1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod anchor;
58mod distinct;
59pub mod grams;
60pub mod graph;
61pub mod host;
62mod prepare;
63mod projection;
64mod run_projection;
65use prepare::Lent;
66pub mod section;
67pub mod stats;
68mod zones;
69
70pub use anchor::{LaneStart, LogAnchor};
71pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
72pub use projection::build_sorted_projection;
73pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
74pub use section::Section;
75pub use zones::{Common, Stripes, ascending, distincts, widths};
76
77const MAGIC: &[u8; 8] = b"RUDBNV10";
78const DIRECTORY: &[u8; 8] = b"RUDBDI10";
79const CATALOG: &[u8; 8] = b"RUDBCA10";
80const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
81const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
82const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
83const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
84const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
85const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
86const MAX_CATALOG_FREQUENCIES: usize = 64;
87const FORMAT: u32 = 30;
88
89const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
124
125const HEADER: u64 = 80;
126const SLOT_BYTES: usize = 28;
127const MAX_PAGE: usize = 256 * 1024 * 1024;
128const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
129const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
130const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
131const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
132const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
140const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
142const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
148const ORDINAL_BOUNDS: &[u8; 8] = b"RUDBFO1\0";
157const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
172const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
192const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
200const KEYS: &[u8; 8] = b"RUDBKY1\0";
207const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
215
216const MAX_SECTIONS: usize = 4096;
223const FREQUENCY_CANDIDATES: usize = 32_768;
224const FREQUENCY_ENTRIES: usize = 512;
225const FREQUENCY_BUILD_RANK: usize = 10;
226const FREQUENCY_ORDINALS: usize = 131_072;
227const MAX_PAIR_FREQUENCIES: usize = 1024;
228const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
233const MAX_FREQUENCY_WORKERS: usize = 32;
240
241fn close_workers() -> usize {
243 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
244}
245
246const CLOSE_BYTES: usize = 1 << 30;
257
258const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
261
262const MAX_ENCODE_WORKERS: usize = 32;
269
270const WRITEBACK_STRETCH: u64 = 32 << 20;
278
279const SIEVE_BUDGET: usize = 8 * 1024;
287
288const PART_BOUND_BYTES: usize = 24;
297
298fn io(error: std::io::Error) -> Error {
299 Error::io(error.to_string())
300}
301
302fn invalid(message: &str) -> Error {
303 Error::invalid_input(format!("invalid rudb native file: {message}"))
304}
305
306fn sum(counts: impl Iterator<Item = u64>) -> u64 {
308 counts.fold(0, u64::saturating_add)
309}
310
311fn span_bytes(spans: &[Span], at: usize) -> u64 {
313 spans.get(at).map_or(0, |span| u64::from(span.length))
314}
315
316fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
318 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
319}
320
321fn dictionary_bytes(table: &Table, at: usize) -> u64 {
323 page_bytes(&table.dictionaries, at)
324 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
325}
326
327fn checksum(bytes: &[u8]) -> u64 {
337 seeded_checksum(bytes, 0)
338}
339
340#[must_use]
347pub fn content_name(bytes: &[u8]) -> u128 {
348 let seed = u64::from(FORMAT);
349 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
350}
351
352#[derive(Debug, Clone)]
358pub struct ContentNamer {
359 seeds: [u64; 2],
360 lanes: [[u64; 4]; 2],
361 held: [u8; 32],
362 filled: usize,
363 length: u64,
364}
365
366impl Default for ContentNamer {
367 fn default() -> Self {
368 let seed = u64::from(FORMAT);
369 let seeds = [seed, !seed];
370 let lanes = seeds.map(|seed| {
371 [
372 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
373 seed.wrapping_add(XXH_P2),
374 seed,
375 seed.wrapping_sub(XXH_P1),
376 ]
377 });
378 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
379 }
380}
381
382impl ContentNamer {
383 pub fn update(&mut self, mut bytes: &[u8]) {
385 self.length += bytes.len() as u64;
386 if self.filled > 0 {
387 let take = (32 - self.filled).min(bytes.len());
388 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
389 self.filled += take;
390 bytes = &bytes[take..];
391 if self.filled < 32 {
392 return;
393 }
394 let block = self.held;
395 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
396 self.filled = 0;
397 }
398 let mut blocks = bytes.chunks_exact(32);
399 for block in blocks.by_ref() {
400 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
401 }
402 let rest = blocks.remainder();
403 self.held[..rest.len()].copy_from_slice(rest);
404 self.filled = rest.len();
405 }
406
407 #[must_use]
409 pub fn finish(&self) -> u128 {
410 let rest = &self.held[..self.filled];
411 let [first, second] = [0, 1].map(|at| {
412 if self.length < 32 {
413 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
414 } else {
415 finish_checksum(self.lanes[at], rest, self.length)
416 }
417 });
418 u128::from(first) << 64 | u128::from(second)
419 }
420}
421
422fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
431 let mut blocks = bytes.chunks_exact(32);
434 let rest = blocks.remainder();
435 if bytes.len() < 32 {
436 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
437 }
438 let mut lanes = [
439 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
440 seed.wrapping_add(XXH_P2),
441 seed,
442 seed.wrapping_sub(XXH_P1),
443 ];
444 for block in blocks.by_ref() {
445 checksum_block(&mut lanes, block);
446 }
447 finish_checksum(lanes, rest, bytes.len() as u64)
448}
449
450const XXH_P1: u64 = 11_400_714_785_074_694_791;
451const XXH_P2: u64 = 14_029_467_366_897_019_727;
452const XXH_P3: u64 = 1_609_587_929_392_839_161;
453const XXH_P4: u64 = 9_650_029_242_287_828_579;
454const XXH_P5: u64 = 2_870_177_450_012_600_261;
455
456fn checksum_round(state: u64, word: u64) -> u64 {
457 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
458}
459
460fn checksum_word(chunk: &[u8]) -> u64 {
461 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
462}
463
464fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
466 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
467 *lane = checksum_round(*lane, checksum_word(chunk));
468 }
469}
470
471fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
473 let merge = |state: u64, lane: u64| {
474 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
475 };
476 let [one, two, three, four] = lanes;
477 let combined = one
478 .rotate_left(1)
479 .wrapping_add(two.rotate_left(7))
480 .wrapping_add(three.rotate_left(12))
481 .wrapping_add(four.rotate_left(18));
482 let hash = merge(merge(merge(merge(combined, one), two), three), four);
483 checksum_tail(hash.wrapping_add(length), rest)
484}
485
486fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
488 let mut words = rest.chunks_exact(8);
489 for chunk in words.by_ref() {
490 hash ^= checksum_round(0, checksum_word(chunk));
491 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
492 }
493 rest = words.remainder();
494 if rest.len() >= 4 {
495 let (head, tail) = rest.split_at(4);
496 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
497 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
498 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
499 rest = tail;
500 }
501 for &byte in rest {
502 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
503 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
504 }
505 hash ^= hash >> 33;
506 hash = hash.wrapping_mul(XXH_P2);
507 hash ^= hash >> 29;
508 hash = hash.wrapping_mul(XXH_P3);
509 hash ^ (hash >> 32)
510}
511
512fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
518 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
519}
520
521fn walk_checksummed(
527 file: &File,
528 offset: u64,
529 length: usize,
530 window: usize,
531 mut each: impl FnMut(&[u8]) -> Result<()>,
532) -> Result<u64> {
533 debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
534 if length < 32 {
535 let mut bytes = vec![0; length];
536 read_at(file, offset, &mut bytes)?;
537 each(&bytes)?;
538 return Ok(checksum(&bytes));
539 }
540 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
541 let mut buffer = vec![0; window.min(length)];
542 let mut read = 0;
543 let (mut whole, mut filled) = (0, 0);
544 while read < length {
545 filled = buffer.len().min(length - read);
546 read_at(file, offset + read as u64, &mut buffer[..filled])?;
547 read += filled;
548 each(&buffer[..filled])?;
549 whole = filled / 32 * 32;
550 for block in buffer[..whole].chunks_exact(32) {
551 checksum_block(&mut lanes, block);
552 }
553 }
554 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
555}
556
557#[derive(Debug, Clone, Copy)]
558struct Slot {
559 offset: u64,
560 length: u32,
561 generation: u64,
562 hash: u64,
563}
564
565impl Slot {
566 fn bytes(self) -> [u8; SLOT_BYTES] {
567 let mut result = [0; SLOT_BYTES];
568 result[..8].copy_from_slice(&self.offset.to_le_bytes());
569 result[8..12].copy_from_slice(&self.length.to_le_bytes());
570 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
571 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
572 result
573 }
574
575 fn read(bytes: &[u8]) -> Self {
576 Self {
577 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
578 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
579 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
580 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
581 }
582 }
583}
584
585#[derive(Debug, Clone, Copy)]
586struct Page {
587 offset: u64,
588 length: u32,
589 hash: u64,
590}
591
592impl Page {
593 fn bytes(&self) -> u64 {
595 u64::from(self.length)
596 }
597}
598
599#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
600enum FrequencyValue {
601 Null,
602 Integer(i128),
603 Code(u32),
604}
605
606type FrequencyMap<V> = HashMap<u64, V, Spread>;
612
613#[derive(Debug)]
627struct Candidates {
628 slots: Vec<Candidate>,
631 held: usize,
632 nulls: u32,
633 decrements: u64,
634 survivors: Vec<Candidate>,
636}
637
638#[derive(Debug, Default, Clone, Copy)]
640struct Candidate {
641 bits: u64,
642 count: u32,
643}
644
645const FIRST_CANDIDATE_SLOTS: usize = 64;
647
648impl Default for Candidates {
649 fn default() -> Self {
650 Self {
651 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
652 held: 0,
653 nulls: 0,
654 decrements: 0,
655 survivors: Vec::new(),
656 }
657 }
658}
659
660impl Candidates {
661 fn add(&mut self, bits: Option<u64>, mut times: u32) {
668 while times > 0 {
669 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
670 match bits {
671 Some(bits) => {
672 let (at, found) = self.find(bits);
673 if found {
674 self.slots[at].count = self.slots[at].count.saturating_add(times);
675 return;
676 }
677 if room {
678 self.place(at, bits, times);
679 return;
680 }
681 }
682 None if self.nulls != 0 => {
683 self.nulls = self.nulls.saturating_add(times);
684 return;
685 }
686 None if room => {
687 self.nulls = times;
688 return;
689 }
690 None => {}
691 }
692 self.decrement();
693 times -= 1;
694 }
695 }
696
697 fn find(&self, bits: u64) -> (usize, bool) {
699 let mask = self.slots.len() - 1;
700 let mut at = home(bits, self.slots.len());
701 loop {
702 let slot = self.slots[at];
703 if slot.count == 0 {
704 return (at, false);
705 }
706 if slot.bits == bits {
707 return (at, true);
708 }
709 at = (at + 1) & mask;
710 }
711 }
712
713 fn position(&self, bits: u64) -> Option<usize> {
715 match self.find(bits) {
716 (at, true) => Some(at),
717 (_, false) => None,
718 }
719 }
720
721 fn place(&mut self, at: usize, bits: u64, count: u32) {
724 let at = if (self.held + 1) * 2 > self.slots.len() {
725 let wider = self.slots.len() * 2;
726 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
727 for slot in old.into_iter().filter(|slot| slot.count != 0) {
728 let (to, _) = self.find(slot.bits);
729 self.slots[to] = slot;
730 }
731 self.find(bits).0
732 } else {
733 at
734 };
735 self.slots[at] = Candidate { bits, count };
736 self.held += 1;
737 }
738
739 fn decrement(&mut self) {
741 let mut survivors = std::mem::take(&mut self.survivors);
742 survivors.clear();
743 survivors.extend(
744 self.slots
745 .iter()
746 .filter(|slot| slot.count > 1)
747 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
748 );
749 self.slots.fill(Candidate::default());
750 self.held = survivors.len();
751 for &slot in &survivors {
752 let (at, _) = self.find(slot.bits);
753 self.slots[at] = slot;
754 }
755 self.survivors = survivors;
756 self.nulls = self.nulls.saturating_sub(1);
757 self.decrements = self.decrements.saturating_add(1);
758 }
759
760 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
762 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
763 }
764}
765
766fn home(bits: u64, slots: usize) -> usize {
771 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
772}
773
774#[derive(Debug, Default)]
776struct Run {
777 bits: Option<u64>,
778 times: u32,
779}
780
781impl Run {
782 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
784 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
785 self.times += 1;
786 return None;
787 }
788 let ended = self.take();
789 self.bits = bits;
790 self.times = 1;
791 ended
792 }
793
794 fn take(&mut self) -> Option<(Option<u64>, u32)> {
796 let times = std::mem::take(&mut self.times);
797 (times != 0).then_some((self.bits, times))
798 }
799}
800
801#[derive(Debug, Default, Clone, Copy)]
803struct Spread;
804
805impl std::hash::BuildHasher for Spread {
806 type Hasher = SpreadHasher;
807
808 fn build_hasher(&self) -> SpreadHasher {
809 SpreadHasher(0)
810 }
811}
812
813#[derive(Debug)]
820struct SpreadHasher(u64);
821
822impl SpreadHasher {
823 fn mix(&mut self, word: u64) {
824 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
825 self.0 = (product as u64) ^ ((product >> 64) as u64);
826 }
827}
828
829impl std::hash::Hasher for SpreadHasher {
830 fn write(&mut self, bytes: &[u8]) {
831 for part in bytes.chunks(8) {
832 let mut word = [0; 8];
833 word[..part.len()].copy_from_slice(part);
834 self.mix(u64::from_le_bytes(word));
835 }
836 }
837
838 fn write_u32(&mut self, value: u32) {
839 self.mix(u64::from(value));
840 }
841
842 fn write_u64(&mut self, value: u64) {
843 self.mix(value);
844 }
845
846 fn write_i128(&mut self, value: i128) {
847 self.mix(value as u64);
848 self.mix((value >> 64) as u64);
849 }
850
851 fn write_isize(&mut self, value: isize) {
852 self.mix(value as u64);
853 }
854
855 fn finish(&self) -> u64 {
856 self.0
857 }
858}
859
860#[derive(Debug, Clone)]
861struct FrequencyEntry {
862 value: FrequencyValue,
863 count: u64,
864}
865
866#[derive(Debug, Clone)]
871struct FrequencySummary {
872 entries: Vec<FrequencyEntry>,
873 omitted_max: u64,
874 ordinals: Vec<u64>,
875 ordinal_entries: Vec<u16>,
876 ordinal_bound: u64,
879}
880
881#[derive(Debug, Clone)]
882struct PairFrequencyEntry {
883 first_entry: u16,
884 second: Option<u32>,
885 count: u64,
886}
887
888#[derive(Debug, Clone)]
894struct PairFrequencySummary {
895 first: u16,
896 second: u16,
897 entries: Vec<PairFrequencyEntry>,
898 omitted_max: u64,
899}
900
901type FrequencyHead = (Vec<FrequencyEntry>, u64);
903
904#[derive(Debug, Clone)]
912enum Frequencies {
913 Held(FrequencySummary),
914 Stored {
917 span: Span,
918 values: bool,
919 entries: usize,
920 },
921}
922
923#[derive(Debug, Clone)]
928pub struct FrequencyPrefix {
929 pub entries: Vec<(Value, u64)>,
931 pub omitted_max: u64,
933}
934
935#[derive(Debug, Clone, PartialEq)]
937pub struct FrequencyOccurrences {
938 pub omitted_max: u64,
940 pub ordinals: Vec<u64>,
942 pub anchors: Vec<Value>,
944 pub anchor_indices: Vec<u16>,
946}
947
948pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
950
951#[derive(Debug, Clone, Copy, Default)]
958struct Span {
959 offset: u64,
960 length: u32,
961}
962
963#[derive(Debug, Clone, Default)]
971struct Pages {
972 columns: usize,
973 held: Box<[StripePage]>,
974}
975
976#[derive(Debug, Clone, Copy)]
978struct StripePage {
979 offset: u64,
980 hash: u64,
981 length: u32,
982 column: u32,
983}
984
985impl Pages {
986 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
988 let mut held = Vec::with_capacity(slots.iter().flatten().count());
989 for (column, page) in slots.iter().enumerate() {
990 if let Some(page) = page {
991 let column =
992 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
993 held.push(StripePage {
994 offset: page.offset,
995 hash: page.hash,
996 length: page.length,
997 column,
998 });
999 }
1000 }
1001 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1002 }
1003
1004 fn get(&self, column: usize) -> Option<Page> {
1006 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1007 let placed = self.held[at];
1008 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1009 }
1010
1011 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1013 (0..self.columns).map(|column| self.get(column))
1014 }
1015
1016 fn bytes(&self, column: usize) -> u64 {
1018 self.get(column).map_or(0, |page| page.bytes())
1019 }
1020}
1021
1022#[derive(Debug, Clone)]
1024pub struct Stripe {
1025 rows: usize,
1026 parts: Vec<u32>,
1029 index: Span,
1033 pages: Vec<Span>,
1034 memberships: Pages,
1035 sieves: Pages,
1038 part_ranges: Pages,
1049 zone: Zone,
1050}
1051
1052impl Stripe {
1053 #[must_use]
1055 pub fn rows(&self) -> usize {
1056 self.rows
1057 }
1058
1059 #[must_use]
1061 pub fn parts(&self) -> usize {
1062 self.parts.len()
1063 }
1064
1065 #[must_use]
1071 pub fn zone(&self) -> &Zone {
1072 &self.zone
1073 }
1074}
1075
1076#[derive(Debug, Clone)]
1078pub struct Table {
1079 name: String,
1080 fields: Vec<Field>,
1081 stripes: Vec<Stripe>,
1082 rows: usize,
1083 dictionaries: Vec<Option<Page>>,
1084 dictionary_payloads: Vec<u64>,
1090 demoted: Vec<bool>,
1096 frequencies: Vec<Option<Frequencies>>,
1097 ordinal_bounds: Vec<u64>,
1100 pair_frequencies: Vec<PairFrequencySummary>,
1101 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1106 host_groups: Option<host::HostSummary>,
1108 distincts: Vec<Option<u64>>,
1118 clustering: Option<Clustering>,
1126 generation: u64,
1140 sections: Vec<Section>,
1147 constraints: Constraints,
1150}
1151
1152#[derive(Debug, Clone, Default, PartialEq, Eq)]
1157pub struct Constraints {
1158 pub keys: Vec<(Vec<u16>, bool)>,
1160 pub foreign: Vec<StoredForeign>,
1162}
1163
1164impl Constraints {
1165 #[must_use]
1167 pub fn is_empty(&self) -> bool {
1168 self.keys.is_empty() && self.foreign.is_empty()
1169 }
1170}
1171
1172#[derive(Debug, Clone, PartialEq, Eq)]
1174pub struct StoredForeign {
1175 pub columns: Vec<u16>,
1177 pub table: String,
1179 pub referenced: Vec<u16>,
1181}
1182
1183impl Table {
1184 #[must_use]
1186 pub fn name(&self) -> &str {
1187 &self.name
1188 }
1189
1190 #[must_use]
1192 pub fn fields(&self) -> &[Field] {
1193 &self.fields
1194 }
1195
1196 #[must_use]
1198 pub fn rows(&self) -> usize {
1199 self.rows
1200 }
1201
1202 #[must_use]
1204 pub fn stripes(&self) -> &[Stripe] {
1205 &self.stripes
1206 }
1207
1208 #[must_use]
1210 pub fn clustering(&self) -> Option<&Clustering> {
1211 self.clustering.as_ref()
1212 }
1213
1214 #[must_use]
1216 pub fn constraints(&self) -> &Constraints {
1217 &self.constraints
1218 }
1219
1220 #[must_use]
1225 pub fn generation(&self) -> u64 {
1226 self.generation
1227 }
1228
1229 #[must_use]
1236 pub fn sections(&self) -> &[Section] {
1237 &self.sections
1238 }
1239}
1240
1241#[derive(Debug, Clone)]
1253struct Entry {
1254 name: String,
1255 fields: Vec<Field>,
1256 rows: usize,
1257 directory: Page,
1259 nonzero: Vec<Option<u64>>,
1262 aggregates: Vec<Option<(i128, u64)>>,
1264 distincts: Vec<Option<u64>>,
1266 extremes: Vec<StoredIntegerExtremes>,
1268 frequencies: Vec<StoredNumericFrequencies>,
1270}
1271
1272type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1273type StoredNumericFrequencies = Option<NumericFrequencies>;
1274
1275#[derive(Debug, Clone, PartialEq, Eq)]
1288pub struct ViewEntry {
1289 pub name: String,
1291 pub sql: String,
1293 pub statement: String,
1295 pub aliases: Vec<String>,
1297 pub columns: Vec<Field>,
1299}
1300
1301#[derive(Debug, Clone)]
1303pub struct ColumnLayout {
1304 pub name: String,
1306 pub kind: String,
1308 pub pages: u64,
1310 pub memberships: u64,
1312 pub sieves: u64,
1314 pub part_ranges: u64,
1316 pub dictionary: u64,
1318}
1319
1320impl ColumnLayout {
1321 #[must_use]
1323 pub fn total(&self) -> u64 {
1324 self.pages
1325 .saturating_add(self.memberships)
1326 .saturating_add(self.sieves)
1327 .saturating_add(self.part_ranges)
1328 .saturating_add(self.dictionary)
1329 }
1330}
1331
1332#[derive(Debug, Clone)]
1343pub struct Layout {
1344 pub file: u64,
1346 pub rows: usize,
1348 pub stripes: usize,
1350 pub parts: usize,
1352 pub columns: Vec<ColumnLayout>,
1354 pub indexes: u64,
1357 pub directory: u64,
1359 pub header: u64,
1361}
1362
1363impl Layout {
1364 #[must_use]
1366 pub fn columns_total(&self) -> u64 {
1367 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1368 }
1369
1370 #[must_use]
1376 pub fn unaccounted(&self) -> u64 {
1377 self.file
1378 .saturating_sub(self.columns_total())
1379 .saturating_sub(self.indexes)
1380 .saturating_sub(self.directory)
1381 .saturating_sub(self.header)
1382 }
1383}
1384
1385#[derive(Debug, Clone)]
1396pub struct StoredPart {
1397 pub stripe: usize,
1399 pub part: usize,
1401 pub row: usize,
1403 pub rows: usize,
1405 pub encoding: String,
1407 pub bytes: u64,
1409 pub page: u64,
1411 pub offset: u64,
1413 pub low: Option<Value>,
1415 pub high: Option<Value>,
1417 pub nulls: Option<usize>,
1419}
1420
1421const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1428
1429#[derive(Debug)]
1454struct GlobalDictionary {
1455 primary: HashMap<u64, u32, Spread>,
1459 collisions: HashMap<u64, Vec<u32>, Spread>,
1460 checks: Vec<u64>,
1462 ends: Vec<u32>,
1464 counts: Vec<u64>,
1465 nulls: u64,
1466 filling: Vec<u8>,
1468 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1474 waiting: Vec<(usize, Vec<u8>)>,
1479 sample: Vec<(usize, Vec<u8>)>,
1485 stride: usize,
1487 shape: Option<chooser::Settled>,
1489 settled: usize,
1491 blocks: Vec<Vec<u8>>,
1496 early: BTreeMap<usize, EncodedBlock>,
1502 placed: Vec<Placed>,
1504 charged: u64,
1507 demoted: bool,
1509}
1510
1511#[derive(Debug, Clone, Copy)]
1513struct Placed {
1514 start: u64,
1515 length: u64,
1516 hash: u64,
1517}
1518
1519type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1521
1522impl GlobalDictionary {
1523 fn new() -> Self {
1524 Self {
1525 primary: HashMap::default(),
1526 collisions: HashMap::default(),
1527 checks: Vec::new(),
1528 ends: Vec::new(),
1529 counts: Vec::new(),
1530 nulls: 0,
1531 filling: Vec::new(),
1532 grams: Vec::new(),
1533 waiting: Vec::new(),
1534 sample: Vec::new(),
1535 stride: 1,
1536 shape: None,
1537 settled: 0,
1538 blocks: Vec::new(),
1539 early: BTreeMap::new(),
1540 placed: Vec::new(),
1541 charged: 0,
1542 demoted: false,
1543 }
1544 }
1545
1546 fn values(&self) -> usize {
1548 self.ends.len()
1549 }
1550
1551 fn closing_bytes(&self) -> usize {
1554 let values = self.values();
1555 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1556 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1557 .sum::<usize>();
1558 let beside = size_of::<u32>().max(size_of::<(u64, Option<u32>)>());
1562 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + beside))
1563 }
1564
1565 fn held_bytes(&self) -> u64 {
1571 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1572 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1573 }
1574 fn spilled<T>(values: &Vec<T>) -> usize {
1575 values.capacity() * size_of::<T>()
1576 }
1577 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1578 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1579 };
1580 let bytes = table(&self.primary)
1581 + table(&self.collisions)
1582 + self.collisions.values().map(spilled).sum::<usize>()
1583 + spilled(&self.checks)
1584 + spilled(&self.ends)
1585 + spilled(&self.counts)
1586 + self.filling.capacity()
1587 + spilled(&self.grams)
1588 + raw(&self.waiting)
1589 + raw(&self.sample)
1590 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1591 + spilled(&self.placed);
1592 bytes as u64
1593 }
1594
1595 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1598 let before = self.charged;
1599 let now = self.held_bytes();
1600 if let Some(profile) = profile {
1601 if now >= before {
1602 profile.hold(now - before);
1603 } else {
1604 profile.release(before - now);
1605 }
1606 }
1607 self.charged = now;
1608 (before, now)
1609 }
1610
1611 fn demote(&mut self) {
1619 if self.demoted {
1620 return;
1621 }
1622 self.seal_rest();
1623 self.release_lookup();
1624 self.demoted = true;
1625 }
1626
1627 fn release_lookup(&mut self) {
1634 self.primary = HashMap::default();
1635 self.collisions = HashMap::default();
1636 self.checks = Vec::new();
1637 self.sample = Vec::new();
1638 self.filling = Vec::new();
1639 }
1640
1641 fn encoded(&self) -> usize {
1643 self.placed.len() + self.blocks.len()
1644 }
1645
1646 #[cfg(test)]
1647 fn code(&mut self, text: &str) -> Result<u32> {
1648 let bytes = text.as_bytes();
1649 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1650 }
1651
1652 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1658 if let Some(&code) = self.primary.get(&hash) {
1659 if self.checks.get(code as usize) == Some(&check) {
1660 return Ok(code);
1661 }
1662 if let Some(codes) = self.collisions.get(&hash)
1663 && let Some(code) =
1664 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1665 {
1666 return Ok(code);
1667 }
1668 let code = self.insert(text, check)?;
1669 self.collisions.entry(hash).or_default().push(code);
1670 return Ok(code);
1671 }
1672 let code = self.insert(text, check)?;
1673 self.primary.insert(hash, code);
1674 Ok(code)
1675 }
1676
1677 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1678 if self.demoted {
1679 return Err(Error::internal("a value was coded against a demoted dictionary"));
1680 }
1681 let code = u32::try_from(self.ends.len())
1682 .map_err(|_| invalid("global dictionary has too many values"))?;
1683 self.filling.extend_from_slice(text);
1684 self.ends.push(
1685 u32::try_from(self.filling.len())
1686 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1687 );
1688 self.checks.push(check);
1689 self.counts.push(0);
1690 if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1691 self.seal();
1692 }
1693 Ok(code)
1694 }
1695
1696 fn seal(&mut self) {
1702 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1703 let bytes = std::mem::take(&mut self.filling);
1704 if at.is_multiple_of(self.stride) {
1705 self.sample.push((at, bytes.clone()));
1706 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1707 self.stride *= 2;
1708 let stride = self.stride;
1709 self.sample.retain(|(at, _)| at % stride == 0);
1710 }
1711 }
1712 self.waiting.push((at, bytes));
1713 }
1714
1715 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1717 block_values(self.block_ends(at), bytes)
1718 }
1719
1720 fn block_ends(&self, at: usize) -> &[u32] {
1722 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1723 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1724 &self.ends[first..last]
1725 }
1726
1727 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1734 let Some(shape) = &self.shape else { return Vec::new() };
1735 let waiting = std::mem::take(&mut self.waiting);
1736 waiting
1737 .into_iter()
1738 .map(|(at, bytes)| Unencoded {
1739 column,
1740 at,
1741 ends: self.block_ends(at).to_vec(),
1742 bytes,
1743 shape: shape.clone(),
1744 })
1745 .collect()
1746 }
1747
1748 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1751 if at < self.encoded() || self.early.insert(at, block).is_some() {
1752 return Err(Error::internal("a dictionary block came back twice"));
1753 }
1754 while let Some(block) = self.early.remove(&self.encoded()) {
1755 self.push_block(block);
1756 }
1757 Ok(())
1758 }
1759
1760 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1762 self.blocks.push(bytes);
1763 self.grams.push(*grams);
1764 }
1765
1766 fn settle(&mut self) -> Result<()> {
1774 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1775 return Ok(());
1776 }
1777 self.settle_on_sample()
1778 }
1779
1780 fn settle_rest(&mut self) -> Result<()> {
1788 if self.shape.is_some() || self.sample.is_empty() {
1789 return Ok(());
1790 }
1791 self.settle_on_sample()
1792 }
1793
1794 fn settle_on_sample(&mut self) -> Result<()> {
1795 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1796 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1797 return Ok(());
1798 }
1799 let sample =
1800 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1801 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1802 self.settled = complete;
1803 Ok(())
1804 }
1805
1806 fn seal_rest(&mut self) {
1808 if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1812 self.seal();
1813 }
1814 }
1815
1816 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1819 let (block, bytes) = &self.waiting[at];
1820 let values = self.slices(*block, bytes);
1821 let encoded = match &self.shape {
1822 Some(shape) => string::encode_with(&values, shape)?,
1823 None => string::encode(&values)?,
1824 };
1825 Ok((encoded, block_grams(&values)))
1826 }
1827
1828 #[cfg(test)]
1830 fn finish_blocks(&mut self) -> Result<()> {
1831 self.seal_rest();
1832 let made = (0..self.waiting.len())
1833 .map(|at| self.encode_waiting(at))
1834 .collect::<Result<Vec<_>>>()?;
1835 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1836 if self.encoded() != at {
1837 return Err(Error::internal("a dictionary block was encoded out of order"));
1838 }
1839 self.push_block(block);
1840 }
1841 Ok(())
1842 }
1843
1844 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1862 let count = self.placed.len() + self.blocks.len();
1863 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1864 return Err(invalid("global dictionary blocks do not cover its values"));
1865 }
1866 let mut bases = Vec::with_capacity(count);
1867 let mut total = 0_usize;
1868 for block in 0..count {
1869 bases.push(total as u64);
1870 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1871 total = total
1872 .checked_add(self.ends[last] as usize)
1873 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1874 }
1875 let mut flat = vec![0_u8; total];
1876 let mut outs = Vec::with_capacity(count);
1877 let mut rest = flat.as_mut_slice();
1878 for block in 0..count {
1879 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1880 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1881 outs.push((block, out));
1882 rest = after;
1883 }
1884 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1885 let mut stored = Vec::new();
1886 for (block, out) in run {
1887 let encoded = match self.placed.get(*block) {
1888 Some(place) => {
1889 let file = file.ok_or_else(|| {
1890 Error::internal("a written dictionary block has no file")
1891 })?;
1892 let length = usize::try_from(place.length).map_err(|_| {
1893 invalid("global dictionary block does not fit in memory")
1894 })?;
1895 stored.resize(length, 0);
1896 read_at(file, place.start, &mut stored)?;
1897 if checksum(&stored) != place.hash {
1898 return Err(invalid(
1899 "a global dictionary block did not read back as written",
1900 ));
1901 }
1902 stored.as_slice()
1903 }
1904 None => &self.blocks[*block - self.placed.len()],
1905 };
1906 let decoded = string::decode_flat(encoded)?;
1907 if decoded.bytes().len() != out.len() {
1908 return Err(invalid(
1909 "a global dictionary block is not the length its ends say",
1910 ));
1911 }
1912 out.copy_from_slice(decoded.bytes());
1913 }
1914 Ok(())
1915 };
1916 let workers = close_workers().min(count / 16).max(1);
1919 if workers <= 1 {
1920 one(&mut outs)?;
1921 } else {
1922 let per = count.div_ceil(workers);
1923 std::thread::scope(|scope| {
1924 outs.chunks_mut(per)
1925 .map(|run| scope.spawn(|| one(run)))
1926 .collect::<Vec<_>>()
1927 .into_iter()
1928 .try_for_each(|handle| {
1929 handle.join().map_err(|_| {
1930 Error::internal("a global dictionary decode worker panicked")
1931 })?
1932 })
1933 })?;
1934 }
1935 drop(outs);
1936 Ok((flat, bases))
1937 }
1938
1939 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1944 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1945 let Some(&end) = ends.get(code) else { return (0, 0) };
1946 let base = base as usize;
1947 let from =
1948 if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1949 (base + from, base + end as usize)
1950 }
1951
1952 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1972 let (flat, bases) = self.decoded(file)?;
1973 let value = |code: u32| {
1974 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1975 flat.get(from..to).unwrap_or_default()
1976 };
1977 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1978 sort_by_value_across(&mut codes, value, close_workers());
1979 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1980 Ok((order, flat, bases))
1981 }
1982
1983 #[cfg(test)]
1984 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1985 self.ranked_with_values(file).map(|(order, _, _)| order)
1986 }
1987}
1988
1989#[derive(Debug)]
1997pub struct Writer {
1998 file: Box<dyn rudb_io::File>,
2001 at: u64,
2009 written_back: u64,
2011 table: Table,
2012 generation: u64,
2013 order: Vec<((u64, u64), (u64, u64))>,
2016 next_order: u64,
2017 dictionaries: Vec<Option<GlobalDictionary>>,
2018 coded: Arc<prepare::Coding>,
2021 gathers: Vec<Option<stats::Gather>>,
2027 lent: Option<Arc<Lent>>,
2030 pending: Vec<PendingChunk>,
2031 closed: Vec<Entry>,
2033 views: Vec<ViewEntry>,
2038 card: Option<KeptCard>,
2040 anchor: Option<LogAnchor>,
2043 profile: Option<Arc<LoadProfile>>,
2049}
2050
2051#[derive(Debug)]
2059struct PendingChunk {
2060 order: (u64, u64),
2061 chunk: Chunk,
2062}
2063
2064#[derive(Debug, Clone, Copy)]
2070struct Part {
2071 order: (u64, u64),
2072 rows: usize,
2073 footprint: usize,
2074}
2075
2076impl Part {
2077 fn of(pending: &PendingChunk) -> Self {
2078 Self {
2079 order: pending.order,
2080 rows: pending.chunk.len(),
2081 footprint: pending.chunk.footprint(),
2082 }
2083 }
2084}
2085
2086#[derive(Debug, Default)]
2092struct ColumnStripe {
2093 pages: Vec<Vec<u8>>,
2094 sums: Vec<u64>,
2097 codes: Vec<Option<Vec<u32>>>,
2098 sieves: Vec<Option<Sieve>>,
2099 ranges: Vec<Range>,
2100}
2101
2102fn coded_type(ty: &LogicalType) -> bool {
2110 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2111}
2112
2113fn dictionary_tag(ty: &LogicalType) -> u8 {
2120 if ty == &LogicalType::Blob { 2 } else { 1 }
2121}
2122
2123fn weight(ty: &LogicalType) -> usize {
2131 match ty {
2132 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2133 LogicalType::HugeInt
2134 | LogicalType::UHugeInt
2135 | LogicalType::Uuid
2136 | LogicalType::Interval => 16,
2137 LogicalType::BigInt
2138 | LogicalType::UBigInt
2139 | LogicalType::Timestamp
2140 | LogicalType::Time
2141 | LogicalType::TimeTz
2142 | LogicalType::TimestampTz
2143 | LogicalType::TimestampS
2144 | LogicalType::TimestampMs
2145 | LogicalType::TimestampNs
2146 | LogicalType::Double
2147 | LogicalType::Decimal { .. } => 8,
2148 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2149 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2150 _ => 1,
2151 }
2152}
2153
2154pub const STRIPE_PARTS: usize = 64;
2161
2162const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2170
2171const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2187
2188const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2190
2191fn index_section(parts: usize) -> Result<usize> {
2193 parts
2194 .checked_mul(INDEX_ENTRY)
2195 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2196 .ok_or_else(|| invalid("index page length overflow"))
2197}
2198
2199impl Writer {
2200 pub fn open(
2219 path: impl AsRef<Path>,
2220 name: impl Into<String>,
2221 fields: Vec<Field>,
2222 ) -> Result<Self> {
2223 Self::open_in(&RealFilesystem::new(), path, name, fields)
2224 }
2225
2226 pub fn open_in(
2233 fs: &dyn Filesystem,
2234 path: impl AsRef<Path>,
2235 name: impl Into<String>,
2236 fields: Vec<Field>,
2237 ) -> Result<Self> {
2238 for field in &fields {
2239 type_tag(&field.ty)?;
2240 }
2241 let name = name.into();
2242 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2243 let size = file.len()?;
2244 let (slot, bytes, _) = committed_slot(&*file, size)?;
2245 let (mut closed, views, card, anchor) = decode_catalog(&bytes, size)?;
2246 let card = card_for(path.as_ref(), card);
2247 if let Some(at) = closed.iter().position(|held| held.name == name) {
2258 if closed[at].rows > 0 {
2259 return Err(invalid("two tables in one native file have the same name"));
2260 }
2261 closed.remove(at);
2262 }
2263 let generation = slot
2268 .generation
2269 .checked_add(1)
2270 .ok_or_else(|| invalid("native file generation overflow"))?;
2271 Ok(Self {
2272 file,
2273 at: size,
2276 written_back: size,
2277 dictionaries: fields
2278 .iter()
2279 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2280 .collect(),
2281 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2282 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2283 lent: None,
2284 table: Table {
2285 name,
2286 dictionaries: vec![None; fields.len()],
2287 dictionary_payloads: Vec::new(),
2288 demoted: Vec::new(),
2289 distincts: vec![None; fields.len()],
2290 fields,
2291 stripes: Vec::new(),
2292 rows: 0,
2293 frequencies: Vec::new(),
2294 ordinal_bounds: Vec::new(),
2295 pair_frequencies: Vec::new(),
2296 frequency_texts: Vec::new(),
2297 host_groups: None,
2298 clustering: None,
2299 constraints: Constraints::default(),
2300 generation,
2301 sections: Vec::new(),
2302 },
2303 generation,
2304 order: Vec::new(),
2305 next_order: 0,
2306 pending: Vec::with_capacity(STRIPE_PARTS),
2307 closed,
2308 views,
2309 card,
2310 anchor,
2311 profile: None,
2312 })
2313 }
2314
2315 pub fn create(
2321 path: impl AsRef<Path>,
2322 name: impl Into<String>,
2323 fields: Vec<Field>,
2324 ) -> Result<Self> {
2325 Self::create_in(&RealFilesystem::new(), path, name, fields)
2326 }
2327
2328 pub fn create_in(
2338 fs: &dyn Filesystem,
2339 path: impl AsRef<Path>,
2340 name: impl Into<String>,
2341 fields: Vec<Field>,
2342 ) -> Result<Self> {
2343 for field in &fields {
2344 type_tag(&field.ty)?;
2345 }
2346 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2347 let mut header = [0; HEADER as usize];
2348 header[..8].copy_from_slice(MAGIC);
2349 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2350 file.write_at(0, &header)?;
2351 Ok(Self {
2352 file,
2353 at: HEADER,
2354 written_back: HEADER,
2355 dictionaries: fields
2356 .iter()
2357 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2358 .collect(),
2359 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2360 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2361 lent: None,
2362 table: Table {
2363 name: name.into(),
2364 dictionaries: vec![None; fields.len()],
2365 dictionary_payloads: Vec::new(),
2366 demoted: Vec::new(),
2367 distincts: vec![None; fields.len()],
2368 fields,
2369 stripes: Vec::new(),
2370 rows: 0,
2371 frequencies: Vec::new(),
2372 ordinal_bounds: Vec::new(),
2373 pair_frequencies: Vec::new(),
2374 frequency_texts: Vec::new(),
2375 host_groups: None,
2376 clustering: None,
2377 constraints: Constraints::default(),
2378 generation: 1,
2379 sections: Vec::new(),
2380 },
2381 generation: 1,
2382 order: Vec::new(),
2383 next_order: 0,
2384 pending: Vec::with_capacity(STRIPE_PARTS),
2385 closed: Vec::new(),
2386 views: Vec::new(),
2387 card: card_for(path.as_ref(), None),
2388 anchor: None,
2389 profile: None,
2390 })
2391 }
2392
2393 pub fn empty(
2417 path: impl AsRef<Path>,
2418 views: &[ViewEntry],
2419 anchor: Option<&LogAnchor>,
2420 ) -> Result<()> {
2421 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2422 let mut header = [0; HEADER as usize];
2423 header[..8].copy_from_slice(MAGIC);
2424 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2425 file.write_at(0, &header)?;
2426 let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref(), anchor)?;
2427 file.write_at(HEADER, &catalog)?;
2428 file.sync()?;
2432 let slot = Slot {
2433 offset: HEADER,
2434 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2435 generation: 1,
2436 hash: checksum(&catalog),
2437 };
2438 file.write_at(slot_offset(1), &slot.bytes())?;
2439 file.sync()?;
2440 Ok(())
2441 }
2442
2443 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2454 for field in &fields {
2455 type_tag(&field.ty)?;
2456 }
2457 let name = name.into();
2458 let entry = self.close()?;
2459 if entry.name == name {
2460 return Err(invalid("two tables in one native file have the same name"));
2461 }
2462 if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2466 if self.closed[at].rows > 0 {
2467 return Err(invalid("two tables in one native file have the same name"));
2468 }
2469 self.closed.remove(at);
2470 }
2471 let Self { file, at, generation, mut closed, views, card, anchor, .. } = self;
2472 closed.push(entry);
2473 Ok(Self {
2474 file,
2475 written_back: at,
2476 at,
2477 generation,
2478 closed,
2479 views,
2480 card,
2481 anchor,
2482 profile: None,
2483 dictionaries: fields
2484 .iter()
2485 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2486 .collect(),
2487 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2488 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2489 lent: None,
2490 table: Table {
2491 name,
2492 dictionaries: vec![None; fields.len()],
2493 dictionary_payloads: Vec::new(),
2494 demoted: Vec::new(),
2495 distincts: vec![None; fields.len()],
2496 fields,
2497 stripes: Vec::new(),
2498 rows: 0,
2499 frequencies: Vec::new(),
2500 ordinal_bounds: Vec::new(),
2501 pair_frequencies: Vec::new(),
2502 frequency_texts: Vec::new(),
2503 host_groups: None,
2504 clustering: None,
2505 constraints: Constraints::default(),
2506 generation,
2507 sections: Vec::new(),
2508 },
2509 order: Vec::new(),
2510 next_order: 0,
2511 pending: Vec::with_capacity(STRIPE_PARTS),
2512 })
2513 }
2514
2515 #[must_use]
2525 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2526 self.views = views;
2527 self
2528 }
2529
2530 #[must_use]
2533 pub fn with_log_anchor(mut self, anchor: LogAnchor) -> Self {
2534 self.anchor = Some(anchor);
2535 self
2536 }
2537
2538 #[must_use]
2544 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2545 self.profile = Some(profile);
2546 self
2547 }
2548
2549 #[must_use]
2553 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2554 self.coded.cap(bytes);
2555 self
2556 }
2557
2558 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2573 self.table.clustering = Some(Clustering::new(
2576 clustering.columns().to_vec(),
2577 clustering.width(),
2578 &self.table.fields,
2579 )?);
2580 Ok(self)
2581 }
2582
2583 pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2591 let width = self.table.fields.len();
2592 let fits = |columns: &[u16]| {
2593 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2594 };
2595 if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2596 || !constraints.foreign.iter().all(|foreign| {
2597 fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2598 })
2599 {
2600 return Err(invalid("a constraint names a column the table does not have"));
2601 }
2602 self.table.constraints = constraints;
2603 Ok(self)
2604 }
2605
2606 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2611 self.file.write_at(self.at, bytes)?;
2612 self.at = self
2613 .at
2614 .checked_add(bytes.len() as u64)
2615 .ok_or_else(|| invalid("native file length overflow"))?;
2616 if self.at - self.written_back >= WRITEBACK_STRETCH {
2617 self.file.start_writeback(self.written_back, self.at - self.written_back);
2618 self.written_back = self.at;
2619 }
2620 Ok(())
2621 }
2622
2623 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2629 let order = (self.next_order, 0);
2630 self.next_order = self.next_order.saturating_add(1);
2631 self.append_at(order, chunk)
2632 }
2633
2634 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2645 if chunk.is_empty() {
2646 return Ok(());
2647 }
2648 self.admit(chunk)?;
2649 if self.pending.last().is_some_and(|last| last.order > order) {
2650 self.flush_pending()?;
2651 }
2652 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2657 if self.pending.len() == STRIPE_PARTS {
2658 self.flush_pending()?;
2659 }
2660 Ok(())
2661 }
2662
2663 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2679 if parts.len() > STRIPE_PARTS {
2680 return Err(invalid("a stripe was handed more parts than it holds"));
2681 }
2682 self.flush_pending()?;
2685 for (order, chunk) in parts {
2686 if chunk.is_empty() {
2687 continue;
2688 }
2689 self.admit(&chunk)?;
2690 self.pending.push(PendingChunk { order, chunk });
2691 }
2692 self.flush_pending()
2693 }
2694
2695 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2697 if chunk.width() != self.table.fields.len() {
2698 return Err(invalid("chunk width differs from table schema"));
2699 }
2700 for (index, field) in self.table.fields.iter().enumerate() {
2701 if chunk.column(index)?.logical_type() != &field.ty {
2702 return Err(invalid("chunk type differs from table schema"));
2703 }
2704 }
2705 self.table.rows = self
2706 .table
2707 .rows
2708 .checked_add(chunk.len())
2709 .ok_or_else(|| invalid("row count overflow"))?;
2710 Ok(())
2711 }
2712
2713 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2715 let mut stripe = ColumnStripe {
2716 pages: Vec::with_capacity(columns.len()),
2717 sums: Vec::with_capacity(columns.len()),
2718 codes: Vec::with_capacity(columns.len()),
2719 sieves: Vec::with_capacity(columns.len()),
2720 ranges: Vec::with_capacity(columns.len()),
2721 };
2722 let mut settling = Settling::default();
2723 for &column in columns {
2724 Self::encode_page(&mut stripe, &mut settling, column)?;
2725 }
2726 Ok(stripe)
2727 }
2728
2729 fn encode_page(
2732 stripe: &mut ColumnStripe,
2733 settling: &mut Settling,
2734 column: &Vector,
2735 ) -> Result<()> {
2736 let bytes = encode(column, settling)?;
2737 if bytes.len() > MAX_PAGE {
2738 return Err(invalid("column page exceeds the configured bound"));
2739 }
2740 let range = Range::of(column);
2743 let sieve =
2754 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2755 stripe.sums.push(checksum(&bytes));
2756 stripe.pages.push(bytes);
2757 stripe.codes.push(None);
2758 stripe.sieves.push(sieve);
2759 stripe.ranges.push(range);
2760 Ok(())
2761 }
2762
2763 fn place_blocks(&mut self) -> Result<()> {
2768 if let Some(lent) = self.lent.clone() {
2769 return self.place_lent_blocks(&lent);
2770 }
2771 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2772 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2773 for block in std::mem::take(&mut dictionary.blocks) {
2774 let start = self.at;
2775 self.put(&block)?;
2776 dictionary.placed.push(Placed {
2777 start,
2778 length: block.len() as u64,
2779 hash: checksum(&block),
2780 });
2781 }
2782 Ok(())
2783 });
2784 self.dictionaries = dictionaries;
2785 placed
2786 }
2787
2788 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2794 for column in lent.columns() {
2795 let Ok(mut held) = column.try_lock() else { continue };
2796 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2797 for block in std::mem::take(&mut dictionary.blocks) {
2798 let start = self.at;
2799 self.put(&block)?;
2800 dictionary.placed.push(Placed {
2801 start,
2802 length: block.len() as u64,
2803 hash: checksum(&block),
2804 });
2805 }
2806 }
2807 Ok(())
2808 }
2809
2810 fn reclaim(&mut self) -> Result<()> {
2814 let Some(lent) = self.lent.take() else { return Ok(()) };
2815 let (dictionaries, gathers) = lent.reclaim()?;
2816 self.dictionaries = dictionaries;
2817 self.gathers = gathers;
2818 Ok(())
2819 }
2820
2821 fn flush_pending(&mut self) -> Result<()> {
2826 if self.pending.is_empty() {
2827 return Ok(());
2828 }
2829 let held = std::mem::take(&mut self.pending);
2830 let prepared = self.preparer().prepare_held(held)?;
2831 let merged = self.merge_held(prepared)?;
2832 let paged = merged.pages()?;
2833 self.write_paged(paged)
2834 }
2835
2836 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2838 let width = self.table.fields.len();
2839 let parts = held.len();
2840 if encoded.len() != width {
2841 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2842 }
2843 let profile = self.profile.clone();
2844 if let Some(profile) = &profile {
2845 let rows = held.iter().map(|part| part.rows as u64).sum();
2846 let raw = held.iter().map(|part| part.footprint as u64).sum();
2847 let pages =
2848 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2849 profile.moved(Stage::Pages, raw, pages, rows);
2850 }
2851 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2854 let before = self.at;
2855 self.place_blocks()?;
2856 drop(timing);
2857 if let Some(profile) = &profile {
2858 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2859 }
2860 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2861 let before = self.at;
2862 let mut pages = Vec::with_capacity(width);
2863 let mut memberships = vec![None; width];
2864 let mut ranges = Vec::with_capacity(width);
2865 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2866 let start = self.at;
2869 let mut out = Vec::with_capacity(width.saturating_mul(parts));
2870 for stripe in &encoded {
2871 let offset = self.at;
2872 let section = index.len();
2873 let mut length = 0_usize;
2874 if stripe.sums.len() != stripe.pages.len() {
2875 return Err(Error::internal("a stripe's pages came without their checksums"));
2876 }
2877 for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2878 put_u32(
2879 &mut index,
2880 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2881 );
2882 put_u64(&mut index, sum);
2883 out.push(bytes.as_slice());
2884 length = length
2885 .checked_add(bytes.len())
2886 .ok_or_else(|| invalid("column page length overflow"))?;
2887 }
2888 let hash = checksum(&index[section..]);
2889 put_u64(&mut index, hash);
2890 if length > MAX_PAGE {
2891 return Err(invalid("column page exceeds the configured bound"));
2892 }
2893 self.at = self
2894 .at
2895 .checked_add(length as u64)
2896 .ok_or_else(|| invalid("native file length overflow"))?;
2897 pages.push(Span {
2898 offset,
2899 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2900 });
2901 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2902 }
2903 self.file.write_parts_at(start, &out)?;
2904 drop(out);
2905 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2906 if stripe.codes.iter().all(Option::is_none) {
2907 continue;
2908 }
2909 let lists = stripe
2910 .codes
2911 .iter()
2912 .map(|codes| codes.clone().unwrap_or_default())
2913 .collect::<Vec<_>>();
2914 let bytes = encode_membership(&merged_codes(lists));
2915 let offset = self.at;
2916 self.put(&bytes)?;
2917 *membership = Some(Page {
2918 offset,
2919 length: u32::try_from(bytes.len())
2920 .map_err(|_| invalid("membership page length overflow"))?,
2921 hash: checksum(&bytes),
2922 });
2923 }
2924 let mut sieves = vec![None; width];
2925 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2926 if stripe.sieves.iter().all(Option::is_none) {
2927 continue;
2928 }
2929 let bytes = encode_sieves(stripe.sieves.iter())?;
2930 let offset = self.at;
2931 self.put(&bytes)?;
2932 *page = Some(Page {
2933 offset,
2934 length: u32::try_from(bytes.len())
2935 .map_err(|_| invalid("sieve page length overflow"))?,
2936 hash: checksum(&bytes),
2937 });
2938 }
2939 let mut part_ranges = vec![None; width];
2945 if parts > 1 {
2946 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2947 let bytes = encode_part_ranges(&stripe.ranges)?;
2948 if bytes.len() >= span.length as usize {
2949 continue;
2950 }
2951 let offset = self.at;
2952 self.put(&bytes)?;
2953 *page = Some(Page {
2954 offset,
2955 length: u32::try_from(bytes.len())
2956 .map_err(|_| invalid("part range page length overflow"))?,
2957 hash: checksum(&bytes),
2958 });
2959 }
2960 }
2961 let offset = self.at;
2962 self.put(&index)?;
2963 let index = Span {
2964 offset,
2965 length: u32::try_from(index.len())
2966 .map_err(|_| invalid("index page length overflow"))?,
2967 };
2968 let mut rows = 0_usize;
2969 let mut lengths = Vec::with_capacity(parts);
2970 let mut span = None;
2971 for part in held {
2972 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2973 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2974 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2975 }
2976 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2977 self.table.stripes.push(Stripe {
2978 rows,
2979 parts: lengths,
2980 index,
2981 pages,
2982 memberships: Pages::from_slots(memberships)?,
2983 sieves: Pages::from_slots(sieves)?,
2984 part_ranges: Pages::from_slots(part_ranges)?,
2985 zone: Zone::from_ranges(ranges),
2986 });
2987 drop(timing);
2988 if let Some(profile) = &profile {
2989 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2990 }
2991 Ok(())
2992 }
2993
2994 fn numeric_frequency(
3014 &self,
3015 column: usize,
3016 counted: bool,
3017 dense: Option<(u64, usize)>,
3018 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
3019 let signed = match self.table.fields[column].ty {
3020 LogicalType::TinyInt
3021 | LogicalType::SmallInt
3022 | LogicalType::Integer
3023 | LogicalType::BigInt
3024 | LogicalType::Date
3025 | LogicalType::Timestamp => true,
3026 LogicalType::UTinyInt
3027 | LogicalType::USmallInt
3028 | LogicalType::UInteger
3029 | LogicalType::UBigInt => false,
3030 _ => return Ok((None, None)),
3031 };
3032 let value_of = |bits: Option<u64>| match bits {
3033 None => FrequencyValue::Null,
3034 Some(bits) => integer_value(bits, signed),
3035 };
3036 let tallied = self
3041 .gathers
3042 .get(column)
3043 .and_then(Option::as_ref)
3044 .filter(|gather| gather.rows() == self.table.rows as u64)
3045 .and_then(stats::Gather::frequencies)
3046 .and_then(|(values, nulls)| {
3047 let entries = values
3048 .iter()
3049 .map(|(value, count)| {
3050 let value = value_of(Some(frequency_bits(value)?));
3051 Some(FrequencyEntry { value, count: *count })
3052 })
3053 .chain((nulls != 0).then_some(Some(FrequencyEntry {
3054 value: FrequencyValue::Null,
3055 count: nulls,
3056 })))
3057 .collect::<Option<Vec<_>>>()?;
3058 Some((entries, values.len() as u64))
3059 });
3060 let exact = match (&tallied, counted) {
3064 (None, true) => self.exact_frequency(column, signed, dense)?,
3065 _ => None,
3066 };
3067 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3068 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3069 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3070 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3071 (None, None) => {
3072 let mut first = Candidates::default();
3076 let mut run = Run::default();
3077 self.visit_numeric(column, signed, |_, bits| {
3078 if let Some((ended, times)) = run.push(bits) {
3079 first.add(ended, times);
3080 }
3081 })?;
3082 if let Some((bits, times)) = run.take() {
3083 first.add(bits, times);
3084 }
3085 let (nulls, decrements) = (first.nulls, first.decrements);
3088 let distinct_count = (decrements == 0).then_some(first.held as u64);
3089 let (exact, null_count) = if decrements == 0 {
3090 let exact = first
3091 .pairs()
3092 .map(|(bits, count)| (bits, u64::from(count)))
3093 .collect::<FrequencyMap<_>>();
3094 (exact, (nulls != 0).then_some(u64::from(nulls)))
3095 } else {
3096 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3097 if nulls != 0 {
3098 lower.push(nulls);
3099 }
3100 lower.sort_unstable_by(|left, right| right.cmp(left));
3101 if lower.len() < FREQUENCY_BUILD_RANK
3102 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3103 {
3104 return Ok((None, distinct_count));
3105 }
3106 let mut recounts = vec![0_u64; first.slots.len()];
3109 let mut null_count = (nulls != 0).then_some(0_u64);
3110 let mut recount = |bits: Option<u64>, times: u32| {
3111 let held = match bits {
3112 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3113 None => null_count.as_mut(),
3114 };
3115 if let Some(count) = held {
3116 *count = count.saturating_add(u64::from(times));
3117 }
3118 };
3119 let mut run = Run::default();
3120 self.visit_numeric(column, signed, |_, bits| {
3121 if let Some((bits, times)) = run.push(bits) {
3122 recount(bits, times);
3123 }
3124 })?;
3125 if let Some((bits, times)) = run.take() {
3126 recount(bits, times);
3127 }
3128 let exact = first
3129 .slots
3130 .iter()
3131 .zip(&recounts)
3132 .filter(|(slot, _)| slot.count != 0)
3133 .map(|(slot, &count)| (slot.bits, count))
3134 .collect::<FrequencyMap<_>>();
3135 (exact, null_count)
3136 };
3137 let entries = exact
3138 .into_iter()
3139 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3140 .chain(
3141 null_count
3142 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3143 )
3144 .collect::<Vec<_>>();
3145 (entries, decrements, distinct_count)
3146 }
3147 };
3148 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3149 if omitted_max == 0 && entries.len() > 1 {
3153 let retained = entries.len().saturating_sub(1).min(2);
3154 omitted_max = entries[retained].count;
3155 entries.truncate(retained);
3156 }
3157 let mut covered = 0;
3161 let mut kept_rows = 0_u64;
3162 for entry in &entries {
3163 match kept_rows.checked_add(entry.count) {
3164 Some(total) if total <= FREQUENCY_ORDINALS as u64 => kept_rows = total,
3165 _ => break,
3166 }
3167 covered += 1;
3168 }
3169 let ordinal_bound = entries.get(covered).map_or(0, |entry| entry.count);
3170 let worth_keeping = covered == entries.len()
3171 || (covered >= FREQUENCY_BUILD_RANK
3172 && entries[FREQUENCY_BUILD_RANK - 1].count > ordinal_bound.max(omitted_max));
3173 let mut ordinals = Vec::new();
3174 let mut ordinal_entries = Vec::new();
3175 if worth_keeping {
3176 let mut kept = FrequencyMap::default();
3177 let mut null_kept = None;
3178 for (at, entry) in entries.iter().enumerate().take(covered) {
3179 let at = u16::try_from(at)
3180 .map_err(|_| invalid("too many retained frequency entries"))?;
3181 match entry.value {
3182 FrequencyValue::Integer(value) => {
3183 kept.insert(value as u64, at);
3184 }
3185 FrequencyValue::Null => null_kept = Some(at),
3186 FrequencyValue::Code(_) => {}
3187 }
3188 }
3189 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3190 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3191 self.visit_numeric(column, signed, |ordinal, bits| {
3192 let held = match bits {
3193 Some(bits) => kept.get(&bits).copied(),
3194 None => null_kept,
3195 };
3196 if let Some(entry) = held {
3197 ordinals.push(ordinal);
3198 ordinal_entries.push(entry);
3199 }
3200 })?;
3201 }
3202 Ok((
3203 Some(FrequencySummary {
3204 entries,
3205 omitted_max,
3206 ordinals,
3207 ordinal_entries,
3208 ordinal_bound: if worth_keeping { ordinal_bound } else { 0 },
3209 }),
3210 distinct_count,
3211 ))
3212 }
3213
3214 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3221 let rows = self.table.rows;
3222 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3223 return None;
3224 }
3225 let (low, high) = gather.span()?;
3226 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3227 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3228 let bits = low as u64;
3229 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3230 }
3231
3232 fn exact_frequency(
3246 &self,
3247 column: usize,
3248 signed: bool,
3249 dense: Option<(u64, usize)>,
3250 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3251 if let Some((low, len)) = dense {
3254 let mut counts = distinct::DenseCounts::new(low, len);
3255 let nulls =
3256 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3257 if let Some(distinct) = counts.count() {
3258 let Some(distinct) = distinct else { return Ok(None) };
3259 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3260 counts.visit(visit);
3261 })));
3262 }
3263 }
3264 let mut set = distinct::ExactCounts::new();
3265 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3266 let Some(distinct) = set.count() else {
3267 return Ok(None);
3268 };
3269 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3270 set.visit(visit);
3271 })))
3272 }
3273
3274 fn count_numeric(
3277 &self,
3278 column: usize,
3279 signed: bool,
3280 mut add: impl FnMut(u64, u32),
3281 ) -> Result<u64> {
3282 let mut nulls = 0_u64;
3283 let mut run = Run::default();
3284 let mut take = |bits: Option<u64>, times: u32| match bits {
3285 Some(bits) => add(bits, times),
3286 None => nulls += u64::from(times),
3287 };
3288 self.visit_numeric(column, signed, |_, bits| {
3289 if let Some((bits, times)) = run.push(bits) {
3290 take(bits, times);
3291 }
3292 })?;
3293 if let Some((bits, times)) = run.take() {
3294 take(bits, times);
3295 }
3296 Ok(nulls)
3297 }
3298
3299 fn frequent_entries(
3302 &self,
3303 signed: bool,
3304 distinct: u64,
3305 nulls: u64,
3306 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3307 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3308 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3310 let mut rank = |count: u64| {
3311 if top.len() <= FREQUENCY_ENTRIES {
3312 top.push(Reverse(count));
3313 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3314 top.pop();
3315 top.push(Reverse(count));
3316 }
3317 };
3318 visit(&mut |_, count| rank(count));
3319 if nulls != 0 {
3320 rank(nulls);
3321 }
3322 let top = top.into_sorted_vec();
3323 let values = distinct + u64::from(nulls != 0);
3324 if values > FREQUENCY_CANDIDATES as u64 {
3325 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3326 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3327 return (None, distinct);
3328 }
3329 }
3330 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3331 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3332 visit(&mut |bits, count| {
3333 if count >= least {
3334 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3335 }
3336 });
3337 if nulls != 0 && nulls >= least {
3338 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3339 }
3340 (Some(entries), distinct)
3341 }
3342
3343 fn visit_numeric(
3350 &self,
3351 column: usize,
3352 signed: bool,
3353 mut visit: impl FnMut(u64, Option<u64>),
3354 ) -> Result<()> {
3355 let ty = &self.table.fields[column].ty;
3356 let mut start = 0_u64;
3357 let mut block = Vec::new();
3358 for stripe in &self.table.stripes {
3359 let spans = read_index(&self.file, stripe, column)?;
3360 let page = stripe.pages[column];
3361 let mut bytes = vec![0; page.length as usize];
3362 read_at(&self.file, page.offset, &mut bytes)?;
3363 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3364 let part = part_bytes(&bytes, *span)?;
3365 if checksum(part) != span.hash {
3366 return Err(invalid("column page checksum differs while building frequencies"));
3367 }
3368 let rows = rows as usize;
3369 let vector = decode(ty, rows, part, None)?;
3370 if signed && vector.signed_block(&mut block) && block.len() == rows {
3374 if vector.none_null() {
3375 for (row, &value) in block.iter().enumerate() {
3376 visit(start.saturating_add(row as u64), Some(value as u64));
3377 }
3378 } else {
3379 for (row, &value) in block.iter().enumerate() {
3380 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3381 visit(start.saturating_add(row as u64), bits);
3382 }
3383 }
3384 start = start.saturating_add(rows as u64);
3385 continue;
3386 }
3387 for row in 0..rows {
3389 let bits = if vector.is_null_at(row) {
3390 None
3391 } else {
3392 let widened = match vector.signed_at(row) {
3396 Some(value) => Some(value as u64),
3397 None => match vector.value_at(row) {
3398 Value::UTinyInt(value) => Some(u64::from(value)),
3399 Value::USmallInt(value) => Some(u64::from(value)),
3400 Value::UInteger(value) => Some(u64::from(value)),
3401 Value::UBigInt(value) => Some(value),
3402 _ => None,
3403 },
3404 };
3405 Some(widened.ok_or_else(|| {
3406 invalid("numeric frequency page did not contain an integer value")
3407 })?)
3408 };
3409 visit(start.saturating_add(row as u64), bits);
3410 }
3411 start = start.saturating_add(rows as u64);
3412 }
3413 }
3414 Ok(())
3415 }
3416
3417 fn numeric_columns(&self) -> Vec<usize> {
3419 self.table
3420 .fields
3421 .iter()
3422 .enumerate()
3423 .filter_map(|(column, field)| {
3424 matches!(
3425 field.ty,
3426 LogicalType::TinyInt
3427 | LogicalType::SmallInt
3428 | LogicalType::Integer
3429 | LogicalType::BigInt
3430 | LogicalType::UTinyInt
3431 | LogicalType::USmallInt
3432 | LogicalType::UInteger
3433 | LogicalType::UBigInt
3434 | LogicalType::Date
3435 | LogicalType::Timestamp
3436 )
3437 .then_some(column)
3438 })
3439 .collect()
3440 }
3441
3442 #[allow(dead_code)]
3444 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3445 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3446 return Ok(None);
3447 }
3448 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3449 return Err(invalid("frequency ordinals are not sorted and unique"));
3450 }
3451 let mut out = Vec::with_capacity(ordinals.len());
3452 let mut wanted = 0;
3453 let mut stripe_start = 0_u64;
3454 for stripe in &self.table.stripes {
3455 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3456 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3457 stripe_start = stripe_end;
3458 continue;
3459 }
3460 let spans = read_index(&self.file, stripe, column)?;
3461 let page = stripe.pages[column];
3462 let mut bytes = vec![0; page.length as usize];
3463 read_at(&self.file, page.offset, &mut bytes)?;
3464 let mut part_start = stripe_start;
3465 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3466 let part_end = part_start.saturating_add(u64::from(rows));
3467 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3468 let part = part_bytes(&bytes, *span)?;
3469 if checksum(part) != span.hash {
3470 return Err(invalid(
3471 "column page checksum differs while building pair frequencies",
3472 ));
3473 }
3474 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3475 let positions = ordinals[wanted..upto]
3476 .iter()
3477 .map(|&ordinal| {
3478 usize::try_from(ordinal.saturating_sub(part_start))
3479 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3480 })
3481 .collect::<Result<Vec<_>>>()?;
3482 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3483 return Ok(None);
3484 }
3485 wanted = upto;
3486 }
3487 part_start = part_end;
3488 }
3489 stripe_start = stripe_end;
3490 }
3491 if wanted != ordinals.len() {
3492 return Err(invalid("frequency ordinal is outside the table"));
3493 }
3494 Ok(Some(out))
3495 }
3496
3497 #[allow(dead_code)]
3499 fn pair_frequencies(
3500 &self,
3501 frequencies: &[Option<Frequencies>],
3502 ) -> Result<Vec<PairFrequencySummary>> {
3503 let anchors = frequencies
3504 .iter()
3505 .enumerate()
3506 .filter_map(|(column, summary)| {
3507 match summary {
3509 Some(Frequencies::Held(summary)) => Some(summary),
3510 _ => None,
3511 }
3512 .filter(|summary| {
3513 !summary.ordinals.is_empty()
3514 && summary.ordinal_entries.len() == summary.ordinals.len()
3515 && summary.ordinal_bound == 0
3516 })
3517 .cloned()
3518 .map(|summary| (column, summary))
3519 })
3520 .collect::<Vec<_>>();
3521 let strings = self
3522 .dictionaries
3523 .iter()
3524 .enumerate()
3525 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3526 .collect::<Vec<_>>();
3527 let mut summaries = Vec::new();
3528 for (first, anchors) in anchors {
3529 for &second in &strings {
3530 if summaries.len() == MAX_PAIR_FREQUENCIES {
3531 return Ok(summaries);
3532 }
3533 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3534 continue;
3535 };
3536 if codes.len() != anchors.ordinal_entries.len() {
3537 return Err(invalid("pair frequency columns have different lengths"));
3538 }
3539 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3540 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3541 *counts.entry((anchor, code)).or_default() += 1;
3542 }
3543 let mut entries = counts
3544 .into_iter()
3545 .map(|((first_entry, second), count)| PairFrequencyEntry {
3546 first_entry,
3547 second,
3548 count,
3549 })
3550 .collect::<Vec<_>>();
3551 entries.sort_unstable_by(|left, right| {
3552 right
3553 .count
3554 .cmp(&left.count)
3555 .then_with(|| left.first_entry.cmp(&right.first_entry))
3556 .then_with(|| left.second.cmp(&right.second))
3557 });
3558 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3559 entries.truncate(FREQUENCY_ENTRIES);
3560 summaries.push(PairFrequencySummary {
3561 first: u16::try_from(first)
3562 .map_err(|_| invalid("pair frequency column index overflows"))?,
3563 second: u16::try_from(second)
3564 .map_err(|_| invalid("pair frequency column index overflows"))?,
3565 entries,
3566 omitted_max: anchors.omitted_max.max(pair_omitted),
3567 });
3568 }
3569 }
3570 Ok(summaries)
3571 }
3572
3573 fn close(&mut self) -> Result<Entry> {
3584 self.reclaim()?;
3585 self.flush_pending()?;
3586 let profile = self.profile.clone();
3590 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3591 let before = self.at;
3592 let mut stripes = std::mem::take(&mut self.order)
3593 .into_iter()
3594 .zip(std::mem::take(&mut self.table.stripes))
3595 .collect::<Vec<_>>();
3596 stripes.sort_by_key(|(order, _)| order.0);
3597 let mut previous: Option<(u64, u64)> = None;
3598 for ((first, last), _) in &stripes {
3599 if previous.is_some_and(|previous| previous >= *first) {
3600 return Err(invalid("chunks did not arrive in source order"));
3601 }
3602 previous = Some(*last);
3603 }
3604 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3605 drop(timing);
3606 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3607 let placing = self.at;
3608 finish_dictionaries(&mut self.dictionaries)?;
3609 self.place_blocks()?;
3610 for dictionary in self.dictionaries.iter_mut().flatten() {
3611 dictionary.release_lookup();
3612 dictionary.recharge(profile.as_deref());
3613 }
3614 let (numeric, closed) = self.close_columns()?;
3615 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3616 numeric.into_iter().unzip();
3617 let frequencies =
3618 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3619 let pairs = Vec::new();
3621 self.table.frequencies = frequencies;
3622 self.table.distincts = distincts;
3623 self.table.pair_frequencies = pairs;
3624 if let Some(profile) = &profile {
3625 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3626 }
3627 self.table.demoted = self
3628 .dictionaries
3629 .iter()
3630 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3631 .collect();
3632 if !self.table.demoted.contains(&true) {
3633 self.table.demoted = Vec::new();
3634 }
3635 self.dictionaries = Vec::new();
3636 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3637 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3638 self.table.host_groups = None;
3639 for (index, closed) in closed.into_iter().enumerate() {
3640 let Some(closed) = closed else { continue };
3641 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3642 self.table.distincts[index] = distinct;
3643 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3644 self.table.frequency_texts[index] = texts;
3645 if hosts.is_some() {
3646 self.table.host_groups = hosts;
3647 }
3648 let offset = self.at;
3649 self.put(&encoded.index)?;
3650 self.put(&encoded.ranks)?;
3651 self.put(&encoded.grams)?;
3652 self.table.dictionary_payloads[index] = payload;
3653 let length = encoded
3654 .index
3655 .len()
3656 .checked_add(encoded.ranks.len())
3657 .and_then(|len| len.checked_add(encoded.grams.len()))
3658 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3659 self.table.dictionaries[index] = Some(Page {
3660 offset,
3661 length: u32::try_from(length)
3662 .map_err(|_| invalid("dictionary page length overflow"))?,
3663 hash: checksum(&encoded.index),
3664 });
3665 }
3666 drop(timing);
3667 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3668 let placed = self.at - placing;
3669 self.write_stats()?;
3670 let directory = encode_directory(&self.table)?;
3671 if directory.len() > MAX_DIRECTORY {
3672 return Err(invalid("directory exceeds the configured bound"));
3673 }
3674 let offset = self.at;
3675 self.put(&directory)?;
3676 drop(timing);
3677 if let Some(profile) = &profile {
3678 profile.moved(Stage::Dictionary, 0, placed, 0);
3679 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3680 }
3681 Ok(Entry {
3682 name: self.table.name.clone(),
3683 fields: self.table.fields.clone(),
3684 rows: self.table.rows,
3685 nonzero: vec![None; self.table.fields.len()],
3686 aggregates: table_aggregate_sums(&self.table),
3687 distincts: self.table.distincts.clone(),
3688 extremes: table_integer_extremes(&self.table),
3689 frequencies: table_complete_numeric_frequencies(&self.table),
3690 directory: Page {
3691 offset,
3692 length: u32::try_from(directory.len())
3693 .map_err(|_| invalid("directory length overflow"))?,
3694 hash: checksum(&directory),
3695 },
3696 })
3697 }
3698
3699 #[allow(clippy::type_complexity)]
3716 fn close_columns(
3717 &self,
3718 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3719 let numeric = self.numeric_columns().into_iter().map(|column| {
3720 let gather = self.gathers.get(column).and_then(Option::as_ref);
3721 let estimate = gather.and_then(stats::Gather::distinct);
3722 let counted = !estimate.is_some_and(distinct::beyond);
3723 let set =
3724 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3725 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3726 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3727 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3728 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3729 });
3730 let dictionaries =
3731 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3732 let dictionary = dictionary.as_ref()?;
3733 let bytes = dictionary.closing_bytes();
3734 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3735 });
3736 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3737 jobs.sort_by_key(|&(_, _, cost)| cost);
3738 let columns = self.table.fields.len();
3739 let mut frequencies = vec![(None, None); columns];
3740 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3741 let profile = self.profile.as_deref();
3742 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3743 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3744 let closed = match job {
3745 Closing::Numeric { column, counted, dense } => {
3746 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3747 Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3748 }
3749 Closing::Dictionary { index, dictionary } => {
3750 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3751 Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3752 }
3753 };
3754 rudb_common::heap::release();
3757 Ok(closed)
3758 };
3759 let workers = close_workers().min(jobs.len());
3760 let pieces = if workers <= 1 {
3761 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3762 } else {
3763 let state = Mutex::new((jobs, 0_usize));
3765 let finished = Condvar::new();
3766 std::thread::scope(|scope| {
3767 (0..workers)
3768 .map(|_| {
3769 scope.spawn(|| {
3770 let mut mine = Vec::new();
3771 loop {
3772 let mut held = state.lock().map_err(|_| {
3773 Error::internal("a native close worker panicked")
3774 })?;
3775 let (job, bytes) = loop {
3776 let (jobs, busy) = &mut *held;
3777 if jobs.is_empty() {
3778 return Ok(mine);
3779 }
3780 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3781 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3782 });
3783 if let Some(at) = fits {
3784 let (job, bytes, _) = jobs.remove(at);
3785 *busy += bytes;
3786 break (job, bytes);
3787 }
3788 held = finished.wait(held).map_err(|_| {
3789 Error::internal("a native close worker panicked")
3790 })?;
3791 };
3792 drop(held);
3793 let _room = Room { state: &state, finished: &finished, bytes };
3796 mine.push(run(job, bytes)?);
3797 }
3798 })
3799 })
3800 .collect::<Vec<_>>()
3801 .into_iter()
3802 .map(|handle| {
3803 handle
3804 .join()
3805 .map_err(|_| Error::internal("a native close worker panicked"))?
3806 })
3807 .collect::<Result<Vec<_>>>()
3808 })?
3809 .into_iter()
3810 .flatten()
3811 .collect()
3812 };
3813 for piece in pieces {
3814 match piece {
3815 Closed::Numeric(column, summary) => frequencies[column] = summary,
3816 Closed::Dictionary(index, one) => closed[index] = Some(one),
3817 }
3818 }
3819 Ok((frequencies, closed))
3820 }
3821
3822 fn close_dictionary(
3829 &self,
3830 _index: usize,
3831 dictionary: &GlobalDictionary,
3832 ) -> Result<ClosedDictionary> {
3833 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3834 let (distinct, frequencies, texts) = if dictionary.demoted {
3839 (None, None, Vec::new())
3840 } else {
3841 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3842 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3843 (Some(distinct), Some(frequencies), texts)
3844 };
3845 let hosts = None;
3847 drop(flat);
3848 drop(bases);
3849 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3850 let payload = dictionary
3851 .placed
3852 .iter()
3853 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3854 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3855 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3856 }
3857
3858 fn write_stats(&mut self) -> Result<()> {
3870 let gathers = std::mem::take(&mut self.gathers);
3871 let rows = self.table.rows as u64;
3872 let mut payloads = Vec::new();
3873 for (column, gather) in gathers.into_iter().enumerate() {
3874 let Some(gather) = gather else { continue };
3875 if gather.rows() != rows {
3881 continue;
3882 }
3883 let Some(stats) = gather.finish() else { continue };
3884 let mut summary = Vec::new();
3885 stats.summary.encode(&mut summary)?;
3886 let mut sketches = Vec::new();
3887 stats.sketches.encode(&mut sketches)?;
3888 payloads.push((column, summary, sketches));
3889 }
3890 if payloads.is_empty() {
3891 return Ok(());
3892 }
3893 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3894 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3895 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3896 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3899 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3900 if !built {
3901 continue;
3902 }
3903 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3904 let sections = [
3905 (*section::SUMMARY, summary, summary.len() as u32),
3908 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3909 ];
3910 let wanted = 1 + usize::from(sketched);
3911 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3912 let written = write_section(
3913 &*self.file,
3914 &mut self.at,
3915 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3916 self.generation,
3917 )?;
3918 self.table.sections.push(written);
3919 }
3920 }
3921 if self.table.sections.len() > MAX_SECTIONS {
3922 return Err(invalid("the table would name more sections than the bound allows"));
3923 }
3924 Ok(())
3925 }
3926
3927 pub fn finish(mut self) -> Result<Table> {
3937 let entry = self.close()?;
3938 let profile = self.profile.take();
3939 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3940 let mut tables = std::mem::take(&mut self.closed);
3941 tables.push(entry);
3942 let catalog =
3943 encode_catalog(&tables, &self.views, self.card.as_ref(), self.anchor.as_ref())?;
3944 if catalog.len() > MAX_DIRECTORY {
3945 return Err(invalid("catalog exceeds the configured bound"));
3946 }
3947 let offset = self.at;
3948 self.put(&catalog)?;
3949 if let Some(profile) = &profile {
3950 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3951 }
3952 synced(&*self.file, profile.as_deref())?;
3956 let slot = Slot {
3957 offset,
3958 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3959 generation: self.generation,
3960 hash: checksum(&catalog),
3961 };
3962 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3967 synced(&*self.file, profile.as_deref())?;
3968 Ok(self.table)
3969 }
3970
3971 pub fn restate(
3990 path: impl AsRef<Path>,
3991 views: &[ViewEntry],
3992 anchor: Option<&LogAnchor>,
3993 ) -> Result<()> {
3994 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3995 let size = file.len()?;
3996 let (slot, bytes, _) = committed_slot(&*file, size)?;
3997 let (closed, _, card, held) = decode_catalog(&bytes, size)?;
3998 let anchor = anchor.cloned().or(held);
3999 let generation = slot
4000 .generation
4001 .checked_add(1)
4002 .ok_or_else(|| invalid("native file generation overflow"))?;
4003 let catalog = encode_catalog(
4004 &closed,
4005 views,
4006 card_for(path.as_ref(), card).as_ref(),
4007 anchor.as_ref(),
4008 )?;
4009 if catalog.len() > MAX_DIRECTORY {
4010 return Err(invalid("catalog exceeds the configured bound"));
4011 }
4012 file.write_at(size, &catalog)?;
4013 file.sync()?;
4014 let slot = Slot {
4015 offset: size,
4016 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4017 generation,
4018 hash: checksum(&catalog),
4019 };
4020 file.write_at(slot_offset(generation), &slot.bytes())?;
4021 file.sync()?;
4022 Ok(())
4023 }
4024
4025 pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
4036 let path = path.as_ref();
4037 let (_, size, _, bytes, _) = slot_bytes(path)?;
4038 let (_, views, held, _) = decode_catalog(&bytes, size)?;
4039 if card_for(path, held.clone()) == held {
4040 return Ok(());
4041 }
4042 Self::restate(path, &views, None)
4043 }
4044
4045 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
4048 let path = path.as_ref();
4049 let (_, size, slot, bytes, _) = slot_bytes(path)?;
4050 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4051 let native = Catalog::open(path)?;
4052 for entry in &mut entries {
4053 let reader = native.table(&entry.name)?;
4054 entry.nonzero.fill(None);
4055 entry.aggregates = reader_aggregate_sums(&reader)?;
4056 entry.distincts = (0..entry.fields.len())
4057 .map(|column| reader.distinct_values(column))
4058 .collect::<Result<Vec<_>>>()?;
4059 entry.extremes = reader_integer_extremes(&reader)?;
4060 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
4061 }
4062 let generation = slot
4063 .generation
4064 .checked_add(1)
4065 .ok_or_else(|| invalid("native file generation overflow"))?;
4066 let catalog =
4067 encode_catalog(&entries, &views, card_for(path, card).as_ref(), anchor.as_ref())?;
4068 if catalog.len() > MAX_DIRECTORY {
4069 return Err(invalid("catalog exceeds the configured bound"));
4070 }
4071 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
4072 file.write_at(size, &catalog)?;
4073 file.sync()?;
4074 let slot = Slot {
4075 offset: size,
4076 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4077 generation,
4078 hash: checksum(&catalog),
4079 };
4080 file.write_at(slot_offset(generation), &slot.bytes())?;
4081 file.sync()?;
4082 Ok(())
4083 }
4084
4085 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4087 Self::certify_summaries(path)
4088 }
4089}
4090
4091fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4097 let offset = *at;
4098 file.write_at(offset, bytes)?;
4099 *at =
4100 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4101 Ok(offset)
4102}
4103
4104fn write_section(
4110 file: &dyn rudb_io::File,
4111 at: &mut u64,
4112 one: §ion::Attachment<'_>,
4113 generation: u64,
4114) -> Result<Section> {
4115 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4119 return Err(invalid("a section's header is longer than its payload"));
4120 }
4121 let mut extents = Vec::new();
4122 let mut first = 0_u64;
4123 let extent_size =
4124 if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4125 run_projection::RLE_PAGE_BYTES
4126 } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4127 1 << 19
4128 } else {
4129 section::MAX_EXTENT as usize
4130 };
4131 for chunk in one.bytes.chunks(extent_size) {
4132 let offset = append(file, at, chunk)?;
4133 extents.push(section::Extent {
4134 offset,
4135 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4136 hash: checksum(chunk),
4137 first,
4138 });
4139 first += chunk.len() as u64;
4140 }
4141 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4142 section::encode_extents(&extents, &mut table)?;
4143 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4147 Ok(Section {
4148 kind: one.kind,
4149 id: one.id,
4150 generation,
4151 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4152 extent_page,
4153 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4154 hash: checksum(&table),
4155 flags: one.flags,
4156 header_bytes: one.header_bytes,
4157 })
4158}
4159
4160pub fn attach(
4184 path: impl AsRef<Path>,
4185 table: &str,
4186 attachments: &[section::Attachment<'_>],
4187) -> Result<Table> {
4188 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4189 let file = &*file;
4190 let size = file.len()?;
4191 let (slot, bytes, _) = committed_slot(file, size)?;
4192 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4193 let card = card_for(path.as_ref(), card);
4194 let at = entries
4195 .iter()
4196 .position(|entry| entry.name == table)
4197 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4198 let mut version = [0; 4];
4199 read_at(file, 8, &mut version)?;
4200 let version = u32::from_le_bytes(version);
4201 if version != FORMAT {
4207 return Err(invalid(&format!(
4208 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4209 to be written again"
4210 )));
4211 }
4212 let mut directory = vec![0; entries[at].directory.length as usize];
4213 read_at(file, entries[at].directory.offset, &mut directory)?;
4214 if checksum(&directory) != entries[at].directory.hash {
4215 return Err(invalid(&format!("the directory of table {table} does not checksum")));
4216 }
4217 let mut held = decode_directory(&directory, size)?;
4218 let mut cursor = size;
4219 for one in attachments {
4220 let written = write_section(file, &mut cursor, one, held.generation)?;
4221 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4222 held.sections.push(written);
4223 }
4224 if held.sections.len() > MAX_SECTIONS {
4225 return Err(invalid("the table would name more sections than the bound allows"));
4226 }
4227 let encoded = encode_directory(&held)?;
4228 if encoded.len() > MAX_DIRECTORY {
4229 return Err(invalid("directory exceeds the configured bound"));
4230 }
4231 let offset = append(file, &mut cursor, &encoded)?;
4232 entries[at].directory = Page {
4233 offset,
4234 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4235 hash: checksum(&encoded),
4236 };
4237 let catalog = encode_catalog(&entries, &views, card.as_ref(), anchor.as_ref())?;
4240 if catalog.len() > MAX_DIRECTORY {
4241 return Err(invalid("catalog exceeds the configured bound"));
4242 }
4243 let offset = append(file, &mut cursor, &catalog)?;
4244 file.sync()?;
4245 let generation =
4246 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4247 let committed = Slot {
4248 offset,
4249 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4250 generation,
4251 hash: checksum(&catalog),
4252 };
4253 file.write_at(slot_offset(generation), &committed.bytes())?;
4254 file.sync()?;
4255 Ok(held)
4256}
4257
4258type Synopsis = Arc<Vec<(Value, u64)>>;
4261
4262#[derive(Debug, Clone)]
4264pub struct Reader {
4265 file: Arc<File>,
4266 table: Arc<Table>,
4267 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4268 loading: Arc<Vec<Mutex<()>>>,
4277 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4280 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4284 frequency_heads: Arc<Vec<OnceLock<Arc<FrequencyHead>>>>,
4288 summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4290 opened: Arc<AtomicUsize>,
4294 sieves: Arc<Vec<Vec<SieveSlot>>>,
4298 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4301 places: Arc<Vec<Place>>,
4303 cache: Arc<Shelf>,
4304 pool: PagePool,
4306 pages: Arc<AtomicUsize>,
4309 indexes: Arc<AtomicUsize>,
4312 verified: Arc<Vec<AtomicU64>>,
4322 text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4325 firsts: Arc<Vec<usize>>,
4327 size: u64,
4329 directory: u64,
4331 opening: Opening,
4333}
4334
4335#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4347pub struct Opening {
4348 pub reads: u32,
4351 pub bytes: u64,
4353}
4354
4355#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4357pub struct Reads {
4358 pub opening: Opening,
4360 pub pages: usize,
4362 pub indexes: usize,
4364 pub dictionaries: usize,
4367}
4368
4369#[derive(Debug, Clone, Copy)]
4371struct Place {
4372 stripe: u32,
4373 part: u32,
4374 rows: u32,
4375}
4376
4377#[derive(Debug, Clone, Copy)]
4379struct PartSpan {
4380 start: usize,
4381 length: usize,
4382 hash: u64,
4383}
4384
4385#[derive(Debug, Clone)]
4391struct CachedColumn {
4392 stripe: usize,
4393 index: Arc<Vec<PartSpan>>,
4394 page: Option<Arc<HeldPage>>,
4395}
4396
4397#[derive(Debug)]
4404struct HeldPage {
4405 bytes: Vec<u8>,
4406 checked: Vec<AtomicBool>,
4407}
4408
4409impl HeldPage {
4410 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4412 let bytes = part_bytes(&self.bytes, span)?;
4413 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4414 if !checked.load(Atomic::Relaxed) {
4415 verify_part(bytes, span)?;
4416 checked.store(true, Atomic::Relaxed);
4417 }
4418 Ok(bytes)
4419 }
4420}
4421
4422fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4424 let got = checksum(bytes);
4425 if got != span.hash {
4426 return Err(invalid(&format!(
4427 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4428 span.start, span.length, span.hash,
4429 )));
4430 }
4431 Ok(())
4432}
4433
4434#[derive(Debug, Default)]
4470struct Cached {
4471 pages: Vec<Option<Resident>>,
4472 loading: Vec<usize>,
4473 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4474 touched: Vec<Vec<u64>>,
4475 passing: VecDeque<usize>,
4476}
4477
4478#[derive(Debug, Clone)]
4480struct Resident {
4481 page: Arc<HeldPage>,
4482 used: Arc<AtomicBool>,
4483}
4484
4485#[derive(Debug)]
4487struct Shelf {
4488 columns: Vec<Mutex<Cached>>,
4489 held: Vec<AtomicUsize>,
4492 kept: AtomicUsize,
4495}
4496
4497#[derive(Debug, Clone, Default)]
4516pub struct PagePool {
4517 ring: Arc<Mutex<Ring>>,
4518 budget: Arc<AtomicUsize>,
4519}
4520
4521#[derive(Debug, Default)]
4522struct Ring {
4523 held: VecDeque<Held>,
4524 bytes: usize,
4525}
4526
4527#[derive(Debug)]
4532struct Held {
4533 shelf: Weak<Shelf>,
4534 column: usize,
4535 stripe: usize,
4536 bytes: usize,
4537 used: Arc<AtomicBool>,
4538}
4539
4540impl PagePool {
4541 #[must_use]
4543 pub fn new(budget: usize) -> Self {
4544 let pool = Self::default();
4545 pool.budget.store(budget, Atomic::Relaxed);
4546 pool
4547 }
4548
4549 #[must_use]
4555 pub fn bytes(&self) -> usize {
4556 self.ring.lock().map_or(0, |ring| ring.bytes)
4557 }
4558
4559 fn admit(&self, held: Held) {
4565 let budget = self.budget.load(Atomic::Relaxed);
4566 let mut gone = Vec::new();
4567 {
4568 let Ok(mut ring) = self.ring.lock() else { return };
4569 ring.bytes += held.bytes;
4570 ring.held.push_back(held);
4571 let mut looked = 0;
4574 let limit = ring.held.len();
4575 while ring.bytes > budget && looked < limit {
4576 looked += 1;
4577 let Some(entry) = ring.held.pop_front() else { break };
4578 let Some(shelf) = entry.shelf.upgrade() else {
4579 ring.bytes -= entry.bytes;
4580 continue;
4581 };
4582 if entry.used.swap(false, Atomic::Relaxed) {
4583 ring.held.push_back(entry);
4584 continue;
4585 }
4586 let count = &shelf.held[entry.column];
4587 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4588 ring.held.push_back(entry);
4589 continue;
4590 }
4591 count.fetch_sub(1, Atomic::Relaxed);
4592 ring.bytes -= entry.bytes;
4593 gone.push((shelf, entry));
4594 }
4595 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4598 if let Some(entry) = ring.held.pop_front() {
4599 ring.bytes -= entry.bytes;
4600 }
4601 }
4602 }
4603 for (shelf, entry) in gone {
4604 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4605 if let Some(slot) = cached.pages.get_mut(entry.stripe)
4606 && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4607 {
4608 *slot = None;
4609 }
4610 }
4611 }
4612}
4613
4614const CACHED_STRIPES_PER_COLUMN: usize = 4;
4626
4627type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4629
4630type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4631
4632#[derive(Debug)]
4633struct NativeText {
4634 file: Arc<File>,
4635 values: usize,
4637 offsets: Vec<u8>,
4649 offset_bits: usize,
4652 value_ends: OnceLock<Option<Vec<u32>>>,
4665 value_lens: OnceLock<Option<Lengths>>,
4675 ends_asked: AtomicUsize,
4681 ranks: usize,
4683 rank_at: u64,
4687 rank_ends: Vec<u64>,
4691 rank_hashes: Vec<u64>,
4692 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4693 code_bits: usize,
4696 code_ranks: OnceLock<Option<Vec<u32>>>,
4703 starts: Vec<u64>,
4710 lengths: Vec<u64>,
4711 hashes: Vec<u64>,
4712 grams: Option<NativeGrams>,
4714 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4716 char_lens: Vec<OnceLock<Box<[u32]>>>,
4725 keep_budget: usize,
4728 payload_kept: AtomicUsize,
4736 swept: Vec<AtomicBool>,
4744 visit_dropped: AtomicUsize,
4759 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4776}
4777
4778#[derive(Debug)]
4779struct NativeGrams {
4780 start: u64,
4781 length: usize,
4782 width: usize,
4784 hash: u64,
4785 verdicts: Mutex<Vec<Verdict>>,
4792}
4793
4794type Verdict = (Vec<u8>, Arc<[bool]>);
4796
4797const GRAM_VERDICTS: usize = 8;
4799
4800impl NativeGrams {
4801 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4806 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4807 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4808 return Ok(Arc::clone(verdict));
4809 }
4810 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4811 let mut verdict = Vec::with_capacity(self.length / self.width);
4812 let window = GRAM_WINDOW / self.width * self.width;
4813 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4814 verdict.extend(bytes.chunks(self.width).map(|bits| {
4815 wanted
4816 .iter()
4817 .flatten()
4818 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4819 }));
4820 Ok(())
4821 })?;
4822 if hash != self.hash {
4823 return Err(invalid("global dictionary substring signatures checksum differs"));
4824 }
4825 let verdict: Arc<[bool]> = verdict.into();
4826 if held.len() >= GRAM_VERDICTS {
4827 held.remove(0);
4828 }
4829 held.push((literal.to_vec(), Arc::clone(&verdict)));
4830 Ok(verdict)
4831 }
4832
4833 fn footprint(&self) -> usize {
4834 self.verdicts.lock().map_or(0, |held| {
4835 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4836 })
4837 }
4838}
4839
4840const TEXT_SEARCH_MEMO: usize = 64;
4845
4846const TEXT_PAYLOAD_VALUES: usize = 1024;
4862
4863const TEXT_GRAM_BYTES: usize = 8192;
4874
4875const NARROW_GRAM_BYTES: usize = 2048;
4877
4878const GRAM_WINDOW: usize = 256 << 10;
4880
4881fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4884 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4885 let mut first = original ^ (original >> 16);
4886 first = first.wrapping_mul(0x7feb_352d);
4887 first ^= first >> 15;
4888 let mut second = original ^ (original >> 17);
4889 second = second.wrapping_mul(0x846c_a68b);
4890 second ^= second >> 16;
4891 let mask = width * 8 - 1;
4892 [(first as usize) & mask, (second as usize) & mask]
4893}
4894
4895const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4916
4917#[derive(Debug)]
4924enum Lengths {
4925 Narrow(Vec<u16>),
4927 Wide(Vec<u32>),
4929}
4930
4931impl Lengths {
4932 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4935 match self {
4936 Lengths::Narrow(lens) => into.extend(
4937 indices
4938 .iter()
4939 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4940 ),
4941 Lengths::Wide(lens) => into.extend(
4942 indices
4943 .iter()
4944 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4945 ),
4946 }
4947 }
4948
4949 fn footprint(&self) -> usize {
4951 match self {
4952 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4953 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4954 }
4955 }
4956}
4957
4958fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4967 match lengths_as::<u16>(ends)? {
4968 Some(narrow) => Some(Lengths::Narrow(narrow)),
4969 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4970 }
4971}
4972
4973fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4976 let mut lens = Vec::with_capacity(ends.len());
4977 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4978 let mut start = 0;
4979 for &end in block {
4980 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4981 return Some(None);
4982 };
4983 lens.push(len);
4984 start = end;
4985 }
4986 }
4987 Some(Some(lens))
4988}
4989
4990const TEXT_OFFSET_RUN: usize = 512;
4997
4998const DICTIONARY_HEADER: usize = 16;
5001
5002const DICTIONARY_SCATTERED: u32 = 1 << 31;
5016const DICTIONARY_GRAMS: u32 = 1 << 30;
5018const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
5021const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
5023
5024const TEXT_RANK_BLOCK: usize = 512;
5035
5036const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
5050
5051impl NativeText {
5052 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
5059 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
5060 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
5061 Ok(Some(bytes.as_slice()))
5062 }
5063
5064 fn block_chars(&self, block: usize) -> Result<&[u32]> {
5071 let slot = self
5072 .char_lens
5073 .get(block)
5074 .ok_or_else(|| invalid("a block past the global dictionary"))?;
5075 if let Some(lens) = slot.get() {
5076 return Ok(lens);
5077 }
5078 let decoded;
5079 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5080 Some(Ok(kept)) => kept,
5081 _ => {
5082 decoded = self.decode_block(block)?;
5083 &decoded
5084 }
5085 };
5086 let first = block * TEXT_PAYLOAD_VALUES;
5087 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5088 let ends = self.ends_within(first, last)?;
5089 if ends.len() != last - first {
5090 return Err(invalid("global dictionary offsets are short"));
5091 }
5092 let mut lens = Vec::with_capacity(ends.len());
5093 let mut start = u64::from(self.start_within(first)?);
5094 for &end in &ends {
5095 let value = usize::try_from(start)
5096 .ok()
5097 .zip(usize::try_from(end).ok())
5098 .and_then(|(from, to)| bytes.get(from..to))
5099 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5100 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5103 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5104 start = end;
5105 }
5106 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5107 }
5108
5109 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5114 let len = self.lengths[block];
5115 let mut stored = vec![
5116 0;
5117 usize::try_from(len).map_err(|_| invalid(
5118 "global dictionary block does not fit in memory"
5119 ))?
5120 ];
5121 read_at(&self.file, self.starts[block], &mut stored)?;
5122 if checksum(&stored) != self.hashes[block] {
5123 return Err(invalid("global dictionary payload checksum differs"));
5124 }
5125 let first = block * TEXT_PAYLOAD_VALUES;
5126 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5127 let want = self.end_within(last - 1)? as usize;
5128 let values = string::decode_flat(&stored)?;
5129 if values.len() != last - first {
5130 return Err(invalid("global dictionary block holds the wrong value count"));
5131 }
5132 let bytes = values.into_bytes();
5133 if bytes.len() != want {
5134 return Err(invalid("global dictionary block decodes to the wrong length"));
5135 }
5136 Ok(bytes)
5137 }
5138
5139 fn loaned_block<'a>(
5148 &'a self,
5149 block: usize,
5150 decoded: &'a mut Vec<u8>,
5151 scattered: bool,
5152 ) -> Result<&'a [u8]> {
5153 let kept = self.blocks.get(block).and_then(OnceLock::get);
5154 if let Some(Ok(kept)) = kept {
5155 return Ok(kept);
5156 }
5157 let again = kept.is_none()
5158 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5159 let keep = again
5160 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5161 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5162 if keep {
5163 let kept = self
5164 .payload_block(block)?
5165 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5166 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5167 return Ok(kept);
5168 }
5169 *decoded = self.decode_block(block)?;
5170 if scattered && again {
5171 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5172 }
5173 Ok(decoded)
5174 }
5175
5176 fn ends_worth_unpacking(&self) -> usize {
5193 self.values.max(TEXT_PAYLOAD_VALUES)
5194 }
5195
5196 fn value_ends(&self) -> Option<&[u32]> {
5198 if let Some(built) = self.value_ends.get() {
5199 return built.as_deref();
5200 }
5201 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5202 return None;
5203 }
5204 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5205 }
5206
5207 fn unpack_ends(&self) -> Option<Vec<u32>> {
5213 let mut ends = vec![0u32; self.values];
5214 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5215 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5216 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5217 u32::try_from(bits).unwrap_or(u32::MAX)
5218 })
5219 .ok()?;
5220 }
5221 if ends.contains(&u32::MAX) { None } else { Some(ends) }
5224 }
5225
5226 fn packed(&self) -> &[u8] {
5228 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5229 }
5230
5231 fn end_within(&self, index: usize) -> Result<u32> {
5233 if let Some(ends) = self.value_ends() {
5234 return ends
5235 .get(index)
5236 .copied()
5237 .ok_or_else(|| invalid("global dictionary offsets are short"));
5238 }
5239 let run = index / TEXT_OFFSET_RUN;
5240 let bytes = self
5241 .packed()
5242 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5243 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5244 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5245 .map_err(|_| invalid("global dictionary offsets are short"))?;
5246 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5247 }
5248
5249 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5267 let mut ends = vec![0u64; last.saturating_sub(first)];
5268 let mut scratch = Vec::new();
5269 let mut at = first;
5270 while at < last {
5271 let run = at / TEXT_OFFSET_RUN;
5272 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5273 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5274 let bytes = self
5275 .packed()
5276 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5277 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5278 let from = at % TEXT_OFFSET_RUN;
5279 let upto = stop - run * TEXT_OFFSET_RUN;
5280 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5281 return Err(invalid("global dictionary offsets are short"));
5282 }
5283 let into = &mut ends[at - first..stop - first];
5284 if from == 0 {
5285 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5286 .map_err(|_| invalid("global dictionary offsets are short"))?;
5287 } else {
5288 scratch.resize(held, 0);
5289 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5290 .map_err(|_| invalid("global dictionary offsets are short"))?;
5291 into.copy_from_slice(&scratch[from..upto]);
5292 }
5293 at = stop;
5294 }
5295 Ok(ends)
5296 }
5297
5298 fn start_within(&self, index: usize) -> Result<u32> {
5301 if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5302 }
5303
5304 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5312 if let Some(ends) = self.value_ends() {
5313 let end =
5314 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5315 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5318 if start > end {
5319 return Err(invalid("global dictionary value ends before it starts"));
5320 }
5321 return Ok((start, end));
5322 }
5323 let within = index % TEXT_OFFSET_RUN;
5324 let (start, end) = if within == 0 {
5325 (self.start_within(index)?, self.end_within(index)?)
5326 } else {
5327 let run = index / TEXT_OFFSET_RUN;
5328 let bytes = self
5329 .packed()
5330 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5331 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5332 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5333 .map_err(|_| invalid("global dictionary offsets are short"))?;
5334 let ends = u32::try_from(end)
5335 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5336 let starts = u32::try_from(start)
5337 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5338 (starts, ends)
5339 };
5340 if start > end {
5341 return Err(invalid("global dictionary value ends before it starts"));
5342 }
5343 Ok((start, end))
5344 }
5345
5346 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5353 let slot = self
5354 .rank_blocks
5355 .get(rank / TEXT_RANK_BLOCK)
5356 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5357 let block = slot
5358 .get_or_init(|| {
5359 let mut bytes = Vec::new();
5360 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5361 Ok(bytes)
5362 })
5363 .as_ref()
5364 .map_err(Clone::clone)?;
5365 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5366 }
5367
5368 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5371 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5372 let end = self.rank_ends[which];
5373 bytes.clear();
5374 bytes.resize((end - start) as usize, 0);
5375 read_at(&self.file, self.rank_at + start, bytes)?;
5376 let expected = self
5377 .rank_hashes
5378 .get(which)
5379 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5380 if checksum(bytes) != *expected {
5381 return Err(invalid("global dictionary rank checksum differs"));
5382 }
5383 Ok(())
5384 }
5385
5386 fn head_at(&self, rank: usize) -> Result<u64> {
5388 let (block, within) = self.rank_parts(rank)?;
5389 let (base, width, packed) = rank_heads(block)?;
5390 let above = bitpack::tail_at(packed, width, within)
5391 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5392 Ok(base.wrapping_add(above))
5393 }
5394
5395 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5397 let (_, width, packed) = rank_heads(block)?;
5398 packed
5399 .get(bitpack::tail_len(count, width)..)
5400 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5401 }
5402
5403 fn rank_block_len(&self, rank: usize) -> usize {
5405 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5406 TEXT_RANK_BLOCK.min(self.ranks - first)
5407 }
5408}
5409
5410fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5412 let header = block
5413 .get(..RANK_BLOCK_HEADER)
5414 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5415 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5416 let width = header[8] as usize;
5417 if width > 64 {
5418 return Err(invalid("global dictionary rank block packs heads past a word"));
5419 }
5420 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5421}
5422
5423fn offset_width(ends: &[u32]) -> usize {
5430 let span = ends.iter().copied().max().unwrap_or(0);
5434 (u32::BITS - span.leading_zeros()) as usize
5435}
5436
5437fn offset_bytes(values: usize, bits: usize) -> usize {
5440 let full = values / TEXT_OFFSET_RUN;
5441 let rest = values % TEXT_OFFSET_RUN;
5442 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5443}
5444
5445fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5449 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5450 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5451 run.clear();
5452 run.extend(chunk.iter().map(|&end| u64::from(end)));
5453 bitpack::pack_tail(&run, bits, out)
5454 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5455 }
5456 Ok(())
5457}
5458
5459fn code_width(values: usize) -> usize {
5461 match u64::try_from(values).unwrap_or(u64::MAX) {
5462 0 | 1 => 0,
5463 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5464 }
5465}
5466
5467impl TextSource for NativeText {
5468 fn len(&self) -> usize {
5469 self.values
5470 }
5471
5472 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5473 let Some(grams) = &self.grams else { return Ok(true) };
5474 if literal.len() < 4 || first >= self.values {
5475 return Ok(true);
5476 }
5477 let verdict = grams.verdicts(&self.file, literal)?;
5478 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5479 }
5480
5481 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5482 if index >= self.values {
5483 return Ok(None);
5484 }
5485 let (start, end) = self.span_within(index)?;
5486 if start == end {
5487 return Ok(Some(&[]));
5488 }
5489 let block = index / TEXT_PAYLOAD_VALUES;
5492 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5493 Ok(bytes.get(start as usize..end as usize))
5494 }
5495
5496 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5497 if index >= self.values {
5498 return Ok(None);
5499 }
5500 let (start, end) = self.span_within(index)?;
5501 Ok(Some((end - start) as usize))
5502 }
5503
5504 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5511 into.reserve(indices.len());
5512 if let Some(Some(lens)) = self.value_lens.get() {
5515 lens.extend_at(indices, into);
5516 return Ok(());
5517 }
5518 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5519 let Some(ends) = self.value_ends() else {
5520 for &index in indices {
5521 into.push(
5522 self.bytes_len_at(index as usize)?
5523 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5524 );
5525 }
5526 return Ok(());
5527 };
5528 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5529 lens.extend_at(indices, into);
5530 return Ok(());
5531 }
5532 for &index in indices {
5533 let index = index as usize;
5534 let Some(&end) = ends.get(index) else {
5536 into.push(0);
5537 continue;
5538 };
5539 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5540 if start > end {
5541 return Err(invalid("global dictionary value ends before it starts"));
5542 }
5543 into.push(i64::from(end - start));
5544 }
5545 Ok(())
5546 }
5547
5548 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5551 into.reserve(indices.len());
5552 for &index in indices {
5553 let index = index as usize;
5554 if index >= self.values {
5556 into.push(0);
5557 continue;
5558 }
5559 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5560 let len = lens
5561 .get(index % TEXT_PAYLOAD_VALUES)
5562 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5563 into.push(i64::from(*len));
5564 }
5565 Ok(())
5566 }
5567
5568 fn sweep(
5581 &self,
5582 first: usize,
5583 limit: usize,
5584 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5585 ) -> Result<usize> {
5586 let limit = limit.min(self.values);
5587 if first >= limit {
5588 return Ok(first);
5589 }
5590 let block = first / TEXT_PAYLOAD_VALUES;
5591 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5592 let mut decoded = Vec::new();
5593 let bytes = self.loaned_block(block, &mut decoded, false)?;
5594 let ends = self.ends_within(first, last)?;
5595 if ends.len() != last - first {
5596 return Err(invalid("global dictionary offsets are short"));
5597 }
5598 let mut start = u64::from(self.start_within(first)?);
5599 for (index, &end) in (first..last).zip(&ends) {
5602 let value = usize::try_from(start)
5603 .ok()
5604 .zip(usize::try_from(end).ok())
5605 .and_then(|(from, to)| bytes.get(from..to))
5606 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5607 body(index, value)?;
5608 start = end;
5609 }
5610 Ok(last)
5611 }
5612
5613 fn visit_at(
5622 &self,
5623 indices: &[u32],
5624 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5625 ) -> Result<()> {
5626 let mut order = (0..indices.len()).collect::<Vec<_>>();
5627 order.sort_unstable_by_key(|&at| indices[at]);
5628 let block_of = |at: usize| {
5629 let index = indices[at] as usize;
5630 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5631 };
5632 let mut decoded = Vec::new();
5633 let mut run = 0;
5634 while run < order.len() {
5635 let Some(block) = block_of(order[run]) else {
5636 for &at in &order[run..] {
5638 body(at, &[])?;
5639 }
5640 break;
5641 };
5642 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5643 let bytes = self.loaned_block(block, &mut decoded, true)?;
5644 for &at in &order[run..upto] {
5645 let (start, end) = self.span_within(indices[at] as usize)?;
5646 let value = bytes
5647 .get(start as usize..end as usize)
5648 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5649 body(at, value)?;
5650 }
5651 run = upto;
5652 }
5653 Ok(())
5654 }
5655
5656 fn visit(
5662 &self,
5663 indices: &[usize],
5664 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5665 ) -> Result<()> {
5666 let mut at = 0;
5667 while at < indices.len() {
5668 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5669 let upto =
5670 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5671 let wanted = &indices[at..upto];
5672 if wanted.iter().any(|&index| index >= self.values) {
5673 return Err(invalid("a visited value is past the global dictionary"));
5674 }
5675 let decoded;
5676 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5677 Some(Ok(kept)) => kept,
5678 _ => {
5679 decoded = self.decode_block(block)?;
5680 &decoded
5681 }
5682 };
5683 for (offset, &index) in wanted.iter().enumerate() {
5684 let (start, end) = self.span_within(index)?;
5685 let value = bytes
5686 .get(start as usize..end as usize)
5687 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5688 body(at + offset, value)?;
5689 }
5690 at = upto;
5691 }
5692 Ok(())
5693 }
5694
5695 fn ranks(&self) -> Option<usize> {
5696 (self.ranks > 0).then_some(self.ranks)
5697 }
5698
5699 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5707 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5708 if let Some(&answer) = memo.get(wanted) {
5709 return Ok(answer);
5710 }
5711 let answer = search_below(self, ranks, wanted)?;
5712 if memo.len() >= TEXT_SEARCH_MEMO {
5713 memo.clear();
5714 }
5715 memo.insert(wanted.to_vec(), answer);
5716 Ok(answer)
5717 }
5718
5719 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5720 let settled = self.head_at(rank)?.cmp(&head(wanted));
5724 if settled != Ordering::Equal {
5725 return Ok(settled);
5726 }
5727 let code = self.code_at_rank(rank)?;
5728 let bytes = self
5729 .bytes_at(code as usize)?
5730 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5731 Ok(bytes.cmp(wanted))
5732 }
5733
5734 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5735 let (block, within) = self.rank_parts(rank)?;
5736 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5737 let code = bitpack::tail_at(codes, self.code_bits, within)
5738 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5739 let code = u32::try_from(code)
5740 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5741 if code as usize >= self.len() {
5742 return Err(invalid("global dictionary order names a code it does not have"));
5743 }
5744 Ok(code)
5745 }
5746
5747 fn code_ranks(&self) -> Option<&[u32]> {
5748 if self.ranks == 0 || self.ranks != self.len() {
5752 return None;
5753 }
5754 self.code_ranks
5755 .get_or_init(|| {
5756 let mut ranks = vec![u32::MAX; self.ranks];
5757 let mut scratch = Vec::new();
5765 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5766 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5767 let which = first / TEXT_RANK_BLOCK;
5768 let block = match self.rank_blocks.get(which)?.get() {
5769 Some(kept) => kept.as_ref().ok()?.as_slice(),
5770 None => {
5771 self.read_rank_block(which, &mut scratch).ok()?;
5772 scratch.as_slice()
5773 }
5774 };
5775 let count = self.rank_block_len(first);
5776 let packed = self.rank_codes(block, count).ok()?;
5777 let codes = codes.get_mut(..count)?;
5778 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5779 for (within, &code) in codes.iter().enumerate() {
5780 let code = usize::try_from(code).ok()?;
5781 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5782 }
5783 }
5784 if ranks.contains(&u32::MAX) {
5785 return None;
5786 }
5787 Some(ranks)
5788 })
5789 .as_deref()
5790 }
5791
5792 fn footprint(&self) -> usize {
5793 self.offsets.capacity()
5794 + self
5795 .value_ends
5796 .get()
5797 .and_then(Option::as_ref)
5798 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5799 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5800 + self
5801 .code_ranks
5802 .get()
5803 .and_then(Option::as_ref)
5804 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5805 + self.rank_hashes.capacity() * size_of::<u64>()
5806 + self.rank_ends.capacity() * size_of::<u64>()
5807 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5808 + self
5809 .rank_blocks
5810 .iter()
5811 .filter_map(OnceLock::get)
5812 .filter_map(|result| result.as_ref().ok())
5813 .map(Vec::capacity)
5814 .sum::<usize>()
5815 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5816 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5817 + self
5818 .char_lens
5819 .iter()
5820 .filter_map(OnceLock::get)
5821 .map(|lens| lens.len() * size_of::<u32>())
5822 .sum::<usize>()
5823 + self.hashes.capacity() * size_of::<u64>()
5824 + self.starts.capacity() * size_of::<u64>()
5825 + self.lengths.capacity() * size_of::<u64>()
5826 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5827 + self
5828 .blocks
5829 .iter()
5830 .filter_map(OnceLock::get)
5831 .filter_map(|result| result.as_ref().ok())
5832 .map(Vec::capacity)
5833 .sum::<usize>()
5834 }
5835}
5836
5837fn places(table: &Table) -> Result<Vec<Place>> {
5839 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5840 for (at, stripe) in table.stripes.iter().enumerate() {
5841 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5842 for (part, &rows) in stripe.parts.iter().enumerate() {
5843 places.push(Place {
5844 stripe: index,
5845 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5846 rows,
5847 });
5848 }
5849 }
5850 Ok(places)
5851}
5852
5853fn read_index<F: Positional + ?Sized>(
5858 file: &F,
5859 stripe: &Stripe,
5860 column: usize,
5861) -> Result<Vec<PartSpan>> {
5862 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5863 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5864}
5865
5866fn read_index_span<F: Positional + ?Sized>(
5867 file: &F,
5868 index: Span,
5869 page: Span,
5870 parts: usize,
5871 column: usize,
5872) -> Result<Vec<PartSpan>> {
5873 let section = index_section(parts)?;
5874 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5875 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5876 if end > index.length as usize {
5877 return Err(invalid("index page is shorter than its columns"));
5878 }
5879 let mut bytes = vec![0; section];
5880 let offset =
5881 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5882 read_at(file, offset, &mut bytes)?;
5883 let entries = section - size_of::<u64>();
5884 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5885 if checksum(&bytes[..entries]) != stored {
5886 return Err(invalid(&format!(
5889 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5890 wanted {stored:016x} and got {:016x}",
5891 checksum(&bytes[..entries]),
5892 )));
5893 }
5894 let mut spans = Vec::with_capacity(parts);
5895 let mut start = 0_usize;
5896 for part in 0..parts {
5897 let at = part * INDEX_ENTRY;
5898 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5899 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5900 spans.push(PartSpan { start, length, hash });
5901 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5902 }
5903 if start != page.length as usize {
5904 return Err(invalid("column page length differs from its index"));
5905 }
5906 Ok(spans)
5907}
5908
5909fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5911 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5912 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5913}
5914
5915fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5918 if bits.is_empty() {
5919 bits.resize(parts.div_ceil(64).max(1), 0);
5920 }
5921 let (word, bit) = (part / 64, 1_u64 << (part % 64));
5922 let Some(held) = bits.get_mut(word) else { return (false, false) };
5923 let again = *held & bit != 0;
5924 *held |= bit;
5925 let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5926 (again, again && through)
5927}
5928
5929fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5935 if let Some(slot) = cached.index.get_mut(held.stripe)
5936 && slot.is_none()
5937 {
5938 *slot = Some(Arc::clone(&held.index));
5939 }
5940 let page = held.page.clone()?;
5941 let slot = cached.pages.get_mut(held.stripe)?;
5942 if slot.is_some() {
5943 return None;
5944 }
5945 let bytes = page.bytes.len();
5946 let used = Arc::new(AtomicBool::new(true));
5949 *slot = Some(Resident { page, used: Arc::clone(&used) });
5950 Some((bytes, used))
5951}
5952
5953#[derive(Debug, Clone)]
5962pub struct Catalog {
5963 file: Arc<File>,
5964 size: u64,
5965 entries: Arc<Vec<Entry>>,
5966 views: Arc<Vec<ViewEntry>>,
5968 anchor: Option<Arc<LogAnchor>>,
5970 opening: Opening,
5971 pool: PagePool,
5973}
5974
5975#[derive(Debug, Clone, PartialEq, Eq)]
5977pub struct CertifiedSums {
5978 pub columns: Vec<(i128, u64)>,
5979 pub rows: u64,
5980}
5981
5982#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5984pub enum IntegerExtremes {
5985 Null,
5986 Values { low: i128, high: i128 },
5987}
5988
5989pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5991
5992impl Catalog {
5993 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6002 Self::open_in(path, &PagePool::default())
6003 }
6004
6005 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
6011 let path = path.as_ref();
6012 let (file, size, _, bytes, opening) = slot_bytes(path)?;
6013 let (entries, views, card, anchor) = decode_catalog(&bytes, size)?;
6014 remember_card(path, card.as_ref());
6015 Ok(Self {
6016 anchor: anchor.map(Arc::new),
6017 file: Arc::new(file),
6018 size,
6019 entries: Arc::new(entries),
6020 views: Arc::new(views),
6021 opening,
6022 pool: pool.clone(),
6023 })
6024 }
6025
6026 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
6028 self.entries.iter().map(|entry| entry.name.as_str())
6029 }
6030
6031 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
6038 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
6039 }
6040
6041 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
6047 self.views.iter()
6048 }
6049
6050 #[must_use]
6052 pub fn log_anchor(&self) -> Option<&LogAnchor> {
6053 self.anchor.as_deref()
6054 }
6055
6056 #[must_use]
6058 pub fn len(&self) -> usize {
6059 self.entries.len()
6060 }
6061
6062 #[must_use]
6065 pub fn is_empty(&self) -> bool {
6066 self.entries.is_empty()
6067 }
6068
6069 pub fn table(&self, name: &str) -> Result<Reader> {
6075 let entry = self
6076 .entries
6077 .iter()
6078 .find(|entry| entry.name == name)
6079 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6080 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6084 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6085 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6086 }
6087 let mut opening = self.opening;
6088 opening.reads += 1;
6089 opening.bytes += u64::from(entry.directory.length);
6090 Reader::build(
6091 Arc::clone(&self.file),
6092 self.size,
6093 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6094 u64::from(entry.directory.length),
6095 opening,
6096 self.pool.clone(),
6097 )
6098 }
6099
6100 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6108 let mut counts = BTreeMap::<i64, u64>::new();
6109 let Some(()) = self.integer_fold(name, column, |value, count| {
6110 let held = counts.entry(value).or_default();
6111 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6112 Ok(())
6113 })?
6114 else {
6115 return Ok(None);
6116 };
6117 Ok(Some(counts.into_iter().collect()))
6118 }
6119
6120 pub fn integer_fold(
6127 &self,
6128 name: &str,
6129 column: usize,
6130 mut emit: impl FnMut(i64, u64) -> Result<()>,
6131 ) -> Result<Option<()>> {
6132 let entry = self
6133 .entries
6134 .iter()
6135 .find(|entry| entry.name == name)
6136 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6137 let field =
6138 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6139 if !signed_integer(&field.ty) {
6140 return Ok(None);
6141 }
6142 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6143 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6144 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6145 }
6146 quick_integer_fold(
6147 &self.file,
6148 Cursor::over(&self.file, offset, length),
6149 entry,
6150 self.size,
6151 column,
6152 &mut emit,
6153 )?;
6154 Ok(Some(()))
6155 }
6156
6157 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6162 let entry = self
6163 .entries
6164 .iter()
6165 .find(|entry| entry.name == name)
6166 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6167 let Some(field) = entry.fields.get(column) else {
6168 return Err(invalid("frequency column index out of range"));
6169 };
6170 if !matches!(
6171 field.ty,
6172 LogicalType::TinyInt
6173 | LogicalType::SmallInt
6174 | LogicalType::Integer
6175 | LogicalType::BigInt
6176 | LogicalType::UTinyInt
6177 | LogicalType::USmallInt
6178 | LogicalType::UInteger
6179 | LogicalType::UBigInt
6180 ) {
6181 return Ok(None);
6182 }
6183 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6184 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6185 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6186 }
6187 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6188 return frequencies
6189 .iter()
6190 .filter(|(value, _)| value.is_some_and(|value| value != 0))
6191 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6192 .map(Some)
6193 .ok_or_else(|| invalid("numeric frequency count overflow"));
6194 }
6195 quick_nonzero(
6196 Cursor::over(&self.file, offset, length),
6197 &entry.name,
6198 &entry.fields,
6199 entry.rows,
6200 column,
6201 )
6202 }
6203
6204 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6207 let entry = self
6208 .entries
6209 .iter()
6210 .find(|entry| entry.name == name)
6211 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6212 let mut sums = Vec::with_capacity(columns.len());
6213 for &column in columns {
6214 let Some(field) = entry.fields.get(column) else {
6215 return Err(invalid("aggregate column index out of range"));
6216 };
6217 if !signed_integer(&field.ty) {
6218 return Ok(None);
6219 }
6220 let Some(sum) = entry.aggregates[column] else {
6221 return Ok(None);
6222 };
6223 sums.push(sum);
6224 }
6225 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6226 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6227 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6228 }
6229 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6230 }
6231
6232 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6234 let entry = self
6235 .entries
6236 .iter()
6237 .find(|entry| entry.name == name)
6238 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6239 let Some(count) = entry.distincts.get(column).copied() else {
6240 return Err(invalid("distinct column index out of range"));
6241 };
6242 let Some(count) = count else { return Ok(None) };
6243 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6244 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6245 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6246 }
6247 Ok(Some(count))
6248 }
6249
6250 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6252 let entry = self
6253 .entries
6254 .iter()
6255 .find(|entry| entry.name == name)
6256 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6257 let Some(extremes) = entry.extremes.get(column).copied() else {
6258 return Err(invalid("extremes column index out of range"));
6259 };
6260 let Some(extremes) = extremes else { return Ok(None) };
6261 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6262 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6263 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6264 }
6265 Ok(Some(match extremes {
6266 None => IntegerExtremes::Null,
6267 Some((low, high)) => IntegerExtremes::Values { low, high },
6268 }))
6269 }
6270
6271 pub fn exact_numeric_frequencies(
6273 &self,
6274 name: &str,
6275 column: usize,
6276 ) -> Result<Option<NumericFrequencies>> {
6277 let entry = self
6278 .entries
6279 .iter()
6280 .find(|entry| entry.name == name)
6281 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6282 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6283 return Err(invalid("numeric frequency column index out of range"));
6284 };
6285 let Some(frequencies) = frequencies else { return Ok(None) };
6286 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6287 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6288 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6289 }
6290 Ok(Some(frequencies))
6291 }
6292
6293 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6295 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6296 }
6297}
6298
6299fn slot_offset(generation: u64) -> u64 {
6304 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6305}
6306
6307fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6312 let file = File::open(path).map_err(io)?;
6313 let size = file.metadata().map_err(io)?.len();
6314 let (slot, bytes, opening) = committed_slot(&file, size)?;
6315 Ok((file, size, slot, bytes, opening))
6316}
6317
6318fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6324 if size < HEADER {
6325 return Err(invalid("file is shorter than its header"));
6326 }
6327 let mut header = [0; HEADER as usize];
6328 read_at(file, 0, &mut header)?;
6329 let mut opening = Opening { reads: 1, bytes: HEADER };
6330 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6331 if &header[..8] != MAGIC {
6336 return Err(invalid("the header does not begin with a rudb native magic"));
6337 }
6338 if !READABLE.contains(&version) {
6339 return Err(invalid(&format!(
6340 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6341 be written again"
6342 )));
6343 }
6344 let mut selected = None;
6345 for start in [16, 16 + SLOT_BYTES] {
6346 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6347 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6348 continue;
6349 }
6350 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6351 if slot.offset < HEADER || end > size {
6352 continue;
6353 }
6354 let mut bytes = vec![0; slot.length as usize];
6355 read_at(file, slot.offset, &mut bytes)?;
6356 opening.reads += 1;
6357 opening.bytes += u64::from(slot.length);
6358 if checksum(&bytes) == slot.hash
6359 && selected
6360 .as_ref()
6361 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6362 {
6363 selected = Some((slot, bytes));
6364 }
6365 }
6366 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6367 Ok((slot, bytes, opening))
6368}
6369
6370impl Reader {
6371 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6378 let catalog = Catalog::open(path)?;
6379 let mut names = catalog.names();
6380 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6381 if names.next().is_some() {
6382 return Err(invalid(
6383 "the file holds more than one table, so it has to be opened by name",
6384 ));
6385 }
6386 catalog.table(&name)
6387 }
6388
6389 fn build(
6391 file: Arc<File>,
6392 size: u64,
6393 table: Table,
6394 directory: u64,
6395 opening: Opening,
6396 pool: PagePool,
6397 ) -> Result<Self> {
6398 let places = places(&table)?;
6399 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6400 let table_fields = table.fields.len();
6401 let stripes = table.stripes.len();
6402 let columns = (0..table.fields.len())
6403 .map(|_| {
6404 Mutex::new(Cached {
6405 pages: (0..stripes).map(|_| None).collect(),
6406 index: (0..stripes).map(|_| None).collect(),
6407 touched: vec![Vec::new(); stripes],
6408 ..Cached::default()
6409 })
6410 })
6411 .collect::<Vec<_>>();
6412 let cache = Shelf {
6413 columns,
6414 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6415 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6416 };
6417 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6418 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6419 .collect();
6420 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6421 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6422 .collect();
6423 let verified = (places.len() * table_fields).div_ceil(64);
6424 let firsts = places
6425 .iter()
6426 .scan(0, |first, place| {
6427 let at = *first;
6428 *first += place.rows as usize;
6429 Some(at)
6430 })
6431 .collect();
6432 Ok(Self {
6433 file,
6434 table: Arc::new(table),
6435 dictionaries: Arc::new(dictionaries),
6436 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6437 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6438 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6439 frequency_heads: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6440 summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6441 opened: Arc::new(AtomicUsize::new(0)),
6442 sieves: Arc::new(sieves),
6443 part_ranges: Arc::new(part_ranges),
6444 places: Arc::new(places),
6445 cache: Arc::new(cache),
6446 pool,
6447 pages: Arc::new(AtomicUsize::new(0)),
6448 indexes: Arc::new(AtomicUsize::new(0)),
6449 verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6450 text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6451 firsts: Arc::new(firsts),
6452 size,
6453 directory,
6454 opening,
6455 })
6456 }
6457
6458 #[must_use]
6465 pub fn reads(&self) -> Reads {
6466 Reads {
6467 opening: self.opening,
6468 pages: self.pages.load(Atomic::Relaxed),
6469 indexes: self.indexes.load(Atomic::Relaxed),
6470 dictionaries: self.opened.load(Atomic::Relaxed),
6471 }
6472 }
6473
6474 #[must_use]
6479 pub fn layout(&self) -> Layout {
6480 let table = &self.table;
6481 let stripes = table.stripes.as_slice();
6482 let columns = table
6483 .fields
6484 .iter()
6485 .enumerate()
6486 .map(|(at, field)| ColumnLayout {
6487 name: field.name.clone(),
6488 kind: field.ty.to_string(),
6489 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6490 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6491 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6492 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6493 dictionary: dictionary_bytes(table, at),
6494 })
6495 .collect();
6496 Layout {
6497 file: self.size,
6498 rows: table.rows,
6499 stripes: stripes.len(),
6500 parts: self.places.len(),
6501 columns,
6502 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6503 directory: self.directory,
6504 header: HEADER,
6505 }
6506 }
6507
6508 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6525 let field = self
6526 .table
6527 .fields
6528 .get(column)
6529 .ok_or_else(|| invalid("stored column index out of range"))?;
6530 let mut stored = Vec::with_capacity(self.places.len());
6531 let mut row = 0;
6532 for (at, stripe) in self.table.stripes.iter().enumerate() {
6533 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6534 let index = read_index(&self.file, stripe, column)?;
6535 let mut bytes = vec![0; page.length as usize];
6536 read_at(&self.file, page.offset, &mut bytes)?;
6537 let ranges = self.stripe_part_ranges(at, column);
6538 for (part, &rows) in stripe.parts.iter().enumerate() {
6539 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6540 let held = part_bytes(&bytes, span)?;
6541 let range = ranges.and_then(|held| held.get(part));
6542 stored.push(StoredPart {
6543 stripe: at,
6544 part,
6545 row,
6546 rows: rows as usize,
6547 encoding: page_encoding(&field.ty, rows as usize, held),
6548 bytes: span.length as u64,
6549 page: page.offset,
6550 offset: span.start as u64,
6551 low: range
6552 .and_then(|range| range.low.clone())
6553 .and_then(|bound| bound.into_value(&field.ty)),
6554 high: range
6555 .and_then(|range| range.high.clone())
6556 .and_then(|bound| bound.into_value(&field.ty)),
6557 nulls: range.map(|range| range.nulls),
6558 });
6559 row += rows as usize;
6560 }
6561 }
6562 Ok(stored)
6563 }
6564
6565 #[must_use]
6567 pub fn parts(&self) -> usize {
6568 self.places.len()
6569 }
6570
6571 #[must_use]
6578 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6579 let mut runs = Vec::with_capacity(self.table.stripes.len());
6580 let mut start = 0;
6581 for stripe in &self.table.stripes {
6582 let end = start + stripe.parts.len();
6583 runs.push(start..end);
6584 start = end;
6585 }
6586 runs
6587 }
6588
6589 #[must_use]
6594 pub fn stripe_rows(&self, stripe: usize) -> usize {
6595 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6596 }
6597
6598 pub fn keep_stripes(&self, stripes: usize) {
6605 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6606 }
6607
6608 #[must_use]
6610 pub fn part_rows(&self, at: usize) -> usize {
6611 self.places.get(at).map_or(0, |place| place.rows as usize)
6612 }
6613
6614 #[must_use]
6616 pub fn table(&self) -> &Table {
6617 &self.table
6618 }
6619
6620 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6629 let field = self
6630 .table
6631 .fields
6632 .get(column)
6633 .ok_or_else(|| invalid("frequency column index out of range"))?;
6634 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6635 return Ok(None);
6636 };
6637 if top == 0 || entries.len() < top {
6638 return Ok(None);
6639 }
6640 let boundary = entries[top - 1].count;
6641 if boundary <= omitted_max {
6642 return Ok(None);
6643 }
6644 self.decode_frequencies(column, &field.ty, &entries).map(|values| Some(Vec::clone(&values)))
6645 }
6646
6647 pub fn top_pair_frequencies(
6655 &self,
6656 first: usize,
6657 second: usize,
6658 _top: usize,
6659 ) -> Result<Option<PairFrequencyCounts>> {
6660 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6661 return Err(invalid("pair frequency column index out of range"));
6662 }
6663 Ok(None)
6664 }
6665
6666 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6686 let Some(prefix) = self.frequency_prefix(column)? else {
6687 return Ok(None);
6688 };
6689 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6690 }
6691
6692 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6715 Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6716 entries: Vec::clone(&entries),
6717 omitted_max,
6718 }))
6719 }
6720
6721 pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6727 let field = self
6728 .table
6729 .fields
6730 .get(column)
6731 .ok_or_else(|| invalid("frequency column index out of range"))?;
6732 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6733 return Ok(None);
6734 };
6735 let entries = self.decode_frequencies(column, &field.ty, &entries)?;
6736 Ok(Some((entries, omitted_max)))
6737 }
6738
6739 fn frequency_head(&self, column: usize) -> Result<Option<(Cow<'_, [FrequencyEntry]>, u64)>> {
6742 let (span, count) = match self.table.frequencies.get(column) {
6743 None | Some(None) => return Ok(None),
6744 Some(Some(Frequencies::Held(summary))) => {
6745 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6746 }
6747 Some(Some(Frequencies::Stored { span, entries, .. })) => (span, *entries),
6748 };
6749 if let Some(summary) = self.frequency_summaries.get(column).and_then(OnceLock::get) {
6750 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6751 }
6752 let slot = self
6753 .frequency_heads
6754 .get(column)
6755 .ok_or_else(|| invalid("frequency column index out of range"))?;
6756 if slot.get().is_none() {
6757 let field = self
6758 .table
6759 .fields
6760 .get(column)
6761 .ok_or_else(|| invalid("frequency column index out of range"))?;
6762 let length = (span.length as usize).min(13 + count * 25);
6765 let mut bytes = vec![0; length];
6766 read_at(&self.file, span.offset, &mut bytes)?;
6767 let head = decode_summary_head(&mut Cursor::new(&bytes), field, self.table.rows)?
6768 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
6769 if head.0.len() != count {
6770 return Err(invalid("a stored synopsis differs from its directory span"));
6771 }
6772 let _ = slot.set(Arc::new(head));
6773 }
6774 let (entries, omitted_max) = slot.get().expect("the synopsis head was stored").as_ref();
6775 Ok(Some((Cow::Borrowed(entries), *omitted_max)))
6776 }
6777
6778 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6780 Ok(match self.table.frequencies.get(column) {
6781 None | Some(None) => None,
6782 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6783 Some(Some(Frequencies::Stored { span, values, entries })) => {
6784 let slot = self
6785 .frequency_summaries
6786 .get(column)
6787 .ok_or_else(|| invalid("frequency column index out of range"))?;
6788 if let Some(summary) = slot.get() {
6789 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6790 }
6791 let field = self
6792 .table
6793 .fields
6794 .get(column)
6795 .ok_or_else(|| invalid("frequency column index out of range"))?;
6796 let mut bytes = vec![0; span.length as usize];
6797 read_at(&self.file, span.offset, &mut bytes)?;
6798 let mut cur = Cursor::new(&bytes);
6799 let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6800 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6801 if !cur.done() || summary.entries.len() != *entries {
6802 return Err(invalid("a stored synopsis differs from its directory span"));
6803 }
6804 let _ = slot.set(Arc::new(summary));
6805 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6806 }
6807 })
6808 }
6809
6810 fn decode_frequencies(
6818 &self,
6819 column: usize,
6820 ty: &LogicalType,
6821 entries: &[FrequencyEntry],
6822 ) -> Result<Synopsis> {
6823 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6824 return Ok(Arc::clone(values));
6825 }
6826 let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6827 if let Some(slot) = self.frequency_values.get(column) {
6828 let _ = slot.set(Arc::clone(&values));
6829 }
6830 Ok(values)
6831 }
6832
6833 fn decode_frequencies_once(
6834 &self,
6835 column: usize,
6836 ty: &LogicalType,
6837 entries: &[FrequencyEntry],
6838 ) -> Result<Vec<(Value, u64)>> {
6839 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6840 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6841 return Err(invalid("frequency text count differs from its synopsis"));
6842 }
6843 let dictionary =
6844 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6845 let mut codes = entries
6846 .iter()
6847 .filter_map(|entry| match entry.value {
6848 FrequencyValue::Code(code) => Some(code as usize),
6849 _ => None,
6850 })
6851 .collect::<Vec<_>>();
6852 codes.sort_unstable();
6853 codes.dedup();
6854 let texts = match &dictionary {
6855 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6856 _ => Vec::new(),
6857 };
6858 let mut out = Vec::with_capacity(entries.len());
6859 for (entry_at, entry) in entries.iter().enumerate() {
6860 let value = match entry.value {
6861 FrequencyValue::Null => {
6862 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6863 return Err(invalid("a null frequency entry has text"));
6864 }
6865 Value::Null
6866 }
6867 FrequencyValue::Integer(value) => match *ty {
6868 LogicalType::TinyInt => Value::TinyInt(
6869 i8::try_from(value)
6870 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6871 ),
6872 LogicalType::UTinyInt => Value::UTinyInt(
6873 u8::try_from(value)
6874 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6875 ),
6876 LogicalType::USmallInt => Value::USmallInt(
6877 u16::try_from(value)
6878 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6879 ),
6880 LogicalType::UInteger => Value::UInteger(
6881 u32::try_from(value)
6882 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6883 ),
6884 LogicalType::UBigInt => Value::UBigInt(
6885 u64::try_from(value)
6886 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6887 ),
6888 LogicalType::SmallInt => Value::SmallInt(
6889 i16::try_from(value)
6890 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6891 ),
6892 LogicalType::Integer => Value::Integer(
6893 i32::try_from(value)
6894 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6895 ),
6896 LogicalType::BigInt => Value::BigInt(
6897 i64::try_from(value)
6898 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6899 ),
6900 LogicalType::Date => Value::Date(
6901 i32::try_from(value)
6902 .map_err(|_| invalid("frequency DATE is out of range"))?,
6903 ),
6904 LogicalType::Timestamp => Value::Timestamp(
6905 i64::try_from(value)
6906 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6907 ),
6908 _ => return Err(invalid("integer frequency belongs to another type")),
6909 },
6910 FrequencyValue::Code(code) => {
6911 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6912 if *ty == LogicalType::Blob {
6913 Value::Blob(text.clone())
6914 } else {
6915 Value::Varchar(
6916 String::from_utf8(text.clone())
6917 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6918 )
6919 }
6920 } else {
6921 if dictionary.is_none() {
6922 return Err(invalid("frequency code has no dictionary or stored text"));
6923 }
6924 let at = codes
6925 .binary_search(&(code as usize))
6926 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6927 texts[at].clone()
6928 }
6929 }
6930 };
6931 out.push((value, entry.count));
6932 }
6933 Ok(out)
6934 }
6935
6936 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6946 let field = self
6947 .table
6948 .fields
6949 .get(column)
6950 .ok_or_else(|| invalid("frequency column index out of range"))?;
6951 let Some(summary) = self.frequency_summary(column)? else {
6952 return Ok(None);
6953 };
6954 if summary.ordinals.is_empty() {
6955 return Ok(None);
6956 }
6957 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6958 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6959 (
6960 entries.iter().map(|(value, _)| value.clone()).collect(),
6961 summary.ordinal_entries.clone(),
6962 )
6963 } else {
6964 (Vec::new(), Vec::new())
6965 };
6966 let stored = self.table.ordinal_bounds.get(column).copied().unwrap_or(0);
6969 Ok(Some(FrequencyOccurrences {
6970 omitted_max: summary.omitted_max.max(summary.ordinal_bound).max(stored),
6971 ordinals: summary.ordinals.clone(),
6972 anchors,
6973 anchor_indices,
6974 }))
6975 }
6976
6977 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
7003 self.table
7004 .distincts
7005 .get(column)
7006 .copied()
7007 .ok_or_else(|| invalid("distinct column index out of range"))
7008 }
7009
7010 pub fn null_count(&self, column: usize) -> Result<u64> {
7021 if column >= self.table.fields.len() {
7022 return Err(invalid("null count column index out of range"));
7023 }
7024 let mut nulls = 0_u64;
7025 for stripe in &self.table.stripes {
7026 let range = stripe
7027 .zone
7028 .column(column)
7029 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7030 nulls = nulls
7031 .checked_add(range.nulls as u64)
7032 .ok_or_else(|| invalid("null count overflow"))?;
7033 }
7034 Ok(nulls)
7035 }
7036
7037 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
7052 if self.null_count(column)? > 0 || self.demoted(column) {
7053 return Ok(None);
7054 }
7055 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
7056 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
7057 if ranks == 0 {
7058 return Ok(None);
7059 }
7060 let low = text_at_rank(&dictionary, 0)?;
7061 let high = text_at_rank(&dictionary, ranks - 1)?;
7062 Ok(Some((low, high)))
7063 }
7064
7065 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
7088 if column >= self.table.fields.len() {
7089 return Err(invalid("extremes column index out of range"));
7090 }
7091 let mut low: Option<Bound> = None;
7092 let mut high: Option<Bound> = None;
7093 for stripe in &self.table.stripes {
7094 let range = stripe
7095 .zone
7096 .column(column)
7097 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7098 if !range.exact {
7099 return Ok(None);
7100 }
7101 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
7106 if stripe.rows > range.nulls {
7107 return Ok(None);
7108 }
7109 continue;
7110 };
7111 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
7112 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
7113 }
7114 Ok(low.zip(high))
7115 }
7116
7117 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
7130 if column >= self.table.fields.len() {
7131 return Err(invalid("sum column index out of range"));
7132 }
7133 let mut total = 0_i128;
7134 let mut rows = 0_u64;
7135 for stripe in &self.table.stripes {
7136 let range = stripe
7137 .zone
7138 .column(column)
7139 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7140 let Some(part) = range.sum else { return Ok(None) };
7141 let Some(sum) = total.checked_add(part) else { return Ok(None) };
7142 total = sum;
7143 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7144 }
7145 Ok(Some((total, rows)))
7146 }
7147
7148 pub fn host_groups(
7150 &self,
7151 column: usize,
7152 _minimum_count: u64,
7153 ) -> Result<Option<Vec<host::HostEntry>>> {
7154 if column >= self.table.fields.len() {
7155 return Err(invalid("host group column index out of range"));
7156 }
7157 Ok(None)
7158 }
7159
7160 #[must_use]
7164 pub fn demoted(&self, column: usize) -> bool {
7165 self.table.demoted.get(column).copied().unwrap_or(false)
7166 }
7167
7168 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7177 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7178 if let Some(dictionary) = self.dictionaries[column].get() {
7179 return Ok(Some(Arc::clone(dictionary)));
7180 }
7181 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7182 if let Some(dictionary) = self.dictionaries[column].get() {
7183 return Ok(Some(Arc::clone(dictionary)));
7184 }
7185 self.opened.fetch_add(1, Atomic::Relaxed);
7186 let dictionary = Arc::new(open_global_dictionary(
7187 Arc::clone(&self.file),
7188 page,
7189 &self.table.fields[column].ty,
7190 TEXT_KEEP_BUDGET,
7191 )?);
7192 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7193 Ok(Some(dictionary))
7194 }
7195
7196 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7203 if of.extent_bytes == 0 {
7204 return Ok(Vec::new());
7205 }
7206 let mut bytes = vec![0; of.extent_bytes as usize];
7207 read_at(&self.file, of.extent_page, &mut bytes)?;
7208 if checksum(&bytes) != of.hash {
7209 return Err(invalid("a section's extent table does not checksum"));
7210 }
7211 let extents = section::decode_extents(&bytes)?;
7212 if extents.len() != of.extents as usize {
7213 return Err(invalid("a section's extent table is not the length the entry says"));
7214 }
7215 Ok(extents)
7216 }
7217
7218 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
7228 let mut bytes = Vec::new();
7229 self.extent_into(of, &mut bytes)?;
7230 Ok(bytes)
7231 }
7232
7233 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7235 let end = of
7236 .offset
7237 .checked_add(u64::from(of.length))
7238 .ok_or_else(|| invalid("an extent overflows the file"))?;
7239 if of.offset < HEADER || end > self.size {
7240 return Err(invalid("an extent is outside the file"));
7241 }
7242 bytes.resize(of.length as usize, 0);
7243 read_at(&self.file, of.offset, bytes)?;
7244 if checksum(bytes) != of.hash {
7245 return Err(invalid("an extent does not checksum"));
7246 }
7247 Ok(())
7248 }
7249
7250 pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7263 let extents = self.extents(of)?;
7264 let Some(first) = extents.first() else { return Ok(Vec::new()) };
7265 let end = first
7266 .offset
7267 .checked_add(u64::from(first.length))
7268 .ok_or_else(|| invalid("an extent overflows the file"))?;
7269 if first.offset < HEADER || end > self.size {
7270 return Err(invalid("an extent is outside the file"));
7271 }
7272 let mut bytes = vec![0; len.min(first.length as usize)];
7273 read_at(&self.file, first.offset, &mut bytes)?;
7274 Ok(bytes)
7275 }
7276
7277 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7286 let extents = self.extents(of)?;
7287 let mut bytes =
7288 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
7289 for one in &extents {
7290 if one.first != bytes.len() as u64 {
7291 return Err(invalid("a section's extents do not join up"));
7292 }
7293 bytes.extend_from_slice(&self.extent(one)?);
7294 }
7295 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7298 return Err(invalid("a section's header is longer than its payload"));
7299 }
7300 Ok(bytes)
7301 }
7302
7303 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7314 self.read_impl(part, columns, true, None)
7315 }
7316
7317 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7327 self.read_impl(part, columns, false, None)
7328 }
7329
7330 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7338 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7339 let field =
7340 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7341 if !matches!(
7342 field.ty,
7343 LogicalType::TinyInt
7344 | LogicalType::SmallInt
7345 | LogicalType::Integer
7346 | LogicalType::BigInt
7347 ) {
7348 return Ok(None);
7349 }
7350 let (rows, counts) = match self.with_part(place, column, |bytes| {
7351 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7352 return Ok(None);
7353 }
7354 integer::tally(&bytes[2..]).map(Some)
7355 })? {
7356 Some(tallied) => tallied,
7357 None => return Ok(None),
7358 };
7359 if rows != place.rows as usize {
7360 return Err(invalid("encoded integer part holds the wrong number of rows"));
7361 }
7362 for &(value, _) in &counts {
7363 let fits = match field.ty {
7364 LogicalType::TinyInt => i8::try_from(value).is_ok(),
7365 LogicalType::SmallInt => i16::try_from(value).is_ok(),
7366 LogicalType::Integer => i32::try_from(value).is_ok(),
7367 LogicalType::BigInt => true,
7368 _ => false,
7369 };
7370 if !fits {
7371 return Err(invalid("encoded integer value is outside its column type"));
7372 }
7373 }
7374 Ok(Some(counts))
7375 }
7376
7377 pub fn rows_holding(
7389 &self,
7390 part: usize,
7391 column: usize,
7392 sequence: &Sequence,
7393 negated: bool,
7394 ) -> Result<Option<Vec<u32>>> {
7395 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7396 let field =
7397 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7398 if field.ty != LogicalType::Varchar {
7399 return Ok(None);
7400 }
7401 let rows = place.rows as usize;
7402 self.with_part(place, column, |bytes| {
7403 if bytes.first() != Some(&6) {
7404 return Ok(None);
7405 }
7406 let mut cur = Cursor::new(bytes);
7407 cur.u8()?;
7408 let mask = match cur.u8()? {
7409 0 => None,
7410 1 => return Ok(Some(Vec::new())),
7411 2 => {
7412 let from = cur.at;
7413 cur.take(rows.div_ceil(8))?;
7414 Some(&bytes[from..cur.at])
7415 }
7416 _ => return Err(invalid("page validity tag differs")),
7417 };
7418 let needs = sequence.needs();
7421 let first = self.firsts.get(part).copied().unwrap_or_default();
7422 let sketch = self
7423 .text_grams
7424 .get(column)
7425 .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7426 .and_then(|words| words.get(first..first + rows));
7427 let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7428 let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7429 return Ok(None);
7430 };
7431 if held.len() != rows {
7432 return Err(invalid("compressed text page holds the wrong number of rows"));
7433 }
7434 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7435 Ok(Some(
7436 (0..rows)
7437 .filter(|&row| held[row] != negated && valid(row))
7438 .map(|row| row as u32)
7439 .collect(),
7440 ))
7441 })
7442 }
7443
7444 fn is_verified(&self, bit: usize) -> bool {
7447 self.verified
7448 .get(bit / 64)
7449 .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7450 }
7451
7452 fn set_verified(&self, bit: usize) {
7454 if let Some(word) = self.verified.get(bit / 64) {
7455 word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7456 }
7457 }
7458
7459 fn with_part<T>(
7462 &self,
7463 place: Place,
7464 column: usize,
7465 read: impl FnOnce(&[u8]) -> Result<T>,
7466 ) -> Result<T> {
7467 let stripe_index = place.stripe as usize;
7468 let stripe = self
7469 .table
7470 .stripes
7471 .get(stripe_index)
7472 .ok_or_else(|| invalid("stripe index out of range"))?;
7473 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7474 let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7475 let span = *held
7476 .index
7477 .get(place.part as usize)
7478 .ok_or_else(|| invalid("part index out of range"))?;
7479 match &held.page {
7480 Some(page) => read(page.part(place.part as usize, span)?),
7481 None => {
7482 let offset = page
7483 .offset
7484 .checked_add(span.start as u64)
7485 .ok_or_else(|| invalid("part range overflow"))?;
7486 let mut bytes = vec![0; span.length];
7487 read_at(&self.file, offset, &mut bytes)?;
7488 verify_part(&bytes, span)?;
7489 read(&bytes)
7490 }
7491 }
7492 }
7493
7494 pub fn read_rows(
7507 &self,
7508 part: usize,
7509 columns: &[usize],
7510 positions: &[u32],
7511 whole: bool,
7512 ) -> Result<Chunk> {
7513 self.read_impl(part, columns, whole, Some(positions))
7514 }
7515
7516 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7523 if self.demoted(column) {
7526 return Ok(false);
7527 }
7528 if candidates.is_empty() {
7529 return Ok(true);
7530 }
7531 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7532 return Err(Error::internal("native code candidates are not sorted and unique"));
7533 }
7534 let stripe = self.stripe_of(part)?;
7535 let Some(page) = stripe.memberships.get(column) else {
7536 return Ok(false);
7537 };
7538 let mut bytes = vec![0; page.length as usize];
7539 read_at(&self.file, page.offset, &mut bytes)?;
7540 if checksum(&bytes) != page.hash {
7541 return Err(invalid("membership page checksum differs"));
7542 }
7543 let codes = decode_membership(&bytes)?;
7544 let mut left = 0;
7545 let mut right = 0;
7546 while left < codes.len() && right < candidates.len() {
7547 match codes[left].cmp(&candidates[right]) {
7548 Ordering::Less => left += 1,
7549 Ordering::Greater => right += 1,
7550 Ordering::Equal => return Ok(false),
7551 }
7552 }
7553 Ok(true)
7554 }
7555
7556 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7557 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7558 self.table
7559 .stripes
7560 .get(place.stripe as usize)
7561 .ok_or_else(|| invalid("stripe index out of range"))
7562 }
7563
7564 fn held(
7581 &self,
7582 at: usize,
7583 part: usize,
7584 stripe: &Stripe,
7585 column: usize,
7586 whole: bool,
7587 ) -> Result<CachedColumn> {
7588 let cache =
7589 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7590 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7591 let known = cached.index.get(at).and_then(Clone::clone);
7592 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7593 slot.used.store(true, Atomic::Relaxed);
7594 Arc::clone(&slot.page)
7595 });
7596 let (again, through) = match cached.touched.get_mut(at) {
7598 Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7599 _ => (false, false),
7600 };
7601 let whole = whole && again;
7602 if let Some(index) = known.clone()
7603 && (!whole || page.is_some())
7604 {
7605 return Ok(CachedColumn { stripe: at, index, page });
7606 }
7607 if cached.loading.contains(&at) {
7608 drop(cached);
7609 if let Some(index) = known {
7613 return Ok(CachedColumn { stripe: at, index, page: None });
7614 }
7615 let held = self.page_of(stripe, column, at, false, None)?;
7616 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7617 remember(&mut cached, &held);
7618 return Ok(held);
7619 }
7620 cached.loading.push(at);
7621 drop(cached);
7622
7623 let read = self.page_of(stripe, column, at, whole, known);
7624
7625 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7629 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7630 cached.loading.remove(position);
7631 }
7632 let held = read?;
7633 let taken = remember(&mut cached, &held);
7634 if taken.is_some() && !through {
7635 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7636 cached.passing.push_back(at);
7637 while cached.passing.len() > floor {
7638 let Some(old) = cached.passing.pop_front() else { break };
7639 if let Some(slot) = cached.pages.get_mut(old) {
7640 *slot = None;
7641 }
7642 }
7643 return Ok(held);
7644 }
7645 drop(cached);
7646 if let Some((bytes, used)) = taken {
7647 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7648 self.pool.admit(Held {
7649 shelf: Arc::downgrade(&self.cache),
7650 column,
7651 stripe: at,
7652 bytes,
7653 used,
7654 });
7655 }
7656 Ok(held)
7657 }
7658
7659 fn page_of(
7665 &self,
7666 stripe: &Stripe,
7667 column: usize,
7668 at: usize,
7669 whole: bool,
7670 known: Option<Arc<Vec<PartSpan>>>,
7671 ) -> Result<CachedColumn> {
7672 let index = match known {
7673 Some(index) => index,
7674 None => {
7675 self.indexes.fetch_add(1, Atomic::Relaxed);
7676 Arc::new(read_index(&self.file, stripe, column)?)
7677 }
7678 };
7679 let page = if whole {
7680 self.pages.fetch_add(1, Atomic::Relaxed);
7681 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7682 let mut bytes = vec![0; span.length as usize];
7683 read_at(&self.file, span.offset, &mut bytes)?;
7684 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7685 Some(Arc::new(HeldPage { bytes, checked }))
7686 } else {
7687 None
7688 };
7689 Ok(CachedColumn { stripe: at, index, page })
7690 }
7691
7692 fn read_impl(
7693 &self,
7694 at: usize,
7695 columns: &[usize],
7696 whole: bool,
7697 positions: Option<&[u32]>,
7698 ) -> Result<Chunk> {
7699 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7700 let index = place.stripe as usize;
7701 let stripe =
7702 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7703 let rows = place.rows as usize;
7704 let mut picked = Vec::with_capacity(columns.len());
7705 for &column in columns {
7706 let field = self
7707 .table
7708 .fields
7709 .get(column)
7710 .ok_or_else(|| invalid("column index out of range"))?;
7711 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7712 let held = self.held(index, place.part as usize, stripe, column, whole)?;
7713 let span = *held
7714 .index
7715 .get(place.part as usize)
7716 .ok_or_else(|| invalid("part index out of range"))?;
7717 let owned;
7718 let bit = at * self.table.fields.len() + column;
7719 let bytes = match &held.page {
7720 Some(held) if self.is_verified(bit) => part_bytes(&held.bytes, span),
7721 Some(held) => {
7722 held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7723 }
7724 None => {
7725 let offset = page
7726 .offset
7727 .checked_add(span.start as u64)
7728 .ok_or_else(|| invalid("part range overflow"))?;
7729 let mut bytes = vec![0; span.length];
7730 read_at(&self.file, offset, &mut bytes)?;
7731 owned = bytes;
7732 if self.is_verified(bit) {
7733 Ok(owned.as_slice())
7734 } else {
7735 verify_part(&owned, span).map(|()| {
7736 self.set_verified(bit);
7737 owned.as_slice()
7738 })
7739 }
7740 }
7741 }
7742 .map_err(|error| {
7743 invalid(&format!(
7744 "{}, column {column} part {} of the page at {}",
7745 error.message(),
7746 place.part,
7747 page.offset,
7748 ))
7749 })?;
7750 let dictionary = self.dictionary(column)?;
7751 let mut vector = match positions {
7757 None => decode(&field.ty, rows, bytes, dictionary)?,
7758 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7759 };
7760 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7764 vector = vector.flatten()?;
7765 }
7766 picked.push(vector.into_pages());
7767 }
7768 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7769 }
7770
7771 #[must_use]
7787 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7788 let Some(place) = self.places.get(part).copied() else { return false };
7789 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7790 if stripe.zone.skips(probes) {
7791 return true;
7792 }
7793 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7794 }
7795
7796 #[must_use]
7803 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7804 let Some(place) = self.places.get(part).copied() else { return false };
7805 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7806 if stripe.zone.column(column).is_some_and(&rule) {
7807 return true;
7808 }
7809 self.stripe_part_ranges(place.stripe as usize, column)
7810 .and_then(|ranges| ranges.get(place.part as usize))
7811 .is_some_and(rule)
7812 }
7813
7814 #[must_use]
7817 pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7818 let place = self.places.get(part).copied()?;
7819 let own = self
7820 .stripe_part_ranges(place.stripe as usize, column)
7821 .and_then(|ranges| ranges.get(place.part as usize));
7822 own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7823 }
7824
7825 #[must_use]
7827 pub fn stripe_ruled_by(
7828 &self,
7829 stripe: usize,
7830 column: usize,
7831 rule: impl Fn(&Range) -> bool,
7832 ) -> bool {
7833 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7834 }
7835
7836 fn outside(&self, place: Place, probe: &Probe) -> bool {
7842 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7843 Some(ranges) => ranges
7844 .get(place.part as usize)
7845 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7846 None => false,
7847 }
7848 }
7849
7850 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7856 let slot = self.part_ranges.get(column)?.get(stripe)?;
7857 if let Some(held) = slot.get() {
7858 return Some(held);
7859 }
7860 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7861 let mut bytes = vec![0; page.length as usize];
7862 read_at(&self.file, page.offset, &mut bytes).ok()?;
7863 if checksum(&bytes) != page.hash {
7864 return None;
7865 }
7866 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7867 let _ = slot.set(ranges);
7868 slot.get().map(|held| held.as_slice())
7869 }
7870
7871 #[must_use]
7888 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7889 let Some(place) = self.places.get(part).copied() else { return false };
7890 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7891 if stripe.zone.certain(probes) {
7892 return true;
7893 }
7894 probes
7895 .iter()
7896 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7897 }
7898
7899 fn inside(&self, place: Place, probe: &Probe) -> bool {
7905 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7906 Some(ranges) => ranges
7907 .get(place.part as usize)
7908 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7909 None => false,
7910 }
7911 }
7912
7913 #[must_use]
7924 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7925 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7926 }
7927
7928 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7934 if probe.op != Op::Equal {
7935 return false;
7936 }
7937 match self.stripe_sieves(place.stripe as usize, probe.column) {
7938 Some(sieves) => sieves
7939 .get(place.part as usize)
7940 .and_then(Option::as_ref)
7941 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7942 None => false,
7943 }
7944 }
7945
7946 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7953 let slot = self.sieves.get(column)?.get(stripe)?;
7954 if let Some(held) = slot.get() {
7955 return Some(held);
7956 }
7957 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7958 let mut bytes = vec![0; page.length as usize];
7959 read_at(&self.file, page.offset, &mut bytes).ok()?;
7960 if checksum(&bytes) != page.hash {
7961 return None;
7962 }
7963 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7964 let _ = slot.set(sieves);
7965 slot.get().map(|held| held.as_slice())
7966 }
7967}
7968
7969fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7971 let code = dictionary.code_at_rank(rank)? as usize;
7972 if dictionary.logical_type() == &LogicalType::Blob {
7973 let bytes = dictionary
7974 .try_bytes_at(code)?
7975 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7976 return Ok(Value::Blob(bytes.to_vec()));
7977 }
7978 let text = dictionary
7979 .try_text_at(code)?
7980 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7981 Ok(Value::Varchar(text.into()))
7982}
7983
7984fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7994 file.fill_at(offset, bytes)
7995}
7996
7997trait Positional {
8005 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
8010}
8011
8012impl<T: Positional + ?Sized> Positional for &T {
8013 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8014 (**self).fill_at(offset, bytes)
8015 }
8016}
8017
8018impl<T: Positional + ?Sized> Positional for Arc<T> {
8019 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8020 (**self).fill_at(offset, bytes)
8021 }
8022}
8023
8024impl<T: Positional + ?Sized> Positional for Box<T> {
8025 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8026 (**self).fill_at(offset, bytes)
8027 }
8028}
8029
8030impl Positional for dyn rudb_io::File + '_ {
8031 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8032 while !bytes.is_empty() {
8033 let read = self.read_at(offset, bytes)?;
8034 if read == 0 {
8035 return Err(invalid("column page ends before its declared length"));
8036 }
8037 offset += read as u64;
8038 bytes = &mut bytes[read..];
8039 }
8040 Ok(())
8041 }
8042}
8043
8044impl Positional for File {
8045 #[cfg(unix)]
8046 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8047 use std::os::unix::fs::FileExt;
8048 while !bytes.is_empty() {
8049 let read = self.read_at(bytes, offset).map_err(io)?;
8050 if read == 0 {
8051 return Err(invalid("column page ends before its declared length"));
8052 }
8053 offset += read as u64;
8054 bytes = &mut bytes[read..];
8055 }
8056 Ok(())
8057 }
8058
8059 #[cfg(windows)]
8065 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8066 use std::os::windows::fs::FileExt;
8067 while !bytes.is_empty() {
8068 let read = self.seek_read(bytes, offset).map_err(io)?;
8069 if read == 0 {
8070 return Err(invalid("column page ends before its declared length"));
8071 }
8072 offset += read as u64;
8073 bytes = &mut bytes[read..];
8074 }
8075 Ok(())
8076 }
8077
8078 #[cfg(not(any(unix, windows)))]
8083 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8084 use std::io::{Read, Seek, SeekFrom};
8085 let mut file = self.try_clone().map_err(io)?;
8086 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8087 file.read_exact(bytes).map_err(io)
8088 }
8089}
8090
8091#[cfg(test)]
8096fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
8097 use std::io::{Seek, SeekFrom, Write};
8098 let mut file = file;
8099 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8100 file.write_all(bytes).map_err(io)
8101}
8102
8103fn type_tag(ty: &LogicalType) -> Result<u8> {
8110 match ty {
8111 LogicalType::SmallInt => Ok(1),
8112 LogicalType::Integer => Ok(2),
8113 LogicalType::BigInt => Ok(3),
8114 LogicalType::Varchar => Ok(4),
8115 LogicalType::Date => Ok(5),
8116 LogicalType::Timestamp => Ok(6),
8117 LogicalType::Boolean => Ok(7),
8118 LogicalType::TinyInt => Ok(8),
8119 LogicalType::UTinyInt => Ok(9),
8120 LogicalType::USmallInt => Ok(10),
8121 LogicalType::UInteger => Ok(11),
8122 LogicalType::UBigInt => Ok(12),
8123 LogicalType::Decimal { .. } => Ok(13),
8124 LogicalType::Float => Ok(14),
8125 LogicalType::Double => Ok(15),
8126 LogicalType::HugeInt => Ok(16),
8127 LogicalType::UHugeInt => Ok(17),
8128 LogicalType::Time => Ok(18),
8129 LogicalType::TimeTz => Ok(19),
8130 LogicalType::TimestampTz => Ok(20),
8131 LogicalType::Interval => Ok(21),
8132 LogicalType::Uuid => Ok(22),
8133 LogicalType::Blob => Ok(23),
8134 LogicalType::Bit => Ok(24),
8135 LogicalType::TimestampS => Ok(25),
8136 LogicalType::TimestampMs => Ok(26),
8137 LogicalType::TimestampNs => Ok(27),
8138 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8139 }
8140}
8141
8142fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8148 out.push(type_tag(ty)?);
8149 if let LogicalType::Decimal { width, scale } = ty {
8150 out.push(*width);
8151 out.push(*scale);
8152 }
8153 Ok(())
8154}
8155
8156fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8158 let tag = cur.u8()?;
8159 if tag == 13 {
8160 let width = cur.u8()?;
8161 let scale = cur.u8()?;
8162 return LogicalType::decimal(width, scale)
8163 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8164 }
8165 tag_type(tag)
8166}
8167
8168fn tag_type(tag: u8) -> Result<LogicalType> {
8169 match tag {
8170 1 => Ok(LogicalType::SmallInt),
8171 2 => Ok(LogicalType::Integer),
8172 3 => Ok(LogicalType::BigInt),
8173 4 => Ok(LogicalType::Varchar),
8174 5 => Ok(LogicalType::Date),
8175 6 => Ok(LogicalType::Timestamp),
8176 7 => Ok(LogicalType::Boolean),
8177 8 => Ok(LogicalType::TinyInt),
8178 9 => Ok(LogicalType::UTinyInt),
8179 10 => Ok(LogicalType::USmallInt),
8180 11 => Ok(LogicalType::UInteger),
8181 12 => Ok(LogicalType::UBigInt),
8182 14 => Ok(LogicalType::Float),
8183 15 => Ok(LogicalType::Double),
8184 16 => Ok(LogicalType::HugeInt),
8185 17 => Ok(LogicalType::UHugeInt),
8186 18 => Ok(LogicalType::Time),
8187 19 => Ok(LogicalType::TimeTz),
8188 20 => Ok(LogicalType::TimestampTz),
8189 21 => Ok(LogicalType::Interval),
8190 22 => Ok(LogicalType::Uuid),
8191 23 => Ok(LogicalType::Blob),
8192 24 => Ok(LogicalType::Bit),
8193 25 => Ok(LogicalType::TimestampS),
8194 26 => Ok(LogicalType::TimestampMs),
8195 27 => Ok(LogicalType::TimestampNs),
8196 _ => Err(invalid("column type tag is unknown")),
8197 }
8198}
8199
8200fn put_u16(out: &mut Vec<u8>, value: u16) {
8201 out.extend_from_slice(&value.to_le_bytes());
8202}
8203fn put_u32(out: &mut Vec<u8>, value: u32) {
8204 out.extend_from_slice(&value.to_le_bytes());
8205}
8206fn put_u64(out: &mut Vec<u8>, value: u64) {
8207 out.extend_from_slice(&value.to_le_bytes());
8208}
8209fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8210 while value >= 0x80 {
8211 out.push((value as u8 & 0x7f) | 0x80);
8212 value >>= 7;
8213 }
8214 out.push(value as u8);
8215}
8216
8217fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8218 match (left, right) {
8219 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8220 (FrequencyValue::Null, _) => Ordering::Less,
8221 (_, FrequencyValue::Null) => Ordering::Greater,
8222 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8223 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8224 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8225 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8226 }
8227}
8228
8229fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8242 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8243 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8244 };
8245 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8246 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8247 let omitted_max = next.count;
8248 entries.truncate(FREQUENCY_ENTRIES);
8249 entries.shrink_to_fit();
8252 omitted_max
8253 } else {
8254 0
8255 };
8256 entries.sort_unstable_by(order);
8257 omitted_max
8258}
8259
8260fn code_frequency(
8261 dictionary: &GlobalDictionary,
8262 flat: &[u8],
8263 bases: &[u64],
8264) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8265 let seen = dictionary.counts.iter().filter(|count| **count != 0).count();
8270 let mut candidates = Vec::with_capacity(seen + usize::from(dictionary.nulls != 0));
8271 candidates.extend(
8272 dictionary
8273 .counts
8274 .iter()
8275 .enumerate()
8276 .filter(|(_, count)| **count != 0)
8277 .map(|(code, &count)| (count, Some(code as u32))),
8278 );
8279 if dictionary.nulls != 0 {
8280 candidates.push((dictionary.nulls, None));
8281 }
8282 let order = |left: &(u64, Option<u32>), right: &(u64, Option<u32>)| {
8284 right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1))
8285 };
8286 let omitted_max = if candidates.len() > FREQUENCY_ENTRIES {
8287 let (_, next, _) = candidates.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8288 let omitted_max = next.0;
8289 candidates.truncate(FREQUENCY_ENTRIES);
8290 omitted_max
8291 } else {
8292 0
8293 };
8294 candidates.sort_unstable_by(order);
8295 let entries = candidates
8296 .into_iter()
8297 .map(|(count, code)| FrequencyEntry {
8298 value: code.map_or(FrequencyValue::Null, FrequencyValue::Code),
8299 count,
8300 })
8301 .collect::<Vec<_>>();
8302 let mut spans = Vec::with_capacity(entries.len());
8303 let mut text_bytes = 0_usize;
8304 for entry in &entries {
8305 let span = match entry.value {
8306 FrequencyValue::Code(code) => {
8307 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8308 let bytes = flat
8309 .get(span.0..span.1)
8310 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8311 text_bytes = text_bytes.saturating_add(bytes.len());
8312 Some(span)
8313 }
8314 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8315 };
8316 spans.push(span);
8317 }
8318 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8319 Vec::new()
8320 } else {
8321 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8322 };
8323 Ok((
8324 FrequencySummary {
8325 entries,
8326 omitted_max,
8327 ordinals: Vec::new(),
8328 ordinal_entries: Vec::new(),
8329 ordinal_bound: 0,
8330 },
8331 texts,
8332 ))
8333}
8334
8335fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8336 let mut out = DIRECTORY.to_vec();
8337 let name = table.name.as_bytes();
8338 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8339 out.extend_from_slice(name);
8340 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8341 for field in &table.fields {
8342 let name = field.name.as_bytes();
8343 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8344 out.extend_from_slice(name);
8345 put_type(&mut out, &field.ty)?;
8346 out.push(u8::from(field.not_null));
8347 }
8348 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8349 match dictionary {
8350 None => out.push(0),
8351 Some(page) => {
8352 out.push(dictionary_tag(&field.ty));
8353 put_u64(&mut out, page.offset);
8354 put_u32(&mut out, page.length);
8355 put_u64(&mut out, page.hash);
8356 }
8357 }
8358 }
8359 for distinct in &table.distincts {
8360 match distinct {
8361 None => out.push(0),
8362 Some(count) => {
8363 out.push(1);
8364 put_u64(&mut out, *count);
8365 }
8366 }
8367 }
8368 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8369 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8370 for stripe in &table.stripes {
8371 put_u32(
8372 &mut out,
8373 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8374 );
8375 for &rows in &stripe.parts {
8376 put_u32(&mut out, rows);
8377 }
8378 put_u64(&mut out, stripe.index.offset);
8379 put_u32(&mut out, stripe.index.length);
8380 for page in &stripe.pages {
8381 put_u64(&mut out, page.offset);
8382 put_u32(&mut out, page.length);
8383 }
8384 for (column, ((field, dictionary), membership)) in
8389 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8390 {
8391 if !coded_type(&field.ty) || dictionary.is_none() {
8392 continue;
8393 }
8394 let page = match membership {
8395 Some(page) => page,
8396 None if table.demoted.get(column).copied().unwrap_or(false) => {
8397 Page { offset: HEADER, length: 0, hash: 0 }
8398 }
8399 None => return Err(invalid("string page has no code membership index")),
8400 };
8401 put_u64(&mut out, page.offset);
8402 put_u32(&mut out, page.length);
8403 put_u64(&mut out, page.hash);
8404 }
8405 for sieve in stripe.sieves.slots() {
8406 match sieve {
8407 None => out.push(0),
8408 Some(page) => {
8409 out.push(1);
8410 put_u64(&mut out, page.offset);
8411 put_u32(&mut out, page.length);
8412 put_u64(&mut out, page.hash);
8413 }
8414 }
8415 }
8416 for held in stripe.part_ranges.slots() {
8417 match held {
8418 None => out.push(0),
8419 Some(page) => {
8420 out.push(1);
8421 put_u64(&mut out, page.offset);
8422 put_u32(&mut out, page.length);
8423 put_u64(&mut out, page.hash);
8424 }
8425 }
8426 }
8427 for range in stripe.zone.columns() {
8428 put_bound(&mut out, range.low.as_ref())?;
8429 put_bound(&mut out, range.high.as_ref())?;
8430 put_u32(
8431 &mut out,
8432 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8433 );
8434 out.push(u8::from(range.exact));
8435 match range.sum {
8436 None => out.push(0),
8437 Some(total) => {
8438 out.push(1);
8439 out.extend_from_slice(&total.to_le_bytes());
8440 }
8441 }
8442 }
8443 }
8444 out.extend_from_slice(FREQUENCIES_SPANS);
8445 put_u16(
8446 &mut out,
8447 u16::try_from(table.frequencies.len())
8448 .map_err(|_| invalid("too many frequency columns"))?,
8449 );
8450 for summary in &table.frequencies {
8451 let summary = match summary {
8452 None => {
8453 put_u32(&mut out, 0);
8454 put_u32(&mut out, 0);
8455 continue;
8456 }
8457 Some(Frequencies::Held(summary)) => summary,
8458 Some(Frequencies::Stored { .. }) => {
8460 return Err(invalid("a synopsis left in the file cannot be written back"));
8461 }
8462 };
8463 let length_at = out.len();
8464 put_u32(&mut out, 0);
8465 put_u32(
8466 &mut out,
8467 u32::try_from(summary.entries.len())
8468 .map_err(|_| invalid("too many frequency entries"))?,
8469 );
8470 let start = out.len();
8471 out.push(1);
8472 put_u64(&mut out, summary.omitted_max);
8473 put_u32(
8474 &mut out,
8475 u32::try_from(summary.entries.len())
8476 .map_err(|_| invalid("too many frequency entries"))?,
8477 );
8478 for entry in &summary.entries {
8479 match entry.value {
8480 FrequencyValue::Null => out.push(0),
8481 FrequencyValue::Integer(value) => {
8482 out.push(1);
8483 out.extend_from_slice(&value.to_le_bytes());
8484 }
8485 FrequencyValue::Code(value) => {
8486 out.push(2);
8487 put_u32(&mut out, value);
8488 }
8489 }
8490 put_u64(&mut out, entry.count);
8491 }
8492 put_u32(
8493 &mut out,
8494 u32::try_from(summary.ordinals.len())
8495 .map_err(|_| invalid("too many frequency ordinals"))?,
8496 );
8497 let mut previous = 0_u64;
8498 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8499 let delta = if at == 0 {
8500 ordinal
8501 } else {
8502 ordinal
8503 .checked_sub(previous)
8504 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8505 };
8506 if at != 0 && delta == 0 {
8507 return Err(invalid("frequency ordinals are not unique"));
8508 }
8509 put_var_u64(&mut out, delta);
8510 previous = ordinal;
8511 }
8512 if summary.ordinal_entries.len() != summary.ordinals.len() {
8513 return Err(invalid("frequency ordinal values have a different length"));
8514 }
8515 for &entry in &summary.ordinal_entries {
8516 if entry as usize >= summary.entries.len() {
8517 return Err(invalid("frequency ordinal value is outside its entries"));
8518 }
8519 put_u16(&mut out, entry);
8520 }
8521 let length = u32::try_from(out.len() - start)
8522 .map_err(|_| invalid("a frequency synopsis is too long"))?;
8523 out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8524 }
8525 let bounds = table
8526 .frequencies
8527 .iter()
8528 .enumerate()
8529 .filter_map(|(column, summary)| match summary {
8530 Some(Frequencies::Held(summary)) if summary.ordinal_bound != 0 => {
8531 Some((column, summary.ordinal_bound))
8532 }
8533 _ => None,
8534 })
8535 .collect::<Vec<_>>();
8536 if !bounds.is_empty() {
8537 out.extend_from_slice(ORDINAL_BOUNDS);
8538 put_u16(&mut out, u16::try_from(bounds.len()).map_err(|_| invalid("too many bounds"))?);
8539 for (column, bound) in bounds {
8540 put_u16(
8541 &mut out,
8542 u16::try_from(column).map_err(|_| invalid("bound column overflows"))?,
8543 );
8544 put_u64(&mut out, bound);
8545 }
8546 }
8547 if !table.pair_frequencies.is_empty() {
8548 out.extend_from_slice(PAIR_FREQUENCIES);
8549 put_u16(
8550 &mut out,
8551 u16::try_from(table.pair_frequencies.len())
8552 .map_err(|_| invalid("too many pair frequency summaries"))?,
8553 );
8554 for summary in &table.pair_frequencies {
8555 put_u16(&mut out, summary.first);
8556 put_u16(&mut out, summary.second);
8557 put_u64(&mut out, summary.omitted_max);
8558 put_u16(
8559 &mut out,
8560 u16::try_from(summary.entries.len())
8561 .map_err(|_| invalid("too many pair frequency entries"))?,
8562 );
8563 for entry in &summary.entries {
8564 put_u16(&mut out, entry.first_entry);
8565 match entry.second {
8566 None => out.push(0),
8567 Some(code) => {
8568 out.push(1);
8569 put_u32(&mut out, code);
8570 }
8571 }
8572 put_u64(&mut out, entry.count);
8573 }
8574 }
8575 }
8576 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8577 if text_columns != 0 {
8578 out.extend_from_slice(FREQUENCY_TEXTS);
8579 put_u16(
8580 &mut out,
8581 u16::try_from(text_columns)
8582 .map_err(|_| invalid("too many string frequency columns"))?,
8583 );
8584 for (column, texts) in table.frequency_texts.iter().enumerate() {
8585 if texts.is_empty() {
8586 continue;
8587 }
8588 put_u16(
8589 &mut out,
8590 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8591 );
8592 put_u16(
8593 &mut out,
8594 u16::try_from(texts.len())
8595 .map_err(|_| invalid("too many frequency text entries"))?,
8596 );
8597 for text in texts {
8598 match text {
8599 None => out.push(0),
8600 Some(text) => {
8601 out.push(1);
8602 put_u32(
8603 &mut out,
8604 u32::try_from(text.len())
8605 .map_err(|_| invalid("frequency text is too long"))?,
8606 );
8607 out.extend_from_slice(text);
8608 }
8609 }
8610 }
8611 }
8612 }
8613 if let Some(summary) = &table.host_groups {
8614 out.extend_from_slice(HOST_GROUPS);
8615 put_u16(
8616 &mut out,
8617 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8618 );
8619 put_u64(&mut out, summary.omitted_max);
8620 put_u16(
8621 &mut out,
8622 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8623 );
8624 for entry in &summary.entries {
8625 put_u32(
8626 &mut out,
8627 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8628 );
8629 out.extend_from_slice(entry.host.as_bytes());
8630 put_u64(&mut out, entry.count);
8631 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8632 put_u32(
8633 &mut out,
8634 u32::try_from(entry.minimum.len())
8635 .map_err(|_| invalid("host minimum is too long"))?,
8636 );
8637 out.extend_from_slice(entry.minimum.as_bytes());
8638 }
8639 }
8640 if let Some(clustering) = &table.clustering {
8643 out.extend_from_slice(CLUSTERING);
8644 out.push(clustering.width().tag());
8645 put_u16(
8646 &mut out,
8647 u16::try_from(clustering.columns().len())
8648 .map_err(|_| invalid("too many clustering columns"))?,
8649 );
8650 for &column in clustering.columns() {
8651 put_u16(
8652 &mut out,
8653 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8654 );
8655 }
8656 }
8657 let demoted = (0..table.fields.len())
8658 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8659 .collect::<Vec<_>>();
8660 if !demoted.is_empty() {
8661 out.extend_from_slice(DEMOTED);
8662 put_u16(
8663 &mut out,
8664 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8665 );
8666 for column in demoted {
8667 put_u16(
8668 &mut out,
8669 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8670 );
8671 }
8672 }
8673 if !table.constraints.is_empty() {
8674 out.extend_from_slice(KEYS);
8675 put_count(&mut out, table.constraints.keys.len())?;
8676 for (columns, primary) in &table.constraints.keys {
8677 out.push(u8::from(*primary));
8678 put_columns(&mut out, columns)?;
8679 }
8680 put_count(&mut out, table.constraints.foreign.len())?;
8681 for foreign in &table.constraints.foreign {
8682 put_columns(&mut out, &foreign.columns)?;
8683 put_columns(&mut out, &foreign.referenced)?;
8684 put_u32(
8685 &mut out,
8686 u32::try_from(foreign.table.len())
8687 .map_err(|_| invalid("table name is too long"))?,
8688 );
8689 out.extend_from_slice(foreign.table.as_bytes());
8690 }
8691 }
8692 out.extend_from_slice(SECTIONS);
8698 put_u64(&mut out, table.generation);
8699 put_u16(
8700 &mut out,
8701 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8702 );
8703 for held in &table.sections {
8704 held.encode(&mut out)?;
8705 }
8706 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8707 out.extend_from_slice(DICTIONARY_PAYLOADS);
8708 put_u16(
8709 &mut out,
8710 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8711 );
8712 for at in 0..table.fields.len() {
8713 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8714 }
8715 }
8716 Ok(out)
8717}
8718
8719fn signed_integer(ty: &LogicalType) -> bool {
8728 matches!(
8729 ty,
8730 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8731 )
8732}
8733
8734fn integer_or_date(ty: &LogicalType) -> bool {
8735 matches!(
8736 ty,
8737 LogicalType::TinyInt
8738 | LogicalType::SmallInt
8739 | LogicalType::Integer
8740 | LogicalType::BigInt
8741 | LogicalType::UTinyInt
8742 | LogicalType::USmallInt
8743 | LogicalType::UInteger
8744 | LogicalType::UBigInt
8745 | LogicalType::Date
8746 )
8747}
8748
8749fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8750 table
8751 .fields
8752 .iter()
8753 .enumerate()
8754 .map(|(column, field)| {
8755 if !integer_or_date(&field.ty) {
8756 return None;
8757 }
8758 let mut low: Option<i128> = None;
8759 let mut high: Option<i128> = None;
8760 for stripe in &table.stripes {
8761 let range = stripe.zone.column(column)?;
8762 if !range.exact {
8763 return None;
8764 }
8765 match (range.low.as_ref(), range.high.as_ref()) {
8766 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8767 low = Some(low.map_or(*small, |held| held.min(*small)));
8768 high = Some(high.map_or(*large, |held| held.max(*large)));
8769 }
8770 (None, None) if stripe.rows == range.nulls => {}
8771 _ => return None,
8772 }
8773 }
8774 Some(low.zip(high))
8775 })
8776 .collect()
8777}
8778
8779fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8780 reader
8781 .table
8782 .fields
8783 .iter()
8784 .enumerate()
8785 .map(|(column, field)| {
8786 if !integer_or_date(&field.ty) {
8787 return Ok(None);
8788 }
8789 match reader.exact_extremes(column)? {
8790 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8791 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8792 _ => Ok(None),
8793 }
8794 })
8795 .collect()
8796}
8797
8798fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8799 table
8800 .fields
8801 .iter()
8802 .enumerate()
8803 .map(|(column, field)| {
8804 if !integer_or_date(&field.ty) {
8805 return None;
8806 }
8807 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8808 return None;
8809 };
8810 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8811 return None;
8812 }
8813 let entries = summary
8814 .entries
8815 .iter()
8816 .map(|entry| {
8817 let value = match entry.value {
8818 FrequencyValue::Null => None,
8819 FrequencyValue::Integer(value) => Some(value),
8820 FrequencyValue::Code(_) => return None,
8821 };
8822 Some((value, entry.count))
8823 })
8824 .collect::<Option<Vec<_>>>()?;
8825 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8826 (rows == table.rows as u64).then_some(entries)
8827 })
8828 .collect()
8829}
8830
8831fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8838 if signed {
8839 FrequencyValue::Integer(i128::from(bits as i64))
8840 } else {
8841 FrequencyValue::Integer(i128::from(bits))
8842 }
8843}
8844
8845fn frequency_bits(value: &Value) -> Option<u64> {
8846 Some(match value {
8847 Value::TinyInt(value) => i64::from(*value) as u64,
8848 Value::SmallInt(value) => i64::from(*value) as u64,
8849 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8850 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8851 Value::UTinyInt(value) => u64::from(*value),
8852 Value::USmallInt(value) => u64::from(*value),
8853 Value::UInteger(value) => u64::from(*value),
8854 Value::UBigInt(value) => *value,
8855 _ => return None,
8856 })
8857}
8858
8859fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8860 Some(match value {
8861 Value::Null => None,
8862 Value::TinyInt(value) => Some(i128::from(*value)),
8863 Value::SmallInt(value) => Some(i128::from(*value)),
8864 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8865 Value::BigInt(value) => Some(i128::from(*value)),
8866 Value::UTinyInt(value) => Some(i128::from(*value)),
8867 Value::USmallInt(value) => Some(i128::from(*value)),
8868 Value::UInteger(value) => Some(i128::from(*value)),
8869 Value::UBigInt(value) => Some(i128::from(*value)),
8870 _ => return None,
8871 })
8872}
8873
8874fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8875 reader
8876 .table
8877 .fields
8878 .iter()
8879 .enumerate()
8880 .map(|(column, field)| {
8881 if !integer_or_date(&field.ty) {
8882 return Ok(None);
8883 }
8884 let Some((entries, omitted_max)) = reader.frequency_head(column)? else {
8885 return Ok(None);
8886 };
8887 if omitted_max != 0 || entries.len() > MAX_CATALOG_FREQUENCIES {
8888 return Ok(None);
8889 }
8890 let entries = reader.decode_frequencies(column, &field.ty, &entries)?;
8891 let Some(entries) = entries
8892 .iter()
8893 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8894 .collect::<Option<Vec<_>>>()
8895 else {
8896 return Ok(None);
8897 };
8898 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8899 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8900 })
8901 .collect()
8902}
8903
8904fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8905 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8906 let range = stripe.zone.column(column)?;
8907 let sum = sum.checked_add(range.sum?)?;
8908 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8909 Some((sum, count.checked_add(nonnull)?))
8910 })
8911}
8912
8913fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8914 table
8915 .fields
8916 .iter()
8917 .enumerate()
8918 .map(|(column, field)| {
8919 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8920 })
8921 .collect()
8922}
8923
8924fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8925 reader
8926 .table
8927 .fields
8928 .iter()
8929 .enumerate()
8930 .map(
8931 |(column, field)| {
8932 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8933 },
8934 )
8935 .collect()
8936}
8937
8938fn encode_catalog(
8939 entries: &[Entry],
8940 views: &[ViewEntry],
8941 card: Option<&KeptCard>,
8942 anchor: Option<&LogAnchor>,
8943) -> Result<Vec<u8>> {
8944 let mut out = CATALOG.to_vec();
8945 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8946 for entry in entries {
8947 let name = entry.name.as_bytes();
8948 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8949 out.extend_from_slice(name);
8950 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8951 put_u16(
8952 &mut out,
8953 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8954 );
8955 for field in &entry.fields {
8956 let name = field.name.as_bytes();
8957 put_u16(
8958 &mut out,
8959 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8960 );
8961 out.extend_from_slice(name);
8962 put_type(&mut out, &field.ty)?;
8963 out.push(u8::from(field.not_null));
8964 }
8965 put_u64(&mut out, entry.directory.offset);
8966 put_u32(&mut out, entry.directory.length);
8967 put_u64(&mut out, entry.directory.hash);
8968 }
8969 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8970 for view in views {
8971 let name = view.name.as_bytes();
8972 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8973 out.extend_from_slice(name);
8974 put_long_text(&mut out, &view.sql, "view body")?;
8975 put_long_text(&mut out, &view.statement, "view statement")?;
8976 put_u16(
8977 &mut out,
8978 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8979 );
8980 for alias in &view.aliases {
8981 let alias = alias.as_bytes();
8982 put_u16(
8983 &mut out,
8984 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8985 );
8986 out.extend_from_slice(alias);
8987 }
8988 put_u16(
8989 &mut out,
8990 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8991 );
8992 for field in &view.columns {
8993 let name = field.name.as_bytes();
8994 put_u16(
8995 &mut out,
8996 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8997 );
8998 out.extend_from_slice(name);
8999 put_type(&mut out, &field.ty)?;
9000 out.push(u8::from(field.not_null));
9001 }
9002 }
9003 out.extend_from_slice(NONZERO_COUNTS);
9004 for entry in entries {
9005 if entry.nonzero.len() != entry.fields.len() {
9006 return Err(invalid("nonzero count width differs from schema"));
9007 }
9008 for count in &entry.nonzero {
9009 match count {
9010 None => out.push(0),
9011 Some(count) => {
9012 out.push(1);
9013 put_u64(&mut out, *count);
9014 }
9015 }
9016 }
9017 }
9018 out.extend_from_slice(AGGREGATE_SUMS);
9019 for entry in entries {
9020 if entry.aggregates.len() != entry.fields.len() {
9021 return Err(invalid("aggregate sum width differs from schema"));
9022 }
9023 for summary in &entry.aggregates {
9024 match summary {
9025 None => out.push(0),
9026 Some((sum, count)) => {
9027 out.push(1);
9028 out.extend_from_slice(&sum.to_le_bytes());
9029 put_u64(&mut out, *count);
9030 }
9031 }
9032 }
9033 }
9034 out.extend_from_slice(DISTINCT_COUNTS);
9035 for entry in entries {
9036 if entry.distincts.len() != entry.fields.len() {
9037 return Err(invalid("distinct count width differs from schema"));
9038 }
9039 for count in &entry.distincts {
9040 match count {
9041 None => out.push(0),
9042 Some(count) => {
9043 if *count > entry.rows as u64 {
9044 return Err(invalid("distinct count exceeds table rows"));
9045 }
9046 out.push(1);
9047 put_u64(&mut out, *count);
9048 }
9049 }
9050 }
9051 }
9052 out.extend_from_slice(INTEGER_EXTREMES);
9053 for entry in entries {
9054 if entry.extremes.len() != entry.fields.len() {
9055 return Err(invalid("integer extremes width differs from schema"));
9056 }
9057 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
9058 match extremes {
9059 None => out.push(0),
9060 Some(None) if integer_or_date(&field.ty) => out.push(1),
9061 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
9062 out.push(2);
9063 out.extend_from_slice(&low.to_le_bytes());
9064 out.extend_from_slice(&high.to_le_bytes());
9065 }
9066 _ => return Err(invalid("integer extremes type or range differs")),
9067 }
9068 }
9069 }
9070 out.extend_from_slice(COMPLETE_FREQUENCIES);
9071 for entry in entries {
9072 if entry.frequencies.len() != entry.fields.len() {
9073 return Err(invalid("numeric frequency width differs from schema"));
9074 }
9075 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
9076 match frequencies {
9077 None => out.push(0),
9078 Some(entries)
9079 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
9080 {
9081 let mut total = 0_u64;
9082 for (at, (value, count)) in entries.iter().enumerate() {
9083 if entries[..at].iter().any(|(held, _)| held == value) {
9084 return Err(invalid("numeric frequency value repeats"));
9085 }
9086 total = total
9087 .checked_add(*count)
9088 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9089 }
9090 if total != entry.rows as u64 {
9091 return Err(invalid("numeric frequencies do not cover table rows"));
9092 }
9093 out.push(1);
9094 out.push(entries.len() as u8);
9095 for (value, count) in entries {
9096 match value {
9097 None => out.push(0),
9098 Some(value) => {
9099 out.push(1);
9100 out.extend_from_slice(&value.to_le_bytes());
9101 }
9102 }
9103 put_u64(&mut out, *count);
9104 }
9105 }
9106 _ => return Err(invalid("numeric frequency type or width differs")),
9107 }
9108 }
9109 }
9110 if let Some(card) = card {
9111 out.extend_from_slice(DEVICE_CARD);
9112 let device = card.device.as_bytes();
9113 put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
9114 out.extend_from_slice(device);
9115 put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
9116 out.extend_from_slice(&card.bytes);
9117 }
9118 if let Some(anchor) = anchor {
9119 anchor.encode(&mut out)?;
9120 }
9121 Ok(out)
9122}
9123
9124fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
9126 let bytes = text.as_bytes();
9127 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
9128 out.extend_from_slice(bytes);
9129 Ok(())
9130}
9131
9132fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
9135 let mut cur = Cursor::new(bytes);
9136 if cur.take(8)? != CATALOG {
9137 return Err(invalid("catalog magic differs"));
9138 }
9139 let count = cur.u32()? as usize;
9140 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
9141 for _ in 0..count {
9142 let name = cur.text()?;
9143 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9144 let width = cur.u16()? as usize;
9145 let mut fields = Vec::with_capacity(width);
9146 for _ in 0..width {
9147 let name = cur.text()?;
9148 let ty = read_type(&mut cur)?;
9149 let not_null = match cur.u8()? {
9150 0 => false,
9151 1 => true,
9152 _ => return Err(invalid("nullability flag differs")),
9153 };
9154 fields.push(Field { name, ty, not_null });
9155 }
9156 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9157 let end = directory
9158 .offset
9159 .checked_add(u64::from(directory.length))
9160 .ok_or_else(|| invalid("table directory offset overflow"))?;
9161 if directory.offset < HEADER
9162 || end > size
9163 || directory.length as usize > MAX_DIRECTORY
9164 || directory.length == 0
9165 {
9166 return Err(invalid("table directory range is outside the file"));
9167 }
9168 if entries.iter().any(|held| held.name == name) {
9169 return Err(invalid("two tables in the catalog have the same name"));
9170 }
9171 let nonzero = vec![None; fields.len()];
9172 let aggregates = vec![None; fields.len()];
9173 let distincts = vec![None; fields.len()];
9174 let extremes = vec![None; fields.len()];
9175 let frequencies = vec![None; fields.len()];
9176 entries.push(Entry {
9177 name,
9178 fields,
9179 rows,
9180 directory,
9181 nonzero,
9182 aggregates,
9183 distincts,
9184 extremes,
9185 frequencies,
9186 });
9187 }
9188 let count = if cur.done() { 0 } else { cur.u32()? as usize };
9193 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9194 for _ in 0..count {
9195 let name = cur.text()?;
9196 let sql = cur.long_text()?;
9197 let statement = cur.long_text()?;
9198 let width = cur.u16()? as usize;
9199 let mut aliases = Vec::with_capacity(width);
9200 for _ in 0..width {
9201 aliases.push(cur.text()?);
9202 }
9203 let width = cur.u16()? as usize;
9204 let mut columns = Vec::with_capacity(width);
9205 for _ in 0..width {
9206 let name = cur.text()?;
9207 let ty = read_type(&mut cur)?;
9208 let not_null = match cur.u8()? {
9209 0 => false,
9210 1 => true,
9211 _ => return Err(invalid("nullability flag differs")),
9212 };
9213 columns.push(Field { name, ty, not_null });
9214 }
9215 if views.iter().any(|held| held.name == name) {
9219 return Err(invalid("two views in the catalog have the same name"));
9220 }
9221 if entries.iter().any(|held| held.name == name) {
9222 return Err(invalid("a table and a view in the catalog have the same name"));
9223 }
9224 views.push(ViewEntry { name, sql, statement, aliases, columns });
9225 }
9226 if !cur.done() {
9227 if cur.take(8)? != NONZERO_COUNTS {
9228 return Err(invalid("catalog extension magic differs"));
9229 }
9230 for entry in &mut entries {
9231 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9232 *count = match cur.u8()? {
9233 0 => None,
9234 1 if matches!(
9235 field.ty,
9236 LogicalType::TinyInt
9237 | LogicalType::SmallInt
9238 | LogicalType::Integer
9239 | LogicalType::BigInt
9240 | LogicalType::UTinyInt
9241 | LogicalType::USmallInt
9242 | LogicalType::UInteger
9243 | LogicalType::UBigInt
9244 ) =>
9245 {
9246 let value = cur.u64()?;
9247 if value > entry.rows as u64 {
9248 return Err(invalid("nonzero count exceeds rows"));
9249 }
9250 Some(value)
9251 }
9252 _ => return Err(invalid("nonzero count tag or column type differs")),
9253 };
9254 }
9255 }
9256 }
9257 if !cur.done() {
9258 if cur.take(8)? != AGGREGATE_SUMS {
9259 return Err(invalid("aggregate catalog extension magic differs"));
9260 }
9261 for entry in &mut entries {
9262 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9263 *summary = match cur.u8()? {
9264 0 => None,
9265 1 if signed_integer(&field.ty) => {
9266 let sum = i128::from_le_bytes(
9267 cur.take(16)?
9268 .try_into()
9269 .map_err(|_| invalid("aggregate sum is truncated"))?,
9270 );
9271 let count = cur.u64()?;
9272 if count > entry.rows as u64 {
9273 return Err(invalid("aggregate count exceeds table rows"));
9274 }
9275 Some((sum, count))
9276 }
9277 _ => return Err(invalid("aggregate sum tag or column type differs")),
9278 };
9279 }
9280 }
9281 }
9282 if !cur.done() {
9283 if cur.take(8)? != DISTINCT_COUNTS {
9284 return Err(invalid("distinct catalog extension magic differs"));
9285 }
9286 for entry in &mut entries {
9287 for count in &mut entry.distincts {
9288 *count = match cur.u8()? {
9289 0 => None,
9290 1 => {
9291 let value = cur.u64()?;
9292 if value > entry.rows as u64 {
9293 return Err(invalid("distinct count exceeds table rows"));
9294 }
9295 Some(value)
9296 }
9297 _ => return Err(invalid("distinct count tag differs")),
9298 };
9299 }
9300 }
9301 }
9302 if !cur.done() {
9303 if cur.take(8)? != INTEGER_EXTREMES {
9304 return Err(invalid("integer extremes catalog extension magic differs"));
9305 }
9306 for entry in &mut entries {
9307 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9308 *extremes = match cur.u8()? {
9309 0 => None,
9310 1 if integer_or_date(&field.ty) => Some(None),
9311 2 if integer_or_date(&field.ty) => {
9312 let low = i128::from_le_bytes(
9313 cur.take(16)?
9314 .try_into()
9315 .map_err(|_| invalid("minimum is truncated"))?,
9316 );
9317 let high = i128::from_le_bytes(
9318 cur.take(16)?
9319 .try_into()
9320 .map_err(|_| invalid("maximum is truncated"))?,
9321 );
9322 if low > high {
9323 return Err(invalid("integer extremes are reversed"));
9324 }
9325 Some(Some((low, high)))
9326 }
9327 _ => return Err(invalid("integer extremes tag or type differs")),
9328 };
9329 }
9330 }
9331 }
9332 if !cur.done() {
9333 if cur.take(8)? != COMPLETE_FREQUENCIES {
9334 return Err(invalid("numeric frequency catalog extension magic differs"));
9335 }
9336 for entry in &mut entries {
9337 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9338 *frequencies = match cur.u8()? {
9339 0 => None,
9340 1 if integer_or_date(&field.ty) => {
9341 let len = cur.u8()? as usize;
9342 if len > MAX_CATALOG_FREQUENCIES {
9343 return Err(invalid("too many catalog numeric frequencies"));
9344 }
9345 let mut values = Vec::with_capacity(len);
9346 let mut total = 0_u64;
9347 for _ in 0..len {
9348 let value = match cur.u8()? {
9349 0 => None,
9350 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9351 |_| invalid("numeric frequency value is truncated"),
9352 )?)),
9353 _ => return Err(invalid("numeric frequency value tag differs")),
9354 };
9355 if values.iter().any(|(held, _)| *held == value) {
9356 return Err(invalid("numeric frequency value repeats"));
9357 }
9358 let count = cur.u64()?;
9359 total = total
9360 .checked_add(count)
9361 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9362 values.push((value, count));
9363 }
9364 if total != entry.rows as u64 {
9365 return Err(invalid("numeric frequencies do not cover table rows"));
9366 }
9367 Some(values)
9368 }
9369 _ => return Err(invalid("numeric frequency tag or type differs")),
9370 };
9371 }
9372 }
9373 }
9374 let mut card = None;
9375 let mut anchor = None;
9376 while !cur.done() {
9379 let tag = cur.take(8)?;
9380 if tag == DEVICE_CARD && card.is_none() && anchor.is_none() {
9381 let device = cur.text()?;
9382 let len = cur.u32()? as usize;
9383 if len > MAX_CARD {
9384 return Err(invalid("device card is longer than any card"));
9385 }
9386 card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9387 } else if tag == anchor::LOG_ANCHOR && anchor.is_none() {
9388 anchor = Some(LogAnchor::decode(&mut cur)?);
9389 } else {
9390 return Err(invalid("catalog extension magic differs or repeats"));
9391 }
9392 }
9393 Ok((entries, views, card, anchor))
9394}
9395
9396type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>, Option<LogAnchor>);
9398
9399const MAX_CARD: usize = 64 << 10;
9401
9402#[derive(Debug, Clone, PartialEq, Eq)]
9409struct KeptCard {
9410 device: String,
9411 bytes: Vec<u8>,
9412}
9413
9414fn directory_of(path: &Path) -> &Path {
9416 path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9417}
9418
9419fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9426 let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9427 return held;
9428 };
9429 match rudb_io::device::kept(&device) {
9430 Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9431 None => held,
9432 }
9433}
9434
9435fn remember_card(path: &Path, card: Option<&KeptCard>) {
9437 let Some(card) = card else { return };
9438 let dir = directory_of(path);
9439 let Ok(device) = rudb_io::device::device_key(dir) else { return };
9440 if device != card.device {
9441 return;
9442 }
9443 if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9444 rudb_io::device::remember(&device, decoded);
9445 }
9446}
9447
9448struct Cursor<'a> {
9456 bytes: &'a [u8],
9457 at: usize,
9458 window: Option<Window<'a>>,
9459}
9460
9461struct Window<'a> {
9463 file: &'a File,
9464 offset: u64,
9465 length: usize,
9466 start: usize,
9468 held: Vec<u8>,
9469 size: usize,
9471}
9472
9473const DIRECTORY_WINDOW: usize = 64 << 10;
9475
9476impl<'a> Cursor<'a> {
9477 fn new(bytes: &'a [u8]) -> Self {
9478 Self { bytes, at: 0, window: None }
9479 }
9480
9481 fn over(file: &'a File, offset: u64, length: usize) -> Self {
9483 let window =
9484 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9485 Self { bytes: &[], at: 0, window: Some(window) }
9486 }
9487
9488 fn len(&self) -> usize {
9490 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9491 }
9492
9493 fn ensure(&mut self, len: usize) -> Result<()> {
9495 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9496 if end > self.len() {
9497 return Err(invalid("directory is truncated"));
9498 }
9499 let Some(window) = &mut self.window else { return Ok(()) };
9500 if self.at < window.start || end > window.start + window.held.len() {
9501 let want = len.max(window.size).min(window.length - self.at);
9502 window.start = self.at;
9503 window.held.resize(want, 0);
9504 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9505 }
9506 Ok(())
9507 }
9508
9509 fn held(&self, at: usize, len: usize) -> &[u8] {
9511 match &self.window {
9512 Some(window) => &window.held[at - window.start..at - window.start + len],
9513 None => &self.bytes[at..at + len],
9514 }
9515 }
9516
9517 #[inline]
9519 fn peek(&mut self, len: usize) -> Result<&[u8]> {
9520 if self.window.is_none() {
9521 let bytes = self.bytes;
9522 return Ok(&bytes[self.at..self.end(len)?]);
9523 }
9524 self.ensure(len)?;
9525 Ok(self.held(self.at, len))
9526 }
9527
9528 #[inline]
9534 fn take(&mut self, len: usize) -> Result<&[u8]> {
9535 if self.window.is_none() {
9536 let bytes = self.bytes;
9537 let (at, end) = (self.at, self.end(len)?);
9538 self.at = end;
9539 return Ok(&bytes[at..end]);
9540 }
9541 self.take_windowed(len)
9542 }
9543
9544 fn skip(&mut self, len: usize) -> Result<()> {
9546 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9547 if end > self.len() {
9548 return Err(invalid("directory is truncated"));
9549 }
9550 self.at = end;
9551 Ok(())
9552 }
9553
9554 fn skip_bound(&mut self) -> Result<()> {
9555 match self.u8()? {
9556 0 => Ok(()),
9557 1 => self.skip(16),
9558 2 => self.skip(8),
9559 3 => {
9560 let length = self.u32()? as usize;
9561 self.skip(length)
9562 }
9563 4 => self.skip(17),
9564 _ => Err(invalid("a stored bound has an unknown tag")),
9565 }
9566 }
9567
9568 #[inline]
9570 fn end(&self, len: usize) -> Result<usize> {
9571 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9572 if end > self.bytes.len() {
9573 return Err(invalid("directory is truncated"));
9574 }
9575 Ok(end)
9576 }
9577
9578 #[inline(never)]
9580 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9581 self.ensure(len)?;
9582 self.at += len;
9583 Ok(self.held(self.at - len, len))
9584 }
9585 #[inline]
9586 fn u8(&mut self) -> Result<u8> {
9587 Ok(self.take(1)?[0])
9588 }
9589 #[inline]
9590 fn u16(&mut self) -> Result<u16> {
9591 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9592 }
9593 #[inline]
9594 fn u32(&mut self) -> Result<u32> {
9595 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9596 }
9597 #[inline]
9598 fn u64(&mut self) -> Result<u64> {
9599 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9600 }
9601 fn var_u64(&mut self) -> Result<u64> {
9602 let mut value = 0_u64;
9603 for shift in (0..=63).step_by(7) {
9604 let byte = self.u8()?;
9605 let part = u64::from(byte & 0x7f);
9606 if shift == 63 && part > 1 {
9607 return Err(invalid("frequency ordinal varint overflows"));
9608 }
9609 value |= part << shift;
9610 if byte & 0x80 == 0 {
9611 return Ok(value);
9612 }
9613 }
9614 Err(invalid("frequency ordinal varint is too long"))
9615 }
9616 fn bound(&mut self) -> Result<Option<Bound>> {
9625 let rest = self.len().saturating_sub(self.at);
9626 let mut want = 32;
9627 loop {
9628 let offered = self.peek(want.min(rest))?;
9629 let mut used = 0;
9630 match bounds::get(offered, &mut used) {
9631 Ok(bound) => {
9632 self.at += used;
9633 return Ok(bound);
9634 }
9635 Err(_) if want < rest => want *= 2,
9636 Err(error) => return Err(error),
9637 }
9638 }
9639 }
9640 fn text(&mut self) -> Result<String> {
9641 let len = self.u16()? as usize;
9642 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9643 }
9644 fn done(&self) -> bool {
9647 self.at >= self.len()
9648 }
9649 fn long_text(&mut self) -> Result<String> {
9656 let len = self.u32()? as usize;
9657 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9658 }
9659}
9660
9661fn decode_summary(
9663 cur: &mut Cursor<'_>,
9664 field: &Field,
9665 rows: usize,
9666 values: bool,
9667) -> Result<Option<FrequencySummary>> {
9668 let Some((entries, omitted_max)) = decode_summary_head(cur, field, rows)? else {
9669 return Ok(None);
9670 };
9671 let ordinals = {
9672 let ordinal_count = cur.u32()? as usize;
9673 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9674 return Err(invalid("frequency ordinal count exceeds its bound"));
9675 }
9676 let mut ordinals = Vec::with_capacity(ordinal_count);
9677 let mut previous = 0_u64;
9678 for at in 0..ordinal_count {
9679 let delta = cur.var_u64()?;
9680 if at != 0 && delta == 0 {
9681 return Err(invalid("frequency ordinals are not increasing"));
9682 }
9683 let ordinal = if at == 0 {
9684 delta
9685 } else {
9686 previous.checked_add(delta).ok_or_else(|| invalid("frequency ordinal overflows"))?
9687 };
9688 if ordinal >= rows as u64 {
9689 return Err(invalid("frequency ordinal is outside the table"));
9690 }
9691 ordinals.push(ordinal);
9692 previous = ordinal;
9693 }
9694 ordinals
9695 };
9696 let ordinal_entries = if values {
9697 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9698 for _ in 0..ordinals.len() {
9699 let entry = cur.u16()?;
9700 if entry as usize >= entries.len() {
9701 return Err(invalid("frequency ordinal value is outside its entries"));
9702 }
9703 ordinal_entries.push(entry);
9704 }
9705 ordinal_entries
9706 } else {
9707 Vec::new()
9708 };
9709 Ok(Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries, ordinal_bound: 0 }))
9710}
9711
9712fn decode_summary_head(
9715 cur: &mut Cursor<'_>,
9716 field: &Field,
9717 rows: usize,
9718) -> Result<Option<(Vec<FrequencyEntry>, u64)>> {
9719 Ok(match cur.u8()? {
9720 0 => None,
9721 1 => {
9722 let omitted_max = cur.u64()?;
9723 let count = cur.u32()? as usize;
9724 if count > FREQUENCY_ENTRIES {
9725 return Err(invalid("frequency entry count exceeds its bound"));
9726 }
9727 let mut entries = Vec::with_capacity(count);
9728 for _ in 0..count {
9730 let value = match cur.u8()? {
9731 0 => FrequencyValue::Null,
9732 1 => FrequencyValue::Integer(i128::from_le_bytes(
9733 cur.take(16)?.try_into().expect("sixteen bytes"),
9734 )),
9735 2 => FrequencyValue::Code(cur.u32()?),
9736 _ => return Err(invalid("frequency value tag differs")),
9737 };
9738 let valid = matches!(
9739 (&field.ty, value),
9740 (_, FrequencyValue::Null)
9741 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9742 | (
9743 LogicalType::TinyInt
9744 | LogicalType::SmallInt
9745 | LogicalType::Integer
9746 | LogicalType::BigInt
9747 | LogicalType::UTinyInt
9748 | LogicalType::USmallInt
9749 | LogicalType::UInteger
9750 | LogicalType::UBigInt
9751 | LogicalType::Date
9752 | LogicalType::Timestamp,
9753 FrequencyValue::Integer(_),
9754 )
9755 );
9756 if !valid {
9757 return Err(invalid("frequency value does not match its column"));
9758 }
9759 let count = cur.u64()?;
9760 if count == 0 || count > rows as u64 {
9761 return Err(invalid("frequency count is outside the table"));
9762 }
9763 entries.push(FrequencyEntry { value, count });
9764 }
9765 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9766 return Err(invalid("frequency entries are not descending"));
9767 }
9768 Some((entries, omitted_max))
9769 }
9770 _ => return Err(invalid("frequency summary tag differs")),
9771 })
9772}
9773
9774fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9776 let length = cur.u32()? as usize;
9777 let entries = cur.u32()? as usize;
9778 if entries > FREQUENCY_ENTRIES {
9779 return Err(invalid("frequency entry count exceeds its bound"));
9780 }
9781 if length == 0 {
9782 if entries != 0 {
9783 return Err(invalid("missing frequency synopsis has entries"));
9784 }
9785 return Ok(None);
9786 }
9787 if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9788 return Err(invalid("frequency synopsis span is outside the directory"));
9789 }
9790 Ok(Some((length, entries)))
9791}
9792
9793fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9796 match cur.u8()? {
9797 0 => Ok(()),
9798 1 => {
9799 cur.skip(8)?;
9800 let entries = cur.u32()? as usize;
9801 if entries > FREQUENCY_ENTRIES {
9802 return Err(invalid("frequency entry count exceeds its bound"));
9803 }
9804 for _ in 0..entries {
9805 match cur.u8()? {
9806 0 => {}
9807 1 => cur.skip(16)?,
9808 2 => cur.skip(4)?,
9809 _ => return Err(invalid("frequency value tag differs")),
9810 }
9811 cur.skip(8)?;
9812 }
9813 let ordinals = cur.u32()? as usize;
9814 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9815 return Err(invalid("frequency ordinal count exceeds its bound"));
9816 }
9817 for _ in 0..ordinals {
9818 cur.var_u64()?;
9819 }
9820 if values {
9821 cur.skip(ordinals * 2)?;
9822 }
9823 Ok(())
9824 }
9825 _ => Err(invalid("frequency summary tag differs")),
9826 }
9827}
9828
9829fn quick_nonzero(
9833 mut cur: Cursor<'_>,
9834 name: &str,
9835 fields: &[Field],
9836 rows: usize,
9837 wanted: usize,
9838) -> Result<Option<u64>> {
9839 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9840 return Err(invalid("table directory differs from the catalog"));
9841 }
9842 let width = cur.u16()? as usize;
9843 if width != fields.len() {
9844 return Err(invalid("table directory width differs from the catalog"));
9845 }
9846 for field in fields {
9847 let stored =
9848 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9849 if &stored != field {
9850 return Err(invalid("table directory schema differs from the catalog"));
9851 }
9852 }
9853 let mut dictionaries = Vec::with_capacity(width);
9854 for field in fields {
9855 let held = match cur.u8()? {
9856 0 => false,
9857 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9858 cur.skip(20)?;
9859 true
9860 }
9861 _ => return Err(invalid("dictionary page tag differs")),
9862 };
9863 dictionaries.push(held);
9864 }
9865 for _ in 0..width {
9866 match cur.u8()? {
9867 0 => {}
9868 1 => cur.skip(8)?,
9869 _ => return Err(invalid("distinct count tag differs")),
9870 }
9871 }
9872 if cur.u64()? != rows as u64 {
9873 return Err(invalid("table row count differs from the catalog"));
9874 }
9875 let stripes = cur.u32()? as usize;
9876 let mut total = 0_usize;
9877 let mut nulls = 0_u64;
9878 for _ in 0..stripes {
9879 let parts = cur.u32()? as usize;
9880 if parts == 0 || parts > STRIPE_PARTS {
9881 return Err(invalid("stripe part count is outside its bound"));
9882 }
9883 let mut stripe_rows = 0_usize;
9884 for _ in 0..parts {
9885 stripe_rows = stripe_rows
9886 .checked_add(cur.u32()? as usize)
9887 .ok_or_else(|| invalid("stripe row count overflow"))?;
9888 }
9889 total =
9890 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9891 cur.skip(12 + width * 12)?;
9892 for (field, held) in fields.iter().zip(&dictionaries) {
9893 if coded_type(&field.ty) && *held {
9894 cur.skip(20)?;
9895 }
9896 }
9897 for _ in 0..width * 2 {
9898 match cur.u8()? {
9899 0 => {}
9900 1 => cur.skip(20)?,
9901 _ => return Err(invalid("stripe page tag differs")),
9902 }
9903 }
9904 for column in 0..width {
9905 cur.skip_bound()?;
9906 cur.skip_bound()?;
9907 let count = cur.u32()? as u64;
9908 if count > stripe_rows as u64 {
9909 return Err(invalid("null count exceeds stripe rows"));
9910 }
9911 if column == wanted {
9912 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9913 }
9914 cur.skip(1)?;
9915 match cur.u8()? {
9916 0 => {}
9917 1 => cur.skip(16)?,
9918 _ => return Err(invalid("a stripe sum has an unknown tag")),
9919 }
9920 }
9921 }
9922 if total != rows {
9923 return Err(invalid("table row count differs from stripes"));
9924 }
9925 if cur.done() {
9926 return Ok(None);
9927 }
9928 let magic = cur.take(8)?;
9929 let spanned = magic == FREQUENCIES_SPANS;
9930 let values = magic == FREQUENCIES || spanned;
9931 if !values && magic != FREQUENCIES_V2 {
9932 return Err(invalid("directory extension magic differs"));
9933 }
9934 if cur.u16()? as usize != width {
9935 return Err(invalid("frequency column count differs"));
9936 }
9937 for _ in 0..wanted {
9938 if spanned {
9939 if let Some((length, _)) = summary_span(&mut cur)? {
9940 cur.skip(length)?;
9941 }
9942 } else {
9943 skip_summary(&mut cur, values, rows)?;
9944 }
9945 }
9946 let summary = if spanned {
9947 let Some((length, entries)) = summary_span(&mut cur)? else {
9948 return Ok(None);
9949 };
9950 let start = cur.at;
9951 let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
9952 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
9953 if cur.at - start != length || summary.entries.len() != entries {
9954 return Err(invalid("a stored synopsis differs from its directory span"));
9955 }
9956 summary
9957 } else {
9958 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9959 return Ok(None);
9960 };
9961 summary
9962 };
9963 let zero = summary
9964 .entries
9965 .iter()
9966 .find(|entry| entry.value == FrequencyValue::Integer(0))
9967 .map(|entry| entry.count)
9968 .or_else(|| (summary.omitted_max == 0).then_some(0));
9969 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9970}
9971
9972fn quick_integer_fold(
9975 file: &File,
9976 mut cur: Cursor<'_>,
9977 entry: &Entry,
9978 size: u64,
9979 wanted: usize,
9980 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9981) -> Result<()> {
9982 let name = &entry.name;
9983 let fields = &entry.fields;
9984 let rows = entry.rows;
9985 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9986 return Err(invalid("table directory differs from the catalog"));
9987 }
9988 let width = cur.u16()? as usize;
9989 if width != fields.len() {
9990 return Err(invalid("table directory width differs from the catalog"));
9991 }
9992 for field in fields {
9993 let stored =
9994 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9995 if &stored != field {
9996 return Err(invalid("table directory schema differs from the catalog"));
9997 }
9998 }
9999 let mut dictionaries = Vec::with_capacity(width);
10000 for field in fields {
10001 dictionaries.push(match cur.u8()? {
10002 0 => false,
10003 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
10004 cur.skip(20)?;
10005 true
10006 }
10007 _ => return Err(invalid("dictionary page tag differs")),
10008 });
10009 }
10010 for _ in 0..width {
10011 match cur.u8()? {
10012 0 => {}
10013 1 => cur.skip(8)?,
10014 _ => return Err(invalid("distinct count tag differs")),
10015 }
10016 }
10017 if cur.u64()? != rows as u64 {
10018 return Err(invalid("table row count differs from the catalog"));
10019 }
10020 let stripes = cur.u32()? as usize;
10021 let mut total = 0_usize;
10022 let mut bytes = Vec::new();
10023 for _ in 0..stripes {
10024 let parts = cur.u32()? as usize;
10025 if parts == 0 || parts > STRIPE_PARTS {
10026 return Err(invalid("stripe part count is outside its bound"));
10027 }
10028 let mut part_rows = Vec::with_capacity(parts);
10029 for _ in 0..parts {
10030 let count = cur.u32()? as usize;
10031 if count == 0 {
10032 return Err(invalid("empty part"));
10033 }
10034 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
10035 part_rows.push(count);
10036 }
10037 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10038 let section = index_section(parts)?;
10039 let index_length =
10040 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
10041 if index.offset < HEADER
10042 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
10043 || index.length as usize != index_length
10044 {
10045 return Err(invalid("index page range is outside the file"));
10046 }
10047 cur.skip(wanted * 12)?;
10048 let page = Span { offset: cur.u64()?, length: cur.u32()? };
10049 if page.offset < HEADER
10050 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
10051 || page.length as usize > MAX_PAGE
10052 {
10053 return Err(invalid("column page range is outside the file"));
10054 }
10055 cur.skip((width - wanted - 1) * 12)?;
10056 for (field, held) in fields.iter().zip(&dictionaries) {
10057 if coded_type(&field.ty) && *held {
10058 cur.skip(20)?;
10059 }
10060 }
10061 for _ in 0..width * 2 {
10062 match cur.u8()? {
10063 0 => {}
10064 1 => cur.skip(20)?,
10065 _ => return Err(invalid("stripe page tag differs")),
10066 }
10067 }
10068 for _ in 0..width {
10069 cur.skip_bound()?;
10070 cur.skip_bound()?;
10071 cur.skip(5)?;
10072 match cur.u8()? {
10073 0 => {}
10074 1 => cur.skip(16)?,
10075 _ => return Err(invalid("a stripe sum has an unknown tag")),
10076 }
10077 }
10078 let spans = read_index_span(file, index, page, parts, wanted)?;
10079 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
10080 bytes.resize(span.length, 0);
10081 let at = page
10082 .offset
10083 .checked_add(span.start as u64)
10084 .ok_or_else(|| invalid("part range overflow"))?;
10085 read_at(file, at, &mut bytes)?;
10086 if checksum(&bytes) != span.hash {
10087 return Err(invalid("integer part checksum differs"));
10088 }
10089 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
10090 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
10091 check_integer_tally_value(value, &fields[wanted].ty)?;
10092 emit(value, count)
10093 })?;
10094 if decoded_rows != expected_rows {
10095 return Err(invalid("encoded integer part holds the wrong number of rows"));
10096 }
10097 } else {
10098 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
10099 if let Some(packed) = column.packed_parts() {
10100 let validity = column.validity();
10101 let all_valid = column.none_null();
10102 let base = packed.base();
10103 let mut codes = [0_u64; 64];
10104 for from in (0..expected_rows).step_by(codes.len()) {
10105 let count = (expected_rows - from).min(codes.len());
10106 packed.unpack(from, &mut codes[..count]);
10107 for (offset, &code) in codes[..count].iter().enumerate() {
10108 if all_valid || validity.is_valid(from + offset) {
10109 emit((base + i128::from(code)) as i64, 1)?;
10111 }
10112 }
10113 }
10114 continue;
10115 }
10116 let column = column.into_flat()?;
10117 let validity = column.validity();
10118 macro_rules! count_decoded {
10119 ($values:expr) => {
10120 for (row, &value) in $values.as_slice().iter().enumerate() {
10121 if validity.is_valid(row) {
10122 emit(i64::from(value), 1)?;
10123 }
10124 }
10125 };
10126 }
10127 match column.data() {
10128 Some(Data::Int8(values)) => count_decoded!(values),
10129 Some(Data::Int16(values)) => count_decoded!(values),
10130 Some(Data::Int32(values)) => count_decoded!(values),
10131 Some(Data::Int64(values)) => count_decoded!(values),
10132 _ => return Err(invalid("decoded integer part has the wrong type")),
10133 }
10134 }
10135 }
10136 }
10137 if total != rows {
10138 return Err(invalid("table row count differs from stripes"));
10139 }
10140 Ok(())
10141}
10142
10143fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
10144 let fits = match ty {
10145 LogicalType::TinyInt => i8::try_from(value).is_ok(),
10146 LogicalType::SmallInt => i16::try_from(value).is_ok(),
10147 LogicalType::Integer => i32::try_from(value).is_ok(),
10148 LogicalType::BigInt => true,
10149 _ => false,
10150 };
10151 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
10152}
10153
10154fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
10155 read_directory(Cursor::new(bytes), size, None)
10156}
10157
10158fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
10163 if cur.take(8)? != DIRECTORY {
10164 return Err(invalid("directory magic differs"));
10165 }
10166 let name = cur.text()?;
10167 let width = cur.u16()? as usize;
10168 let mut fields = Vec::with_capacity(width);
10169 for _ in 0..width {
10170 let name = cur.text()?;
10171 let ty = read_type(&mut cur)?;
10172 let not_null = match cur.u8()? {
10173 0 => false,
10174 1 => true,
10175 _ => return Err(invalid("nullability flag differs")),
10176 };
10177 fields.push(Field { name, ty, not_null });
10178 }
10179 let mut dictionaries = Vec::with_capacity(width);
10180 for field in &fields {
10181 dictionaries.push(match cur.u8()? {
10182 0 => None,
10183 tag if tag == dictionary_tag(&field.ty) => {
10184 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10185 let end = page
10186 .offset
10187 .checked_add(u64::from(page.length))
10188 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
10189 if page.offset < HEADER || end > size {
10194 return Err(invalid("dictionary page range is outside the file"));
10195 }
10196 Some(page)
10197 }
10198 _ => return Err(invalid("dictionary page tag differs")),
10199 });
10200 }
10201 let mut distincts = Vec::with_capacity(width);
10202 for _ in 0..width {
10203 distincts.push(match cur.u8()? {
10204 0 => None,
10205 1 => Some(cur.u64()?),
10206 _ => return Err(invalid("distinct count tag differs")),
10207 });
10208 }
10209 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10210 let count = cur.u32()? as usize;
10211 let mut stripes = Vec::with_capacity(count);
10212 let mut total = 0_usize;
10213 for _ in 0..count {
10214 let count = cur.u32()? as usize;
10215 if count == 0 || count > STRIPE_PARTS {
10216 return Err(invalid("stripe part count is outside its bound"));
10217 }
10218 let mut parts = Vec::with_capacity(count);
10219 let mut stripe_rows = 0_usize;
10220 for _ in 0..count {
10221 let rows = cur.u32()?;
10222 if rows == 0 {
10223 return Err(invalid("empty part"));
10224 }
10225 parts.push(rows);
10226 stripe_rows = stripe_rows
10227 .checked_add(rows as usize)
10228 .ok_or_else(|| invalid("stripe row count overflow"))?;
10229 }
10230 total =
10231 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10232 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10233 let section = index_section(count)?;
10234 let wanted = section
10235 .checked_mul(width)
10236 .and_then(|bytes| u32::try_from(bytes).ok())
10237 .ok_or_else(|| invalid("index page length overflow"))?;
10238 let end = index
10239 .offset
10240 .checked_add(u64::from(index.length))
10241 .ok_or_else(|| invalid("index page offset overflow"))?;
10242 if index.offset < HEADER || end > size || index.length != wanted {
10243 return Err(invalid("index page range is outside the file"));
10244 }
10245 let mut pages = Vec::with_capacity(width);
10246 for _ in 0..width {
10247 let offset = cur.u64()?;
10248 let length = cur.u32()?;
10249 let end = offset
10250 .checked_add(u64::from(length))
10251 .ok_or_else(|| invalid("page offset overflow"))?;
10252 if offset < HEADER || end > size || length as usize > MAX_PAGE {
10253 return Err(invalid("page range is outside the file"));
10254 }
10255 pages.push(Span { offset, length });
10256 }
10257 let mut memberships = vec![None; width];
10258 for (column, field) in fields.iter().enumerate() {
10259 if !coded_type(&field.ty) || dictionaries[column].is_none() {
10260 continue;
10261 }
10262 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10263 let end = page
10264 .offset
10265 .checked_add(u64::from(page.length))
10266 .ok_or_else(|| invalid("membership page offset overflow"))?;
10267 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10268 return Err(invalid("membership page range is outside the file"));
10269 }
10270 if page.length != 0 {
10273 memberships[column] = Some(page);
10274 }
10275 }
10276 let mut sieves = vec![None; width];
10277 for sieve in sieves.iter_mut().take(width) {
10278 match cur.u8()? {
10279 0 => continue,
10280 1 => {}
10281 _ => return Err(invalid("a sieve page has an unknown tag")),
10282 }
10283 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10284 let end = page
10285 .offset
10286 .checked_add(u64::from(page.length))
10287 .ok_or_else(|| invalid("sieve page offset overflow"))?;
10288 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10289 return Err(invalid("sieve page range is outside the file"));
10290 }
10291 *sieve = Some(page);
10292 }
10293 let mut part_ranges = vec![None; width];
10294 for held in part_ranges.iter_mut().take(width) {
10295 match cur.u8()? {
10296 0 => continue,
10297 1 => {}
10298 _ => return Err(invalid("a part range page has an unknown tag")),
10299 }
10300 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10301 let end = page
10302 .offset
10303 .checked_add(u64::from(page.length))
10304 .ok_or_else(|| invalid("part range page offset overflow"))?;
10305 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10306 return Err(invalid("part range page range is outside the file"));
10307 }
10308 *held = Some(page);
10309 }
10310 let mut ranges = Vec::with_capacity(width);
10311 for column in 0..width {
10312 let low = cur.bound()?;
10313 let high = cur.bound()?;
10314 let nulls = cur.u32()? as usize;
10315 if nulls > stripe_rows {
10316 return Err(invalid("null count exceeds stripe rows"));
10317 }
10318 let exact = cur.u8()? != 0;
10319 let sum = match cur.u8()? {
10320 0 => None,
10321 1 => Some(i128::from_le_bytes(
10322 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10323 )),
10324 _ => return Err(invalid("a stripe sum has an unknown tag")),
10325 };
10326 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10332 let low = low.map(|bound| scaled_as(bound, ty));
10333 let high = high.map(|bound| scaled_as(bound, ty));
10334 ranges.push(Range { low, high, nulls, exact, sum });
10335 }
10336 stripes.push(Stripe {
10337 rows: stripe_rows,
10338 parts,
10339 index,
10340 pages,
10341 memberships: Pages::from_slots(memberships)?,
10342 sieves: Pages::from_slots(sieves)?,
10343 part_ranges: Pages::from_slots(part_ranges)?,
10344 zone: Zone::from_ranges(ranges),
10345 });
10346 }
10347 if total != rows {
10348 return Err(invalid("table row count differs from stripes"));
10349 }
10350 let mut entry_counts = vec![0; width];
10353 let frequencies = if cur.done() {
10354 vec![None; width]
10355 } else {
10356 let frequency_magic = cur.take(8)?;
10357 let spanned = frequency_magic == FREQUENCIES_SPANS;
10358 let frequency_values = frequency_magic == FREQUENCIES || spanned;
10359 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10360 return Err(invalid("directory extension magic differs"));
10361 }
10362 if cur.u16()? as usize != width {
10363 return Err(invalid("frequency column count differs"));
10364 }
10365 let mut frequencies = Vec::with_capacity(width);
10366 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10367 if spanned {
10368 let Some((length, entries)) = summary_span(&mut cur)? else {
10369 frequencies.push(None);
10370 continue;
10371 };
10372 *entry_count = entries;
10373 let start = cur.at;
10374 if let Some(offset) = stored_at {
10375 cur.skip(length)?;
10376 frequencies.push(Some(Frequencies::Stored {
10377 span: Span {
10378 offset: offset
10379 .checked_add(start as u64)
10380 .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10381 length: u32::try_from(length)
10382 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10383 },
10384 values: true,
10385 entries,
10386 }));
10387 } else {
10388 let summary = decode_summary(&mut cur, field, rows, true)?
10389 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10390 if cur.at - start != length || summary.entries.len() != entries {
10391 return Err(invalid("a stored synopsis differs from its directory span"));
10392 }
10393 frequencies.push(Some(Frequencies::Held(summary)));
10394 }
10395 continue;
10396 }
10397 let start = cur.at;
10398 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10399 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10400 frequencies.push(match (summary, stored_at) {
10401 (None, _) => None,
10402 (Some(summary), None) => Some(Frequencies::Held(summary)),
10403 (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10404 span: Span {
10405 offset: offset + start as u64,
10406 length: u32::try_from(cur.at - start)
10407 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10408 },
10409 values: frequency_values,
10410 entries: summary.entries.len(),
10411 }),
10412 });
10413 }
10414 frequencies
10415 };
10416 let mut clustering = None;
10426 let mut sections = Vec::new();
10427 let mut pair_frequencies = Vec::new();
10428 let mut seen_pair_frequencies = false;
10429 let mut ordinal_bounds = Vec::new();
10430 let mut seen_ordinal_bounds = false;
10431 let mut frequency_texts = vec![Vec::new(); width];
10432 let mut seen_frequency_texts = false;
10433 let mut host_groups = None;
10434 let mut demoted = Vec::new();
10435 let mut seen_sections = false;
10436 let mut dictionary_payloads = Vec::new();
10437 let mut seen_payloads = false;
10438 let mut constraints = Constraints::default();
10439 let mut generation = 0;
10442 while !cur.done() {
10443 let mut tag = [0u8; 8];
10444 tag.copy_from_slice(cur.take(8)?);
10445 if &tag == PAIR_FREQUENCIES {
10446 if seen_pair_frequencies {
10447 return Err(invalid("directory names two pair frequency blocks"));
10448 }
10449 seen_pair_frequencies = true;
10450 let count = cur.u16()? as usize;
10451 if count > MAX_PAIR_FREQUENCIES {
10452 return Err(invalid("pair frequency count exceeds its bound"));
10453 }
10454 pair_frequencies = Vec::with_capacity(count);
10455 for _ in 0..count {
10456 let first = cur.u16()?;
10457 let second = cur.u16()?;
10458 let first_at = first as usize;
10459 let second_at = second as usize;
10460 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10461 return Err(invalid("pair frequency first column has no synopsis"));
10462 }
10463 let first_entries = entry_counts[first_at];
10464 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10465 || dictionaries.get(second_at).copied().flatten().is_none()
10466 {
10467 return Err(invalid("pair frequency second column has no stable dictionary"));
10468 }
10469 if pair_frequencies
10470 .iter()
10471 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10472 {
10473 return Err(invalid("directory repeats a pair frequency summary"));
10474 }
10475 let omitted_max = cur.u64()?;
10476 if omitted_max > rows as u64 {
10477 return Err(invalid("pair frequency omitted count exceeds the table"));
10478 }
10479 let entries_count = cur.u16()? as usize;
10480 if entries_count > FREQUENCY_ENTRIES {
10481 return Err(invalid("pair frequency entry count exceeds its bound"));
10482 }
10483 let mut entries = Vec::with_capacity(entries_count);
10484 for _ in 0..entries_count {
10485 let first_entry = cur.u16()?;
10486 if first_entry as usize >= first_entries {
10487 return Err(invalid("pair frequency anchor is outside its synopsis"));
10488 }
10489 let second = match cur.u8()? {
10490 0 => None,
10491 1 => Some(cur.u32()?),
10492 _ => return Err(invalid("pair frequency string tag differs")),
10493 };
10494 let count = cur.u64()?;
10495 if count == 0 || count > rows as u64 {
10496 return Err(invalid("pair frequency count is outside the table"));
10497 }
10498 entries.push(PairFrequencyEntry { first_entry, second, count });
10499 }
10500 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10501 return Err(invalid("pair frequency entries are not descending"));
10502 }
10503 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10504 }
10505 } else if &tag == ORDINAL_BOUNDS {
10506 if seen_ordinal_bounds {
10507 return Err(invalid("directory names two ordinal bound blocks"));
10508 }
10509 seen_ordinal_bounds = true;
10510 ordinal_bounds = vec![0; width];
10511 let count = cur.u16()? as usize;
10512 if count > width {
10513 return Err(invalid("ordinal bound count exceeds the columns"));
10514 }
10515 for _ in 0..count {
10516 let column = cur.u16()? as usize;
10517 let bound = cur.u64()?;
10518 if column >= width || frequencies.get(column).and_then(Option::as_ref).is_none() {
10519 return Err(invalid("ordinal bound names a column with no synopsis"));
10520 }
10521 if bound == 0 || bound > rows as u64 || ordinal_bounds[column] != 0 {
10522 return Err(invalid("ordinal bound is outside the table or repeated"));
10523 }
10524 ordinal_bounds[column] = bound;
10525 }
10526 } else if &tag == FREQUENCY_TEXTS {
10527 if seen_frequency_texts {
10528 return Err(invalid("directory names two frequency text blocks"));
10529 }
10530 seen_frequency_texts = true;
10531 let columns = cur.u16()? as usize;
10532 if columns > width {
10533 return Err(invalid("frequency text column count exceeds the schema"));
10534 }
10535 for _ in 0..columns {
10536 let column = cur.u16()? as usize;
10537 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10538 return Err(invalid("frequency text column is repeated or out of range"));
10539 }
10540 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10541 || dictionaries.get(column).copied().flatten().is_none()
10542 || frequencies.get(column).and_then(Option::as_ref).is_none()
10543 {
10544 return Err(invalid("frequency texts belong to a non-string synopsis"));
10545 }
10546 let count = cur.u16()? as usize;
10547 if count == 0 || count != entry_counts[column] {
10548 return Err(invalid("frequency text count differs from its synopsis"));
10549 }
10550 let mut texts = Vec::with_capacity(count);
10551 for _ in 0..count {
10552 texts.push(match cur.u8()? {
10553 0 => None,
10554 1 => {
10555 let length = cur.u32()? as usize;
10556 let bytes = cur.take(length)?.to_vec();
10557 if fields[column].ty == LogicalType::Varchar {
10558 std::str::from_utf8(&bytes)
10559 .map_err(|_| invalid("frequency text is not UTF-8"))?;
10560 }
10561 Some(bytes)
10562 }
10563 _ => return Err(invalid("frequency text tag differs")),
10564 });
10565 }
10566 frequency_texts[column] = texts;
10567 }
10568 } else if &tag == HOST_GROUPS {
10569 if host_groups.is_some() {
10570 return Err(invalid("directory names two host group blocks"));
10571 }
10572 let column = cur.u16()? as usize;
10573 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10574 || dictionaries.get(column).copied().flatten().is_none()
10575 {
10576 return Err(invalid("host groups belong to a non-string dictionary"));
10577 }
10578 let omitted_max = cur.u64()?;
10579 if omitted_max > rows as u64 {
10580 return Err(invalid("host group bound exceeds the table"));
10581 }
10582 let count = cur.u16()? as usize;
10583 if count > host::CAPACITY {
10584 return Err(invalid("host group count exceeds its bound"));
10585 }
10586 let mut entries = Vec::with_capacity(count);
10587 let mut bytes = 0_usize;
10588 for _ in 0..count {
10589 let host_len = cur.u32()? as usize;
10590 bytes =
10591 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10592 if bytes > host::BYTE_BUDGET {
10593 return Err(invalid("host groups exceed their byte budget"));
10594 }
10595 let host = std::str::from_utf8(cur.take(host_len)?)
10596 .map_err(|_| invalid("host is not UTF-8"))?
10597 .to_owned();
10598 let count = cur.u64()?;
10599 if count == 0 || count > rows as u64 {
10600 return Err(invalid("host group count exceeds the table"));
10601 }
10602 let bytes_sum = i128::from_le_bytes(
10603 cur.take(16)?
10604 .try_into()
10605 .map_err(|_| invalid("host length sum is truncated"))?,
10606 );
10607 if bytes_sum < 0 {
10608 return Err(invalid("host length sum is negative"));
10609 }
10610 let minimum_len = cur.u32()? as usize;
10611 bytes =
10612 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10613 if bytes > host::BYTE_BUDGET {
10614 return Err(invalid("host groups exceed their byte budget"));
10615 }
10616 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10617 .map_err(|_| invalid("host minimum is not UTF-8"))?
10618 .to_owned();
10619 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10620 }
10621 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10622 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10623 {
10624 return Err(invalid("host groups are not in certified order"));
10625 }
10626 host_groups = Some(host::HostSummary { column, omitted_max, entries });
10627 } else if &tag == CLUSTERING {
10628 if clustering.is_some() {
10629 return Err(invalid("directory names two clustering declarations"));
10630 }
10631 let bucket = Width::from_tag(cur.u8()?)
10632 .ok_or_else(|| invalid("clustering width tag differs"))?;
10633 let count = cur.u16()? as usize;
10634 let mut columns = Vec::with_capacity(count.min(fields.len()));
10635 for _ in 0..count {
10636 columns.push(u32::from(cur.u16()?));
10637 }
10638 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10641 invalid("stored clustering declaration does not match the table it is on")
10642 })?);
10643 } else if &tag == DEMOTED {
10644 if !demoted.is_empty() {
10645 return Err(invalid("directory names two demoted column blocks"));
10646 }
10647 let count = cur.u16()? as usize;
10648 if count == 0 || count > width {
10649 return Err(invalid("demoted column count is outside the schema"));
10650 }
10651 demoted = vec![false; width];
10652 for _ in 0..count {
10653 let column = cur.u16()? as usize;
10654 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10655 return Err(invalid("a demoted column is repeated or has no dictionary"));
10656 }
10657 demoted[column] = true;
10658 }
10659 } else if &tag == SECTIONS {
10660 if seen_sections {
10661 return Err(invalid("directory names two section tables"));
10662 }
10663 seen_sections = true;
10664 generation = cur.u64()?;
10665 let count = cur.u16()? as usize;
10666 if count > MAX_SECTIONS {
10667 return Err(invalid("section count exceeds its bound"));
10668 }
10669 sections = Vec::with_capacity(count);
10670 for _ in 0..count {
10673 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10674 }
10675 for held in §ions {
10676 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10677 return Err(invalid("a section's extent table overflows the file"));
10678 };
10679 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10683 return Err(invalid("a section's extent table is outside the file"));
10684 }
10685 if held.extents == 0 && held.extent_bytes != 0 {
10686 return Err(invalid("a section with no extents names an extent table"));
10687 }
10688 }
10689 } else if &tag == DICTIONARY_PAYLOADS {
10690 if seen_payloads {
10691 return Err(invalid("directory names two dictionary payload blocks"));
10692 }
10693 seen_payloads = true;
10694 let count = cur.u16()? as usize;
10695 if count != fields.len() {
10696 return Err(invalid("dictionary payload block does not match the table's columns"));
10697 }
10698 dictionary_payloads = Vec::with_capacity(count);
10699 for _ in 0..count {
10700 let bytes = cur.u64()?;
10701 if bytes > size {
10702 return Err(invalid("a dictionary payload is larger than the file"));
10703 }
10704 dictionary_payloads.push(bytes);
10705 }
10706 } else if &tag == KEYS {
10707 if !constraints.is_empty() {
10708 return Err(invalid("directory names two key blocks"));
10709 }
10710 let fits = |columns: &[u16]| {
10711 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10712 };
10713 let count = cur.u16()? as usize;
10714 for _ in 0..count {
10715 let primary = cur.u8()? != 0;
10716 let columns = columns_of(&mut cur)?;
10717 if !fits(&columns) {
10718 return Err(invalid("a stored key names a column the table does not have"));
10719 }
10720 constraints.keys.push((columns, primary));
10721 }
10722 let count = cur.u16()? as usize;
10723 for _ in 0..count {
10724 let columns = columns_of(&mut cur)?;
10725 let referenced = columns_of(&mut cur)?;
10726 let len = cur.u32()? as usize;
10727 let table = std::str::from_utf8(cur.take(len)?)
10728 .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10729 .to_owned();
10730 if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10731 return Err(invalid("a stored foreign key does not match its table"));
10732 }
10733 constraints.foreign.push(StoredForeign { columns, table, referenced });
10734 }
10735 if constraints.is_empty() {
10736 return Err(invalid("a key block holds no key"));
10737 }
10738 } else {
10739 return Err(invalid("directory extension magic differs"));
10740 }
10741 }
10742 if !cur.done() {
10743 return Err(invalid("directory has trailing bytes"));
10744 }
10745 for stripe in &stripes {
10746 for (column, field) in fields.iter().enumerate() {
10747 if coded_type(&field.ty)
10748 && dictionaries[column].is_some()
10749 && stripe.memberships.get(column).is_none()
10750 && !demoted.get(column).copied().unwrap_or(false)
10751 {
10752 return Err(invalid("string page has no code membership index"));
10753 }
10754 }
10755 }
10756 Ok(Table {
10757 name,
10758 fields,
10759 stripes,
10760 rows,
10761 dictionaries,
10762 dictionary_payloads,
10763 demoted,
10764 distincts,
10765 frequencies,
10766 ordinal_bounds,
10767 pair_frequencies,
10768 frequency_texts,
10769 host_groups,
10770 clustering,
10771 generation,
10772 sections,
10773 constraints,
10774 })
10775}
10776
10777fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10779 put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10780 Ok(())
10781}
10782
10783fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10785 put_count(out, columns.len())?;
10786 for &column in columns {
10787 put_u16(out, column);
10788 }
10789 Ok(())
10790}
10791
10792fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10794 let count = cur.u16()? as usize;
10795 (0..count).map(|_| cur.u16()).collect()
10796}
10797
10798fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10800 bounds::put(out, bound)
10801}
10802
10803#[derive(Debug)]
10820struct Codes;
10821
10822impl chooser::Chooser for Codes {
10823 fn name(&self) -> &'static str {
10824 "codes"
10825 }
10826
10827 fn narrow_strings(
10828 &self,
10829 _values: &[&[u8]],
10830 offered: &[string::Kind],
10831 _depth: u8,
10832 ) -> Vec<string::Kind> {
10833 offered.to_vec()
10836 }
10837
10838 fn narrow_integers(
10839 &self,
10840 _values: &[i64],
10841 offered: &[integer::Kind],
10842 depth: u8,
10843 ) -> Vec<integer::Kind> {
10844 narrowed_to(Codes::keep(depth), offered)
10847 }
10848
10849 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10850 Codes::keep(depth).contains(&kind)
10851 }
10852}
10853
10854impl Codes {
10855 fn keep(depth: u8) -> &'static [integer::Kind] {
10856 if depth == 0 {
10857 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10858 } else {
10859 &[integer::Kind::Constant, integer::Kind::Packed]
10860 }
10861 }
10862}
10863
10864fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10872 let narrowed: Vec<integer::Kind> =
10873 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10874 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10875}
10876
10877#[derive(Debug)]
10889struct Fixed;
10890
10891impl chooser::Chooser for Fixed {
10892 fn name(&self) -> &'static str {
10893 "fixed"
10894 }
10895
10896 fn narrow_strings(
10897 &self,
10898 _values: &[&[u8]],
10899 offered: &[string::Kind],
10900 _depth: u8,
10901 ) -> Vec<string::Kind> {
10902 offered.to_vec()
10903 }
10904
10905 fn narrow_integers(
10906 &self,
10907 _values: &[i64],
10908 offered: &[integer::Kind],
10909 depth: u8,
10910 ) -> Vec<integer::Kind> {
10911 narrowed_to(Fixed::keep(depth), offered)
10912 }
10913
10914 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10915 Fixed::keep(depth).contains(&kind)
10916 }
10917}
10918
10919impl Fixed {
10920 fn keep(depth: u8) -> &'static [integer::Kind] {
10921 if depth == 0 {
10922 &[
10923 integer::Kind::Constant,
10924 integer::Kind::Packed,
10925 integer::Kind::Delta,
10926 integer::Kind::Rle,
10927 integer::Kind::Sparse,
10928 integer::Kind::Strided,
10929 ]
10930 } else {
10931 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10932 }
10933 }
10934}
10935
10936fn widened(data: &Data) -> Option<Vec<i64>> {
10943 match data {
10944 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10945 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10946 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10947 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10948 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10949 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10950 Data::Int64(values) => Some(values.to_vec()),
10951 _ => None,
10952 }
10953}
10954
10955fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10961 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10962 let values = integer::decode_as::<T>(bytes)
10963 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10964 if values.len() != rows {
10965 return Err(invalid("cascade page holds the wrong number of rows"));
10966 }
10967 Ok(values)
10968 }
10969 Ok(match ty {
10970 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10971 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10972 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10973 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10974 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10975 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10976 LogicalType::BigInt
10977 | LogicalType::Timestamp
10978 | LogicalType::Time
10979 | LogicalType::TimeTz
10980 | LogicalType::TimestampTz
10981 | LogicalType::TimestampS
10982 | LogicalType::TimestampMs
10983 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10984 LogicalType::Decimal { .. } => match ty.physical() {
10987 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10988 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10989 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10990 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10991 },
10992 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10993 })
10994}
10995
10996fn plain_width(ty: &LogicalType) -> Option<usize> {
10999 Some(match ty {
11000 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
11001 LogicalType::SmallInt | LogicalType::USmallInt => 2,
11002 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
11003 LogicalType::BigInt
11004 | LogicalType::Timestamp
11005 | LogicalType::Time
11006 | LogicalType::TimeTz
11007 | LogicalType::TimestampTz
11008 | LogicalType::TimestampS
11009 | LogicalType::TimestampMs
11010 | LogicalType::TimestampNs => 8,
11011 LogicalType::Decimal { .. } => match ty.physical() {
11012 PhysicalType::Int16 => 2,
11013 PhysicalType::Int32 => 4,
11014 PhysicalType::Int64 => 8,
11015 _ => return None,
11018 },
11019 _ => return None,
11020 })
11021}
11022
11023fn cascaded(
11029 flat: &Vector,
11030 ty: &LogicalType,
11031 packed: Option<&Packed<'_>>,
11032 settling: &mut Settling,
11033) -> Result<Option<Vec<u8>>> {
11034 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
11035 let Some(values) = widened(data) else { return Ok(None) };
11036 let plain = values.len().saturating_mul(width);
11037 let best = match packed {
11038 Some(packed) => plain.min(21 + size_of_val(packed.words())),
11040 None => plain,
11041 };
11042 let out = settling.encode(&values)?;
11043 Ok((out.len() < best).then_some(out))
11044}
11045
11046const SEARCH_EVERY: usize = 16;
11053
11054#[derive(Debug, Default)]
11060struct Settling {
11061 shape: Option<Shape>,
11064 since: usize,
11066 symbols: Option<Symbols>,
11068}
11069
11070#[derive(Debug)]
11073struct Symbols {
11074 shape: chooser::Settled,
11075 len: usize,
11078 payload: usize,
11079 since: usize,
11080}
11081
11082impl Settling {
11083 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
11091 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
11092 {
11093 let out = string::encode_fsst(values, &symbols.shape)?;
11094 let held = match &out {
11096 None => symbols.len == 0,
11097 Some(out) => {
11098 (out.len() as u128) * (symbols.payload as u128) * 4
11099 <= (symbols.len as u128) * (payload as u128) * 5
11100 }
11101 };
11102 if held {
11103 symbols.since += 1;
11104 return Ok(out);
11105 }
11106 }
11107 let shape = string::fsst_shape(values);
11108 let out = string::encode_fsst(values, &shape)?;
11109 let len = out.as_ref().map_or(0, Vec::len);
11110 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
11111 Ok(out)
11112 }
11113
11114 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
11121 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
11122 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
11123 let out = integer::encode_with(values, &replay)?;
11124 if !replay.held() {
11125 self.settle(&out, values.len(), replay.first_offered())?;
11126 return Ok(out);
11127 }
11128 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
11129 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
11130 self.since += 1;
11131 return Ok(out);
11132 }
11133 }
11134 let search = chooser::Replay::new(&[], &Fixed);
11136 let out = integer::encode_with(values, &search)?;
11137 self.settle(&out, values.len(), search.first_offered())?;
11138 Ok(out)
11139 }
11140
11141 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
11142 let kinds = integer::shape(out)?;
11143 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
11144 self.since = 0;
11145 Ok(())
11146 }
11147}
11148
11149#[derive(Debug)]
11151struct Shape {
11152 kinds: Vec<integer::Kind>,
11153 offered: Vec<integer::Kind>,
11154 len: usize,
11155 rows: usize,
11156}
11157
11158fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
11199 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
11200 let mut payload = 0_usize;
11201 for row in 0..flat.len() {
11202 let text = flat.bytes_at(row).unwrap_or(b"");
11205 payload = payload.saturating_add(text.len());
11206 values.push(text);
11207 }
11208 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
11210 let Some(out) = settling.text(&values, payload)? else {
11211 return Ok(None);
11212 };
11213 Ok((out.len() < plain).then_some(out))
11214}
11215
11216fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
11217 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
11218 let coded = integer::encode_with(&wide, &Codes)?;
11219 let plain = codes.len().saturating_mul(size_of::<u32>());
11220 Ok((coded.len() < plain).then_some(coded))
11221}
11222
11223fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
11226 let flag = match flat.validity() {
11227 Validity::AllValid => 0,
11228 Validity::AllInvalid => 1,
11229 Validity::Mask(_) => 2,
11230 };
11231 out.push(flag);
11232 if flag == 2 {
11233 for group in (0..flat.len()).step_by(8) {
11234 let mut bits = 0_u8;
11235 for bit in 0..8 {
11236 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11237 bits |= 1 << bit;
11238 }
11239 }
11240 out.push(bits);
11241 }
11242 }
11243}
11244
11245fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11252 let coded = encoded_codes(codes)?;
11253 let mut out = Vec::with_capacity(
11254 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11255 );
11256 out.push(if coded.is_some() { 4 } else { 3 });
11257 out.extend_from_slice(validity);
11258 match coded {
11259 Some(coded) => out.extend_from_slice(&coded),
11260 None => {
11261 for &code in codes {
11262 put_u32(&mut out, code);
11263 }
11264 }
11265 }
11266 Ok(out)
11267}
11268
11269fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11272 let ty = vector.logical_type();
11273 let flat = vector.flatten()?;
11275 let mut out = Vec::new();
11276 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11277 let compressed_text = if dictionary.is_none() && coded_type(ty) {
11278 text_compressed(&flat, settling)?
11279 } else {
11280 None
11281 };
11282 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11283 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11284 let cascade =
11288 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11289 out.push(if cascade.is_some() {
11290 5
11291 } else if dictionary.is_some() {
11292 1
11293 } else if compressed_text.is_some() {
11294 6
11295 } else if packed.is_some() {
11296 2
11297 } else {
11298 0
11299 });
11300 push_validity(&mut out, &flat);
11301 if let Some(cascade) = cascade {
11302 out.extend_from_slice(&cascade);
11303 return Ok(out);
11304 }
11305 if let Some(dictionary) = dictionary {
11306 out.extend_from_slice(&dictionary);
11307 return Ok(out);
11308 }
11309 if let Some(compressed_text) = compressed_text {
11310 out.extend_from_slice(&compressed_text);
11311 return Ok(out);
11312 }
11313 if let Some(packed) = packed {
11314 if packed.offset() != 0 {
11315 return Err(invalid("writer received a sliced packed vector"));
11316 }
11317 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11318 out.extend_from_slice(&packed.base().to_le_bytes());
11319 put_u32(
11320 &mut out,
11321 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11322 );
11323 for word in packed.words() {
11324 put_u64(&mut out, *word);
11325 }
11326 return Ok(out);
11327 }
11328 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11329 match (ty, data) {
11330 (LogicalType::TinyInt, Data::Int8(values)) => {
11331 for value in &**values {
11332 out.extend_from_slice(&value.to_le_bytes());
11333 }
11334 }
11335 (LogicalType::UTinyInt, Data::UInt8(values)) => {
11336 for value in &**values {
11337 out.extend_from_slice(&value.to_le_bytes());
11338 }
11339 }
11340 (LogicalType::SmallInt, Data::Int16(values)) => {
11341 for value in &**values {
11342 out.extend_from_slice(&value.to_le_bytes());
11343 }
11344 }
11345 (LogicalType::USmallInt, Data::UInt16(values)) => {
11346 for value in &**values {
11347 out.extend_from_slice(&value.to_le_bytes());
11348 }
11349 }
11350 (LogicalType::UInteger, Data::UInt32(values)) => {
11351 for value in &**values {
11352 out.extend_from_slice(&value.to_le_bytes());
11353 }
11354 }
11355 (LogicalType::UBigInt, Data::UInt64(values)) => {
11356 for value in &**values {
11357 out.extend_from_slice(&value.to_le_bytes());
11358 }
11359 }
11360 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11361 for value in &**values {
11362 out.extend_from_slice(&value.to_le_bytes());
11363 }
11364 }
11365 (
11366 LogicalType::BigInt
11367 | LogicalType::Timestamp
11368 | LogicalType::Time
11369 | LogicalType::TimeTz
11370 | LogicalType::TimestampTz
11371 | LogicalType::TimestampS
11372 | LogicalType::TimestampMs
11373 | LogicalType::TimestampNs,
11374 Data::Int64(values),
11375 ) => {
11376 for value in &**values {
11377 out.extend_from_slice(&value.to_le_bytes());
11378 }
11379 }
11380 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11383 for value in &**values {
11384 out.extend_from_slice(&value.to_le_bytes());
11385 }
11386 }
11387 (LogicalType::UHugeInt, Data::UInt128(values)) => {
11388 for value in &**values {
11389 out.extend_from_slice(&value.to_le_bytes());
11390 }
11391 }
11392 (LogicalType::Float, Data::Float32(values)) => {
11395 for value in &**values {
11396 out.extend_from_slice(&value.to_le_bytes());
11397 }
11398 }
11399 (LogicalType::Double, Data::Float64(values)) => {
11400 for value in &**values {
11401 out.extend_from_slice(&value.to_le_bytes());
11402 }
11403 }
11404 (LogicalType::Interval, Data::Interval(values)) => {
11408 for (months, days, micros) in &**values {
11409 out.extend_from_slice(&months.to_le_bytes());
11410 out.extend_from_slice(&days.to_le_bytes());
11411 out.extend_from_slice(µs.to_le_bytes());
11412 }
11413 }
11414 (LogicalType::Boolean, Data::Bool(values)) => {
11415 for value in &**values {
11416 out.push(u8::from(*value));
11417 }
11418 }
11419 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11422 for value in &**values {
11423 out.extend_from_slice(&value.to_le_bytes());
11424 }
11425 }
11426 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11427 for value in &**values {
11428 out.extend_from_slice(&value.to_le_bytes());
11429 }
11430 }
11431 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11432 for value in &**values {
11433 out.extend_from_slice(&value.to_le_bytes());
11434 }
11435 }
11436 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11437 for value in &**values {
11438 out.extend_from_slice(&value.to_le_bytes());
11439 }
11440 }
11441 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11446 let mut bytes = Vec::new();
11447 put_u32(&mut out, 0);
11448 for row in 0..vector.len() {
11449 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11450 bytes.extend_from_slice(value);
11451 put_u32(
11452 &mut out,
11453 u32::try_from(bytes.len())
11454 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11455 );
11456 }
11457 out.extend_from_slice(&bytes);
11458 }
11459 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11460 }
11461 Ok(out)
11462}
11463
11464fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11465 while value >= 0x80 {
11466 out.push((value as u8 & 0x7f) | 0x80);
11467 value >>= 7;
11468 }
11469 out.push(value as u8);
11470}
11471
11472fn unique_codes(codes: &[u32]) -> Vec<u32> {
11474 let mut unique = codes.to_vec();
11475 unique.sort_unstable();
11476 unique.dedup();
11477 unique
11478}
11479
11480fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11486 let mut lists = lists;
11487 while lists.len() > 1 {
11488 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11489 for pair in lists.chunks(2) {
11490 match pair {
11491 [left, right] => next.push(merged_pair(left, right)),
11492 [only] => next.push(only.clone()),
11493 _ => {}
11494 }
11495 }
11496 lists = next;
11497 }
11498 lists.pop().unwrap_or_default()
11499}
11500
11501fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11502 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11503 let mut at = 0;
11504 let mut to = 0;
11505 while at < left.len() && to < right.len() {
11506 match left[at].cmp(&right[to]) {
11507 Ordering::Less => {
11508 out.push(left[at]);
11509 at += 1;
11510 }
11511 Ordering::Greater => {
11512 out.push(right[to]);
11513 to += 1;
11514 }
11515 Ordering::Equal => {
11516 out.push(left[at]);
11517 at += 1;
11518 to += 1;
11519 }
11520 }
11521 }
11522 out.extend_from_slice(&left[at..]);
11523 out.extend_from_slice(&right[to..]);
11524 out
11525}
11526
11527fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11532 let mut merged = Range::default();
11533 let mut first = true;
11534 for range in ranges {
11535 merged.nulls = merged.nulls.saturating_add(range.nulls);
11536 merged.sum = match (merged.sum.take(), range.sum) {
11540 (Some(held), Some(next)) if !first => held.checked_add(next),
11541 (_, next) if first => next,
11542 _ => None,
11543 };
11544 merged.exact = if first { range.exact } else { merged.exact && range.exact };
11545 if first {
11546 merged.low = range.low;
11547 merged.high = range.high;
11548 first = false;
11549 continue;
11550 }
11551 merged.low = match (merged.low.take(), range.low) {
11552 (Some(held), Some(next)) => Some(held.smaller(next)),
11553 _ => None,
11554 };
11555 merged.high = match (merged.high.take(), range.high) {
11556 (Some(held), Some(next)) => Some(held.larger(next)),
11557 _ => None,
11558 };
11559 }
11560 merged
11561}
11562
11563fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11576 match bound {
11577 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11578 value.truncate(PART_BOUND_BYTES);
11579 if !high {
11580 return Some(Bound::Bytes(value));
11581 }
11582 while let Some(last) = value.pop() {
11583 if last < u8::MAX {
11584 value.push(last + 1);
11585 return Some(Bound::Bytes(value));
11586 }
11587 }
11588 None
11589 }
11590 other => other,
11591 }
11592}
11593
11594fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11602 let mut out = Vec::new();
11603 put_u32(
11604 &mut out,
11605 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11606 );
11607 for range in ranges {
11608 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11609 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11610 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11611 }
11612 Ok(out)
11613}
11614
11615fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11617 let mut cur = Cursor::new(bytes);
11618 let parts = cur.u32()? as usize;
11619 let mut out = Vec::new();
11620 for _ in 0..parts {
11621 let low = cur.bound()?;
11622 let high = cur.bound()?;
11623 let nulls = cur.u32()? as usize;
11624 out.push(Range { low, high, nulls, exact: false, sum: None });
11625 }
11626 Ok(out)
11627}
11628
11629fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11630 let held: Vec<&Option<Sieve>> = sieves.collect();
11631 let mut out = Vec::new();
11632 put_u32(
11633 &mut out,
11634 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11635 );
11636 for sieve in &held {
11637 let length = sieve.as_ref().map_or(0, Sieve::len);
11638 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11639 }
11640 for sieve in held.into_iter().flatten() {
11642 out.extend_from_slice(&sieve.to_bytes());
11643 }
11644 Ok(out)
11645}
11646
11647fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11653 let parts = u32::from_le_bytes(
11654 bytes
11655 .get(..4)
11656 .ok_or_else(|| invalid("sieve page is truncated"))?
11657 .try_into()
11658 .map_err(|_| invalid("sieve page is truncated"))?,
11659 ) as usize;
11660 let mut lengths = Vec::with_capacity(parts);
11661 for part in 0..parts {
11662 let at = 4 + part * 4;
11663 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11664 lengths.push(u32::from_le_bytes(
11665 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11666 ) as usize);
11667 }
11668 let mut at = 4 + parts * 4;
11669 let mut out = Vec::with_capacity(parts);
11670 for length in lengths {
11671 if length == 0 {
11672 out.push(None);
11673 continue;
11674 }
11675 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11676 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11677 out.push(Sieve::from_bytes(field));
11678 at = end;
11679 }
11680 if at != bytes.len() {
11681 return Err(invalid("sieve page has trailing bytes"));
11682 }
11683 Ok(out)
11684}
11685
11686fn encode_membership(unique: &[u32]) -> Vec<u8> {
11692 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11693 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11694 let mut previous = 0;
11695 for (at, &code) in unique.iter().enumerate() {
11696 put_varint(&mut out, if at == 0 { code } else { code - previous });
11697 previous = code;
11698 }
11699 out
11700}
11701
11702fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11703 let mut value = 0_u32;
11704 for shift in (0..35).step_by(7) {
11705 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11706 *at += 1;
11707 let part = u32::from(byte & 0x7f);
11708 if shift == 28 && part > 0x0f {
11709 return Err(invalid("membership varint overflow"));
11710 }
11711 value = value
11712 .checked_add(
11713 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11714 )
11715 .ok_or_else(|| invalid("membership varint overflow"))?;
11716 if byte & 0x80 == 0 {
11717 return Ok(value);
11718 }
11719 }
11720 Err(invalid("membership varint is too long"))
11721}
11722
11723fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11724 let mut at = 0;
11725 let count = take_varint(bytes, &mut at)? as usize;
11726 let mut codes = Vec::with_capacity(count);
11727 let mut previous = 0_u32;
11728 for index in 0..count {
11729 let delta = take_varint(bytes, &mut at)?;
11730 let code = if index == 0 {
11731 delta
11732 } else {
11733 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11734 };
11735 if index > 0 && code <= previous {
11736 return Err(invalid("membership codes are not increasing"));
11737 }
11738 codes.push(code);
11739 previous = code;
11740 }
11741 if at != bytes.len() {
11742 return Err(invalid("membership page has trailing bytes"));
11743 }
11744 Ok(codes)
11745}
11746
11747fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11755 let mut by_text: HashMap<&[u8], u32, Spread> =
11756 HashMap::with_capacity_and_hasher(vector.len(), Spread);
11757 let mut values = Vec::new();
11758 let mut codes = Vec::with_capacity(vector.len());
11759 let mut plain_bytes = 0_usize;
11760 for row in 0..vector.len() {
11761 let text = vector.bytes_at(row).unwrap_or(b"");
11762 plain_bytes = plain_bytes.saturating_add(text.len());
11763 let code = match by_text.get(text) {
11764 Some(&code) => code,
11765 None => {
11766 let code = u32::try_from(values.len())
11767 .map_err(|_| invalid("too many dictionary values"))?;
11768 by_text.insert(text, code);
11769 values.push(text);
11770 code
11771 }
11772 };
11773 codes.push(code);
11774 }
11775 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11776 let encoded = 8_usize
11777 .saturating_add((values.len() + 1).saturating_mul(4))
11778 .saturating_add(dictionary_bytes)
11779 .saturating_add(codes.len().saturating_mul(4));
11780 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11781 if encoded >= plain {
11782 return Ok(None);
11783 }
11784 let mut out = Vec::with_capacity(encoded);
11785 put_u32(
11786 &mut out,
11787 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11788 );
11789 put_u32(
11790 &mut out,
11791 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11792 );
11793 let mut offset = 0_u32;
11794 put_u32(&mut out, offset);
11795 for value in &values {
11796 offset = offset
11797 .checked_add(
11798 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11799 )
11800 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11801 put_u32(&mut out, offset);
11802 }
11803 for value in values {
11804 out.extend_from_slice(value);
11805 }
11806 for code in codes {
11807 put_u32(&mut out, code);
11808 }
11809 Ok(Some(out))
11810}
11811
11812struct Room<'a, T> {
11814 state: &'a Mutex<(T, usize)>,
11815 finished: &'a Condvar,
11816 bytes: usize,
11817}
11818
11819impl<T> Drop for Room<'_, T> {
11820 fn drop(&mut self) {
11821 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11822 held.1 -= self.bytes;
11823 drop(held);
11824 self.finished.notify_all();
11825 }
11826}
11827
11828enum Closing<'a> {
11830 Numeric {
11833 column: usize,
11834 counted: bool,
11835 dense: Option<(u64, usize)>,
11836 },
11837 Dictionary {
11838 index: usize,
11839 dictionary: &'a GlobalDictionary,
11840 },
11841}
11842
11843enum Closed {
11845 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11846 Dictionary(usize, ClosedDictionary),
11847}
11848
11849struct ClosedDictionary {
11851 distinct: Option<u64>,
11853 frequencies: Option<FrequencySummary>,
11854 texts: Vec<Option<Vec<u8>>>,
11855 hosts: Option<host::HostSummary>,
11856 encoded: EncodedDictionary,
11857 payload: u64,
11859}
11860
11861struct EncodedDictionary {
11862 index: Vec<u8>,
11863 ranks: Vec<u8>,
11864 grams: Vec<u8>,
11865}
11866
11867fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11908 let mut work = vec![(0, codes.len(), 0)];
11909 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11910 while let Some((from, to, depth)) = work.pop() {
11911 let part = &mut codes[from..to];
11912 keyed.clear();
11913 keyed.extend(part.iter().map(|&code| {
11914 let value = values(code);
11915 let rest = value.get(depth..).unwrap_or_default();
11916 (head(rest), rest.len().min(8) as u8, code)
11917 }));
11918 keyed.sort_unstable();
11919 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11920 *slot = entry.2;
11921 }
11922 let mut start = 0;
11923 while start < keyed.len() {
11924 let (key, taken, _) = keyed[start];
11925 let mut end = start + 1;
11926 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11927 end += 1;
11928 }
11929 if taken == 8 && end - start > 1 {
11930 work.push((from + start, from + end, depth + 8));
11931 }
11932 start = end;
11933 }
11934 }
11935}
11936
11937const PARALLEL_SORT_MIN: usize = 1 << 16;
11939
11940const BUCKETS_PER_WORKER: usize = 4;
11943
11944const SAMPLES_PER_BUCKET: usize = 32;
11946
11947fn sort_by_value_across<'a>(
11965 codes: &mut [u32],
11966 values: impl Fn(u32) -> &'a [u8] + Sync,
11967 workers: usize,
11968) {
11969 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11970 sort_by_value(codes, values);
11971 return;
11972 }
11973 let buckets = workers * BUCKETS_PER_WORKER;
11974 let wanted = buckets * SAMPLES_PER_BUCKET;
11975 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11976 sort_by_value(&mut sample, &values);
11977 let splitters =
11978 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11979 let values = &values;
11980 let splitters = &splitters;
11981 let per = codes.len().div_ceil(workers);
11982 let places = std::thread::scope(|scope| {
11984 codes
11985 .chunks(per)
11986 .map(|run| {
11987 scope.spawn(move || {
11988 run.iter()
11989 .map(|&code| {
11990 let value = values(code);
11991 splitters.partition_point(|splitter| *splitter <= value) as u32
11992 })
11993 .collect::<Vec<_>>()
11994 })
11995 })
11996 .collect::<Vec<_>>()
11997 .into_iter()
11998 .flat_map(|handle| {
11999 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
12000 })
12001 .collect::<Vec<_>>()
12002 });
12003 let mut starts = vec![0_usize; buckets + 1];
12004 for &place in &places {
12005 starts[place as usize + 1] += 1;
12006 }
12007 for bucket in 0..buckets {
12008 starts[bucket + 1] += starts[bucket];
12009 }
12010 let mut laid = vec![0_u32; codes.len()];
12011 let mut next = starts.clone();
12012 for (&code, &place) in codes.iter().zip(&places) {
12013 laid[next[place as usize]] = code;
12014 next[place as usize] += 1;
12015 }
12016 drop(places);
12017 let mut runs = Vec::with_capacity(buckets);
12018 let mut rest = laid.as_mut_slice();
12019 for bucket in 0..buckets {
12020 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
12021 runs.push(run);
12022 rest = after;
12023 }
12024 runs.sort_by_key(|run| run.len());
12026 let queue = Mutex::new(runs);
12027 std::thread::scope(|scope| {
12028 for _ in 0..workers {
12029 scope.spawn(|| {
12030 loop {
12031 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
12032 let Some(run) = taken else { break };
12033 sort_by_value(run, values);
12034 }
12035 });
12036 }
12037 });
12038 codes.copy_from_slice(&laid);
12039}
12040
12041fn head(bytes: &[u8]) -> u64 {
12049 if let Some(word) = bytes.first_chunk::<8>() {
12050 return u64::from_be_bytes(*word);
12051 }
12052 let len = bytes.len();
12053 if len >= 4 {
12054 let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
12055 let back = &bytes[len - 4..];
12056 let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
12057 return (front << 32) | (back << (8 * (8 - len)));
12058 }
12059 bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
12060}
12061
12062fn encode_global_dictionary(
12073 dictionary: &GlobalDictionary,
12074 order: &[(u64, u32)],
12075 places: &[Placed],
12076 scattered: bool,
12077) -> Result<EncodedDictionary> {
12078 let values = dictionary.values();
12079 if order.len() != values {
12080 return Err(invalid("global dictionary order does not cover its values"));
12081 }
12082 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
12083 if places.len() != blocks {
12084 return Err(invalid("global dictionary payload is not the blocks it says it is"));
12085 }
12086 if dictionary.grams.len() != blocks {
12087 return Err(invalid("global dictionary signatures do not cover its blocks"));
12088 }
12089 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
12090 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
12091 let offset_bits = offset_width(&dictionary.ends);
12092 let payload_words = if scattered { 3 } else { 2 };
12093 let index_len = DICTIONARY_HEADER
12094 .checked_add(offset_bytes(values, offset_bits))
12095 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
12096 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12097 .and_then(|len| len.checked_add(8))
12098 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
12099 let mut index = Vec::with_capacity(index_len);
12100 put_u32(
12101 &mut index,
12102 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
12103 );
12104 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
12105 put_u32(
12106 &mut index,
12107 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
12108 );
12109 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
12110 | DICTIONARY_GRAMS
12111 | DICTIONARY_WIDE_GRAMS;
12112 put_u32(&mut index, offset_bits as u32 | flag);
12113 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
12114 let mut end = 0_u64;
12119 for place in places {
12120 if scattered {
12121 put_u64(&mut index, place.start);
12122 put_u64(&mut index, place.length);
12123 } else {
12124 end = end
12125 .checked_add(place.length)
12126 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
12127 put_u64(&mut index, end);
12128 }
12129 }
12130 for place in places {
12131 put_u64(&mut index, place.hash);
12132 }
12133 if rank_ends.len() != rank_blocks {
12136 return Err(invalid("global dictionary order is not the blocks it says it is"));
12137 }
12138 for end in &rank_ends {
12139 put_u64(&mut index, *end);
12140 }
12141 let mut at = 0_usize;
12142 for end in &rank_ends {
12143 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
12144 put_u64(&mut index, checksum(&ranks[at..end]));
12145 at = end;
12146 }
12147 let gram_len = blocks
12148 .checked_mul(TEXT_GRAM_BYTES)
12149 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
12150 let mut grams = Vec::with_capacity(gram_len);
12151 for block in &dictionary.grams {
12152 grams.extend_from_slice(block);
12153 }
12154 put_u64(&mut index, checksum(&grams));
12155 if index.len() != index_len {
12156 return Err(invalid("global dictionary index is not the length it was laid out for"));
12157 }
12158 Ok(EncodedDictionary { index, ranks, grams })
12159}
12160
12161const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
12168
12169fn payload_shapes() -> Vec<chooser::Settled> {
12195 let integers = vec![integer::Kind::Packed];
12196 [
12197 vec![string::Kind::Front, string::Kind::Lz],
12198 vec![string::Kind::Lz, string::Kind::Fsst],
12199 vec![string::Kind::Lz, string::Kind::Plain],
12200 vec![string::Kind::Fsst],
12201 vec![string::Kind::Plain],
12202 ]
12203 .into_iter()
12204 .map(|strings| chooser::Settled::new(strings, integers.clone()))
12205 .collect()
12206}
12207
12208fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
12215 let started = profile.map(|_| std::time::Instant::now());
12216 file.sync()?;
12217 if let (Some(profile), Some(started)) = (profile, started) {
12218 profile.waited(
12219 Stage::Publish,
12220 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
12221 );
12222 }
12223 Ok(())
12224}
12225
12226#[derive(Debug)]
12231pub(crate) struct Unencoded {
12232 column: usize,
12233 at: usize,
12234 ends: Vec<u32>,
12235 bytes: Vec<u8>,
12236 shape: chooser::Settled,
12237}
12238
12239impl Unencoded {
12240 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12242 let values = block_values(&self.ends, &self.bytes);
12243 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12244 }
12245
12246 pub(crate) fn place(&self) -> (usize, usize) {
12248 (self.column, self.at)
12249 }
12250}
12251
12252pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12256
12257fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12259 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12260 for value in values {
12261 for gram in value.windows(4) {
12262 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12263 grams[bit / 8] |= 1 << (bit % 8);
12264 }
12265 }
12266 }
12267 grams
12268}
12269
12270fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12272 let mut out = Vec::with_capacity(ends.len());
12273 let mut from = 0;
12274 for &to in ends {
12275 out.push(&bytes[from..to as usize]);
12276 from = to as usize;
12277 }
12278 out
12279}
12280
12281fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12288 for dictionary in dictionaries.iter_mut().flatten() {
12289 if !dictionary.early.is_empty() {
12290 return Err(Error::internal("a dictionary block handed out never came back"));
12291 }
12292 dictionary.seal_rest();
12293 dictionary.settle_rest()?;
12294 }
12295 encode_waiting(dictionaries)?;
12296 if dictionaries
12299 .iter()
12300 .flatten()
12301 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12302 {
12303 return Err(Error::internal("a dictionary block handed out never came back"));
12304 }
12305 Ok(())
12306}
12307
12308fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12311 let jobs = dictionaries
12312 .iter()
12313 .enumerate()
12314 .flat_map(|(column, held)| {
12315 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12316 })
12317 .collect::<Vec<_>>();
12318 if jobs.is_empty() {
12319 return Ok(());
12320 }
12321 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12322 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12323 Ok((column, at, held.encode_waiting(at)?))
12324 };
12325 let workers = std::thread::available_parallelism()
12326 .map_or(1, usize::from)
12327 .min(MAX_FREQUENCY_WORKERS)
12328 .min(jobs.len());
12329 let made = if workers <= 1 {
12330 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12331 } else {
12332 let next = AtomicUsize::new(0);
12333 let jobs = &jobs;
12334 let pieces = std::thread::scope(|scope| {
12335 (0..workers)
12336 .map(|_| {
12337 scope.spawn(|| {
12338 let mut mine = Vec::new();
12339 loop {
12340 let job = next.fetch_add(1, Atomic::Relaxed);
12341 let Some(&(column, at)) = jobs.get(job) else { break };
12342 mine.push(one(column, at)?);
12343 }
12344 Ok(mine)
12345 })
12346 })
12347 .collect::<Vec<_>>()
12348 .into_iter()
12349 .map(|handle| {
12350 handle
12351 .join()
12352 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12353 })
12354 .collect::<Result<Vec<_>>>()
12355 })?;
12356 pieces.into_iter().flatten().collect()
12357 };
12358 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12359 (0..dictionaries.len()).map(|_| Vec::new()).collect();
12360 for (column, at, bytes) in made {
12361 done[column].push((at, bytes));
12362 }
12363 for (column, mut made) in done.into_iter().enumerate() {
12364 if made.is_empty() {
12365 continue;
12366 }
12367 let Some(held) = dictionaries[column].as_mut() else { continue };
12368 made.sort_by_key(|(at, _)| *at);
12369 let waiting = std::mem::take(&mut held.waiting);
12370 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12371 if held.encoded() != at {
12372 return Err(Error::internal("a dictionary block was encoded out of order"));
12373 }
12374 held.push_block(block);
12375 }
12376 }
12377 Ok(())
12378}
12379
12380fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12390 let mut best: Option<(chooser::Settled, usize)> = None;
12391 for shape in payload_shapes() {
12392 let mut size = 0;
12393 for block in sample {
12394 size += string::encode_with(block, &shape)?.len();
12395 }
12396 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12397 best = Some((shape, size));
12398 }
12399 }
12400 best.map(|(shape, _)| shape)
12401 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12402}
12403
12404fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12411 let mut out = Vec::with_capacity(order.len() * 4);
12412 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12413 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12414 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12415 for block in order.chunks(TEXT_RANK_BLOCK) {
12416 let base = block.first().map_or(0, |&(head, _)| head);
12419 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12420 let width = (u64::BITS - span.leading_zeros()) as usize;
12421 heads.clear();
12422 codes.clear();
12423 for &(head, code) in block {
12424 heads.push(head.wrapping_sub(base));
12425 codes.push(u64::from(code));
12426 }
12427 put_u64(&mut out, base);
12428 out.push(width as u8);
12429 bitpack::pack_tail(&heads, width, &mut out)
12430 .map_err(|_| invalid("global dictionary heads do not pack"))?;
12431 bitpack::pack_tail(&codes, code_bits, &mut out)
12432 .map_err(|_| invalid("global dictionary codes do not pack"))?;
12433 ends.push(out.len() as u64);
12434 }
12435 Ok((out, ends))
12436}
12437
12438fn open_global_dictionary(
12445 file: Arc<File>,
12446 page: Page,
12447 ty: &LogicalType,
12448 keep_budget: usize,
12449) -> Result<Vector> {
12450 if !coded_type(ty) {
12451 return Err(invalid("global dictionary belongs to a non-string column"));
12452 }
12453 let mut header = [0; DICTIONARY_HEADER];
12454 read_at(&file, page.offset, &mut header)?;
12455 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12456 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12457 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12458 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12459 let scattered = width & DICTIONARY_SCATTERED != 0;
12460 let has_grams = width & DICTIONARY_GRAMS != 0;
12461 let gram_width =
12462 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12463 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12464 if per_block != TEXT_PAYLOAD_VALUES {
12465 return Err(invalid("global dictionary block width differs"));
12466 }
12467 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12468 return Err(invalid("global dictionary block count differs from its value count"));
12469 }
12470 if offset_bits > u32::BITS as usize {
12471 return Err(invalid("global dictionary packs offsets past a payload"));
12472 }
12473 let offset_len = offset_bytes(count, offset_bits);
12474 let ranks = count;
12479 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12480 let payload_words = if scattered { 3 } else { 2 };
12484 let hash_len = blocks
12485 .checked_mul(payload_words * 8)
12486 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12487 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12488 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12489 let gram_len = if has_grams {
12490 blocks
12491 .checked_mul(gram_width)
12492 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12493 } else {
12494 0
12495 };
12496 let index_len = DICTIONARY_HEADER
12497 .checked_add(offset_len)
12498 .and_then(|len| len.checked_add(hash_len))
12499 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12500 if index_len > page.length as usize {
12501 return Err(invalid("global dictionary offset index exceeds its page"));
12502 }
12503 let mut index = vec![0; index_len];
12504 index[..DICTIONARY_HEADER].copy_from_slice(&header);
12505 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12506 if checksum(&index) != page.hash {
12507 return Err(invalid("global dictionary index checksum differs"));
12508 }
12509 let word_end = index_len - usize::from(has_grams) * 8;
12510 let gram_hash = has_grams
12511 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12512 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12513 .chunks_exact(8)
12514 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12515 .collect::<Vec<_>>();
12516 let mut rest = words.split_off(blocks * payload_words);
12517 let rank_hashes = rest.split_off(rank_blocks);
12518 let rank_ends = rest;
12519 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12522 return Err(invalid("global dictionary order blocks do not rise"));
12523 }
12524 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12525 .map_err(|_| invalid("global dictionary rank overflow"))?;
12526 let body_len = index_len
12527 .checked_add(rank_len)
12528 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12529 if body_len > page.length as usize {
12530 return Err(invalid("global dictionary order exceeds its page"));
12531 }
12532 let gram_end = body_len
12533 .checked_add(gram_len)
12534 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12535 if gram_end > page.length as usize {
12536 return Err(invalid("global dictionary signatures exceed their page"));
12537 }
12538 let grams = gram_hash.map(|hash| NativeGrams {
12539 start: page.offset + body_len as u64,
12540 length: gram_len,
12541 width: gram_width,
12542 hash,
12543 verdicts: Mutex::new(Vec::new()),
12544 });
12545 let mut offsets = index;
12549 offsets.truncate(DICTIONARY_HEADER + offset_len);
12550 let hashes = words.split_off(blocks * (payload_words - 1));
12551 let (starts, lengths) = if scattered {
12552 let mut starts = Vec::with_capacity(blocks);
12553 let mut lengths = Vec::with_capacity(blocks);
12554 for pair in words.chunks_exact(2) {
12555 starts.push(pair[0]);
12556 lengths.push(pair[1]);
12557 }
12558 (starts, lengths)
12559 } else {
12560 let base = page.offset + gram_end as u64;
12564 let mut starts = Vec::with_capacity(blocks);
12565 let mut lengths = Vec::with_capacity(blocks);
12566 let mut at = 0_u64;
12567 for &end in &words {
12568 let len = end
12569 .checked_sub(at)
12570 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12571 starts.push(base + at);
12572 lengths.push(len);
12573 at = end;
12574 }
12575 (starts, lengths)
12576 };
12577 let stored_len = page.length as u64 - gram_end as u64;
12583 if scattered && stored_len == 0 {
12584 let size = file.metadata().map_err(io)?.len();
12585 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12586 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12587 });
12588 if !inside {
12589 return Err(invalid("global dictionary block lies outside the file"));
12590 }
12591 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12592 return Err(invalid("global dictionary blocks do not bound the payload"));
12593 }
12594 Vector::external_text(
12595 ty.clone(),
12596 Arc::new(NativeText {
12597 file,
12598 values: count,
12599 offsets,
12600 offset_bits,
12601 value_ends: OnceLock::new(),
12602 value_lens: OnceLock::new(),
12603 ends_asked: AtomicUsize::new(0),
12604 ranks,
12605 rank_at: page.offset + index_len as u64,
12606 rank_ends,
12607 rank_hashes,
12608 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12609 code_bits: code_width(count),
12610 code_ranks: OnceLock::new(),
12611 starts,
12612 lengths,
12613 hashes,
12614 grams,
12615 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12616 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12617 keep_budget,
12618 payload_kept: AtomicUsize::new(0),
12619 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12620 visit_dropped: AtomicUsize::new(0),
12621 searched: Mutex::new(HashMap::new()),
12622 }),
12623 )
12624}
12625
12626fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12639 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12641 let mut cur = Cursor::new(bytes);
12642 let codec = cur.u8()?;
12643 if cur.u8()? == 2 {
12644 cur.take(rows.div_ceil(8))?;
12645 }
12646 Ok((codec, cur.at))
12647 }
12648 let Ok((codec, at)) = cascade_at(rows, bytes) else {
12649 return "UNREADABLE".to_string();
12650 };
12651 let tail = &bytes[at..];
12652 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12653 match codec {
12654 0 => match ty {
12655 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12656 _ => "FIXED".to_string(),
12657 },
12658 1 => "DICT(PLAIN)".to_string(),
12659 2 => "FOR+BITPACK".to_string(),
12660 3 => "TABLE DICT".to_string(),
12661 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12662 5 => described(integer::describe(tail)),
12663 6 => described(string::describe(tail)),
12664 other => format!("CODEC {other}"),
12665 }
12666}
12667
12668fn decode_selected_stable_codes(
12673 rows: usize,
12674 bytes: &[u8],
12675 positions: &[usize],
12676 out: &mut Vec<Option<u32>>,
12677) -> Result<bool> {
12678 if positions.windows(2).any(|pair| pair[0] >= pair[1])
12679 || positions.last().is_some_and(|&position| position >= rows)
12680 {
12681 return Err(invalid("selected code positions are not sorted and in range"));
12682 }
12683 let mut cur = Cursor::new(bytes);
12684 let codec = cur.u8()?;
12685 if codec != 3 && codec != 4 {
12686 return Ok(false);
12687 }
12688 let flag = cur.u8()?;
12689 let mask = match flag {
12690 0 | 1 => None,
12691 2 => {
12692 let at = cur.at;
12693 let len = rows.div_ceil(8);
12694 cur.take(len)?;
12695 Some((at, len))
12696 }
12697 _ => return Err(invalid("page validity tag differs")),
12698 };
12699 let valid = |row: usize| match flag {
12700 0 => true,
12701 1 => false,
12702 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12703 _ => unreachable!("the validity tag was checked"),
12704 };
12705 if codec == 4 {
12706 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12707 for (&row, code) in positions.iter().zip(wide) {
12708 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12709 out.push(valid(row).then_some(code));
12710 }
12711 return Ok(true);
12712 }
12713 let codes_at = cur.at;
12714 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12715 cur.take(codes_len)?;
12716 if cur.at != bytes.len() {
12717 return Err(invalid("global code page has trailing bytes"));
12718 }
12719 let codes = &bytes[codes_at..codes_at + codes_len];
12720 for &row in positions {
12721 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12722 let code = u32::from_le_bytes(
12723 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12724 );
12725 out.push(valid(row).then_some(code));
12726 }
12727 Ok(true)
12728}
12729
12730fn decode_at(
12736 ty: &LogicalType,
12737 rows: usize,
12738 bytes: &[u8],
12739 global: Option<Arc<Vector>>,
12740 positions: &[u32],
12741) -> Result<Vector> {
12742 if positions.last().is_some_and(|&last| last as usize >= rows) {
12743 return Err(invalid("a position is past the end of the part"));
12744 }
12745 if bytes.first() == Some(&5)
12748 && positions.len().saturating_mul(8) <= rows
12749 && bytes
12751 .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12752 .is_some_and(integer::pointed)
12753 {
12754 return cascade_at(ty, rows, bytes, positions);
12755 }
12756 if bytes.first() != Some(&6) {
12757 return decode(ty, rows, bytes, global)?.gather(positions);
12758 }
12759 if !coded_type(ty) {
12760 return Err(invalid("compressed text codec belongs to a non-string page"));
12761 }
12762 let mut cur = Cursor::new(bytes);
12763 cur.u8()?;
12764 let validity = match cur.u8()? {
12765 0 => Validity::AllValid,
12766 1 => Validity::AllInvalid,
12767 2 => {
12768 let mask = cur.take(rows.div_ceil(8))?;
12769 Validity::from_iter(positions.len(), |at| {
12770 let row = positions[at] as usize;
12771 mask[row / 8] >> (row % 8) & 1 == 1
12772 })
12773 }
12774 _ => return Err(invalid("page validity tag differs")),
12775 };
12776 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12777 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12778 push_values(&mut values, ty, &ends)?;
12779 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12780}
12781
12782fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12788 fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12789 values
12790 .iter()
12791 .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12792 .collect()
12793 }
12794 let mut cur = Cursor::new(bytes);
12795 cur.u8()?;
12796 let validity = match cur.u8()? {
12797 0 => Validity::AllValid,
12798 1 => Validity::AllInvalid,
12799 2 => {
12800 let mask = cur.take(rows.div_ceil(8))?;
12801 Validity::from_iter(positions.len(), |at| {
12802 let row = positions[at] as usize;
12803 mask[row / 8] >> (row % 8) & 1 == 1
12804 })
12805 }
12806 _ => return Err(invalid("page validity tag differs")),
12807 };
12808 let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12809 let values = integer::decode_selected(&bytes[cur.at..], &at)
12810 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12811 if values.len() != positions.len() {
12812 return Err(invalid("cascade page holds the wrong number of rows"));
12813 }
12814 let data = match ty {
12815 LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12816 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12817 LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12818 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12819 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12820 LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12821 LogicalType::BigInt
12822 | LogicalType::Timestamp
12823 | LogicalType::Time
12824 | LogicalType::TimeTz
12825 | LogicalType::TimestampTz
12826 | LogicalType::TimestampS
12827 | LogicalType::TimestampMs
12828 | LogicalType::TimestampNs => Data::Int64(values.into()),
12829 LogicalType::Decimal { .. } => match ty.physical() {
12830 PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12831 PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12832 PhysicalType::Int64 => Data::Int64(values.into()),
12833 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12834 },
12835 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12836 };
12837 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12838}
12839
12840fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12844 if ty == &LogicalType::Varchar {
12845 return values.push_run_in_place(0, ends);
12846 }
12847 let mut start = 0;
12848 for &end in ends {
12849 let len = end
12850 .checked_sub(start)
12851 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12852 values.push_bytes_in_place(start, len)?;
12853 start = end;
12854 }
12855 Ok(())
12856}
12857
12858fn decode(
12859 ty: &LogicalType,
12860 rows: usize,
12861 bytes: &[u8],
12862 global: Option<Arc<Vector>>,
12863) -> Result<Vector> {
12864 let mut cur = Cursor::new(bytes);
12865 let codec = cur.u8()?;
12866 let flag = cur.u8()?;
12867 let validity = match flag {
12868 0 => Validity::AllValid,
12869 1 => Validity::AllInvalid,
12870 2 => {
12871 let mask = cur.take(rows.div_ceil(8))?;
12872 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12873 }
12874 _ => return Err(invalid("page validity tag differs")),
12875 };
12876 if codec == 1 {
12877 if !coded_type(ty) {
12878 return Err(invalid("dictionary codec belongs to a non-string page"));
12879 }
12880 let count = cur.u32()? as usize;
12881 let payload_len = cur.u32()? as usize;
12882 let offset_bytes = cur.take(
12883 (count + 1)
12884 .checked_mul(4)
12885 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12886 )?;
12887 let offsets = offset_bytes
12888 .chunks_exact(4)
12889 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12890 .collect::<Vec<_>>();
12891 let payload = cur.take(payload_len)?.to_vec();
12892 if offsets.first() != Some(&0)
12893 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12894 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12895 {
12896 return Err(invalid("dictionary offsets do not bound the payload"));
12897 }
12898 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12901 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12902 push_values(&mut strings, ty, &ends)?;
12903 let mut codes = Vec::with_capacity(rows);
12904 for _ in 0..rows {
12905 codes.push(cur.u32()?);
12906 }
12907 if codes.iter().any(|code| *code as usize >= count) {
12908 return Err(invalid("dictionary code is out of range"));
12909 }
12910 if cur.at != bytes.len() {
12911 return Err(invalid("dictionary page has trailing bytes"));
12912 }
12913 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12914 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12915 }
12916 if codec == 3 || codec == 4 {
12917 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12918 let codes = if codec == 4 {
12919 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12924 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12925 if codes.len() != rows {
12926 return Err(invalid("encoded code page holds the wrong number of rows"));
12927 }
12928 codes
12929 } else {
12930 let mut codes = Vec::with_capacity(rows);
12931 for _ in 0..rows {
12932 codes.push(cur.u32()?);
12933 }
12934 if cur.at != bytes.len() {
12935 return Err(invalid("global code page has trailing bytes"));
12936 }
12937 codes
12938 };
12939 let highest = codes.iter().copied().max();
12940 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12941 .with_validity(validity));
12942 }
12943 if codec == 6 {
12944 if !coded_type(ty) {
12945 return Err(invalid("compressed text codec belongs to a non-string page"));
12946 }
12947 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12951 if ends.len() != rows {
12952 return Err(invalid("compressed text page holds the wrong number of rows"));
12953 }
12954 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12957 push_values(&mut values, ty, &ends)?;
12958 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12959 }
12960 if codec == 5 {
12961 let data = cascade(ty, &bytes[cur.at..], rows)?;
12963 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12964 }
12965 if codec == 2 {
12966 let width = u32::from(cur.u8()?);
12967 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12968 let count = cur.u32()? as usize;
12969 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12970 let words: Vec<u64> = cur
12971 .take(length)?
12972 .chunks_exact(8)
12973 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12974 .collect();
12975 if cur.at != bytes.len() {
12976 return Err(invalid("packed page has trailing bytes"));
12977 }
12978 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12979 }
12980 if codec != 0 {
12981 return Err(invalid("page codec is unknown"));
12982 }
12983 let data = match ty {
12984 LogicalType::TinyInt => {
12985 let values = cur.take(rows)?;
12986 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12987 }
12988 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12989 LogicalType::SmallInt => {
12990 let values =
12991 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12992 Data::Int16(
12993 values
12994 .chunks_exact(2)
12995 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12996 .collect::<Vec<_>>()
12997 .into(),
12998 )
12999 }
13000 LogicalType::USmallInt => {
13001 let values =
13002 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13003 Data::UInt16(
13004 values
13005 .chunks_exact(2)
13006 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
13007 .collect::<Vec<_>>()
13008 .into(),
13009 )
13010 }
13011 LogicalType::UInteger => {
13012 let values =
13013 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13014 Data::UInt32(
13015 values
13016 .chunks_exact(4)
13017 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
13018 .collect::<Vec<_>>()
13019 .into(),
13020 )
13021 }
13022 LogicalType::UBigInt => {
13023 let values =
13024 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13025 Data::UInt64(
13026 values
13027 .chunks_exact(8)
13028 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
13029 .collect::<Vec<_>>()
13030 .into(),
13031 )
13032 }
13033 LogicalType::Integer | LogicalType::Date => {
13034 let values =
13035 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13036 Data::Int32(
13037 values
13038 .chunks_exact(4)
13039 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13040 .collect::<Vec<_>>()
13041 .into(),
13042 )
13043 }
13044 LogicalType::BigInt
13045 | LogicalType::Timestamp
13046 | LogicalType::Time
13047 | LogicalType::TimeTz
13048 | LogicalType::TimestampTz
13049 | LogicalType::TimestampS
13050 | LogicalType::TimestampMs
13051 | LogicalType::TimestampNs => {
13052 let values =
13053 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13054 Data::Int64(
13055 values
13056 .chunks_exact(8)
13057 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13058 .collect::<Vec<_>>()
13059 .into(),
13060 )
13061 }
13062 LogicalType::HugeInt | LogicalType::Uuid => {
13063 let values =
13064 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13065 Data::Int128(
13066 values
13067 .chunks_exact(16)
13068 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13069 .collect::<Vec<_>>()
13070 .into(),
13071 )
13072 }
13073 LogicalType::UHugeInt => {
13074 let values =
13075 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13076 Data::UInt128(
13077 values
13078 .chunks_exact(16)
13079 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13080 .collect::<Vec<_>>()
13081 .into(),
13082 )
13083 }
13084 LogicalType::Float => {
13085 let values =
13086 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13087 Data::Float32(
13088 values
13089 .chunks_exact(4)
13090 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
13091 .collect::<Vec<_>>()
13092 .into(),
13093 )
13094 }
13095 LogicalType::Double => {
13096 let values =
13097 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13098 Data::Float64(
13099 values
13100 .chunks_exact(8)
13101 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
13102 .collect::<Vec<_>>()
13103 .into(),
13104 )
13105 }
13106 LogicalType::Interval => {
13107 let values =
13108 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13109 Data::Interval(
13110 values
13111 .chunks_exact(16)
13112 .map(|item| {
13113 (
13114 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
13115 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
13116 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
13117 )
13118 })
13119 .collect::<Vec<_>>()
13120 .into(),
13121 )
13122 }
13123 LogicalType::Boolean => {
13124 let values = cur.take(rows)?;
13125 if values.iter().any(|value| *value > 1) {
13126 return Err(invalid("boolean page has another value"));
13127 }
13128 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
13129 }
13130 LogicalType::Decimal { .. } => match ty.physical() {
13133 PhysicalType::Int16 => {
13134 let values =
13135 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13136 Data::Int16(
13137 values
13138 .chunks_exact(2)
13139 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13140 .collect::<Vec<_>>()
13141 .into(),
13142 )
13143 }
13144 PhysicalType::Int32 => {
13145 let values =
13146 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13147 Data::Int32(
13148 values
13149 .chunks_exact(4)
13150 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13151 .collect::<Vec<_>>()
13152 .into(),
13153 )
13154 }
13155 PhysicalType::Int64 => {
13156 let values =
13157 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13158 Data::Int64(
13159 values
13160 .chunks_exact(8)
13161 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13162 .collect::<Vec<_>>()
13163 .into(),
13164 )
13165 }
13166 _ => {
13167 let values =
13168 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13169 Data::Int128(
13170 values
13171 .chunks_exact(16)
13172 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13173 .collect::<Vec<_>>()
13174 .into(),
13175 )
13176 }
13177 },
13178 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
13179 let offset_bytes = cur
13180 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
13181 let offsets = offset_bytes
13182 .chunks_exact(4)
13183 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13184 .collect::<Vec<_>>();
13185 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
13186 if offsets.first() != Some(&0)
13187 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13188 || offsets.windows(2).any(|pair| pair[0] > pair[1])
13189 {
13190 return Err(invalid("string offsets do not bound the payload"));
13191 }
13192 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13200 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13201 push_values(&mut values, ty, &ends)?;
13202 Data::Varlen(values)
13203 }
13204 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
13205 };
13206 if cur.at != bytes.len() {
13207 return Err(invalid("page has trailing bytes"));
13208 }
13209 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
13210}
13211
13212#[cfg(test)]
13213mod tests {
13214 use std::fs::{self, OpenOptions};
13215 use std::io::{Seek, SeekFrom, Write};
13216 use std::path::PathBuf;
13217 use std::time::{SystemTime, UNIX_EPOCH};
13218
13219 use rudb_common::Stat;
13220 use rudb_common::Value;
13221 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
13222 use rudb_common::stat::Provenance;
13223
13224 use super::*;
13225
13226 #[test]
13227 fn head_is_the_value_padded_to_eight_bytes() {
13228 let bytes: Vec<u8> = (1..=12).collect();
13229 for len in 0..=bytes.len() {
13230 let value = &bytes[..len];
13231 let mut word = [0; 8];
13232 let take = len.min(8);
13233 word[..take].copy_from_slice(&value[..take]);
13234 assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13235 }
13236 assert!(head(b"ab") < head(b"ab\x01"));
13237 assert!(head(b"abcd") < head(b"abce"));
13238 }
13239
13240 #[test]
13241 fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13242 for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13243 let mut bytes = Vec::new();
13244 put_u32(&mut bytes, length);
13245 put_u32(&mut bytes, entries);
13246 bytes.push(1);
13247 assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13248 }
13249 let mut bytes = Vec::new();
13250 put_u32(&mut bytes, 1);
13251 put_u32(&mut bytes, 0);
13252 bytes.push(1);
13253 assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13254 }
13255
13256 #[test]
13257 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13258 let bytes: Vec<u8> =
13259 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13260 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13261 let whole = content_name(&bytes[..length]);
13262 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13263 let mut namer = ContentNamer::default();
13264 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13265 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13266 }
13267 }
13268 }
13269
13270 #[derive(Debug)]
13273 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13274
13275 impl chooser::Chooser for TestsEverything<'_> {
13276 fn name(&self) -> &'static str {
13277 "tests everything"
13278 }
13279
13280 fn narrow_strings(
13281 &self,
13282 values: &[&[u8]],
13283 offered: &[string::Kind],
13284 depth: u8,
13285 ) -> Vec<string::Kind> {
13286 self.0.narrow_strings(values, offered, depth)
13287 }
13288
13289 fn narrow_integers(
13290 &self,
13291 values: &[i64],
13292 offered: &[integer::Kind],
13293 depth: u8,
13294 ) -> Vec<integer::Kind> {
13295 self.0.narrow_integers(values, offered, depth)
13296 }
13297 }
13298
13299 #[test]
13300 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13301 let columns: Vec<Vec<i64>> = vec![
13302 vec![],
13303 vec![5; 1000],
13304 (0..1000).collect(),
13305 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13306 (0..1000).map(|row| row / 50).collect(),
13307 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13308 (0..1000).map(|row| (row * 7919) % 13).collect(),
13309 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13310 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13311 (0..1000).map(|row| i64::MIN + row % 3).collect(),
13312 ];
13313 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13314 for column in &columns {
13315 for chooser in choosers {
13316 let quick = integer::encode_with(column, chooser).unwrap();
13317 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13318 assert_eq!(
13319 quick,
13320 full,
13321 "{} on {:?}",
13322 chooser.name(),
13323 &column[..column.len().min(8)]
13324 );
13325 }
13326 }
13327 }
13328
13329 #[test]
13332 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13333 let mut settling = Settling::default();
13334 for part in 0..STRIPE_PARTS as i64 {
13335 let values: Vec<i64> = (0..2048)
13336 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13337 .collect();
13338 let searched = integer::encode_with(&values, &Fixed).unwrap();
13339 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13340 }
13341 }
13342
13343 #[test]
13347 fn text_pages_share_a_table_until_the_text_changes() {
13348 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13349 let english: Vec<Vec<u8>> = (0..1024)
13350 .map(|row: usize| {
13351 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13352 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13353 })
13354 .collect();
13355 let digits: Vec<Vec<u8>> =
13356 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13357 let mut settling = Settling::default();
13358 for page in 0..8 {
13359 let values: Vec<&[u8]> =
13360 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13361 let payload = values.iter().map(|value| value.len()).sum();
13362 let out = settling.text(&values, payload).unwrap().unwrap();
13363 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13364 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13365 assert!(
13366 out.len() * 4 <= alone.len() * 5,
13367 "page {page}: {} against {}",
13368 out.len(),
13369 alone.len()
13370 );
13371 let since = settling.symbols.as_ref().unwrap().since;
13372 assert_eq!(since, page % 4, "page {page}");
13373 }
13374 }
13375
13376 #[test]
13380 fn a_column_that_changes_under_the_shape_is_searched_again() {
13381 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13382 let mut noise = move || {
13383 state ^= state << 13;
13384 state ^= state >> 7;
13385 state ^= state << 17;
13386 (state % 1_000_000) as i64
13387 };
13388 let mut settling = Settling::default();
13389 for part in 0..STRIPE_PARTS as i64 {
13390 let values: Vec<i64> = match part / 16 {
13391 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13392 1 => (0..2048).map(|_| noise()).collect(),
13393 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13394 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13395 };
13396 let settled = settling.encode(&values).unwrap();
13397 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13398 let searched = integer::encode_with(&values, &Fixed).unwrap();
13399 assert!(
13400 settled.len() * 4 <= searched.len() * 5,
13401 "part {part}: {} settled against {} searched, {} against {}",
13402 settled.len(),
13403 searched.len(),
13404 integer::describe(&settled).unwrap(),
13405 integer::describe(&searched).unwrap(),
13406 );
13407 }
13408 }
13409
13410 #[test]
13411 fn checksum_matches_fixed_vectors() {
13412 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13413 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13414 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13415 }
13416
13417 #[test]
13418 fn sorting_across_threads_matches_sorting_on_one() {
13419 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13420 let mut next = move || {
13421 state ^= state << 13;
13422 state ^= state >> 7;
13423 state ^= state << 17;
13424 state
13425 };
13426 let mut values = Vec::new();
13427 for at in 0..150_000_u64 {
13428 let value = match next() % 6 {
13429 0 => Vec::new(),
13430 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13431 2 => format!("https://example.com/path/{at}").into_bytes(),
13432 3 => b"same".to_vec(),
13433 4 => vec![0xff; (next() % 12) as usize],
13434 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13435 };
13436 values.push(value);
13437 }
13438 let value = |code: u32| values[code as usize].as_slice();
13439 for workers in [1, 2, 3, 8, 32] {
13440 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13441 let mut across = one.clone();
13442 sort_by_value(&mut one, value);
13443 sort_by_value_across(&mut across, value, workers);
13444 assert_eq!(one, across, "{workers} workers");
13445 }
13446 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13447 sort_by_value_across(&mut sorted, value, 8);
13448 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13449 }
13450
13451 fn path(label: &str) -> PathBuf {
13452 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13453 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13454 }
13455
13456 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13461 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13462 (0..dictionary.values())
13463 .map(|code| {
13464 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13465 flat[from..to].to_vec()
13466 })
13467 .collect()
13468 }
13469
13470 fn attached(table: &Table) -> Vec<&Section> {
13477 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13478 }
13479
13480 #[test]
13482 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13483 const SPANS: usize = 64;
13484 const SPAN: usize = 512;
13485 let path = path("positional");
13486 let content: Vec<u8> =
13487 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13488 fs::write(&path, &content).expect("the file is written");
13489 let file = Arc::new(File::open(&path).expect("the file opens"));
13490 std::thread::scope(|scope| {
13491 for _ in 0..8 {
13492 let file = Arc::clone(&file);
13493 scope.spawn(move || {
13494 for _ in 0..64 {
13495 for span in 0..SPANS {
13496 let mut bytes = [0_u8; SPAN];
13497 read_at(&file, (span * SPAN) as u64, &mut bytes)
13498 .expect("the span reads");
13499 assert!(
13500 bytes.iter().all(|byte| *byte == span as u8),
13501 "span {span} came back as {}",
13502 bytes[0],
13503 );
13504 }
13505 }
13506 });
13507 }
13508 });
13509 let mut past = [0_u8; SPAN];
13510 let end = (SPANS * SPAN) as u64;
13511 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13512 assert!(error.message().contains("ends before its declared length"), "{error}");
13513 drop(file);
13514 let _ = fs::remove_file(&path);
13515 }
13516
13517 #[test]
13524 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13525 let path = path("cursor");
13526 let mut writer = Writer::create(
13527 &path,
13528 "items",
13529 vec![
13530 Field::required("id", LogicalType::Integer),
13531 Field::new("text", LogicalType::Varchar),
13532 ],
13533 )
13534 .expect("new file");
13535 writer.append(&sample()).expect("first part");
13536 writer.append(&sample()).expect("second part");
13537 writer.finish().expect("commit");
13538 let reader = Reader::open(&path).expect("reopen from disk");
13539 assert_eq!(reader.table().rows(), 6);
13540 let ids = reader.read(0, &[0]).expect("the integer page reads back");
13541 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13542 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13543 let text = reader.read(1, &[1]).expect("the text page reads back");
13544 assert_eq!(text.value_at(1, 0), Value::Null);
13545 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13546 let end = reader.table().stripes().iter().flat_map(|stripe| {
13549 stripe
13550 .pages
13551 .iter()
13552 .map(|page| page.offset + u64::from(page.length))
13553 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13554 });
13555 let last = end.fold(HEADER, u64::max);
13556 let directory = fs::metadata(&path).expect("the file is there").len();
13557 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13558 fs::remove_file(path).expect("remove scratch file");
13559 }
13560
13561 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13567 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13568 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13569 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13570 let bits = (width & !DICTIONARY_FLAGS) as usize;
13571 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13572 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13573 DICTIONARY_HEADER as u64
13574 + offset_bytes(count as usize, bits) as u64
13575 + blocks * payload_words * 8
13576 + rank_blocks * 16
13577 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13578 }
13579
13580 fn sample() -> Chunk {
13581 Chunk::new(vec![
13582 Vector::from_values(
13583 LogicalType::Integer,
13584 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13585 )
13586 .expect("integers"),
13587 Vector::from_values(
13588 LogicalType::Varchar,
13589 &[
13590 Value::Varchar("alpha".into()),
13591 Value::Null,
13592 Value::Varchar("long text after a slash".into()),
13593 ],
13594 )
13595 .expect("strings"),
13596 ])
13597 .expect("matching rows")
13598 }
13599
13600 fn sample_ids() -> Chunk {
13601 Chunk::new(vec![
13602 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13603 .expect("integers"),
13604 ])
13605 .expect("one column")
13606 }
13607
13608 #[test]
13609 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13610 let path = path("nulls_for_the_planner");
13613 let mut writer =
13614 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13615 .expect("new file");
13616 let rows = Chunk::new(vec![
13617 Vector::from_values(
13618 LogicalType::Integer,
13619 &[
13620 Value::Integer(4),
13621 Value::Null,
13622 Value::Integer(9),
13623 Value::Null,
13624 Value::Integer(1),
13625 Value::Integer(2),
13626 ],
13627 )
13628 .expect("integers"),
13629 ])
13630 .expect("one column");
13631 writer.append(&rows).expect("the only part");
13632 writer.finish().expect("commit");
13633 let reader = Reader::open(&path).expect("reopen from disk");
13634 let stripes = Stripes::new(reader);
13635 let column = stripes.column("a").expect("the file has that column");
13636 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13637 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13640 fs::remove_file(&path).expect("clean up");
13641 }
13642
13643 #[test]
13644 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13645 let path = path("frequencies_for_the_planner");
13648 let mut writer =
13649 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13650 .expect("new file");
13651 let rows = Chunk::new(vec![
13652 Vector::from_values(
13653 LogicalType::Integer,
13654 &[
13655 Value::Integer(4),
13656 Value::Integer(4),
13657 Value::Integer(4),
13658 Value::Integer(9),
13659 Value::Integer(9),
13660 Value::Integer(1),
13661 ],
13662 )
13663 .expect("integers"),
13664 ])
13665 .expect("one column");
13666 writer.append(&rows).expect("the only part");
13667 writer.finish().expect("commit");
13668 let reader = Reader::open(&path).expect("reopen from disk");
13669 let common = Common::new(reader);
13670 assert_eq!(common.rows(), 6);
13671 let column = common.column("id").expect("the file has that column");
13672 assert_eq!(common.column("nothing"), None);
13673 assert_eq!(
13674 common.rows_with(column, &Bound::Int(4)),
13675 Stat::exact(3, Provenance::FrequencySynopsis)
13676 );
13677 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13679 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13682 assert!(common.remainder(column).is_some());
13683 fs::remove_file(&path).expect("clean up");
13684 }
13685
13686 #[test]
13687 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13688 let path = path("string_frequencies_for_the_planner");
13689 let mut writer =
13690 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13691 .expect("new file");
13692 let rows = Chunk::new(vec![
13693 Vector::from_values(
13694 LogicalType::Varchar,
13695 &[
13696 Value::Varchar(String::new()),
13697 Value::Varchar("alpha".into()),
13698 Value::Varchar(String::new()),
13699 Value::Varchar("beta".into()),
13700 Value::Varchar(String::new()),
13701 ],
13702 )
13703 .expect("strings"),
13704 ])
13705 .expect("one column");
13706 writer.append(&rows).expect("the only part");
13707 writer.finish().expect("commit");
13708
13709 let reader = Reader::open(&path).expect("reopen from disk");
13710 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13711 let common = Common::new(reader.clone());
13712 let column = common.column("text").expect("the file has that column");
13713 assert_eq!(
13714 common.rows_with(column, &Bound::Bytes(Vec::new())),
13715 Stat::exact(3, Provenance::FrequencySynopsis)
13716 );
13717 assert_eq!(
13718 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13719 Stat::exact(0, Provenance::FrequencySynopsis)
13720 );
13721 assert_eq!(
13722 reader.reads().dictionaries,
13723 0,
13724 "the bounded spellings answer without opening the dictionary index"
13725 );
13726 fs::remove_file(&path).expect("clean up");
13727 }
13728
13729 #[test]
13730 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13731 let path = path("certified_host_groups");
13732 let mut writer =
13733 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13734 .expect("new file");
13735 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13736 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13737 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13738 values.push(Value::Varchar(String::new()));
13739 for part in values.chunks(512) {
13740 writer
13741 .append(
13742 &Chunk::new(vec![
13743 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13744 ])
13745 .expect("one column"),
13746 )
13747 .expect("part written");
13748 }
13749 writer.finish().expect("commit");
13750 let reader = Reader::open(&path).expect("reopen");
13751 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13752 fs::remove_file(&path).expect("clean up");
13753 }
13754
13755 fn bare_table(sections: Vec<Section>) -> Table {
13760 Table {
13761 name: "linked".to_owned(),
13762 fields: vec![Field::required("id", LogicalType::Integer)],
13763 stripes: Vec::new(),
13764 rows: 0,
13765 dictionaries: vec![None],
13766 dictionary_payloads: Vec::new(),
13767 demoted: Vec::new(),
13768 distincts: vec![None],
13769 frequencies: vec![None],
13770 ordinal_bounds: Vec::new(),
13771 pair_frequencies: Vec::new(),
13772 frequency_texts: Vec::new(),
13773 host_groups: None,
13774 clustering: None,
13775 constraints: Constraints::default(),
13776 generation: 1,
13777 sections,
13778 }
13779 }
13780
13781 fn a_key_map_section() -> Section {
13782 Section {
13783 kind: *section::KEY_MAP,
13784 id: 1,
13785 generation: 3,
13786 extents: 1,
13787 extent_page: HEADER,
13788 extent_bytes: section::EXTENT_BYTES as u32,
13789 hash: 0x1234_5678_9abc_def0,
13790 flags: 0,
13791 header_bytes: 24,
13792 }
13793 }
13794
13795 #[test]
13796 fn a_section_table_round_trips_through_a_directory() {
13797 let mut later = a_key_map_section();
13798 later.kind = *b"RUDBZZ9\0";
13799 later.id = 2;
13800 let table = bare_table(vec![a_key_map_section(), later]);
13801 let directory = encode_directory(&table).expect("directory");
13802 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13803 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13804 assert!(decoded.sections()[0].known());
13808 assert!(!decoded.sections()[1].known());
13809 }
13810
13811 #[test]
13812 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13813 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13817 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13818 let older = &directory[..directory.len() - block];
13819 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13820 assert!(decoded.sections().is_empty());
13821 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13822 assert_eq!(decoded.name(), "linked");
13823 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13824 }
13825
13826 #[test]
13827 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13828 let path = path("format_twenty_two");
13835 let mut writer =
13836 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13837 .expect("new file");
13838 let rows = Chunk::new(vec![
13839 Vector::from_values(
13840 LogicalType::Integer,
13841 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13842 )
13843 .expect("integers"),
13844 ])
13845 .expect("one column");
13846 writer.append(&rows).expect("the only part");
13847 writer.finish().expect("commit");
13848
13849 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13850 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13851 drop(file);
13852
13853 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13854 assert_eq!(reader.table().rows(), 3);
13855 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13860
13861 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13864 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13865 drop(file);
13866 let error = Reader::open(&path).expect_err("format 21 is not readable");
13867 assert!(error.to_string().contains("format 21"), "{error}");
13868
13869 fs::remove_file(&path).expect("clean up");
13870 }
13871
13872 #[test]
13873 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13874 let mut past = a_key_map_section();
13879 past.extent_page = 1 << 30;
13880 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13881 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13882 assert!(error.to_string().contains("outside the file"), "{error}");
13883
13884 let mut inside_the_header = a_key_map_section();
13885 inside_the_header.extent_page = 8;
13886 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13887 assert!(
13888 decode_directory(&directory, 1 << 20).is_err(),
13889 "a section may not overlap a header"
13890 );
13891 }
13892
13893 #[test]
13894 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13895 let not_built = Section {
13899 kind: *section::FORWARD_LINK,
13900 id: 9,
13901 generation: 3,
13902 extents: 0,
13903 extent_page: 0,
13904 extent_bytes: 0,
13905 hash: 0,
13906 flags: 0,
13907 header_bytes: 0,
13908 };
13909 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13910 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13911 assert_eq!(decoded.sections(), &[not_built]);
13912
13913 let mut incoherent = not_built;
13916 incoherent.extent_bytes = 28;
13917 incoherent.extent_page = HEADER;
13918 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13919 assert!(decode_directory(&directory, 1 << 20).is_err());
13920 }
13921
13922 #[test]
13923 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13924 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13925 let mut torn = directory.clone();
13926 let count_at = torn.len() - size_of::<u16>();
13927 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13928 assert!(decode_directory(&torn, 1 << 20).is_err());
13931 }
13932
13933 fn linked_file(label: &str, rows: i32) -> PathBuf {
13935 let path = path(label);
13936 let mut writer =
13937 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13938 .expect("new file");
13939 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13940 let chunk =
13941 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13942 .expect("one column");
13943 writer.append(&chunk).expect("the only part");
13944 writer.finish().expect("commit");
13945 path
13946 }
13947
13948 fn a_key_map_payload() -> Vec<u8> {
13949 (0..512_u32).flat_map(u32::to_le_bytes).collect()
13952 }
13953
13954 #[test]
13955 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13956 let path = linked_file("attach", 64);
13957 let payload = a_key_map_payload();
13958 let table = attach(
13959 &path,
13960 "items",
13961 &[section::Attachment {
13962 kind: *section::KEY_MAP,
13963 id: 0,
13964 flags: 2,
13965 header_bytes: 40,
13966 bytes: &payload,
13967 }],
13968 )
13969 .expect("attach a key map");
13970 assert_eq!(attached(&table).len(), 1);
13971
13972 let reader = Reader::open(&path).expect("reopen after the attach");
13973 let held = attached(reader.table());
13974 assert_eq!(held.len(), 1);
13975 assert_eq!(held[0].kind, *section::KEY_MAP);
13976 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13977 assert_eq!(held[0].header_bytes, 40);
13978 assert_eq!(held[0].generation, 1);
13982 assert!(held[0].usable(reader.table().generation()));
13983 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13984 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13985
13986 fs::remove_file(&path).expect("clean up");
13987 }
13988
13989 #[test]
13990 fn attaching_a_section_answers_every_row_exactly_as_before() {
13991 let path = linked_file("attach_changes_nothing", 300);
13996 let before = Reader::open(&path).expect("open before");
13997 let rows = before.table().rows();
13998 let first = before.read(0, &[0]).expect("read before");
13999 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
14000 let layout = before.layout().columns_total();
14001 drop(before);
14002
14003 let payload = a_key_map_payload();
14004 attach(
14005 &path,
14006 "items",
14007 &[section::Attachment {
14008 kind: *section::KEY_MAP,
14009 id: 0,
14010 flags: 0,
14011 header_bytes: 0,
14012 bytes: &payload,
14013 }],
14014 )
14015 .expect("attach");
14016
14017 let after = Reader::open(&path).expect("open after");
14018 assert_eq!(after.table().rows(), rows);
14019 let read = after.read(0, &[0]).expect("read after");
14020 for (at, value) in values.iter().enumerate() {
14021 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
14022 }
14023 assert_eq!(
14024 after.layout().columns_total(),
14025 layout,
14026 "an attach appends and does not rewrite a column page"
14027 );
14028
14029 fs::remove_file(&path).expect("clean up");
14030 }
14031
14032 #[test]
14033 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
14034 let path = linked_file("attach_twice", 32);
14038 let one = a_key_map_payload();
14039 let two = vec![7_u8; 1024];
14040 let entry = |bytes| section::Attachment {
14041 kind: *section::KEY_MAP,
14042 id: 4,
14043 flags: 1,
14044 header_bytes: 0,
14045 bytes,
14046 };
14047 attach(&path, "items", &[entry(&one)]).expect("first build");
14048 attach(&path, "items", &[entry(&two)]).expect("rebuild");
14049
14050 let reader = Reader::open(&path).expect("reopen");
14051 let held = attached(reader.table());
14052 assert_eq!(held.len(), 1, "one map per column and not one per build");
14053 assert_eq!(reader.payload(held[0]).expect("payload"), two);
14054
14055 fs::remove_file(&path).expect("clean up");
14056 }
14057
14058 #[test]
14059 fn an_attach_carries_through_a_kind_it_does_not_know() {
14060 let path = linked_file("attach_unknown", 16);
14064 let payload = vec![3_u8; 96];
14065 attach(
14066 &path,
14067 "items",
14068 &[section::Attachment {
14069 kind: *b"RUDBZZ9\0",
14070 id: 1,
14071 flags: 0,
14072 header_bytes: 0,
14073 bytes: &payload,
14074 }],
14075 )
14076 .expect("a kind this build does not know still writes");
14077 let key_map = a_key_map_payload();
14078 attach(
14079 &path,
14080 "items",
14081 &[section::Attachment {
14082 kind: *section::KEY_MAP,
14083 id: 0,
14084 flags: 0,
14085 header_bytes: 0,
14086 bytes: &key_map,
14087 }],
14088 )
14089 .expect("attach beside it");
14090
14091 let reader = Reader::open(&path).expect("reopen");
14092 let held = attached(reader.table());
14093 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
14094 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
14095 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
14096
14097 fs::remove_file(&path).expect("clean up");
14098 }
14099
14100 #[test]
14101 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
14102 let path = linked_file("attach_not_built", 8);
14103 attach(
14104 &path,
14105 "items",
14106 &[section::Attachment {
14107 kind: *section::FORWARD_LINK,
14108 id: 2,
14109 flags: 0,
14110 header_bytes: 0,
14111 bytes: &[],
14112 }],
14113 )
14114 .expect("record a link that did not fit the budget");
14115
14116 let reader = Reader::open(&path).expect("reopen");
14117 let held = attached(reader.table());
14118 assert_eq!(held.len(), 1);
14119 assert_eq!(held[0].extents, 0);
14120 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
14121 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
14122 assert!(reader.payload(held[0]).expect("no payload").is_empty());
14123
14124 fs::remove_file(&path).expect("clean up");
14125 }
14126
14127 #[test]
14128 fn a_payload_past_one_extent_is_split_and_joined_back() {
14129 let path = linked_file("attach_two_extents", 8);
14133 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
14134 attach(
14135 &path,
14136 "items",
14137 &[section::Attachment {
14138 kind: *section::KEY_MAP,
14139 id: 0,
14140 flags: 0,
14141 header_bytes: 0,
14142 bytes: &payload,
14143 }],
14144 )
14145 .expect("attach a payload past the bound");
14146
14147 let reader = Reader::open(&path).expect("reopen");
14148 let held = attached(reader.table());
14149 let extents = reader.extents(held[0]).expect("extent table");
14150 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
14151 assert_eq!(extents[0].length, section::MAX_EXTENT);
14152 assert_eq!(extents[1].length, 1);
14153 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
14154 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
14156 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
14157
14158 fs::remove_file(&path).expect("clean up");
14159 }
14160
14161 #[test]
14162 fn a_torn_extent_is_refused_rather_than_decoded() {
14163 let path = linked_file("attach_torn", 8);
14164 let payload = a_key_map_payload();
14165 attach(
14166 &path,
14167 "items",
14168 &[section::Attachment {
14169 kind: *section::KEY_MAP,
14170 id: 0,
14171 flags: 0,
14172 header_bytes: 0,
14173 bytes: &payload,
14174 }],
14175 )
14176 .expect("attach");
14177
14178 let reader = Reader::open(&path).expect("reopen");
14179 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
14180 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
14181 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
14182 drop(file);
14183
14184 let reader = Reader::open(&path).expect("the table still opens");
14185 let error = reader
14186 .payload(&reader.table().sections()[0])
14187 .expect_err("a corrupt payload is not handed out");
14188 assert!(error.to_string().contains("checksum"), "{error}");
14189 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
14192
14193 fs::remove_file(&path).expect("clean up");
14194 }
14195
14196 #[test]
14197 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
14198 let path = linked_file("attach_old_format", 8);
14201 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
14202 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
14203 drop(file);
14204
14205 let payload = a_key_map_payload();
14206 let error = attach(
14207 &path,
14208 "items",
14209 &[section::Attachment {
14210 kind: *section::KEY_MAP,
14211 id: 0,
14212 flags: 0,
14213 header_bytes: 0,
14214 bytes: &payload,
14215 }],
14216 )
14217 .expect_err("format 22 cannot gain a section");
14218 assert!(error.to_string().contains("format 22"), "{error}");
14219 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
14220
14221 fs::remove_file(&path).expect("clean up");
14222 }
14223
14224 #[test]
14225 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
14226 let path = linked_file("attach_bad_header", 8);
14227 let error = attach(
14228 &path,
14229 "items",
14230 &[section::Attachment {
14231 kind: *section::KEY_MAP,
14232 id: 0,
14233 flags: 0,
14234 header_bytes: 40,
14235 bytes: &[1, 2, 3],
14236 }],
14237 )
14238 .expect_err("a writer's bug stops at the write");
14239 assert!(error.to_string().contains("header is longer"), "{error}");
14240
14241 fs::remove_file(&path).expect("clean up");
14242 }
14243
14244 #[test]
14245 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14246 let path = linked_file("attach_wrong_name", 8);
14247 let error = attach(&path, "orders", &[]).expect_err("no such table");
14248 assert!(error.to_string().contains("orders"), "{error}");
14249 fs::remove_file(&path).expect("clean up");
14250 }
14251
14252 #[test]
14253 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14254 let path = path("frequency_prefix_for_the_planner");
14261 let mut writer =
14262 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14263 .expect("new file");
14264 let mut values = vec![Value::Integer(1); 10_000];
14265 for _ in 0..10 {
14266 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14267 }
14268 for part in values.chunks(8_000) {
14271 let rows = Chunk::new(vec![
14272 Vector::from_values(LogicalType::Integer, part).expect("integers"),
14273 ])
14274 .expect("one column");
14275 writer.append(&rows).expect("a part");
14276 }
14277 writer.finish().expect("commit");
14278 let reader = Reader::open(&path).expect("reopen from disk");
14279 let prefix =
14280 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14281 assert_eq!(prefix.entries.len(), 512);
14284 assert_eq!(prefix.omitted_max, 10);
14285 let common = Common::new(reader);
14286 assert_eq!(common.rows(), 16_000);
14287 let column = common.column("id").expect("the file has that column");
14288 assert_eq!(
14289 common.rows_with(column, &Bound::Int(1)),
14290 Stat::exact(10_000, Provenance::FrequencySynopsis)
14291 );
14292 assert_eq!(
14294 common.rows_with(column, &Bound::Int(1_100)),
14295 Stat::exact(10, Provenance::FrequencySynopsis)
14296 );
14297 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14300 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14303 let remainder = common.remainder(column).expect("the list is a prefix");
14307 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14308 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14309 fs::remove_file(&path).expect("clean up");
14310 }
14311
14312 #[test]
14314 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14315 let path = path("empty");
14316 Writer::empty(&path, &[], None).expect("a file with nothing in it");
14317 let catalog = Catalog::open(&path).expect("the empty file opens");
14318 assert_eq!(catalog.len(), 0);
14319 assert!(catalog.is_empty());
14320 assert_eq!(catalog.names().count(), 0);
14321 let mut writer =
14324 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14325 .expect("a table goes into the empty file");
14326 writer.append(&sample_ids()).expect("rows");
14327 writer.finish().expect("commit");
14328 let catalog = Catalog::open(&path).expect("the file opens again");
14329 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14330 fs::remove_file(&path).expect("clean up");
14331 }
14332
14333 #[test]
14343 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14344 let path = path("empty-name");
14345 let field = || vec![Field::required("id", LogicalType::Integer)];
14346 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14347 let catalog = Catalog::open(&path).expect("the file opens");
14348 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14349
14350 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14351 writer.append(&sample_ids()).expect("rows");
14352 writer.finish().expect("commit");
14353 let catalog = Catalog::open(&path).expect("the file opens again");
14354 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14356 let held = catalog.rows().collect::<Vec<_>>();
14357 assert_eq!(held.len(), 1);
14358 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14359
14360 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14362 assert!(error.to_string().contains("same name"), "{error}");
14363 fs::remove_file(&path).expect("clean up");
14364 }
14365
14366 #[test]
14367 fn a_log_anchor_rides_the_catalog_after_the_card_and_goes_forward_with_every_commit() {
14368 let entry = || Entry {
14369 name: "items".to_string(),
14370 fields: vec![Field::required("id", LogicalType::Integer)],
14371 rows: 1,
14372 directory: Page { offset: HEADER, length: 8, hash: 0 },
14373 nonzero: vec![None],
14374 aggregates: vec![None],
14375 distincts: vec![None],
14376 extremes: vec![None],
14377 frequencies: vec![None],
14378 };
14379 let anchor = LogAnchor {
14380 database: 0xfeed,
14381 durable: 41,
14382 lanes: vec![LaneStart { sequence: 3, offset: 4096 }],
14383 voids: vec![43, 47],
14384 };
14385 assert!(!anchor.replays(41) && anchor.replays(42) && !anchor.replays(43));
14386 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14387 for card in [None, Some(&card)] {
14388 let bytes = encode_catalog(&[entry()], &[], card, Some(&anchor)).expect("encodes");
14389 let (_, _, kept, held) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14390 assert_eq!((kept.as_ref(), held.as_ref()), (card, Some(&anchor)));
14391 }
14392 let mut twice = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14393 anchor.encode(&mut twice).expect("encodes");
14394 assert!(decode_catalog(&twice, HEADER + 8).is_err(), "a second anchor");
14395 let mut after = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14396 after.extend_from_slice(DEVICE_CARD);
14397 assert!(decode_catalog(&after, HEADER + 8).is_err(), "a card after the anchor");
14398 let under = LogAnchor { voids: vec![40], ..anchor.clone() };
14399 let bytes = encode_catalog(&[entry()], &[], None, Some(&under)).expect("encodes");
14400 assert!(decode_catalog(&bytes, HEADER + 8).is_err(), "a void under the cut");
14401
14402 let path = path("anchored");
14403 let mut writer =
14404 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14405 .expect("new file");
14406 writer.append(&sample_ids()).expect("rows");
14407 writer.with_log_anchor(anchor.clone()).finish().expect("commit");
14408 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14409 Writer::restate(&path, &[sample_view("v")], None).expect("a view");
14410 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14411 let mut writer =
14412 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14413 .expect("a second table");
14414 writer.append(&sample_ids()).expect("rows");
14415 writer.finish().expect("commit");
14416 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14417 let next = LogAnchor { durable: 90, voids: Vec::new(), ..anchor };
14418 Writer::restate(&path, &[], Some(&next)).expect("a new cut");
14419 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&next));
14420 fs::remove_file(&path).expect("clean up");
14421 let empty = self::path("anchoredempty");
14422 Writer::empty(&empty, &[], Some(&next)).expect("an empty file");
14423 assert_eq!(Catalog::open(&empty).expect("reopen").log_anchor(), Some(&next));
14424 fs::remove_file(&empty).expect("clean up");
14425 }
14426
14427 #[test]
14428 fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14429 let entry = || Entry {
14430 name: "items".to_string(),
14431 fields: vec![Field::required("id", LogicalType::Integer)],
14432 rows: 1,
14433 directory: Page { offset: HEADER, length: 8, hash: 0 },
14434 nonzero: vec![None],
14435 aggregates: vec![None],
14436 distincts: vec![None],
14437 extremes: vec![None],
14438 frequencies: vec![None],
14439 };
14440 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14441 let bytes = encode_catalog(&[entry()], &[], Some(&card), None).expect("encodes");
14442 let (entries, views, kept, _) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14443 assert_eq!((entries.len(), views.len()), (1, 0));
14444 assert_eq!(kept, Some(card));
14445 let bytes = encode_catalog(&[entry()], &[], None, None).expect("encodes");
14446 assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14447 }
14448
14449 fn sample_view(name: &str) -> ViewEntry {
14451 ViewEntry {
14452 name: name.to_string(),
14453 sql: "SELECT id FROM items WHERE id > 0".to_string(),
14454 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14455 aliases: vec!["n".to_string()],
14456 columns: vec![Field::new("n", LogicalType::Integer)],
14457 }
14458 }
14459
14460 #[test]
14461 fn a_view_written_into_the_catalog_comes_back_whole() {
14462 let path = path("views");
14463 let mut writer =
14464 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14465 .expect("new file");
14466 writer.append(&sample_ids()).expect("rows");
14467 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14468 let catalog = Catalog::open(&path).expect("reopen");
14469 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14470 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14473 fs::remove_file(&path).expect("clean up");
14474 }
14475
14476 #[test]
14478 fn appending_a_table_carries_the_views_forward() {
14479 let path = path("viewscarry");
14480 let mut writer =
14481 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14482 .expect("new file");
14483 writer.append(&sample_ids()).expect("rows");
14484 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14485 let mut writer =
14486 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14487 .expect("a second table");
14488 writer.append(&sample_ids()).expect("rows");
14489 writer.finish().expect("commit");
14490 let catalog = Catalog::open(&path).expect("reopen");
14491 assert_eq!(catalog.views().count(), 1);
14492 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14493 fs::remove_file(&path).expect("clean up");
14494 }
14495
14496 #[test]
14498 fn restating_the_views_leaves_every_table_where_it_was() {
14499 let path = path("restate");
14500 let mut writer =
14501 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14502 .expect("new file");
14503 writer.append(&sample_ids()).expect("rows");
14504 writer.finish().expect("commit");
14505 let before = fs::metadata(&path).expect("the file is there").len();
14506 Writer::restate(&path, &[sample_view("v"), sample_view("w")], None).expect("two views");
14507 let catalog = Catalog::open(&path).expect("reopen");
14508 assert_eq!(catalog.views().count(), 2);
14509 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14510 let after = fs::metadata(&path).expect("the file is there").len();
14513 assert!(after > before, "a generation was written");
14514 assert!(after - before < before, "the table was not written again");
14515 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14518 assert_eq!(reader.table().rows, 3);
14519 Writer::restate(&path, &[], None).expect("no views at all");
14522 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14523 fs::remove_file(&path).expect("clean up");
14524 }
14525
14526 #[test]
14528 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14529 let bytes = encode_catalog(
14530 &[Entry {
14531 name: "items".to_string(),
14532 fields: vec![Field::required("id", LogicalType::Integer)],
14533 rows: 1,
14534 directory: Page { offset: HEADER, length: 8, hash: 0 },
14535 nonzero: vec![None],
14536 aggregates: vec![None],
14537 distincts: vec![None],
14538 extremes: vec![None],
14539 frequencies: vec![None],
14540 }],
14541 &[sample_view("items")],
14542 None,
14543 None,
14544 )
14545 .expect("it encodes, because encoding does not look");
14546 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14547 assert!(error.to_string().contains("same name"), "{error}");
14548 }
14549
14550 #[test]
14553 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14554 let rows: usize = 300;
14555 let text: Vec<String> =
14556 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14557 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14558 let mut page = vec![6, 2];
14559 page.extend((0..rows.div_ceil(8)).map(|byte| {
14560 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14561 }));
14562 let compressed = string::encode_only(string::Kind::Fsst, &values)
14563 .expect("encoded")
14564 .expect("text this repetitive compresses");
14565 page.extend_from_slice(&compressed);
14566 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14567 let positions = [0_u32, 3, 8, 13, 200, 299];
14568 let some =
14569 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14570 assert_eq!(some.len(), positions.len());
14571 for (at, &row) in positions.iter().enumerate() {
14572 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14573 }
14574 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14575 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14576 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14577 }
14578
14579 #[test]
14582 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14583 let path = path("rows");
14584 let mut writer = Writer::create(
14585 &path,
14586 "items",
14587 vec![
14588 Field::required("id", LogicalType::Integer),
14589 Field::new("text", LogicalType::Varchar),
14590 ],
14591 )
14592 .expect("new file");
14593 let rows = 2_000;
14594 let chunk = Chunk::new(vec![
14595 Vector::from_values(
14596 LogicalType::Integer,
14597 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14598 )
14599 .expect("integers"),
14600 Vector::from_values(
14601 LogicalType::Varchar,
14602 &(0..rows)
14603 .map(|row| {
14604 if row % 7 == 2 {
14605 Value::Null
14606 } else {
14607 Value::Varchar(format!("a comment about order {}", row * 13))
14608 }
14609 })
14610 .collect::<Vec<_>>(),
14611 )
14612 .expect("strings"),
14613 ])
14614 .expect("matching rows");
14615 writer.append(&chunk).expect("one part");
14616 writer.finish().expect("commit");
14617 let reader = Reader::open(&path).expect("reopen from disk");
14618 let positions = [1_u32, 2, 9, 1_000, 1_999];
14619 for whole in [true, false] {
14620 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14621 let all = reader.read(0, &[0, 1]).expect("the whole part");
14622 assert_eq!(some.len(), positions.len());
14623 for column in 0..2 {
14624 for (at, &row) in positions.iter().enumerate() {
14625 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14626 }
14627 }
14628 }
14629 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14630 }
14631
14632 #[test]
14633 fn committed_file_reopens_and_reads_only_requested_columns() {
14634 let path = path("reopen");
14635 let mut writer = Writer::create(
14636 &path,
14637 "items",
14638 vec![
14639 Field::required("id", LogicalType::Integer),
14640 Field::new("text", LogicalType::Varchar),
14641 ],
14642 )
14643 .expect("new file");
14644 writer.append(&sample()).expect("first part");
14645 writer.append(&sample()).expect("second part");
14646 writer.finish().expect("commit");
14647 let reader = Reader::open(&path).expect("reopen from disk");
14648 assert_eq!(reader.table().rows(), 6);
14649 assert_eq!(reader.table().stripes().len(), 1);
14652 assert_eq!(reader.parts(), 2);
14653 assert_eq!(reader.part_rows(0), 3);
14654 assert_eq!(reader.part_rows(1), 3);
14655 let text = reader.read(1, &[1]).expect("only text page");
14656 assert_eq!(text.width(), 1);
14657 assert_eq!(text.value_at(1, 0), Value::Null);
14658 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14659 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14660 assert_eq!(sparse.width(), 1);
14661 assert_eq!(sparse.value_at(1, 0), Value::Null);
14662 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14663 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14664 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14665 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14666 let count = reader.read(0, &[]).expect("no page is needed for count");
14667 assert_eq!(count.len(), 3);
14668 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14669 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14670 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14671 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14672 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14673 assert_eq!(integers.omitted_max, 2);
14674 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14675 assert_eq!(strings.len(), 3);
14676 assert!(strings.contains(&(Value::Null, 2)));
14677 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14678 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14679 fs::remove_file(path).expect("remove scratch file");
14680 }
14681
14682 #[test]
14690 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14691 let path = path("interleaved-runs");
14692 let mut writer =
14693 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14694 .expect("new file");
14695 for morsel in [2_u64, 0, 3, 1] {
14696 let parts = (0..4_u64)
14697 .map(|chunk| {
14698 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14699 let values =
14700 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14701 let column =
14702 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14703 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14704 })
14705 .collect::<Vec<_>>();
14706 writer.append_stripe(parts).expect("a stripe");
14707 }
14708 writer.finish().expect("commit");
14709
14710 let reader = Reader::open(&path).expect("valid directory");
14711 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14712 assert_eq!(reader.table().rows(), 128);
14713 for part in 0..16_usize {
14714 let read = reader.read(part, &[0]).expect("a part back");
14715 for row in 0..8_usize {
14716 let want = i64::try_from(part * 8 + row).expect("small");
14717 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14718 }
14719 }
14720 fs::remove_file(path).expect("remove scratch file");
14721 }
14722
14723 #[test]
14726 fn runs_that_overlap_each_other_are_refused_at_commit() {
14727 let path = path("overlapping-runs");
14728 let mut writer =
14729 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14730 .expect("new file");
14731 let one = |order: (u64, u64)| {
14732 let column =
14733 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14734 (order, Chunk::new(vec![column]).expect("one column"))
14735 };
14736 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14739 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14740 let error = writer.finish().expect_err("the runs overlap");
14741 assert!(error.message().contains("source order"), "{error}");
14742 fs::remove_file(path).expect("remove scratch file");
14743 }
14744
14745 #[test]
14748 fn a_run_longer_than_a_stripe_is_refused() {
14749 let path = path("overlong-run");
14750 let mut writer =
14751 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14752 .expect("new file");
14753 let parts = (0..=STRIPE_PARTS)
14754 .map(|at| {
14755 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14756 .expect("a column");
14757 let chunk = Chunk::new(vec![column]).expect("one column");
14758 ((0, u64::try_from(at).expect("small")), chunk)
14759 })
14760 .collect::<Vec<_>>();
14761 let error = writer.append_stripe(parts).expect_err("one part too many");
14762 assert!(error.message().contains("more parts than it holds"), "{error}");
14763 fs::remove_file(path).expect("remove scratch file");
14764 }
14765
14766 #[test]
14772 fn parts_past_the_stripe_bound_start_a_new_stripe() {
14773 let path = path("stripe-bound");
14774 let mut writer = Writer::create(
14775 &path,
14776 "items",
14777 vec![
14778 Field::required("id", LogicalType::Integer),
14779 Field::new("text", LogicalType::Varchar),
14780 ],
14781 )
14782 .expect("new file");
14783 let parts = STRIPE_PARTS * 2 + 3;
14784 for part in 0..parts {
14785 let id = part as i32;
14786 let chunk = Chunk::new(vec![
14787 Vector::from_values(
14788 LogicalType::Integer,
14789 &[Value::Integer(id), Value::Integer(-id)],
14790 )
14791 .expect("integers"),
14792 Vector::from_values(
14793 LogicalType::Varchar,
14794 &[Value::Varchar(format!("value {part}")), Value::Null],
14795 )
14796 .expect("strings"),
14797 ])
14798 .expect("matching rows");
14799 writer.append(&chunk).expect("one part");
14800 }
14801 writer.finish().expect("commit");
14802
14803 let reader = Reader::open(&path).expect("reopen from disk");
14804 assert_eq!(reader.parts(), parts);
14805 assert_eq!(reader.table().rows(), parts * 2);
14806 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14807 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14808 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14809 assert_eq!(reader.table().stripes()[2].parts(), 3);
14810 for part in (0..parts).rev() {
14813 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14814 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14815 for chunk in [&dense, &sparse] {
14816 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14817 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14818 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14819 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14820 assert_eq!(chunk.value_at(1, 1), Value::Null);
14821 }
14822 }
14823 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14826 assert!(reader.skips(0, &above), "the first stripe stops at 63");
14827 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14828 fs::remove_file(path).expect("remove scratch file");
14829 }
14830
14831 fn scattered(n: i64) -> i64 {
14833 n.wrapping_mul(-7_046_029_254_386_353_131)
14834 }
14835
14836 #[test]
14842 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14843 let path = path("sieve-skip");
14844 let mut writer =
14845 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14846 .expect("new file");
14847 let parts = STRIPE_PARTS + 3;
14848 let per_part = 128;
14852 for part in 0..parts {
14853 let held: Vec<Value> = (0..per_part)
14854 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14855 .collect();
14856 let chunk =
14857 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14858 .expect("one column");
14859 writer.append(&chunk).expect("one part");
14860 }
14861 writer.finish().expect("commit");
14862
14863 let reader = Reader::open(&path).expect("reopen from disk");
14864 let probe = |value: i64| Probe {
14865 column: 0,
14866 op: Op::Equal,
14867 value: Bound::Int(i128::from(scattered(value))),
14868 };
14869 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14870 let tests = [probe(wanted)];
14871 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14872 let home = wanted as usize / per_part;
14873 assert!(kept.contains(&home), "the part holding {wanted} is read");
14874 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14878 }
14879 let absent = [probe((parts * per_part) as i64 + 1)];
14880 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
14881 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
14882 let tests = [probe(0)];
14885 assert!(
14886 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
14887 "the bounds rule out no stripe at all"
14888 );
14889 fs::remove_file(path).expect("remove scratch file");
14890 }
14891
14892 #[test]
14898 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
14899 let path = path("part-range-skip");
14900 let mut writer =
14901 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14902 .expect("new file");
14903 let parts = STRIPE_PARTS + 3;
14904 let per_part = 128;
14905 for part in 0..parts {
14906 let held: Vec<Value> = (0..per_part)
14910 .map(|row| {
14911 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14912 })
14913 .collect();
14914 let chunk =
14915 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14916 .expect("one column");
14917 writer.append(&chunk).expect("one part");
14918 }
14919 writer.finish().expect("commit");
14920
14921 let reader = Reader::open(&path).expect("reopen from disk");
14922 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14923 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
14924 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
14925 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
14927 fs::remove_file(path).expect("remove scratch file");
14928 }
14929
14930 #[test]
14934 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
14935 let path = path("part-range-certain");
14936 let mut writer =
14937 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14938 .expect("new file");
14939 let parts = STRIPE_PARTS + 3;
14940 let per_part = 128;
14941 for part in 0..parts {
14942 let held: Vec<Value> = (0..per_part)
14943 .map(|row| {
14944 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14945 })
14946 .collect();
14947 let chunk =
14948 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14949 .expect("one column");
14950 writer.append(&chunk).expect("one part");
14951 }
14952 writer.finish().expect("commit");
14953
14954 let reader = Reader::open(&path).expect("reopen from disk");
14955 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14956 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
14957 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
14958 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
14961 fs::remove_file(path).expect("remove scratch file");
14962 }
14963
14964 #[test]
14967 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14968 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14969 let path = path("part-range-page");
14970 let mut writer =
14971 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14972 .expect("new file");
14973 for part in 0..parts {
14974 let held: Vec<Value> = (0..128)
14975 .map(|row| {
14976 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14977 })
14978 .collect();
14979 let chunk = Chunk::new(vec![
14980 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14981 ])
14982 .expect("one column");
14983 writer.append(&chunk).expect("one part");
14984 }
14985 writer.finish().expect("commit");
14986 let reader = Reader::open(&path).expect("reopen from disk");
14987 let bytes = reader.layout().columns[0].part_ranges;
14988 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14989 fs::remove_file(path).expect("remove scratch file");
14990 }
14991 }
14992
14993 #[test]
14996 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14997 let long = vec![b'a'; PART_BOUND_BYTES * 2];
14998 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14999 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
15000 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
15001 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
15002 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
15003 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
15004 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
15005 }
15006
15007 #[test]
15010 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
15011 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
15012 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
15013 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
15014 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
15015 }
15016
15017 #[test]
15029 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
15030 let parts = 4;
15031 let per_part = 1024;
15032 let rows = parts * per_part;
15033 let written = |name: &str, keys: &[i64]| {
15034 let path = path(name);
15035 let fields = vec![Field::required("key", LogicalType::BigInt)];
15036 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
15037 for part in 0..parts {
15038 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
15039 .iter()
15040 .map(|key| Value::BigInt(*key))
15041 .collect();
15042 let chunk = Chunk::new(vec![
15043 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
15044 ])
15045 .expect("one column");
15046 writer.append(&chunk).expect("one part");
15047 }
15048 writer.finish().expect("commit");
15049 path
15050 };
15051 let climbing = |step: &dyn Fn(usize) -> i64| {
15054 let mut key = 0;
15055 (0..rows)
15056 .map(|row| {
15057 key += step(row);
15058 key
15059 })
15060 .collect::<Vec<i64>>()
15061 };
15062 let ascending = climbing(&|row| (row % 3) as i64);
15063 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
15067 let near_path = written("stored-near", &ascending);
15068 let far_path = written("stored-far", &sparse);
15069
15070 let one = Reader::open(&near_path).expect("reopen from disk");
15071 let other = Reader::open(&far_path).expect("reopen from disk");
15072 let near = one.stored(0).expect("the column is stored");
15073 let far = other.stored(0).expect("the column is stored");
15074 assert_eq!(near.len(), parts, "one row per part");
15075 assert_eq!(far.len(), parts);
15076 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
15079 assert_eq!(total(&near), one.layout().columns[0].pages);
15080 assert_eq!(total(&far), other.layout().columns[0].pages);
15081 assert!(
15082 total(&near) * 2 < total(&far),
15083 "the sparse keys cost more, {} against {}",
15084 total(&far),
15085 total(&near)
15086 );
15087 for (at, part) in near.iter().enumerate() {
15089 assert_eq!(part.part, at);
15090 assert_eq!(part.row, at * per_part);
15091 assert_eq!(part.rows, per_part);
15092 let held = &ascending[at * per_part..(at + 1) * per_part];
15093 assert_eq!(part.low, Some(Value::BigInt(held[0])));
15094 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
15095 assert_eq!(part.nulls, Some(0));
15096 }
15097 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
15100 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
15101 assert_ne!(near[0].encoding, far[0].encoding);
15102 fs::remove_file(near_path).expect("remove scratch file");
15103 fs::remove_file(far_path).expect("remove scratch file");
15104 }
15105
15106 #[test]
15116 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
15117 let path = path("sieve-pays");
15118 let fields = vec![
15119 Field::required("spread", LogicalType::BigInt),
15120 Field::required("repeated", LogicalType::BigInt),
15121 ];
15122 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
15123 let parts = 3;
15124 let per_part = 1024;
15125 for part in 0..parts {
15126 let base = (part * per_part) as i64;
15127 let spread: Vec<Value> =
15128 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
15129 let repeated: Vec<Value> =
15130 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
15131 let chunk = Chunk::new(vec![
15132 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
15133 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
15134 ])
15135 .expect("two columns");
15136 writer.append(&chunk).expect("one part");
15137 }
15138 writer.finish().expect("commit");
15139
15140 let reader = Reader::open(&path).expect("reopen from disk");
15141 let layout = reader.layout();
15142 let spread = &layout.columns[0];
15143 let repeated = &layout.columns[1];
15144 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
15145 assert_eq!(
15146 repeated.sieves, 0,
15147 "a column whose filter costs more than its parts keeps none"
15148 );
15149 for column in &layout.columns {
15152 assert!(
15153 column.sieves < column.pages,
15154 "{} spends {} on sieves over {} of data",
15155 column.name,
15156 column.sieves,
15157 column.pages
15158 );
15159 }
15160 let absent = [Probe {
15162 column: 0,
15163 op: Op::Equal,
15164 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
15165 }];
15166 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
15167 fs::remove_file(path).expect("remove scratch file");
15168 }
15169
15170 #[test]
15176 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
15177 let path = path("sieve-damaged");
15178 let mut writer =
15179 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
15180 .expect("new file");
15181 let rows = 128;
15182 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
15183 let chunk =
15184 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15185 .expect("one column");
15186 writer.append(&chunk).expect("one part");
15187 writer.finish().expect("commit");
15188
15189 let page = Reader::open(&path).expect("reopen").table.stripes[0]
15190 .sieves
15191 .get(0)
15192 .expect("a sieve page");
15193 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
15194 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
15195 file.write_all(&[0xff]).expect("damage one byte");
15196 drop(file);
15197
15198 let reader = Reader::open(&path).expect("reopen the damaged file");
15199 let absent =
15200 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
15201 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
15202 assert_eq!(
15203 reader.read(0, &[0]).expect("the rows are untouched").len(),
15204 usize::try_from(rows).expect("a small count")
15205 );
15206 fs::remove_file(path).expect("remove scratch file");
15207 }
15208
15209 #[test]
15215 fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
15216 let path = path("asked-twice");
15217 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15218 let mut writer =
15219 Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
15220 .expect("new file");
15221 for part in 0..parts {
15222 let chunk = Chunk::new(vec![
15223 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15224 .expect("integers"),
15225 ])
15226 .expect("matching rows");
15227 writer.append(&chunk).expect("one part");
15228 }
15229 writer.finish().expect("commit");
15230
15231 let pool = PagePool::new(usize::MAX);
15232 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15233 let a = catalog.table("a").expect("a");
15234 let stripes = a.table().stripes().len();
15235 for part in 0..parts {
15236 for _ in 0..2 {
15237 let chunk = a.read(part, &[0]).expect("a part");
15238 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15239 }
15240 }
15241 assert_eq!(
15242 a.pages.load(Atomic::Relaxed),
15243 stripes,
15244 "a page a stripe, read on the second ask"
15245 );
15246 assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
15247 let column = a.cache.columns[0].lock().expect("the column");
15248 assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
15249 drop(column);
15250 drop((a, catalog));
15251 fs::remove_file(path).expect("remove scratch file");
15252 }
15253
15254 #[test]
15265 fn workers_that_want_the_same_stripe_read_it_once() {
15266 let path = path("single-flight");
15267 let mut writer =
15268 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15269 .expect("new file");
15270 for part in 0..STRIPE_PARTS {
15271 let id = part as i32;
15272 let chunk = Chunk::new(vec![
15273 Vector::from_values(
15274 LogicalType::Integer,
15275 &[Value::Integer(id), Value::Integer(-id)],
15276 )
15277 .expect("integers"),
15278 ])
15279 .expect("matching rows");
15280 writer.append(&chunk).expect("one part");
15281 }
15282 writer.finish().expect("commit");
15283
15284 let reader = Reader::open(&path).expect("reopen from disk");
15285 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
15286 for part in 0..STRIPE_PARTS {
15289 reader.read(part, &[0]).expect("a part");
15290 }
15291 assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
15292 let barrier = std::sync::Barrier::new(8);
15293 std::thread::scope(|scope| {
15294 for worker in 0..8 {
15295 let reader = &reader;
15296 let barrier = &barrier;
15297 scope.spawn(move || {
15298 barrier.wait();
15299 for part in (worker..STRIPE_PARTS).step_by(8) {
15300 let chunk = reader.read(part, &[0]).expect("a whole page read");
15301 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15302 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15303 }
15304 });
15305 }
15306 });
15307 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15308 fs::remove_file(path).expect("remove scratch file");
15309 }
15310
15311 #[test]
15324 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15325 let opened = |label: &str, rows_per_part: i32| {
15326 let path = path(label);
15327 let mut writer =
15328 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15329 .expect("new file");
15330 for part in 0..STRIPE_PARTS * 3 {
15331 let values = (0..rows_per_part)
15335 .map(|row| {
15336 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15337 })
15338 .collect::<Vec<_>>();
15339 let chunk = Chunk::new(vec![
15340 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15341 ])
15342 .expect("matching rows");
15343 writer.append(&chunk).expect("one part");
15344 }
15345 writer.finish().expect("commit");
15346 let reader = Reader::open(&path).expect("reopen from disk");
15347 let size = fs::metadata(&path).expect("the file is there").len();
15348 let out = (reader.reads(), reader.table().stripes().len(), size);
15349 fs::remove_file(path).expect("remove scratch file");
15350 out
15351 };
15352
15353 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15354 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15355 assert_eq!(
15356 thin_stripes, fat_stripes,
15357 "the same stripe count is what makes this a fair ask"
15358 );
15359 assert!(
15360 fat_size > thin_size * 50,
15361 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15362 );
15363
15364 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15365 assert_eq!(thin.pages, 0, "opening read a page");
15366 assert_eq!(fat.pages, 0, "opening read a page");
15367 assert_eq!(thin.indexes, 0, "opening read an index");
15368 assert_eq!(fat.indexes, 0, "opening read an index");
15369 assert!(
15372 fat.opening.bytes < thin.opening.bytes * 2,
15373 "opening the thin file read {} bytes and the fat one read {}",
15374 thin.opening.bytes,
15375 fat.opening.bytes
15376 );
15377 }
15378
15379 #[test]
15387 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15388 let path = path("open-twice");
15389 let mut writer =
15390 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15391 .expect("new file");
15392 for part in 0..STRIPE_PARTS * 3 {
15393 let chunk = Chunk::new(vec![
15394 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15395 .expect("integers"),
15396 ])
15397 .expect("matching rows");
15398 writer.append(&chunk).expect("one part");
15399 }
15400 writer.finish().expect("commit");
15401
15402 let first = Reader::open(&path).expect("open");
15403 for part in 0..first.parts() {
15406 first.read(part, &[0]).expect("a part");
15407 }
15408 assert!(first.reads().indexes > 0, "the scan has to have read something");
15409 let second = Reader::open(&path).expect("open again");
15410
15411 assert_eq!(first.reads().opening, second.reads().opening);
15412 assert_eq!(
15413 second.reads().pages,
15414 0,
15415 "the second open read a page off the back of the first"
15416 );
15417 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15418 fs::remove_file(path).expect("remove scratch file");
15419 }
15420
15421 #[test]
15429 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15430 let path = path("index-cache");
15431 let mut writer =
15432 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15433 .expect("new file");
15434 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15435 for part in 0..parts {
15436 let id = part as i32;
15437 let chunk = Chunk::new(vec![
15438 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15439 ])
15440 .expect("matching rows");
15441 writer.append(&chunk).expect("one part");
15442 }
15443 writer.finish().expect("commit");
15444
15445 let reader = Reader::open(&path).expect("reopen from disk");
15446 let stripes = reader.table().stripes().len();
15447 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15448 for _ in 0..3 {
15451 for part in 0..parts {
15452 let chunk = reader.read(part, &[0]).expect("a part");
15453 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15454 }
15455 }
15456 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15457 assert!(
15458 reader.pages.load(Atomic::Relaxed) > stripes,
15459 "the pages are the ones that get read again, which is what makes the index count mean \
15460 something"
15461 );
15462 fs::remove_file(path).expect("remove scratch file");
15463 }
15464
15465 #[test]
15472 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15473 let path = path("page-pool");
15474 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15475 let fields = || vec![Field::required("id", LogicalType::Integer)];
15476 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15477 for table in ["a", "b"] {
15478 if table == "b" {
15479 writer = writer.next("b".to_string(), fields()).expect("a second table");
15480 }
15481 for part in 0..parts {
15482 let chunk = Chunk::new(vec![
15483 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15484 .expect("integers"),
15485 ])
15486 .expect("matching rows");
15487 writer.append(&chunk).expect("one part");
15488 }
15489 }
15490 writer.finish().expect("commit");
15491
15492 let pool = PagePool::new(usize::MAX);
15493 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15494 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15495 let stripes = a.table().stripes().len();
15496 assert!(
15497 stripes > CACHED_STRIPES_PER_COLUMN * 2,
15498 "the floor has to be smaller than a table"
15499 );
15500 let scan = |reader: &Reader| {
15501 for part in 0..parts {
15502 let chunk = reader.read(part, &[0]).expect("a part");
15503 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15504 }
15505 };
15506 scan(&a);
15509 assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15510 assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15511 scan(&a);
15512 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15513 scan(&a);
15514 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15515 let one = pool.bytes();
15516 assert!(one > 0, "the pool counts what the reader holds");
15517
15518 pool.budget.store(one, Atomic::Relaxed);
15520 scan(&b);
15521 scan(&b);
15522 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15523 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15524 let column = a.cache.columns[0].lock().expect("the column");
15525 let held = column.pages.iter().flatten().count();
15526 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15527 drop(column);
15528
15529 drop((a, b, catalog));
15531 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15532 scan(&c);
15533 scan(&c);
15534 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15535 fs::remove_file(path).expect("remove scratch file");
15536 }
15537
15538 #[test]
15547 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15548 let workers = CACHED_STRIPES_PER_COLUMN + 4;
15549 let path = path("stripe-per-worker");
15550 let mut writer =
15551 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15552 .expect("new file");
15553 for part in 0..STRIPE_PARTS * workers {
15554 let chunk = Chunk::new(vec![
15555 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15556 .expect("integers"),
15557 ])
15558 .expect("matching rows");
15559 writer.append(&chunk).expect("one part");
15560 }
15561 writer.finish().expect("commit");
15562
15563 let read = |told: bool| {
15564 let reader = Reader::open(&path).expect("reopen from disk");
15565 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15566 if told {
15567 reader.keep_stripes(workers);
15568 }
15569 for part in 0..reader.parts() {
15571 reader.read(part, &[0]).expect("a part");
15572 }
15573 let barrier = std::sync::Barrier::new(workers);
15574 std::thread::scope(|scope| {
15575 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15576 let reader = &reader;
15577 let barrier = &barrier;
15578 scope.spawn(move || {
15579 for part in run {
15580 barrier.wait();
15581 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15582 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15583 }
15584 assert!(worker < workers);
15585 });
15586 }
15587 });
15588 reader.pages.load(Atomic::Relaxed)
15589 };
15590
15591 assert_eq!(read(true), workers, "one page read per stripe and no more");
15592 assert!(read(false) > workers, "a cache that small is read again on every part");
15593 fs::remove_file(path).expect("remove scratch file");
15594 }
15595
15596 #[test]
15601 fn a_damaged_index_page_is_an_error() {
15602 let path = path("damaged-index");
15603 let mut writer =
15604 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15605 .expect("new file");
15606 writer.append(&sample_ids()).expect("first part");
15607 writer.append(&sample_ids()).expect("second part");
15608 writer.finish().expect("commit");
15609
15610 let reader = Reader::open(&path).expect("valid directory");
15611 let index = reader.table.stripes[0].index;
15612 let mut byte = [0; 1];
15613 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15614 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15615 file.seek(SeekFrom::Start(index.offset)).expect("index start");
15616 file.write_all(&[!byte[0]]).expect("damage the first part length");
15617 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15618 assert!(error.message().contains("index page section checksum differs"), "{error}");
15619 fs::remove_file(path).expect("remove scratch file");
15620 }
15621
15622 #[test]
15629 fn every_integer_width_round_trips_through_a_page() {
15630 let path = path("integer-widths");
15631 let columns = [
15632 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15633 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15634 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15635 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15636 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15637 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15638 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15639 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15640 ];
15641 let fields = columns
15642 .iter()
15643 .enumerate()
15644 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15645 .collect::<Vec<_>>();
15646 let vectors = columns
15647 .iter()
15648 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15649 .collect::<Vec<_>>();
15650 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15651 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15652 writer.finish().expect("commit");
15653
15654 let reader = Reader::open(&path).expect("reopen from disk");
15655 let wanted = (0..columns.len()).collect::<Vec<_>>();
15656 let read = reader.read(0, &wanted).expect("every column");
15657 assert_eq!(read.len(), 2);
15658 for (at, (ty, values)) in columns.iter().enumerate() {
15660 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15661 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15662 }
15663 fs::remove_file(path).expect("remove scratch file");
15664 }
15665
15666 #[test]
15677 fn every_other_type_the_format_knows_round_trips_through_a_page() {
15678 let path = path("other-types");
15679 let columns = [
15680 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15681 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15682 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15683 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15684 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15685 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15686 (
15687 LogicalType::TimestampTz,
15688 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15689 ),
15690 (
15691 LogicalType::Interval,
15692 vec![
15693 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15694 Value::Interval { months: 13, days: -1, micros: 1 },
15695 ],
15696 ),
15697 (
15698 LogicalType::Blob,
15699 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15700 ),
15701 ];
15702 let fields = columns
15703 .iter()
15704 .enumerate()
15705 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15706 .collect::<Vec<_>>();
15707 let vectors = columns
15708 .iter()
15709 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15710 .collect::<Vec<_>>();
15711 let mut writer = Writer::create(&path, "others", fields).expect("new file");
15712 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15713 writer.finish().expect("commit");
15714
15715 let reader = Reader::open(&path).expect("reopen from disk");
15716 let wanted = (0..columns.len()).collect::<Vec<_>>();
15717 let read = reader.read(0, &wanted).expect("every column");
15718 assert_eq!(read.len(), 2);
15719 for (at, (ty, values)) in columns.iter().enumerate() {
15720 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15721 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15722 }
15723 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15726 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15727
15728 fs::remove_file(path).expect("remove scratch file");
15729 }
15730
15731 #[test]
15737 fn a_nan_survives_being_written_down() {
15738 let path = path("nan");
15739 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15740 .expect("a NaN vector");
15741 let mut writer =
15742 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15743 .expect("new file");
15744 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15745 writer.finish().expect("commit");
15746 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15747 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15748 assert!(back.is_nan(), "a NaN came back as {back}");
15749 fs::remove_file(path).expect("remove scratch file");
15750 }
15751
15752 #[test]
15759 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15760 let path = path("uuid-and-bit");
15761 let uuids = vec![0_i128, i128::MIN, -1];
15762 let mut bits = StringColumn::new();
15763 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15764 bits.push_bytes(value);
15765 }
15766 let expected = bits.clone();
15767 let fields =
15768 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15769 let vectors = vec![
15770 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15771 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15772 ];
15773 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15774 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15775 writer.finish().expect("commit");
15776
15777 let reader = Reader::open(&path).expect("reopen from disk");
15778 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15779 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15780 panic!("a uuid column is the 128 bit lane")
15781 };
15782 assert_eq!(back.as_slice(), uuids.as_slice());
15783 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15784 panic!("a bit column is bytes")
15785 };
15786 for row in 0..expected.len() {
15787 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15788 }
15789 fs::remove_file(path).expect("remove scratch file");
15790 }
15791
15792 #[test]
15795 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15796 let mut rows: Vec<Option<u64>> = Vec::new();
15797 let mut state = 0x2545_f491_4f6c_dd1d_u64;
15798 for index in 0..400_000_u64 {
15799 state ^= state << 13;
15800 state ^= state >> 7;
15801 state ^= state << 17;
15802 let times = 1 + (state % 7) as usize;
15803 let bits = match state % 11 {
15804 0 => None,
15805 1..=3 => Some(state % 16),
15806 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15807 };
15808 rows.extend(std::iter::repeat_n(bits, times));
15809 }
15810 let mut by_row = Candidates::default();
15811 for &bits in &rows {
15812 by_row.add(bits, 1);
15813 }
15814 let mut by_run = Candidates::default();
15815 let mut run = Run::default();
15816 let mut runs = 0_usize;
15817 for &bits in &rows {
15818 if let Some((bits, times)) = run.push(bits) {
15819 by_run.add(bits, times);
15820 runs += 1;
15821 }
15822 }
15823 if let Some((bits, times)) = run.take() {
15824 by_run.add(bits, times);
15825 }
15826 assert!(runs < rows.len() / 2, "the rows came in runs");
15827 assert!(by_row.decrements > 0, "the table filled and turned values away");
15828 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15829 assert_eq!(by_run.nulls, by_row.nulls);
15830 assert_eq!(by_run.decrements, by_row.decrements);
15831 }
15832
15833 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15834 let mut pairs = candidates.pairs().collect::<Vec<_>>();
15835 pairs.sort_unstable();
15836 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15837 pairs
15838 }
15839
15840 #[derive(Default)]
15843 struct MapCandidates {
15844 counts: HashMap<u64, u32>,
15845 nulls: u32,
15846 decrements: u64,
15847 }
15848
15849 impl MapCandidates {
15850 fn add(&mut self, bits: Option<u64>, mut times: u32) {
15851 while times > 0 {
15852 let held = match bits {
15853 Some(bits) => self.counts.get_mut(&bits),
15854 None if self.nulls != 0 => Some(&mut self.nulls),
15855 None => None,
15856 };
15857 if let Some(count) = held {
15858 *count = count.saturating_add(times);
15859 return;
15860 }
15861 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15862 match bits {
15863 Some(bits) => {
15864 self.counts.insert(bits, times);
15865 }
15866 None => self.nulls = times,
15867 }
15868 return;
15869 }
15870 self.counts.retain(|_, count| {
15871 *count -= 1;
15872 *count != 0
15873 });
15874 self.nulls = self.nulls.saturating_sub(1);
15875 self.decrements = self.decrements.saturating_add(1);
15876 times -= 1;
15877 }
15878 }
15879 }
15880
15881 #[test]
15885 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
15886 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
15887 let mut table = Candidates::default();
15888 let mut oracle = MapCandidates::default();
15889 let mut state = seed;
15890 for index in 0..300_000_u64 {
15891 state ^= state << 13;
15892 state ^= state >> 7;
15893 state ^= state << 17;
15894 let bits = match state % 13 {
15895 0 => None,
15896 1..=4 => Some(state % 40),
15897 5 => Some((index % 1000) * 1_000_000),
15898 _ => Some(state),
15899 };
15900 let times = 1 + (state >> 60) as u32 % 3;
15901 table.add(bits, times);
15902 oracle.add(bits, times);
15903 if index % 50_000 == 0 {
15904 let mut expected =
15905 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15906 expected.sort_unstable();
15907 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
15908 }
15909 }
15910 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15911 expected.sort_unstable();
15912 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
15913 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
15914 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
15915 assert!(table.decrements > 0, "seed {seed} never filled the table");
15916 for &(bits, _) in &expected {
15917 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
15918 }
15919 }
15920 }
15921
15922 #[test]
15923 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
15924 let path = path("frequency-ordinals");
15925 let mut writer =
15926 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
15927 .expect("new file");
15928 let mut values = Vec::new();
15929 for leader in 0..10_i64 {
15930 values.extend(std::iter::repeat_n(leader, 100));
15931 }
15932 values.extend(1_000_i64..41_000);
15933 for part in values.chunks(1_024) {
15934 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
15935 .expect("big integers");
15936 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
15937 }
15938 writer.finish().expect("commit");
15939
15940 let reader = Reader::open(&path).expect("reopen from disk");
15941 let occurrences =
15942 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
15943 assert!(occurrences.omitted_max < 100);
15944 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
15945 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
15946 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
15947 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
15948 assert_eq!(
15949 &occurrences.anchor_indices[..1_000]
15950 .iter()
15951 .map(|&entry| occurrences.anchors[entry as usize].clone())
15952 .collect::<Vec<_>>(),
15953 &(0_i64..10)
15954 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
15955 .collect::<Vec<_>>()
15956 );
15957 fs::remove_file(path).expect("remove scratch file");
15958 }
15959
15960 #[test]
15961 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
15962 let path = path("frequency-bits");
15967 let mut writer = Writer::create(
15968 &path,
15969 "items",
15970 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
15971 )
15972 .expect("new file");
15973 let mut rows = Vec::new();
15974 let mut leaders = Vec::new();
15975 for leader in 0..10_u64 {
15976 let count = 300 - leader * 10;
15977 let (unsigned, signed) = if leader == 0 {
15978 (Value::Null, Value::Null)
15979 } else {
15980 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
15981 };
15982 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
15983 leaders.push(((unsigned, count), (signed, count)));
15984 }
15985 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
15986 for part in rows.chunks(1_024) {
15987 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
15988 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
15989 let chunk = Chunk::new(vec![
15990 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
15991 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
15992 ])
15993 .expect("matching columns");
15994 writer.append(&chunk).expect("rows");
15995 }
15996 writer.finish().expect("commit");
15997
15998 let reader = Reader::open(&path).expect("reopen from disk");
15999 for column in 0..2 {
16000 let prefix =
16001 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16002 let wanted = leaders
16003 .iter()
16004 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
16005 .cloned()
16006 .collect::<Vec<_>>();
16007 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
16008 assert!(prefix.omitted_max < 210, "column {column}");
16009 assert_eq!(
16010 reader.distinct_values(column).expect("valid metadata"),
16011 Some(9 + 40_000),
16012 "column {column}"
16013 );
16014 }
16015 fs::remove_file(path).expect("remove scratch file");
16016 }
16017
16018 #[test]
16019 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
16020 let path = path("frequency-tally");
16026 let types = [
16027 LogicalType::TinyInt,
16028 LogicalType::UInteger,
16029 LogicalType::Date,
16030 LogicalType::Timestamp,
16031 ];
16032 let value = |ty: &LogicalType, at: i64| match ty {
16033 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
16034 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
16035 LogicalType::Date => Value::Date(19_000 - at as i32),
16036 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
16037 };
16038 let fields = types
16039 .iter()
16040 .enumerate()
16041 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
16042 .collect::<Vec<_>>();
16043 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16044 let mut rows = Vec::new();
16045 for at in 0..250_i64 {
16046 for _ in 0..=(at % 37) {
16047 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
16048 }
16049 }
16050 for part in rows.chunks(1_000) {
16051 let columns = types
16052 .iter()
16053 .map(|ty| {
16054 let values = part
16055 .iter()
16056 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
16057 .collect::<Vec<_>>();
16058 Vector::from_values(ty.clone(), &values).expect("a column")
16059 })
16060 .collect();
16061 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
16062 }
16063 writer.finish().expect("commit");
16064
16065 let reader = Reader::open(&path).expect("reopen from disk");
16066 for (column, ty) in types.iter().enumerate() {
16067 let mut counts = HashMap::<Option<i64>, u64>::new();
16068 for row in &rows {
16069 *counts.entry(*row).or_default() += 1;
16070 }
16071 let wanted = counts
16072 .into_iter()
16073 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
16074 .collect::<Vec<_>>();
16075 let prefix =
16076 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16077 assert_eq!(prefix.entries.len(), 2, "column {column}");
16078 assert!(prefix.omitted_max > 0, "column {column}");
16079 for (value, count) in &prefix.entries {
16080 let held =
16081 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
16082 assert_eq!(held, Some(count), "column {column} value {value:?}");
16083 }
16084 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
16085 assert_eq!(
16086 reader.distinct_values(column).expect("valid metadata"),
16087 Some(wanted.len() as u64 - 1),
16088 "column {column}"
16089 );
16090 }
16091 fs::remove_file(path).expect("remove scratch file");
16092 }
16093
16094 #[test]
16095 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
16096 let edge = FREQUENCY_CANDIDATES as i64;
16101 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
16102 for with_null in [false, true] {
16103 let path = path("distinct-edge");
16104 let mut writer =
16105 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
16106 .expect("new file");
16107 let mut values = Vec::new();
16108 for round in 0..2 {
16109 for value in 0..distinct {
16110 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
16111 values.extend(std::iter::repeat_n(
16112 Value::BigInt(value * 7_919 % distinct),
16113 repeat,
16114 ));
16115 if with_null && value % 1_000 == 0 {
16116 values.push(Value::Null);
16117 }
16118 }
16119 }
16120 if with_null {
16121 values.push(Value::Null);
16122 }
16123 for part in values.chunks(1_024) {
16124 let chunk = Chunk::new(vec![
16125 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
16126 ])
16127 .expect("one column");
16128 writer.append(&chunk).expect("rows");
16129 }
16130 writer.finish().expect("commit");
16131 let reader = Reader::open(&path).expect("reopen from disk");
16132 assert_eq!(
16133 reader.distinct_values(0).expect("valid metadata"),
16134 Some(distinct as u64),
16135 "{distinct} values, null {with_null}"
16136 );
16137 fs::remove_file(path).expect("remove scratch file");
16138 }
16139 }
16140 }
16141
16142 #[test]
16143 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
16144 let path = path("quick-nonzero");
16145 let mut writer = Writer::create(
16146 &path,
16147 "items",
16148 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
16149 )
16150 .expect("create");
16151 for ids in [
16152 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
16153 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
16154 ] {
16155 let labels = vec![Value::Varchar("same".into()); ids.len()];
16156 writer
16157 .append(
16158 &Chunk::new(vec![
16159 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
16160 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
16161 ])
16162 .expect("chunk"),
16163 )
16164 .expect("append");
16165 }
16166 writer.finish().expect("finish");
16167 let catalog = Catalog::open(&path).expect("catalog");
16168 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
16169 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
16170 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
16171 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
16172 let prefix = catalog
16173 .table("items")
16174 .expect("reader")
16175 .frequency_prefix(1)
16176 .expect("valid metadata")
16177 .expect("partial frequencies");
16178 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
16179 assert_eq!(prefix.omitted_max, 1);
16180 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
16181 assert_eq!(
16182 catalog.integer_extremes("items", 1).expect("extremes"),
16183 Some(IntegerExtremes::Values { low: 0, high: 7 })
16184 );
16185 assert_eq!(
16186 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
16187 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16188 );
16189 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
16190 let mut legacy = catalog.clone();
16191 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
16192 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
16193 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
16194 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
16195 Writer::certify_counts(&path).expect("recertify");
16196 assert_eq!(
16197 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
16198 Some(2)
16199 );
16200 assert_eq!(
16201 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
16202 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16203 );
16204 assert_eq!(
16205 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
16206 Some(3)
16207 );
16208 assert_eq!(
16209 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
16210 Some(IntegerExtremes::Values { low: 0, high: 7 })
16211 );
16212 assert_eq!(
16213 Catalog::open(&path)
16214 .expect("reopen")
16215 .exact_numeric_frequencies("items", 1)
16216 .expect("frequencies"),
16217 None
16218 );
16219 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
16220 fs::remove_file(path).expect("remove scratch file");
16221 }
16222
16223 #[test]
16224 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
16225 let path = path("pair-frequencies");
16226 let mut pairs = Vec::new();
16227 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
16228 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
16229 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
16230 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
16231 let mut writer = Writer::create(
16232 &path,
16233 "items",
16234 vec![
16235 Field::required("id", LogicalType::BigInt),
16236 Field::required("phrase", LogicalType::Varchar),
16237 ],
16238 )
16239 .expect("new file");
16240 for part in pairs.chunks(1_024) {
16241 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
16242 let phrases =
16243 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
16244 writer
16245 .append(
16246 &Chunk::new(vec![
16247 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
16248 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
16249 ])
16250 .expect("matching columns"),
16251 )
16252 .expect("rows");
16253 }
16254 writer.finish().expect("commit");
16255
16256 let reader = Reader::open(&path).expect("reopen from disk");
16257 assert!(
16258 reader.table.pair_frequencies.is_empty(),
16259 "no query-specific pair result is stored"
16260 );
16261 fs::remove_file(path).expect("remove scratch file");
16262 }
16263
16264 #[test]
16265 fn legacy_group_answers_are_ignored() {
16266 let path = path("legacy-group-answers");
16267 let mut writer = Writer::create(
16268 &path,
16269 "items",
16270 vec![
16271 Field::required("id", LogicalType::BigInt),
16272 Field::required("text", LogicalType::Varchar),
16273 ],
16274 )
16275 .expect("new file");
16276 writer
16277 .append(
16278 &Chunk::new(vec![
16279 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
16280 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
16281 .expect("text"),
16282 ])
16283 .expect("row"),
16284 )
16285 .expect("append");
16286 writer.finish().expect("commit");
16287 let mut reader = Reader::open(&path).expect("reopen");
16288 let table = Arc::make_mut(&mut reader.table);
16289 table.pair_frequencies.push(PairFrequencySummary {
16290 first: 0,
16291 second: 1,
16292 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
16293 omitted_max: 0,
16294 });
16295 table.host_groups = Some(host::HostSummary {
16296 column: 1,
16297 omitted_max: 0,
16298 entries: vec![host::HostEntry {
16299 host: "fake.test".into(),
16300 count: 999,
16301 bytes_sum: 999,
16302 minimum: "x".into(),
16303 }],
16304 });
16305 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16306 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16307 fs::remove_file(path).expect("remove scratch file");
16308 }
16309
16310 #[test]
16316 fn a_file_from_another_format_says_which_format_it_is() {
16317 let older = path("older-format");
16318 let mut writer =
16319 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16320 .expect("new file");
16321 let chunk = Chunk::new(vec![
16322 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16323 .expect("integers"),
16324 ])
16325 .expect("chunk");
16326 writer.append(&chunk).expect("page written");
16327 writer.finish().expect("commit");
16328
16329 let unreadable =
16333 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16334 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16335 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16336 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16337 drop(file);
16338 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16339 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16340 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16341
16342 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16343 file.seek(SeekFrom::Start(0)).expect("the magic is first");
16344 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16345 drop(file);
16346 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16347 assert!(complaint.contains("magic"), "{complaint}");
16348 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16349 fs::remove_file(older).expect("remove scratch file");
16350 }
16351
16352 #[test]
16353 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16354 let unfinished = path("unfinished");
16355 let mut writer =
16356 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16357 .expect("new file");
16358 let chunk = Chunk::new(vec![
16359 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16360 .expect("integers"),
16361 ])
16362 .expect("chunk");
16363 writer.append(&chunk).expect("page written");
16364 drop(writer);
16365 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16366 fs::remove_file(unfinished).expect("remove scratch file");
16367
16368 let damaged = path("damaged");
16369 let mut writer =
16370 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16371 .expect("new file");
16372 writer.append(&chunk).expect("page written");
16373 writer.finish().expect("commit");
16374 let reader = Reader::open(&damaged).expect("valid directory");
16375 let mut file =
16376 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16377 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16378 file.write_all(&[255]).expect("damage one byte");
16379 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16380 fs::remove_file(damaged).expect("remove scratch file");
16381 }
16382
16383 #[test]
16384 fn damaged_lazy_dictionary_payload_is_an_error() {
16385 let path = path("damaged-dictionary");
16386 let mut writer = Writer::create(
16387 &path,
16388 "items",
16389 vec![
16390 Field::required("id", LogicalType::Integer),
16391 Field::new("text", LogicalType::Varchar),
16392 ],
16393 )
16394 .expect("new file");
16395 writer.append(&sample()).expect("stripe written");
16396 writer.finish().expect("commit");
16397
16398 let reader = Reader::open(&path).expect("valid directory");
16399 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16400 let mut header = [0; DICTIONARY_HEADER];
16403 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16404 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16407 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16408 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16409 let bits = (width & !DICTIONARY_FLAGS) as usize;
16410 let mut start = [0; 8];
16411 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16412 read_at(&reader.file, at, &mut start).expect("the first block's start");
16413 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16414 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16415 file.write_all(&[255]).expect("damage dictionary payload");
16416
16417 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16418 let error =
16419 chunk.validate_external().expect_err("payload corruption must reach the caller");
16420 assert!(error.message().contains("payload checksum differs"), "{error}");
16421 fs::remove_file(path).expect("remove scratch file");
16422 }
16423
16424 #[test]
16434 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16435 let path = path("dictionary-decide");
16436 let rows = 20_000;
16437 let unique =
16439 |row: usize| format!("{row:09} a value that appears exactly once in the table");
16440 let repeated = |row: usize| unique(row / 40);
16442 let mut writer = Writer::create(
16443 &path,
16444 "items",
16445 vec![
16446 Field::required("unique", LogicalType::Varchar),
16447 Field::required("repeated", LogicalType::Varchar),
16448 ],
16449 )
16450 .expect("new file");
16451 for part in (0..rows).step_by(1_000) {
16452 let span = part..(part + 1_000).min(rows);
16453 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16454 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16455 writer
16456 .append(
16457 &Chunk::new(vec![
16458 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16459 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16460 ])
16461 .expect("two columns"),
16462 )
16463 .expect("a part");
16464 }
16465 writer.finish().expect("commit");
16466
16467 let reader = Reader::open(&path).expect("reopen from disk");
16468 assert!(
16469 reader.table.dictionaries[0].is_none(),
16470 "a column with no repeats has nothing to say twice"
16471 );
16472 assert!(
16473 reader.table.dictionaries[1].is_some(),
16474 "a column whose values come round again keeps its dictionary"
16475 );
16476 let mut first = 0;
16477 for part in 0..reader.parts() {
16478 let chunk = reader.read(part, &[0, 1]).expect("a part");
16479 for row in 0..chunk.len() {
16480 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16481 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16482 }
16483 first += chunk.len();
16484 }
16485 assert_eq!(first, rows, "every row was read back");
16486 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16487 let size = fs::metadata(&path).expect("the file is there").len() as usize;
16488 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16489 fs::remove_file(path).expect("remove scratch file");
16490 }
16491
16492 #[test]
16505 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16506 let path = path("dictionary-blocks");
16507 let value = |row: usize| {
16508 let row = row.saturating_sub(8_000);
16509 format!("{row:07} a value long enough to be worth a payload block")
16510 };
16511 let parts = 40;
16512 let per_part = 1000;
16513 let mut writer =
16514 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16515 .expect("new file");
16516 for part in 0..parts {
16517 let values = (0..per_part)
16518 .map(|row| Value::Varchar(value(part * per_part + row)))
16519 .collect::<Vec<_>>();
16520 let chunk = Chunk::new(vec![
16521 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16522 ])
16523 .expect("matching rows");
16524 writer.append(&chunk).expect("a part");
16525 }
16526 writer.finish().expect("commit");
16527
16528 let reader = Reader::open(&path).expect("reopen from disk");
16529 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16530 assert!(
16531 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16532 "the dictionary has to be several blocks for this to be testing anything"
16533 );
16534 for part in [0, parts - 1] {
16535 let chunk = reader.read(part, &[0]).expect("a part");
16536 chunk.validate_external().expect("every payload block checks out");
16537 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16538 }
16539
16540 let mut header = [0; DICTIONARY_HEADER];
16542 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16543 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16544 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16545 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16546 let bits = (width & !DICTIONARY_FLAGS) as usize;
16547 let mut place = [0; 16];
16548 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16549 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16550 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16551 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16552 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16553 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16554 file.write_all(&[255]).expect("damage the last payload block");
16555 let reader = Reader::open(&path).expect("the directory and the index are untouched");
16556 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16557 let error = chunk.validate_external().expect_err("the damage must reach the caller");
16558 assert!(error.message().contains("payload checksum differs"), "{error}");
16559 fs::remove_file(path).expect("remove scratch file");
16560 }
16561
16562 #[test]
16576 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16577 let path = path("dictionary-offsets");
16578 let value = |row: usize| {
16579 let row = row % 5_000;
16580 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16581 };
16582 let rows = 6_000;
16583 let mut writer =
16584 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16585 .expect("new file");
16586 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16587 for part in values.chunks(1_000) {
16588 let chunk =
16589 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16590 .expect("matching rows");
16591 writer.append(&chunk).expect("a part");
16592 }
16593 writer.finish().expect("commit");
16594
16595 let reader = Reader::open(&path).expect("reopen from disk");
16596 assert!(
16597 rows > TEXT_PAYLOAD_VALUES * 4,
16598 "the dictionary has to be several blocks for this to be testing anything"
16599 );
16600 for part in 0..rows / 1_000 {
16601 let chunk = reader.read(part, &[0]).expect("a part");
16602 for row in 0..1_000 {
16603 let row = part * 1_000 + row;
16604 assert_eq!(
16605 chunk.value_at(row % 1_000, 0),
16606 Value::Varchar(value(row)),
16607 "value {row}"
16608 );
16609 }
16610 }
16611 for _ in 0..2 {
16614 for part in 0..rows / 1_000 {
16615 let chunk = reader.read(part, &[0]).expect("a part");
16616 let mut lens = vec![0_i64; 1_000];
16617 let column = chunk.column(0).expect("one column");
16618 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16619 for (row, &len) in lens.iter().enumerate() {
16620 let row = part * 1_000 + row;
16621 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16622 }
16623 }
16624 }
16625 fs::remove_file(path).expect("remove scratch file");
16626 }
16627
16628 #[test]
16630 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16631 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16632 ends.extend([3, 3, 10]);
16633 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16634 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16635 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16636 let long = [5, 70_005, 70_006];
16638 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16639 assert_eq!(lens, [5, 70_000, 1]);
16640 let mut read = Vec::new();
16641 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16642 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16643 ends.push(9);
16644 assert!(lengths_of(&ends).is_none());
16645 }
16646
16647 #[test]
16659 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16660 let path = path("dictionary-once");
16661 let parts = 8;
16662 let per_part = 500;
16663 let value =
16664 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16665 let mut writer =
16666 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16667 .expect("new file");
16668 for part in 0..parts {
16669 let values = (0..per_part)
16670 .map(|row| Value::Varchar(value(part * per_part + row)))
16671 .collect::<Vec<_>>();
16672 let chunk = Chunk::new(vec![
16673 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16674 ])
16675 .expect("matching rows");
16676 writer.append(&chunk).expect("a part");
16677 }
16678 writer.finish().expect("commit");
16679
16680 let reader = Reader::open(&path).expect("reopen from disk");
16681 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16682 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16683
16684 let workers = 16;
16685 let gate = std::sync::Barrier::new(workers);
16686 std::thread::scope(|scope| {
16687 for worker in 0..workers {
16688 let reader = reader.clone();
16689 let gate = &gate;
16690 scope.spawn(move || {
16691 gate.wait();
16692 let chunk = reader.read(worker % parts, &[0]).expect("a part");
16693 assert_eq!(
16694 chunk.value_at(0, 0),
16695 Value::Varchar(value((worker % parts) * per_part))
16696 );
16697 });
16698 }
16699 });
16700
16701 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16702 fs::remove_file(path).expect("remove scratch file");
16703 }
16704
16705 #[test]
16710 fn a_damaged_sorted_order_is_an_error() {
16711 let path = path("damaged-order");
16712 let mut writer = Writer::create(
16713 &path,
16714 "items",
16715 vec![
16716 Field::required("id", LogicalType::Integer),
16717 Field::new("text", LogicalType::Varchar),
16718 ],
16719 )
16720 .expect("new file");
16721 writer.append(&sample()).expect("stripe written");
16722 writer.finish().expect("commit");
16723
16724 let reader = Reader::open(&path).expect("valid directory");
16725 let page = reader.table.dictionaries[1].expect("string dictionary page");
16726 let mut header = [0; DICTIONARY_HEADER];
16727 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16728 let index_len = dictionary_index_len(&header);
16729 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16730 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16731 file.write_all(&[255]).expect("damage the order");
16732
16733 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16734 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16735 assert!(error.message().contains("rank checksum differs"), "{error}");
16736 fs::remove_file(path).expect("remove scratch file");
16737 }
16738
16739 #[test]
16743 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16744 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16747 let path = path("dictionary-order");
16748 let mut writer =
16749 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16750 .expect("new file");
16751 writer
16752 .append(
16753 &Chunk::new(vec![
16754 Vector::from_values(
16755 LogicalType::Varchar,
16756 &spellings.map(|text| Value::Varchar(text.into())),
16757 )
16758 .expect("strings"),
16759 ])
16760 .expect("one column"),
16761 )
16762 .expect("stripe written");
16763 writer.finish().expect("commit");
16764
16765 let reader = Reader::open(&path).expect("valid directory");
16766 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16767 let count = dictionary.ranks().expect("a v10 file stores one");
16768 assert_eq!(count, spellings.len(), "every distinct value has a rank");
16769 let order = (0..count)
16770 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16771 .collect::<Vec<_>>();
16772 let mut seen = order.clone();
16773 seen.sort_unstable();
16774 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16775
16776 let ranked = order
16777 .iter()
16778 .map(|&code| {
16779 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16780 })
16781 .collect::<Vec<_>>();
16782 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16783 expected.sort();
16784 assert_eq!(ranked, expected, "rank order is value order");
16785
16786 for (rank, value) in expected.iter().enumerate() {
16789 assert_eq!(
16790 dictionary.compare_rank(rank, value).expect("compare"),
16791 Ordering::Equal,
16792 "rank {rank} is its own value"
16793 );
16794 if rank > 0 {
16795 assert_eq!(
16796 dictionary.compare_rank(rank - 1, value).expect("compare"),
16797 Ordering::Less,
16798 "rank {rank} follows the one before it"
16799 );
16800 }
16801 }
16802 fs::remove_file(path).expect("remove scratch file");
16803 }
16804
16805 #[test]
16812 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16813 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16814 let path = path("dictionaries-at-once");
16815 let fields = (0..sizes.len())
16816 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16817 .collect::<Vec<_>>();
16818 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16819 let rows = 10_000_usize;
16820 for start in (0..rows).step_by(1_024) {
16821 let columns = sizes
16822 .iter()
16823 .enumerate()
16824 .map(|(column, &size)| {
16825 let values = (start..(start + 1_024).min(rows))
16826 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16827 .collect::<Vec<_>>();
16828 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16829 })
16830 .collect::<Vec<_>>();
16831 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16832 }
16833 writer.finish().expect("commit");
16834
16835 let reader = Reader::open(&path).expect("valid directory");
16836 for (column, &size) in sizes.iter().enumerate() {
16837 let dictionary =
16838 reader.dictionary(column).expect("read").expect("a string column has one");
16839 let count = dictionary.ranks().expect("a v10 file stores one");
16840 assert_eq!(count, size, "column {column} has its own distinct count");
16841 let ranked = (0..count)
16842 .map(|rank| {
16843 let code = dictionary.code_at_rank(rank).expect("a code");
16844 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16845 })
16846 .collect::<Vec<_>>();
16847 let expected = (0..size)
16848 .map(|value| format!("c{column}-{value:05}").into_bytes())
16849 .collect::<Vec<_>>();
16850 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16851 }
16852 fs::remove_file(path).expect("remove scratch file");
16853 }
16854
16855 #[test]
16863 fn a_large_dictionary_ranks_in_value_order() {
16864 let path = path("dictionary-large-rank");
16865 let value = |row: u64| {
16866 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16867 match row % 3 {
16868 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16869 1 => format!("{mixed}"),
16870 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16871 }
16872 };
16873 let distinct = 70_000;
16874 let parts = 4 * distinct / 1000;
16875 let per_part = 1000;
16876 let mut writer =
16877 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16878 .expect("new file");
16879 for part in 0..parts {
16880 let values = (0..per_part)
16881 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
16882 .collect::<Vec<_>>();
16883 let chunk = Chunk::new(vec![
16884 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16885 ])
16886 .expect("matching rows");
16887 writer.append(&chunk).expect("a part");
16888 }
16889 writer.finish().expect("commit");
16890
16891 let reader = Reader::open(&path).expect("reopen from disk");
16892 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16893 let count = dictionary.ranks().expect("a ranked dictionary");
16894 assert_eq!(count, distinct as usize, "every distinct value has a rank");
16895 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
16896 let ranked = (0..count)
16897 .map(|rank| {
16898 let code = dictionary.code_at_rank(rank).expect("a code");
16899 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16900 })
16901 .collect::<Vec<_>>();
16902 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
16903 expected.sort();
16904 assert_eq!(ranked, expected, "rank order is value order");
16905 fs::remove_file(path).expect("remove scratch file");
16906 }
16907
16908 #[test]
16921 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
16922 let path = path("windowed-directory");
16923 let fields = vec![
16924 Field::required("id", LogicalType::BigInt),
16925 Field::required("word", LogicalType::Varchar),
16926 Field::new("score", LogicalType::Double),
16927 ];
16928 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16929 for part in 0..70_i64 {
16930 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
16931 let words = (0..100)
16932 .map(|row| Value::Varchar(format!("word {}", row % 13)))
16933 .collect::<Vec<_>>();
16934 let scores = (0..100)
16935 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
16936 .collect::<Vec<_>>();
16937 let chunk = Chunk::new(vec![
16938 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
16939 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
16940 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
16941 ])
16942 .expect("three columns");
16943 writer.append(&chunk).expect("a part");
16944 }
16945 writer.finish().expect("commit");
16946
16947 let catalog = Catalog::open(&path).expect("reopen");
16948 let entry = catalog.entries.first().expect("one table").directory;
16949 let (offset, length) = (entry.offset, entry.length as usize);
16950 let mut bytes = vec![0; length];
16951 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
16952 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
16953 let whole = decode_directory(&bytes, catalog.size).expect("whole");
16954 assert!(whole.stripes.len() > 1, "the table should span stripes");
16955 for size in [1, 7, 33, 4_096] {
16956 let mut cursor = Cursor::over(&catalog.file, offset, length);
16957 cursor.window.as_mut().expect("a window").size = size;
16958 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
16959 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
16960 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
16961 let mut stored = 0;
16962 for (column, (left, held)) in
16963 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
16964 {
16965 match (left, held) {
16966 (None, None) => {}
16967 (
16968 Some(super::Frequencies::Stored { span, values, entries }),
16969 Some(super::Frequencies::Held(summary)),
16970 ) => {
16971 let mut one = vec![0; span.length as usize];
16972 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
16973 let read = decode_summary(
16974 &mut Cursor::new(&one),
16975 &whole.fields[column],
16976 whole.rows,
16977 *values,
16978 )
16979 .expect("a valid synopsis")
16980 .expect("one is there");
16981 assert_eq!(*entries, read.entries.len());
16982 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
16983 stored += 1;
16984 }
16985 other => panic!("column {column} came back as {other:?}"),
16986 }
16987 }
16988 assert!(stored >= 2, "only {stored} synopses were left in the file");
16989 }
16990 let reader = catalog.table("items").expect("the table");
16991 assert!(reader.frequency_heads[1].get().is_none());
16992 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
16993 let first = reader.frequency_heads[1].get().expect("decoded synopsis");
16994 let clone = reader.clone();
16995 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
16996 assert!(Arc::ptr_eq(first, clone.frequency_heads[1].get().expect("same synopsis")));
16997 fs::remove_file(path).expect("remove scratch file");
16998 }
16999
17000 #[test]
17001 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
17002 let path = path("file-checksum");
17003 let bytes = (0..200_000_u32)
17004 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
17005 .collect::<Vec<_>>();
17006 fs::write(&path, &bytes).expect("scratch file");
17007 let file = File::open(&path).expect("open");
17008 for (offset, length) in [
17009 (0, 0),
17010 (3, 1),
17011 (5, 31),
17012 (0, 32),
17013 (9, 33),
17014 (1, 65_536),
17015 (7, 65_567),
17016 (0, 200_000),
17017 (11, 131_101),
17018 ] {
17019 let whole = checksum(&bytes[offset..offset + length]);
17020 assert_eq!(
17021 file_checksum(&file, offset as u64, length).expect("read"),
17022 whole,
17023 "{offset} {length}"
17024 );
17025 }
17026 fs::remove_file(path).expect("remove scratch file");
17027 }
17028
17029 #[test]
17030 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
17031 let path = path("synopsis-keeps-no-block");
17032 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
17033 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
17034 for _ in 0..3 {
17035 values.extend((0..3_000).step_by(5).map(spelled));
17036 }
17037 let mut writer =
17038 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17039 .expect("new file");
17040 for part in values.chunks(1_024) {
17041 writer
17042 .append(
17043 &Chunk::new(vec![
17044 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17045 ])
17046 .expect("one column"),
17047 )
17048 .expect("a part");
17049 }
17050 writer.finish().expect("commit");
17051
17052 let reader = Reader::open(&path).expect("reopen from disk");
17053 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17054 let resting = dictionary.footprint();
17055 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17056 assert_eq!(prefix.entries.len(), 512);
17057 for (value, count) in &prefix.entries {
17058 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
17059 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
17060 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
17061 }
17062 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
17063 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17064 assert_eq!(again.entries, prefix.entries);
17065 fs::remove_file(path).expect("remove scratch file");
17066 }
17067
17068 #[test]
17075 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
17076 let path = path("character-lengths");
17077 let spellings = (0..2_500)
17078 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
17079 .collect::<Vec<_>>();
17080 let mut writer =
17081 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17082 .expect("new file");
17083 for part in spellings.chunks(1_024) {
17084 writer
17085 .append(
17086 &Chunk::new(vec![
17087 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17088 ])
17089 .expect("one column"),
17090 )
17091 .expect("a part");
17092 }
17093 writer.finish().expect("commit");
17094
17095 let reader = Reader::open(&path).expect("reopen from disk");
17096 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17097 let resting = dictionary.footprint();
17098 let mut lens = Vec::new();
17099 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
17100 let counted = dictionary.footprint() - resting;
17101 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17102 assert!(
17103 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17104 "counting kept {counted} bytes, more than a count a value"
17105 );
17106 let expected = (0..dictionary.len())
17107 .map(|code| {
17108 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
17109 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
17110 .expect("small")
17111 })
17112 .collect::<Vec<_>>();
17113 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
17114 let mut again = Vec::new();
17115 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
17116 assert_eq!(again, lens, "the kept counts answer the second time");
17117 fs::remove_file(path).expect("remove scratch file");
17118 }
17119
17120 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
17122 let path = path(label);
17123 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
17124 let mut writer =
17125 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17126 .expect("new file");
17127 for part in values.chunks(1_024) {
17128 writer
17129 .append(
17130 &Chunk::new(vec![
17131 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17132 ])
17133 .expect("one column"),
17134 )
17135 .expect("a part");
17136 }
17137 writer.finish().expect("commit");
17138 let reader = Reader::open(&path).expect("reopen from disk");
17139 (path, reader)
17140 }
17141
17142 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
17148 let codes = (0..len)
17149 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
17150 .collect::<Vec<_>>();
17151 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
17152 (codes, valid)
17153 }
17154
17155 #[test]
17162 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
17163 let spellings = (0..2_500)
17164 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
17165 .collect::<Vec<_>>();
17166 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
17167 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17168 let (codes, valid) = scattered_rows(spellings.len());
17169 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
17170 .expect("every code is inside")
17171 .with_validity(Validity::from_run(&valid));
17172
17173 let resting = dictionary.footprint();
17174 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
17175 .expect("length reads");
17176 let counted = dictionary.footprint() - resting;
17177 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17178 assert!(
17179 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17180 "length over a vector with nulls kept {counted} bytes, more than a count a value"
17181 );
17182 let expected = (0..rows.len())
17183 .map(|row| match valid[row] {
17184 true => Value::BigInt(
17185 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
17186 ),
17187 false => Value::Null,
17188 })
17189 .collect::<Vec<_>>();
17190 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
17191 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
17192 fs::remove_file(path).expect("remove scratch file");
17193 }
17194
17195 #[test]
17205 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
17206 let spellings = (0..2_500)
17207 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
17208 .collect::<Vec<_>>();
17209 let (path, reader) = stored_spellings("string-kernels", &spellings);
17210 let page = reader.table.dictionaries[0].expect("a string column has one");
17211 let starved =
17212 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
17213 .expect("a dictionary opens whatever it may keep");
17214 let starved = Arc::new(starved);
17215 let (codes, valid) = scattered_rows(spellings.len());
17216 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
17217 .expect("every code is inside")
17218 .with_validity(Validity::from_run(&valid));
17219 let expected = |each: &dyn Fn(&str) -> String| {
17220 (0..rows.len())
17221 .map(|row| match valid[row] {
17222 true => Value::Varchar(each(&spellings[codes[row] as usize])),
17223 false => Value::Null,
17224 })
17225 .collect::<Vec<_>>()
17226 };
17227 let answers =
17228 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
17229
17230 let resting = starved.footprint();
17233 let ends = spellings.len() * size_of::<u32>();
17234 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
17235 .expect("lower reads");
17236 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
17237 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
17238
17239 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
17240 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
17241 let cut =
17242 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
17243 .expect("substring reads");
17244 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
17245 assert_eq!(answers(&cut), expected(&cut_of), "substring");
17246 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
17247
17248 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17251 .expect("upper reads");
17252 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
17253 let payload = spellings.iter().map(String::len).sum::<usize>();
17254 assert!(
17255 starved.footprint() >= resting + payload,
17256 "a visit that has dropped a column's worth of blocks keeps what it reads"
17257 );
17258 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17259 .expect("upper reads kept blocks");
17260 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
17261 fs::remove_file(path).expect("remove scratch file");
17262 }
17263
17264 #[test]
17274 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
17275 let path = path("dictionary-sweep");
17276 let spellings = (0..2_500)
17279 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17280 .collect::<Vec<_>>();
17281 let mut writer =
17282 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17283 .expect("new file");
17284 for part in spellings.chunks(1_024) {
17287 writer
17288 .append(
17289 &Chunk::new(vec![
17290 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17291 ])
17292 .expect("one column"),
17293 )
17294 .expect("stripe written");
17295 }
17296 writer.finish().expect("commit");
17297
17298 let reader = Reader::open(&path).expect("valid directory");
17299 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17300 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17301 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17302 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17303 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17304 }
17305
17306 let resting = dictionary.footprint();
17307 let sweep = || {
17308 let mut swept: Vec<Vec<u8>> = Vec::new();
17309 let mut at = 0;
17310 let mut calls = 0;
17311 while at < dictionary.len() {
17312 let stopped = dictionary
17313 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17314 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17315 swept.push(text.to_vec());
17316 Ok(())
17317 })
17318 .expect("a sweep reads");
17319 assert!(stopped > at, "a sweep moves");
17320 at = stopped;
17321 calls += 1;
17322 }
17323 assert_eq!(calls, 3, "a sweep hands over one block at a time");
17324 swept
17325 };
17326 let swept = sweep();
17327 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17328 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17329 let after = dictionary.footprint();
17330 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17331
17332 let read = (0..dictionary.len())
17333 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17334 .collect::<Vec<_>>();
17335 assert_eq!(swept, read, "a sweep answers what a point read answers");
17336 let grown = dictionary.footprint() - after;
17340 assert!(
17341 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17342 "a point read of a kept block decodes nothing, and {grown} bytes grew"
17343 );
17344 fs::remove_file(path).expect("remove scratch file");
17345 }
17346
17347 #[test]
17348 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17349 let path = path("narrow-substring-signature");
17350 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17351 let mut grams = Vec::new();
17352 for text in blocks {
17353 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17354 for gram in text.windows(4) {
17355 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17356 bits[bit / 8] |= 1 << (bit % 8);
17357 }
17358 }
17359 grams.extend(bits);
17360 }
17361 fs::write(&path, &grams).expect("scratch file");
17362 let file = File::open(&path).expect("open scratch file");
17363 let signatures = NativeGrams {
17364 start: 0,
17365 length: grams.len(),
17366 width: NARROW_GRAM_BYTES,
17367 hash: checksum(&grams),
17368 verdicts: Mutex::new(Vec::new()),
17369 };
17370 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17371 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17372 assert!(signatures.footprint() > 0, "a verdict is remembered");
17373 let again = signatures.verdicts(&file, b"google").expect("remembered");
17374 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17375
17376 let damaged = NativeGrams {
17377 hash: signatures.hash ^ 1,
17378 verdicts: Mutex::new(Vec::new()),
17379 ..signatures
17380 };
17381 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17382 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17383 fs::remove_file(path).expect("remove scratch file");
17384 }
17385
17386 #[test]
17387 fn a_damaged_substring_signature_is_checked_only_when_used() {
17388 let path = path("damaged-substring-signature");
17389 let mut writer =
17390 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17391 .expect("new file");
17392 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17393 writer
17394 .append(
17395 &Chunk::new(vec![
17396 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17397 ])
17398 .expect("one column"),
17399 )
17400 .expect("stripe written");
17401 writer.finish().expect("commit");
17402
17403 let reader = Reader::open(&path).expect("valid directory");
17404 let page = reader.table.dictionaries[0].expect("string dictionary page");
17405 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17406 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17407 .expect("last signature byte");
17408 file.write_all(&[255]).expect("damage signature");
17409 let reader = Reader::open(&path).expect("the directory is still valid");
17410 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17411 let error = dictionary
17412 .text_block_might_contain(0, b"goog")
17413 .expect_err("a used signature checks its own checksum");
17414 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17415 fs::remove_file(path).expect("remove scratch file");
17416 }
17417
17418 #[test]
17429 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17430 let path = path("dictionary-sweep-short-run");
17431 let spellings = (0..2_800)
17432 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17433 .collect::<Vec<_>>();
17434 let mut writer =
17435 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17436 .expect("new file");
17437 for part in spellings.chunks(1_024) {
17438 writer
17439 .append(
17440 &Chunk::new(vec![
17441 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17442 ])
17443 .expect("one column"),
17444 )
17445 .expect("stripe written");
17446 }
17447 writer.finish().expect("commit");
17448
17449 let reader = Reader::open(&path).expect("valid directory");
17450 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17451 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17452 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17453 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17454 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17455
17456 let mut swept: Vec<Vec<u8>> = Vec::new();
17457 let mut at = 0;
17458 while at < dictionary.len() {
17459 let stopped = dictionary
17460 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17461 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17462 swept.push(text.to_vec());
17463 Ok(())
17464 })
17465 .expect("a sweep reads");
17466 assert!(stopped > at, "a sweep moves");
17467 at = stopped;
17468 }
17469 let read = (0..dictionary.len())
17470 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17471 .collect::<Vec<_>>();
17472 assert_eq!(swept, read, "a sweep answers what a point read answers");
17473 fs::remove_file(path).expect("remove scratch file");
17474 }
17475
17476 #[test]
17485 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17486 let path = path("dictionary-unpacked-ends");
17487 let spellings = (0..2_800)
17488 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17489 .collect::<Vec<_>>();
17490 let mut writer =
17491 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17492 .expect("new file");
17493 for part in spellings.chunks(1_024) {
17494 writer
17495 .append(
17496 &Chunk::new(vec![
17497 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17498 ])
17499 .expect("one column"),
17500 )
17501 .expect("stripe written");
17502 }
17503 writer.finish().expect("commit");
17504
17505 let reader = Reader::open(&path).expect("valid directory");
17506 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17507 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17508 let wanted = (0..spellings.len())
17509 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17510 .collect::<Vec<_>>();
17511
17512 let pass = |what: &str| {
17513 for (index, value) in wanted.iter().enumerate() {
17514 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17515 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17516 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17517 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17518 }
17519 };
17520 pass("the first pass");
17521 pass("the second pass");
17522
17523 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17527 let mut whole = vec![0i64; wanted.len()];
17528 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17529 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17530 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17531 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17532 let mut through = vec![0i64; codes.len()];
17533 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17534 for (row, &code) in codes.iter().enumerate() {
17535 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17536 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17537 assert_eq!(through[row], one as i64, "row {row} a row at a time");
17538 }
17539
17540 let fresh = Reader::open(&path).expect("valid directory");
17543 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17544 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17545 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17546 let mut short = vec![0i64; few.len()];
17547 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17548 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17549 assert_eq!(short, expected, "the packed ends answer what the table answers");
17550 fs::remove_file(path).expect("remove scratch file");
17551 }
17552
17553 #[test]
17568 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17569 let spellings = (0..3_000)
17570 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17571 .collect::<Vec<_>>();
17572 let mut read = Vec::new();
17573 for layout in ["outside", "inside", "behind"] {
17574 let mut dictionary = GlobalDictionary::new();
17575 for text in &spellings {
17576 dictionary.code(text).expect("a code for every spelling");
17577 }
17578 dictionary.finish_blocks().expect("the last block encodes");
17579 let order = dictionary.ranked(None).expect("a sorted order");
17580 let laid = |from: u64| {
17582 let mut at = from;
17583 dictionary
17584 .blocks
17585 .iter()
17586 .map(|block| {
17587 let place =
17588 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17589 at += block.len() as u64;
17590 place
17591 })
17592 .collect::<Vec<_>>()
17593 };
17594 let payload = dictionary.blocks.concat();
17595 let scattered = layout != "behind";
17596 let (bytes, encoded, offset, length) = if layout == "outside" {
17597 let mut bytes = vec![0; HEADER as usize];
17598 bytes.extend_from_slice(&payload);
17599 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17600 .expect("an encoding");
17601 let offset = bytes.len() as u64;
17602 bytes.extend_from_slice(&encoded.index);
17603 bytes.extend_from_slice(&encoded.ranks);
17604 bytes.extend_from_slice(&encoded.grams);
17605 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17606 (bytes, encoded, offset, length)
17607 } else {
17608 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17611 .expect("an encoding");
17612 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17613 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17614 .expect("an encoding");
17615 let mut bytes = encoded.index.clone();
17616 bytes.extend_from_slice(&encoded.ranks);
17617 bytes.extend_from_slice(&encoded.grams);
17618 bytes.extend_from_slice(&payload);
17619 let length = bytes.len();
17620 (bytes, encoded, 0, length)
17621 };
17622 let path = path(&format!("blocks-{layout}"));
17623 fs::write(&path, &bytes).expect("the dictionary is written on its own");
17624 let file = Arc::new(File::open(&path).expect("it opens again"));
17625 let page = Page {
17626 offset,
17627 length: u32::try_from(length).expect("a test dictionary is small"),
17628 hash: checksum(&encoded.index),
17629 };
17630 let opened =
17631 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17632 .expect("a dictionary laid out either way opens");
17633 let mut swept: Vec<Vec<u8>> = Vec::new();
17634 let mut at = 0;
17635 while at < opened.len() {
17636 at = opened
17637 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17638 swept.push(text.to_vec());
17639 Ok(())
17640 })
17641 .expect("a sweep reads");
17642 }
17643 fs::remove_file(&path).expect("clean up");
17644 read.push(swept);
17645 }
17646 let wanted =
17647 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17648 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17649 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17650 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17651 }
17652
17653 #[test]
17661 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17662 let path = path("dictionary-budget");
17663 let spellings = (0..2_500)
17664 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17665 .collect::<Vec<_>>();
17666 let mut writer =
17667 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17668 .expect("new file");
17669 for part in spellings.chunks(1_024) {
17670 writer
17671 .append(
17672 &Chunk::new(vec![
17673 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17674 ])
17675 .expect("one column"),
17676 )
17677 .expect("stripe written");
17678 }
17679 writer.finish().expect("commit");
17680
17681 let reader = Reader::open(&path).expect("valid directory");
17682 let page = reader.table.dictionaries[0].expect("a string column has one");
17683 let file = Arc::clone(&reader.file);
17684 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17685 .expect("a dictionary opens whatever it may keep");
17686
17687 let resting = starved.footprint();
17688 let mut swept: Vec<Vec<u8>> = Vec::new();
17689 let mut at = 0;
17690 while at < starved.len() {
17691 at = starved
17692 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17693 swept.push(text.to_vec());
17694 Ok(())
17695 })
17696 .expect("a sweep reads");
17697 }
17698 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17699 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17700
17701 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17702 let read = (0..generous.len())
17703 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17704 .collect::<Vec<_>>();
17705 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17706 fs::remove_file(path).expect("remove scratch file");
17707 }
17708
17709 #[test]
17712 fn a_part_is_checked_once_per_open_reader() {
17713 let path = path("checked-once");
17714 let mut writer = Writer::create(
17715 &path,
17716 "items",
17717 vec![
17718 Field::required("id", LogicalType::Integer),
17719 Field::new("text", LogicalType::Varchar),
17720 ],
17721 )
17722 .expect("new file");
17723 writer.append(&sample()).expect("stripe written");
17724 writer.finish().expect("commit");
17725
17726 let reader = Reader::open(&path).expect("valid directory");
17727 let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17728 assert!(reader.is_verified(0), "the part is remembered as checked");
17729 let page = reader.table.stripes[0].pages[0];
17730 let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17731 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17732 file.write_all(&[0xa5]).expect("damage page");
17733 if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17734 assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17735 }
17736 let fresh = Reader::open(&path).expect("valid directory");
17737 let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17738 assert!(error.message().contains("column page checksum differs"), "{error}");
17739 assert_eq!(first.len(), 2);
17740 fs::remove_file(path).expect("remove scratch file");
17741 }
17742
17743 #[test]
17744 fn damaged_membership_cannot_skip_a_string_page() {
17745 let path = path("damaged-membership");
17746 let mut writer = Writer::create(
17747 &path,
17748 "items",
17749 vec![
17750 Field::required("id", LogicalType::Integer),
17751 Field::new("text", LogicalType::Varchar),
17752 ],
17753 )
17754 .expect("new file");
17755 writer.append(&sample()).expect("stripe written");
17756 writer.finish().expect("commit");
17757
17758 let reader = Reader::open(&path).expect("valid directory");
17759 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17760 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17761 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17762 file.write_all(&[255]).expect("damage membership");
17763 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17764 assert!(error.message().contains("membership page checksum differs"), "{error}");
17765 fs::remove_file(path).expect("remove scratch file");
17766 }
17767
17768 #[test]
17769 fn membership_delta_stream_is_sorted_exact_and_bounded() {
17770 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17771 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17772 let encoded = encode_membership(&unique);
17773 assert_eq!(
17774 decode_membership(&encoded).expect("valid membership"),
17775 [4, 9, 72, 900, u32::MAX]
17776 );
17777 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17780 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17781 assert_eq!(
17782 decode_membership(&encode_membership(&merged)).expect("valid membership"),
17783 unique
17784 );
17785 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17786 assert!(
17787 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17788 "a value past u32 is invalid"
17789 );
17790 }
17791
17792 #[test]
17793 fn a_global_dictionary_may_be_larger_than_one_column_page() {
17794 let dictionary = Page {
17795 offset: HEADER,
17796 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17797 hash: 0,
17798 };
17799 let table = Table {
17800 name: "items".to_owned(),
17801 fields: vec![Field::new("text", LogicalType::Varchar)],
17802 stripes: Vec::new(),
17803 rows: 0,
17804 dictionaries: vec![Some(dictionary)],
17805 dictionary_payloads: Vec::new(),
17806 demoted: Vec::new(),
17807 distincts: vec![None],
17808 frequencies: vec![None],
17809 ordinal_bounds: Vec::new(),
17810 pair_frequencies: Vec::new(),
17811 frequency_texts: Vec::new(),
17812 host_groups: None,
17813 clustering: None,
17814 constraints: Constraints::default(),
17815 generation: 1,
17816 sections: Vec::new(),
17817 };
17818 let directory = encode_directory(&table).expect("directory");
17819 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17820
17821 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17822 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17823 }
17824
17825 #[test]
17826 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17827 let path = path("constant-codes");
17828 let mut writer =
17829 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17830 .expect("new file");
17831 let empty = vec![Value::Varchar(String::new()); 1024];
17832 for _ in 0..4 {
17833 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17834 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17835 }
17836 writer.finish().expect("commit");
17837
17838 let reader = Reader::open(&path).expect("valid directory");
17839 let pages = reader.layout().columns.first().expect("one column").pages;
17840 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17844 let read = reader.read(3, &[0]).expect("the last part back");
17845 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17846 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17847 fs::remove_file(path).expect("remove scratch file");
17848 }
17849
17850 #[test]
17851 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17852 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17855 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17856 assert!(format!("{error}").contains("not of its type"), "{error}");
17857 let low = integer::encode(&[i64::MIN]).expect("a chunk");
17858 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17859 let zero = integer::encode(&[0]).expect("a chunk");
17860 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17861 }
17862
17863 #[test]
17864 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17865 let mut state: u32 = 0x9e37_79b9;
17869 let spread: Vec<u32> = (0..1024)
17870 .map(|_| {
17871 state ^= state << 13;
17872 state ^= state >> 17;
17873 state ^= state << 5;
17874 state
17875 })
17876 .collect();
17877 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17878 let near: Vec<u32> = (0..1024).collect();
17879 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
17880 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
17881 }
17882
17883 #[test]
17889 fn two_writes_of_the_same_rows_give_the_same_bytes() {
17890 fn written(path: &PathBuf) {
17891 let fields = (0..40)
17892 .map(|column| {
17893 let ty =
17894 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
17895 Field::new(format!("c{column}"), ty)
17896 })
17897 .collect::<Vec<_>>();
17898 let mut writer = Writer::create(path, "wide", fields).expect("new file");
17899 for part in 0..70_u64 {
17900 let columns = (0..40)
17901 .map(|column| {
17902 let values = (0..64_u64)
17903 .map(|row| {
17904 let seed = part.wrapping_mul(31).wrapping_add(row);
17905 if column % 4 == 0 {
17906 Value::Varchar(format!("v{}", seed % 17))
17907 } else {
17908 Value::BigInt(i64::try_from(seed % 97).expect("small"))
17909 }
17910 })
17911 .collect::<Vec<_>>();
17912 let ty = if column % 4 == 0 {
17913 LogicalType::Varchar
17914 } else {
17915 LogicalType::BigInt
17916 };
17917 Vector::from_values(ty, &values).expect("a column")
17918 })
17919 .collect::<Vec<_>>();
17920 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
17921 }
17922 writer.finish().expect("commit");
17923 }
17924
17925 let first = path("repeatable-one");
17926 let second = path("repeatable-two");
17927 written(&first);
17928 written(&second);
17929 let left = fs::read(&first).expect("the first file");
17930 let right = fs::read(&second).expect("the second file");
17931 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
17932 assert!(left == right, "two writes of the same rows differ in their bytes");
17933
17934 let reader = Reader::open(&first).expect("valid directory");
17937 assert_eq!(reader.table().rows(), 70 * 64);
17938 let read = reader.read(0, &[0, 1]).expect("the first part back");
17939 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
17940 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
17941 fs::remove_file(first).expect("remove scratch file");
17942 fs::remove_file(second).expect("remove scratch file");
17943 }
17944
17945 fn three_tables(path: &PathBuf) {
17947 let writer = Writer::create(
17948 path,
17949 "region",
17950 vec![
17951 Field::new("r_key", LogicalType::Integer),
17952 Field::new("r_name", LogicalType::Varchar),
17953 ],
17954 )
17955 .expect("new file");
17956 let mut writer = writer;
17957 writer
17958 .append(
17959 &Chunk::new(vec![
17960 Vector::from_values(
17961 LogicalType::Integer,
17962 &[Value::Integer(0), Value::Integer(1)],
17963 )
17964 .expect("keys"),
17965 Vector::from_values(
17966 LogicalType::Varchar,
17967 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
17968 )
17969 .expect("names"),
17970 ])
17971 .expect("two columns"),
17972 )
17973 .expect("a part");
17974 let mut writer = writer
17975 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
17976 .expect("a second table");
17977 writer
17978 .append(
17979 &Chunk::new(vec![
17980 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
17981 ])
17982 .expect("one column"),
17983 )
17984 .expect("a part");
17985 let mut writer =
17986 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
17987 for part in 0..70_i64 {
17988 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
17989 writer
17990 .append(
17991 &Chunk::new(vec![
17992 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
17993 ])
17994 .expect("one column"),
17995 )
17996 .expect("a part");
17997 }
17998 writer.finish().expect("commit");
17999 }
18000
18001 #[test]
18002 fn three_tables_in_one_file_read_back_by_name() {
18003 let file = path("three-tables");
18004 three_tables(&file);
18005 let catalog = Catalog::open(&file).expect("a committed catalog");
18006 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
18007
18008 let region = catalog.table("region").expect("the first table");
18009 assert_eq!(region.table().rows(), 2);
18010 assert_eq!(
18011 region.read(0, &[1]).expect("names").value_at(1, 0),
18012 Value::Varchar("ASIA".to_owned())
18013 );
18014
18015 let wide = catalog.table("wide").expect("the third table");
18016 assert_eq!(wide.table().rows(), 70 * 64);
18017 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
18018
18019 let empty = catalog.table("empty").expect("the second table");
18022 assert_eq!(empty.table().rows(), 1);
18023 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
18024
18025 fs::remove_file(file).expect("remove scratch file");
18026 }
18027
18028 #[test]
18029 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
18030 let file = path("three-tables-missing");
18031 three_tables(&file);
18032 let catalog = Catalog::open(&file).expect("a committed catalog");
18033 let error = catalog.table("nation").expect_err("no such table");
18034 assert!(error.message().contains("nation"), "{}", error.message());
18035 fs::remove_file(file).expect("remove scratch file");
18036 }
18037
18038 #[test]
18039 fn a_file_of_three_tables_will_not_open_as_one() {
18040 let file = path("three-tables-unnamed");
18041 three_tables(&file);
18042 let error = Reader::open(&file).expect_err("more than one table");
18043 assert!(error.message().contains("more than one table"), "{}", error.message());
18044 fs::remove_file(file).expect("remove scratch file");
18045 }
18046
18047 #[test]
18049 fn decimals_of_every_storage_width_round_trip() {
18050 let file = path("decimals");
18051 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
18052 let fields = widths
18053 .iter()
18054 .enumerate()
18055 .map(|(index, (width, scale))| {
18056 Field::new(
18057 format!("d{index}"),
18058 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18059 )
18060 })
18061 .collect::<Vec<_>>();
18062 let mut writer = Writer::create(&file, "money", fields).expect("new file");
18063 let rows: [i128; 3] = [-1234, 0, 999];
18064 let columns = widths
18065 .iter()
18066 .map(|(width, scale)| {
18067 let values = rows
18068 .iter()
18069 .map(|unscaled| Value::Decimal {
18070 unscaled: *unscaled,
18071 width: *width,
18072 scale: *scale,
18073 })
18074 .collect::<Vec<_>>();
18075 Vector::from_values(
18076 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18077 &values,
18078 )
18079 .expect("a decimal column")
18080 })
18081 .collect::<Vec<_>>();
18082 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
18083 writer.finish().expect("commit");
18084
18085 let reader = Reader::open(&file).expect("a committed file");
18086 for (index, (width, scale)) in widths.iter().enumerate() {
18087 assert_eq!(
18088 reader.table().fields()[index].ty,
18089 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18090 "column {index} came back as another type"
18091 );
18092 let column = reader.read(0, &[index]).expect("the column");
18093 for (row, unscaled) in rows.iter().enumerate() {
18094 assert_eq!(
18095 column.value_at(row, 0),
18096 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
18097 "column {index} row {row}"
18098 );
18099 }
18100 }
18101 fs::remove_file(file).expect("remove scratch file");
18102 }
18103
18104 #[test]
18105 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
18106 let file = path("two-of-a-name");
18107 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
18108 .expect("new file");
18109 let error = writer
18110 .next("t", vec![Field::new("a", LogicalType::BigInt)])
18111 .expect_err("the same name twice");
18112 assert!(error.message().contains("same name"), "{}", error.message());
18113 fs::remove_file(file).expect("remove scratch file");
18114 }
18115
18116 #[test]
18117 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
18118 let file = path("integer-tally");
18119 let mut writer =
18120 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
18121 .expect("new file");
18122 let mut values = vec![Value::SmallInt(0); 1024];
18123 values[7] = Value::SmallInt(3);
18124 values[99] = Value::SmallInt(-2);
18125 values[1001] = Value::SmallInt(3);
18126 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
18127 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
18128 values[0] = Value::Null;
18129 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
18130 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
18131 writer.finish().expect("commit");
18132
18133 let reader = Reader::open(&file).expect("read file");
18134 assert_eq!(
18135 reader.integer_tally(0, 0).expect("valid part"),
18136 Some(vec![(-2, 1), (0, 1021), (3, 2)])
18137 );
18138 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
18139 let catalog = Catalog::open(&file).expect("catalog");
18140 assert_eq!(
18141 catalog.integer_tally("events", 0).expect("nullable column"),
18142 Some(vec![(-2, 2), (0, 2041), (3, 4)])
18143 );
18144 fs::remove_file(file).expect("remove scratch file");
18145 }
18146
18147 #[test]
18148 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
18149 let file = path("catalog-integer-tally");
18150 let mut writer = Writer::create(
18151 &file,
18152 "events",
18153 vec![
18154 Field::new("noise", LogicalType::SmallInt),
18155 Field::new("source", LogicalType::SmallInt),
18156 ],
18157 )
18158 .expect("new file");
18159 let noise = vec![Value::SmallInt(9); 1024];
18160 let mut source = vec![Value::SmallInt(0); 1024];
18161 source[7] = Value::SmallInt(3);
18162 source[99] = Value::SmallInt(-2);
18163 let chunk = Chunk::new(vec![
18164 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
18165 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
18166 ])
18167 .expect("two columns");
18168 writer.append(&chunk).expect("append");
18169 writer.finish().expect("commit");
18170
18171 let catalog = Catalog::open(&file).expect("catalog");
18172 assert_eq!(
18173 catalog.integer_tally("events", 1).expect("selected column"),
18174 Some(vec![(-2, 1), (0, 1022), (3, 1)])
18175 );
18176 assert_eq!(
18177 catalog.integer_tally("events", 0).expect("other column"),
18178 Some(vec![(9, 1024)])
18179 );
18180 fs::remove_file(file).expect("remove scratch file");
18181 }
18182
18183 #[test]
18184 fn opening_the_catalog_reads_no_table_directory() {
18185 let file = path("catalog-only");
18186 three_tables(&file);
18187 let catalog = Catalog::open(&file).expect("a committed catalog");
18188 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
18191 assert_eq!(catalog.names().len(), 3);
18192 fs::remove_file(file).expect("remove scratch file");
18193 }
18194
18195 #[test]
18206 fn the_checksum_answers_what_it_has_always_answered() {
18207 let bytes: Vec<u8> =
18208 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
18209 for (length, expected) in [
18210 (0, 0xef46_db37_51d8_e999),
18211 (1, 0xa96c_7f0c_e858_bbb7),
18212 (3, 0x56e6_9576_32a4_87f9),
18213 (4, 0xc60d_15b1_e3ff_8f04),
18214 (5, 0x8088_1585_8624_dd4e),
18215 (7, 0xafbe_fc3d_6c6f_9a8e),
18216 (8, 0x3da5_c7aa_2696_83e0),
18217 (9, 0x465e_c429_b13c_3892),
18218 (15, 0xdee8_9d8a_065a_6233),
18219 (16, 0x1330_489a_7767_9c80),
18220 (31, 0x3391_303d_485e_846e),
18221 (32, 0x40b7_aff7_5d45_bbc8),
18222 (33, 0x4997_cae4_951c_17a5),
18223 (39, 0x5807_28fd_5c14_5739),
18224 (40, 0xf95c_f6f5_c08a_3d3b),
18225 (63, 0x2944_b4da_fc69_b206),
18226 (64, 0xbb76_f6ef_19bd_5a1b),
18227 (65, 0x814e_0c65_4a9f_d640),
18228 (127, 0x00de_aab1_31cf_f89b),
18229 (1000, 0x9e33_00c1_cde3_c58d),
18230 ] {
18231 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
18232 }
18233 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
18234 }
18235 #[test]
18242 fn a_declared_order_comes_back_out_of_the_file() {
18243 let path = path("clustered");
18244 let shipped = vec![
18245 Field::new("key", LogicalType::BigInt),
18246 Field::new("line", LogicalType::Integer),
18247 Field::new("shipdate", LogicalType::Date),
18248 ];
18249 let plain = vec![Field::new("a", LogicalType::Integer)];
18250 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
18251
18252 let mut writer = Writer::create(&path, "lineitem", shipped)
18253 .expect("new file")
18254 .declare(stage_zero.clone())
18255 .expect("the columns are the table's");
18256 let column = |ty: LogicalType, values: &[Value]| {
18257 Vector::from_values(ty, values).expect("the values match the type")
18258 };
18259 writer
18260 .append(
18261 &Chunk::new(vec![
18262 column(
18263 LogicalType::BigInt,
18264 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
18265 ),
18266 column(
18267 LogicalType::Integer,
18268 &[
18269 Value::Integer(1),
18270 Value::Integer(1),
18271 Value::Integer(1),
18272 Value::Integer(1),
18273 ],
18274 ),
18275 column(
18276 LogicalType::Date,
18277 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
18278 ),
18279 ])
18280 .expect("three columns"),
18281 )
18282 .expect("four rows");
18283 let mut writer = writer.next("nation", plain).expect("a second table");
18284 writer
18285 .append(
18286 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
18287 .expect("one column"),
18288 )
18289 .expect("one row");
18290 writer.finish().expect("commit");
18291
18292 let catalog = Catalog::open(&path).expect("reopen");
18293 let lineitem = catalog.table("lineitem").expect("the clustered table");
18294 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
18295 let nation = catalog.table("nation").expect("the plain table");
18296 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18297
18298 assert_eq!(lineitem.table().rows(), 4);
18301 assert_eq!(nation.table().rows(), 1);
18302 fs::remove_file(&path).ok();
18303 }
18304
18305 #[test]
18307 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18308 let path = path("clustered-bad");
18309 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18310 .expect("new file");
18311 let four =
18312 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18313 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18314 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18315 fs::remove_file(&path).ok();
18316 }
18317
18318 #[test]
18324 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18325 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18326 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18327 .collect::<Vec<_>>();
18328 let filled = || {
18329 let mut dictionary = GlobalDictionary::new();
18330 for value in &values {
18331 dictionary.code(value).expect("a code for every value");
18332 }
18333 dictionary.settle().expect("a shape");
18334 dictionary
18335 };
18336 let mut in_place = filled();
18337 in_place.finish_blocks().expect("every block encodes");
18338
18339 let mut handed = filled();
18340 let out = handed.hand_out(3);
18341 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18342 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18343 for job in out.iter().rev() {
18344 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18345 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18346 }
18347 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18348 handed.finish_blocks().expect("the last block encodes");
18349
18350 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18351 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18352 }
18353
18354 #[test]
18356 fn a_block_given_back_twice_is_refused() {
18357 let mut dictionary = GlobalDictionary::new();
18358 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18359 dictionary.code(&format!("value {at}")).expect("a code");
18360 }
18361 dictionary.settle().expect("a shape");
18362 let out = dictionary.hand_out(0);
18363 let last = out.last().expect("blocks went out");
18364 let at = last.place().1;
18365 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18366 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18367 }
18368
18369 #[test]
18375 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18376 let mut values = vec![String::new(), "http://".to_owned()];
18377 for host in 0..7 {
18378 for path in 0..30 {
18379 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18380 values.push(format!("http://example{host}.test/page/{path:04}"));
18381 }
18382 }
18383 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18384
18385 let mut dictionary = GlobalDictionary::new();
18386 for value in &values {
18387 dictionary.code(value).expect("a code for every value");
18388 }
18389 dictionary.finish_blocks().expect("the last block encodes");
18390 let ranked = dictionary.ranked(None).expect("a sorted order");
18391 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18392
18393 let spellings = dictionary_values(&dictionary);
18394 let seen = ranked
18395 .iter()
18396 .map(|&(_, code)| {
18397 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18398 })
18399 .collect::<Vec<_>>();
18400 let mut wanted = values.clone();
18401 wanted.sort_unstable();
18402 assert_eq!(seen, wanted, "the order is the order the bytes give");
18403
18404 for &(carried, code) in &ranked {
18405 let value = &spellings[code as usize];
18406 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18407 }
18408 }
18409
18410 #[test]
18415 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18416 let entry =
18417 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18418 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18419 .map(|code| entry(code, u64::from(code % 7) + 1))
18420 .collect::<Vec<_>>();
18421 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18422
18423 let mut sorted = all.clone();
18424 sorted.sort_unstable_by(|left, right| {
18425 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18426 });
18427 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18428 sorted.truncate(FREQUENCY_ENTRIES);
18429
18430 let mut picked = all.clone();
18431 let omitted = keep_most_frequent(&mut picked);
18432 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18433 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18434 assert!(
18435 picked
18436 .iter()
18437 .zip(&sorted)
18438 .all(|(one, two)| one.value == two.value && one.count == two.count),
18439 "the same entries in the same order"
18440 );
18441
18442 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18443 let omitted = keep_most_frequent(&mut short);
18444 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18445 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18446 }
18447
18448 #[test]
18450 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18451 let empty = GlobalDictionary::new();
18452 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18453
18454 let mut dictionary = GlobalDictionary::new();
18455 for value in ["pear", "apple", "", "apples", "app"] {
18456 dictionary.code(value).expect("a code for every value");
18457 }
18458 dictionary.finish_blocks().expect("the one block encodes");
18459 let spellings = dictionary_values(&dictionary);
18460 let seen = dictionary
18461 .ranked(None)
18462 .expect("a sorted order")
18463 .iter()
18464 .map(|&(_, code)| spellings[code as usize].clone())
18465 .collect::<Vec<_>>();
18466 let wanted: Vec<Vec<u8>> =
18467 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18468 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18469 }
18470
18471 #[test]
18474 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18475 let profile = LoadProfile::begin("demoted");
18476 let mut dictionary = GlobalDictionary::new();
18477 for value in 0..50_000 {
18478 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18479 }
18480 let (_, grown) = dictionary.recharge(Some(&profile));
18481 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18482
18483 dictionary.demote();
18484 let (before, after) = dictionary.recharge(Some(&profile));
18485 assert_eq!(before, grown);
18486 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18489 assert_eq!(profile.held(), after, "the profile was told about the drop");
18490 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18491
18492 dictionary.demote();
18493 assert_eq!(
18494 dictionary.recharge(Some(&profile)),
18495 (after, after),
18496 "demoting twice is a no-op"
18497 );
18498 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18499 }
18500}