1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, Mapped, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod anchor;
58mod distinct;
59pub mod grams;
60pub mod graph;
61pub mod host;
62mod prepare;
63mod projection;
64mod run_projection;
65use prepare::Lent;
66pub mod section;
67pub mod stats;
68mod zones;
69
70pub use anchor::{LaneStart, LogAnchor};
71pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
72pub use projection::build_sorted_projection;
73pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
74pub use section::Section;
75pub use zones::{Common, Stripes, ascending, distincts, facts, widths};
76
77const MAGIC: &[u8; 8] = b"RUDBNV10";
78const DIRECTORY: &[u8; 8] = b"RUDBDI10";
79const CATALOG: &[u8; 8] = b"RUDBCA10";
80const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
81const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
82const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
83const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
84const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
85const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
86const MAX_CATALOG_FREQUENCIES: usize = 64;
87const FORMAT: u32 = 30;
88
89const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
124
125const HEADER: u64 = 80;
126const SLOT_BYTES: usize = 28;
127const MAX_PAGE: usize = 256 * 1024 * 1024;
128const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
129const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
130const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
131const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
132const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
140const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
142const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
148const ORDINAL_BOUNDS: &[u8; 8] = b"RUDBFO1\0";
157const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
172const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
192const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
200const KEYS: &[u8; 8] = b"RUDBKY1\0";
207const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
215
216const MAX_SECTIONS: usize = 4096;
223const FREQUENCY_CANDIDATES: usize = 32_768;
224const FREQUENCY_ENTRIES: usize = 512;
225const FREQUENCY_BUILD_RANK: usize = 10;
226const FREQUENCY_ORDINALS: usize = 131_072;
227const MAX_PAIR_FREQUENCIES: usize = 1024;
228const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
233const MAX_FREQUENCY_WORKERS: usize = 32;
240
241fn close_workers() -> usize {
243 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
244}
245
246const CLOSE_BYTES: usize = 1 << 30;
257
258const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
261
262const MAX_ENCODE_WORKERS: usize = 32;
269
270const WRITEBACK_STRETCH: u64 = 32 << 20;
278
279const SIEVE_BUDGET: usize = 8 * 1024;
287
288const PART_BOUND_BYTES: usize = 24;
297
298fn io(error: std::io::Error) -> Error {
299 Error::io(error.to_string())
300}
301
302fn invalid(message: &str) -> Error {
303 Error::invalid_input(format!("invalid rudb native file: {message}"))
304}
305
306fn sum(counts: impl Iterator<Item = u64>) -> u64 {
308 counts.fold(0, u64::saturating_add)
309}
310
311fn span_bytes(spans: &[Span], at: usize) -> u64 {
313 spans.get(at).map_or(0, |span| u64::from(span.length))
314}
315
316fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
318 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
319}
320
321fn dictionary_bytes(table: &Table, at: usize) -> u64 {
323 page_bytes(&table.dictionaries, at)
324 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
325}
326
327fn checksum(bytes: &[u8]) -> u64 {
337 seeded_checksum(bytes, 0)
338}
339
340#[must_use]
347pub fn content_name(bytes: &[u8]) -> u128 {
348 let seed = u64::from(FORMAT);
349 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
350}
351
352#[derive(Debug, Clone)]
358pub struct ContentNamer {
359 seeds: [u64; 2],
360 lanes: [[u64; 4]; 2],
361 held: [u8; 32],
362 filled: usize,
363 length: u64,
364}
365
366impl Default for ContentNamer {
367 fn default() -> Self {
368 let seed = u64::from(FORMAT);
369 let seeds = [seed, !seed];
370 let lanes = seeds.map(|seed| {
371 [
372 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
373 seed.wrapping_add(XXH_P2),
374 seed,
375 seed.wrapping_sub(XXH_P1),
376 ]
377 });
378 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
379 }
380}
381
382impl ContentNamer {
383 pub fn update(&mut self, mut bytes: &[u8]) {
385 self.length += bytes.len() as u64;
386 if self.filled > 0 {
387 let take = (32 - self.filled).min(bytes.len());
388 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
389 self.filled += take;
390 bytes = &bytes[take..];
391 if self.filled < 32 {
392 return;
393 }
394 let block = self.held;
395 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
396 self.filled = 0;
397 }
398 let mut blocks = bytes.chunks_exact(32);
399 for block in blocks.by_ref() {
400 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
401 }
402 let rest = blocks.remainder();
403 self.held[..rest.len()].copy_from_slice(rest);
404 self.filled = rest.len();
405 }
406
407 #[must_use]
409 pub fn finish(&self) -> u128 {
410 let rest = &self.held[..self.filled];
411 let [first, second] = [0, 1].map(|at| {
412 if self.length < 32 {
413 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
414 } else {
415 finish_checksum(self.lanes[at], rest, self.length)
416 }
417 });
418 u128::from(first) << 64 | u128::from(second)
419 }
420}
421
422fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
431 let mut blocks = bytes.chunks_exact(32);
434 let rest = blocks.remainder();
435 if bytes.len() < 32 {
436 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
437 }
438 let mut lanes = [
439 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
440 seed.wrapping_add(XXH_P2),
441 seed,
442 seed.wrapping_sub(XXH_P1),
443 ];
444 for block in blocks.by_ref() {
445 checksum_block(&mut lanes, block);
446 }
447 finish_checksum(lanes, rest, bytes.len() as u64)
448}
449
450const XXH_P1: u64 = 11_400_714_785_074_694_791;
451const XXH_P2: u64 = 14_029_467_366_897_019_727;
452const XXH_P3: u64 = 1_609_587_929_392_839_161;
453const XXH_P4: u64 = 9_650_029_242_287_828_579;
454const XXH_P5: u64 = 2_870_177_450_012_600_261;
455
456fn checksum_round(state: u64, word: u64) -> u64 {
457 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
458}
459
460fn checksum_word(chunk: &[u8]) -> u64 {
461 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
462}
463
464fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
466 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
467 *lane = checksum_round(*lane, checksum_word(chunk));
468 }
469}
470
471fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
473 let merge = |state: u64, lane: u64| {
474 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
475 };
476 let [one, two, three, four] = lanes;
477 let combined = one
478 .rotate_left(1)
479 .wrapping_add(two.rotate_left(7))
480 .wrapping_add(three.rotate_left(12))
481 .wrapping_add(four.rotate_left(18));
482 let hash = merge(merge(merge(merge(combined, one), two), three), four);
483 checksum_tail(hash.wrapping_add(length), rest)
484}
485
486fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
488 let mut words = rest.chunks_exact(8);
489 for chunk in words.by_ref() {
490 hash ^= checksum_round(0, checksum_word(chunk));
491 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
492 }
493 rest = words.remainder();
494 if rest.len() >= 4 {
495 let (head, tail) = rest.split_at(4);
496 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
497 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
498 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
499 rest = tail;
500 }
501 for &byte in rest {
502 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
503 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
504 }
505 hash ^= hash >> 33;
506 hash = hash.wrapping_mul(XXH_P2);
507 hash ^= hash >> 29;
508 hash = hash.wrapping_mul(XXH_P3);
509 hash ^ (hash >> 32)
510}
511
512fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
518 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
519}
520
521fn walk_checksummed(
527 file: &File,
528 offset: u64,
529 length: usize,
530 window: usize,
531 mut each: impl FnMut(&[u8]) -> Result<()>,
532) -> Result<u64> {
533 debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
534 if length < 32 {
535 let mut bytes = vec![0; length];
536 read_at(file, offset, &mut bytes)?;
537 each(&bytes)?;
538 return Ok(checksum(&bytes));
539 }
540 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
541 let mut buffer = vec![0; window.min(length)];
542 let mut read = 0;
543 let (mut whole, mut filled) = (0, 0);
544 while read < length {
545 filled = buffer.len().min(length - read);
546 read_at(file, offset + read as u64, &mut buffer[..filled])?;
547 read += filled;
548 each(&buffer[..filled])?;
549 whole = filled / 32 * 32;
550 for block in buffer[..whole].chunks_exact(32) {
551 checksum_block(&mut lanes, block);
552 }
553 }
554 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
555}
556
557#[derive(Debug, Clone, Copy)]
558struct Slot {
559 offset: u64,
560 length: u32,
561 generation: u64,
562 hash: u64,
563}
564
565impl Slot {
566 fn bytes(self) -> [u8; SLOT_BYTES] {
567 let mut result = [0; SLOT_BYTES];
568 result[..8].copy_from_slice(&self.offset.to_le_bytes());
569 result[8..12].copy_from_slice(&self.length.to_le_bytes());
570 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
571 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
572 result
573 }
574
575 fn read(bytes: &[u8]) -> Self {
576 Self {
577 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
578 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
579 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
580 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
581 }
582 }
583}
584
585#[derive(Debug, Clone, Copy)]
586struct Page {
587 offset: u64,
588 length: u32,
589 hash: u64,
590}
591
592impl Page {
593 fn bytes(&self) -> u64 {
595 u64::from(self.length)
596 }
597}
598
599#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
600enum FrequencyValue {
601 Null,
602 Integer(i128),
603 Code(u32),
604}
605
606type FrequencyMap<V> = HashMap<u64, V, Spread>;
612
613#[derive(Debug)]
627struct Candidates {
628 slots: Vec<Candidate>,
631 held: usize,
632 nulls: u32,
633 decrements: u64,
634 survivors: Vec<Candidate>,
636}
637
638#[derive(Debug, Default, Clone, Copy)]
640struct Candidate {
641 bits: u64,
642 count: u32,
643}
644
645const FIRST_CANDIDATE_SLOTS: usize = 64;
647
648impl Default for Candidates {
649 fn default() -> Self {
650 Self {
651 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
652 held: 0,
653 nulls: 0,
654 decrements: 0,
655 survivors: Vec::new(),
656 }
657 }
658}
659
660impl Candidates {
661 fn add(&mut self, bits: Option<u64>, mut times: u32) {
668 while times > 0 {
669 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
670 match bits {
671 Some(bits) => {
672 let (at, found) = self.find(bits);
673 if found {
674 self.slots[at].count = self.slots[at].count.saturating_add(times);
675 return;
676 }
677 if room {
678 self.place(at, bits, times);
679 return;
680 }
681 }
682 None if self.nulls != 0 => {
683 self.nulls = self.nulls.saturating_add(times);
684 return;
685 }
686 None if room => {
687 self.nulls = times;
688 return;
689 }
690 None => {}
691 }
692 self.decrement();
693 times -= 1;
694 }
695 }
696
697 fn find(&self, bits: u64) -> (usize, bool) {
699 let mask = self.slots.len() - 1;
700 let mut at = home(bits, self.slots.len());
701 loop {
702 let slot = self.slots[at];
703 if slot.count == 0 {
704 return (at, false);
705 }
706 if slot.bits == bits {
707 return (at, true);
708 }
709 at = (at + 1) & mask;
710 }
711 }
712
713 fn position(&self, bits: u64) -> Option<usize> {
715 match self.find(bits) {
716 (at, true) => Some(at),
717 (_, false) => None,
718 }
719 }
720
721 fn place(&mut self, at: usize, bits: u64, count: u32) {
724 let at = if (self.held + 1) * 2 > self.slots.len() {
725 let wider = self.slots.len() * 2;
726 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
727 for slot in old.into_iter().filter(|slot| slot.count != 0) {
728 let (to, _) = self.find(slot.bits);
729 self.slots[to] = slot;
730 }
731 self.find(bits).0
732 } else {
733 at
734 };
735 self.slots[at] = Candidate { bits, count };
736 self.held += 1;
737 }
738
739 fn decrement(&mut self) {
741 let mut survivors = std::mem::take(&mut self.survivors);
742 survivors.clear();
743 survivors.extend(
744 self.slots
745 .iter()
746 .filter(|slot| slot.count > 1)
747 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
748 );
749 self.slots.fill(Candidate::default());
750 self.held = survivors.len();
751 for &slot in &survivors {
752 let (at, _) = self.find(slot.bits);
753 self.slots[at] = slot;
754 }
755 self.survivors = survivors;
756 self.nulls = self.nulls.saturating_sub(1);
757 self.decrements = self.decrements.saturating_add(1);
758 }
759
760 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
762 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
763 }
764}
765
766fn home(bits: u64, slots: usize) -> usize {
771 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
772}
773
774#[derive(Debug, Default)]
776struct Run {
777 bits: Option<u64>,
778 times: u32,
779}
780
781impl Run {
782 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
784 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
785 self.times += 1;
786 return None;
787 }
788 let ended = self.take();
789 self.bits = bits;
790 self.times = 1;
791 ended
792 }
793
794 fn take(&mut self) -> Option<(Option<u64>, u32)> {
796 let times = std::mem::take(&mut self.times);
797 (times != 0).then_some((self.bits, times))
798 }
799}
800
801#[derive(Debug, Default, Clone, Copy)]
803struct Spread;
804
805impl std::hash::BuildHasher for Spread {
806 type Hasher = SpreadHasher;
807
808 fn build_hasher(&self) -> SpreadHasher {
809 SpreadHasher(0)
810 }
811}
812
813#[derive(Debug)]
820struct SpreadHasher(u64);
821
822impl SpreadHasher {
823 fn mix(&mut self, word: u64) {
824 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
825 self.0 = (product as u64) ^ ((product >> 64) as u64);
826 }
827}
828
829impl std::hash::Hasher for SpreadHasher {
830 fn write(&mut self, bytes: &[u8]) {
831 for part in bytes.chunks(8) {
832 let mut word = [0; 8];
833 word[..part.len()].copy_from_slice(part);
834 self.mix(u64::from_le_bytes(word));
835 }
836 }
837
838 fn write_u32(&mut self, value: u32) {
839 self.mix(u64::from(value));
840 }
841
842 fn write_u64(&mut self, value: u64) {
843 self.mix(value);
844 }
845
846 fn write_i128(&mut self, value: i128) {
847 self.mix(value as u64);
848 self.mix((value >> 64) as u64);
849 }
850
851 fn write_isize(&mut self, value: isize) {
852 self.mix(value as u64);
853 }
854
855 fn finish(&self) -> u64 {
856 self.0
857 }
858}
859
860#[derive(Debug, Clone)]
861struct FrequencyEntry {
862 value: FrequencyValue,
863 count: u64,
864}
865
866#[derive(Debug, Clone)]
871struct FrequencySummary {
872 entries: Vec<FrequencyEntry>,
873 omitted_max: u64,
874 ordinals: Vec<u64>,
875 ordinal_entries: Vec<u16>,
876 ordinal_bound: u64,
879}
880
881#[derive(Debug, Clone)]
882struct PairFrequencyEntry {
883 first_entry: u16,
884 second: Option<u32>,
885 count: u64,
886}
887
888#[derive(Debug, Clone)]
894struct PairFrequencySummary {
895 first: u16,
896 second: u16,
897 entries: Vec<PairFrequencyEntry>,
898 omitted_max: u64,
899}
900
901type FrequencyHead = (Vec<FrequencyEntry>, u64);
903
904#[derive(Debug, Clone)]
912enum Frequencies {
913 Held(FrequencySummary),
914 Stored {
917 span: Span,
918 values: bool,
919 entries: usize,
920 },
921}
922
923#[derive(Debug, Clone)]
928pub struct FrequencyPrefix {
929 pub entries: Vec<(Value, u64)>,
931 pub omitted_max: u64,
933}
934
935#[derive(Debug, Clone, PartialEq)]
937pub struct FrequencyOccurrences {
938 pub omitted_max: u64,
940 pub ordinals: Vec<u64>,
942 pub anchors: Vec<Value>,
944 pub anchor_indices: Vec<u16>,
946}
947
948pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
950
951#[derive(Debug, Clone, Copy, Default)]
958struct Span {
959 offset: u64,
960 length: u32,
961}
962
963#[derive(Debug, Clone, Default)]
971struct Pages {
972 columns: usize,
973 held: Box<[StripePage]>,
974}
975
976#[derive(Debug, Clone, Copy)]
978struct StripePage {
979 offset: u64,
980 hash: u64,
981 length: u32,
982 column: u32,
983}
984
985impl Pages {
986 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
988 let mut held = Vec::with_capacity(slots.iter().flatten().count());
989 for (column, page) in slots.iter().enumerate() {
990 if let Some(page) = page {
991 let column =
992 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
993 held.push(StripePage {
994 offset: page.offset,
995 hash: page.hash,
996 length: page.length,
997 column,
998 });
999 }
1000 }
1001 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1002 }
1003
1004 fn get(&self, column: usize) -> Option<Page> {
1006 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1007 let placed = self.held[at];
1008 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1009 }
1010
1011 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1013 (0..self.columns).map(|column| self.get(column))
1014 }
1015
1016 fn bytes(&self, column: usize) -> u64 {
1018 self.get(column).map_or(0, |page| page.bytes())
1019 }
1020}
1021
1022#[derive(Debug, Clone)]
1024pub struct Stripe {
1025 rows: usize,
1026 parts: Vec<u32>,
1029 index: Span,
1033 pages: Vec<Span>,
1034 memberships: Pages,
1035 sieves: Pages,
1038 part_ranges: Pages,
1049 zone: Zone,
1050}
1051
1052impl Stripe {
1053 #[must_use]
1055 pub fn rows(&self) -> usize {
1056 self.rows
1057 }
1058
1059 #[must_use]
1061 pub fn parts(&self) -> usize {
1062 self.parts.len()
1063 }
1064
1065 #[must_use]
1071 pub fn zone(&self) -> &Zone {
1072 &self.zone
1073 }
1074}
1075
1076#[derive(Debug, Clone)]
1078pub struct Table {
1079 name: String,
1080 fields: Vec<Field>,
1081 stripes: Vec<Stripe>,
1082 rows: usize,
1083 dictionaries: Vec<Option<Page>>,
1084 dictionary_payloads: Vec<u64>,
1090 demoted: Vec<bool>,
1096 frequencies: Vec<Option<Frequencies>>,
1097 ordinal_bounds: Vec<u64>,
1100 pair_frequencies: Vec<PairFrequencySummary>,
1101 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1106 host_groups: Option<host::HostSummary>,
1108 distincts: Vec<Option<u64>>,
1118 clustering: Option<Clustering>,
1126 generation: u64,
1140 sections: Vec<Section>,
1147 constraints: Constraints,
1150}
1151
1152#[derive(Debug, Clone, Default, PartialEq, Eq)]
1157pub struct Constraints {
1158 pub keys: Vec<(Vec<u16>, bool)>,
1160 pub foreign: Vec<StoredForeign>,
1162}
1163
1164impl Constraints {
1165 #[must_use]
1167 pub fn is_empty(&self) -> bool {
1168 self.keys.is_empty() && self.foreign.is_empty()
1169 }
1170}
1171
1172#[derive(Debug, Clone, PartialEq, Eq)]
1174pub struct StoredForeign {
1175 pub columns: Vec<u16>,
1177 pub table: String,
1179 pub referenced: Vec<u16>,
1181}
1182
1183impl Table {
1184 #[must_use]
1186 pub fn name(&self) -> &str {
1187 &self.name
1188 }
1189
1190 #[must_use]
1192 pub fn fields(&self) -> &[Field] {
1193 &self.fields
1194 }
1195
1196 #[must_use]
1198 pub fn rows(&self) -> usize {
1199 self.rows
1200 }
1201
1202 #[must_use]
1204 pub fn stripes(&self) -> &[Stripe] {
1205 &self.stripes
1206 }
1207
1208 #[must_use]
1210 pub fn clustering(&self) -> Option<&Clustering> {
1211 self.clustering.as_ref()
1212 }
1213
1214 #[must_use]
1216 pub fn constraints(&self) -> &Constraints {
1217 &self.constraints
1218 }
1219
1220 #[must_use]
1225 pub fn generation(&self) -> u64 {
1226 self.generation
1227 }
1228
1229 #[must_use]
1236 pub fn sections(&self) -> &[Section] {
1237 &self.sections
1238 }
1239}
1240
1241#[derive(Debug, Clone)]
1253struct Entry {
1254 name: String,
1255 fields: Vec<Field>,
1256 rows: usize,
1257 directory: Page,
1259 nonzero: Vec<Option<u64>>,
1262 aggregates: Vec<Option<(i128, u64)>>,
1264 distincts: Vec<Option<u64>>,
1266 extremes: Vec<StoredIntegerExtremes>,
1268 frequencies: Vec<StoredNumericFrequencies>,
1270}
1271
1272type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1273type StoredNumericFrequencies = Option<NumericFrequencies>;
1274
1275#[derive(Debug, Clone, PartialEq, Eq)]
1288pub struct ViewEntry {
1289 pub name: String,
1291 pub sql: String,
1293 pub statement: String,
1295 pub aliases: Vec<String>,
1297 pub columns: Vec<Field>,
1299}
1300
1301#[derive(Debug, Clone)]
1303pub struct ColumnLayout {
1304 pub name: String,
1306 pub kind: String,
1308 pub pages: u64,
1310 pub memberships: u64,
1312 pub sieves: u64,
1314 pub part_ranges: u64,
1316 pub dictionary: u64,
1318}
1319
1320impl ColumnLayout {
1321 #[must_use]
1323 pub fn total(&self) -> u64 {
1324 self.pages
1325 .saturating_add(self.memberships)
1326 .saturating_add(self.sieves)
1327 .saturating_add(self.part_ranges)
1328 .saturating_add(self.dictionary)
1329 }
1330}
1331
1332#[derive(Debug, Clone)]
1343pub struct Layout {
1344 pub file: u64,
1346 pub rows: usize,
1348 pub stripes: usize,
1350 pub parts: usize,
1352 pub columns: Vec<ColumnLayout>,
1354 pub indexes: u64,
1357 pub directory: u64,
1359 pub header: u64,
1361}
1362
1363impl Layout {
1364 #[must_use]
1366 pub fn columns_total(&self) -> u64 {
1367 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1368 }
1369
1370 #[must_use]
1376 pub fn unaccounted(&self) -> u64 {
1377 self.file
1378 .saturating_sub(self.columns_total())
1379 .saturating_sub(self.indexes)
1380 .saturating_sub(self.directory)
1381 .saturating_sub(self.header)
1382 }
1383}
1384
1385#[derive(Debug, Clone)]
1396pub struct StoredPart {
1397 pub stripe: usize,
1399 pub part: usize,
1401 pub row: usize,
1403 pub rows: usize,
1405 pub encoding: String,
1407 pub bytes: u64,
1409 pub page: u64,
1411 pub offset: u64,
1413 pub low: Option<Value>,
1415 pub high: Option<Value>,
1417 pub nulls: Option<usize>,
1419}
1420
1421const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1428
1429#[derive(Debug)]
1454struct GlobalDictionary {
1455 primary: HashMap<u64, u32, Spread>,
1459 collisions: HashMap<u64, Vec<u32>, Spread>,
1460 checks: Vec<u64>,
1462 ends: Vec<u32>,
1464 counts: Vec<u64>,
1465 nulls: u64,
1466 filling: Vec<u8>,
1468 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1474 waiting: Vec<(usize, Vec<u8>)>,
1479 sample: Vec<(usize, Vec<u8>)>,
1485 stride: usize,
1487 shape: Option<chooser::Settled>,
1489 settled: usize,
1491 blocks: Vec<Vec<u8>>,
1496 early: BTreeMap<usize, EncodedBlock>,
1502 placed: Vec<Placed>,
1504 charged: u64,
1507 demoted: bool,
1509}
1510
1511#[derive(Debug, Clone, Copy)]
1513struct Placed {
1514 start: u64,
1515 length: u64,
1516 hash: u64,
1517}
1518
1519type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1521
1522impl GlobalDictionary {
1523 fn new() -> Self {
1524 Self {
1525 primary: HashMap::default(),
1526 collisions: HashMap::default(),
1527 checks: Vec::new(),
1528 ends: Vec::new(),
1529 counts: Vec::new(),
1530 nulls: 0,
1531 filling: Vec::new(),
1532 grams: Vec::new(),
1533 waiting: Vec::new(),
1534 sample: Vec::new(),
1535 stride: 1,
1536 shape: None,
1537 settled: 0,
1538 blocks: Vec::new(),
1539 early: BTreeMap::new(),
1540 placed: Vec::new(),
1541 charged: 0,
1542 demoted: false,
1543 }
1544 }
1545
1546 fn values(&self) -> usize {
1548 self.ends.len()
1549 }
1550
1551 fn closing_bytes(&self) -> usize {
1554 let values = self.values();
1555 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1556 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1557 .sum::<usize>();
1558 let beside = size_of::<u32>().max(size_of::<(u64, Option<u32>)>());
1562 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + beside))
1563 }
1564
1565 fn held_bytes(&self) -> u64 {
1571 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1572 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1573 }
1574 fn spilled<T>(values: &Vec<T>) -> usize {
1575 values.capacity() * size_of::<T>()
1576 }
1577 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1578 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1579 };
1580 let bytes = table(&self.primary)
1581 + table(&self.collisions)
1582 + self.collisions.values().map(spilled).sum::<usize>()
1583 + spilled(&self.checks)
1584 + spilled(&self.ends)
1585 + spilled(&self.counts)
1586 + self.filling.capacity()
1587 + spilled(&self.grams)
1588 + raw(&self.waiting)
1589 + raw(&self.sample)
1590 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1591 + spilled(&self.placed);
1592 bytes as u64
1593 }
1594
1595 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1598 let before = self.charged;
1599 let now = self.held_bytes();
1600 if let Some(profile) = profile {
1601 if now >= before {
1602 profile.hold(now - before);
1603 } else {
1604 profile.release(before - now);
1605 }
1606 }
1607 self.charged = now;
1608 (before, now)
1609 }
1610
1611 fn demote(&mut self) {
1619 if self.demoted {
1620 return;
1621 }
1622 self.seal_rest();
1623 self.release_lookup();
1624 self.demoted = true;
1625 }
1626
1627 fn release_lookup(&mut self) {
1634 self.primary = HashMap::default();
1635 self.collisions = HashMap::default();
1636 self.checks = Vec::new();
1637 self.sample = Vec::new();
1638 self.filling = Vec::new();
1639 }
1640
1641 fn encoded(&self) -> usize {
1643 self.placed.len() + self.blocks.len()
1644 }
1645
1646 #[cfg(test)]
1647 fn code(&mut self, text: &str) -> Result<u32> {
1648 let bytes = text.as_bytes();
1649 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1650 }
1651
1652 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1658 if let Some(&code) = self.primary.get(&hash) {
1659 if self.checks.get(code as usize) == Some(&check) {
1660 return Ok(code);
1661 }
1662 if let Some(codes) = self.collisions.get(&hash)
1663 && let Some(code) =
1664 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1665 {
1666 return Ok(code);
1667 }
1668 let code = self.insert(text, check)?;
1669 self.collisions.entry(hash).or_default().push(code);
1670 return Ok(code);
1671 }
1672 let code = self.insert(text, check)?;
1673 self.primary.insert(hash, code);
1674 Ok(code)
1675 }
1676
1677 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1678 if self.demoted {
1679 return Err(Error::internal("a value was coded against a demoted dictionary"));
1680 }
1681 let code = u32::try_from(self.ends.len())
1682 .map_err(|_| invalid("global dictionary has too many values"))?;
1683 self.filling.extend_from_slice(text);
1684 self.ends.push(
1685 u32::try_from(self.filling.len())
1686 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1687 );
1688 self.checks.push(check);
1689 self.counts.push(0);
1690 if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1691 self.seal();
1692 }
1693 Ok(code)
1694 }
1695
1696 fn seal(&mut self) {
1702 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1703 let bytes = std::mem::take(&mut self.filling);
1704 if at.is_multiple_of(self.stride) {
1705 self.sample.push((at, bytes.clone()));
1706 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1707 self.stride *= 2;
1708 let stride = self.stride;
1709 self.sample.retain(|(at, _)| at % stride == 0);
1710 }
1711 }
1712 self.waiting.push((at, bytes));
1713 }
1714
1715 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1717 block_values(self.block_ends(at), bytes)
1718 }
1719
1720 fn block_ends(&self, at: usize) -> &[u32] {
1722 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1723 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1724 &self.ends[first..last]
1725 }
1726
1727 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1734 let Some(shape) = &self.shape else { return Vec::new() };
1735 let waiting = std::mem::take(&mut self.waiting);
1736 waiting
1737 .into_iter()
1738 .map(|(at, bytes)| Unencoded {
1739 column,
1740 at,
1741 ends: self.block_ends(at).to_vec(),
1742 bytes,
1743 shape: shape.clone(),
1744 })
1745 .collect()
1746 }
1747
1748 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1751 if at < self.encoded() || self.early.insert(at, block).is_some() {
1752 return Err(Error::internal("a dictionary block came back twice"));
1753 }
1754 while let Some(block) = self.early.remove(&self.encoded()) {
1755 self.push_block(block);
1756 }
1757 Ok(())
1758 }
1759
1760 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1762 self.blocks.push(bytes);
1763 self.grams.push(*grams);
1764 }
1765
1766 fn settle(&mut self) -> Result<()> {
1774 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1775 return Ok(());
1776 }
1777 self.settle_on_sample()
1778 }
1779
1780 fn settle_rest(&mut self) -> Result<()> {
1788 if self.shape.is_some() || self.sample.is_empty() {
1789 return Ok(());
1790 }
1791 self.settle_on_sample()
1792 }
1793
1794 fn settle_on_sample(&mut self) -> Result<()> {
1795 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1796 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1797 return Ok(());
1798 }
1799 let sample =
1800 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1801 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1802 self.settled = complete;
1803 Ok(())
1804 }
1805
1806 fn seal_rest(&mut self) {
1808 if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1812 self.seal();
1813 }
1814 }
1815
1816 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1819 let (block, bytes) = &self.waiting[at];
1820 let values = self.slices(*block, bytes);
1821 let encoded = match &self.shape {
1822 Some(shape) => string::encode_with(&values, shape)?,
1823 None => string::encode(&values)?,
1824 };
1825 Ok((encoded, block_grams(&values)))
1826 }
1827
1828 #[cfg(test)]
1830 fn finish_blocks(&mut self) -> Result<()> {
1831 self.seal_rest();
1832 let made = (0..self.waiting.len())
1833 .map(|at| self.encode_waiting(at))
1834 .collect::<Result<Vec<_>>>()?;
1835 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1836 if self.encoded() != at {
1837 return Err(Error::internal("a dictionary block was encoded out of order"));
1838 }
1839 self.push_block(block);
1840 }
1841 Ok(())
1842 }
1843
1844 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1862 let count = self.placed.len() + self.blocks.len();
1863 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1864 return Err(invalid("global dictionary blocks do not cover its values"));
1865 }
1866 let mut bases = Vec::with_capacity(count);
1867 let mut total = 0_usize;
1868 for block in 0..count {
1869 bases.push(total as u64);
1870 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1871 total = total
1872 .checked_add(self.ends[last] as usize)
1873 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1874 }
1875 let mut flat = vec![0_u8; total];
1876 let mut outs = Vec::with_capacity(count);
1877 let mut rest = flat.as_mut_slice();
1878 for block in 0..count {
1879 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1880 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1881 outs.push((block, out));
1882 rest = after;
1883 }
1884 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1885 let mut stored = Vec::new();
1886 for (block, out) in run {
1887 let encoded = match self.placed.get(*block) {
1888 Some(place) => {
1889 let file = file.ok_or_else(|| {
1890 Error::internal("a written dictionary block has no file")
1891 })?;
1892 let length = usize::try_from(place.length).map_err(|_| {
1893 invalid("global dictionary block does not fit in memory")
1894 })?;
1895 stored.resize(length, 0);
1896 read_at(file, place.start, &mut stored)?;
1897 if checksum(&stored) != place.hash {
1898 return Err(invalid(
1899 "a global dictionary block did not read back as written",
1900 ));
1901 }
1902 stored.as_slice()
1903 }
1904 None => &self.blocks[*block - self.placed.len()],
1905 };
1906 let decoded = string::decode_flat(encoded)?;
1907 if decoded.bytes().len() != out.len() {
1908 return Err(invalid(
1909 "a global dictionary block is not the length its ends say",
1910 ));
1911 }
1912 out.copy_from_slice(decoded.bytes());
1913 }
1914 Ok(())
1915 };
1916 let workers = close_workers().min(count / 16).max(1);
1919 if workers <= 1 {
1920 one(&mut outs)?;
1921 } else {
1922 let per = count.div_ceil(workers);
1923 std::thread::scope(|scope| {
1924 outs.chunks_mut(per)
1925 .map(|run| scope.spawn(|| one(run)))
1926 .collect::<Vec<_>>()
1927 .into_iter()
1928 .try_for_each(|handle| {
1929 handle.join().map_err(|_| {
1930 Error::internal("a global dictionary decode worker panicked")
1931 })?
1932 })
1933 })?;
1934 }
1935 drop(outs);
1936 Ok((flat, bases))
1937 }
1938
1939 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1944 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1945 let Some(&end) = ends.get(code) else { return (0, 0) };
1946 let base = base as usize;
1947 let from =
1948 if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1949 (base + from, base + end as usize)
1950 }
1951
1952 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1972 let (flat, bases) = self.decoded(file)?;
1973 let value = |code: u32| {
1974 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1975 flat.get(from..to).unwrap_or_default()
1976 };
1977 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1978 sort_by_value_across(&mut codes, value, close_workers());
1979 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1980 Ok((order, flat, bases))
1981 }
1982
1983 #[cfg(test)]
1984 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1985 self.ranked_with_values(file).map(|(order, _, _)| order)
1986 }
1987}
1988
1989#[derive(Debug)]
1997pub struct Writer {
1998 file: Box<dyn rudb_io::File>,
2001 at: u64,
2009 written_back: u64,
2011 table: Table,
2012 generation: u64,
2013 order: Vec<((u64, u64), (u64, u64))>,
2016 next_order: u64,
2017 dictionaries: Vec<Option<GlobalDictionary>>,
2018 coded: Arc<prepare::Coding>,
2021 gathers: Vec<Option<stats::Gather>>,
2027 lent: Option<Arc<Lent>>,
2030 pending: Vec<PendingChunk>,
2031 closed: Vec<Entry>,
2033 views: Vec<ViewEntry>,
2038 card: Option<KeptCard>,
2040 anchor: Option<LogAnchor>,
2043 profile: Option<Arc<LoadProfile>>,
2049}
2050
2051#[derive(Debug)]
2059struct PendingChunk {
2060 order: (u64, u64),
2061 chunk: Chunk,
2062}
2063
2064#[derive(Debug, Clone, Copy)]
2070struct Part {
2071 order: (u64, u64),
2072 rows: usize,
2073 footprint: usize,
2074}
2075
2076impl Part {
2077 fn of(pending: &PendingChunk) -> Self {
2078 Self {
2079 order: pending.order,
2080 rows: pending.chunk.len(),
2081 footprint: pending.chunk.footprint(),
2082 }
2083 }
2084}
2085
2086#[derive(Debug, Default)]
2092struct ColumnStripe {
2093 pages: Vec<Vec<u8>>,
2094 sums: Vec<u64>,
2097 codes: Vec<Option<Vec<u32>>>,
2098 sieves: Vec<Option<Sieve>>,
2099 ranges: Vec<Range>,
2100}
2101
2102fn coded_type(ty: &LogicalType) -> bool {
2110 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2111}
2112
2113fn dictionary_tag(ty: &LogicalType) -> u8 {
2120 if ty == &LogicalType::Blob { 2 } else { 1 }
2121}
2122
2123fn weight(ty: &LogicalType) -> usize {
2131 match ty {
2132 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2133 LogicalType::HugeInt
2134 | LogicalType::UHugeInt
2135 | LogicalType::Uuid
2136 | LogicalType::Interval => 16,
2137 LogicalType::BigInt
2138 | LogicalType::UBigInt
2139 | LogicalType::Timestamp
2140 | LogicalType::Time
2141 | LogicalType::TimeTz
2142 | LogicalType::TimestampTz
2143 | LogicalType::TimestampS
2144 | LogicalType::TimestampMs
2145 | LogicalType::TimestampNs
2146 | LogicalType::Double
2147 | LogicalType::Decimal { .. } => 8,
2148 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2149 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2150 _ => 1,
2151 }
2152}
2153
2154pub const STRIPE_PARTS: usize = 64;
2161
2162const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2170
2171const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2187
2188const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2190
2191fn index_section(parts: usize) -> Result<usize> {
2193 parts
2194 .checked_mul(INDEX_ENTRY)
2195 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2196 .ok_or_else(|| invalid("index page length overflow"))
2197}
2198
2199impl Writer {
2200 pub fn open(
2219 path: impl AsRef<Path>,
2220 name: impl Into<String>,
2221 fields: Vec<Field>,
2222 ) -> Result<Self> {
2223 Self::open_in(&RealFilesystem::new(), path, name, fields)
2224 }
2225
2226 pub fn open_in(
2233 fs: &dyn Filesystem,
2234 path: impl AsRef<Path>,
2235 name: impl Into<String>,
2236 fields: Vec<Field>,
2237 ) -> Result<Self> {
2238 for field in &fields {
2239 type_tag(&field.ty)?;
2240 }
2241 let name = name.into();
2242 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2243 let size = file.len()?;
2244 let (slot, bytes, _) = committed_slot(&*file, size)?;
2245 let (mut closed, views, card, anchor) = decode_catalog(&bytes, size)?;
2246 let card = card_for(path.as_ref(), card);
2247 if let Some(at) = closed.iter().position(|held| held.name == name) {
2258 if closed[at].rows > 0 {
2259 return Err(invalid("two tables in one native file have the same name"));
2260 }
2261 closed.remove(at);
2262 }
2263 let generation = slot
2268 .generation
2269 .checked_add(1)
2270 .ok_or_else(|| invalid("native file generation overflow"))?;
2271 Ok(Self {
2272 file,
2273 at: size,
2276 written_back: size,
2277 dictionaries: fields
2278 .iter()
2279 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2280 .collect(),
2281 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2282 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2283 lent: None,
2284 table: Table {
2285 name,
2286 dictionaries: vec![None; fields.len()],
2287 dictionary_payloads: Vec::new(),
2288 demoted: Vec::new(),
2289 distincts: vec![None; fields.len()],
2290 fields,
2291 stripes: Vec::new(),
2292 rows: 0,
2293 frequencies: Vec::new(),
2294 ordinal_bounds: Vec::new(),
2295 pair_frequencies: Vec::new(),
2296 frequency_texts: Vec::new(),
2297 host_groups: None,
2298 clustering: None,
2299 constraints: Constraints::default(),
2300 generation,
2301 sections: Vec::new(),
2302 },
2303 generation,
2304 order: Vec::new(),
2305 next_order: 0,
2306 pending: Vec::with_capacity(STRIPE_PARTS),
2307 closed,
2308 views,
2309 card,
2310 anchor,
2311 profile: None,
2312 })
2313 }
2314
2315 pub fn create(
2321 path: impl AsRef<Path>,
2322 name: impl Into<String>,
2323 fields: Vec<Field>,
2324 ) -> Result<Self> {
2325 Self::create_in(&RealFilesystem::new(), path, name, fields)
2326 }
2327
2328 pub fn create_in(
2338 fs: &dyn Filesystem,
2339 path: impl AsRef<Path>,
2340 name: impl Into<String>,
2341 fields: Vec<Field>,
2342 ) -> Result<Self> {
2343 for field in &fields {
2344 type_tag(&field.ty)?;
2345 }
2346 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2347 let mut header = [0; HEADER as usize];
2348 header[..8].copy_from_slice(MAGIC);
2349 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2350 file.write_at(0, &header)?;
2351 Ok(Self {
2352 file,
2353 at: HEADER,
2354 written_back: HEADER,
2355 dictionaries: fields
2356 .iter()
2357 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2358 .collect(),
2359 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2360 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2361 lent: None,
2362 table: Table {
2363 name: name.into(),
2364 dictionaries: vec![None; fields.len()],
2365 dictionary_payloads: Vec::new(),
2366 demoted: Vec::new(),
2367 distincts: vec![None; fields.len()],
2368 fields,
2369 stripes: Vec::new(),
2370 rows: 0,
2371 frequencies: Vec::new(),
2372 ordinal_bounds: Vec::new(),
2373 pair_frequencies: Vec::new(),
2374 frequency_texts: Vec::new(),
2375 host_groups: None,
2376 clustering: None,
2377 constraints: Constraints::default(),
2378 generation: 1,
2379 sections: Vec::new(),
2380 },
2381 generation: 1,
2382 order: Vec::new(),
2383 next_order: 0,
2384 pending: Vec::with_capacity(STRIPE_PARTS),
2385 closed: Vec::new(),
2386 views: Vec::new(),
2387 card: card_for(path.as_ref(), None),
2388 anchor: None,
2389 profile: None,
2390 })
2391 }
2392
2393 pub fn empty(
2417 path: impl AsRef<Path>,
2418 views: &[ViewEntry],
2419 anchor: Option<&LogAnchor>,
2420 ) -> Result<()> {
2421 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2422 let mut header = [0; HEADER as usize];
2423 header[..8].copy_from_slice(MAGIC);
2424 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2425 file.write_at(0, &header)?;
2426 let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref(), anchor)?;
2427 file.write_at(HEADER, &catalog)?;
2428 file.sync()?;
2432 let slot = Slot {
2433 offset: HEADER,
2434 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2435 generation: 1,
2436 hash: checksum(&catalog),
2437 };
2438 file.write_at(slot_offset(1), &slot.bytes())?;
2439 file.sync()?;
2440 Ok(())
2441 }
2442
2443 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2454 for field in &fields {
2455 type_tag(&field.ty)?;
2456 }
2457 let name = name.into();
2458 let entry = self.close()?;
2459 if entry.name == name {
2460 return Err(invalid("two tables in one native file have the same name"));
2461 }
2462 if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2466 if self.closed[at].rows > 0 {
2467 return Err(invalid("two tables in one native file have the same name"));
2468 }
2469 self.closed.remove(at);
2470 }
2471 let Self { file, at, generation, mut closed, views, card, anchor, .. } = self;
2472 closed.push(entry);
2473 Ok(Self {
2474 file,
2475 written_back: at,
2476 at,
2477 generation,
2478 closed,
2479 views,
2480 card,
2481 anchor,
2482 profile: None,
2483 dictionaries: fields
2484 .iter()
2485 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2486 .collect(),
2487 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2488 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2489 lent: None,
2490 table: Table {
2491 name,
2492 dictionaries: vec![None; fields.len()],
2493 dictionary_payloads: Vec::new(),
2494 demoted: Vec::new(),
2495 distincts: vec![None; fields.len()],
2496 fields,
2497 stripes: Vec::new(),
2498 rows: 0,
2499 frequencies: Vec::new(),
2500 ordinal_bounds: Vec::new(),
2501 pair_frequencies: Vec::new(),
2502 frequency_texts: Vec::new(),
2503 host_groups: None,
2504 clustering: None,
2505 constraints: Constraints::default(),
2506 generation,
2507 sections: Vec::new(),
2508 },
2509 order: Vec::new(),
2510 next_order: 0,
2511 pending: Vec::with_capacity(STRIPE_PARTS),
2512 })
2513 }
2514
2515 #[must_use]
2525 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2526 self.views = views;
2527 self
2528 }
2529
2530 #[must_use]
2533 pub fn with_log_anchor(mut self, anchor: LogAnchor) -> Self {
2534 self.anchor = Some(anchor);
2535 self
2536 }
2537
2538 #[must_use]
2544 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2545 self.profile = Some(profile);
2546 self
2547 }
2548
2549 #[must_use]
2553 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2554 self.coded.cap(bytes);
2555 self
2556 }
2557
2558 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2573 self.table.clustering = Some(Clustering::new(
2576 clustering.columns().to_vec(),
2577 clustering.width(),
2578 &self.table.fields,
2579 )?);
2580 Ok(self)
2581 }
2582
2583 pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2591 let width = self.table.fields.len();
2592 let fits = |columns: &[u16]| {
2593 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2594 };
2595 if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2596 || !constraints.foreign.iter().all(|foreign| {
2597 fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2598 })
2599 {
2600 return Err(invalid("a constraint names a column the table does not have"));
2601 }
2602 self.table.constraints = constraints;
2603 Ok(self)
2604 }
2605
2606 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2611 self.file.write_at(self.at, bytes)?;
2612 self.at = self
2613 .at
2614 .checked_add(bytes.len() as u64)
2615 .ok_or_else(|| invalid("native file length overflow"))?;
2616 if self.at - self.written_back >= WRITEBACK_STRETCH {
2617 self.file.start_writeback(self.written_back, self.at - self.written_back);
2618 self.written_back = self.at;
2619 }
2620 Ok(())
2621 }
2622
2623 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2629 let order = (self.next_order, 0);
2630 self.next_order = self.next_order.saturating_add(1);
2631 self.append_at(order, chunk)
2632 }
2633
2634 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2645 if chunk.is_empty() {
2646 return Ok(());
2647 }
2648 self.admit(chunk)?;
2649 if self.pending.last().is_some_and(|last| last.order > order) {
2650 self.flush_pending()?;
2651 }
2652 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2657 if self.pending.len() == STRIPE_PARTS {
2658 self.flush_pending()?;
2659 }
2660 Ok(())
2661 }
2662
2663 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2679 if parts.len() > STRIPE_PARTS {
2680 return Err(invalid("a stripe was handed more parts than it holds"));
2681 }
2682 self.flush_pending()?;
2685 for (order, chunk) in parts {
2686 if chunk.is_empty() {
2687 continue;
2688 }
2689 self.admit(&chunk)?;
2690 self.pending.push(PendingChunk { order, chunk });
2691 }
2692 self.flush_pending()
2693 }
2694
2695 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2697 if chunk.width() != self.table.fields.len() {
2698 return Err(invalid("chunk width differs from table schema"));
2699 }
2700 for (index, field) in self.table.fields.iter().enumerate() {
2701 if chunk.column(index)?.logical_type() != &field.ty {
2702 return Err(invalid("chunk type differs from table schema"));
2703 }
2704 }
2705 self.table.rows = self
2706 .table
2707 .rows
2708 .checked_add(chunk.len())
2709 .ok_or_else(|| invalid("row count overflow"))?;
2710 Ok(())
2711 }
2712
2713 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2715 let mut stripe = ColumnStripe {
2716 pages: Vec::with_capacity(columns.len()),
2717 sums: Vec::with_capacity(columns.len()),
2718 codes: Vec::with_capacity(columns.len()),
2719 sieves: Vec::with_capacity(columns.len()),
2720 ranges: Vec::with_capacity(columns.len()),
2721 };
2722 let mut settling = Settling::default();
2723 for &column in columns {
2724 Self::encode_page(&mut stripe, &mut settling, column)?;
2725 }
2726 Ok(stripe)
2727 }
2728
2729 fn encode_page(
2732 stripe: &mut ColumnStripe,
2733 settling: &mut Settling,
2734 column: &Vector,
2735 ) -> Result<()> {
2736 let bytes = encode(column, settling)?;
2737 if bytes.len() > MAX_PAGE {
2738 return Err(invalid("column page exceeds the configured bound"));
2739 }
2740 let range = Range::of(column);
2743 let sieve =
2754 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2755 stripe.sums.push(checksum(&bytes));
2756 stripe.pages.push(bytes);
2757 stripe.codes.push(None);
2758 stripe.sieves.push(sieve);
2759 stripe.ranges.push(range);
2760 Ok(())
2761 }
2762
2763 fn place_blocks(&mut self) -> Result<()> {
2768 if let Some(lent) = self.lent.clone() {
2769 return self.place_lent_blocks(&lent);
2770 }
2771 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2772 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2773 for block in std::mem::take(&mut dictionary.blocks) {
2774 let start = self.at;
2775 self.put(&block)?;
2776 dictionary.placed.push(Placed {
2777 start,
2778 length: block.len() as u64,
2779 hash: checksum(&block),
2780 });
2781 }
2782 Ok(())
2783 });
2784 self.dictionaries = dictionaries;
2785 placed
2786 }
2787
2788 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2794 for column in lent.columns() {
2795 let Ok(mut held) = column.try_lock() else { continue };
2796 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2797 for block in std::mem::take(&mut dictionary.blocks) {
2798 let start = self.at;
2799 self.put(&block)?;
2800 dictionary.placed.push(Placed {
2801 start,
2802 length: block.len() as u64,
2803 hash: checksum(&block),
2804 });
2805 }
2806 }
2807 Ok(())
2808 }
2809
2810 fn reclaim(&mut self) -> Result<()> {
2814 let Some(lent) = self.lent.take() else { return Ok(()) };
2815 let (dictionaries, gathers) = lent.reclaim()?;
2816 self.dictionaries = dictionaries;
2817 self.gathers = gathers;
2818 Ok(())
2819 }
2820
2821 fn flush_pending(&mut self) -> Result<()> {
2826 if self.pending.is_empty() {
2827 return Ok(());
2828 }
2829 let held = std::mem::take(&mut self.pending);
2830 let prepared = self.preparer().prepare_held(held)?;
2831 let merged = self.merge_held(prepared)?;
2832 let paged = merged.pages()?;
2833 self.write_paged(paged)
2834 }
2835
2836 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2838 let width = self.table.fields.len();
2839 let parts = held.len();
2840 if encoded.len() != width {
2841 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2842 }
2843 let profile = self.profile.clone();
2844 if let Some(profile) = &profile {
2845 let rows = held.iter().map(|part| part.rows as u64).sum();
2846 let raw = held.iter().map(|part| part.footprint as u64).sum();
2847 let pages =
2848 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2849 profile.moved(Stage::Pages, raw, pages, rows);
2850 }
2851 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2854 let before = self.at;
2855 self.place_blocks()?;
2856 drop(timing);
2857 if let Some(profile) = &profile {
2858 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2859 }
2860 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2861 let before = self.at;
2862 let mut pages = Vec::with_capacity(width);
2863 let mut memberships = vec![None; width];
2864 let mut ranges = Vec::with_capacity(width);
2865 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2866 let start = self.at;
2869 let mut out = Vec::with_capacity(width.saturating_mul(parts));
2870 for stripe in &encoded {
2871 let offset = self.at;
2872 let section = index.len();
2873 let mut length = 0_usize;
2874 if stripe.sums.len() != stripe.pages.len() {
2875 return Err(Error::internal("a stripe's pages came without their checksums"));
2876 }
2877 for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2878 put_u32(
2879 &mut index,
2880 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2881 );
2882 put_u64(&mut index, sum);
2883 out.push(bytes.as_slice());
2884 length = length
2885 .checked_add(bytes.len())
2886 .ok_or_else(|| invalid("column page length overflow"))?;
2887 }
2888 let hash = checksum(&index[section..]);
2889 put_u64(&mut index, hash);
2890 if length > MAX_PAGE {
2891 return Err(invalid("column page exceeds the configured bound"));
2892 }
2893 self.at = self
2894 .at
2895 .checked_add(length as u64)
2896 .ok_or_else(|| invalid("native file length overflow"))?;
2897 pages.push(Span {
2898 offset,
2899 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2900 });
2901 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2902 }
2903 self.file.write_parts_at(start, &out)?;
2904 drop(out);
2905 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2906 if stripe.codes.iter().all(Option::is_none) {
2907 continue;
2908 }
2909 let lists = stripe
2910 .codes
2911 .iter()
2912 .map(|codes| codes.clone().unwrap_or_default())
2913 .collect::<Vec<_>>();
2914 let bytes = encode_membership(&merged_codes(lists));
2915 let offset = self.at;
2916 self.put(&bytes)?;
2917 *membership = Some(Page {
2918 offset,
2919 length: u32::try_from(bytes.len())
2920 .map_err(|_| invalid("membership page length overflow"))?,
2921 hash: checksum(&bytes),
2922 });
2923 }
2924 let mut sieves = vec![None; width];
2925 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2926 if stripe.sieves.iter().all(Option::is_none) {
2927 continue;
2928 }
2929 let bytes = encode_sieves(stripe.sieves.iter())?;
2930 let offset = self.at;
2931 self.put(&bytes)?;
2932 *page = Some(Page {
2933 offset,
2934 length: u32::try_from(bytes.len())
2935 .map_err(|_| invalid("sieve page length overflow"))?,
2936 hash: checksum(&bytes),
2937 });
2938 }
2939 let mut part_ranges = vec![None; width];
2945 if parts > 1 {
2946 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2947 let bytes = encode_part_ranges(&stripe.ranges)?;
2948 if bytes.len() >= span.length as usize {
2949 continue;
2950 }
2951 let offset = self.at;
2952 self.put(&bytes)?;
2953 *page = Some(Page {
2954 offset,
2955 length: u32::try_from(bytes.len())
2956 .map_err(|_| invalid("part range page length overflow"))?,
2957 hash: checksum(&bytes),
2958 });
2959 }
2960 }
2961 let offset = self.at;
2962 self.put(&index)?;
2963 let index = Span {
2964 offset,
2965 length: u32::try_from(index.len())
2966 .map_err(|_| invalid("index page length overflow"))?,
2967 };
2968 let mut rows = 0_usize;
2969 let mut lengths = Vec::with_capacity(parts);
2970 let mut span = None;
2971 for part in held {
2972 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2973 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2974 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2975 }
2976 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2977 self.table.stripes.push(Stripe {
2978 rows,
2979 parts: lengths,
2980 index,
2981 pages,
2982 memberships: Pages::from_slots(memberships)?,
2983 sieves: Pages::from_slots(sieves)?,
2984 part_ranges: Pages::from_slots(part_ranges)?,
2985 zone: Zone::from_ranges(ranges),
2986 });
2987 drop(timing);
2988 if let Some(profile) = &profile {
2989 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2990 }
2991 Ok(())
2992 }
2993
2994 fn numeric_frequency(
3014 &self,
3015 column: usize,
3016 counted: bool,
3017 dense: Option<(u64, usize)>,
3018 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
3019 let signed = match self.table.fields[column].ty {
3020 LogicalType::TinyInt
3021 | LogicalType::SmallInt
3022 | LogicalType::Integer
3023 | LogicalType::BigInt
3024 | LogicalType::Date
3025 | LogicalType::Timestamp => true,
3026 LogicalType::UTinyInt
3027 | LogicalType::USmallInt
3028 | LogicalType::UInteger
3029 | LogicalType::UBigInt => false,
3030 _ => return Ok((None, None)),
3031 };
3032 let value_of = |bits: Option<u64>| match bits {
3033 None => FrequencyValue::Null,
3034 Some(bits) => integer_value(bits, signed),
3035 };
3036 let tallied = self
3041 .gathers
3042 .get(column)
3043 .and_then(Option::as_ref)
3044 .filter(|gather| gather.rows() == self.table.rows as u64)
3045 .and_then(stats::Gather::frequencies)
3046 .and_then(|(values, nulls)| {
3047 let entries = values
3048 .iter()
3049 .map(|(value, count)| {
3050 let value = value_of(Some(frequency_bits(value)?));
3051 Some(FrequencyEntry { value, count: *count })
3052 })
3053 .chain((nulls != 0).then_some(Some(FrequencyEntry {
3054 value: FrequencyValue::Null,
3055 count: nulls,
3056 })))
3057 .collect::<Option<Vec<_>>>()?;
3058 Some((entries, values.len() as u64))
3059 });
3060 let exact = match (&tallied, counted) {
3064 (None, true) => self.exact_frequency(column, signed, dense)?,
3065 _ => None,
3066 };
3067 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3068 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3069 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3070 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3071 (None, None) => {
3072 let mut first = Candidates::default();
3076 let mut run = Run::default();
3077 self.visit_numeric(column, signed, |_, bits| {
3078 if let Some((ended, times)) = run.push(bits) {
3079 first.add(ended, times);
3080 }
3081 })?;
3082 if let Some((bits, times)) = run.take() {
3083 first.add(bits, times);
3084 }
3085 let (nulls, decrements) = (first.nulls, first.decrements);
3088 let distinct_count = (decrements == 0).then_some(first.held as u64);
3089 let (exact, null_count) = if decrements == 0 {
3090 let exact = first
3091 .pairs()
3092 .map(|(bits, count)| (bits, u64::from(count)))
3093 .collect::<FrequencyMap<_>>();
3094 (exact, (nulls != 0).then_some(u64::from(nulls)))
3095 } else {
3096 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3097 if nulls != 0 {
3098 lower.push(nulls);
3099 }
3100 lower.sort_unstable_by(|left, right| right.cmp(left));
3101 if lower.len() < FREQUENCY_BUILD_RANK
3102 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3103 {
3104 return Ok((None, distinct_count));
3105 }
3106 let mut recounts = vec![0_u64; first.slots.len()];
3109 let mut null_count = (nulls != 0).then_some(0_u64);
3110 let mut recount = |bits: Option<u64>, times: u32| {
3111 let held = match bits {
3112 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3113 None => null_count.as_mut(),
3114 };
3115 if let Some(count) = held {
3116 *count = count.saturating_add(u64::from(times));
3117 }
3118 };
3119 let mut run = Run::default();
3120 self.visit_numeric(column, signed, |_, bits| {
3121 if let Some((bits, times)) = run.push(bits) {
3122 recount(bits, times);
3123 }
3124 })?;
3125 if let Some((bits, times)) = run.take() {
3126 recount(bits, times);
3127 }
3128 let exact = first
3129 .slots
3130 .iter()
3131 .zip(&recounts)
3132 .filter(|(slot, _)| slot.count != 0)
3133 .map(|(slot, &count)| (slot.bits, count))
3134 .collect::<FrequencyMap<_>>();
3135 (exact, null_count)
3136 };
3137 let entries = exact
3138 .into_iter()
3139 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3140 .chain(
3141 null_count
3142 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3143 )
3144 .collect::<Vec<_>>();
3145 (entries, decrements, distinct_count)
3146 }
3147 };
3148 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3149 if omitted_max == 0 && entries.len() > 1 {
3153 let retained = entries.len().saturating_sub(1).min(2);
3154 omitted_max = entries[retained].count;
3155 entries.truncate(retained);
3156 }
3157 let mut covered = 0;
3161 let mut kept_rows = 0_u64;
3162 for entry in &entries {
3163 match kept_rows.checked_add(entry.count) {
3164 Some(total) if total <= FREQUENCY_ORDINALS as u64 => kept_rows = total,
3165 _ => break,
3166 }
3167 covered += 1;
3168 }
3169 let ordinal_bound = entries.get(covered).map_or(0, |entry| entry.count);
3170 let worth_keeping = covered == entries.len()
3171 || (covered >= FREQUENCY_BUILD_RANK
3172 && entries[FREQUENCY_BUILD_RANK - 1].count > ordinal_bound.max(omitted_max));
3173 let mut ordinals = Vec::new();
3174 let mut ordinal_entries = Vec::new();
3175 if worth_keeping {
3176 let mut kept = FrequencyMap::default();
3177 let mut null_kept = None;
3178 for (at, entry) in entries.iter().enumerate().take(covered) {
3179 let at = u16::try_from(at)
3180 .map_err(|_| invalid("too many retained frequency entries"))?;
3181 match entry.value {
3182 FrequencyValue::Integer(value) => {
3183 kept.insert(value as u64, at);
3184 }
3185 FrequencyValue::Null => null_kept = Some(at),
3186 FrequencyValue::Code(_) => {}
3187 }
3188 }
3189 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3190 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3191 self.visit_numeric(column, signed, |ordinal, bits| {
3192 let held = match bits {
3193 Some(bits) => kept.get(&bits).copied(),
3194 None => null_kept,
3195 };
3196 if let Some(entry) = held {
3197 ordinals.push(ordinal);
3198 ordinal_entries.push(entry);
3199 }
3200 })?;
3201 }
3202 Ok((
3203 Some(FrequencySummary {
3204 entries,
3205 omitted_max,
3206 ordinals,
3207 ordinal_entries,
3208 ordinal_bound: if worth_keeping { ordinal_bound } else { 0 },
3209 }),
3210 distinct_count,
3211 ))
3212 }
3213
3214 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3221 let rows = self.table.rows;
3222 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3223 return None;
3224 }
3225 let (low, high) = gather.span()?;
3226 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3227 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3228 let bits = low as u64;
3229 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3230 }
3231
3232 fn exact_frequency(
3246 &self,
3247 column: usize,
3248 signed: bool,
3249 dense: Option<(u64, usize)>,
3250 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3251 if let Some((low, len)) = dense {
3254 let mut counts = distinct::DenseCounts::new(low, len);
3255 let nulls =
3256 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3257 if let Some(distinct) = counts.count() {
3258 let Some(distinct) = distinct else { return Ok(None) };
3259 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3260 counts.visit(visit);
3261 })));
3262 }
3263 }
3264 let mut set = distinct::ExactCounts::new();
3265 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3266 let Some(distinct) = set.count() else {
3267 return Ok(None);
3268 };
3269 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3270 set.visit(visit);
3271 })))
3272 }
3273
3274 fn count_numeric(
3277 &self,
3278 column: usize,
3279 signed: bool,
3280 mut add: impl FnMut(u64, u32),
3281 ) -> Result<u64> {
3282 let mut nulls = 0_u64;
3283 let mut run = Run::default();
3284 let mut take = |bits: Option<u64>, times: u32| match bits {
3285 Some(bits) => add(bits, times),
3286 None => nulls += u64::from(times),
3287 };
3288 self.visit_numeric(column, signed, |_, bits| {
3289 if let Some((bits, times)) = run.push(bits) {
3290 take(bits, times);
3291 }
3292 })?;
3293 if let Some((bits, times)) = run.take() {
3294 take(bits, times);
3295 }
3296 Ok(nulls)
3297 }
3298
3299 fn frequent_entries(
3302 &self,
3303 signed: bool,
3304 distinct: u64,
3305 nulls: u64,
3306 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3307 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3308 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3310 let mut rank = |count: u64| {
3311 if top.len() <= FREQUENCY_ENTRIES {
3312 top.push(Reverse(count));
3313 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3314 top.pop();
3315 top.push(Reverse(count));
3316 }
3317 };
3318 visit(&mut |_, count| rank(count));
3319 if nulls != 0 {
3320 rank(nulls);
3321 }
3322 let top = top.into_sorted_vec();
3323 let values = distinct + u64::from(nulls != 0);
3324 if values > FREQUENCY_CANDIDATES as u64 {
3325 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3326 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3327 return (None, distinct);
3328 }
3329 }
3330 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3331 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3332 visit(&mut |bits, count| {
3333 if count >= least {
3334 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3335 }
3336 });
3337 if nulls != 0 && nulls >= least {
3338 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3339 }
3340 (Some(entries), distinct)
3341 }
3342
3343 fn visit_numeric(
3350 &self,
3351 column: usize,
3352 signed: bool,
3353 mut visit: impl FnMut(u64, Option<u64>),
3354 ) -> Result<()> {
3355 let ty = &self.table.fields[column].ty;
3356 let mut start = 0_u64;
3357 let mut block = Vec::new();
3358 for stripe in &self.table.stripes {
3359 let spans = read_index(&self.file, stripe, column)?;
3360 let page = stripe.pages[column];
3361 let mut bytes = vec![0; page.length as usize];
3362 read_at(&self.file, page.offset, &mut bytes)?;
3363 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3364 let part = part_bytes(&bytes, *span)?;
3365 if checksum(part) != span.hash {
3366 return Err(invalid("column page checksum differs while building frequencies"));
3367 }
3368 let rows = rows as usize;
3369 let vector = decode(ty, rows, part, None)?;
3370 if signed && vector.signed_block(&mut block) && block.len() == rows {
3374 if vector.none_null() {
3375 for (row, &value) in block.iter().enumerate() {
3376 visit(start.saturating_add(row as u64), Some(value as u64));
3377 }
3378 } else {
3379 for (row, &value) in block.iter().enumerate() {
3380 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3381 visit(start.saturating_add(row as u64), bits);
3382 }
3383 }
3384 start = start.saturating_add(rows as u64);
3385 continue;
3386 }
3387 for row in 0..rows {
3389 let bits = if vector.is_null_at(row) {
3390 None
3391 } else {
3392 let widened = match vector.signed_at(row) {
3396 Some(value) => Some(value as u64),
3397 None => match vector.value_at(row) {
3398 Value::UTinyInt(value) => Some(u64::from(value)),
3399 Value::USmallInt(value) => Some(u64::from(value)),
3400 Value::UInteger(value) => Some(u64::from(value)),
3401 Value::UBigInt(value) => Some(value),
3402 _ => None,
3403 },
3404 };
3405 Some(widened.ok_or_else(|| {
3406 invalid("numeric frequency page did not contain an integer value")
3407 })?)
3408 };
3409 visit(start.saturating_add(row as u64), bits);
3410 }
3411 start = start.saturating_add(rows as u64);
3412 }
3413 }
3414 Ok(())
3415 }
3416
3417 fn numeric_columns(&self) -> Vec<usize> {
3419 self.table
3420 .fields
3421 .iter()
3422 .enumerate()
3423 .filter_map(|(column, field)| {
3424 matches!(
3425 field.ty,
3426 LogicalType::TinyInt
3427 | LogicalType::SmallInt
3428 | LogicalType::Integer
3429 | LogicalType::BigInt
3430 | LogicalType::UTinyInt
3431 | LogicalType::USmallInt
3432 | LogicalType::UInteger
3433 | LogicalType::UBigInt
3434 | LogicalType::Date
3435 | LogicalType::Timestamp
3436 )
3437 .then_some(column)
3438 })
3439 .collect()
3440 }
3441
3442 #[allow(dead_code)]
3444 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3445 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3446 return Ok(None);
3447 }
3448 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3449 return Err(invalid("frequency ordinals are not sorted and unique"));
3450 }
3451 let mut out = Vec::with_capacity(ordinals.len());
3452 let mut wanted = 0;
3453 let mut stripe_start = 0_u64;
3454 for stripe in &self.table.stripes {
3455 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3456 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3457 stripe_start = stripe_end;
3458 continue;
3459 }
3460 let spans = read_index(&self.file, stripe, column)?;
3461 let page = stripe.pages[column];
3462 let mut bytes = vec![0; page.length as usize];
3463 read_at(&self.file, page.offset, &mut bytes)?;
3464 let mut part_start = stripe_start;
3465 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3466 let part_end = part_start.saturating_add(u64::from(rows));
3467 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3468 let part = part_bytes(&bytes, *span)?;
3469 if checksum(part) != span.hash {
3470 return Err(invalid(
3471 "column page checksum differs while building pair frequencies",
3472 ));
3473 }
3474 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3475 let positions = ordinals[wanted..upto]
3476 .iter()
3477 .map(|&ordinal| {
3478 usize::try_from(ordinal.saturating_sub(part_start))
3479 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3480 })
3481 .collect::<Result<Vec<_>>>()?;
3482 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3483 return Ok(None);
3484 }
3485 wanted = upto;
3486 }
3487 part_start = part_end;
3488 }
3489 stripe_start = stripe_end;
3490 }
3491 if wanted != ordinals.len() {
3492 return Err(invalid("frequency ordinal is outside the table"));
3493 }
3494 Ok(Some(out))
3495 }
3496
3497 #[allow(dead_code)]
3499 fn pair_frequencies(
3500 &self,
3501 frequencies: &[Option<Frequencies>],
3502 ) -> Result<Vec<PairFrequencySummary>> {
3503 let anchors = frequencies
3504 .iter()
3505 .enumerate()
3506 .filter_map(|(column, summary)| {
3507 match summary {
3509 Some(Frequencies::Held(summary)) => Some(summary),
3510 _ => None,
3511 }
3512 .filter(|summary| {
3513 !summary.ordinals.is_empty()
3514 && summary.ordinal_entries.len() == summary.ordinals.len()
3515 && summary.ordinal_bound == 0
3516 })
3517 .cloned()
3518 .map(|summary| (column, summary))
3519 })
3520 .collect::<Vec<_>>();
3521 let strings = self
3522 .dictionaries
3523 .iter()
3524 .enumerate()
3525 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3526 .collect::<Vec<_>>();
3527 let mut summaries = Vec::new();
3528 for (first, anchors) in anchors {
3529 for &second in &strings {
3530 if summaries.len() == MAX_PAIR_FREQUENCIES {
3531 return Ok(summaries);
3532 }
3533 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3534 continue;
3535 };
3536 if codes.len() != anchors.ordinal_entries.len() {
3537 return Err(invalid("pair frequency columns have different lengths"));
3538 }
3539 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3540 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3541 *counts.entry((anchor, code)).or_default() += 1;
3542 }
3543 let mut entries = counts
3544 .into_iter()
3545 .map(|((first_entry, second), count)| PairFrequencyEntry {
3546 first_entry,
3547 second,
3548 count,
3549 })
3550 .collect::<Vec<_>>();
3551 entries.sort_unstable_by(|left, right| {
3552 right
3553 .count
3554 .cmp(&left.count)
3555 .then_with(|| left.first_entry.cmp(&right.first_entry))
3556 .then_with(|| left.second.cmp(&right.second))
3557 });
3558 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3559 entries.truncate(FREQUENCY_ENTRIES);
3560 summaries.push(PairFrequencySummary {
3561 first: u16::try_from(first)
3562 .map_err(|_| invalid("pair frequency column index overflows"))?,
3563 second: u16::try_from(second)
3564 .map_err(|_| invalid("pair frequency column index overflows"))?,
3565 entries,
3566 omitted_max: anchors.omitted_max.max(pair_omitted),
3567 });
3568 }
3569 }
3570 Ok(summaries)
3571 }
3572
3573 fn close(&mut self) -> Result<Entry> {
3584 self.reclaim()?;
3585 self.flush_pending()?;
3586 let profile = self.profile.clone();
3590 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3591 let before = self.at;
3592 let mut stripes = std::mem::take(&mut self.order)
3593 .into_iter()
3594 .zip(std::mem::take(&mut self.table.stripes))
3595 .collect::<Vec<_>>();
3596 stripes.sort_by_key(|(order, _)| order.0);
3597 let mut previous: Option<(u64, u64)> = None;
3598 for ((first, last), _) in &stripes {
3599 if previous.is_some_and(|previous| previous >= *first) {
3600 return Err(invalid("chunks did not arrive in source order"));
3601 }
3602 previous = Some(*last);
3603 }
3604 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3605 drop(timing);
3606 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3607 let placing = self.at;
3608 finish_dictionaries(&mut self.dictionaries)?;
3609 self.place_blocks()?;
3610 for dictionary in self.dictionaries.iter_mut().flatten() {
3611 dictionary.release_lookup();
3612 dictionary.recharge(profile.as_deref());
3613 }
3614 let (numeric, closed) = self.close_columns()?;
3615 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3616 numeric.into_iter().unzip();
3617 let frequencies =
3618 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3619 let pairs = Vec::new();
3621 self.table.frequencies = frequencies;
3622 self.table.distincts = distincts;
3623 self.table.pair_frequencies = pairs;
3624 if let Some(profile) = &profile {
3625 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3626 }
3627 self.table.demoted = self
3628 .dictionaries
3629 .iter()
3630 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3631 .collect();
3632 if !self.table.demoted.contains(&true) {
3633 self.table.demoted = Vec::new();
3634 }
3635 self.dictionaries = Vec::new();
3636 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3637 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3638 self.table.host_groups = None;
3639 for (index, closed) in closed.into_iter().enumerate() {
3640 let Some(closed) = closed else { continue };
3641 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3642 self.table.distincts[index] = distinct;
3643 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3644 self.table.frequency_texts[index] = texts;
3645 if hosts.is_some() {
3646 self.table.host_groups = hosts;
3647 }
3648 let offset = self.at;
3649 self.put(&encoded.index)?;
3650 self.put(&encoded.ranks)?;
3651 self.put(&encoded.grams)?;
3652 self.table.dictionary_payloads[index] = payload;
3653 let length = encoded
3654 .index
3655 .len()
3656 .checked_add(encoded.ranks.len())
3657 .and_then(|len| len.checked_add(encoded.grams.len()))
3658 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3659 self.table.dictionaries[index] = Some(Page {
3660 offset,
3661 length: u32::try_from(length)
3662 .map_err(|_| invalid("dictionary page length overflow"))?,
3663 hash: checksum(&encoded.index),
3664 });
3665 }
3666 drop(timing);
3667 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3668 let placed = self.at - placing;
3669 self.write_stats()?;
3670 let directory = encode_directory(&self.table)?;
3671 if directory.len() > MAX_DIRECTORY {
3672 return Err(invalid("directory exceeds the configured bound"));
3673 }
3674 let offset = self.at;
3675 self.put(&directory)?;
3676 drop(timing);
3677 if let Some(profile) = &profile {
3678 profile.moved(Stage::Dictionary, 0, placed, 0);
3679 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3680 }
3681 Ok(Entry {
3682 name: self.table.name.clone(),
3683 fields: self.table.fields.clone(),
3684 rows: self.table.rows,
3685 nonzero: vec![None; self.table.fields.len()],
3686 aggregates: table_aggregate_sums(&self.table),
3687 distincts: self.table.distincts.clone(),
3688 extremes: table_integer_extremes(&self.table),
3689 frequencies: table_complete_numeric_frequencies(&self.table),
3690 directory: Page {
3691 offset,
3692 length: u32::try_from(directory.len())
3693 .map_err(|_| invalid("directory length overflow"))?,
3694 hash: checksum(&directory),
3695 },
3696 })
3697 }
3698
3699 #[allow(clippy::type_complexity)]
3716 fn close_columns(
3717 &self,
3718 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3719 let numeric = self.numeric_columns().into_iter().map(|column| {
3720 let gather = self.gathers.get(column).and_then(Option::as_ref);
3721 let estimate = gather.and_then(stats::Gather::distinct);
3722 let counted = !estimate.is_some_and(distinct::beyond);
3723 let set =
3724 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3725 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3726 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3727 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3728 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3729 });
3730 let dictionaries =
3731 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3732 let dictionary = dictionary.as_ref()?;
3733 let bytes = dictionary.closing_bytes();
3734 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3735 });
3736 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3737 jobs.sort_by_key(|&(_, _, cost)| cost);
3738 let columns = self.table.fields.len();
3739 let mut frequencies = vec![(None, None); columns];
3740 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3741 let profile = self.profile.as_deref();
3742 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3743 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3744 let closed = match job {
3745 Closing::Numeric { column, counted, dense } => {
3746 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3747 Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3748 }
3749 Closing::Dictionary { index, dictionary } => {
3750 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3751 Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3752 }
3753 };
3754 rudb_common::heap::release();
3757 Ok(closed)
3758 };
3759 let workers = close_workers().min(jobs.len());
3760 let pieces = if workers <= 1 {
3761 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3762 } else {
3763 let state = Mutex::new((jobs, 0_usize));
3765 let finished = Condvar::new();
3766 std::thread::scope(|scope| {
3767 (0..workers)
3768 .map(|_| {
3769 scope.spawn(|| {
3770 let mut mine = Vec::new();
3771 loop {
3772 let mut held = state.lock().map_err(|_| {
3773 Error::internal("a native close worker panicked")
3774 })?;
3775 let (job, bytes) = loop {
3776 let (jobs, busy) = &mut *held;
3777 if jobs.is_empty() {
3778 return Ok(mine);
3779 }
3780 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3781 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3782 });
3783 if let Some(at) = fits {
3784 let (job, bytes, _) = jobs.remove(at);
3785 *busy += bytes;
3786 break (job, bytes);
3787 }
3788 held = finished.wait(held).map_err(|_| {
3789 Error::internal("a native close worker panicked")
3790 })?;
3791 };
3792 drop(held);
3793 let _room = Room { state: &state, finished: &finished, bytes };
3796 mine.push(run(job, bytes)?);
3797 }
3798 })
3799 })
3800 .collect::<Vec<_>>()
3801 .into_iter()
3802 .map(|handle| {
3803 handle
3804 .join()
3805 .map_err(|_| Error::internal("a native close worker panicked"))?
3806 })
3807 .collect::<Result<Vec<_>>>()
3808 })?
3809 .into_iter()
3810 .flatten()
3811 .collect()
3812 };
3813 for piece in pieces {
3814 match piece {
3815 Closed::Numeric(column, summary) => frequencies[column] = summary,
3816 Closed::Dictionary(index, one) => closed[index] = Some(one),
3817 }
3818 }
3819 Ok((frequencies, closed))
3820 }
3821
3822 fn close_dictionary(
3829 &self,
3830 _index: usize,
3831 dictionary: &GlobalDictionary,
3832 ) -> Result<ClosedDictionary> {
3833 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3834 let (distinct, frequencies, texts) = if dictionary.demoted {
3839 (None, None, Vec::new())
3840 } else {
3841 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3842 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3843 (Some(distinct), Some(frequencies), texts)
3844 };
3845 let hosts = None;
3847 drop(flat);
3848 drop(bases);
3849 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3850 let payload = dictionary
3851 .placed
3852 .iter()
3853 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3854 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3855 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3856 }
3857
3858 fn write_stats(&mut self) -> Result<()> {
3870 let gathers = std::mem::take(&mut self.gathers);
3871 let rows = self.table.rows as u64;
3872 let mut payloads = Vec::new();
3873 for (column, gather) in gathers.into_iter().enumerate() {
3874 let Some(gather) = gather else { continue };
3875 if gather.rows() != rows {
3881 continue;
3882 }
3883 let Some(stats) = gather.finish() else { continue };
3884 let mut summary = Vec::new();
3885 stats.summary.encode(&mut summary)?;
3886 let mut sketches = Vec::new();
3887 stats.sketches.encode(&mut sketches)?;
3888 payloads.push((column, summary, sketches));
3889 }
3890 if payloads.is_empty() {
3891 return Ok(());
3892 }
3893 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3894 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3895 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3896 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3899 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3900 if !built {
3901 continue;
3902 }
3903 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3904 let sections = [
3905 (*section::SUMMARY, summary, summary.len() as u32),
3908 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3909 ];
3910 let wanted = 1 + usize::from(sketched);
3911 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3912 let written = write_section(
3913 &*self.file,
3914 &mut self.at,
3915 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3916 self.generation,
3917 )?;
3918 self.table.sections.push(written);
3919 }
3920 }
3921 if self.table.sections.len() > MAX_SECTIONS {
3922 return Err(invalid("the table would name more sections than the bound allows"));
3923 }
3924 Ok(())
3925 }
3926
3927 pub fn finish(mut self) -> Result<Table> {
3937 let entry = self.close()?;
3938 let profile = self.profile.take();
3939 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3940 let mut tables = std::mem::take(&mut self.closed);
3941 tables.push(entry);
3942 let catalog =
3943 encode_catalog(&tables, &self.views, self.card.as_ref(), self.anchor.as_ref())?;
3944 if catalog.len() > MAX_DIRECTORY {
3945 return Err(invalid("catalog exceeds the configured bound"));
3946 }
3947 let offset = self.at;
3948 self.put(&catalog)?;
3949 if let Some(profile) = &profile {
3950 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3951 }
3952 synced(&*self.file, profile.as_deref())?;
3956 let slot = Slot {
3957 offset,
3958 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3959 generation: self.generation,
3960 hash: checksum(&catalog),
3961 };
3962 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3967 synced(&*self.file, profile.as_deref())?;
3968 Ok(self.table)
3969 }
3970
3971 pub fn restate(
3990 path: impl AsRef<Path>,
3991 views: &[ViewEntry],
3992 anchor: Option<&LogAnchor>,
3993 ) -> Result<()> {
3994 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3995 let size = file.len()?;
3996 let (slot, bytes, _) = committed_slot(&*file, size)?;
3997 let (closed, _, card, held) = decode_catalog(&bytes, size)?;
3998 let anchor = anchor.cloned().or(held);
3999 let generation = slot
4000 .generation
4001 .checked_add(1)
4002 .ok_or_else(|| invalid("native file generation overflow"))?;
4003 let catalog = encode_catalog(
4004 &closed,
4005 views,
4006 card_for(path.as_ref(), card).as_ref(),
4007 anchor.as_ref(),
4008 )?;
4009 if catalog.len() > MAX_DIRECTORY {
4010 return Err(invalid("catalog exceeds the configured bound"));
4011 }
4012 file.write_at(size, &catalog)?;
4013 file.sync()?;
4014 let slot = Slot {
4015 offset: size,
4016 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4017 generation,
4018 hash: checksum(&catalog),
4019 };
4020 file.write_at(slot_offset(generation), &slot.bytes())?;
4021 file.sync()?;
4022 Ok(())
4023 }
4024
4025 pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
4036 let path = path.as_ref();
4037 let (_, size, _, bytes, _) = slot_bytes(path)?;
4038 let (_, views, held, _) = decode_catalog(&bytes, size)?;
4039 if card_for(path, held.clone()) == held {
4040 return Ok(());
4041 }
4042 Self::restate(path, &views, None)
4043 }
4044
4045 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
4048 let path = path.as_ref();
4049 let (_, size, slot, bytes, _) = slot_bytes(path)?;
4050 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4051 let native = Catalog::open(path)?;
4052 for entry in &mut entries {
4053 let reader = native.table(&entry.name)?;
4054 entry.nonzero.fill(None);
4055 entry.aggregates = reader_aggregate_sums(&reader)?;
4056 entry.distincts = (0..entry.fields.len())
4057 .map(|column| reader.distinct_values(column))
4058 .collect::<Result<Vec<_>>>()?;
4059 entry.extremes = reader_integer_extremes(&reader)?;
4060 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
4061 }
4062 let generation = slot
4063 .generation
4064 .checked_add(1)
4065 .ok_or_else(|| invalid("native file generation overflow"))?;
4066 let catalog =
4067 encode_catalog(&entries, &views, card_for(path, card).as_ref(), anchor.as_ref())?;
4068 if catalog.len() > MAX_DIRECTORY {
4069 return Err(invalid("catalog exceeds the configured bound"));
4070 }
4071 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
4072 file.write_at(size, &catalog)?;
4073 file.sync()?;
4074 let slot = Slot {
4075 offset: size,
4076 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4077 generation,
4078 hash: checksum(&catalog),
4079 };
4080 file.write_at(slot_offset(generation), &slot.bytes())?;
4081 file.sync()?;
4082 Ok(())
4083 }
4084
4085 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4087 Self::certify_summaries(path)
4088 }
4089}
4090
4091fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4097 let offset = *at;
4098 file.write_at(offset, bytes)?;
4099 *at =
4100 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4101 Ok(offset)
4102}
4103
4104fn write_section(
4110 file: &dyn rudb_io::File,
4111 at: &mut u64,
4112 one: §ion::Attachment<'_>,
4113 generation: u64,
4114) -> Result<Section> {
4115 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4119 return Err(invalid("a section's header is longer than its payload"));
4120 }
4121 let mut extents = Vec::new();
4122 let mut first = 0_u64;
4123 let extent_size =
4124 if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4125 run_projection::RLE_PAGE_BYTES
4126 } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4127 1 << 19
4128 } else {
4129 section::MAX_EXTENT as usize
4130 };
4131 for chunk in one.bytes.chunks(extent_size) {
4132 let offset = append(file, at, chunk)?;
4133 extents.push(section::Extent {
4134 offset,
4135 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4136 hash: checksum(chunk),
4137 first,
4138 });
4139 first += chunk.len() as u64;
4140 }
4141 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4142 section::encode_extents(&extents, &mut table)?;
4143 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4147 Ok(Section {
4148 kind: one.kind,
4149 id: one.id,
4150 generation,
4151 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4152 extent_page,
4153 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4154 hash: checksum(&table),
4155 flags: one.flags,
4156 header_bytes: one.header_bytes,
4157 })
4158}
4159
4160pub fn attach(
4184 path: impl AsRef<Path>,
4185 table: &str,
4186 attachments: &[section::Attachment<'_>],
4187) -> Result<Table> {
4188 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4189 let file = &*file;
4190 let size = file.len()?;
4191 let (slot, bytes, _) = committed_slot(file, size)?;
4192 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4193 let card = card_for(path.as_ref(), card);
4194 let at = entries
4195 .iter()
4196 .position(|entry| entry.name == table)
4197 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4198 let mut version = [0; 4];
4199 read_at(file, 8, &mut version)?;
4200 let version = u32::from_le_bytes(version);
4201 if version != FORMAT {
4207 return Err(invalid(&format!(
4208 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4209 to be written again"
4210 )));
4211 }
4212 let mut directory = vec![0; entries[at].directory.length as usize];
4213 read_at(file, entries[at].directory.offset, &mut directory)?;
4214 if checksum(&directory) != entries[at].directory.hash {
4215 return Err(invalid(&format!("the directory of table {table} does not checksum")));
4216 }
4217 let mut held = decode_directory(&directory, size)?;
4218 let mut cursor = size;
4219 for one in attachments {
4220 let written = write_section(file, &mut cursor, one, held.generation)?;
4221 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4222 held.sections.push(written);
4223 }
4224 if held.sections.len() > MAX_SECTIONS {
4225 return Err(invalid("the table would name more sections than the bound allows"));
4226 }
4227 let encoded = encode_directory(&held)?;
4228 if encoded.len() > MAX_DIRECTORY {
4229 return Err(invalid("directory exceeds the configured bound"));
4230 }
4231 let offset = append(file, &mut cursor, &encoded)?;
4232 entries[at].directory = Page {
4233 offset,
4234 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4235 hash: checksum(&encoded),
4236 };
4237 let catalog = encode_catalog(&entries, &views, card.as_ref(), anchor.as_ref())?;
4240 if catalog.len() > MAX_DIRECTORY {
4241 return Err(invalid("catalog exceeds the configured bound"));
4242 }
4243 let offset = append(file, &mut cursor, &catalog)?;
4244 file.sync()?;
4245 let generation =
4246 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4247 let committed = Slot {
4248 offset,
4249 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4250 generation,
4251 hash: checksum(&catalog),
4252 };
4253 file.write_at(slot_offset(generation), &committed.bytes())?;
4254 file.sync()?;
4255 Ok(held)
4256}
4257
4258type Synopsis = Arc<Vec<(Value, u64)>>;
4261
4262#[derive(Debug, Clone)]
4264pub struct Reader {
4265 file: Arc<File>,
4266 map: Option<Arc<Mapped>>,
4268 table: Arc<Table>,
4269 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4270 loading: Arc<Vec<Mutex<()>>>,
4279 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4282 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4286 frequency_heads: Arc<Vec<OnceLock<Arc<FrequencyHead>>>>,
4290 summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4292 facts: Arc<OnceLock<Arc<rudb_common::ColumnFacts>>>,
4295 opened: Arc<AtomicUsize>,
4299 sieves: Arc<Vec<OnceLock<Box<[SieveSlot]>>>>,
4304 part_ranges: Arc<Vec<OnceLock<Box<[RangeSlot]>>>>,
4307 places: Arc<Vec<Place>>,
4309 cache: Arc<Shelf>,
4310 pool: PagePool,
4312 pages: Arc<AtomicUsize>,
4315 indexes: Arc<AtomicUsize>,
4318 verified: Arc<Vec<AtomicU64>>,
4328 unreleased: Arc<Vec<AtomicU32>>,
4336 text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4339 firsts: Arc<Vec<usize>>,
4341 graph: Arc<graph::Decoded>,
4344 size: u64,
4346 directory: u64,
4348 opening: Opening,
4350}
4351
4352#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4364pub struct Opening {
4365 pub reads: u32,
4368 pub bytes: u64,
4370}
4371
4372#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4374pub struct Reads {
4375 pub opening: Opening,
4377 pub pages: usize,
4379 pub indexes: usize,
4381 pub dictionaries: usize,
4384}
4385
4386#[derive(Debug, Clone, Copy)]
4388struct Place {
4389 stripe: u32,
4390 part: u32,
4391 rows: u32,
4392}
4393
4394#[derive(Debug, Clone, Copy)]
4396struct PartSpan {
4397 start: usize,
4398 length: usize,
4399 hash: u64,
4400}
4401
4402#[derive(Debug, Clone)]
4408struct CachedColumn {
4409 stripe: usize,
4410 index: Arc<Vec<PartSpan>>,
4411 page: Option<Arc<HeldPage>>,
4412}
4413
4414#[derive(Debug)]
4421struct HeldPage {
4422 bytes: PageBytes,
4423 checked: Vec<AtomicBool>,
4424}
4425
4426#[derive(Debug)]
4428enum PageBytes {
4429 Read(Vec<u8>),
4430 Mapped { map: Arc<Mapped>, offset: u64, length: usize },
4431}
4432
4433impl PageBytes {
4434 fn bytes(&self) -> &[u8] {
4435 match self {
4436 Self::Read(bytes) => bytes,
4437 Self::Mapped { map, offset, length } => map.get(*offset, *length).unwrap_or_default(),
4439 }
4440 }
4441}
4442
4443impl HeldPage {
4444 fn bytes(&self) -> &[u8] {
4445 self.bytes.bytes()
4446 }
4447
4448 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4450 let bytes = part_bytes(self.bytes(), span)?;
4451 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4452 if !checked.load(Atomic::Relaxed) {
4453 verify_part(bytes, span)?;
4454 checked.store(true, Atomic::Relaxed);
4455 }
4456 Ok(bytes)
4457 }
4458}
4459
4460fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4462 let got = checksum(bytes);
4463 if got != span.hash {
4464 return Err(invalid(&format!(
4465 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4466 span.start, span.length, span.hash,
4467 )));
4468 }
4469 Ok(())
4470}
4471
4472#[derive(Debug, Default)]
4512struct Cached {
4513 pages: Vec<Option<Resident>>,
4514 loading: Vec<usize>,
4515 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4516 touched: Vec<Vec<u64>>,
4517 passing: VecDeque<usize>,
4518}
4519
4520#[derive(Debug, Clone)]
4522struct Resident {
4523 page: Arc<HeldPage>,
4524 used: Arc<AtomicBool>,
4525}
4526
4527#[derive(Debug)]
4529struct Shelf {
4530 columns: Vec<Mutex<Cached>>,
4531 held: Vec<AtomicUsize>,
4534 kept: AtomicUsize,
4537}
4538
4539#[derive(Debug, Clone, Default)]
4558pub struct PagePool {
4559 ring: Arc<Mutex<Ring>>,
4560 budget: Arc<AtomicUsize>,
4561}
4562
4563#[derive(Debug, Default)]
4564struct Ring {
4565 held: VecDeque<Held>,
4566 bytes: usize,
4567}
4568
4569#[derive(Debug)]
4574struct Held {
4575 shelf: Weak<Shelf>,
4576 column: usize,
4577 stripe: usize,
4578 bytes: usize,
4579 used: Arc<AtomicBool>,
4580}
4581
4582impl PagePool {
4583 #[must_use]
4585 pub fn new(budget: usize) -> Self {
4586 let pool = Self::default();
4587 pool.budget.store(budget, Atomic::Relaxed);
4588 pool
4589 }
4590
4591 #[must_use]
4597 pub fn bytes(&self) -> usize {
4598 self.ring.lock().map_or(0, |ring| ring.bytes)
4599 }
4600
4601 fn admit(&self, held: Held) {
4607 let budget = self.budget.load(Atomic::Relaxed);
4608 let mut gone = Vec::new();
4609 {
4610 let Ok(mut ring) = self.ring.lock() else { return };
4611 ring.bytes += held.bytes;
4612 ring.held.push_back(held);
4613 let mut looked = 0;
4616 let limit = ring.held.len();
4617 while ring.bytes > budget && looked < limit {
4618 looked += 1;
4619 let Some(entry) = ring.held.pop_front() else { break };
4620 let Some(shelf) = entry.shelf.upgrade() else {
4621 ring.bytes -= entry.bytes;
4622 continue;
4623 };
4624 if entry.used.swap(false, Atomic::Relaxed) {
4625 ring.held.push_back(entry);
4626 continue;
4627 }
4628 let count = &shelf.held[entry.column];
4629 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4630 ring.held.push_back(entry);
4631 continue;
4632 }
4633 count.fetch_sub(1, Atomic::Relaxed);
4634 ring.bytes -= entry.bytes;
4635 gone.push((shelf, entry));
4636 }
4637 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4640 if let Some(entry) = ring.held.pop_front() {
4641 ring.bytes -= entry.bytes;
4642 }
4643 }
4644 }
4645 for (shelf, entry) in gone {
4646 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4647 if let Some(slot) = cached.pages.get_mut(entry.stripe)
4648 && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4649 {
4650 *slot = None;
4651 }
4652 }
4653 }
4654}
4655
4656const CACHED_STRIPES_PER_COLUMN: usize = 4;
4668
4669type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4671
4672type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4673
4674#[derive(Debug)]
4675struct NativeText {
4676 file: Arc<File>,
4677 values: usize,
4679 offsets: Vec<u8>,
4691 offset_bits: usize,
4694 value_ends: OnceLock<Option<Vec<u32>>>,
4707 value_lens: OnceLock<Option<Lengths>>,
4717 ends_asked: AtomicUsize,
4723 ranks: usize,
4725 rank_at: u64,
4729 rank_ends: Vec<u64>,
4733 rank_hashes: Vec<u64>,
4734 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4735 code_bits: usize,
4738 code_ranks: OnceLock<Option<Vec<u32>>>,
4745 starts: Vec<u64>,
4752 lengths: Vec<u64>,
4753 hashes: Vec<u64>,
4754 grams: Option<NativeGrams>,
4756 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4758 char_lens: Vec<OnceLock<Box<[u32]>>>,
4767 keep_budget: usize,
4770 payload_kept: AtomicUsize,
4778 swept: Vec<AtomicBool>,
4786 visit_dropped: AtomicUsize,
4801 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4818}
4819
4820#[derive(Debug)]
4821struct NativeGrams {
4822 start: u64,
4823 length: usize,
4824 width: usize,
4826 hash: u64,
4827 verdicts: Mutex<Vec<Verdict>>,
4834}
4835
4836type Verdict = (Vec<u8>, Arc<[bool]>);
4838
4839const GRAM_VERDICTS: usize = 8;
4841
4842impl NativeGrams {
4843 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4848 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4849 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4850 return Ok(Arc::clone(verdict));
4851 }
4852 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4853 let mut verdict = Vec::with_capacity(self.length / self.width);
4854 let window = GRAM_WINDOW / self.width * self.width;
4855 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4856 verdict.extend(bytes.chunks(self.width).map(|bits| {
4857 wanted
4858 .iter()
4859 .flatten()
4860 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4861 }));
4862 Ok(())
4863 })?;
4864 if hash != self.hash {
4865 return Err(invalid("global dictionary substring signatures checksum differs"));
4866 }
4867 let verdict: Arc<[bool]> = verdict.into();
4868 if held.len() >= GRAM_VERDICTS {
4869 held.remove(0);
4870 }
4871 held.push((literal.to_vec(), Arc::clone(&verdict)));
4872 Ok(verdict)
4873 }
4874
4875 fn footprint(&self) -> usize {
4876 self.verdicts.lock().map_or(0, |held| {
4877 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4878 })
4879 }
4880}
4881
4882const TEXT_SEARCH_MEMO: usize = 64;
4887
4888const TEXT_PAYLOAD_VALUES: usize = 1024;
4904
4905const TEXT_GRAM_BYTES: usize = 8192;
4916
4917const NARROW_GRAM_BYTES: usize = 2048;
4919
4920const GRAM_WINDOW: usize = 256 << 10;
4922
4923fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4926 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4927 let mut first = original ^ (original >> 16);
4928 first = first.wrapping_mul(0x7feb_352d);
4929 first ^= first >> 15;
4930 let mut second = original ^ (original >> 17);
4931 second = second.wrapping_mul(0x846c_a68b);
4932 second ^= second >> 16;
4933 let mask = width * 8 - 1;
4934 [(first as usize) & mask, (second as usize) & mask]
4935}
4936
4937const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4958
4959#[derive(Debug)]
4966enum Lengths {
4967 Narrow(Vec<u16>),
4969 Wide(Vec<u32>),
4971}
4972
4973impl Lengths {
4974 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4977 match self {
4978 Lengths::Narrow(lens) => into.extend(
4979 indices
4980 .iter()
4981 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4982 ),
4983 Lengths::Wide(lens) => into.extend(
4984 indices
4985 .iter()
4986 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4987 ),
4988 }
4989 }
4990
4991 fn footprint(&self) -> usize {
4993 match self {
4994 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4995 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4996 }
4997 }
4998}
4999
5000fn lengths_of(ends: &[u32]) -> Option<Lengths> {
5009 match lengths_as::<u16>(ends)? {
5010 Some(narrow) => Some(Lengths::Narrow(narrow)),
5011 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
5012 }
5013}
5014
5015fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
5018 let mut lens = Vec::with_capacity(ends.len());
5019 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
5020 let mut start = 0;
5021 for &end in block {
5022 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
5023 return Some(None);
5024 };
5025 lens.push(len);
5026 start = end;
5027 }
5028 }
5029 Some(Some(lens))
5030}
5031
5032const TEXT_OFFSET_RUN: usize = 512;
5039
5040const DICTIONARY_HEADER: usize = 16;
5043
5044const DICTIONARY_SCATTERED: u32 = 1 << 31;
5058const DICTIONARY_GRAMS: u32 = 1 << 30;
5060const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
5063const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
5065
5066const TEXT_RANK_BLOCK: usize = 512;
5077
5078const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
5092
5093impl NativeText {
5094 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
5101 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
5102 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
5103 Ok(Some(bytes.as_slice()))
5104 }
5105
5106 fn block_chars(&self, block: usize) -> Result<&[u32]> {
5113 let slot = self
5114 .char_lens
5115 .get(block)
5116 .ok_or_else(|| invalid("a block past the global dictionary"))?;
5117 if let Some(lens) = slot.get() {
5118 return Ok(lens);
5119 }
5120 let decoded;
5121 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5122 Some(Ok(kept)) => kept,
5123 _ => {
5124 decoded = self.decode_block(block)?;
5125 &decoded
5126 }
5127 };
5128 let first = block * TEXT_PAYLOAD_VALUES;
5129 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5130 let ends = self.ends_within(first, last)?;
5131 if ends.len() != last - first {
5132 return Err(invalid("global dictionary offsets are short"));
5133 }
5134 let mut lens = Vec::with_capacity(ends.len());
5135 let mut start = u64::from(self.start_within(first)?);
5136 for &end in &ends {
5137 let value = usize::try_from(start)
5138 .ok()
5139 .zip(usize::try_from(end).ok())
5140 .and_then(|(from, to)| bytes.get(from..to))
5141 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5142 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5145 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5146 start = end;
5147 }
5148 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5149 }
5150
5151 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5156 let len = self.lengths[block];
5157 let mut stored = vec![
5158 0;
5159 usize::try_from(len).map_err(|_| invalid(
5160 "global dictionary block does not fit in memory"
5161 ))?
5162 ];
5163 read_at(&self.file, self.starts[block], &mut stored)?;
5164 if checksum(&stored) != self.hashes[block] {
5165 return Err(invalid("global dictionary payload checksum differs"));
5166 }
5167 let first = block * TEXT_PAYLOAD_VALUES;
5168 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5169 let want = self.end_within(last - 1)? as usize;
5170 let values = string::decode_flat(&stored)?;
5171 if values.len() != last - first {
5172 return Err(invalid("global dictionary block holds the wrong value count"));
5173 }
5174 let bytes = values.into_bytes();
5175 if bytes.len() != want {
5176 return Err(invalid("global dictionary block decodes to the wrong length"));
5177 }
5178 Ok(bytes)
5179 }
5180
5181 fn loaned_block<'a>(
5190 &'a self,
5191 block: usize,
5192 decoded: &'a mut Vec<u8>,
5193 scattered: bool,
5194 ) -> Result<&'a [u8]> {
5195 let kept = self.blocks.get(block).and_then(OnceLock::get);
5196 if let Some(Ok(kept)) = kept {
5197 return Ok(kept);
5198 }
5199 let again = kept.is_none()
5200 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5201 let keep = again
5202 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5203 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5204 if keep {
5205 let kept = self
5206 .payload_block(block)?
5207 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5208 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5209 return Ok(kept);
5210 }
5211 *decoded = self.decode_block(block)?;
5212 if scattered && again {
5213 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5214 }
5215 Ok(decoded)
5216 }
5217
5218 fn ends_worth_unpacking(&self) -> usize {
5235 self.values.max(TEXT_PAYLOAD_VALUES)
5236 }
5237
5238 fn value_ends(&self) -> Option<&[u32]> {
5240 if let Some(built) = self.value_ends.get() {
5241 return built.as_deref();
5242 }
5243 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5244 return None;
5245 }
5246 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5247 }
5248
5249 fn unpack_ends(&self) -> Option<Vec<u32>> {
5255 let mut ends = vec![0u32; self.values];
5256 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5257 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5258 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5259 u32::try_from(bits).unwrap_or(u32::MAX)
5260 })
5261 .ok()?;
5262 }
5263 if ends.contains(&u32::MAX) { None } else { Some(ends) }
5266 }
5267
5268 fn packed(&self) -> &[u8] {
5270 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5271 }
5272
5273 fn end_within(&self, index: usize) -> Result<u32> {
5275 if let Some(ends) = self.value_ends() {
5276 return ends
5277 .get(index)
5278 .copied()
5279 .ok_or_else(|| invalid("global dictionary offsets are short"));
5280 }
5281 self.packed_end(index)
5282 }
5283
5284 fn packed_end(&self, index: usize) -> Result<u32> {
5286 let run = index / TEXT_OFFSET_RUN;
5287 let bytes = self
5288 .packed()
5289 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5290 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5291 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5292 .map_err(|_| invalid("global dictionary offsets are short"))?;
5293 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5294 }
5295
5296 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5314 let mut ends = vec![0u64; last.saturating_sub(first)];
5315 let mut scratch = Vec::new();
5316 let mut at = first;
5317 while at < last {
5318 let run = at / TEXT_OFFSET_RUN;
5319 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5320 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5321 let bytes = self
5322 .packed()
5323 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5324 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5325 let from = at % TEXT_OFFSET_RUN;
5326 let upto = stop - run * TEXT_OFFSET_RUN;
5327 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5328 return Err(invalid("global dictionary offsets are short"));
5329 }
5330 let into = &mut ends[at - first..stop - first];
5331 if from == 0 {
5332 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5333 .map_err(|_| invalid("global dictionary offsets are short"))?;
5334 } else {
5335 scratch.resize(held, 0);
5336 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5337 .map_err(|_| invalid("global dictionary offsets are short"))?;
5338 into.copy_from_slice(&scratch[from..upto]);
5339 }
5340 at = stop;
5341 }
5342 Ok(ends)
5343 }
5344
5345 fn start_within(&self, index: usize) -> Result<u32> {
5348 if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5349 }
5350
5351 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5359 if let Some(ends) = self.value_ends() {
5360 let end =
5361 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5362 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5365 if start > end {
5366 return Err(invalid("global dictionary value ends before it starts"));
5367 }
5368 return Ok((start, end));
5369 }
5370 self.packed_span(index)
5371 }
5372
5373 fn packed_span(&self, index: usize) -> Result<(u32, u32)> {
5375 let within = index % TEXT_OFFSET_RUN;
5376 let (start, end) = if within == 0 {
5377 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) {
5378 0
5379 } else {
5380 self.packed_end(index - 1)?
5381 };
5382 (start, self.packed_end(index)?)
5383 } else {
5384 let run = index / TEXT_OFFSET_RUN;
5385 let bytes = self
5386 .packed()
5387 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5388 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5389 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5390 .map_err(|_| invalid("global dictionary offsets are short"))?;
5391 let ends = u32::try_from(end)
5392 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5393 let starts = u32::try_from(start)
5394 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5395 (starts, ends)
5396 };
5397 if start > end {
5398 return Err(invalid("global dictionary value ends before it starts"));
5399 }
5400 Ok((start, end))
5401 }
5402
5403 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5410 let slot = self
5411 .rank_blocks
5412 .get(rank / TEXT_RANK_BLOCK)
5413 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5414 let block = slot
5415 .get_or_init(|| {
5416 let mut bytes = Vec::new();
5417 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5418 Ok(bytes)
5419 })
5420 .as_ref()
5421 .map_err(Clone::clone)?;
5422 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5423 }
5424
5425 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5428 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5429 let end = self.rank_ends[which];
5430 bytes.clear();
5431 bytes.resize((end - start) as usize, 0);
5432 read_at(&self.file, self.rank_at + start, bytes)?;
5433 let expected = self
5434 .rank_hashes
5435 .get(which)
5436 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5437 if checksum(bytes) != *expected {
5438 return Err(invalid("global dictionary rank checksum differs"));
5439 }
5440 Ok(())
5441 }
5442
5443 fn head_at(&self, rank: usize) -> Result<u64> {
5445 let (block, within) = self.rank_parts(rank)?;
5446 let (base, width, packed) = rank_heads(block)?;
5447 let above = bitpack::tail_at(packed, width, within)
5448 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5449 Ok(base.wrapping_add(above))
5450 }
5451
5452 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5454 let (_, width, packed) = rank_heads(block)?;
5455 packed
5456 .get(bitpack::tail_len(count, width)..)
5457 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5458 }
5459
5460 fn rank_block_len(&self, rank: usize) -> usize {
5462 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5463 TEXT_RANK_BLOCK.min(self.ranks - first)
5464 }
5465}
5466
5467fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5469 let header = block
5470 .get(..RANK_BLOCK_HEADER)
5471 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5472 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5473 let width = header[8] as usize;
5474 if width > 64 {
5475 return Err(invalid("global dictionary rank block packs heads past a word"));
5476 }
5477 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5478}
5479
5480fn offset_width(ends: &[u32]) -> usize {
5487 let span = ends.iter().copied().max().unwrap_or(0);
5491 (u32::BITS - span.leading_zeros()) as usize
5492}
5493
5494fn offset_bytes(values: usize, bits: usize) -> usize {
5497 let full = values / TEXT_OFFSET_RUN;
5498 let rest = values % TEXT_OFFSET_RUN;
5499 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5500}
5501
5502fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5506 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5507 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5508 run.clear();
5509 run.extend(chunk.iter().map(|&end| u64::from(end)));
5510 bitpack::pack_tail(&run, bits, out)
5511 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5512 }
5513 Ok(())
5514}
5515
5516fn code_width(values: usize) -> usize {
5518 match u64::try_from(values).unwrap_or(u64::MAX) {
5519 0 | 1 => 0,
5520 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5521 }
5522}
5523
5524impl TextSource for NativeText {
5525 fn len(&self) -> usize {
5526 self.values
5527 }
5528
5529 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5530 let Some(grams) = &self.grams else { return Ok(true) };
5531 if literal.len() < 4 || first >= self.values {
5532 return Ok(true);
5533 }
5534 let verdict = grams.verdicts(&self.file, literal)?;
5535 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5536 }
5537
5538 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5539 if index >= self.values {
5540 return Ok(None);
5541 }
5542 let (start, end) = self.span_within(index)?;
5543 if start == end {
5544 return Ok(Some(&[]));
5545 }
5546 let block = index / TEXT_PAYLOAD_VALUES;
5549 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5550 Ok(bytes.get(start as usize..end as usize))
5551 }
5552
5553 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5554 if index >= self.values {
5555 return Ok(None);
5556 }
5557 let (start, end) = self.span_within(index)?;
5558 Ok(Some((end - start) as usize))
5559 }
5560
5561 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5568 into.reserve(indices.len());
5569 if let Some(Some(lens)) = self.value_lens.get() {
5572 lens.extend_at(indices, into);
5573 return Ok(());
5574 }
5575 let asked = self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed) + indices.len();
5580 let lens = match self.value_ends.get() {
5581 Some(Some(ends)) => self.value_lens.get_or_init(|| lengths_of(ends)).as_ref(),
5582 _ if asked >= self.ends_worth_unpacking() => self
5583 .value_lens
5584 .get_or_init(|| self.unpack_ends().and_then(|ends| lengths_of(&ends)))
5585 .as_ref(),
5586 _ => None,
5587 };
5588 if let Some(lens) = lens {
5589 lens.extend_at(indices, into);
5590 return Ok(());
5591 }
5592 for &index in indices {
5593 let index = index as usize;
5594 if index >= self.values {
5596 into.push(0);
5597 continue;
5598 }
5599 let (start, end) = self.packed_span(index)?;
5600 into.push(i64::from(end - start));
5601 }
5602 Ok(())
5603 }
5604
5605 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5608 into.reserve(indices.len());
5609 for &index in indices {
5610 let index = index as usize;
5611 if index >= self.values {
5613 into.push(0);
5614 continue;
5615 }
5616 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5617 let len = lens
5618 .get(index % TEXT_PAYLOAD_VALUES)
5619 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5620 into.push(i64::from(*len));
5621 }
5622 Ok(())
5623 }
5624
5625 fn sweep(
5638 &self,
5639 first: usize,
5640 limit: usize,
5641 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5642 ) -> Result<usize> {
5643 let limit = limit.min(self.values);
5644 if first >= limit {
5645 return Ok(first);
5646 }
5647 let block = first / TEXT_PAYLOAD_VALUES;
5648 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5649 let mut decoded = Vec::new();
5650 let bytes = self.loaned_block(block, &mut decoded, false)?;
5651 let ends = self.ends_within(first, last)?;
5652 if ends.len() != last - first {
5653 return Err(invalid("global dictionary offsets are short"));
5654 }
5655 let mut start = u64::from(self.start_within(first)?);
5656 for (index, &end) in (first..last).zip(&ends) {
5659 let value = usize::try_from(start)
5660 .ok()
5661 .zip(usize::try_from(end).ok())
5662 .and_then(|(from, to)| bytes.get(from..to))
5663 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5664 body(index, value)?;
5665 start = end;
5666 }
5667 Ok(last)
5668 }
5669
5670 fn visit_at(
5679 &self,
5680 indices: &[u32],
5681 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5682 ) -> Result<()> {
5683 let mut order = (0..indices.len()).collect::<Vec<_>>();
5684 order.sort_unstable_by_key(|&at| indices[at]);
5685 let block_of = |at: usize| {
5686 let index = indices[at] as usize;
5687 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5688 };
5689 let mut decoded = Vec::new();
5690 let mut run = 0;
5691 while run < order.len() {
5692 let Some(block) = block_of(order[run]) else {
5693 for &at in &order[run..] {
5695 body(at, &[])?;
5696 }
5697 break;
5698 };
5699 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5700 let bytes = self.loaned_block(block, &mut decoded, true)?;
5701 for &at in &order[run..upto] {
5702 let (start, end) = self.span_within(indices[at] as usize)?;
5703 let value = bytes
5704 .get(start as usize..end as usize)
5705 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5706 body(at, value)?;
5707 }
5708 run = upto;
5709 }
5710 Ok(())
5711 }
5712
5713 fn visit(
5719 &self,
5720 indices: &[usize],
5721 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5722 ) -> Result<()> {
5723 let mut at = 0;
5724 while at < indices.len() {
5725 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5726 let upto =
5727 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5728 let wanted = &indices[at..upto];
5729 if wanted.iter().any(|&index| index >= self.values) {
5730 return Err(invalid("a visited value is past the global dictionary"));
5731 }
5732 let decoded;
5733 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5734 Some(Ok(kept)) => kept,
5735 _ => {
5736 decoded = self.decode_block(block)?;
5737 &decoded
5738 }
5739 };
5740 for (offset, &index) in wanted.iter().enumerate() {
5741 let (start, end) = self.span_within(index)?;
5742 let value = bytes
5743 .get(start as usize..end as usize)
5744 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5745 body(at + offset, value)?;
5746 }
5747 at = upto;
5748 }
5749 Ok(())
5750 }
5751
5752 fn ranks(&self) -> Option<usize> {
5753 (self.ranks > 0).then_some(self.ranks)
5754 }
5755
5756 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5764 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5765 if let Some(&answer) = memo.get(wanted) {
5766 return Ok(answer);
5767 }
5768 let answer = search_below(self, ranks, wanted)?;
5769 if memo.len() >= TEXT_SEARCH_MEMO {
5770 memo.clear();
5771 }
5772 memo.insert(wanted.to_vec(), answer);
5773 Ok(answer)
5774 }
5775
5776 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5777 let settled = self.head_at(rank)?.cmp(&head(wanted));
5781 if settled != Ordering::Equal {
5782 return Ok(settled);
5783 }
5784 let code = self.code_at_rank(rank)?;
5785 let bytes = self
5786 .bytes_at(code as usize)?
5787 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5788 Ok(bytes.cmp(wanted))
5789 }
5790
5791 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5792 let (block, within) = self.rank_parts(rank)?;
5793 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5794 let code = bitpack::tail_at(codes, self.code_bits, within)
5795 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5796 let code = u32::try_from(code)
5797 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5798 if code as usize >= self.len() {
5799 return Err(invalid("global dictionary order names a code it does not have"));
5800 }
5801 Ok(code)
5802 }
5803
5804 fn code_ranks(&self) -> Option<&[u32]> {
5805 if self.ranks == 0 || self.ranks != self.len() {
5809 return None;
5810 }
5811 self.code_ranks
5812 .get_or_init(|| {
5813 let mut ranks = vec![u32::MAX; self.ranks];
5814 let mut scratch = Vec::new();
5822 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5823 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5824 let which = first / TEXT_RANK_BLOCK;
5825 let block = match self.rank_blocks.get(which)?.get() {
5826 Some(kept) => kept.as_ref().ok()?.as_slice(),
5827 None => {
5828 self.read_rank_block(which, &mut scratch).ok()?;
5829 scratch.as_slice()
5830 }
5831 };
5832 let count = self.rank_block_len(first);
5833 let packed = self.rank_codes(block, count).ok()?;
5834 let codes = codes.get_mut(..count)?;
5835 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5836 for (within, &code) in codes.iter().enumerate() {
5837 let code = usize::try_from(code).ok()?;
5838 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5839 }
5840 }
5841 if ranks.contains(&u32::MAX) {
5842 return None;
5843 }
5844 Some(ranks)
5845 })
5846 .as_deref()
5847 }
5848
5849 fn footprint(&self) -> usize {
5850 self.offsets.capacity()
5851 + self
5852 .value_ends
5853 .get()
5854 .and_then(Option::as_ref)
5855 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5856 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5857 + self
5858 .code_ranks
5859 .get()
5860 .and_then(Option::as_ref)
5861 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5862 + self.rank_hashes.capacity() * size_of::<u64>()
5863 + self.rank_ends.capacity() * size_of::<u64>()
5864 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5865 + self
5866 .rank_blocks
5867 .iter()
5868 .filter_map(OnceLock::get)
5869 .filter_map(|result| result.as_ref().ok())
5870 .map(Vec::capacity)
5871 .sum::<usize>()
5872 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5873 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5874 + self
5875 .char_lens
5876 .iter()
5877 .filter_map(OnceLock::get)
5878 .map(|lens| lens.len() * size_of::<u32>())
5879 .sum::<usize>()
5880 + self.hashes.capacity() * size_of::<u64>()
5881 + self.starts.capacity() * size_of::<u64>()
5882 + self.lengths.capacity() * size_of::<u64>()
5883 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5884 + self
5885 .blocks
5886 .iter()
5887 .filter_map(OnceLock::get)
5888 .filter_map(|result| result.as_ref().ok())
5889 .map(Vec::capacity)
5890 .sum::<usize>()
5891 }
5892}
5893
5894fn places(table: &Table) -> Result<Vec<Place>> {
5896 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5897 for (at, stripe) in table.stripes.iter().enumerate() {
5898 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5899 for (part, &rows) in stripe.parts.iter().enumerate() {
5900 places.push(Place {
5901 stripe: index,
5902 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5903 rows,
5904 });
5905 }
5906 }
5907 Ok(places)
5908}
5909
5910fn read_index<F: Positional + ?Sized>(
5915 file: &F,
5916 stripe: &Stripe,
5917 column: usize,
5918) -> Result<Vec<PartSpan>> {
5919 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5920 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5921}
5922
5923fn read_index_span<F: Positional + ?Sized>(
5924 file: &F,
5925 index: Span,
5926 page: Span,
5927 parts: usize,
5928 column: usize,
5929) -> Result<Vec<PartSpan>> {
5930 let section = index_section(parts)?;
5931 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5932 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5933 if end > index.length as usize {
5934 return Err(invalid("index page is shorter than its columns"));
5935 }
5936 let mut bytes = vec![0; section];
5937 let offset =
5938 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5939 read_at(file, offset, &mut bytes)?;
5940 let entries = section - size_of::<u64>();
5941 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5942 if checksum(&bytes[..entries]) != stored {
5943 return Err(invalid(&format!(
5946 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5947 wanted {stored:016x} and got {:016x}",
5948 checksum(&bytes[..entries]),
5949 )));
5950 }
5951 let mut spans = Vec::with_capacity(parts);
5952 let mut start = 0_usize;
5953 for part in 0..parts {
5954 let at = part * INDEX_ENTRY;
5955 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5956 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5957 spans.push(PartSpan { start, length, hash });
5958 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5959 }
5960 if start != page.length as usize {
5961 return Err(invalid("column page length differs from its index"));
5962 }
5963 Ok(spans)
5964}
5965
5966fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5968 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5969 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5970}
5971
5972fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5975 if bits.is_empty() {
5976 bits.resize(parts.div_ceil(64).max(1), 0);
5977 }
5978 let (word, bit) = (part / 64, 1_u64 << (part % 64));
5979 let Some(held) = bits.get_mut(word) else { return (false, false) };
5980 let again = *held & bit != 0;
5981 *held |= bit;
5982 let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5983 (again, again && through)
5984}
5985
5986fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5992 if let Some(slot) = cached.index.get_mut(held.stripe)
5993 && slot.is_none()
5994 {
5995 *slot = Some(Arc::clone(&held.index));
5996 }
5997 let page = held.page.clone()?;
5998 let slot = cached.pages.get_mut(held.stripe)?;
5999 if slot.is_some() {
6000 return None;
6001 }
6002 let bytes = page.bytes().len();
6003 let used = Arc::new(AtomicBool::new(true));
6006 *slot = Some(Resident { page, used: Arc::clone(&used) });
6007 Some((bytes, used))
6008}
6009
6010#[derive(Debug, Clone)]
6019pub struct Catalog {
6020 file: Arc<File>,
6021 size: u64,
6022 map: Option<Arc<Mapped>>,
6025 entries: Arc<Vec<Entry>>,
6026 views: Arc<Vec<ViewEntry>>,
6028 anchor: Option<Arc<LogAnchor>>,
6030 opening: Opening,
6031 pool: PagePool,
6033}
6034
6035#[derive(Debug, Clone, PartialEq, Eq)]
6037pub struct CertifiedSums {
6038 pub columns: Vec<(i128, u64)>,
6039 pub rows: u64,
6040}
6041
6042#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6044pub enum IntegerExtremes {
6045 Null,
6046 Values { low: i128, high: i128 },
6047}
6048
6049pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
6051
6052impl Catalog {
6053 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6062 Self::open_in(path, &PagePool::default())
6063 }
6064
6065 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
6071 let path = path.as_ref();
6072 let (file, size, _, bytes, opening) = slot_bytes(path)?;
6073 let (entries, views, card, anchor) = decode_catalog(&bytes, size)?;
6074 remember_card(path, card.as_ref());
6075 let map = Mapped::open(&file, size).map(Arc::new);
6076 Ok(Self {
6077 anchor: anchor.map(Arc::new),
6078 file: Arc::new(file),
6079 size,
6080 map,
6081 entries: Arc::new(entries),
6082 views: Arc::new(views),
6083 opening,
6084 pool: pool.clone(),
6085 })
6086 }
6087
6088 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
6090 self.entries.iter().map(|entry| entry.name.as_str())
6091 }
6092
6093 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
6100 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
6101 }
6102
6103 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
6109 self.views.iter()
6110 }
6111
6112 #[must_use]
6114 pub fn log_anchor(&self) -> Option<&LogAnchor> {
6115 self.anchor.as_deref()
6116 }
6117
6118 #[must_use]
6120 pub fn len(&self) -> usize {
6121 self.entries.len()
6122 }
6123
6124 #[must_use]
6127 pub fn is_empty(&self) -> bool {
6128 self.entries.is_empty()
6129 }
6130
6131 pub fn table(&self, name: &str) -> Result<Reader> {
6137 let entry = self
6138 .entries
6139 .iter()
6140 .find(|entry| entry.name == name)
6141 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6142 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6146 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6147 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6148 }
6149 let mut opening = self.opening;
6150 opening.reads += 1;
6151 opening.bytes += u64::from(entry.directory.length);
6152 Reader::build(
6153 Arc::clone(&self.file),
6154 self.map.clone(),
6155 self.size,
6156 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6157 u64::from(entry.directory.length),
6158 opening,
6159 self.pool.clone(),
6160 )
6161 }
6162
6163 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6171 let mut counts = BTreeMap::<i64, u64>::new();
6172 let Some(()) = self.integer_fold(name, column, |value, count| {
6173 let held = counts.entry(value).or_default();
6174 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6175 Ok(())
6176 })?
6177 else {
6178 return Ok(None);
6179 };
6180 Ok(Some(counts.into_iter().collect()))
6181 }
6182
6183 pub fn integer_fold(
6190 &self,
6191 name: &str,
6192 column: usize,
6193 mut emit: impl FnMut(i64, u64) -> Result<()>,
6194 ) -> Result<Option<()>> {
6195 let entry = self
6196 .entries
6197 .iter()
6198 .find(|entry| entry.name == name)
6199 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6200 let field =
6201 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6202 if !signed_integer(&field.ty) {
6203 return Ok(None);
6204 }
6205 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6206 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6207 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6208 }
6209 quick_integer_fold(
6210 &self.file,
6211 Cursor::over(&self.file, offset, length),
6212 entry,
6213 self.size,
6214 column,
6215 &mut emit,
6216 )?;
6217 Ok(Some(()))
6218 }
6219
6220 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6225 let entry = self
6226 .entries
6227 .iter()
6228 .find(|entry| entry.name == name)
6229 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6230 let Some(field) = entry.fields.get(column) else {
6231 return Err(invalid("frequency column index out of range"));
6232 };
6233 if !matches!(
6234 field.ty,
6235 LogicalType::TinyInt
6236 | LogicalType::SmallInt
6237 | LogicalType::Integer
6238 | LogicalType::BigInt
6239 | LogicalType::UTinyInt
6240 | LogicalType::USmallInt
6241 | LogicalType::UInteger
6242 | LogicalType::UBigInt
6243 ) {
6244 return Ok(None);
6245 }
6246 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6247 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6248 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6249 }
6250 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6251 return frequencies
6252 .iter()
6253 .filter(|(value, _)| value.is_some_and(|value| value != 0))
6254 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6255 .map(Some)
6256 .ok_or_else(|| invalid("numeric frequency count overflow"));
6257 }
6258 quick_nonzero(
6259 Cursor::over(&self.file, offset, length),
6260 &entry.name,
6261 &entry.fields,
6262 entry.rows,
6263 column,
6264 )
6265 }
6266
6267 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6270 let entry = self
6271 .entries
6272 .iter()
6273 .find(|entry| entry.name == name)
6274 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6275 let mut sums = Vec::with_capacity(columns.len());
6276 for &column in columns {
6277 let Some(field) = entry.fields.get(column) else {
6278 return Err(invalid("aggregate column index out of range"));
6279 };
6280 if !signed_integer(&field.ty) {
6281 return Ok(None);
6282 }
6283 let Some(sum) = entry.aggregates[column] else {
6284 return Ok(None);
6285 };
6286 sums.push(sum);
6287 }
6288 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6289 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6290 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6291 }
6292 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6293 }
6294
6295 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6297 let entry = self
6298 .entries
6299 .iter()
6300 .find(|entry| entry.name == name)
6301 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6302 let Some(count) = entry.distincts.get(column).copied() else {
6303 return Err(invalid("distinct column index out of range"));
6304 };
6305 let Some(count) = count else { return Ok(None) };
6306 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6307 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6308 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6309 }
6310 Ok(Some(count))
6311 }
6312
6313 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6315 let entry = self
6316 .entries
6317 .iter()
6318 .find(|entry| entry.name == name)
6319 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6320 let Some(extremes) = entry.extremes.get(column).copied() else {
6321 return Err(invalid("extremes column index out of range"));
6322 };
6323 let Some(extremes) = extremes else { return Ok(None) };
6324 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6325 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6326 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6327 }
6328 Ok(Some(match extremes {
6329 None => IntegerExtremes::Null,
6330 Some((low, high)) => IntegerExtremes::Values { low, high },
6331 }))
6332 }
6333
6334 pub fn exact_numeric_frequencies(
6336 &self,
6337 name: &str,
6338 column: usize,
6339 ) -> Result<Option<NumericFrequencies>> {
6340 let entry = self
6341 .entries
6342 .iter()
6343 .find(|entry| entry.name == name)
6344 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6345 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6346 return Err(invalid("numeric frequency column index out of range"));
6347 };
6348 let Some(frequencies) = frequencies else { return Ok(None) };
6349 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6350 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6351 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6352 }
6353 Ok(Some(frequencies))
6354 }
6355
6356 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6358 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6359 }
6360}
6361
6362fn slot_offset(generation: u64) -> u64 {
6367 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6368}
6369
6370fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6375 let file = File::open(path).map_err(io)?;
6376 let size = file.metadata().map_err(io)?.len();
6377 let (slot, bytes, opening) = committed_slot(&file, size)?;
6378 Ok((file, size, slot, bytes, opening))
6379}
6380
6381fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6387 if size < HEADER {
6388 return Err(invalid("file is shorter than its header"));
6389 }
6390 let mut header = [0; HEADER as usize];
6391 read_at(file, 0, &mut header)?;
6392 let mut opening = Opening { reads: 1, bytes: HEADER };
6393 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6394 if &header[..8] != MAGIC {
6399 return Err(invalid("the header does not begin with a rudb native magic"));
6400 }
6401 if !READABLE.contains(&version) {
6402 return Err(invalid(&format!(
6403 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6404 be written again"
6405 )));
6406 }
6407 let mut selected = None;
6408 for start in [16, 16 + SLOT_BYTES] {
6409 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6410 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6411 continue;
6412 }
6413 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6414 if slot.offset < HEADER || end > size {
6415 continue;
6416 }
6417 let mut bytes = vec![0; slot.length as usize];
6418 read_at(file, slot.offset, &mut bytes)?;
6419 opening.reads += 1;
6420 opening.bytes += u64::from(slot.length);
6421 if checksum(&bytes) == slot.hash
6422 && selected
6423 .as_ref()
6424 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6425 {
6426 selected = Some((slot, bytes));
6427 }
6428 }
6429 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6430 Ok((slot, bytes, opening))
6431}
6432
6433impl Reader {
6434 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6441 let catalog = Catalog::open(path)?;
6442 let mut names = catalog.names();
6443 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6444 if names.next().is_some() {
6445 return Err(invalid(
6446 "the file holds more than one table, so it has to be opened by name",
6447 ));
6448 }
6449 catalog.table(&name)
6450 }
6451
6452 fn build(
6454 file: Arc<File>,
6455 map: Option<Arc<Mapped>>,
6456 size: u64,
6457 table: Table,
6458 directory: u64,
6459 opening: Opening,
6460 pool: PagePool,
6461 ) -> Result<Self> {
6462 let places = places(&table)?;
6463 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6464 let table_fields = table.fields.len();
6465 let columns = (0..table_fields).map(|_| Mutex::new(Cached::default())).collect::<Vec<_>>();
6466 let cache = Shelf {
6467 columns,
6468 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6469 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6470 };
6471 let sieves = (0..table_fields).map(|_| OnceLock::new()).collect();
6472 let part_ranges = (0..table_fields).map(|_| OnceLock::new()).collect();
6473 let verified = (places.len() * table_fields).div_ceil(64);
6474 let unreleased = table
6475 .stripes
6476 .iter()
6477 .flat_map(|stripe| {
6478 let parts = u32::try_from(stripe.parts.len()).unwrap_or(u32::MAX);
6479 (0..table_fields).map(move |_| AtomicU32::new(parts))
6480 })
6481 .collect();
6482 let firsts = places
6483 .iter()
6484 .scan(0, |first, place| {
6485 let at = *first;
6486 *first += place.rows as usize;
6487 Some(at)
6488 })
6489 .collect();
6490 Ok(Self {
6491 file,
6492 map,
6493 table: Arc::new(table),
6494 dictionaries: Arc::new(dictionaries),
6495 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6496 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6497 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6498 frequency_heads: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6499 summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6500 facts: Arc::new(OnceLock::new()),
6501 opened: Arc::new(AtomicUsize::new(0)),
6502 sieves: Arc::new(sieves),
6503 part_ranges: Arc::new(part_ranges),
6504 places: Arc::new(places),
6505 cache: Arc::new(cache),
6506 pool,
6507 pages: Arc::new(AtomicUsize::new(0)),
6508 indexes: Arc::new(AtomicUsize::new(0)),
6509 verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6510 unreleased: Arc::new(unreleased),
6511 text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6512 firsts: Arc::new(firsts),
6513 graph: Arc::default(),
6514 size,
6515 directory,
6516 opening,
6517 })
6518 }
6519
6520 #[must_use]
6527 pub fn reads(&self) -> Reads {
6528 Reads {
6529 opening: self.opening,
6530 pages: self.pages.load(Atomic::Relaxed),
6531 indexes: self.indexes.load(Atomic::Relaxed),
6532 dictionaries: self.opened.load(Atomic::Relaxed),
6533 }
6534 }
6535
6536 #[must_use]
6541 pub fn layout(&self) -> Layout {
6542 let table = &self.table;
6543 let stripes = table.stripes.as_slice();
6544 let columns = table
6545 .fields
6546 .iter()
6547 .enumerate()
6548 .map(|(at, field)| ColumnLayout {
6549 name: field.name.clone(),
6550 kind: field.ty.to_string(),
6551 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6552 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6553 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6554 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6555 dictionary: dictionary_bytes(table, at),
6556 })
6557 .collect();
6558 Layout {
6559 file: self.size,
6560 rows: table.rows,
6561 stripes: stripes.len(),
6562 parts: self.places.len(),
6563 columns,
6564 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6565 directory: self.directory,
6566 header: HEADER,
6567 }
6568 }
6569
6570 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6587 let field = self
6588 .table
6589 .fields
6590 .get(column)
6591 .ok_or_else(|| invalid("stored column index out of range"))?;
6592 let mut stored = Vec::with_capacity(self.places.len());
6593 let mut row = 0;
6594 for (at, stripe) in self.table.stripes.iter().enumerate() {
6595 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6596 let index = read_index(&self.file, stripe, column)?;
6597 let mut bytes = vec![0; page.length as usize];
6598 read_at(&self.file, page.offset, &mut bytes)?;
6599 let ranges = self.stripe_part_ranges(at, column);
6600 for (part, &rows) in stripe.parts.iter().enumerate() {
6601 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6602 let held = part_bytes(&bytes, span)?;
6603 let range = ranges.and_then(|held| held.get(part));
6604 stored.push(StoredPart {
6605 stripe: at,
6606 part,
6607 row,
6608 rows: rows as usize,
6609 encoding: page_encoding(&field.ty, rows as usize, held),
6610 bytes: span.length as u64,
6611 page: page.offset,
6612 offset: span.start as u64,
6613 low: range
6614 .and_then(|range| range.low.clone())
6615 .and_then(|bound| bound.into_value(&field.ty)),
6616 high: range
6617 .and_then(|range| range.high.clone())
6618 .and_then(|bound| bound.into_value(&field.ty)),
6619 nulls: range.map(|range| range.nulls),
6620 });
6621 row += rows as usize;
6622 }
6623 }
6624 Ok(stored)
6625 }
6626
6627 #[must_use]
6629 pub fn parts(&self) -> usize {
6630 self.places.len()
6631 }
6632
6633 #[must_use]
6640 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6641 let mut runs = Vec::with_capacity(self.table.stripes.len());
6642 let mut start = 0;
6643 for stripe in &self.table.stripes {
6644 let end = start + stripe.parts.len();
6645 runs.push(start..end);
6646 start = end;
6647 }
6648 runs
6649 }
6650
6651 #[must_use]
6656 pub fn stripe_rows(&self, stripe: usize) -> usize {
6657 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6658 }
6659
6660 pub fn keep_stripes(&self, stripes: usize) {
6667 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6668 }
6669
6670 #[must_use]
6672 pub fn part_rows(&self, at: usize) -> usize {
6673 self.places.get(at).map_or(0, |place| place.rows as usize)
6674 }
6675
6676 #[must_use]
6678 pub fn table(&self) -> &Table {
6679 &self.table
6680 }
6681
6682 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6691 let field = self
6692 .table
6693 .fields
6694 .get(column)
6695 .ok_or_else(|| invalid("frequency column index out of range"))?;
6696 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6697 return Ok(None);
6698 };
6699 if top == 0 || entries.len() < top {
6700 return Ok(None);
6701 }
6702 let boundary = entries[top - 1].count;
6703 if boundary <= omitted_max {
6704 return Ok(None);
6705 }
6706 self.decode_frequencies(column, &field.ty, &entries).map(|values| Some(Vec::clone(&values)))
6707 }
6708
6709 pub fn top_pair_frequencies(
6717 &self,
6718 first: usize,
6719 second: usize,
6720 _top: usize,
6721 ) -> Result<Option<PairFrequencyCounts>> {
6722 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6723 return Err(invalid("pair frequency column index out of range"));
6724 }
6725 Ok(None)
6726 }
6727
6728 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6748 let Some(prefix) = self.frequency_prefix(column)? else {
6749 return Ok(None);
6750 };
6751 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6752 }
6753
6754 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6777 Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6778 entries: Vec::clone(&entries),
6779 omitted_max,
6780 }))
6781 }
6782
6783 pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6789 let field = self
6790 .table
6791 .fields
6792 .get(column)
6793 .ok_or_else(|| invalid("frequency column index out of range"))?;
6794 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6795 return Ok(None);
6796 };
6797 let entries = self.decode_frequencies(column, &field.ty, &entries)?;
6798 Ok(Some((entries, omitted_max)))
6799 }
6800
6801 fn frequency_head(&self, column: usize) -> Result<Option<(Cow<'_, [FrequencyEntry]>, u64)>> {
6804 let (span, count) = match self.table.frequencies.get(column) {
6805 None | Some(None) => return Ok(None),
6806 Some(Some(Frequencies::Held(summary))) => {
6807 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6808 }
6809 Some(Some(Frequencies::Stored { span, entries, .. })) => (span, *entries),
6810 };
6811 if let Some(summary) = self.frequency_summaries.get(column).and_then(OnceLock::get) {
6812 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6813 }
6814 let slot = self
6815 .frequency_heads
6816 .get(column)
6817 .ok_or_else(|| invalid("frequency column index out of range"))?;
6818 if slot.get().is_none() {
6819 let field = self
6820 .table
6821 .fields
6822 .get(column)
6823 .ok_or_else(|| invalid("frequency column index out of range"))?;
6824 let length = (span.length as usize).min(13 + count * 25);
6827 let mut bytes = vec![0; length];
6828 read_at(&self.file, span.offset, &mut bytes)?;
6829 let head = decode_summary_head(&mut Cursor::new(&bytes), field, self.table.rows)?
6830 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
6831 if head.0.len() != count {
6832 return Err(invalid("a stored synopsis differs from its directory span"));
6833 }
6834 let _ = slot.set(Arc::new(head));
6835 }
6836 let (entries, omitted_max) = slot.get().expect("the synopsis head was stored").as_ref();
6837 Ok(Some((Cow::Borrowed(entries), *omitted_max)))
6838 }
6839
6840 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6842 Ok(match self.table.frequencies.get(column) {
6843 None | Some(None) => None,
6844 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6845 Some(Some(Frequencies::Stored { span, values, entries })) => {
6846 let slot = self
6847 .frequency_summaries
6848 .get(column)
6849 .ok_or_else(|| invalid("frequency column index out of range"))?;
6850 if let Some(summary) = slot.get() {
6851 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6852 }
6853 let field = self
6854 .table
6855 .fields
6856 .get(column)
6857 .ok_or_else(|| invalid("frequency column index out of range"))?;
6858 let mut bytes = vec![0; span.length as usize];
6859 read_at(&self.file, span.offset, &mut bytes)?;
6860 let mut cur = Cursor::new(&bytes);
6861 let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6862 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6863 if !cur.done() || summary.entries.len() != *entries {
6864 return Err(invalid("a stored synopsis differs from its directory span"));
6865 }
6866 let _ = slot.set(Arc::new(summary));
6867 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6868 }
6869 })
6870 }
6871
6872 fn decode_frequencies(
6880 &self,
6881 column: usize,
6882 ty: &LogicalType,
6883 entries: &[FrequencyEntry],
6884 ) -> Result<Synopsis> {
6885 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6886 return Ok(Arc::clone(values));
6887 }
6888 let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6889 if let Some(slot) = self.frequency_values.get(column) {
6890 let _ = slot.set(Arc::clone(&values));
6891 }
6892 Ok(values)
6893 }
6894
6895 fn decode_frequencies_once(
6896 &self,
6897 column: usize,
6898 ty: &LogicalType,
6899 entries: &[FrequencyEntry],
6900 ) -> Result<Vec<(Value, u64)>> {
6901 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6902 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6903 return Err(invalid("frequency text count differs from its synopsis"));
6904 }
6905 let dictionary =
6906 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6907 let mut codes = entries
6908 .iter()
6909 .filter_map(|entry| match entry.value {
6910 FrequencyValue::Code(code) => Some(code as usize),
6911 _ => None,
6912 })
6913 .collect::<Vec<_>>();
6914 codes.sort_unstable();
6915 codes.dedup();
6916 let texts = match &dictionary {
6917 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6918 _ => Vec::new(),
6919 };
6920 let mut out = Vec::with_capacity(entries.len());
6921 for (entry_at, entry) in entries.iter().enumerate() {
6922 let value = match entry.value {
6923 FrequencyValue::Null => {
6924 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6925 return Err(invalid("a null frequency entry has text"));
6926 }
6927 Value::Null
6928 }
6929 FrequencyValue::Integer(value) => match *ty {
6930 LogicalType::TinyInt => Value::TinyInt(
6931 i8::try_from(value)
6932 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6933 ),
6934 LogicalType::UTinyInt => Value::UTinyInt(
6935 u8::try_from(value)
6936 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6937 ),
6938 LogicalType::USmallInt => Value::USmallInt(
6939 u16::try_from(value)
6940 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6941 ),
6942 LogicalType::UInteger => Value::UInteger(
6943 u32::try_from(value)
6944 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6945 ),
6946 LogicalType::UBigInt => Value::UBigInt(
6947 u64::try_from(value)
6948 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6949 ),
6950 LogicalType::SmallInt => Value::SmallInt(
6951 i16::try_from(value)
6952 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6953 ),
6954 LogicalType::Integer => Value::Integer(
6955 i32::try_from(value)
6956 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6957 ),
6958 LogicalType::BigInt => Value::BigInt(
6959 i64::try_from(value)
6960 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6961 ),
6962 LogicalType::Date => Value::Date(
6963 i32::try_from(value)
6964 .map_err(|_| invalid("frequency DATE is out of range"))?,
6965 ),
6966 LogicalType::Timestamp => Value::Timestamp(
6967 i64::try_from(value)
6968 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6969 ),
6970 _ => return Err(invalid("integer frequency belongs to another type")),
6971 },
6972 FrequencyValue::Code(code) => {
6973 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6974 if *ty == LogicalType::Blob {
6975 Value::Blob(text.clone())
6976 } else {
6977 Value::Varchar(
6978 String::from_utf8(text.clone())
6979 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6980 )
6981 }
6982 } else {
6983 if dictionary.is_none() {
6984 return Err(invalid("frequency code has no dictionary or stored text"));
6985 }
6986 let at = codes
6987 .binary_search(&(code as usize))
6988 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6989 texts[at].clone()
6990 }
6991 }
6992 };
6993 out.push((value, entry.count));
6994 }
6995 Ok(out)
6996 }
6997
6998 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
7008 let field = self
7009 .table
7010 .fields
7011 .get(column)
7012 .ok_or_else(|| invalid("frequency column index out of range"))?;
7013 let Some(summary) = self.frequency_summary(column)? else {
7014 return Ok(None);
7015 };
7016 if summary.ordinals.is_empty() {
7017 return Ok(None);
7018 }
7019 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
7020 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
7021 (
7022 entries.iter().map(|(value, _)| value.clone()).collect(),
7023 summary.ordinal_entries.clone(),
7024 )
7025 } else {
7026 (Vec::new(), Vec::new())
7027 };
7028 let stored = self.table.ordinal_bounds.get(column).copied().unwrap_or(0);
7031 Ok(Some(FrequencyOccurrences {
7032 omitted_max: summary.omitted_max.max(summary.ordinal_bound).max(stored),
7033 ordinals: summary.ordinals.clone(),
7034 anchors,
7035 anchor_indices,
7036 }))
7037 }
7038
7039 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
7065 self.table
7066 .distincts
7067 .get(column)
7068 .copied()
7069 .ok_or_else(|| invalid("distinct column index out of range"))
7070 }
7071
7072 pub fn null_count(&self, column: usize) -> Result<u64> {
7083 if column >= self.table.fields.len() {
7084 return Err(invalid("null count column index out of range"));
7085 }
7086 let mut nulls = 0_u64;
7087 for stripe in &self.table.stripes {
7088 let range = stripe
7089 .zone
7090 .column(column)
7091 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7092 nulls = nulls
7093 .checked_add(range.nulls as u64)
7094 .ok_or_else(|| invalid("null count overflow"))?;
7095 }
7096 Ok(nulls)
7097 }
7098
7099 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
7114 if self.null_count(column)? > 0 || self.demoted(column) {
7115 return Ok(None);
7116 }
7117 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
7118 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
7119 if ranks == 0 {
7120 return Ok(None);
7121 }
7122 let low = text_at_rank(&dictionary, 0)?;
7123 let high = text_at_rank(&dictionary, ranks - 1)?;
7124 Ok(Some((low, high)))
7125 }
7126
7127 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
7150 if column >= self.table.fields.len() {
7151 return Err(invalid("extremes column index out of range"));
7152 }
7153 let mut low: Option<Bound> = None;
7154 let mut high: Option<Bound> = None;
7155 for stripe in &self.table.stripes {
7156 let range = stripe
7157 .zone
7158 .column(column)
7159 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7160 if !range.exact {
7161 return Ok(None);
7162 }
7163 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
7168 if stripe.rows > range.nulls {
7169 return Ok(None);
7170 }
7171 continue;
7172 };
7173 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
7174 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
7175 }
7176 Ok(low.zip(high))
7177 }
7178
7179 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
7192 if column >= self.table.fields.len() {
7193 return Err(invalid("sum column index out of range"));
7194 }
7195 let mut total = 0_i128;
7196 let mut rows = 0_u64;
7197 for stripe in &self.table.stripes {
7198 let range = stripe
7199 .zone
7200 .column(column)
7201 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7202 let Some(part) = range.sum else { return Ok(None) };
7203 let Some(sum) = total.checked_add(part) else { return Ok(None) };
7204 total = sum;
7205 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7206 }
7207 Ok(Some((total, rows)))
7208 }
7209
7210 pub fn host_groups(
7212 &self,
7213 column: usize,
7214 _minimum_count: u64,
7215 ) -> Result<Option<Vec<host::HostEntry>>> {
7216 if column >= self.table.fields.len() {
7217 return Err(invalid("host group column index out of range"));
7218 }
7219 Ok(None)
7220 }
7221
7222 #[must_use]
7226 pub fn demoted(&self, column: usize) -> bool {
7227 self.table.demoted.get(column).copied().unwrap_or(false)
7228 }
7229
7230 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7239 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7240 if let Some(dictionary) = self.dictionaries[column].get() {
7241 return Ok(Some(Arc::clone(dictionary)));
7242 }
7243 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7244 if let Some(dictionary) = self.dictionaries[column].get() {
7245 return Ok(Some(Arc::clone(dictionary)));
7246 }
7247 self.opened.fetch_add(1, Atomic::Relaxed);
7248 let dictionary = Arc::new(open_global_dictionary(
7249 Arc::clone(&self.file),
7250 page,
7251 &self.table.fields[column].ty,
7252 TEXT_KEEP_BUDGET,
7253 )?);
7254 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7255 Ok(Some(dictionary))
7256 }
7257
7258 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7265 if of.extent_bytes == 0 {
7266 return Ok(Vec::new());
7267 }
7268 let mut bytes = vec![0; of.extent_bytes as usize];
7269 read_at(&self.file, of.extent_page, &mut bytes)?;
7270 if checksum(&bytes) != of.hash {
7271 return Err(invalid("a section's extent table does not checksum"));
7272 }
7273 let extents = section::decode_extents(&bytes)?;
7274 if extents.len() != of.extents as usize {
7275 return Err(invalid("a section's extent table is not the length the entry says"));
7276 }
7277 Ok(extents)
7278 }
7279
7280 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
7290 let mut bytes = Vec::new();
7291 self.extent_into(of, &mut bytes)?;
7292 Ok(bytes)
7293 }
7294
7295 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7297 bytes.resize(of.length as usize, 0);
7298 self.extent_in_place(of, bytes)
7299 }
7300
7301 fn extent_in_place(&self, of: §ion::Extent, bytes: &mut [u8]) -> Result<()> {
7303 let end = of
7304 .offset
7305 .checked_add(u64::from(of.length))
7306 .ok_or_else(|| invalid("an extent overflows the file"))?;
7307 if of.offset < HEADER || end > self.size || bytes.len() != of.length as usize {
7308 return Err(invalid("an extent is outside the file"));
7309 }
7310 read_at(&self.file, of.offset, bytes)?;
7311 if checksum(bytes) != of.hash {
7312 return Err(invalid("an extent does not checksum"));
7313 }
7314 Ok(())
7315 }
7316
7317 pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7330 let extents = self.extents(of)?;
7331 let Some(first) = extents.first() else { return Ok(Vec::new()) };
7332 let end = first
7333 .offset
7334 .checked_add(u64::from(first.length))
7335 .ok_or_else(|| invalid("an extent overflows the file"))?;
7336 if first.offset < HEADER || end > self.size {
7337 return Err(invalid("an extent is outside the file"));
7338 }
7339 let mut bytes = vec![0; len.min(first.length as usize)];
7340 read_at(&self.file, first.offset, &mut bytes)?;
7341 Ok(bytes)
7342 }
7343
7344 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7357 let extents = self.extents(of)?;
7358 let total = usize::try_from(sum(extents.iter().map(|one| u64::from(one.length))))
7359 .map_err(|_| invalid("a section longer than fits in memory"))?;
7360 let mut bytes = vec![0; total];
7361 let mut at = 0;
7362 for one in &extents {
7363 if one.first != at as u64 {
7364 return Err(invalid("a section's extents do not join up"));
7365 }
7366 let end = at + one.length as usize;
7367 self.extent_in_place(one, &mut bytes[at..end])?;
7368 at = end;
7369 }
7370 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7373 return Err(invalid("a section's header is longer than its payload"));
7374 }
7375 Ok(bytes)
7376 }
7377
7378 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7389 self.read_impl(part, columns, true, None)
7390 }
7391
7392 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7402 self.read_impl(part, columns, false, None)
7403 }
7404
7405 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7413 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7414 let field =
7415 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7416 if !matches!(
7417 field.ty,
7418 LogicalType::TinyInt
7419 | LogicalType::SmallInt
7420 | LogicalType::Integer
7421 | LogicalType::BigInt
7422 ) {
7423 return Ok(None);
7424 }
7425 let (rows, counts) = match self.with_part(place, column, |bytes| {
7426 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7427 return Ok(None);
7428 }
7429 integer::tally(&bytes[2..]).map(Some)
7430 })? {
7431 Some(tallied) => tallied,
7432 None => return Ok(None),
7433 };
7434 if rows != place.rows as usize {
7435 return Err(invalid("encoded integer part holds the wrong number of rows"));
7436 }
7437 for &(value, _) in &counts {
7438 let fits = match field.ty {
7439 LogicalType::TinyInt => i8::try_from(value).is_ok(),
7440 LogicalType::SmallInt => i16::try_from(value).is_ok(),
7441 LogicalType::Integer => i32::try_from(value).is_ok(),
7442 LogicalType::BigInt => true,
7443 _ => false,
7444 };
7445 if !fits {
7446 return Err(invalid("encoded integer value is outside its column type"));
7447 }
7448 }
7449 Ok(Some(counts))
7450 }
7451
7452 pub fn rows_holding(
7464 &self,
7465 part: usize,
7466 column: usize,
7467 sequence: &Sequence,
7468 negated: bool,
7469 ) -> Result<Option<Vec<u32>>> {
7470 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7471 let field =
7472 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7473 if field.ty != LogicalType::Varchar {
7474 return Ok(None);
7475 }
7476 let rows = place.rows as usize;
7477 self.with_part(place, column, |bytes| {
7478 if bytes.first() != Some(&6) {
7479 return Ok(None);
7480 }
7481 let mut cur = Cursor::new(bytes);
7482 cur.u8()?;
7483 let mask = match cur.u8()? {
7484 0 => None,
7485 1 => return Ok(Some(Vec::new())),
7486 2 => {
7487 let from = cur.at;
7488 cur.take(rows.div_ceil(8))?;
7489 Some(&bytes[from..cur.at])
7490 }
7491 _ => return Err(invalid("page validity tag differs")),
7492 };
7493 let needs = sequence.needs();
7496 let first = self.firsts.get(part).copied().unwrap_or_default();
7497 let sketch = self
7498 .text_grams
7499 .get(column)
7500 .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7501 .and_then(|words| words.get(first..first + rows));
7502 let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7503 let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7504 return Ok(None);
7505 };
7506 if held.len() != rows {
7507 return Err(invalid("compressed text page holds the wrong number of rows"));
7508 }
7509 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7510 Ok(Some(
7511 (0..rows)
7512 .filter(|&row| held[row] != negated && valid(row))
7513 .map(|row| row as u32)
7514 .collect(),
7515 ))
7516 })
7517 }
7518
7519 fn is_verified(&self, bit: usize) -> bool {
7522 self.verified
7523 .get(bit / 64)
7524 .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7525 }
7526
7527 fn set_verified(&self, bit: usize) {
7529 if let Some(word) = self.verified.get(bit / 64) {
7530 word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7531 }
7532 }
7533
7534 fn with_part<T>(
7537 &self,
7538 place: Place,
7539 column: usize,
7540 read: impl FnOnce(&[u8]) -> Result<T>,
7541 ) -> Result<T> {
7542 let stripe_index = place.stripe as usize;
7543 let stripe = self
7544 .table
7545 .stripes
7546 .get(stripe_index)
7547 .ok_or_else(|| invalid("stripe index out of range"))?;
7548 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7549 let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7550 let span = *held
7551 .index
7552 .get(place.part as usize)
7553 .ok_or_else(|| invalid("part index out of range"))?;
7554 match &held.page {
7555 Some(page) => read(page.part(place.part as usize, span)?),
7556 None => {
7557 let offset = page
7558 .offset
7559 .checked_add(span.start as u64)
7560 .ok_or_else(|| invalid("part range overflow"))?;
7561 let mut bytes = vec![0; span.length];
7562 read_at(&self.file, offset, &mut bytes)?;
7563 verify_part(&bytes, span)?;
7564 read(&bytes)
7565 }
7566 }
7567 }
7568
7569 pub fn read_rows(
7582 &self,
7583 part: usize,
7584 columns: &[usize],
7585 positions: &[u32],
7586 whole: bool,
7587 ) -> Result<Chunk> {
7588 self.read_impl(part, columns, whole, Some(positions))
7589 }
7590
7591 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7598 if self.demoted(column) {
7601 return Ok(false);
7602 }
7603 if candidates.is_empty() {
7604 return Ok(true);
7605 }
7606 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7607 return Err(Error::internal("native code candidates are not sorted and unique"));
7608 }
7609 let stripe = self.stripe_of(part)?;
7610 let Some(page) = stripe.memberships.get(column) else {
7611 return Ok(false);
7612 };
7613 let mut bytes = vec![0; page.length as usize];
7614 read_at(&self.file, page.offset, &mut bytes)?;
7615 if checksum(&bytes) != page.hash {
7616 return Err(invalid("membership page checksum differs"));
7617 }
7618 let codes = decode_membership(&bytes)?;
7619 let mut left = 0;
7620 let mut right = 0;
7621 while left < codes.len() && right < candidates.len() {
7622 match codes[left].cmp(&candidates[right]) {
7623 Ordering::Less => left += 1,
7624 Ordering::Greater => right += 1,
7625 Ordering::Equal => return Ok(false),
7626 }
7627 }
7628 Ok(true)
7629 }
7630
7631 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7632 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7633 self.table
7634 .stripes
7635 .get(place.stripe as usize)
7636 .ok_or_else(|| invalid("stripe index out of range"))
7637 }
7638
7639 fn held(
7656 &self,
7657 at: usize,
7658 part: usize,
7659 stripe: &Stripe,
7660 column: usize,
7661 whole: bool,
7662 ) -> Result<CachedColumn> {
7663 let cache =
7664 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7665 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7666 if cached.index.is_empty() {
7667 let stripes = self.table.stripes.len();
7668 cached.pages = (0..stripes).map(|_| None).collect();
7669 cached.index = vec![None; stripes];
7670 cached.touched = vec![Vec::new(); stripes];
7671 }
7672 let known = cached.index.get(at).and_then(Clone::clone);
7673 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7674 slot.used.store(true, Atomic::Relaxed);
7675 Arc::clone(&slot.page)
7676 });
7677 let (again, through) = match cached.touched.get_mut(at) {
7679 Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7680 _ => (false, false),
7681 };
7682 let whole = whole && again;
7683 if let Some(index) = known.clone()
7684 && (!whole || page.is_some())
7685 {
7686 return Ok(CachedColumn { stripe: at, index, page });
7687 }
7688 if cached.loading.contains(&at) {
7689 drop(cached);
7690 if let Some(index) = known {
7694 return Ok(CachedColumn { stripe: at, index, page: None });
7695 }
7696 let held = self.page_of(stripe, column, at, false, None)?;
7697 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7698 remember(&mut cached, &held);
7699 return Ok(held);
7700 }
7701 cached.loading.push(at);
7702 drop(cached);
7703
7704 let read = self.page_of(stripe, column, at, whole, known);
7705
7706 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7710 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7711 cached.loading.remove(position);
7712 }
7713 let held = read?;
7714 let taken = remember(&mut cached, &held);
7715 if taken.is_some() && !through {
7716 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7717 cached.passing.push_back(at);
7718 while cached.passing.len() > floor {
7719 let Some(old) = cached.passing.pop_front() else { break };
7720 if let Some(slot) = cached.pages.get_mut(old) {
7721 *slot = None;
7722 }
7723 }
7724 return Ok(held);
7725 }
7726 drop(cached);
7727 if let Some((bytes, used)) = taken {
7728 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7729 self.pool.admit(Held {
7730 shelf: Arc::downgrade(&self.cache),
7731 column,
7732 stripe: at,
7733 bytes,
7734 used,
7735 });
7736 }
7737 Ok(held)
7738 }
7739
7740 fn page_of(
7746 &self,
7747 stripe: &Stripe,
7748 column: usize,
7749 at: usize,
7750 whole: bool,
7751 known: Option<Arc<Vec<PartSpan>>>,
7752 ) -> Result<CachedColumn> {
7753 let index = match known {
7754 Some(index) => index,
7755 None => {
7756 self.indexes.fetch_add(1, Atomic::Relaxed);
7757 Arc::new(read_index(&self.file, stripe, column)?)
7758 }
7759 };
7760 let page = if whole {
7761 self.pages.fetch_add(1, Atomic::Relaxed);
7762 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7763 let length = span.length as usize;
7764 let bytes = match &self.map {
7765 Some(map) if map.get(span.offset, length).is_some() => {
7766 PageBytes::Mapped { map: Arc::clone(map), offset: span.offset, length }
7767 }
7768 _ => {
7769 let mut bytes = vec![0; length];
7770 read_at(&self.file, span.offset, &mut bytes)?;
7771 PageBytes::Read(bytes)
7772 }
7773 };
7774 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7775 Some(Arc::new(HeldPage { bytes, checked }))
7776 } else {
7777 None
7778 };
7779 Ok(CachedColumn { stripe: at, index, page })
7780 }
7781
7782 fn read_impl(
7783 &self,
7784 at: usize,
7785 columns: &[usize],
7786 whole: bool,
7787 positions: Option<&[u32]>,
7788 ) -> Result<Chunk> {
7789 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7790 let index = place.stripe as usize;
7791 let stripe =
7792 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7793 let rows = place.rows as usize;
7794 let mut picked = Vec::with_capacity(columns.len());
7795 for &column in columns {
7796 let field = self
7797 .table
7798 .fields
7799 .get(column)
7800 .ok_or_else(|| invalid("column index out of range"))?;
7801 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7802 let held = self.held(index, place.part as usize, stripe, column, whole)?;
7803 let span = *held
7804 .index
7805 .get(place.part as usize)
7806 .ok_or_else(|| invalid("part index out of range"))?;
7807 let owned;
7808 let mut mapped = false;
7809 let bit = at * self.table.fields.len() + column;
7810 let bytes = match &held.page {
7811 Some(held) if self.is_verified(bit) => part_bytes(held.bytes(), span),
7812 Some(held) => {
7813 held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7814 }
7815 None => {
7816 let offset = page
7817 .offset
7818 .checked_add(span.start as u64)
7819 .ok_or_else(|| invalid("part range overflow"))?;
7820 let bytes =
7821 match self.map.as_deref().and_then(|map| map.get(offset, span.length)) {
7822 Some(bytes) => {
7823 mapped = true;
7824 bytes
7825 }
7826 None => {
7827 let mut bytes = vec![0; span.length];
7828 read_at(&self.file, offset, &mut bytes)?;
7829 owned = bytes;
7830 owned.as_slice()
7831 }
7832 };
7833 if self.is_verified(bit) {
7834 Ok(bytes)
7835 } else {
7836 verify_part(bytes, span).map(|()| {
7837 self.set_verified(bit);
7838 bytes
7839 })
7840 }
7841 }
7842 }
7843 .map_err(|error| {
7844 invalid(&format!(
7845 "{}, column {column} part {} of the page at {}",
7846 error.message(),
7847 place.part,
7848 page.offset,
7849 ))
7850 })?;
7851 let dictionary = self.dictionary(column)?;
7852 let mut vector = match positions {
7858 None => decode(&field.ty, rows, bytes, dictionary)?,
7859 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7860 };
7861 if mapped
7865 && let Some(map) = self.map.as_deref()
7866 && let Some(left) = self.unreleased.get(index * self.table.fields.len() + column)
7867 && left.fetch_update(Atomic::Relaxed, Atomic::Relaxed, |left| left.checked_sub(1))
7868 == Ok(1)
7869 {
7870 map.release(page.offset, page.length as usize);
7871 }
7872 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7876 vector = vector.flatten()?;
7877 }
7878 picked.push(vector.into_pages());
7879 }
7880 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7881 }
7882
7883 #[must_use]
7899 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7900 let Some(place) = self.places.get(part).copied() else { return false };
7901 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7902 if stripe.zone.skips(probes) {
7903 return true;
7904 }
7905 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7906 }
7907
7908 #[must_use]
7915 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7916 let Some(place) = self.places.get(part).copied() else { return false };
7917 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7918 if stripe.zone.column(column).is_some_and(&rule) {
7919 return true;
7920 }
7921 self.stripe_part_ranges(place.stripe as usize, column)
7922 .and_then(|ranges| ranges.get(place.part as usize))
7923 .is_some_and(rule)
7924 }
7925
7926 #[must_use]
7929 pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7930 let place = self.places.get(part).copied()?;
7931 let own = self
7932 .stripe_part_ranges(place.stripe as usize, column)
7933 .and_then(|ranges| ranges.get(place.part as usize));
7934 own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7935 }
7936
7937 #[must_use]
7939 pub fn stripe_ruled_by(
7940 &self,
7941 stripe: usize,
7942 column: usize,
7943 rule: impl Fn(&Range) -> bool,
7944 ) -> bool {
7945 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7946 }
7947
7948 fn outside(&self, place: Place, probe: &Probe) -> bool {
7954 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7955 Some(ranges) => ranges
7956 .get(place.part as usize)
7957 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7958 None => false,
7959 }
7960 }
7961
7962 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7968 let slot = self
7969 .part_ranges
7970 .get(column)?
7971 .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
7972 .get(stripe)?;
7973 if let Some(held) = slot.get() {
7974 return Some(held);
7975 }
7976 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7977 let mut bytes = vec![0; page.length as usize];
7978 read_at(&self.file, page.offset, &mut bytes).ok()?;
7979 if checksum(&bytes) != page.hash {
7980 return None;
7981 }
7982 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7983 let _ = slot.set(ranges);
7984 slot.get().map(|held| held.as_slice())
7985 }
7986
7987 #[must_use]
8004 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
8005 let Some(place) = self.places.get(part).copied() else { return false };
8006 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
8007 if stripe.zone.certain(probes) {
8008 return true;
8009 }
8010 probes
8011 .iter()
8012 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
8013 }
8014
8015 fn inside(&self, place: Place, probe: &Probe) -> bool {
8021 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
8022 Some(ranges) => ranges
8023 .get(place.part as usize)
8024 .is_some_and(|range| range.certain(probe.op, &probe.value)),
8025 None => false,
8026 }
8027 }
8028
8029 #[must_use]
8040 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
8041 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
8042 }
8043
8044 fn sifted(&self, place: Place, probe: &Probe) -> bool {
8050 if probe.op != Op::Equal {
8051 return false;
8052 }
8053 match self.stripe_sieves(place.stripe as usize, probe.column) {
8054 Some(sieves) => sieves
8055 .get(place.part as usize)
8056 .and_then(Option::as_ref)
8057 .is_some_and(|sieve| sieve.excludes(&probe.value)),
8058 None => false,
8059 }
8060 }
8061
8062 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
8069 let slot = self
8070 .sieves
8071 .get(column)?
8072 .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
8073 .get(stripe)?;
8074 if let Some(held) = slot.get() {
8075 return Some(held);
8076 }
8077 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
8078 let mut bytes = vec![0; page.length as usize];
8079 read_at(&self.file, page.offset, &mut bytes).ok()?;
8080 if checksum(&bytes) != page.hash {
8081 return None;
8082 }
8083 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
8084 let _ = slot.set(sieves);
8085 slot.get().map(|held| held.as_slice())
8086 }
8087}
8088
8089fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
8091 let code = dictionary.code_at_rank(rank)? as usize;
8092 if dictionary.logical_type() == &LogicalType::Blob {
8093 let bytes = dictionary
8094 .try_bytes_at(code)?
8095 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8096 return Ok(Value::Blob(bytes.to_vec()));
8097 }
8098 let text = dictionary
8099 .try_text_at(code)?
8100 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8101 Ok(Value::Varchar(text.into()))
8102}
8103
8104fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
8114 file.fill_at(offset, bytes)
8115}
8116
8117trait Positional {
8125 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
8130}
8131
8132impl<T: Positional + ?Sized> Positional for &T {
8133 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8134 (**self).fill_at(offset, bytes)
8135 }
8136}
8137
8138impl<T: Positional + ?Sized> Positional for Arc<T> {
8139 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8140 (**self).fill_at(offset, bytes)
8141 }
8142}
8143
8144impl<T: Positional + ?Sized> Positional for Box<T> {
8145 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8146 (**self).fill_at(offset, bytes)
8147 }
8148}
8149
8150impl Positional for dyn rudb_io::File + '_ {
8151 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8152 while !bytes.is_empty() {
8153 let read = self.read_at(offset, bytes)?;
8154 if read == 0 {
8155 return Err(invalid("column page ends before its declared length"));
8156 }
8157 offset += read as u64;
8158 bytes = &mut bytes[read..];
8159 }
8160 Ok(())
8161 }
8162}
8163
8164impl Positional for File {
8165 #[cfg(unix)]
8166 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8167 use std::os::unix::fs::FileExt;
8168 while !bytes.is_empty() {
8169 let read = self.read_at(bytes, offset).map_err(io)?;
8170 if read == 0 {
8171 return Err(invalid("column page ends before its declared length"));
8172 }
8173 offset += read as u64;
8174 bytes = &mut bytes[read..];
8175 }
8176 Ok(())
8177 }
8178
8179 #[cfg(windows)]
8185 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8186 use std::os::windows::fs::FileExt;
8187 while !bytes.is_empty() {
8188 let read = self.seek_read(bytes, offset).map_err(io)?;
8189 if read == 0 {
8190 return Err(invalid("column page ends before its declared length"));
8191 }
8192 offset += read as u64;
8193 bytes = &mut bytes[read..];
8194 }
8195 Ok(())
8196 }
8197
8198 #[cfg(not(any(unix, windows)))]
8203 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8204 use std::io::{Read, Seek, SeekFrom};
8205 let mut file = self.try_clone().map_err(io)?;
8206 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8207 file.read_exact(bytes).map_err(io)
8208 }
8209}
8210
8211#[cfg(test)]
8216fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
8217 use std::io::{Seek, SeekFrom, Write};
8218 let mut file = file;
8219 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8220 file.write_all(bytes).map_err(io)
8221}
8222
8223fn type_tag(ty: &LogicalType) -> Result<u8> {
8230 match ty {
8231 LogicalType::SmallInt => Ok(1),
8232 LogicalType::Integer => Ok(2),
8233 LogicalType::BigInt => Ok(3),
8234 LogicalType::Varchar => Ok(4),
8235 LogicalType::Date => Ok(5),
8236 LogicalType::Timestamp => Ok(6),
8237 LogicalType::Boolean => Ok(7),
8238 LogicalType::TinyInt => Ok(8),
8239 LogicalType::UTinyInt => Ok(9),
8240 LogicalType::USmallInt => Ok(10),
8241 LogicalType::UInteger => Ok(11),
8242 LogicalType::UBigInt => Ok(12),
8243 LogicalType::Decimal { .. } => Ok(13),
8244 LogicalType::Float => Ok(14),
8245 LogicalType::Double => Ok(15),
8246 LogicalType::HugeInt => Ok(16),
8247 LogicalType::UHugeInt => Ok(17),
8248 LogicalType::Time => Ok(18),
8249 LogicalType::TimeTz => Ok(19),
8250 LogicalType::TimestampTz => Ok(20),
8251 LogicalType::Interval => Ok(21),
8252 LogicalType::Uuid => Ok(22),
8253 LogicalType::Blob => Ok(23),
8254 LogicalType::Bit => Ok(24),
8255 LogicalType::TimestampS => Ok(25),
8256 LogicalType::TimestampMs => Ok(26),
8257 LogicalType::TimestampNs => Ok(27),
8258 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8259 }
8260}
8261
8262fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8268 out.push(type_tag(ty)?);
8269 if let LogicalType::Decimal { width, scale } = ty {
8270 out.push(*width);
8271 out.push(*scale);
8272 }
8273 Ok(())
8274}
8275
8276fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8278 let tag = cur.u8()?;
8279 if tag == 13 {
8280 let width = cur.u8()?;
8281 let scale = cur.u8()?;
8282 return LogicalType::decimal(width, scale)
8283 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8284 }
8285 tag_type(tag)
8286}
8287
8288fn tag_type(tag: u8) -> Result<LogicalType> {
8289 match tag {
8290 1 => Ok(LogicalType::SmallInt),
8291 2 => Ok(LogicalType::Integer),
8292 3 => Ok(LogicalType::BigInt),
8293 4 => Ok(LogicalType::Varchar),
8294 5 => Ok(LogicalType::Date),
8295 6 => Ok(LogicalType::Timestamp),
8296 7 => Ok(LogicalType::Boolean),
8297 8 => Ok(LogicalType::TinyInt),
8298 9 => Ok(LogicalType::UTinyInt),
8299 10 => Ok(LogicalType::USmallInt),
8300 11 => Ok(LogicalType::UInteger),
8301 12 => Ok(LogicalType::UBigInt),
8302 14 => Ok(LogicalType::Float),
8303 15 => Ok(LogicalType::Double),
8304 16 => Ok(LogicalType::HugeInt),
8305 17 => Ok(LogicalType::UHugeInt),
8306 18 => Ok(LogicalType::Time),
8307 19 => Ok(LogicalType::TimeTz),
8308 20 => Ok(LogicalType::TimestampTz),
8309 21 => Ok(LogicalType::Interval),
8310 22 => Ok(LogicalType::Uuid),
8311 23 => Ok(LogicalType::Blob),
8312 24 => Ok(LogicalType::Bit),
8313 25 => Ok(LogicalType::TimestampS),
8314 26 => Ok(LogicalType::TimestampMs),
8315 27 => Ok(LogicalType::TimestampNs),
8316 _ => Err(invalid("column type tag is unknown")),
8317 }
8318}
8319
8320fn put_u16(out: &mut Vec<u8>, value: u16) {
8321 out.extend_from_slice(&value.to_le_bytes());
8322}
8323fn put_u32(out: &mut Vec<u8>, value: u32) {
8324 out.extend_from_slice(&value.to_le_bytes());
8325}
8326fn put_u64(out: &mut Vec<u8>, value: u64) {
8327 out.extend_from_slice(&value.to_le_bytes());
8328}
8329fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8330 while value >= 0x80 {
8331 out.push((value as u8 & 0x7f) | 0x80);
8332 value >>= 7;
8333 }
8334 out.push(value as u8);
8335}
8336
8337fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8338 match (left, right) {
8339 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8340 (FrequencyValue::Null, _) => Ordering::Less,
8341 (_, FrequencyValue::Null) => Ordering::Greater,
8342 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8343 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8344 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8345 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8346 }
8347}
8348
8349fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8362 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8363 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8364 };
8365 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8366 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8367 let omitted_max = next.count;
8368 entries.truncate(FREQUENCY_ENTRIES);
8369 entries.shrink_to_fit();
8372 omitted_max
8373 } else {
8374 0
8375 };
8376 entries.sort_unstable_by(order);
8377 omitted_max
8378}
8379
8380fn code_frequency(
8381 dictionary: &GlobalDictionary,
8382 flat: &[u8],
8383 bases: &[u64],
8384) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8385 let seen = dictionary.counts.iter().filter(|count| **count != 0).count();
8390 let mut candidates = Vec::with_capacity(seen + usize::from(dictionary.nulls != 0));
8391 candidates.extend(
8392 dictionary
8393 .counts
8394 .iter()
8395 .enumerate()
8396 .filter(|(_, count)| **count != 0)
8397 .map(|(code, &count)| (count, Some(code as u32))),
8398 );
8399 if dictionary.nulls != 0 {
8400 candidates.push((dictionary.nulls, None));
8401 }
8402 let order = |left: &(u64, Option<u32>), right: &(u64, Option<u32>)| {
8404 right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1))
8405 };
8406 let omitted_max = if candidates.len() > FREQUENCY_ENTRIES {
8407 let (_, next, _) = candidates.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8408 let omitted_max = next.0;
8409 candidates.truncate(FREQUENCY_ENTRIES);
8410 omitted_max
8411 } else {
8412 0
8413 };
8414 candidates.sort_unstable_by(order);
8415 let entries = candidates
8416 .into_iter()
8417 .map(|(count, code)| FrequencyEntry {
8418 value: code.map_or(FrequencyValue::Null, FrequencyValue::Code),
8419 count,
8420 })
8421 .collect::<Vec<_>>();
8422 let mut spans = Vec::with_capacity(entries.len());
8423 let mut text_bytes = 0_usize;
8424 for entry in &entries {
8425 let span = match entry.value {
8426 FrequencyValue::Code(code) => {
8427 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8428 let bytes = flat
8429 .get(span.0..span.1)
8430 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8431 text_bytes = text_bytes.saturating_add(bytes.len());
8432 Some(span)
8433 }
8434 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8435 };
8436 spans.push(span);
8437 }
8438 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8439 Vec::new()
8440 } else {
8441 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8442 };
8443 Ok((
8444 FrequencySummary {
8445 entries,
8446 omitted_max,
8447 ordinals: Vec::new(),
8448 ordinal_entries: Vec::new(),
8449 ordinal_bound: 0,
8450 },
8451 texts,
8452 ))
8453}
8454
8455fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8456 let mut out = DIRECTORY.to_vec();
8457 let name = table.name.as_bytes();
8458 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8459 out.extend_from_slice(name);
8460 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8461 for field in &table.fields {
8462 let name = field.name.as_bytes();
8463 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8464 out.extend_from_slice(name);
8465 put_type(&mut out, &field.ty)?;
8466 out.push(u8::from(field.not_null));
8467 }
8468 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8469 match dictionary {
8470 None => out.push(0),
8471 Some(page) => {
8472 out.push(dictionary_tag(&field.ty));
8473 put_u64(&mut out, page.offset);
8474 put_u32(&mut out, page.length);
8475 put_u64(&mut out, page.hash);
8476 }
8477 }
8478 }
8479 for distinct in &table.distincts {
8480 match distinct {
8481 None => out.push(0),
8482 Some(count) => {
8483 out.push(1);
8484 put_u64(&mut out, *count);
8485 }
8486 }
8487 }
8488 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8489 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8490 for stripe in &table.stripes {
8491 put_u32(
8492 &mut out,
8493 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8494 );
8495 for &rows in &stripe.parts {
8496 put_u32(&mut out, rows);
8497 }
8498 put_u64(&mut out, stripe.index.offset);
8499 put_u32(&mut out, stripe.index.length);
8500 for page in &stripe.pages {
8501 put_u64(&mut out, page.offset);
8502 put_u32(&mut out, page.length);
8503 }
8504 for (column, ((field, dictionary), membership)) in
8509 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8510 {
8511 if !coded_type(&field.ty) || dictionary.is_none() {
8512 continue;
8513 }
8514 let page = match membership {
8515 Some(page) => page,
8516 None if table.demoted.get(column).copied().unwrap_or(false) => {
8517 Page { offset: HEADER, length: 0, hash: 0 }
8518 }
8519 None => return Err(invalid("string page has no code membership index")),
8520 };
8521 put_u64(&mut out, page.offset);
8522 put_u32(&mut out, page.length);
8523 put_u64(&mut out, page.hash);
8524 }
8525 for sieve in stripe.sieves.slots() {
8526 match sieve {
8527 None => out.push(0),
8528 Some(page) => {
8529 out.push(1);
8530 put_u64(&mut out, page.offset);
8531 put_u32(&mut out, page.length);
8532 put_u64(&mut out, page.hash);
8533 }
8534 }
8535 }
8536 for held in stripe.part_ranges.slots() {
8537 match held {
8538 None => out.push(0),
8539 Some(page) => {
8540 out.push(1);
8541 put_u64(&mut out, page.offset);
8542 put_u32(&mut out, page.length);
8543 put_u64(&mut out, page.hash);
8544 }
8545 }
8546 }
8547 for range in stripe.zone.columns() {
8548 put_bound(&mut out, range.low.as_ref())?;
8549 put_bound(&mut out, range.high.as_ref())?;
8550 put_u32(
8551 &mut out,
8552 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8553 );
8554 out.push(u8::from(range.exact));
8555 match range.sum {
8556 None => out.push(0),
8557 Some(total) => {
8558 out.push(1);
8559 out.extend_from_slice(&total.to_le_bytes());
8560 }
8561 }
8562 }
8563 }
8564 out.extend_from_slice(FREQUENCIES_SPANS);
8565 put_u16(
8566 &mut out,
8567 u16::try_from(table.frequencies.len())
8568 .map_err(|_| invalid("too many frequency columns"))?,
8569 );
8570 for summary in &table.frequencies {
8571 let summary = match summary {
8572 None => {
8573 put_u32(&mut out, 0);
8574 put_u32(&mut out, 0);
8575 continue;
8576 }
8577 Some(Frequencies::Held(summary)) => summary,
8578 Some(Frequencies::Stored { .. }) => {
8580 return Err(invalid("a synopsis left in the file cannot be written back"));
8581 }
8582 };
8583 let length_at = out.len();
8584 put_u32(&mut out, 0);
8585 put_u32(
8586 &mut out,
8587 u32::try_from(summary.entries.len())
8588 .map_err(|_| invalid("too many frequency entries"))?,
8589 );
8590 let start = out.len();
8591 out.push(1);
8592 put_u64(&mut out, summary.omitted_max);
8593 put_u32(
8594 &mut out,
8595 u32::try_from(summary.entries.len())
8596 .map_err(|_| invalid("too many frequency entries"))?,
8597 );
8598 for entry in &summary.entries {
8599 match entry.value {
8600 FrequencyValue::Null => out.push(0),
8601 FrequencyValue::Integer(value) => {
8602 out.push(1);
8603 out.extend_from_slice(&value.to_le_bytes());
8604 }
8605 FrequencyValue::Code(value) => {
8606 out.push(2);
8607 put_u32(&mut out, value);
8608 }
8609 }
8610 put_u64(&mut out, entry.count);
8611 }
8612 put_u32(
8613 &mut out,
8614 u32::try_from(summary.ordinals.len())
8615 .map_err(|_| invalid("too many frequency ordinals"))?,
8616 );
8617 let mut previous = 0_u64;
8618 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8619 let delta = if at == 0 {
8620 ordinal
8621 } else {
8622 ordinal
8623 .checked_sub(previous)
8624 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8625 };
8626 if at != 0 && delta == 0 {
8627 return Err(invalid("frequency ordinals are not unique"));
8628 }
8629 put_var_u64(&mut out, delta);
8630 previous = ordinal;
8631 }
8632 if summary.ordinal_entries.len() != summary.ordinals.len() {
8633 return Err(invalid("frequency ordinal values have a different length"));
8634 }
8635 for &entry in &summary.ordinal_entries {
8636 if entry as usize >= summary.entries.len() {
8637 return Err(invalid("frequency ordinal value is outside its entries"));
8638 }
8639 put_u16(&mut out, entry);
8640 }
8641 let length = u32::try_from(out.len() - start)
8642 .map_err(|_| invalid("a frequency synopsis is too long"))?;
8643 out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8644 }
8645 let bounds = table
8646 .frequencies
8647 .iter()
8648 .enumerate()
8649 .filter_map(|(column, summary)| match summary {
8650 Some(Frequencies::Held(summary)) if summary.ordinal_bound != 0 => {
8651 Some((column, summary.ordinal_bound))
8652 }
8653 _ => None,
8654 })
8655 .collect::<Vec<_>>();
8656 if !bounds.is_empty() {
8657 out.extend_from_slice(ORDINAL_BOUNDS);
8658 put_u16(&mut out, u16::try_from(bounds.len()).map_err(|_| invalid("too many bounds"))?);
8659 for (column, bound) in bounds {
8660 put_u16(
8661 &mut out,
8662 u16::try_from(column).map_err(|_| invalid("bound column overflows"))?,
8663 );
8664 put_u64(&mut out, bound);
8665 }
8666 }
8667 if !table.pair_frequencies.is_empty() {
8668 out.extend_from_slice(PAIR_FREQUENCIES);
8669 put_u16(
8670 &mut out,
8671 u16::try_from(table.pair_frequencies.len())
8672 .map_err(|_| invalid("too many pair frequency summaries"))?,
8673 );
8674 for summary in &table.pair_frequencies {
8675 put_u16(&mut out, summary.first);
8676 put_u16(&mut out, summary.second);
8677 put_u64(&mut out, summary.omitted_max);
8678 put_u16(
8679 &mut out,
8680 u16::try_from(summary.entries.len())
8681 .map_err(|_| invalid("too many pair frequency entries"))?,
8682 );
8683 for entry in &summary.entries {
8684 put_u16(&mut out, entry.first_entry);
8685 match entry.second {
8686 None => out.push(0),
8687 Some(code) => {
8688 out.push(1);
8689 put_u32(&mut out, code);
8690 }
8691 }
8692 put_u64(&mut out, entry.count);
8693 }
8694 }
8695 }
8696 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8697 if text_columns != 0 {
8698 out.extend_from_slice(FREQUENCY_TEXTS);
8699 put_u16(
8700 &mut out,
8701 u16::try_from(text_columns)
8702 .map_err(|_| invalid("too many string frequency columns"))?,
8703 );
8704 for (column, texts) in table.frequency_texts.iter().enumerate() {
8705 if texts.is_empty() {
8706 continue;
8707 }
8708 put_u16(
8709 &mut out,
8710 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8711 );
8712 put_u16(
8713 &mut out,
8714 u16::try_from(texts.len())
8715 .map_err(|_| invalid("too many frequency text entries"))?,
8716 );
8717 for text in texts {
8718 match text {
8719 None => out.push(0),
8720 Some(text) => {
8721 out.push(1);
8722 put_u32(
8723 &mut out,
8724 u32::try_from(text.len())
8725 .map_err(|_| invalid("frequency text is too long"))?,
8726 );
8727 out.extend_from_slice(text);
8728 }
8729 }
8730 }
8731 }
8732 }
8733 if let Some(summary) = &table.host_groups {
8734 out.extend_from_slice(HOST_GROUPS);
8735 put_u16(
8736 &mut out,
8737 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8738 );
8739 put_u64(&mut out, summary.omitted_max);
8740 put_u16(
8741 &mut out,
8742 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8743 );
8744 for entry in &summary.entries {
8745 put_u32(
8746 &mut out,
8747 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8748 );
8749 out.extend_from_slice(entry.host.as_bytes());
8750 put_u64(&mut out, entry.count);
8751 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8752 put_u32(
8753 &mut out,
8754 u32::try_from(entry.minimum.len())
8755 .map_err(|_| invalid("host minimum is too long"))?,
8756 );
8757 out.extend_from_slice(entry.minimum.as_bytes());
8758 }
8759 }
8760 if let Some(clustering) = &table.clustering {
8763 out.extend_from_slice(CLUSTERING);
8764 out.push(clustering.width().tag());
8765 put_u16(
8766 &mut out,
8767 u16::try_from(clustering.columns().len())
8768 .map_err(|_| invalid("too many clustering columns"))?,
8769 );
8770 for &column in clustering.columns() {
8771 put_u16(
8772 &mut out,
8773 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8774 );
8775 }
8776 }
8777 let demoted = (0..table.fields.len())
8778 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8779 .collect::<Vec<_>>();
8780 if !demoted.is_empty() {
8781 out.extend_from_slice(DEMOTED);
8782 put_u16(
8783 &mut out,
8784 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8785 );
8786 for column in demoted {
8787 put_u16(
8788 &mut out,
8789 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8790 );
8791 }
8792 }
8793 if !table.constraints.is_empty() {
8794 out.extend_from_slice(KEYS);
8795 put_count(&mut out, table.constraints.keys.len())?;
8796 for (columns, primary) in &table.constraints.keys {
8797 out.push(u8::from(*primary));
8798 put_columns(&mut out, columns)?;
8799 }
8800 put_count(&mut out, table.constraints.foreign.len())?;
8801 for foreign in &table.constraints.foreign {
8802 put_columns(&mut out, &foreign.columns)?;
8803 put_columns(&mut out, &foreign.referenced)?;
8804 put_u32(
8805 &mut out,
8806 u32::try_from(foreign.table.len())
8807 .map_err(|_| invalid("table name is too long"))?,
8808 );
8809 out.extend_from_slice(foreign.table.as_bytes());
8810 }
8811 }
8812 out.extend_from_slice(SECTIONS);
8818 put_u64(&mut out, table.generation);
8819 put_u16(
8820 &mut out,
8821 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8822 );
8823 for held in &table.sections {
8824 held.encode(&mut out)?;
8825 }
8826 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8827 out.extend_from_slice(DICTIONARY_PAYLOADS);
8828 put_u16(
8829 &mut out,
8830 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8831 );
8832 for at in 0..table.fields.len() {
8833 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8834 }
8835 }
8836 Ok(out)
8837}
8838
8839fn signed_integer(ty: &LogicalType) -> bool {
8848 matches!(
8849 ty,
8850 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8851 )
8852}
8853
8854fn integer_or_date(ty: &LogicalType) -> bool {
8855 matches!(
8856 ty,
8857 LogicalType::TinyInt
8858 | LogicalType::SmallInt
8859 | LogicalType::Integer
8860 | LogicalType::BigInt
8861 | LogicalType::UTinyInt
8862 | LogicalType::USmallInt
8863 | LogicalType::UInteger
8864 | LogicalType::UBigInt
8865 | LogicalType::Date
8866 )
8867}
8868
8869fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8870 table
8871 .fields
8872 .iter()
8873 .enumerate()
8874 .map(|(column, field)| {
8875 if !integer_or_date(&field.ty) {
8876 return None;
8877 }
8878 let mut low: Option<i128> = None;
8879 let mut high: Option<i128> = None;
8880 for stripe in &table.stripes {
8881 let range = stripe.zone.column(column)?;
8882 if !range.exact {
8883 return None;
8884 }
8885 match (range.low.as_ref(), range.high.as_ref()) {
8886 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8887 low = Some(low.map_or(*small, |held| held.min(*small)));
8888 high = Some(high.map_or(*large, |held| held.max(*large)));
8889 }
8890 (None, None) if stripe.rows == range.nulls => {}
8891 _ => return None,
8892 }
8893 }
8894 Some(low.zip(high))
8895 })
8896 .collect()
8897}
8898
8899fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8900 reader
8901 .table
8902 .fields
8903 .iter()
8904 .enumerate()
8905 .map(|(column, field)| {
8906 if !integer_or_date(&field.ty) {
8907 return Ok(None);
8908 }
8909 match reader.exact_extremes(column)? {
8910 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8911 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8912 _ => Ok(None),
8913 }
8914 })
8915 .collect()
8916}
8917
8918fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8919 table
8920 .fields
8921 .iter()
8922 .enumerate()
8923 .map(|(column, field)| {
8924 if !integer_or_date(&field.ty) {
8925 return None;
8926 }
8927 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8928 return None;
8929 };
8930 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8931 return None;
8932 }
8933 let entries = summary
8934 .entries
8935 .iter()
8936 .map(|entry| {
8937 let value = match entry.value {
8938 FrequencyValue::Null => None,
8939 FrequencyValue::Integer(value) => Some(value),
8940 FrequencyValue::Code(_) => return None,
8941 };
8942 Some((value, entry.count))
8943 })
8944 .collect::<Option<Vec<_>>>()?;
8945 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8946 (rows == table.rows as u64).then_some(entries)
8947 })
8948 .collect()
8949}
8950
8951fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8958 if signed {
8959 FrequencyValue::Integer(i128::from(bits as i64))
8960 } else {
8961 FrequencyValue::Integer(i128::from(bits))
8962 }
8963}
8964
8965fn frequency_bits(value: &Value) -> Option<u64> {
8966 Some(match value {
8967 Value::TinyInt(value) => i64::from(*value) as u64,
8968 Value::SmallInt(value) => i64::from(*value) as u64,
8969 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8970 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8971 Value::UTinyInt(value) => u64::from(*value),
8972 Value::USmallInt(value) => u64::from(*value),
8973 Value::UInteger(value) => u64::from(*value),
8974 Value::UBigInt(value) => *value,
8975 _ => return None,
8976 })
8977}
8978
8979fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8980 Some(match value {
8981 Value::Null => None,
8982 Value::TinyInt(value) => Some(i128::from(*value)),
8983 Value::SmallInt(value) => Some(i128::from(*value)),
8984 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8985 Value::BigInt(value) => Some(i128::from(*value)),
8986 Value::UTinyInt(value) => Some(i128::from(*value)),
8987 Value::USmallInt(value) => Some(i128::from(*value)),
8988 Value::UInteger(value) => Some(i128::from(*value)),
8989 Value::UBigInt(value) => Some(i128::from(*value)),
8990 _ => return None,
8991 })
8992}
8993
8994fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8995 reader
8996 .table
8997 .fields
8998 .iter()
8999 .enumerate()
9000 .map(|(column, field)| {
9001 if !integer_or_date(&field.ty) {
9002 return Ok(None);
9003 }
9004 let Some((entries, omitted_max)) = reader.frequency_head(column)? else {
9005 return Ok(None);
9006 };
9007 if omitted_max != 0 || entries.len() > MAX_CATALOG_FREQUENCIES {
9008 return Ok(None);
9009 }
9010 let entries = reader.decode_frequencies(column, &field.ty, &entries)?;
9011 let Some(entries) = entries
9012 .iter()
9013 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
9014 .collect::<Option<Vec<_>>>()
9015 else {
9016 return Ok(None);
9017 };
9018 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
9019 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
9020 })
9021 .collect()
9022}
9023
9024fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
9025 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
9026 let range = stripe.zone.column(column)?;
9027 let sum = sum.checked_add(range.sum?)?;
9028 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
9029 Some((sum, count.checked_add(nonnull)?))
9030 })
9031}
9032
9033fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
9034 table
9035 .fields
9036 .iter()
9037 .enumerate()
9038 .map(|(column, field)| {
9039 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
9040 })
9041 .collect()
9042}
9043
9044fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
9045 reader
9046 .table
9047 .fields
9048 .iter()
9049 .enumerate()
9050 .map(
9051 |(column, field)| {
9052 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
9053 },
9054 )
9055 .collect()
9056}
9057
9058fn encode_catalog(
9059 entries: &[Entry],
9060 views: &[ViewEntry],
9061 card: Option<&KeptCard>,
9062 anchor: Option<&LogAnchor>,
9063) -> Result<Vec<u8>> {
9064 let mut out = CATALOG.to_vec();
9065 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
9066 for entry in entries {
9067 let name = entry.name.as_bytes();
9068 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
9069 out.extend_from_slice(name);
9070 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
9071 put_u16(
9072 &mut out,
9073 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
9074 );
9075 for field in &entry.fields {
9076 let name = field.name.as_bytes();
9077 put_u16(
9078 &mut out,
9079 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9080 );
9081 out.extend_from_slice(name);
9082 put_type(&mut out, &field.ty)?;
9083 out.push(u8::from(field.not_null));
9084 }
9085 put_u64(&mut out, entry.directory.offset);
9086 put_u32(&mut out, entry.directory.length);
9087 put_u64(&mut out, entry.directory.hash);
9088 }
9089 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
9090 for view in views {
9091 let name = view.name.as_bytes();
9092 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
9093 out.extend_from_slice(name);
9094 put_long_text(&mut out, &view.sql, "view body")?;
9095 put_long_text(&mut out, &view.statement, "view statement")?;
9096 put_u16(
9097 &mut out,
9098 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
9099 );
9100 for alias in &view.aliases {
9101 let alias = alias.as_bytes();
9102 put_u16(
9103 &mut out,
9104 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
9105 );
9106 out.extend_from_slice(alias);
9107 }
9108 put_u16(
9109 &mut out,
9110 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
9111 );
9112 for field in &view.columns {
9113 let name = field.name.as_bytes();
9114 put_u16(
9115 &mut out,
9116 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9117 );
9118 out.extend_from_slice(name);
9119 put_type(&mut out, &field.ty)?;
9120 out.push(u8::from(field.not_null));
9121 }
9122 }
9123 out.extend_from_slice(NONZERO_COUNTS);
9124 for entry in entries {
9125 if entry.nonzero.len() != entry.fields.len() {
9126 return Err(invalid("nonzero count width differs from schema"));
9127 }
9128 for count in &entry.nonzero {
9129 match count {
9130 None => out.push(0),
9131 Some(count) => {
9132 out.push(1);
9133 put_u64(&mut out, *count);
9134 }
9135 }
9136 }
9137 }
9138 out.extend_from_slice(AGGREGATE_SUMS);
9139 for entry in entries {
9140 if entry.aggregates.len() != entry.fields.len() {
9141 return Err(invalid("aggregate sum width differs from schema"));
9142 }
9143 for summary in &entry.aggregates {
9144 match summary {
9145 None => out.push(0),
9146 Some((sum, count)) => {
9147 out.push(1);
9148 out.extend_from_slice(&sum.to_le_bytes());
9149 put_u64(&mut out, *count);
9150 }
9151 }
9152 }
9153 }
9154 out.extend_from_slice(DISTINCT_COUNTS);
9155 for entry in entries {
9156 if entry.distincts.len() != entry.fields.len() {
9157 return Err(invalid("distinct count width differs from schema"));
9158 }
9159 for count in &entry.distincts {
9160 match count {
9161 None => out.push(0),
9162 Some(count) => {
9163 if *count > entry.rows as u64 {
9164 return Err(invalid("distinct count exceeds table rows"));
9165 }
9166 out.push(1);
9167 put_u64(&mut out, *count);
9168 }
9169 }
9170 }
9171 }
9172 out.extend_from_slice(INTEGER_EXTREMES);
9173 for entry in entries {
9174 if entry.extremes.len() != entry.fields.len() {
9175 return Err(invalid("integer extremes width differs from schema"));
9176 }
9177 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
9178 match extremes {
9179 None => out.push(0),
9180 Some(None) if integer_or_date(&field.ty) => out.push(1),
9181 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
9182 out.push(2);
9183 out.extend_from_slice(&low.to_le_bytes());
9184 out.extend_from_slice(&high.to_le_bytes());
9185 }
9186 _ => return Err(invalid("integer extremes type or range differs")),
9187 }
9188 }
9189 }
9190 out.extend_from_slice(COMPLETE_FREQUENCIES);
9191 for entry in entries {
9192 if entry.frequencies.len() != entry.fields.len() {
9193 return Err(invalid("numeric frequency width differs from schema"));
9194 }
9195 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
9196 match frequencies {
9197 None => out.push(0),
9198 Some(entries)
9199 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
9200 {
9201 let mut total = 0_u64;
9202 for (at, (value, count)) in entries.iter().enumerate() {
9203 if entries[..at].iter().any(|(held, _)| held == value) {
9204 return Err(invalid("numeric frequency value repeats"));
9205 }
9206 total = total
9207 .checked_add(*count)
9208 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9209 }
9210 if total != entry.rows as u64 {
9211 return Err(invalid("numeric frequencies do not cover table rows"));
9212 }
9213 out.push(1);
9214 out.push(entries.len() as u8);
9215 for (value, count) in entries {
9216 match value {
9217 None => out.push(0),
9218 Some(value) => {
9219 out.push(1);
9220 out.extend_from_slice(&value.to_le_bytes());
9221 }
9222 }
9223 put_u64(&mut out, *count);
9224 }
9225 }
9226 _ => return Err(invalid("numeric frequency type or width differs")),
9227 }
9228 }
9229 }
9230 if let Some(card) = card {
9231 out.extend_from_slice(DEVICE_CARD);
9232 let device = card.device.as_bytes();
9233 put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
9234 out.extend_from_slice(device);
9235 put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
9236 out.extend_from_slice(&card.bytes);
9237 }
9238 if let Some(anchor) = anchor {
9239 anchor.encode(&mut out)?;
9240 }
9241 Ok(out)
9242}
9243
9244fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
9246 let bytes = text.as_bytes();
9247 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
9248 out.extend_from_slice(bytes);
9249 Ok(())
9250}
9251
9252fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
9255 let mut cur = Cursor::new(bytes);
9256 if cur.take(8)? != CATALOG {
9257 return Err(invalid("catalog magic differs"));
9258 }
9259 let count = cur.u32()? as usize;
9260 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
9261 for _ in 0..count {
9262 let name = cur.text()?;
9263 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9264 let width = cur.u16()? as usize;
9265 let mut fields = Vec::with_capacity(width);
9266 for _ in 0..width {
9267 let name = cur.text()?;
9268 let ty = read_type(&mut cur)?;
9269 let not_null = match cur.u8()? {
9270 0 => false,
9271 1 => true,
9272 _ => return Err(invalid("nullability flag differs")),
9273 };
9274 fields.push(Field { name, ty, not_null });
9275 }
9276 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9277 let end = directory
9278 .offset
9279 .checked_add(u64::from(directory.length))
9280 .ok_or_else(|| invalid("table directory offset overflow"))?;
9281 if directory.offset < HEADER
9282 || end > size
9283 || directory.length as usize > MAX_DIRECTORY
9284 || directory.length == 0
9285 {
9286 return Err(invalid("table directory range is outside the file"));
9287 }
9288 if entries.iter().any(|held| held.name == name) {
9289 return Err(invalid("two tables in the catalog have the same name"));
9290 }
9291 let nonzero = vec![None; fields.len()];
9292 let aggregates = vec![None; fields.len()];
9293 let distincts = vec![None; fields.len()];
9294 let extremes = vec![None; fields.len()];
9295 let frequencies = vec![None; fields.len()];
9296 entries.push(Entry {
9297 name,
9298 fields,
9299 rows,
9300 directory,
9301 nonzero,
9302 aggregates,
9303 distincts,
9304 extremes,
9305 frequencies,
9306 });
9307 }
9308 let count = if cur.done() { 0 } else { cur.u32()? as usize };
9313 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9314 for _ in 0..count {
9315 let name = cur.text()?;
9316 let sql = cur.long_text()?;
9317 let statement = cur.long_text()?;
9318 let width = cur.u16()? as usize;
9319 let mut aliases = Vec::with_capacity(width);
9320 for _ in 0..width {
9321 aliases.push(cur.text()?);
9322 }
9323 let width = cur.u16()? as usize;
9324 let mut columns = Vec::with_capacity(width);
9325 for _ in 0..width {
9326 let name = cur.text()?;
9327 let ty = read_type(&mut cur)?;
9328 let not_null = match cur.u8()? {
9329 0 => false,
9330 1 => true,
9331 _ => return Err(invalid("nullability flag differs")),
9332 };
9333 columns.push(Field { name, ty, not_null });
9334 }
9335 if views.iter().any(|held| held.name == name) {
9339 return Err(invalid("two views in the catalog have the same name"));
9340 }
9341 if entries.iter().any(|held| held.name == name) {
9342 return Err(invalid("a table and a view in the catalog have the same name"));
9343 }
9344 views.push(ViewEntry { name, sql, statement, aliases, columns });
9345 }
9346 if !cur.done() {
9347 if cur.take(8)? != NONZERO_COUNTS {
9348 return Err(invalid("catalog extension magic differs"));
9349 }
9350 for entry in &mut entries {
9351 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9352 *count = match cur.u8()? {
9353 0 => None,
9354 1 if matches!(
9355 field.ty,
9356 LogicalType::TinyInt
9357 | LogicalType::SmallInt
9358 | LogicalType::Integer
9359 | LogicalType::BigInt
9360 | LogicalType::UTinyInt
9361 | LogicalType::USmallInt
9362 | LogicalType::UInteger
9363 | LogicalType::UBigInt
9364 ) =>
9365 {
9366 let value = cur.u64()?;
9367 if value > entry.rows as u64 {
9368 return Err(invalid("nonzero count exceeds rows"));
9369 }
9370 Some(value)
9371 }
9372 _ => return Err(invalid("nonzero count tag or column type differs")),
9373 };
9374 }
9375 }
9376 }
9377 if !cur.done() {
9378 if cur.take(8)? != AGGREGATE_SUMS {
9379 return Err(invalid("aggregate catalog extension magic differs"));
9380 }
9381 for entry in &mut entries {
9382 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9383 *summary = match cur.u8()? {
9384 0 => None,
9385 1 if signed_integer(&field.ty) => {
9386 let sum = i128::from_le_bytes(
9387 cur.take(16)?
9388 .try_into()
9389 .map_err(|_| invalid("aggregate sum is truncated"))?,
9390 );
9391 let count = cur.u64()?;
9392 if count > entry.rows as u64 {
9393 return Err(invalid("aggregate count exceeds table rows"));
9394 }
9395 Some((sum, count))
9396 }
9397 _ => return Err(invalid("aggregate sum tag or column type differs")),
9398 };
9399 }
9400 }
9401 }
9402 if !cur.done() {
9403 if cur.take(8)? != DISTINCT_COUNTS {
9404 return Err(invalid("distinct catalog extension magic differs"));
9405 }
9406 for entry in &mut entries {
9407 for count in &mut entry.distincts {
9408 *count = match cur.u8()? {
9409 0 => None,
9410 1 => {
9411 let value = cur.u64()?;
9412 if value > entry.rows as u64 {
9413 return Err(invalid("distinct count exceeds table rows"));
9414 }
9415 Some(value)
9416 }
9417 _ => return Err(invalid("distinct count tag differs")),
9418 };
9419 }
9420 }
9421 }
9422 if !cur.done() {
9423 if cur.take(8)? != INTEGER_EXTREMES {
9424 return Err(invalid("integer extremes catalog extension magic differs"));
9425 }
9426 for entry in &mut entries {
9427 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9428 *extremes = match cur.u8()? {
9429 0 => None,
9430 1 if integer_or_date(&field.ty) => Some(None),
9431 2 if integer_or_date(&field.ty) => {
9432 let low = i128::from_le_bytes(
9433 cur.take(16)?
9434 .try_into()
9435 .map_err(|_| invalid("minimum is truncated"))?,
9436 );
9437 let high = i128::from_le_bytes(
9438 cur.take(16)?
9439 .try_into()
9440 .map_err(|_| invalid("maximum is truncated"))?,
9441 );
9442 if low > high {
9443 return Err(invalid("integer extremes are reversed"));
9444 }
9445 Some(Some((low, high)))
9446 }
9447 _ => return Err(invalid("integer extremes tag or type differs")),
9448 };
9449 }
9450 }
9451 }
9452 if !cur.done() {
9453 if cur.take(8)? != COMPLETE_FREQUENCIES {
9454 return Err(invalid("numeric frequency catalog extension magic differs"));
9455 }
9456 for entry in &mut entries {
9457 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9458 *frequencies = match cur.u8()? {
9459 0 => None,
9460 1 if integer_or_date(&field.ty) => {
9461 let len = cur.u8()? as usize;
9462 if len > MAX_CATALOG_FREQUENCIES {
9463 return Err(invalid("too many catalog numeric frequencies"));
9464 }
9465 let mut values = Vec::with_capacity(len);
9466 let mut total = 0_u64;
9467 for _ in 0..len {
9468 let value = match cur.u8()? {
9469 0 => None,
9470 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9471 |_| invalid("numeric frequency value is truncated"),
9472 )?)),
9473 _ => return Err(invalid("numeric frequency value tag differs")),
9474 };
9475 if values.iter().any(|(held, _)| *held == value) {
9476 return Err(invalid("numeric frequency value repeats"));
9477 }
9478 let count = cur.u64()?;
9479 total = total
9480 .checked_add(count)
9481 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9482 values.push((value, count));
9483 }
9484 if total != entry.rows as u64 {
9485 return Err(invalid("numeric frequencies do not cover table rows"));
9486 }
9487 Some(values)
9488 }
9489 _ => return Err(invalid("numeric frequency tag or type differs")),
9490 };
9491 }
9492 }
9493 }
9494 let mut card = None;
9495 let mut anchor = None;
9496 while !cur.done() {
9499 let tag = cur.take(8)?;
9500 if tag == DEVICE_CARD && card.is_none() && anchor.is_none() {
9501 let device = cur.text()?;
9502 let len = cur.u32()? as usize;
9503 if len > MAX_CARD {
9504 return Err(invalid("device card is longer than any card"));
9505 }
9506 card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9507 } else if tag == anchor::LOG_ANCHOR && anchor.is_none() {
9508 anchor = Some(LogAnchor::decode(&mut cur)?);
9509 } else {
9510 return Err(invalid("catalog extension magic differs or repeats"));
9511 }
9512 }
9513 Ok((entries, views, card, anchor))
9514}
9515
9516type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>, Option<LogAnchor>);
9518
9519const MAX_CARD: usize = 64 << 10;
9521
9522#[derive(Debug, Clone, PartialEq, Eq)]
9529struct KeptCard {
9530 device: String,
9531 bytes: Vec<u8>,
9532}
9533
9534fn directory_of(path: &Path) -> &Path {
9536 path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9537}
9538
9539fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9546 let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9547 return held;
9548 };
9549 match rudb_io::device::kept(&device) {
9550 Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9551 None => held,
9552 }
9553}
9554
9555fn remember_card(path: &Path, card: Option<&KeptCard>) {
9557 let Some(card) = card else { return };
9558 let dir = directory_of(path);
9559 let Ok(device) = rudb_io::device::device_key(dir) else { return };
9560 if device != card.device {
9561 return;
9562 }
9563 if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9564 rudb_io::device::remember(&device, decoded);
9565 }
9566}
9567
9568struct Cursor<'a> {
9576 bytes: &'a [u8],
9577 at: usize,
9578 window: Option<Window<'a>>,
9579}
9580
9581struct Window<'a> {
9583 file: &'a File,
9584 offset: u64,
9585 length: usize,
9586 start: usize,
9588 held: Vec<u8>,
9589 size: usize,
9591}
9592
9593const DIRECTORY_WINDOW: usize = 64 << 10;
9595
9596impl<'a> Cursor<'a> {
9597 fn new(bytes: &'a [u8]) -> Self {
9598 Self { bytes, at: 0, window: None }
9599 }
9600
9601 fn over(file: &'a File, offset: u64, length: usize) -> Self {
9603 let window =
9604 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9605 Self { bytes: &[], at: 0, window: Some(window) }
9606 }
9607
9608 fn len(&self) -> usize {
9610 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9611 }
9612
9613 fn ensure(&mut self, len: usize) -> Result<()> {
9615 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9616 if end > self.len() {
9617 return Err(invalid("directory is truncated"));
9618 }
9619 let Some(window) = &mut self.window else { return Ok(()) };
9620 if self.at < window.start || end > window.start + window.held.len() {
9621 let want = len.max(window.size).min(window.length - self.at);
9622 window.start = self.at;
9623 window.held.resize(want, 0);
9624 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9625 }
9626 Ok(())
9627 }
9628
9629 fn held(&self, at: usize, len: usize) -> &[u8] {
9631 match &self.window {
9632 Some(window) => &window.held[at - window.start..at - window.start + len],
9633 None => &self.bytes[at..at + len],
9634 }
9635 }
9636
9637 #[inline]
9639 fn peek(&mut self, len: usize) -> Result<&[u8]> {
9640 if self.window.is_none() {
9641 let bytes = self.bytes;
9642 return Ok(&bytes[self.at..self.end(len)?]);
9643 }
9644 self.ensure(len)?;
9645 Ok(self.held(self.at, len))
9646 }
9647
9648 #[inline]
9654 fn take(&mut self, len: usize) -> Result<&[u8]> {
9655 if self.window.is_none() {
9656 let bytes = self.bytes;
9657 let (at, end) = (self.at, self.end(len)?);
9658 self.at = end;
9659 return Ok(&bytes[at..end]);
9660 }
9661 self.take_windowed(len)
9662 }
9663
9664 fn skip(&mut self, len: usize) -> Result<()> {
9666 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9667 if end > self.len() {
9668 return Err(invalid("directory is truncated"));
9669 }
9670 self.at = end;
9671 Ok(())
9672 }
9673
9674 fn skip_bound(&mut self) -> Result<()> {
9675 match self.u8()? {
9676 0 => Ok(()),
9677 1 => self.skip(16),
9678 2 => self.skip(8),
9679 3 => {
9680 let length = self.u32()? as usize;
9681 self.skip(length)
9682 }
9683 4 => self.skip(17),
9684 _ => Err(invalid("a stored bound has an unknown tag")),
9685 }
9686 }
9687
9688 #[inline]
9690 fn end(&self, len: usize) -> Result<usize> {
9691 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9692 if end > self.bytes.len() {
9693 return Err(invalid("directory is truncated"));
9694 }
9695 Ok(end)
9696 }
9697
9698 #[inline(never)]
9700 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9701 self.ensure(len)?;
9702 self.at += len;
9703 Ok(self.held(self.at - len, len))
9704 }
9705 #[inline]
9706 fn u8(&mut self) -> Result<u8> {
9707 Ok(self.take(1)?[0])
9708 }
9709 #[inline]
9710 fn u16(&mut self) -> Result<u16> {
9711 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9712 }
9713 #[inline]
9714 fn u32(&mut self) -> Result<u32> {
9715 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9716 }
9717 #[inline]
9718 fn u64(&mut self) -> Result<u64> {
9719 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9720 }
9721 fn var_u64(&mut self) -> Result<u64> {
9722 let mut value = 0_u64;
9723 for shift in (0..=63).step_by(7) {
9724 let byte = self.u8()?;
9725 let part = u64::from(byte & 0x7f);
9726 if shift == 63 && part > 1 {
9727 return Err(invalid("frequency ordinal varint overflows"));
9728 }
9729 value |= part << shift;
9730 if byte & 0x80 == 0 {
9731 return Ok(value);
9732 }
9733 }
9734 Err(invalid("frequency ordinal varint is too long"))
9735 }
9736 fn bound(&mut self) -> Result<Option<Bound>> {
9745 let rest = self.len().saturating_sub(self.at);
9746 let mut want = 32;
9747 loop {
9748 let offered = self.peek(want.min(rest))?;
9749 let mut used = 0;
9750 match bounds::get(offered, &mut used) {
9751 Ok(bound) => {
9752 self.at += used;
9753 return Ok(bound);
9754 }
9755 Err(_) if want < rest => want *= 2,
9756 Err(error) => return Err(error),
9757 }
9758 }
9759 }
9760 fn text(&mut self) -> Result<String> {
9761 let len = self.u16()? as usize;
9762 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9763 }
9764 fn done(&self) -> bool {
9767 self.at >= self.len()
9768 }
9769 fn long_text(&mut self) -> Result<String> {
9776 let len = self.u32()? as usize;
9777 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9778 }
9779}
9780
9781fn decode_summary(
9783 cur: &mut Cursor<'_>,
9784 field: &Field,
9785 rows: usize,
9786 values: bool,
9787) -> Result<Option<FrequencySummary>> {
9788 let Some((entries, omitted_max)) = decode_summary_head(cur, field, rows)? else {
9789 return Ok(None);
9790 };
9791 let ordinals = {
9792 let ordinal_count = cur.u32()? as usize;
9793 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9794 return Err(invalid("frequency ordinal count exceeds its bound"));
9795 }
9796 let mut ordinals = Vec::with_capacity(ordinal_count);
9797 let mut previous = 0_u64;
9798 for at in 0..ordinal_count {
9799 let delta = cur.var_u64()?;
9800 if at != 0 && delta == 0 {
9801 return Err(invalid("frequency ordinals are not increasing"));
9802 }
9803 let ordinal = if at == 0 {
9804 delta
9805 } else {
9806 previous.checked_add(delta).ok_or_else(|| invalid("frequency ordinal overflows"))?
9807 };
9808 if ordinal >= rows as u64 {
9809 return Err(invalid("frequency ordinal is outside the table"));
9810 }
9811 ordinals.push(ordinal);
9812 previous = ordinal;
9813 }
9814 ordinals
9815 };
9816 let ordinal_entries = if values {
9817 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9818 for _ in 0..ordinals.len() {
9819 let entry = cur.u16()?;
9820 if entry as usize >= entries.len() {
9821 return Err(invalid("frequency ordinal value is outside its entries"));
9822 }
9823 ordinal_entries.push(entry);
9824 }
9825 ordinal_entries
9826 } else {
9827 Vec::new()
9828 };
9829 Ok(Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries, ordinal_bound: 0 }))
9830}
9831
9832fn decode_summary_head(
9835 cur: &mut Cursor<'_>,
9836 field: &Field,
9837 rows: usize,
9838) -> Result<Option<(Vec<FrequencyEntry>, u64)>> {
9839 Ok(match cur.u8()? {
9840 0 => None,
9841 1 => {
9842 let omitted_max = cur.u64()?;
9843 let count = cur.u32()? as usize;
9844 if count > FREQUENCY_ENTRIES {
9845 return Err(invalid("frequency entry count exceeds its bound"));
9846 }
9847 let mut entries = Vec::with_capacity(count);
9848 for _ in 0..count {
9850 let value = match cur.u8()? {
9851 0 => FrequencyValue::Null,
9852 1 => FrequencyValue::Integer(i128::from_le_bytes(
9853 cur.take(16)?.try_into().expect("sixteen bytes"),
9854 )),
9855 2 => FrequencyValue::Code(cur.u32()?),
9856 _ => return Err(invalid("frequency value tag differs")),
9857 };
9858 let valid = matches!(
9859 (&field.ty, value),
9860 (_, FrequencyValue::Null)
9861 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9862 | (
9863 LogicalType::TinyInt
9864 | LogicalType::SmallInt
9865 | LogicalType::Integer
9866 | LogicalType::BigInt
9867 | LogicalType::UTinyInt
9868 | LogicalType::USmallInt
9869 | LogicalType::UInteger
9870 | LogicalType::UBigInt
9871 | LogicalType::Date
9872 | LogicalType::Timestamp,
9873 FrequencyValue::Integer(_),
9874 )
9875 );
9876 if !valid {
9877 return Err(invalid("frequency value does not match its column"));
9878 }
9879 let count = cur.u64()?;
9880 if count == 0 || count > rows as u64 {
9881 return Err(invalid("frequency count is outside the table"));
9882 }
9883 entries.push(FrequencyEntry { value, count });
9884 }
9885 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9886 return Err(invalid("frequency entries are not descending"));
9887 }
9888 Some((entries, omitted_max))
9889 }
9890 _ => return Err(invalid("frequency summary tag differs")),
9891 })
9892}
9893
9894fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9896 let length = cur.u32()? as usize;
9897 let entries = cur.u32()? as usize;
9898 if entries > FREQUENCY_ENTRIES {
9899 return Err(invalid("frequency entry count exceeds its bound"));
9900 }
9901 if length == 0 {
9902 if entries != 0 {
9903 return Err(invalid("missing frequency synopsis has entries"));
9904 }
9905 return Ok(None);
9906 }
9907 if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9908 return Err(invalid("frequency synopsis span is outside the directory"));
9909 }
9910 Ok(Some((length, entries)))
9911}
9912
9913fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9916 match cur.u8()? {
9917 0 => Ok(()),
9918 1 => {
9919 cur.skip(8)?;
9920 let entries = cur.u32()? as usize;
9921 if entries > FREQUENCY_ENTRIES {
9922 return Err(invalid("frequency entry count exceeds its bound"));
9923 }
9924 for _ in 0..entries {
9925 match cur.u8()? {
9926 0 => {}
9927 1 => cur.skip(16)?,
9928 2 => cur.skip(4)?,
9929 _ => return Err(invalid("frequency value tag differs")),
9930 }
9931 cur.skip(8)?;
9932 }
9933 let ordinals = cur.u32()? as usize;
9934 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9935 return Err(invalid("frequency ordinal count exceeds its bound"));
9936 }
9937 for _ in 0..ordinals {
9938 cur.var_u64()?;
9939 }
9940 if values {
9941 cur.skip(ordinals * 2)?;
9942 }
9943 Ok(())
9944 }
9945 _ => Err(invalid("frequency summary tag differs")),
9946 }
9947}
9948
9949fn quick_nonzero(
9953 mut cur: Cursor<'_>,
9954 name: &str,
9955 fields: &[Field],
9956 rows: usize,
9957 wanted: usize,
9958) -> Result<Option<u64>> {
9959 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9960 return Err(invalid("table directory differs from the catalog"));
9961 }
9962 let width = cur.u16()? as usize;
9963 if width != fields.len() {
9964 return Err(invalid("table directory width differs from the catalog"));
9965 }
9966 for field in fields {
9967 let stored =
9968 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9969 if &stored != field {
9970 return Err(invalid("table directory schema differs from the catalog"));
9971 }
9972 }
9973 let mut dictionaries = Vec::with_capacity(width);
9974 for field in fields {
9975 let held = match cur.u8()? {
9976 0 => false,
9977 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9978 cur.skip(20)?;
9979 true
9980 }
9981 _ => return Err(invalid("dictionary page tag differs")),
9982 };
9983 dictionaries.push(held);
9984 }
9985 for _ in 0..width {
9986 match cur.u8()? {
9987 0 => {}
9988 1 => cur.skip(8)?,
9989 _ => return Err(invalid("distinct count tag differs")),
9990 }
9991 }
9992 if cur.u64()? != rows as u64 {
9993 return Err(invalid("table row count differs from the catalog"));
9994 }
9995 let stripes = cur.u32()? as usize;
9996 let mut total = 0_usize;
9997 let mut nulls = 0_u64;
9998 for _ in 0..stripes {
9999 let parts = cur.u32()? as usize;
10000 if parts == 0 || parts > STRIPE_PARTS {
10001 return Err(invalid("stripe part count is outside its bound"));
10002 }
10003 let mut stripe_rows = 0_usize;
10004 for _ in 0..parts {
10005 stripe_rows = stripe_rows
10006 .checked_add(cur.u32()? as usize)
10007 .ok_or_else(|| invalid("stripe row count overflow"))?;
10008 }
10009 total =
10010 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10011 cur.skip(12 + width * 12)?;
10012 for (field, held) in fields.iter().zip(&dictionaries) {
10013 if coded_type(&field.ty) && *held {
10014 cur.skip(20)?;
10015 }
10016 }
10017 for _ in 0..width * 2 {
10018 match cur.u8()? {
10019 0 => {}
10020 1 => cur.skip(20)?,
10021 _ => return Err(invalid("stripe page tag differs")),
10022 }
10023 }
10024 for column in 0..width {
10025 cur.skip_bound()?;
10026 cur.skip_bound()?;
10027 let count = cur.u32()? as u64;
10028 if count > stripe_rows as u64 {
10029 return Err(invalid("null count exceeds stripe rows"));
10030 }
10031 if column == wanted {
10032 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
10033 }
10034 cur.skip(1)?;
10035 match cur.u8()? {
10036 0 => {}
10037 1 => cur.skip(16)?,
10038 _ => return Err(invalid("a stripe sum has an unknown tag")),
10039 }
10040 }
10041 }
10042 if total != rows {
10043 return Err(invalid("table row count differs from stripes"));
10044 }
10045 if cur.done() {
10046 return Ok(None);
10047 }
10048 let magic = cur.take(8)?;
10049 let spanned = magic == FREQUENCIES_SPANS;
10050 let values = magic == FREQUENCIES || spanned;
10051 if !values && magic != FREQUENCIES_V2 {
10052 return Err(invalid("directory extension magic differs"));
10053 }
10054 if cur.u16()? as usize != width {
10055 return Err(invalid("frequency column count differs"));
10056 }
10057 for _ in 0..wanted {
10058 if spanned {
10059 if let Some((length, _)) = summary_span(&mut cur)? {
10060 cur.skip(length)?;
10061 }
10062 } else {
10063 skip_summary(&mut cur, values, rows)?;
10064 }
10065 }
10066 let summary = if spanned {
10067 let Some((length, entries)) = summary_span(&mut cur)? else {
10068 return Ok(None);
10069 };
10070 let start = cur.at;
10071 let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
10072 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10073 if cur.at - start != length || summary.entries.len() != entries {
10074 return Err(invalid("a stored synopsis differs from its directory span"));
10075 }
10076 summary
10077 } else {
10078 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
10079 return Ok(None);
10080 };
10081 summary
10082 };
10083 let zero = summary
10084 .entries
10085 .iter()
10086 .find(|entry| entry.value == FrequencyValue::Integer(0))
10087 .map(|entry| entry.count)
10088 .or_else(|| (summary.omitted_max == 0).then_some(0));
10089 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
10090}
10091
10092fn quick_integer_fold(
10095 file: &File,
10096 mut cur: Cursor<'_>,
10097 entry: &Entry,
10098 size: u64,
10099 wanted: usize,
10100 emit: &mut impl FnMut(i64, u64) -> Result<()>,
10101) -> Result<()> {
10102 let name = &entry.name;
10103 let fields = &entry.fields;
10104 let rows = entry.rows;
10105 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
10106 return Err(invalid("table directory differs from the catalog"));
10107 }
10108 let width = cur.u16()? as usize;
10109 if width != fields.len() {
10110 return Err(invalid("table directory width differs from the catalog"));
10111 }
10112 for field in fields {
10113 let stored =
10114 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
10115 if &stored != field {
10116 return Err(invalid("table directory schema differs from the catalog"));
10117 }
10118 }
10119 let mut dictionaries = Vec::with_capacity(width);
10120 for field in fields {
10121 dictionaries.push(match cur.u8()? {
10122 0 => false,
10123 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
10124 cur.skip(20)?;
10125 true
10126 }
10127 _ => return Err(invalid("dictionary page tag differs")),
10128 });
10129 }
10130 for _ in 0..width {
10131 match cur.u8()? {
10132 0 => {}
10133 1 => cur.skip(8)?,
10134 _ => return Err(invalid("distinct count tag differs")),
10135 }
10136 }
10137 if cur.u64()? != rows as u64 {
10138 return Err(invalid("table row count differs from the catalog"));
10139 }
10140 let stripes = cur.u32()? as usize;
10141 let mut total = 0_usize;
10142 let mut bytes = Vec::new();
10143 for _ in 0..stripes {
10144 let parts = cur.u32()? as usize;
10145 if parts == 0 || parts > STRIPE_PARTS {
10146 return Err(invalid("stripe part count is outside its bound"));
10147 }
10148 let mut part_rows = Vec::with_capacity(parts);
10149 for _ in 0..parts {
10150 let count = cur.u32()? as usize;
10151 if count == 0 {
10152 return Err(invalid("empty part"));
10153 }
10154 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
10155 part_rows.push(count);
10156 }
10157 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10158 let section = index_section(parts)?;
10159 let index_length =
10160 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
10161 if index.offset < HEADER
10162 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
10163 || index.length as usize != index_length
10164 {
10165 return Err(invalid("index page range is outside the file"));
10166 }
10167 cur.skip(wanted * 12)?;
10168 let page = Span { offset: cur.u64()?, length: cur.u32()? };
10169 if page.offset < HEADER
10170 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
10171 || page.length as usize > MAX_PAGE
10172 {
10173 return Err(invalid("column page range is outside the file"));
10174 }
10175 cur.skip((width - wanted - 1) * 12)?;
10176 for (field, held) in fields.iter().zip(&dictionaries) {
10177 if coded_type(&field.ty) && *held {
10178 cur.skip(20)?;
10179 }
10180 }
10181 for _ in 0..width * 2 {
10182 match cur.u8()? {
10183 0 => {}
10184 1 => cur.skip(20)?,
10185 _ => return Err(invalid("stripe page tag differs")),
10186 }
10187 }
10188 for _ in 0..width {
10189 cur.skip_bound()?;
10190 cur.skip_bound()?;
10191 cur.skip(5)?;
10192 match cur.u8()? {
10193 0 => {}
10194 1 => cur.skip(16)?,
10195 _ => return Err(invalid("a stripe sum has an unknown tag")),
10196 }
10197 }
10198 let spans = read_index_span(file, index, page, parts, wanted)?;
10199 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
10200 bytes.resize(span.length, 0);
10201 let at = page
10202 .offset
10203 .checked_add(span.start as u64)
10204 .ok_or_else(|| invalid("part range overflow"))?;
10205 read_at(file, at, &mut bytes)?;
10206 if checksum(&bytes) != span.hash {
10207 return Err(invalid("integer part checksum differs"));
10208 }
10209 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
10210 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
10211 check_integer_tally_value(value, &fields[wanted].ty)?;
10212 emit(value, count)
10213 })?;
10214 if decoded_rows != expected_rows {
10215 return Err(invalid("encoded integer part holds the wrong number of rows"));
10216 }
10217 } else {
10218 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
10219 if let Some(packed) = column.packed_parts() {
10220 let validity = column.validity();
10221 let all_valid = column.none_null();
10222 let base = packed.base();
10223 let mut codes = [0_u64; 64];
10224 for from in (0..expected_rows).step_by(codes.len()) {
10225 let count = (expected_rows - from).min(codes.len());
10226 packed.unpack(from, &mut codes[..count]);
10227 for (offset, &code) in codes[..count].iter().enumerate() {
10228 if all_valid || validity.is_valid(from + offset) {
10229 emit((base + i128::from(code)) as i64, 1)?;
10231 }
10232 }
10233 }
10234 continue;
10235 }
10236 let column = column.into_flat()?;
10237 let validity = column.validity();
10238 macro_rules! count_decoded {
10239 ($values:expr) => {
10240 for (row, &value) in $values.as_slice().iter().enumerate() {
10241 if validity.is_valid(row) {
10242 emit(i64::from(value), 1)?;
10243 }
10244 }
10245 };
10246 }
10247 match column.data() {
10248 Some(Data::Int8(values)) => count_decoded!(values),
10249 Some(Data::Int16(values)) => count_decoded!(values),
10250 Some(Data::Int32(values)) => count_decoded!(values),
10251 Some(Data::Int64(values)) => count_decoded!(values),
10252 _ => return Err(invalid("decoded integer part has the wrong type")),
10253 }
10254 }
10255 }
10256 }
10257 if total != rows {
10258 return Err(invalid("table row count differs from stripes"));
10259 }
10260 Ok(())
10261}
10262
10263fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
10264 let fits = match ty {
10265 LogicalType::TinyInt => i8::try_from(value).is_ok(),
10266 LogicalType::SmallInt => i16::try_from(value).is_ok(),
10267 LogicalType::Integer => i32::try_from(value).is_ok(),
10268 LogicalType::BigInt => true,
10269 _ => false,
10270 };
10271 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
10272}
10273
10274fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
10275 read_directory(Cursor::new(bytes), size, None)
10276}
10277
10278fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
10283 if cur.take(8)? != DIRECTORY {
10284 return Err(invalid("directory magic differs"));
10285 }
10286 let name = cur.text()?;
10287 let width = cur.u16()? as usize;
10288 let mut fields = Vec::with_capacity(width);
10289 for _ in 0..width {
10290 let name = cur.text()?;
10291 let ty = read_type(&mut cur)?;
10292 let not_null = match cur.u8()? {
10293 0 => false,
10294 1 => true,
10295 _ => return Err(invalid("nullability flag differs")),
10296 };
10297 fields.push(Field { name, ty, not_null });
10298 }
10299 let mut dictionaries = Vec::with_capacity(width);
10300 for field in &fields {
10301 dictionaries.push(match cur.u8()? {
10302 0 => None,
10303 tag if tag == dictionary_tag(&field.ty) => {
10304 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10305 let end = page
10306 .offset
10307 .checked_add(u64::from(page.length))
10308 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
10309 if page.offset < HEADER || end > size {
10314 return Err(invalid("dictionary page range is outside the file"));
10315 }
10316 Some(page)
10317 }
10318 _ => return Err(invalid("dictionary page tag differs")),
10319 });
10320 }
10321 let mut distincts = Vec::with_capacity(width);
10322 for _ in 0..width {
10323 distincts.push(match cur.u8()? {
10324 0 => None,
10325 1 => Some(cur.u64()?),
10326 _ => return Err(invalid("distinct count tag differs")),
10327 });
10328 }
10329 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10330 let count = cur.u32()? as usize;
10331 let mut stripes = Vec::with_capacity(count);
10332 let mut total = 0_usize;
10333 for _ in 0..count {
10334 let count = cur.u32()? as usize;
10335 if count == 0 || count > STRIPE_PARTS {
10336 return Err(invalid("stripe part count is outside its bound"));
10337 }
10338 let mut parts = Vec::with_capacity(count);
10339 let mut stripe_rows = 0_usize;
10340 for _ in 0..count {
10341 let rows = cur.u32()?;
10342 if rows == 0 {
10343 return Err(invalid("empty part"));
10344 }
10345 parts.push(rows);
10346 stripe_rows = stripe_rows
10347 .checked_add(rows as usize)
10348 .ok_or_else(|| invalid("stripe row count overflow"))?;
10349 }
10350 total =
10351 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10352 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10353 let section = index_section(count)?;
10354 let wanted = section
10355 .checked_mul(width)
10356 .and_then(|bytes| u32::try_from(bytes).ok())
10357 .ok_or_else(|| invalid("index page length overflow"))?;
10358 let end = index
10359 .offset
10360 .checked_add(u64::from(index.length))
10361 .ok_or_else(|| invalid("index page offset overflow"))?;
10362 if index.offset < HEADER || end > size || index.length != wanted {
10363 return Err(invalid("index page range is outside the file"));
10364 }
10365 let mut pages = Vec::with_capacity(width);
10366 for _ in 0..width {
10367 let offset = cur.u64()?;
10368 let length = cur.u32()?;
10369 let end = offset
10370 .checked_add(u64::from(length))
10371 .ok_or_else(|| invalid("page offset overflow"))?;
10372 if offset < HEADER || end > size || length as usize > MAX_PAGE {
10373 return Err(invalid("page range is outside the file"));
10374 }
10375 pages.push(Span { offset, length });
10376 }
10377 let mut memberships = vec![None; width];
10378 for (column, field) in fields.iter().enumerate() {
10379 if !coded_type(&field.ty) || dictionaries[column].is_none() {
10380 continue;
10381 }
10382 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10383 let end = page
10384 .offset
10385 .checked_add(u64::from(page.length))
10386 .ok_or_else(|| invalid("membership page offset overflow"))?;
10387 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10388 return Err(invalid("membership page range is outside the file"));
10389 }
10390 if page.length != 0 {
10393 memberships[column] = Some(page);
10394 }
10395 }
10396 let mut sieves = vec![None; width];
10397 for sieve in sieves.iter_mut().take(width) {
10398 match cur.u8()? {
10399 0 => continue,
10400 1 => {}
10401 _ => return Err(invalid("a sieve page has an unknown tag")),
10402 }
10403 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10404 let end = page
10405 .offset
10406 .checked_add(u64::from(page.length))
10407 .ok_or_else(|| invalid("sieve page offset overflow"))?;
10408 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10409 return Err(invalid("sieve page range is outside the file"));
10410 }
10411 *sieve = Some(page);
10412 }
10413 let mut part_ranges = vec![None; width];
10414 for held in part_ranges.iter_mut().take(width) {
10415 match cur.u8()? {
10416 0 => continue,
10417 1 => {}
10418 _ => return Err(invalid("a part range page has an unknown tag")),
10419 }
10420 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10421 let end = page
10422 .offset
10423 .checked_add(u64::from(page.length))
10424 .ok_or_else(|| invalid("part range page offset overflow"))?;
10425 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10426 return Err(invalid("part range page range is outside the file"));
10427 }
10428 *held = Some(page);
10429 }
10430 let mut ranges = Vec::with_capacity(width);
10431 for column in 0..width {
10432 let low = cur.bound()?;
10433 let high = cur.bound()?;
10434 let nulls = cur.u32()? as usize;
10435 if nulls > stripe_rows {
10436 return Err(invalid("null count exceeds stripe rows"));
10437 }
10438 let exact = cur.u8()? != 0;
10439 let sum = match cur.u8()? {
10440 0 => None,
10441 1 => Some(i128::from_le_bytes(
10442 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10443 )),
10444 _ => return Err(invalid("a stripe sum has an unknown tag")),
10445 };
10446 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10452 let low = low.map(|bound| scaled_as(bound, ty));
10453 let high = high.map(|bound| scaled_as(bound, ty));
10454 ranges.push(Range { low, high, nulls, exact, sum });
10455 }
10456 stripes.push(Stripe {
10457 rows: stripe_rows,
10458 parts,
10459 index,
10460 pages,
10461 memberships: Pages::from_slots(memberships)?,
10462 sieves: Pages::from_slots(sieves)?,
10463 part_ranges: Pages::from_slots(part_ranges)?,
10464 zone: Zone::from_ranges(ranges),
10465 });
10466 }
10467 if total != rows {
10468 return Err(invalid("table row count differs from stripes"));
10469 }
10470 let mut entry_counts = vec![0; width];
10473 let frequencies = if cur.done() {
10474 vec![None; width]
10475 } else {
10476 let frequency_magic = cur.take(8)?;
10477 let spanned = frequency_magic == FREQUENCIES_SPANS;
10478 let frequency_values = frequency_magic == FREQUENCIES || spanned;
10479 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10480 return Err(invalid("directory extension magic differs"));
10481 }
10482 if cur.u16()? as usize != width {
10483 return Err(invalid("frequency column count differs"));
10484 }
10485 let mut frequencies = Vec::with_capacity(width);
10486 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10487 if spanned {
10488 let Some((length, entries)) = summary_span(&mut cur)? else {
10489 frequencies.push(None);
10490 continue;
10491 };
10492 *entry_count = entries;
10493 let start = cur.at;
10494 if let Some(offset) = stored_at {
10495 cur.skip(length)?;
10496 frequencies.push(Some(Frequencies::Stored {
10497 span: Span {
10498 offset: offset
10499 .checked_add(start as u64)
10500 .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10501 length: u32::try_from(length)
10502 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10503 },
10504 values: true,
10505 entries,
10506 }));
10507 } else {
10508 let summary = decode_summary(&mut cur, field, rows, true)?
10509 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10510 if cur.at - start != length || summary.entries.len() != entries {
10511 return Err(invalid("a stored synopsis differs from its directory span"));
10512 }
10513 frequencies.push(Some(Frequencies::Held(summary)));
10514 }
10515 continue;
10516 }
10517 let start = cur.at;
10518 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10519 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10520 frequencies.push(match (summary, stored_at) {
10521 (None, _) => None,
10522 (Some(summary), None) => Some(Frequencies::Held(summary)),
10523 (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10524 span: Span {
10525 offset: offset + start as u64,
10526 length: u32::try_from(cur.at - start)
10527 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10528 },
10529 values: frequency_values,
10530 entries: summary.entries.len(),
10531 }),
10532 });
10533 }
10534 frequencies
10535 };
10536 let mut clustering = None;
10546 let mut sections = Vec::new();
10547 let mut pair_frequencies = Vec::new();
10548 let mut seen_pair_frequencies = false;
10549 let mut ordinal_bounds = Vec::new();
10550 let mut seen_ordinal_bounds = false;
10551 let mut frequency_texts = vec![Vec::new(); width];
10552 let mut seen_frequency_texts = false;
10553 let mut host_groups = None;
10554 let mut demoted = Vec::new();
10555 let mut seen_sections = false;
10556 let mut dictionary_payloads = Vec::new();
10557 let mut seen_payloads = false;
10558 let mut constraints = Constraints::default();
10559 let mut generation = 0;
10562 while !cur.done() {
10563 let mut tag = [0u8; 8];
10564 tag.copy_from_slice(cur.take(8)?);
10565 if &tag == PAIR_FREQUENCIES {
10566 if seen_pair_frequencies {
10567 return Err(invalid("directory names two pair frequency blocks"));
10568 }
10569 seen_pair_frequencies = true;
10570 let count = cur.u16()? as usize;
10571 if count > MAX_PAIR_FREQUENCIES {
10572 return Err(invalid("pair frequency count exceeds its bound"));
10573 }
10574 pair_frequencies = Vec::with_capacity(count);
10575 for _ in 0..count {
10576 let first = cur.u16()?;
10577 let second = cur.u16()?;
10578 let first_at = first as usize;
10579 let second_at = second as usize;
10580 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10581 return Err(invalid("pair frequency first column has no synopsis"));
10582 }
10583 let first_entries = entry_counts[first_at];
10584 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10585 || dictionaries.get(second_at).copied().flatten().is_none()
10586 {
10587 return Err(invalid("pair frequency second column has no stable dictionary"));
10588 }
10589 if pair_frequencies
10590 .iter()
10591 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10592 {
10593 return Err(invalid("directory repeats a pair frequency summary"));
10594 }
10595 let omitted_max = cur.u64()?;
10596 if omitted_max > rows as u64 {
10597 return Err(invalid("pair frequency omitted count exceeds the table"));
10598 }
10599 let entries_count = cur.u16()? as usize;
10600 if entries_count > FREQUENCY_ENTRIES {
10601 return Err(invalid("pair frequency entry count exceeds its bound"));
10602 }
10603 let mut entries = Vec::with_capacity(entries_count);
10604 for _ in 0..entries_count {
10605 let first_entry = cur.u16()?;
10606 if first_entry as usize >= first_entries {
10607 return Err(invalid("pair frequency anchor is outside its synopsis"));
10608 }
10609 let second = match cur.u8()? {
10610 0 => None,
10611 1 => Some(cur.u32()?),
10612 _ => return Err(invalid("pair frequency string tag differs")),
10613 };
10614 let count = cur.u64()?;
10615 if count == 0 || count > rows as u64 {
10616 return Err(invalid("pair frequency count is outside the table"));
10617 }
10618 entries.push(PairFrequencyEntry { first_entry, second, count });
10619 }
10620 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10621 return Err(invalid("pair frequency entries are not descending"));
10622 }
10623 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10624 }
10625 } else if &tag == ORDINAL_BOUNDS {
10626 if seen_ordinal_bounds {
10627 return Err(invalid("directory names two ordinal bound blocks"));
10628 }
10629 seen_ordinal_bounds = true;
10630 ordinal_bounds = vec![0; width];
10631 let count = cur.u16()? as usize;
10632 if count > width {
10633 return Err(invalid("ordinal bound count exceeds the columns"));
10634 }
10635 for _ in 0..count {
10636 let column = cur.u16()? as usize;
10637 let bound = cur.u64()?;
10638 if column >= width || frequencies.get(column).and_then(Option::as_ref).is_none() {
10639 return Err(invalid("ordinal bound names a column with no synopsis"));
10640 }
10641 if bound == 0 || bound > rows as u64 || ordinal_bounds[column] != 0 {
10642 return Err(invalid("ordinal bound is outside the table or repeated"));
10643 }
10644 ordinal_bounds[column] = bound;
10645 }
10646 } else if &tag == FREQUENCY_TEXTS {
10647 if seen_frequency_texts {
10648 return Err(invalid("directory names two frequency text blocks"));
10649 }
10650 seen_frequency_texts = true;
10651 let columns = cur.u16()? as usize;
10652 if columns > width {
10653 return Err(invalid("frequency text column count exceeds the schema"));
10654 }
10655 for _ in 0..columns {
10656 let column = cur.u16()? as usize;
10657 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10658 return Err(invalid("frequency text column is repeated or out of range"));
10659 }
10660 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10661 || dictionaries.get(column).copied().flatten().is_none()
10662 || frequencies.get(column).and_then(Option::as_ref).is_none()
10663 {
10664 return Err(invalid("frequency texts belong to a non-string synopsis"));
10665 }
10666 let count = cur.u16()? as usize;
10667 if count == 0 || count != entry_counts[column] {
10668 return Err(invalid("frequency text count differs from its synopsis"));
10669 }
10670 let mut texts = Vec::with_capacity(count);
10671 for _ in 0..count {
10672 texts.push(match cur.u8()? {
10673 0 => None,
10674 1 => {
10675 let length = cur.u32()? as usize;
10676 let bytes = cur.take(length)?.to_vec();
10677 if fields[column].ty == LogicalType::Varchar {
10678 std::str::from_utf8(&bytes)
10679 .map_err(|_| invalid("frequency text is not UTF-8"))?;
10680 }
10681 Some(bytes)
10682 }
10683 _ => return Err(invalid("frequency text tag differs")),
10684 });
10685 }
10686 frequency_texts[column] = texts;
10687 }
10688 } else if &tag == HOST_GROUPS {
10689 if host_groups.is_some() {
10690 return Err(invalid("directory names two host group blocks"));
10691 }
10692 let column = cur.u16()? as usize;
10693 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10694 || dictionaries.get(column).copied().flatten().is_none()
10695 {
10696 return Err(invalid("host groups belong to a non-string dictionary"));
10697 }
10698 let omitted_max = cur.u64()?;
10699 if omitted_max > rows as u64 {
10700 return Err(invalid("host group bound exceeds the table"));
10701 }
10702 let count = cur.u16()? as usize;
10703 if count > host::CAPACITY {
10704 return Err(invalid("host group count exceeds its bound"));
10705 }
10706 let mut entries = Vec::with_capacity(count);
10707 let mut bytes = 0_usize;
10708 for _ in 0..count {
10709 let host_len = cur.u32()? as usize;
10710 bytes =
10711 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10712 if bytes > host::BYTE_BUDGET {
10713 return Err(invalid("host groups exceed their byte budget"));
10714 }
10715 let host = std::str::from_utf8(cur.take(host_len)?)
10716 .map_err(|_| invalid("host is not UTF-8"))?
10717 .to_owned();
10718 let count = cur.u64()?;
10719 if count == 0 || count > rows as u64 {
10720 return Err(invalid("host group count exceeds the table"));
10721 }
10722 let bytes_sum = i128::from_le_bytes(
10723 cur.take(16)?
10724 .try_into()
10725 .map_err(|_| invalid("host length sum is truncated"))?,
10726 );
10727 if bytes_sum < 0 {
10728 return Err(invalid("host length sum is negative"));
10729 }
10730 let minimum_len = cur.u32()? as usize;
10731 bytes =
10732 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10733 if bytes > host::BYTE_BUDGET {
10734 return Err(invalid("host groups exceed their byte budget"));
10735 }
10736 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10737 .map_err(|_| invalid("host minimum is not UTF-8"))?
10738 .to_owned();
10739 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10740 }
10741 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10742 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10743 {
10744 return Err(invalid("host groups are not in certified order"));
10745 }
10746 host_groups = Some(host::HostSummary { column, omitted_max, entries });
10747 } else if &tag == CLUSTERING {
10748 if clustering.is_some() {
10749 return Err(invalid("directory names two clustering declarations"));
10750 }
10751 let bucket = Width::from_tag(cur.u8()?)
10752 .ok_or_else(|| invalid("clustering width tag differs"))?;
10753 let count = cur.u16()? as usize;
10754 let mut columns = Vec::with_capacity(count.min(fields.len()));
10755 for _ in 0..count {
10756 columns.push(u32::from(cur.u16()?));
10757 }
10758 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10761 invalid("stored clustering declaration does not match the table it is on")
10762 })?);
10763 } else if &tag == DEMOTED {
10764 if !demoted.is_empty() {
10765 return Err(invalid("directory names two demoted column blocks"));
10766 }
10767 let count = cur.u16()? as usize;
10768 if count == 0 || count > width {
10769 return Err(invalid("demoted column count is outside the schema"));
10770 }
10771 demoted = vec![false; width];
10772 for _ in 0..count {
10773 let column = cur.u16()? as usize;
10774 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10775 return Err(invalid("a demoted column is repeated or has no dictionary"));
10776 }
10777 demoted[column] = true;
10778 }
10779 } else if &tag == SECTIONS {
10780 if seen_sections {
10781 return Err(invalid("directory names two section tables"));
10782 }
10783 seen_sections = true;
10784 generation = cur.u64()?;
10785 let count = cur.u16()? as usize;
10786 if count > MAX_SECTIONS {
10787 return Err(invalid("section count exceeds its bound"));
10788 }
10789 sections = Vec::with_capacity(count);
10790 for _ in 0..count {
10793 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10794 }
10795 for held in §ions {
10796 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10797 return Err(invalid("a section's extent table overflows the file"));
10798 };
10799 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10803 return Err(invalid("a section's extent table is outside the file"));
10804 }
10805 if held.extents == 0 && held.extent_bytes != 0 {
10806 return Err(invalid("a section with no extents names an extent table"));
10807 }
10808 }
10809 } else if &tag == DICTIONARY_PAYLOADS {
10810 if seen_payloads {
10811 return Err(invalid("directory names two dictionary payload blocks"));
10812 }
10813 seen_payloads = true;
10814 let count = cur.u16()? as usize;
10815 if count != fields.len() {
10816 return Err(invalid("dictionary payload block does not match the table's columns"));
10817 }
10818 dictionary_payloads = Vec::with_capacity(count);
10819 for _ in 0..count {
10820 let bytes = cur.u64()?;
10821 if bytes > size {
10822 return Err(invalid("a dictionary payload is larger than the file"));
10823 }
10824 dictionary_payloads.push(bytes);
10825 }
10826 } else if &tag == KEYS {
10827 if !constraints.is_empty() {
10828 return Err(invalid("directory names two key blocks"));
10829 }
10830 let fits = |columns: &[u16]| {
10831 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10832 };
10833 let count = cur.u16()? as usize;
10834 for _ in 0..count {
10835 let primary = cur.u8()? != 0;
10836 let columns = columns_of(&mut cur)?;
10837 if !fits(&columns) {
10838 return Err(invalid("a stored key names a column the table does not have"));
10839 }
10840 constraints.keys.push((columns, primary));
10841 }
10842 let count = cur.u16()? as usize;
10843 for _ in 0..count {
10844 let columns = columns_of(&mut cur)?;
10845 let referenced = columns_of(&mut cur)?;
10846 let len = cur.u32()? as usize;
10847 let table = std::str::from_utf8(cur.take(len)?)
10848 .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10849 .to_owned();
10850 if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10851 return Err(invalid("a stored foreign key does not match its table"));
10852 }
10853 constraints.foreign.push(StoredForeign { columns, table, referenced });
10854 }
10855 if constraints.is_empty() {
10856 return Err(invalid("a key block holds no key"));
10857 }
10858 } else {
10859 return Err(invalid("directory extension magic differs"));
10860 }
10861 }
10862 if !cur.done() {
10863 return Err(invalid("directory has trailing bytes"));
10864 }
10865 for stripe in &stripes {
10866 for (column, field) in fields.iter().enumerate() {
10867 if coded_type(&field.ty)
10868 && dictionaries[column].is_some()
10869 && stripe.memberships.get(column).is_none()
10870 && !demoted.get(column).copied().unwrap_or(false)
10871 {
10872 return Err(invalid("string page has no code membership index"));
10873 }
10874 }
10875 }
10876 Ok(Table {
10877 name,
10878 fields,
10879 stripes,
10880 rows,
10881 dictionaries,
10882 dictionary_payloads,
10883 demoted,
10884 distincts,
10885 frequencies,
10886 ordinal_bounds,
10887 pair_frequencies,
10888 frequency_texts,
10889 host_groups,
10890 clustering,
10891 generation,
10892 sections,
10893 constraints,
10894 })
10895}
10896
10897fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10899 put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10900 Ok(())
10901}
10902
10903fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10905 put_count(out, columns.len())?;
10906 for &column in columns {
10907 put_u16(out, column);
10908 }
10909 Ok(())
10910}
10911
10912fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10914 let count = cur.u16()? as usize;
10915 (0..count).map(|_| cur.u16()).collect()
10916}
10917
10918fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10920 bounds::put(out, bound)
10921}
10922
10923#[derive(Debug)]
10940struct Codes;
10941
10942impl chooser::Chooser for Codes {
10943 fn name(&self) -> &'static str {
10944 "codes"
10945 }
10946
10947 fn narrow_strings(
10948 &self,
10949 _values: &[&[u8]],
10950 offered: &[string::Kind],
10951 _depth: u8,
10952 ) -> Vec<string::Kind> {
10953 offered.to_vec()
10956 }
10957
10958 fn narrow_integers(
10959 &self,
10960 _values: &[i64],
10961 offered: &[integer::Kind],
10962 depth: u8,
10963 ) -> Vec<integer::Kind> {
10964 narrowed_to(Codes::keep(depth), offered)
10967 }
10968
10969 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10970 Codes::keep(depth).contains(&kind)
10971 }
10972}
10973
10974impl Codes {
10975 fn keep(depth: u8) -> &'static [integer::Kind] {
10976 if depth == 0 {
10977 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10978 } else {
10979 &[integer::Kind::Constant, integer::Kind::Packed]
10980 }
10981 }
10982}
10983
10984fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10992 let narrowed: Vec<integer::Kind> =
10993 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10994 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10995}
10996
10997#[derive(Debug)]
11009struct Fixed;
11010
11011impl chooser::Chooser for Fixed {
11012 fn name(&self) -> &'static str {
11013 "fixed"
11014 }
11015
11016 fn narrow_strings(
11017 &self,
11018 _values: &[&[u8]],
11019 offered: &[string::Kind],
11020 _depth: u8,
11021 ) -> Vec<string::Kind> {
11022 offered.to_vec()
11023 }
11024
11025 fn narrow_integers(
11026 &self,
11027 _values: &[i64],
11028 offered: &[integer::Kind],
11029 depth: u8,
11030 ) -> Vec<integer::Kind> {
11031 narrowed_to(Fixed::keep(depth), offered)
11032 }
11033
11034 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
11035 Fixed::keep(depth).contains(&kind)
11036 }
11037}
11038
11039impl Fixed {
11040 fn keep(depth: u8) -> &'static [integer::Kind] {
11041 if depth == 0 {
11042 &[
11043 integer::Kind::Constant,
11044 integer::Kind::Packed,
11045 integer::Kind::Delta,
11046 integer::Kind::Rle,
11047 integer::Kind::Sparse,
11048 integer::Kind::Strided,
11049 ]
11050 } else {
11051 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
11052 }
11053 }
11054}
11055
11056fn widened(data: &Data) -> Option<Vec<i64>> {
11063 match data {
11064 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11065 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11066 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11067 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11068 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11069 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11070 Data::Int64(values) => Some(values.to_vec()),
11071 _ => None,
11072 }
11073}
11074
11075fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
11081 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
11082 let values = integer::decode_as::<T>(bytes)
11083 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
11084 if values.len() != rows {
11085 return Err(invalid("cascade page holds the wrong number of rows"));
11086 }
11087 Ok(values)
11088 }
11089 Ok(match ty {
11090 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
11091 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
11092 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11093 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
11094 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11095 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
11096 LogicalType::BigInt
11097 | LogicalType::Timestamp
11098 | LogicalType::Time
11099 | LogicalType::TimeTz
11100 | LogicalType::TimestampTz
11101 | LogicalType::TimestampS
11102 | LogicalType::TimestampMs
11103 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11104 LogicalType::Decimal { .. } => match ty.physical() {
11107 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11108 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11109 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11110 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
11111 },
11112 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
11113 })
11114}
11115
11116fn plain_width(ty: &LogicalType) -> Option<usize> {
11119 Some(match ty {
11120 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
11121 LogicalType::SmallInt | LogicalType::USmallInt => 2,
11122 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
11123 LogicalType::BigInt
11124 | LogicalType::Timestamp
11125 | LogicalType::Time
11126 | LogicalType::TimeTz
11127 | LogicalType::TimestampTz
11128 | LogicalType::TimestampS
11129 | LogicalType::TimestampMs
11130 | LogicalType::TimestampNs => 8,
11131 LogicalType::Decimal { .. } => match ty.physical() {
11132 PhysicalType::Int16 => 2,
11133 PhysicalType::Int32 => 4,
11134 PhysicalType::Int64 => 8,
11135 _ => return None,
11138 },
11139 _ => return None,
11140 })
11141}
11142
11143fn cascaded(
11149 flat: &Vector,
11150 ty: &LogicalType,
11151 packed: Option<&Packed<'_>>,
11152 settling: &mut Settling,
11153) -> Result<Option<Vec<u8>>> {
11154 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
11155 let Some(values) = widened(data) else { return Ok(None) };
11156 let plain = values.len().saturating_mul(width);
11157 let best = match packed {
11158 Some(packed) => plain.min(21 + size_of_val(packed.words())),
11160 None => plain,
11161 };
11162 let out = settling.encode(&values)?;
11163 Ok((out.len() < best).then_some(out))
11164}
11165
11166const SEARCH_EVERY: usize = 16;
11173
11174#[derive(Debug, Default)]
11180struct Settling {
11181 shape: Option<Shape>,
11184 since: usize,
11186 symbols: Option<Symbols>,
11188}
11189
11190#[derive(Debug)]
11193struct Symbols {
11194 shape: chooser::Settled,
11195 len: usize,
11198 payload: usize,
11199 since: usize,
11200}
11201
11202impl Settling {
11203 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
11211 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
11212 {
11213 let out = string::encode_fsst(values, &symbols.shape)?;
11214 let held = match &out {
11216 None => symbols.len == 0,
11217 Some(out) => {
11218 (out.len() as u128) * (symbols.payload as u128) * 4
11219 <= (symbols.len as u128) * (payload as u128) * 5
11220 }
11221 };
11222 if held {
11223 symbols.since += 1;
11224 return Ok(out);
11225 }
11226 }
11227 let shape = string::fsst_shape(values);
11228 let out = string::encode_fsst(values, &shape)?;
11229 let len = out.as_ref().map_or(0, Vec::len);
11230 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
11231 Ok(out)
11232 }
11233
11234 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
11241 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
11242 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
11243 let out = integer::encode_with(values, &replay)?;
11244 if !replay.held() {
11245 self.settle(&out, values.len(), replay.first_offered())?;
11246 return Ok(out);
11247 }
11248 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
11249 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
11250 self.since += 1;
11251 return Ok(out);
11252 }
11253 }
11254 let search = chooser::Replay::new(&[], &Fixed);
11256 let out = integer::encode_with(values, &search)?;
11257 self.settle(&out, values.len(), search.first_offered())?;
11258 Ok(out)
11259 }
11260
11261 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
11262 let kinds = integer::shape(out)?;
11263 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
11264 self.since = 0;
11265 Ok(())
11266 }
11267}
11268
11269#[derive(Debug)]
11271struct Shape {
11272 kinds: Vec<integer::Kind>,
11273 offered: Vec<integer::Kind>,
11274 len: usize,
11275 rows: usize,
11276}
11277
11278fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
11319 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
11320 let mut payload = 0_usize;
11321 for row in 0..flat.len() {
11322 let text = flat.bytes_at(row).unwrap_or(b"");
11325 payload = payload.saturating_add(text.len());
11326 values.push(text);
11327 }
11328 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
11330 let Some(out) = settling.text(&values, payload)? else {
11331 return Ok(None);
11332 };
11333 Ok((out.len() < plain).then_some(out))
11334}
11335
11336fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
11337 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
11338 let coded = integer::encode_with(&wide, &Codes)?;
11339 let plain = codes.len().saturating_mul(size_of::<u32>());
11340 Ok((coded.len() < plain).then_some(coded))
11341}
11342
11343fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
11346 let flag = match flat.validity() {
11347 Validity::AllValid => 0,
11348 Validity::AllInvalid => 1,
11349 Validity::Mask(_) => 2,
11350 };
11351 out.push(flag);
11352 if flag == 2 {
11353 for group in (0..flat.len()).step_by(8) {
11354 let mut bits = 0_u8;
11355 for bit in 0..8 {
11356 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11357 bits |= 1 << bit;
11358 }
11359 }
11360 out.push(bits);
11361 }
11362 }
11363}
11364
11365fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11372 let coded = encoded_codes(codes)?;
11373 let mut out = Vec::with_capacity(
11374 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11375 );
11376 out.push(if coded.is_some() { 4 } else { 3 });
11377 out.extend_from_slice(validity);
11378 match coded {
11379 Some(coded) => out.extend_from_slice(&coded),
11380 None => {
11381 for &code in codes {
11382 put_u32(&mut out, code);
11383 }
11384 }
11385 }
11386 Ok(out)
11387}
11388
11389fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11392 let ty = vector.logical_type();
11393 let flat = vector.flatten()?;
11395 let mut out = Vec::new();
11396 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11397 let compressed_text = if dictionary.is_none() && coded_type(ty) {
11398 text_compressed(&flat, settling)?
11399 } else {
11400 None
11401 };
11402 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11403 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11404 let cascade =
11408 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11409 out.push(if cascade.is_some() {
11410 5
11411 } else if dictionary.is_some() {
11412 1
11413 } else if compressed_text.is_some() {
11414 6
11415 } else if packed.is_some() {
11416 2
11417 } else {
11418 0
11419 });
11420 push_validity(&mut out, &flat);
11421 if let Some(cascade) = cascade {
11422 out.extend_from_slice(&cascade);
11423 return Ok(out);
11424 }
11425 if let Some(dictionary) = dictionary {
11426 out.extend_from_slice(&dictionary);
11427 return Ok(out);
11428 }
11429 if let Some(compressed_text) = compressed_text {
11430 out.extend_from_slice(&compressed_text);
11431 return Ok(out);
11432 }
11433 if let Some(packed) = packed {
11434 if packed.offset() != 0 {
11435 return Err(invalid("writer received a sliced packed vector"));
11436 }
11437 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11438 out.extend_from_slice(&packed.base().to_le_bytes());
11439 put_u32(
11440 &mut out,
11441 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11442 );
11443 for word in packed.words() {
11444 put_u64(&mut out, *word);
11445 }
11446 return Ok(out);
11447 }
11448 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11449 match (ty, data) {
11450 (LogicalType::TinyInt, Data::Int8(values)) => {
11451 for value in &**values {
11452 out.extend_from_slice(&value.to_le_bytes());
11453 }
11454 }
11455 (LogicalType::UTinyInt, Data::UInt8(values)) => {
11456 for value in &**values {
11457 out.extend_from_slice(&value.to_le_bytes());
11458 }
11459 }
11460 (LogicalType::SmallInt, Data::Int16(values)) => {
11461 for value in &**values {
11462 out.extend_from_slice(&value.to_le_bytes());
11463 }
11464 }
11465 (LogicalType::USmallInt, Data::UInt16(values)) => {
11466 for value in &**values {
11467 out.extend_from_slice(&value.to_le_bytes());
11468 }
11469 }
11470 (LogicalType::UInteger, Data::UInt32(values)) => {
11471 for value in &**values {
11472 out.extend_from_slice(&value.to_le_bytes());
11473 }
11474 }
11475 (LogicalType::UBigInt, Data::UInt64(values)) => {
11476 for value in &**values {
11477 out.extend_from_slice(&value.to_le_bytes());
11478 }
11479 }
11480 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11481 for value in &**values {
11482 out.extend_from_slice(&value.to_le_bytes());
11483 }
11484 }
11485 (
11486 LogicalType::BigInt
11487 | LogicalType::Timestamp
11488 | LogicalType::Time
11489 | LogicalType::TimeTz
11490 | LogicalType::TimestampTz
11491 | LogicalType::TimestampS
11492 | LogicalType::TimestampMs
11493 | LogicalType::TimestampNs,
11494 Data::Int64(values),
11495 ) => {
11496 for value in &**values {
11497 out.extend_from_slice(&value.to_le_bytes());
11498 }
11499 }
11500 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11503 for value in &**values {
11504 out.extend_from_slice(&value.to_le_bytes());
11505 }
11506 }
11507 (LogicalType::UHugeInt, Data::UInt128(values)) => {
11508 for value in &**values {
11509 out.extend_from_slice(&value.to_le_bytes());
11510 }
11511 }
11512 (LogicalType::Float, Data::Float32(values)) => {
11515 for value in &**values {
11516 out.extend_from_slice(&value.to_le_bytes());
11517 }
11518 }
11519 (LogicalType::Double, Data::Float64(values)) => {
11520 for value in &**values {
11521 out.extend_from_slice(&value.to_le_bytes());
11522 }
11523 }
11524 (LogicalType::Interval, Data::Interval(values)) => {
11528 for (months, days, micros) in &**values {
11529 out.extend_from_slice(&months.to_le_bytes());
11530 out.extend_from_slice(&days.to_le_bytes());
11531 out.extend_from_slice(µs.to_le_bytes());
11532 }
11533 }
11534 (LogicalType::Boolean, Data::Bool(values)) => {
11535 for value in &**values {
11536 out.push(u8::from(*value));
11537 }
11538 }
11539 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11542 for value in &**values {
11543 out.extend_from_slice(&value.to_le_bytes());
11544 }
11545 }
11546 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11547 for value in &**values {
11548 out.extend_from_slice(&value.to_le_bytes());
11549 }
11550 }
11551 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11552 for value in &**values {
11553 out.extend_from_slice(&value.to_le_bytes());
11554 }
11555 }
11556 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11557 for value in &**values {
11558 out.extend_from_slice(&value.to_le_bytes());
11559 }
11560 }
11561 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11566 let mut bytes = Vec::new();
11567 put_u32(&mut out, 0);
11568 for row in 0..vector.len() {
11569 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11570 bytes.extend_from_slice(value);
11571 put_u32(
11572 &mut out,
11573 u32::try_from(bytes.len())
11574 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11575 );
11576 }
11577 out.extend_from_slice(&bytes);
11578 }
11579 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11580 }
11581 Ok(out)
11582}
11583
11584fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11585 while value >= 0x80 {
11586 out.push((value as u8 & 0x7f) | 0x80);
11587 value >>= 7;
11588 }
11589 out.push(value as u8);
11590}
11591
11592fn unique_codes(codes: &[u32]) -> Vec<u32> {
11594 let mut unique = codes.to_vec();
11595 unique.sort_unstable();
11596 unique.dedup();
11597 unique
11598}
11599
11600fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11606 let mut lists = lists;
11607 while lists.len() > 1 {
11608 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11609 for pair in lists.chunks(2) {
11610 match pair {
11611 [left, right] => next.push(merged_pair(left, right)),
11612 [only] => next.push(only.clone()),
11613 _ => {}
11614 }
11615 }
11616 lists = next;
11617 }
11618 lists.pop().unwrap_or_default()
11619}
11620
11621fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11622 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11623 let mut at = 0;
11624 let mut to = 0;
11625 while at < left.len() && to < right.len() {
11626 match left[at].cmp(&right[to]) {
11627 Ordering::Less => {
11628 out.push(left[at]);
11629 at += 1;
11630 }
11631 Ordering::Greater => {
11632 out.push(right[to]);
11633 to += 1;
11634 }
11635 Ordering::Equal => {
11636 out.push(left[at]);
11637 at += 1;
11638 to += 1;
11639 }
11640 }
11641 }
11642 out.extend_from_slice(&left[at..]);
11643 out.extend_from_slice(&right[to..]);
11644 out
11645}
11646
11647fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11652 let mut merged = Range::default();
11653 let mut first = true;
11654 for range in ranges {
11655 merged.nulls = merged.nulls.saturating_add(range.nulls);
11656 merged.sum = match (merged.sum.take(), range.sum) {
11660 (Some(held), Some(next)) if !first => held.checked_add(next),
11661 (_, next) if first => next,
11662 _ => None,
11663 };
11664 merged.exact = if first { range.exact } else { merged.exact && range.exact };
11665 if first {
11666 merged.low = range.low;
11667 merged.high = range.high;
11668 first = false;
11669 continue;
11670 }
11671 merged.low = match (merged.low.take(), range.low) {
11672 (Some(held), Some(next)) => Some(held.smaller(next)),
11673 _ => None,
11674 };
11675 merged.high = match (merged.high.take(), range.high) {
11676 (Some(held), Some(next)) => Some(held.larger(next)),
11677 _ => None,
11678 };
11679 }
11680 merged
11681}
11682
11683fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11696 match bound {
11697 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11698 value.truncate(PART_BOUND_BYTES);
11699 if !high {
11700 return Some(Bound::Bytes(value));
11701 }
11702 while let Some(last) = value.pop() {
11703 if last < u8::MAX {
11704 value.push(last + 1);
11705 return Some(Bound::Bytes(value));
11706 }
11707 }
11708 None
11709 }
11710 other => other,
11711 }
11712}
11713
11714fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11722 let mut out = Vec::new();
11723 put_u32(
11724 &mut out,
11725 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11726 );
11727 for range in ranges {
11728 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11729 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11730 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11731 }
11732 Ok(out)
11733}
11734
11735fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11737 let mut cur = Cursor::new(bytes);
11738 let parts = cur.u32()? as usize;
11739 let mut out = Vec::new();
11740 for _ in 0..parts {
11741 let low = cur.bound()?;
11742 let high = cur.bound()?;
11743 let nulls = cur.u32()? as usize;
11744 out.push(Range { low, high, nulls, exact: false, sum: None });
11745 }
11746 Ok(out)
11747}
11748
11749fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11750 let held: Vec<&Option<Sieve>> = sieves.collect();
11751 let mut out = Vec::new();
11752 put_u32(
11753 &mut out,
11754 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11755 );
11756 for sieve in &held {
11757 let length = sieve.as_ref().map_or(0, Sieve::len);
11758 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11759 }
11760 for sieve in held.into_iter().flatten() {
11762 out.extend_from_slice(&sieve.to_bytes());
11763 }
11764 Ok(out)
11765}
11766
11767fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11773 let parts = u32::from_le_bytes(
11774 bytes
11775 .get(..4)
11776 .ok_or_else(|| invalid("sieve page is truncated"))?
11777 .try_into()
11778 .map_err(|_| invalid("sieve page is truncated"))?,
11779 ) as usize;
11780 let mut lengths = Vec::with_capacity(parts);
11781 for part in 0..parts {
11782 let at = 4 + part * 4;
11783 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11784 lengths.push(u32::from_le_bytes(
11785 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11786 ) as usize);
11787 }
11788 let mut at = 4 + parts * 4;
11789 let mut out = Vec::with_capacity(parts);
11790 for length in lengths {
11791 if length == 0 {
11792 out.push(None);
11793 continue;
11794 }
11795 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11796 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11797 out.push(Sieve::from_bytes(field));
11798 at = end;
11799 }
11800 if at != bytes.len() {
11801 return Err(invalid("sieve page has trailing bytes"));
11802 }
11803 Ok(out)
11804}
11805
11806fn encode_membership(unique: &[u32]) -> Vec<u8> {
11812 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11813 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11814 let mut previous = 0;
11815 for (at, &code) in unique.iter().enumerate() {
11816 put_varint(&mut out, if at == 0 { code } else { code - previous });
11817 previous = code;
11818 }
11819 out
11820}
11821
11822fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11823 let mut value = 0_u32;
11824 for shift in (0..35).step_by(7) {
11825 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11826 *at += 1;
11827 let part = u32::from(byte & 0x7f);
11828 if shift == 28 && part > 0x0f {
11829 return Err(invalid("membership varint overflow"));
11830 }
11831 value = value
11832 .checked_add(
11833 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11834 )
11835 .ok_or_else(|| invalid("membership varint overflow"))?;
11836 if byte & 0x80 == 0 {
11837 return Ok(value);
11838 }
11839 }
11840 Err(invalid("membership varint is too long"))
11841}
11842
11843fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11844 let mut at = 0;
11845 let count = take_varint(bytes, &mut at)? as usize;
11846 let mut codes = Vec::with_capacity(count);
11847 let mut previous = 0_u32;
11848 for index in 0..count {
11849 let delta = take_varint(bytes, &mut at)?;
11850 let code = if index == 0 {
11851 delta
11852 } else {
11853 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11854 };
11855 if index > 0 && code <= previous {
11856 return Err(invalid("membership codes are not increasing"));
11857 }
11858 codes.push(code);
11859 previous = code;
11860 }
11861 if at != bytes.len() {
11862 return Err(invalid("membership page has trailing bytes"));
11863 }
11864 Ok(codes)
11865}
11866
11867fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11875 let mut by_text: HashMap<&[u8], u32, Spread> =
11876 HashMap::with_capacity_and_hasher(vector.len(), Spread);
11877 let mut values = Vec::new();
11878 let mut codes = Vec::with_capacity(vector.len());
11879 let mut plain_bytes = 0_usize;
11880 for row in 0..vector.len() {
11881 let text = vector.bytes_at(row).unwrap_or(b"");
11882 plain_bytes = plain_bytes.saturating_add(text.len());
11883 let code = match by_text.get(text) {
11884 Some(&code) => code,
11885 None => {
11886 let code = u32::try_from(values.len())
11887 .map_err(|_| invalid("too many dictionary values"))?;
11888 by_text.insert(text, code);
11889 values.push(text);
11890 code
11891 }
11892 };
11893 codes.push(code);
11894 }
11895 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11896 let encoded = 8_usize
11897 .saturating_add((values.len() + 1).saturating_mul(4))
11898 .saturating_add(dictionary_bytes)
11899 .saturating_add(codes.len().saturating_mul(4));
11900 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11901 if encoded >= plain {
11902 return Ok(None);
11903 }
11904 let mut out = Vec::with_capacity(encoded);
11905 put_u32(
11906 &mut out,
11907 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11908 );
11909 put_u32(
11910 &mut out,
11911 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11912 );
11913 let mut offset = 0_u32;
11914 put_u32(&mut out, offset);
11915 for value in &values {
11916 offset = offset
11917 .checked_add(
11918 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11919 )
11920 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11921 put_u32(&mut out, offset);
11922 }
11923 for value in values {
11924 out.extend_from_slice(value);
11925 }
11926 for code in codes {
11927 put_u32(&mut out, code);
11928 }
11929 Ok(Some(out))
11930}
11931
11932struct Room<'a, T> {
11934 state: &'a Mutex<(T, usize)>,
11935 finished: &'a Condvar,
11936 bytes: usize,
11937}
11938
11939impl<T> Drop for Room<'_, T> {
11940 fn drop(&mut self) {
11941 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11942 held.1 -= self.bytes;
11943 drop(held);
11944 self.finished.notify_all();
11945 }
11946}
11947
11948enum Closing<'a> {
11950 Numeric {
11953 column: usize,
11954 counted: bool,
11955 dense: Option<(u64, usize)>,
11956 },
11957 Dictionary {
11958 index: usize,
11959 dictionary: &'a GlobalDictionary,
11960 },
11961}
11962
11963enum Closed {
11965 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11966 Dictionary(usize, ClosedDictionary),
11967}
11968
11969struct ClosedDictionary {
11971 distinct: Option<u64>,
11973 frequencies: Option<FrequencySummary>,
11974 texts: Vec<Option<Vec<u8>>>,
11975 hosts: Option<host::HostSummary>,
11976 encoded: EncodedDictionary,
11977 payload: u64,
11979}
11980
11981struct EncodedDictionary {
11982 index: Vec<u8>,
11983 ranks: Vec<u8>,
11984 grams: Vec<u8>,
11985}
11986
11987fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
12028 let mut work = vec![(0, codes.len(), 0)];
12029 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
12030 while let Some((from, to, depth)) = work.pop() {
12031 let part = &mut codes[from..to];
12032 keyed.clear();
12033 keyed.extend(part.iter().map(|&code| {
12034 let value = values(code);
12035 let rest = value.get(depth..).unwrap_or_default();
12036 (head(rest), rest.len().min(8) as u8, code)
12037 }));
12038 keyed.sort_unstable();
12039 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
12040 *slot = entry.2;
12041 }
12042 let mut start = 0;
12043 while start < keyed.len() {
12044 let (key, taken, _) = keyed[start];
12045 let mut end = start + 1;
12046 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
12047 end += 1;
12048 }
12049 if taken == 8 && end - start > 1 {
12050 work.push((from + start, from + end, depth + 8));
12051 }
12052 start = end;
12053 }
12054 }
12055}
12056
12057const PARALLEL_SORT_MIN: usize = 1 << 16;
12059
12060const BUCKETS_PER_WORKER: usize = 4;
12063
12064const SAMPLES_PER_BUCKET: usize = 32;
12066
12067fn sort_by_value_across<'a>(
12085 codes: &mut [u32],
12086 values: impl Fn(u32) -> &'a [u8] + Sync,
12087 workers: usize,
12088) {
12089 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
12090 sort_by_value(codes, values);
12091 return;
12092 }
12093 let buckets = workers * BUCKETS_PER_WORKER;
12094 let wanted = buckets * SAMPLES_PER_BUCKET;
12095 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
12096 sort_by_value(&mut sample, &values);
12097 let splitters =
12098 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
12099 let values = &values;
12100 let splitters = &splitters;
12101 let per = codes.len().div_ceil(workers);
12102 let places = std::thread::scope(|scope| {
12104 codes
12105 .chunks(per)
12106 .map(|run| {
12107 scope.spawn(move || {
12108 run.iter()
12109 .map(|&code| {
12110 let value = values(code);
12111 splitters.partition_point(|splitter| *splitter <= value) as u32
12112 })
12113 .collect::<Vec<_>>()
12114 })
12115 })
12116 .collect::<Vec<_>>()
12117 .into_iter()
12118 .flat_map(|handle| {
12119 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
12120 })
12121 .collect::<Vec<_>>()
12122 });
12123 let mut starts = vec![0_usize; buckets + 1];
12124 for &place in &places {
12125 starts[place as usize + 1] += 1;
12126 }
12127 for bucket in 0..buckets {
12128 starts[bucket + 1] += starts[bucket];
12129 }
12130 let mut laid = vec![0_u32; codes.len()];
12131 let mut next = starts.clone();
12132 for (&code, &place) in codes.iter().zip(&places) {
12133 laid[next[place as usize]] = code;
12134 next[place as usize] += 1;
12135 }
12136 drop(places);
12137 let mut runs = Vec::with_capacity(buckets);
12138 let mut rest = laid.as_mut_slice();
12139 for bucket in 0..buckets {
12140 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
12141 runs.push(run);
12142 rest = after;
12143 }
12144 runs.sort_by_key(|run| run.len());
12146 let queue = Mutex::new(runs);
12147 std::thread::scope(|scope| {
12148 for _ in 0..workers {
12149 scope.spawn(|| {
12150 loop {
12151 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
12152 let Some(run) = taken else { break };
12153 sort_by_value(run, values);
12154 }
12155 });
12156 }
12157 });
12158 codes.copy_from_slice(&laid);
12159}
12160
12161fn head(bytes: &[u8]) -> u64 {
12169 if let Some(word) = bytes.first_chunk::<8>() {
12170 return u64::from_be_bytes(*word);
12171 }
12172 let len = bytes.len();
12173 if len >= 4 {
12174 let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
12175 let back = &bytes[len - 4..];
12176 let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
12177 return (front << 32) | (back << (8 * (8 - len)));
12178 }
12179 bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
12180}
12181
12182fn encode_global_dictionary(
12193 dictionary: &GlobalDictionary,
12194 order: &[(u64, u32)],
12195 places: &[Placed],
12196 scattered: bool,
12197) -> Result<EncodedDictionary> {
12198 let values = dictionary.values();
12199 if order.len() != values {
12200 return Err(invalid("global dictionary order does not cover its values"));
12201 }
12202 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
12203 if places.len() != blocks {
12204 return Err(invalid("global dictionary payload is not the blocks it says it is"));
12205 }
12206 if dictionary.grams.len() != blocks {
12207 return Err(invalid("global dictionary signatures do not cover its blocks"));
12208 }
12209 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
12210 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
12211 let offset_bits = offset_width(&dictionary.ends);
12212 let payload_words = if scattered { 3 } else { 2 };
12213 let index_len = DICTIONARY_HEADER
12214 .checked_add(offset_bytes(values, offset_bits))
12215 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
12216 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12217 .and_then(|len| len.checked_add(8))
12218 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
12219 let mut index = Vec::with_capacity(index_len);
12220 put_u32(
12221 &mut index,
12222 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
12223 );
12224 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
12225 put_u32(
12226 &mut index,
12227 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
12228 );
12229 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
12230 | DICTIONARY_GRAMS
12231 | DICTIONARY_WIDE_GRAMS;
12232 put_u32(&mut index, offset_bits as u32 | flag);
12233 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
12234 let mut end = 0_u64;
12239 for place in places {
12240 if scattered {
12241 put_u64(&mut index, place.start);
12242 put_u64(&mut index, place.length);
12243 } else {
12244 end = end
12245 .checked_add(place.length)
12246 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
12247 put_u64(&mut index, end);
12248 }
12249 }
12250 for place in places {
12251 put_u64(&mut index, place.hash);
12252 }
12253 if rank_ends.len() != rank_blocks {
12256 return Err(invalid("global dictionary order is not the blocks it says it is"));
12257 }
12258 for end in &rank_ends {
12259 put_u64(&mut index, *end);
12260 }
12261 let mut at = 0_usize;
12262 for end in &rank_ends {
12263 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
12264 put_u64(&mut index, checksum(&ranks[at..end]));
12265 at = end;
12266 }
12267 let gram_len = blocks
12268 .checked_mul(TEXT_GRAM_BYTES)
12269 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
12270 let mut grams = Vec::with_capacity(gram_len);
12271 for block in &dictionary.grams {
12272 grams.extend_from_slice(block);
12273 }
12274 put_u64(&mut index, checksum(&grams));
12275 if index.len() != index_len {
12276 return Err(invalid("global dictionary index is not the length it was laid out for"));
12277 }
12278 Ok(EncodedDictionary { index, ranks, grams })
12279}
12280
12281const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
12288
12289fn payload_shapes() -> Vec<chooser::Settled> {
12315 let integers = vec![integer::Kind::Packed];
12316 [
12317 vec![string::Kind::Front, string::Kind::Lz],
12318 vec![string::Kind::Lz, string::Kind::Fsst],
12319 vec![string::Kind::Lz, string::Kind::Plain],
12320 vec![string::Kind::Fsst],
12321 vec![string::Kind::Plain],
12322 ]
12323 .into_iter()
12324 .map(|strings| chooser::Settled::new(strings, integers.clone()))
12325 .collect()
12326}
12327
12328fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
12335 let started = profile.map(|_| std::time::Instant::now());
12336 file.sync()?;
12337 if let (Some(profile), Some(started)) = (profile, started) {
12338 profile.waited(
12339 Stage::Publish,
12340 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
12341 );
12342 }
12343 Ok(())
12344}
12345
12346#[derive(Debug)]
12351pub(crate) struct Unencoded {
12352 column: usize,
12353 at: usize,
12354 ends: Vec<u32>,
12355 bytes: Vec<u8>,
12356 shape: chooser::Settled,
12357}
12358
12359impl Unencoded {
12360 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12362 let values = block_values(&self.ends, &self.bytes);
12363 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12364 }
12365
12366 pub(crate) fn place(&self) -> (usize, usize) {
12368 (self.column, self.at)
12369 }
12370}
12371
12372pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12376
12377fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12379 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12380 for value in values {
12381 for gram in value.windows(4) {
12382 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12383 grams[bit / 8] |= 1 << (bit % 8);
12384 }
12385 }
12386 }
12387 grams
12388}
12389
12390fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12392 let mut out = Vec::with_capacity(ends.len());
12393 let mut from = 0;
12394 for &to in ends {
12395 out.push(&bytes[from..to as usize]);
12396 from = to as usize;
12397 }
12398 out
12399}
12400
12401fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12408 for dictionary in dictionaries.iter_mut().flatten() {
12409 if !dictionary.early.is_empty() {
12410 return Err(Error::internal("a dictionary block handed out never came back"));
12411 }
12412 dictionary.seal_rest();
12413 dictionary.settle_rest()?;
12414 }
12415 encode_waiting(dictionaries)?;
12416 if dictionaries
12419 .iter()
12420 .flatten()
12421 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12422 {
12423 return Err(Error::internal("a dictionary block handed out never came back"));
12424 }
12425 Ok(())
12426}
12427
12428fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12431 let jobs = dictionaries
12432 .iter()
12433 .enumerate()
12434 .flat_map(|(column, held)| {
12435 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12436 })
12437 .collect::<Vec<_>>();
12438 if jobs.is_empty() {
12439 return Ok(());
12440 }
12441 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12442 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12443 Ok((column, at, held.encode_waiting(at)?))
12444 };
12445 let workers = std::thread::available_parallelism()
12446 .map_or(1, usize::from)
12447 .min(MAX_FREQUENCY_WORKERS)
12448 .min(jobs.len());
12449 let made = if workers <= 1 {
12450 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12451 } else {
12452 let next = AtomicUsize::new(0);
12453 let jobs = &jobs;
12454 let pieces = std::thread::scope(|scope| {
12455 (0..workers)
12456 .map(|_| {
12457 scope.spawn(|| {
12458 let mut mine = Vec::new();
12459 loop {
12460 let job = next.fetch_add(1, Atomic::Relaxed);
12461 let Some(&(column, at)) = jobs.get(job) else { break };
12462 mine.push(one(column, at)?);
12463 }
12464 Ok(mine)
12465 })
12466 })
12467 .collect::<Vec<_>>()
12468 .into_iter()
12469 .map(|handle| {
12470 handle
12471 .join()
12472 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12473 })
12474 .collect::<Result<Vec<_>>>()
12475 })?;
12476 pieces.into_iter().flatten().collect()
12477 };
12478 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12479 (0..dictionaries.len()).map(|_| Vec::new()).collect();
12480 for (column, at, bytes) in made {
12481 done[column].push((at, bytes));
12482 }
12483 for (column, mut made) in done.into_iter().enumerate() {
12484 if made.is_empty() {
12485 continue;
12486 }
12487 let Some(held) = dictionaries[column].as_mut() else { continue };
12488 made.sort_by_key(|(at, _)| *at);
12489 let waiting = std::mem::take(&mut held.waiting);
12490 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12491 if held.encoded() != at {
12492 return Err(Error::internal("a dictionary block was encoded out of order"));
12493 }
12494 held.push_block(block);
12495 }
12496 }
12497 Ok(())
12498}
12499
12500fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12510 let mut best: Option<(chooser::Settled, usize)> = None;
12511 for shape in payload_shapes() {
12512 let mut size = 0;
12513 for block in sample {
12514 size += string::encode_with(block, &shape)?.len();
12515 }
12516 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12517 best = Some((shape, size));
12518 }
12519 }
12520 best.map(|(shape, _)| shape)
12521 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12522}
12523
12524fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12531 let mut out = Vec::with_capacity(order.len() * 4);
12532 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12533 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12534 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12535 for block in order.chunks(TEXT_RANK_BLOCK) {
12536 let base = block.first().map_or(0, |&(head, _)| head);
12539 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12540 let width = (u64::BITS - span.leading_zeros()) as usize;
12541 heads.clear();
12542 codes.clear();
12543 for &(head, code) in block {
12544 heads.push(head.wrapping_sub(base));
12545 codes.push(u64::from(code));
12546 }
12547 put_u64(&mut out, base);
12548 out.push(width as u8);
12549 bitpack::pack_tail(&heads, width, &mut out)
12550 .map_err(|_| invalid("global dictionary heads do not pack"))?;
12551 bitpack::pack_tail(&codes, code_bits, &mut out)
12552 .map_err(|_| invalid("global dictionary codes do not pack"))?;
12553 ends.push(out.len() as u64);
12554 }
12555 Ok((out, ends))
12556}
12557
12558fn open_global_dictionary(
12565 file: Arc<File>,
12566 page: Page,
12567 ty: &LogicalType,
12568 keep_budget: usize,
12569) -> Result<Vector> {
12570 if !coded_type(ty) {
12571 return Err(invalid("global dictionary belongs to a non-string column"));
12572 }
12573 let mut header = [0; DICTIONARY_HEADER];
12574 read_at(&file, page.offset, &mut header)?;
12575 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12576 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12577 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12578 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12579 let scattered = width & DICTIONARY_SCATTERED != 0;
12580 let has_grams = width & DICTIONARY_GRAMS != 0;
12581 let gram_width =
12582 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12583 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12584 if per_block != TEXT_PAYLOAD_VALUES {
12585 return Err(invalid("global dictionary block width differs"));
12586 }
12587 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12588 return Err(invalid("global dictionary block count differs from its value count"));
12589 }
12590 if offset_bits > u32::BITS as usize {
12591 return Err(invalid("global dictionary packs offsets past a payload"));
12592 }
12593 let offset_len = offset_bytes(count, offset_bits);
12594 let ranks = count;
12599 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12600 let payload_words = if scattered { 3 } else { 2 };
12604 let hash_len = blocks
12605 .checked_mul(payload_words * 8)
12606 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12607 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12608 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12609 let gram_len = if has_grams {
12610 blocks
12611 .checked_mul(gram_width)
12612 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12613 } else {
12614 0
12615 };
12616 let index_len = DICTIONARY_HEADER
12617 .checked_add(offset_len)
12618 .and_then(|len| len.checked_add(hash_len))
12619 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12620 if index_len > page.length as usize {
12621 return Err(invalid("global dictionary offset index exceeds its page"));
12622 }
12623 let mut index = vec![0; index_len];
12624 index[..DICTIONARY_HEADER].copy_from_slice(&header);
12625 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12626 if checksum(&index) != page.hash {
12627 return Err(invalid("global dictionary index checksum differs"));
12628 }
12629 let word_end = index_len - usize::from(has_grams) * 8;
12630 let gram_hash = has_grams
12631 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12632 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12633 .chunks_exact(8)
12634 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12635 .collect::<Vec<_>>();
12636 let mut rest = words.split_off(blocks * payload_words);
12637 let rank_hashes = rest.split_off(rank_blocks);
12638 let rank_ends = rest;
12639 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12642 return Err(invalid("global dictionary order blocks do not rise"));
12643 }
12644 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12645 .map_err(|_| invalid("global dictionary rank overflow"))?;
12646 let body_len = index_len
12647 .checked_add(rank_len)
12648 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12649 if body_len > page.length as usize {
12650 return Err(invalid("global dictionary order exceeds its page"));
12651 }
12652 let gram_end = body_len
12653 .checked_add(gram_len)
12654 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12655 if gram_end > page.length as usize {
12656 return Err(invalid("global dictionary signatures exceed their page"));
12657 }
12658 let grams = gram_hash.map(|hash| NativeGrams {
12659 start: page.offset + body_len as u64,
12660 length: gram_len,
12661 width: gram_width,
12662 hash,
12663 verdicts: Mutex::new(Vec::new()),
12664 });
12665 let mut offsets = index;
12669 offsets.truncate(DICTIONARY_HEADER + offset_len);
12670 let hashes = words.split_off(blocks * (payload_words - 1));
12671 let (starts, lengths) = if scattered {
12672 let mut starts = Vec::with_capacity(blocks);
12673 let mut lengths = Vec::with_capacity(blocks);
12674 for pair in words.chunks_exact(2) {
12675 starts.push(pair[0]);
12676 lengths.push(pair[1]);
12677 }
12678 (starts, lengths)
12679 } else {
12680 let base = page.offset + gram_end as u64;
12684 let mut starts = Vec::with_capacity(blocks);
12685 let mut lengths = Vec::with_capacity(blocks);
12686 let mut at = 0_u64;
12687 for &end in &words {
12688 let len = end
12689 .checked_sub(at)
12690 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12691 starts.push(base + at);
12692 lengths.push(len);
12693 at = end;
12694 }
12695 (starts, lengths)
12696 };
12697 let stored_len = page.length as u64 - gram_end as u64;
12703 if scattered && stored_len == 0 {
12704 let size = file.metadata().map_err(io)?.len();
12705 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12706 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12707 });
12708 if !inside {
12709 return Err(invalid("global dictionary block lies outside the file"));
12710 }
12711 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12712 return Err(invalid("global dictionary blocks do not bound the payload"));
12713 }
12714 Vector::external_text(
12715 ty.clone(),
12716 Arc::new(NativeText {
12717 file,
12718 values: count,
12719 offsets,
12720 offset_bits,
12721 value_ends: OnceLock::new(),
12722 value_lens: OnceLock::new(),
12723 ends_asked: AtomicUsize::new(0),
12724 ranks,
12725 rank_at: page.offset + index_len as u64,
12726 rank_ends,
12727 rank_hashes,
12728 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12729 code_bits: code_width(count),
12730 code_ranks: OnceLock::new(),
12731 starts,
12732 lengths,
12733 hashes,
12734 grams,
12735 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12736 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12737 keep_budget,
12738 payload_kept: AtomicUsize::new(0),
12739 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12740 visit_dropped: AtomicUsize::new(0),
12741 searched: Mutex::new(HashMap::new()),
12742 }),
12743 )
12744}
12745
12746fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12759 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12761 let mut cur = Cursor::new(bytes);
12762 let codec = cur.u8()?;
12763 if cur.u8()? == 2 {
12764 cur.take(rows.div_ceil(8))?;
12765 }
12766 Ok((codec, cur.at))
12767 }
12768 let Ok((codec, at)) = cascade_at(rows, bytes) else {
12769 return "UNREADABLE".to_string();
12770 };
12771 let tail = &bytes[at..];
12772 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12773 match codec {
12774 0 => match ty {
12775 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12776 _ => "FIXED".to_string(),
12777 },
12778 1 => "DICT(PLAIN)".to_string(),
12779 2 => "FOR+BITPACK".to_string(),
12780 3 => "TABLE DICT".to_string(),
12781 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12782 5 => described(integer::describe(tail)),
12783 6 => described(string::describe(tail)),
12784 other => format!("CODEC {other}"),
12785 }
12786}
12787
12788fn decode_selected_stable_codes(
12793 rows: usize,
12794 bytes: &[u8],
12795 positions: &[usize],
12796 out: &mut Vec<Option<u32>>,
12797) -> Result<bool> {
12798 if positions.windows(2).any(|pair| pair[0] >= pair[1])
12799 || positions.last().is_some_and(|&position| position >= rows)
12800 {
12801 return Err(invalid("selected code positions are not sorted and in range"));
12802 }
12803 let mut cur = Cursor::new(bytes);
12804 let codec = cur.u8()?;
12805 if codec != 3 && codec != 4 {
12806 return Ok(false);
12807 }
12808 let flag = cur.u8()?;
12809 let mask = match flag {
12810 0 | 1 => None,
12811 2 => {
12812 let at = cur.at;
12813 let len = rows.div_ceil(8);
12814 cur.take(len)?;
12815 Some((at, len))
12816 }
12817 _ => return Err(invalid("page validity tag differs")),
12818 };
12819 let valid = |row: usize| match flag {
12820 0 => true,
12821 1 => false,
12822 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12823 _ => unreachable!("the validity tag was checked"),
12824 };
12825 if codec == 4 {
12826 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12827 for (&row, code) in positions.iter().zip(wide) {
12828 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12829 out.push(valid(row).then_some(code));
12830 }
12831 return Ok(true);
12832 }
12833 let codes_at = cur.at;
12834 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12835 cur.take(codes_len)?;
12836 if cur.at != bytes.len() {
12837 return Err(invalid("global code page has trailing bytes"));
12838 }
12839 let codes = &bytes[codes_at..codes_at + codes_len];
12840 for &row in positions {
12841 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12842 let code = u32::from_le_bytes(
12843 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12844 );
12845 out.push(valid(row).then_some(code));
12846 }
12847 Ok(true)
12848}
12849
12850fn decode_at(
12856 ty: &LogicalType,
12857 rows: usize,
12858 bytes: &[u8],
12859 global: Option<Arc<Vector>>,
12860 positions: &[u32],
12861) -> Result<Vector> {
12862 if positions.last().is_some_and(|&last| last as usize >= rows) {
12863 return Err(invalid("a position is past the end of the part"));
12864 }
12865 if bytes.first() == Some(&5)
12868 && positions.len().saturating_mul(8) <= rows
12869 && bytes
12871 .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12872 .is_some_and(|body| integer::pointed(body) || integer::run_length(body))
12873 {
12874 return cascade_at(ty, rows, bytes, positions);
12875 }
12876 if bytes.first() != Some(&6) {
12877 return decode(ty, rows, bytes, global)?.gather(positions);
12878 }
12879 if !coded_type(ty) {
12880 return Err(invalid("compressed text codec belongs to a non-string page"));
12881 }
12882 let mut cur = Cursor::new(bytes);
12883 cur.u8()?;
12884 let validity = match cur.u8()? {
12885 0 => Validity::AllValid,
12886 1 => Validity::AllInvalid,
12887 2 => {
12888 let mask = cur.take(rows.div_ceil(8))?;
12889 Validity::from_iter(positions.len(), |at| {
12890 let row = positions[at] as usize;
12891 mask[row / 8] >> (row % 8) & 1 == 1
12892 })
12893 }
12894 _ => return Err(invalid("page validity tag differs")),
12895 };
12896 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12897 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12898 push_values(&mut values, ty, &ends)?;
12899 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12900}
12901
12902fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12908 fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12909 values
12910 .iter()
12911 .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12912 .collect()
12913 }
12914 let mut cur = Cursor::new(bytes);
12915 cur.u8()?;
12916 let validity = match cur.u8()? {
12917 0 => Validity::AllValid,
12918 1 => Validity::AllInvalid,
12919 2 => {
12920 let mask = cur.take(rows.div_ceil(8))?;
12921 Validity::from_iter(positions.len(), |at| {
12922 let row = positions[at] as usize;
12923 mask[row / 8] >> (row % 8) & 1 == 1
12924 })
12925 }
12926 _ => return Err(invalid("page validity tag differs")),
12927 };
12928 let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12929 let values = integer::decode_selected(&bytes[cur.at..], &at)
12930 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12931 if values.len() != positions.len() {
12932 return Err(invalid("cascade page holds the wrong number of rows"));
12933 }
12934 let data = match ty {
12935 LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12936 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12937 LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12938 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12939 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12940 LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12941 LogicalType::BigInt
12942 | LogicalType::Timestamp
12943 | LogicalType::Time
12944 | LogicalType::TimeTz
12945 | LogicalType::TimestampTz
12946 | LogicalType::TimestampS
12947 | LogicalType::TimestampMs
12948 | LogicalType::TimestampNs => Data::Int64(values.into()),
12949 LogicalType::Decimal { .. } => match ty.physical() {
12950 PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12951 PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12952 PhysicalType::Int64 => Data::Int64(values.into()),
12953 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12954 },
12955 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12956 };
12957 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12958}
12959
12960fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12964 if ty == &LogicalType::Varchar {
12965 return values.push_run_in_place(0, ends);
12966 }
12967 let mut start = 0;
12968 for &end in ends {
12969 let len = end
12970 .checked_sub(start)
12971 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12972 values.push_bytes_in_place(start, len)?;
12973 start = end;
12974 }
12975 Ok(())
12976}
12977
12978fn decode(
12979 ty: &LogicalType,
12980 rows: usize,
12981 bytes: &[u8],
12982 global: Option<Arc<Vector>>,
12983) -> Result<Vector> {
12984 let mut cur = Cursor::new(bytes);
12985 let codec = cur.u8()?;
12986 let flag = cur.u8()?;
12987 let validity = match flag {
12988 0 => Validity::AllValid,
12989 1 => Validity::AllInvalid,
12990 2 => {
12991 let mask = cur.take(rows.div_ceil(8))?;
12992 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12993 }
12994 _ => return Err(invalid("page validity tag differs")),
12995 };
12996 if codec == 1 {
12997 if !coded_type(ty) {
12998 return Err(invalid("dictionary codec belongs to a non-string page"));
12999 }
13000 let count = cur.u32()? as usize;
13001 let payload_len = cur.u32()? as usize;
13002 let offset_bytes = cur.take(
13003 (count + 1)
13004 .checked_mul(4)
13005 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
13006 )?;
13007 let offsets = offset_bytes
13008 .chunks_exact(4)
13009 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13010 .collect::<Vec<_>>();
13011 let payload = cur.take(payload_len)?.to_vec();
13012 if offsets.first() != Some(&0)
13013 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13014 || offsets.windows(2).any(|pair| pair[0] > pair[1])
13015 {
13016 return Err(invalid("dictionary offsets do not bound the payload"));
13017 }
13018 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
13021 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13022 push_values(&mut strings, ty, &ends)?;
13023 let mut codes = Vec::with_capacity(rows);
13024 for _ in 0..rows {
13025 codes.push(cur.u32()?);
13026 }
13027 if codes.iter().any(|code| *code as usize >= count) {
13028 return Err(invalid("dictionary code is out of range"));
13029 }
13030 if cur.at != bytes.len() {
13031 return Err(invalid("dictionary page has trailing bytes"));
13032 }
13033 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
13034 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
13035 }
13036 if codec == 3 || codec == 4 {
13037 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
13038 let codes = if codec == 4 {
13039 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
13044 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
13045 if codes.len() != rows {
13046 return Err(invalid("encoded code page holds the wrong number of rows"));
13047 }
13048 codes
13049 } else {
13050 let mut codes = Vec::with_capacity(rows);
13051 for _ in 0..rows {
13052 codes.push(cur.u32()?);
13053 }
13054 if cur.at != bytes.len() {
13055 return Err(invalid("global code page has trailing bytes"));
13056 }
13057 codes
13058 };
13059 let highest = codes.iter().copied().max();
13060 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
13061 .with_validity(validity));
13062 }
13063 if codec == 6 {
13064 if !coded_type(ty) {
13065 return Err(invalid("compressed text codec belongs to a non-string page"));
13066 }
13067 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
13071 if ends.len() != rows {
13072 return Err(invalid("compressed text page holds the wrong number of rows"));
13073 }
13074 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13077 push_values(&mut values, ty, &ends)?;
13078 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
13079 }
13080 if codec == 5 {
13081 let data = cascade(ty, &bytes[cur.at..], rows)?;
13083 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
13084 }
13085 if codec == 2 {
13086 let width = u32::from(cur.u8()?);
13087 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
13088 let count = cur.u32()? as usize;
13089 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
13090 let words: Vec<u64> = cur
13091 .take(length)?
13092 .chunks_exact(8)
13093 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
13094 .collect();
13095 if cur.at != bytes.len() {
13096 return Err(invalid("packed page has trailing bytes"));
13097 }
13098 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
13099 }
13100 if codec != 0 {
13101 return Err(invalid("page codec is unknown"));
13102 }
13103 let data = match ty {
13104 LogicalType::TinyInt => {
13105 let values = cur.take(rows)?;
13106 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
13107 }
13108 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
13109 LogicalType::SmallInt => {
13110 let values =
13111 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13112 Data::Int16(
13113 values
13114 .chunks_exact(2)
13115 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13116 .collect::<Vec<_>>()
13117 .into(),
13118 )
13119 }
13120 LogicalType::USmallInt => {
13121 let values =
13122 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13123 Data::UInt16(
13124 values
13125 .chunks_exact(2)
13126 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
13127 .collect::<Vec<_>>()
13128 .into(),
13129 )
13130 }
13131 LogicalType::UInteger => {
13132 let values =
13133 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13134 Data::UInt32(
13135 values
13136 .chunks_exact(4)
13137 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
13138 .collect::<Vec<_>>()
13139 .into(),
13140 )
13141 }
13142 LogicalType::UBigInt => {
13143 let values =
13144 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13145 Data::UInt64(
13146 values
13147 .chunks_exact(8)
13148 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
13149 .collect::<Vec<_>>()
13150 .into(),
13151 )
13152 }
13153 LogicalType::Integer | LogicalType::Date => {
13154 let values =
13155 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13156 Data::Int32(
13157 values
13158 .chunks_exact(4)
13159 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13160 .collect::<Vec<_>>()
13161 .into(),
13162 )
13163 }
13164 LogicalType::BigInt
13165 | LogicalType::Timestamp
13166 | LogicalType::Time
13167 | LogicalType::TimeTz
13168 | LogicalType::TimestampTz
13169 | LogicalType::TimestampS
13170 | LogicalType::TimestampMs
13171 | LogicalType::TimestampNs => {
13172 let values =
13173 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13174 Data::Int64(
13175 values
13176 .chunks_exact(8)
13177 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13178 .collect::<Vec<_>>()
13179 .into(),
13180 )
13181 }
13182 LogicalType::HugeInt | LogicalType::Uuid => {
13183 let values =
13184 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13185 Data::Int128(
13186 values
13187 .chunks_exact(16)
13188 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13189 .collect::<Vec<_>>()
13190 .into(),
13191 )
13192 }
13193 LogicalType::UHugeInt => {
13194 let values =
13195 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13196 Data::UInt128(
13197 values
13198 .chunks_exact(16)
13199 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13200 .collect::<Vec<_>>()
13201 .into(),
13202 )
13203 }
13204 LogicalType::Float => {
13205 let values =
13206 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13207 Data::Float32(
13208 values
13209 .chunks_exact(4)
13210 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
13211 .collect::<Vec<_>>()
13212 .into(),
13213 )
13214 }
13215 LogicalType::Double => {
13216 let values =
13217 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13218 Data::Float64(
13219 values
13220 .chunks_exact(8)
13221 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
13222 .collect::<Vec<_>>()
13223 .into(),
13224 )
13225 }
13226 LogicalType::Interval => {
13227 let values =
13228 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13229 Data::Interval(
13230 values
13231 .chunks_exact(16)
13232 .map(|item| {
13233 (
13234 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
13235 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
13236 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
13237 )
13238 })
13239 .collect::<Vec<_>>()
13240 .into(),
13241 )
13242 }
13243 LogicalType::Boolean => {
13244 let values = cur.take(rows)?;
13245 if values.iter().any(|value| *value > 1) {
13246 return Err(invalid("boolean page has another value"));
13247 }
13248 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
13249 }
13250 LogicalType::Decimal { .. } => match ty.physical() {
13253 PhysicalType::Int16 => {
13254 let values =
13255 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13256 Data::Int16(
13257 values
13258 .chunks_exact(2)
13259 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13260 .collect::<Vec<_>>()
13261 .into(),
13262 )
13263 }
13264 PhysicalType::Int32 => {
13265 let values =
13266 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13267 Data::Int32(
13268 values
13269 .chunks_exact(4)
13270 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13271 .collect::<Vec<_>>()
13272 .into(),
13273 )
13274 }
13275 PhysicalType::Int64 => {
13276 let values =
13277 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13278 Data::Int64(
13279 values
13280 .chunks_exact(8)
13281 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13282 .collect::<Vec<_>>()
13283 .into(),
13284 )
13285 }
13286 _ => {
13287 let values =
13288 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13289 Data::Int128(
13290 values
13291 .chunks_exact(16)
13292 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13293 .collect::<Vec<_>>()
13294 .into(),
13295 )
13296 }
13297 },
13298 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
13299 let offset_bytes = cur
13300 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
13301 let offsets = offset_bytes
13302 .chunks_exact(4)
13303 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13304 .collect::<Vec<_>>();
13305 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
13306 if offsets.first() != Some(&0)
13307 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13308 || offsets.windows(2).any(|pair| pair[0] > pair[1])
13309 {
13310 return Err(invalid("string offsets do not bound the payload"));
13311 }
13312 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13320 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13321 push_values(&mut values, ty, &ends)?;
13322 Data::Varlen(values)
13323 }
13324 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
13325 };
13326 if cur.at != bytes.len() {
13327 return Err(invalid("page has trailing bytes"));
13328 }
13329 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
13330}
13331
13332#[cfg(test)]
13333mod tests {
13334 use std::fs::{self, OpenOptions};
13335 use std::io::{Seek, SeekFrom, Write};
13336 use std::path::PathBuf;
13337 use std::time::{SystemTime, UNIX_EPOCH};
13338
13339 use rudb_common::Stat;
13340 use rudb_common::Value;
13341 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
13342 use rudb_common::stat::Provenance;
13343
13344 use super::*;
13345
13346 #[test]
13347 fn head_is_the_value_padded_to_eight_bytes() {
13348 let bytes: Vec<u8> = (1..=12).collect();
13349 for len in 0..=bytes.len() {
13350 let value = &bytes[..len];
13351 let mut word = [0; 8];
13352 let take = len.min(8);
13353 word[..take].copy_from_slice(&value[..take]);
13354 assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13355 }
13356 assert!(head(b"ab") < head(b"ab\x01"));
13357 assert!(head(b"abcd") < head(b"abce"));
13358 }
13359
13360 #[test]
13361 fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13362 for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13363 let mut bytes = Vec::new();
13364 put_u32(&mut bytes, length);
13365 put_u32(&mut bytes, entries);
13366 bytes.push(1);
13367 assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13368 }
13369 let mut bytes = Vec::new();
13370 put_u32(&mut bytes, 1);
13371 put_u32(&mut bytes, 0);
13372 bytes.push(1);
13373 assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13374 }
13375
13376 #[test]
13377 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13378 let bytes: Vec<u8> =
13379 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13380 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13381 let whole = content_name(&bytes[..length]);
13382 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13383 let mut namer = ContentNamer::default();
13384 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13385 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13386 }
13387 }
13388 }
13389
13390 #[derive(Debug)]
13393 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13394
13395 impl chooser::Chooser for TestsEverything<'_> {
13396 fn name(&self) -> &'static str {
13397 "tests everything"
13398 }
13399
13400 fn narrow_strings(
13401 &self,
13402 values: &[&[u8]],
13403 offered: &[string::Kind],
13404 depth: u8,
13405 ) -> Vec<string::Kind> {
13406 self.0.narrow_strings(values, offered, depth)
13407 }
13408
13409 fn narrow_integers(
13410 &self,
13411 values: &[i64],
13412 offered: &[integer::Kind],
13413 depth: u8,
13414 ) -> Vec<integer::Kind> {
13415 self.0.narrow_integers(values, offered, depth)
13416 }
13417 }
13418
13419 #[test]
13420 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13421 let columns: Vec<Vec<i64>> = vec![
13422 vec![],
13423 vec![5; 1000],
13424 (0..1000).collect(),
13425 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13426 (0..1000).map(|row| row / 50).collect(),
13427 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13428 (0..1000).map(|row| (row * 7919) % 13).collect(),
13429 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13430 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13431 (0..1000).map(|row| i64::MIN + row % 3).collect(),
13432 ];
13433 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13434 for column in &columns {
13435 for chooser in choosers {
13436 let quick = integer::encode_with(column, chooser).unwrap();
13437 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13438 assert_eq!(
13439 quick,
13440 full,
13441 "{} on {:?}",
13442 chooser.name(),
13443 &column[..column.len().min(8)]
13444 );
13445 }
13446 }
13447 }
13448
13449 #[test]
13452 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13453 let mut settling = Settling::default();
13454 for part in 0..STRIPE_PARTS as i64 {
13455 let values: Vec<i64> = (0..2048)
13456 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13457 .collect();
13458 let searched = integer::encode_with(&values, &Fixed).unwrap();
13459 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13460 }
13461 }
13462
13463 #[test]
13467 fn text_pages_share_a_table_until_the_text_changes() {
13468 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13469 let english: Vec<Vec<u8>> = (0..1024)
13470 .map(|row: usize| {
13471 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13472 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13473 })
13474 .collect();
13475 let digits: Vec<Vec<u8>> =
13476 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13477 let mut settling = Settling::default();
13478 for page in 0..8 {
13479 let values: Vec<&[u8]> =
13480 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13481 let payload = values.iter().map(|value| value.len()).sum();
13482 let out = settling.text(&values, payload).unwrap().unwrap();
13483 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13484 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13485 assert!(
13486 out.len() * 4 <= alone.len() * 5,
13487 "page {page}: {} against {}",
13488 out.len(),
13489 alone.len()
13490 );
13491 let since = settling.symbols.as_ref().unwrap().since;
13492 assert_eq!(since, page % 4, "page {page}");
13493 }
13494 }
13495
13496 #[test]
13500 fn a_column_that_changes_under_the_shape_is_searched_again() {
13501 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13502 let mut noise = move || {
13503 state ^= state << 13;
13504 state ^= state >> 7;
13505 state ^= state << 17;
13506 (state % 1_000_000) as i64
13507 };
13508 let mut settling = Settling::default();
13509 for part in 0..STRIPE_PARTS as i64 {
13510 let values: Vec<i64> = match part / 16 {
13511 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13512 1 => (0..2048).map(|_| noise()).collect(),
13513 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13514 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13515 };
13516 let settled = settling.encode(&values).unwrap();
13517 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13518 let searched = integer::encode_with(&values, &Fixed).unwrap();
13519 assert!(
13520 settled.len() * 4 <= searched.len() * 5,
13521 "part {part}: {} settled against {} searched, {} against {}",
13522 settled.len(),
13523 searched.len(),
13524 integer::describe(&settled).unwrap(),
13525 integer::describe(&searched).unwrap(),
13526 );
13527 }
13528 }
13529
13530 #[test]
13531 fn checksum_matches_fixed_vectors() {
13532 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13533 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13534 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13535 }
13536
13537 #[test]
13538 fn sorting_across_threads_matches_sorting_on_one() {
13539 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13540 let mut next = move || {
13541 state ^= state << 13;
13542 state ^= state >> 7;
13543 state ^= state << 17;
13544 state
13545 };
13546 let mut values = Vec::new();
13547 for at in 0..150_000_u64 {
13548 let value = match next() % 6 {
13549 0 => Vec::new(),
13550 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13551 2 => format!("https://example.com/path/{at}").into_bytes(),
13552 3 => b"same".to_vec(),
13553 4 => vec![0xff; (next() % 12) as usize],
13554 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13555 };
13556 values.push(value);
13557 }
13558 let value = |code: u32| values[code as usize].as_slice();
13559 for workers in [1, 2, 3, 8, 32] {
13560 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13561 let mut across = one.clone();
13562 sort_by_value(&mut one, value);
13563 sort_by_value_across(&mut across, value, workers);
13564 assert_eq!(one, across, "{workers} workers");
13565 }
13566 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13567 sort_by_value_across(&mut sorted, value, 8);
13568 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13569 }
13570
13571 fn path(label: &str) -> PathBuf {
13572 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13573 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13574 }
13575
13576 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13581 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13582 (0..dictionary.values())
13583 .map(|code| {
13584 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13585 flat[from..to].to_vec()
13586 })
13587 .collect()
13588 }
13589
13590 fn attached(table: &Table) -> Vec<&Section> {
13597 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13598 }
13599
13600 #[test]
13602 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13603 const SPANS: usize = 64;
13604 const SPAN: usize = 512;
13605 let path = path("positional");
13606 let content: Vec<u8> =
13607 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13608 fs::write(&path, &content).expect("the file is written");
13609 let file = Arc::new(File::open(&path).expect("the file opens"));
13610 std::thread::scope(|scope| {
13611 for _ in 0..8 {
13612 let file = Arc::clone(&file);
13613 scope.spawn(move || {
13614 for _ in 0..64 {
13615 for span in 0..SPANS {
13616 let mut bytes = [0_u8; SPAN];
13617 read_at(&file, (span * SPAN) as u64, &mut bytes)
13618 .expect("the span reads");
13619 assert!(
13620 bytes.iter().all(|byte| *byte == span as u8),
13621 "span {span} came back as {}",
13622 bytes[0],
13623 );
13624 }
13625 }
13626 });
13627 }
13628 });
13629 let mut past = [0_u8; SPAN];
13630 let end = (SPANS * SPAN) as u64;
13631 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13632 assert!(error.message().contains("ends before its declared length"), "{error}");
13633 drop(file);
13634 let _ = fs::remove_file(&path);
13635 }
13636
13637 #[test]
13644 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13645 let path = path("cursor");
13646 let mut writer = Writer::create(
13647 &path,
13648 "items",
13649 vec![
13650 Field::required("id", LogicalType::Integer),
13651 Field::new("text", LogicalType::Varchar),
13652 ],
13653 )
13654 .expect("new file");
13655 writer.append(&sample()).expect("first part");
13656 writer.append(&sample()).expect("second part");
13657 writer.finish().expect("commit");
13658 let reader = Reader::open(&path).expect("reopen from disk");
13659 assert_eq!(reader.table().rows(), 6);
13660 let ids = reader.read(0, &[0]).expect("the integer page reads back");
13661 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13662 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13663 let text = reader.read(1, &[1]).expect("the text page reads back");
13664 assert_eq!(text.value_at(1, 0), Value::Null);
13665 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13666 let end = reader.table().stripes().iter().flat_map(|stripe| {
13669 stripe
13670 .pages
13671 .iter()
13672 .map(|page| page.offset + u64::from(page.length))
13673 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13674 });
13675 let last = end.fold(HEADER, u64::max);
13676 let directory = fs::metadata(&path).expect("the file is there").len();
13677 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13678 fs::remove_file(path).expect("remove scratch file");
13679 }
13680
13681 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13687 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13688 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13689 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13690 let bits = (width & !DICTIONARY_FLAGS) as usize;
13691 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13692 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13693 DICTIONARY_HEADER as u64
13694 + offset_bytes(count as usize, bits) as u64
13695 + blocks * payload_words * 8
13696 + rank_blocks * 16
13697 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13698 }
13699
13700 fn sample() -> Chunk {
13701 Chunk::new(vec![
13702 Vector::from_values(
13703 LogicalType::Integer,
13704 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13705 )
13706 .expect("integers"),
13707 Vector::from_values(
13708 LogicalType::Varchar,
13709 &[
13710 Value::Varchar("alpha".into()),
13711 Value::Null,
13712 Value::Varchar("long text after a slash".into()),
13713 ],
13714 )
13715 .expect("strings"),
13716 ])
13717 .expect("matching rows")
13718 }
13719
13720 fn sample_ids() -> Chunk {
13721 Chunk::new(vec![
13722 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13723 .expect("integers"),
13724 ])
13725 .expect("one column")
13726 }
13727
13728 #[test]
13729 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13730 let path = path("nulls_for_the_planner");
13733 let mut writer =
13734 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13735 .expect("new file");
13736 let rows = Chunk::new(vec![
13737 Vector::from_values(
13738 LogicalType::Integer,
13739 &[
13740 Value::Integer(4),
13741 Value::Null,
13742 Value::Integer(9),
13743 Value::Null,
13744 Value::Integer(1),
13745 Value::Integer(2),
13746 ],
13747 )
13748 .expect("integers"),
13749 ])
13750 .expect("one column");
13751 writer.append(&rows).expect("the only part");
13752 writer.finish().expect("commit");
13753 let reader = Reader::open(&path).expect("reopen from disk");
13754 let stripes = Stripes::new(reader);
13755 let column = stripes.column("a").expect("the file has that column");
13756 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13757 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13760 fs::remove_file(&path).expect("clean up");
13761 }
13762
13763 #[test]
13764 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13765 let path = path("frequencies_for_the_planner");
13768 let mut writer =
13769 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13770 .expect("new file");
13771 let rows = Chunk::new(vec![
13772 Vector::from_values(
13773 LogicalType::Integer,
13774 &[
13775 Value::Integer(4),
13776 Value::Integer(4),
13777 Value::Integer(4),
13778 Value::Integer(9),
13779 Value::Integer(9),
13780 Value::Integer(1),
13781 ],
13782 )
13783 .expect("integers"),
13784 ])
13785 .expect("one column");
13786 writer.append(&rows).expect("the only part");
13787 writer.finish().expect("commit");
13788 let reader = Reader::open(&path).expect("reopen from disk");
13789 let common = Common::new(reader);
13790 assert_eq!(common.rows(), 6);
13791 let column = common.column("id").expect("the file has that column");
13792 assert_eq!(common.column("nothing"), None);
13793 assert_eq!(
13794 common.rows_with(column, &Bound::Int(4)),
13795 Stat::exact(3, Provenance::FrequencySynopsis)
13796 );
13797 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13799 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13802 assert!(common.remainder(column).is_some());
13803 fs::remove_file(&path).expect("clean up");
13804 }
13805
13806 #[test]
13807 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13808 let path = path("string_frequencies_for_the_planner");
13809 let mut writer =
13810 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13811 .expect("new file");
13812 let rows = Chunk::new(vec![
13813 Vector::from_values(
13814 LogicalType::Varchar,
13815 &[
13816 Value::Varchar(String::new()),
13817 Value::Varchar("alpha".into()),
13818 Value::Varchar(String::new()),
13819 Value::Varchar("beta".into()),
13820 Value::Varchar(String::new()),
13821 ],
13822 )
13823 .expect("strings"),
13824 ])
13825 .expect("one column");
13826 writer.append(&rows).expect("the only part");
13827 writer.finish().expect("commit");
13828
13829 let reader = Reader::open(&path).expect("reopen from disk");
13830 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13831 let common = Common::new(reader.clone());
13832 let column = common.column("text").expect("the file has that column");
13833 assert_eq!(
13834 common.rows_with(column, &Bound::Bytes(Vec::new())),
13835 Stat::exact(3, Provenance::FrequencySynopsis)
13836 );
13837 assert_eq!(
13838 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13839 Stat::exact(0, Provenance::FrequencySynopsis)
13840 );
13841 assert_eq!(
13842 reader.reads().dictionaries,
13843 0,
13844 "the bounded spellings answer without opening the dictionary index"
13845 );
13846 fs::remove_file(&path).expect("clean up");
13847 }
13848
13849 #[test]
13850 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13851 let path = path("certified_host_groups");
13852 let mut writer =
13853 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13854 .expect("new file");
13855 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13856 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13857 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13858 values.push(Value::Varchar(String::new()));
13859 for part in values.chunks(512) {
13860 writer
13861 .append(
13862 &Chunk::new(vec![
13863 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13864 ])
13865 .expect("one column"),
13866 )
13867 .expect("part written");
13868 }
13869 writer.finish().expect("commit");
13870 let reader = Reader::open(&path).expect("reopen");
13871 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13872 fs::remove_file(&path).expect("clean up");
13873 }
13874
13875 fn bare_table(sections: Vec<Section>) -> Table {
13880 Table {
13881 name: "linked".to_owned(),
13882 fields: vec![Field::required("id", LogicalType::Integer)],
13883 stripes: Vec::new(),
13884 rows: 0,
13885 dictionaries: vec![None],
13886 dictionary_payloads: Vec::new(),
13887 demoted: Vec::new(),
13888 distincts: vec![None],
13889 frequencies: vec![None],
13890 ordinal_bounds: Vec::new(),
13891 pair_frequencies: Vec::new(),
13892 frequency_texts: Vec::new(),
13893 host_groups: None,
13894 clustering: None,
13895 constraints: Constraints::default(),
13896 generation: 1,
13897 sections,
13898 }
13899 }
13900
13901 fn a_key_map_section() -> Section {
13902 Section {
13903 kind: *section::KEY_MAP,
13904 id: 1,
13905 generation: 3,
13906 extents: 1,
13907 extent_page: HEADER,
13908 extent_bytes: section::EXTENT_BYTES as u32,
13909 hash: 0x1234_5678_9abc_def0,
13910 flags: 0,
13911 header_bytes: 24,
13912 }
13913 }
13914
13915 #[test]
13916 fn a_section_table_round_trips_through_a_directory() {
13917 let mut later = a_key_map_section();
13918 later.kind = *b"RUDBZZ9\0";
13919 later.id = 2;
13920 let table = bare_table(vec![a_key_map_section(), later]);
13921 let directory = encode_directory(&table).expect("directory");
13922 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13923 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13924 assert!(decoded.sections()[0].known());
13928 assert!(!decoded.sections()[1].known());
13929 }
13930
13931 #[test]
13932 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13933 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13937 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13938 let older = &directory[..directory.len() - block];
13939 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13940 assert!(decoded.sections().is_empty());
13941 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13942 assert_eq!(decoded.name(), "linked");
13943 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13944 }
13945
13946 #[test]
13947 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13948 let path = path("format_twenty_two");
13955 let mut writer =
13956 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13957 .expect("new file");
13958 let rows = Chunk::new(vec![
13959 Vector::from_values(
13960 LogicalType::Integer,
13961 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13962 )
13963 .expect("integers"),
13964 ])
13965 .expect("one column");
13966 writer.append(&rows).expect("the only part");
13967 writer.finish().expect("commit");
13968
13969 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13970 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13971 drop(file);
13972
13973 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13974 assert_eq!(reader.table().rows(), 3);
13975 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13980
13981 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13984 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13985 drop(file);
13986 let error = Reader::open(&path).expect_err("format 21 is not readable");
13987 assert!(error.to_string().contains("format 21"), "{error}");
13988
13989 fs::remove_file(&path).expect("clean up");
13990 }
13991
13992 #[test]
13993 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13994 let mut past = a_key_map_section();
13999 past.extent_page = 1 << 30;
14000 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
14001 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
14002 assert!(error.to_string().contains("outside the file"), "{error}");
14003
14004 let mut inside_the_header = a_key_map_section();
14005 inside_the_header.extent_page = 8;
14006 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
14007 assert!(
14008 decode_directory(&directory, 1 << 20).is_err(),
14009 "a section may not overlap a header"
14010 );
14011 }
14012
14013 #[test]
14014 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
14015 let not_built = Section {
14019 kind: *section::FORWARD_LINK,
14020 id: 9,
14021 generation: 3,
14022 extents: 0,
14023 extent_page: 0,
14024 extent_bytes: 0,
14025 hash: 0,
14026 flags: 0,
14027 header_bytes: 0,
14028 };
14029 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
14030 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
14031 assert_eq!(decoded.sections(), &[not_built]);
14032
14033 let mut incoherent = not_built;
14036 incoherent.extent_bytes = 28;
14037 incoherent.extent_page = HEADER;
14038 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
14039 assert!(decode_directory(&directory, 1 << 20).is_err());
14040 }
14041
14042 #[test]
14043 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
14044 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
14045 let mut torn = directory.clone();
14046 let count_at = torn.len() - size_of::<u16>();
14047 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
14048 assert!(decode_directory(&torn, 1 << 20).is_err());
14051 }
14052
14053 fn linked_file(label: &str, rows: i32) -> PathBuf {
14055 let path = path(label);
14056 let mut writer =
14057 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14058 .expect("new file");
14059 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
14060 let chunk =
14061 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
14062 .expect("one column");
14063 writer.append(&chunk).expect("the only part");
14064 writer.finish().expect("commit");
14065 path
14066 }
14067
14068 fn a_key_map_payload() -> Vec<u8> {
14069 (0..512_u32).flat_map(u32::to_le_bytes).collect()
14072 }
14073
14074 #[test]
14075 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
14076 let path = linked_file("attach", 64);
14077 let payload = a_key_map_payload();
14078 let table = attach(
14079 &path,
14080 "items",
14081 &[section::Attachment {
14082 kind: *section::KEY_MAP,
14083 id: 0,
14084 flags: 2,
14085 header_bytes: 40,
14086 bytes: &payload,
14087 }],
14088 )
14089 .expect("attach a key map");
14090 assert_eq!(attached(&table).len(), 1);
14091
14092 let reader = Reader::open(&path).expect("reopen after the attach");
14093 let held = attached(reader.table());
14094 assert_eq!(held.len(), 1);
14095 assert_eq!(held[0].kind, *section::KEY_MAP);
14096 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
14097 assert_eq!(held[0].header_bytes, 40);
14098 assert_eq!(held[0].generation, 1);
14102 assert!(held[0].usable(reader.table().generation()));
14103 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
14104 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
14105
14106 fs::remove_file(&path).expect("clean up");
14107 }
14108
14109 #[test]
14110 fn attaching_a_section_answers_every_row_exactly_as_before() {
14111 let path = linked_file("attach_changes_nothing", 300);
14116 let before = Reader::open(&path).expect("open before");
14117 let rows = before.table().rows();
14118 let first = before.read(0, &[0]).expect("read before");
14119 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
14120 let layout = before.layout().columns_total();
14121 drop(before);
14122
14123 let payload = a_key_map_payload();
14124 attach(
14125 &path,
14126 "items",
14127 &[section::Attachment {
14128 kind: *section::KEY_MAP,
14129 id: 0,
14130 flags: 0,
14131 header_bytes: 0,
14132 bytes: &payload,
14133 }],
14134 )
14135 .expect("attach");
14136
14137 let after = Reader::open(&path).expect("open after");
14138 assert_eq!(after.table().rows(), rows);
14139 let read = after.read(0, &[0]).expect("read after");
14140 for (at, value) in values.iter().enumerate() {
14141 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
14142 }
14143 assert_eq!(
14144 after.layout().columns_total(),
14145 layout,
14146 "an attach appends and does not rewrite a column page"
14147 );
14148
14149 fs::remove_file(&path).expect("clean up");
14150 }
14151
14152 #[test]
14153 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
14154 let path = linked_file("attach_twice", 32);
14158 let one = a_key_map_payload();
14159 let two = vec![7_u8; 1024];
14160 let entry = |bytes| section::Attachment {
14161 kind: *section::KEY_MAP,
14162 id: 4,
14163 flags: 1,
14164 header_bytes: 0,
14165 bytes,
14166 };
14167 attach(&path, "items", &[entry(&one)]).expect("first build");
14168 attach(&path, "items", &[entry(&two)]).expect("rebuild");
14169
14170 let reader = Reader::open(&path).expect("reopen");
14171 let held = attached(reader.table());
14172 assert_eq!(held.len(), 1, "one map per column and not one per build");
14173 assert_eq!(reader.payload(held[0]).expect("payload"), two);
14174
14175 fs::remove_file(&path).expect("clean up");
14176 }
14177
14178 #[test]
14179 fn an_attach_carries_through_a_kind_it_does_not_know() {
14180 let path = linked_file("attach_unknown", 16);
14184 let payload = vec![3_u8; 96];
14185 attach(
14186 &path,
14187 "items",
14188 &[section::Attachment {
14189 kind: *b"RUDBZZ9\0",
14190 id: 1,
14191 flags: 0,
14192 header_bytes: 0,
14193 bytes: &payload,
14194 }],
14195 )
14196 .expect("a kind this build does not know still writes");
14197 let key_map = a_key_map_payload();
14198 attach(
14199 &path,
14200 "items",
14201 &[section::Attachment {
14202 kind: *section::KEY_MAP,
14203 id: 0,
14204 flags: 0,
14205 header_bytes: 0,
14206 bytes: &key_map,
14207 }],
14208 )
14209 .expect("attach beside it");
14210
14211 let reader = Reader::open(&path).expect("reopen");
14212 let held = attached(reader.table());
14213 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
14214 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
14215 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
14216
14217 fs::remove_file(&path).expect("clean up");
14218 }
14219
14220 #[test]
14221 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
14222 let path = linked_file("attach_not_built", 8);
14223 attach(
14224 &path,
14225 "items",
14226 &[section::Attachment {
14227 kind: *section::FORWARD_LINK,
14228 id: 2,
14229 flags: 0,
14230 header_bytes: 0,
14231 bytes: &[],
14232 }],
14233 )
14234 .expect("record a link that did not fit the budget");
14235
14236 let reader = Reader::open(&path).expect("reopen");
14237 let held = attached(reader.table());
14238 assert_eq!(held.len(), 1);
14239 assert_eq!(held[0].extents, 0);
14240 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
14241 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
14242 assert!(reader.payload(held[0]).expect("no payload").is_empty());
14243
14244 fs::remove_file(&path).expect("clean up");
14245 }
14246
14247 #[test]
14248 fn a_payload_past_one_extent_is_split_and_joined_back() {
14249 let path = linked_file("attach_two_extents", 8);
14253 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
14254 attach(
14255 &path,
14256 "items",
14257 &[section::Attachment {
14258 kind: *section::KEY_MAP,
14259 id: 0,
14260 flags: 0,
14261 header_bytes: 0,
14262 bytes: &payload,
14263 }],
14264 )
14265 .expect("attach a payload past the bound");
14266
14267 let reader = Reader::open(&path).expect("reopen");
14268 let held = attached(reader.table());
14269 let extents = reader.extents(held[0]).expect("extent table");
14270 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
14271 assert_eq!(extents[0].length, section::MAX_EXTENT);
14272 assert_eq!(extents[1].length, 1);
14273 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
14274 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
14276 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
14277
14278 fs::remove_file(&path).expect("clean up");
14279 }
14280
14281 #[test]
14282 fn a_torn_extent_is_refused_rather_than_decoded() {
14283 let path = linked_file("attach_torn", 8);
14284 let payload = a_key_map_payload();
14285 attach(
14286 &path,
14287 "items",
14288 &[section::Attachment {
14289 kind: *section::KEY_MAP,
14290 id: 0,
14291 flags: 0,
14292 header_bytes: 0,
14293 bytes: &payload,
14294 }],
14295 )
14296 .expect("attach");
14297
14298 let reader = Reader::open(&path).expect("reopen");
14299 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
14300 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
14301 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
14302 drop(file);
14303
14304 let reader = Reader::open(&path).expect("the table still opens");
14305 let error = reader
14306 .payload(&reader.table().sections()[0])
14307 .expect_err("a corrupt payload is not handed out");
14308 assert!(error.to_string().contains("checksum"), "{error}");
14309 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
14312
14313 fs::remove_file(&path).expect("clean up");
14314 }
14315
14316 #[test]
14317 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
14318 let path = linked_file("attach_old_format", 8);
14321 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
14322 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
14323 drop(file);
14324
14325 let payload = a_key_map_payload();
14326 let error = attach(
14327 &path,
14328 "items",
14329 &[section::Attachment {
14330 kind: *section::KEY_MAP,
14331 id: 0,
14332 flags: 0,
14333 header_bytes: 0,
14334 bytes: &payload,
14335 }],
14336 )
14337 .expect_err("format 22 cannot gain a section");
14338 assert!(error.to_string().contains("format 22"), "{error}");
14339 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
14340
14341 fs::remove_file(&path).expect("clean up");
14342 }
14343
14344 #[test]
14345 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
14346 let path = linked_file("attach_bad_header", 8);
14347 let error = attach(
14348 &path,
14349 "items",
14350 &[section::Attachment {
14351 kind: *section::KEY_MAP,
14352 id: 0,
14353 flags: 0,
14354 header_bytes: 40,
14355 bytes: &[1, 2, 3],
14356 }],
14357 )
14358 .expect_err("a writer's bug stops at the write");
14359 assert!(error.to_string().contains("header is longer"), "{error}");
14360
14361 fs::remove_file(&path).expect("clean up");
14362 }
14363
14364 #[test]
14365 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14366 let path = linked_file("attach_wrong_name", 8);
14367 let error = attach(&path, "orders", &[]).expect_err("no such table");
14368 assert!(error.to_string().contains("orders"), "{error}");
14369 fs::remove_file(&path).expect("clean up");
14370 }
14371
14372 #[test]
14373 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14374 let path = path("frequency_prefix_for_the_planner");
14381 let mut writer =
14382 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14383 .expect("new file");
14384 let mut values = vec![Value::Integer(1); 10_000];
14385 for _ in 0..10 {
14386 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14387 }
14388 for part in values.chunks(8_000) {
14391 let rows = Chunk::new(vec![
14392 Vector::from_values(LogicalType::Integer, part).expect("integers"),
14393 ])
14394 .expect("one column");
14395 writer.append(&rows).expect("a part");
14396 }
14397 writer.finish().expect("commit");
14398 let reader = Reader::open(&path).expect("reopen from disk");
14399 let prefix =
14400 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14401 assert_eq!(prefix.entries.len(), 512);
14404 assert_eq!(prefix.omitted_max, 10);
14405 let common = Common::new(reader);
14406 assert_eq!(common.rows(), 16_000);
14407 let column = common.column("id").expect("the file has that column");
14408 assert_eq!(
14409 common.rows_with(column, &Bound::Int(1)),
14410 Stat::exact(10_000, Provenance::FrequencySynopsis)
14411 );
14412 assert_eq!(
14414 common.rows_with(column, &Bound::Int(1_100)),
14415 Stat::exact(10, Provenance::FrequencySynopsis)
14416 );
14417 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14420 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14423 let remainder = common.remainder(column).expect("the list is a prefix");
14427 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14428 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14429 fs::remove_file(&path).expect("clean up");
14430 }
14431
14432 #[test]
14434 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14435 let path = path("empty");
14436 Writer::empty(&path, &[], None).expect("a file with nothing in it");
14437 let catalog = Catalog::open(&path).expect("the empty file opens");
14438 assert_eq!(catalog.len(), 0);
14439 assert!(catalog.is_empty());
14440 assert_eq!(catalog.names().count(), 0);
14441 let mut writer =
14444 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14445 .expect("a table goes into the empty file");
14446 writer.append(&sample_ids()).expect("rows");
14447 writer.finish().expect("commit");
14448 let catalog = Catalog::open(&path).expect("the file opens again");
14449 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14450 fs::remove_file(&path).expect("clean up");
14451 }
14452
14453 #[test]
14463 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14464 let path = path("empty-name");
14465 let field = || vec![Field::required("id", LogicalType::Integer)];
14466 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14467 let catalog = Catalog::open(&path).expect("the file opens");
14468 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14469
14470 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14471 writer.append(&sample_ids()).expect("rows");
14472 writer.finish().expect("commit");
14473 let catalog = Catalog::open(&path).expect("the file opens again");
14474 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14476 let held = catalog.rows().collect::<Vec<_>>();
14477 assert_eq!(held.len(), 1);
14478 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14479
14480 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14482 assert!(error.to_string().contains("same name"), "{error}");
14483 fs::remove_file(&path).expect("clean up");
14484 }
14485
14486 #[test]
14487 fn a_log_anchor_rides_the_catalog_after_the_card_and_goes_forward_with_every_commit() {
14488 let entry = || Entry {
14489 name: "items".to_string(),
14490 fields: vec![Field::required("id", LogicalType::Integer)],
14491 rows: 1,
14492 directory: Page { offset: HEADER, length: 8, hash: 0 },
14493 nonzero: vec![None],
14494 aggregates: vec![None],
14495 distincts: vec![None],
14496 extremes: vec![None],
14497 frequencies: vec![None],
14498 };
14499 let anchor = LogAnchor {
14500 database: 0xfeed,
14501 durable: 41,
14502 lanes: vec![LaneStart { sequence: 3, offset: 4096 }],
14503 voids: vec![43, 47],
14504 };
14505 assert!(!anchor.replays(41) && anchor.replays(42) && !anchor.replays(43));
14506 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14507 for card in [None, Some(&card)] {
14508 let bytes = encode_catalog(&[entry()], &[], card, Some(&anchor)).expect("encodes");
14509 let (_, _, kept, held) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14510 assert_eq!((kept.as_ref(), held.as_ref()), (card, Some(&anchor)));
14511 }
14512 let mut twice = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14513 anchor.encode(&mut twice).expect("encodes");
14514 assert!(decode_catalog(&twice, HEADER + 8).is_err(), "a second anchor");
14515 let mut after = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14516 after.extend_from_slice(DEVICE_CARD);
14517 assert!(decode_catalog(&after, HEADER + 8).is_err(), "a card after the anchor");
14518 let under = LogAnchor { voids: vec![40], ..anchor.clone() };
14519 let bytes = encode_catalog(&[entry()], &[], None, Some(&under)).expect("encodes");
14520 assert!(decode_catalog(&bytes, HEADER + 8).is_err(), "a void under the cut");
14521
14522 let path = path("anchored");
14523 let mut writer =
14524 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14525 .expect("new file");
14526 writer.append(&sample_ids()).expect("rows");
14527 writer.with_log_anchor(anchor.clone()).finish().expect("commit");
14528 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14529 Writer::restate(&path, &[sample_view("v")], None).expect("a view");
14530 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14531 let mut writer =
14532 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14533 .expect("a second table");
14534 writer.append(&sample_ids()).expect("rows");
14535 writer.finish().expect("commit");
14536 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14537 let next = LogAnchor { durable: 90, voids: Vec::new(), ..anchor };
14538 Writer::restate(&path, &[], Some(&next)).expect("a new cut");
14539 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&next));
14540 fs::remove_file(&path).expect("clean up");
14541 let empty = self::path("anchoredempty");
14542 Writer::empty(&empty, &[], Some(&next)).expect("an empty file");
14543 assert_eq!(Catalog::open(&empty).expect("reopen").log_anchor(), Some(&next));
14544 fs::remove_file(&empty).expect("clean up");
14545 }
14546
14547 #[test]
14548 fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14549 let entry = || Entry {
14550 name: "items".to_string(),
14551 fields: vec![Field::required("id", LogicalType::Integer)],
14552 rows: 1,
14553 directory: Page { offset: HEADER, length: 8, hash: 0 },
14554 nonzero: vec![None],
14555 aggregates: vec![None],
14556 distincts: vec![None],
14557 extremes: vec![None],
14558 frequencies: vec![None],
14559 };
14560 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14561 let bytes = encode_catalog(&[entry()], &[], Some(&card), None).expect("encodes");
14562 let (entries, views, kept, _) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14563 assert_eq!((entries.len(), views.len()), (1, 0));
14564 assert_eq!(kept, Some(card));
14565 let bytes = encode_catalog(&[entry()], &[], None, None).expect("encodes");
14566 assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14567 }
14568
14569 fn sample_view(name: &str) -> ViewEntry {
14571 ViewEntry {
14572 name: name.to_string(),
14573 sql: "SELECT id FROM items WHERE id > 0".to_string(),
14574 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14575 aliases: vec!["n".to_string()],
14576 columns: vec![Field::new("n", LogicalType::Integer)],
14577 }
14578 }
14579
14580 #[test]
14581 fn a_view_written_into_the_catalog_comes_back_whole() {
14582 let path = path("views");
14583 let mut writer =
14584 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14585 .expect("new file");
14586 writer.append(&sample_ids()).expect("rows");
14587 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14588 let catalog = Catalog::open(&path).expect("reopen");
14589 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14590 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14593 fs::remove_file(&path).expect("clean up");
14594 }
14595
14596 #[test]
14598 fn appending_a_table_carries_the_views_forward() {
14599 let path = path("viewscarry");
14600 let mut writer =
14601 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14602 .expect("new file");
14603 writer.append(&sample_ids()).expect("rows");
14604 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14605 let mut writer =
14606 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14607 .expect("a second table");
14608 writer.append(&sample_ids()).expect("rows");
14609 writer.finish().expect("commit");
14610 let catalog = Catalog::open(&path).expect("reopen");
14611 assert_eq!(catalog.views().count(), 1);
14612 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14613 fs::remove_file(&path).expect("clean up");
14614 }
14615
14616 #[test]
14618 fn restating_the_views_leaves_every_table_where_it_was() {
14619 let path = path("restate");
14620 let mut writer =
14621 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14622 .expect("new file");
14623 writer.append(&sample_ids()).expect("rows");
14624 writer.finish().expect("commit");
14625 let before = fs::metadata(&path).expect("the file is there").len();
14626 Writer::restate(&path, &[sample_view("v"), sample_view("w")], None).expect("two views");
14627 let catalog = Catalog::open(&path).expect("reopen");
14628 assert_eq!(catalog.views().count(), 2);
14629 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14630 let after = fs::metadata(&path).expect("the file is there").len();
14633 assert!(after > before, "a generation was written");
14634 assert!(after - before < before, "the table was not written again");
14635 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14638 assert_eq!(reader.table().rows, 3);
14639 Writer::restate(&path, &[], None).expect("no views at all");
14642 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14643 fs::remove_file(&path).expect("clean up");
14644 }
14645
14646 #[test]
14648 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14649 let bytes = encode_catalog(
14650 &[Entry {
14651 name: "items".to_string(),
14652 fields: vec![Field::required("id", LogicalType::Integer)],
14653 rows: 1,
14654 directory: Page { offset: HEADER, length: 8, hash: 0 },
14655 nonzero: vec![None],
14656 aggregates: vec![None],
14657 distincts: vec![None],
14658 extremes: vec![None],
14659 frequencies: vec![None],
14660 }],
14661 &[sample_view("items")],
14662 None,
14663 None,
14664 )
14665 .expect("it encodes, because encoding does not look");
14666 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14667 assert!(error.to_string().contains("same name"), "{error}");
14668 }
14669
14670 #[test]
14673 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14674 let rows: usize = 300;
14675 let text: Vec<String> =
14676 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14677 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14678 let mut page = vec![6, 2];
14679 page.extend((0..rows.div_ceil(8)).map(|byte| {
14680 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14681 }));
14682 let compressed = string::encode_only(string::Kind::Fsst, &values)
14683 .expect("encoded")
14684 .expect("text this repetitive compresses");
14685 page.extend_from_slice(&compressed);
14686 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14687 let positions = [0_u32, 3, 8, 13, 200, 299];
14688 let some =
14689 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14690 assert_eq!(some.len(), positions.len());
14691 for (at, &row) in positions.iter().enumerate() {
14692 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14693 }
14694 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14695 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14696 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14697 }
14698
14699 #[test]
14702 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14703 let path = path("rows");
14704 let mut writer = Writer::create(
14705 &path,
14706 "items",
14707 vec![
14708 Field::required("id", LogicalType::Integer),
14709 Field::new("text", LogicalType::Varchar),
14710 ],
14711 )
14712 .expect("new file");
14713 let rows = 2_000;
14714 let chunk = Chunk::new(vec![
14715 Vector::from_values(
14716 LogicalType::Integer,
14717 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14718 )
14719 .expect("integers"),
14720 Vector::from_values(
14721 LogicalType::Varchar,
14722 &(0..rows)
14723 .map(|row| {
14724 if row % 7 == 2 {
14725 Value::Null
14726 } else {
14727 Value::Varchar(format!("a comment about order {}", row * 13))
14728 }
14729 })
14730 .collect::<Vec<_>>(),
14731 )
14732 .expect("strings"),
14733 ])
14734 .expect("matching rows");
14735 writer.append(&chunk).expect("one part");
14736 writer.finish().expect("commit");
14737 let reader = Reader::open(&path).expect("reopen from disk");
14738 let positions = [1_u32, 2, 9, 1_000, 1_999];
14739 for whole in [true, false] {
14740 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14741 let all = reader.read(0, &[0, 1]).expect("the whole part");
14742 assert_eq!(some.len(), positions.len());
14743 for column in 0..2 {
14744 for (at, &row) in positions.iter().enumerate() {
14745 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14746 }
14747 }
14748 }
14749 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14750 }
14751
14752 #[test]
14753 fn committed_file_reopens_and_reads_only_requested_columns() {
14754 let path = path("reopen");
14755 let mut writer = Writer::create(
14756 &path,
14757 "items",
14758 vec![
14759 Field::required("id", LogicalType::Integer),
14760 Field::new("text", LogicalType::Varchar),
14761 ],
14762 )
14763 .expect("new file");
14764 writer.append(&sample()).expect("first part");
14765 writer.append(&sample()).expect("second part");
14766 writer.finish().expect("commit");
14767 let reader = Reader::open(&path).expect("reopen from disk");
14768 assert_eq!(reader.table().rows(), 6);
14769 assert_eq!(reader.table().stripes().len(), 1);
14772 assert_eq!(reader.parts(), 2);
14773 assert_eq!(reader.part_rows(0), 3);
14774 assert_eq!(reader.part_rows(1), 3);
14775 let text = reader.read(1, &[1]).expect("only text page");
14776 assert_eq!(text.width(), 1);
14777 assert_eq!(text.value_at(1, 0), Value::Null);
14778 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14779 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14780 assert_eq!(sparse.width(), 1);
14781 assert_eq!(sparse.value_at(1, 0), Value::Null);
14782 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14783 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14784 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14785 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14786 let count = reader.read(0, &[]).expect("no page is needed for count");
14787 assert_eq!(count.len(), 3);
14788 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14789 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14790 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14791 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14792 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14793 assert_eq!(integers.omitted_max, 2);
14794 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14795 assert_eq!(strings.len(), 3);
14796 assert!(strings.contains(&(Value::Null, 2)));
14797 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14798 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14799 fs::remove_file(path).expect("remove scratch file");
14800 }
14801
14802 #[test]
14810 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14811 let path = path("interleaved-runs");
14812 let mut writer =
14813 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14814 .expect("new file");
14815 for morsel in [2_u64, 0, 3, 1] {
14816 let parts = (0..4_u64)
14817 .map(|chunk| {
14818 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14819 let values =
14820 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14821 let column =
14822 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14823 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14824 })
14825 .collect::<Vec<_>>();
14826 writer.append_stripe(parts).expect("a stripe");
14827 }
14828 writer.finish().expect("commit");
14829
14830 let reader = Reader::open(&path).expect("valid directory");
14831 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14832 assert_eq!(reader.table().rows(), 128);
14833 for part in 0..16_usize {
14834 let read = reader.read(part, &[0]).expect("a part back");
14835 for row in 0..8_usize {
14836 let want = i64::try_from(part * 8 + row).expect("small");
14837 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14838 }
14839 }
14840 fs::remove_file(path).expect("remove scratch file");
14841 }
14842
14843 #[test]
14846 fn runs_that_overlap_each_other_are_refused_at_commit() {
14847 let path = path("overlapping-runs");
14848 let mut writer =
14849 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14850 .expect("new file");
14851 let one = |order: (u64, u64)| {
14852 let column =
14853 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14854 (order, Chunk::new(vec![column]).expect("one column"))
14855 };
14856 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14859 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14860 let error = writer.finish().expect_err("the runs overlap");
14861 assert!(error.message().contains("source order"), "{error}");
14862 fs::remove_file(path).expect("remove scratch file");
14863 }
14864
14865 #[test]
14868 fn a_run_longer_than_a_stripe_is_refused() {
14869 let path = path("overlong-run");
14870 let mut writer =
14871 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14872 .expect("new file");
14873 let parts = (0..=STRIPE_PARTS)
14874 .map(|at| {
14875 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14876 .expect("a column");
14877 let chunk = Chunk::new(vec![column]).expect("one column");
14878 ((0, u64::try_from(at).expect("small")), chunk)
14879 })
14880 .collect::<Vec<_>>();
14881 let error = writer.append_stripe(parts).expect_err("one part too many");
14882 assert!(error.message().contains("more parts than it holds"), "{error}");
14883 fs::remove_file(path).expect("remove scratch file");
14884 }
14885
14886 #[test]
14892 fn parts_past_the_stripe_bound_start_a_new_stripe() {
14893 let path = path("stripe-bound");
14894 let mut writer = Writer::create(
14895 &path,
14896 "items",
14897 vec![
14898 Field::required("id", LogicalType::Integer),
14899 Field::new("text", LogicalType::Varchar),
14900 ],
14901 )
14902 .expect("new file");
14903 let parts = STRIPE_PARTS * 2 + 3;
14904 for part in 0..parts {
14905 let id = part as i32;
14906 let chunk = Chunk::new(vec![
14907 Vector::from_values(
14908 LogicalType::Integer,
14909 &[Value::Integer(id), Value::Integer(-id)],
14910 )
14911 .expect("integers"),
14912 Vector::from_values(
14913 LogicalType::Varchar,
14914 &[Value::Varchar(format!("value {part}")), Value::Null],
14915 )
14916 .expect("strings"),
14917 ])
14918 .expect("matching rows");
14919 writer.append(&chunk).expect("one part");
14920 }
14921 writer.finish().expect("commit");
14922
14923 let reader = Reader::open(&path).expect("reopen from disk");
14924 assert_eq!(reader.parts(), parts);
14925 assert_eq!(reader.table().rows(), parts * 2);
14926 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14927 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14928 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14929 assert_eq!(reader.table().stripes()[2].parts(), 3);
14930 for part in (0..parts).rev() {
14933 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14934 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14935 for chunk in [&dense, &sparse] {
14936 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14937 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14938 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14939 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14940 assert_eq!(chunk.value_at(1, 1), Value::Null);
14941 }
14942 }
14943 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14946 assert!(reader.skips(0, &above), "the first stripe stops at 63");
14947 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14948 fs::remove_file(path).expect("remove scratch file");
14949 }
14950
14951 fn scattered(n: i64) -> i64 {
14953 n.wrapping_mul(-7_046_029_254_386_353_131)
14954 }
14955
14956 #[test]
14962 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14963 let path = path("sieve-skip");
14964 let mut writer =
14965 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14966 .expect("new file");
14967 let parts = STRIPE_PARTS + 3;
14968 let per_part = 128;
14972 for part in 0..parts {
14973 let held: Vec<Value> = (0..per_part)
14974 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14975 .collect();
14976 let chunk =
14977 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14978 .expect("one column");
14979 writer.append(&chunk).expect("one part");
14980 }
14981 writer.finish().expect("commit");
14982
14983 let reader = Reader::open(&path).expect("reopen from disk");
14984 let probe = |value: i64| Probe {
14985 column: 0,
14986 op: Op::Equal,
14987 value: Bound::Int(i128::from(scattered(value))),
14988 };
14989 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14990 let tests = [probe(wanted)];
14991 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14992 let home = wanted as usize / per_part;
14993 assert!(kept.contains(&home), "the part holding {wanted} is read");
14994 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14998 }
14999 let absent = [probe((parts * per_part) as i64 + 1)];
15000 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
15001 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
15002 let tests = [probe(0)];
15005 assert!(
15006 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
15007 "the bounds rule out no stripe at all"
15008 );
15009 fs::remove_file(path).expect("remove scratch file");
15010 }
15011
15012 #[test]
15018 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
15019 let path = path("part-range-skip");
15020 let mut writer =
15021 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15022 .expect("new file");
15023 let parts = STRIPE_PARTS + 3;
15024 let per_part = 128;
15025 for part in 0..parts {
15026 let held: Vec<Value> = (0..per_part)
15030 .map(|row| {
15031 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15032 })
15033 .collect();
15034 let chunk =
15035 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15036 .expect("one column");
15037 writer.append(&chunk).expect("one part");
15038 }
15039 writer.finish().expect("commit");
15040
15041 let reader = Reader::open(&path).expect("reopen from disk");
15042 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15043 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
15044 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
15045 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
15047 fs::remove_file(path).expect("remove scratch file");
15048 }
15049
15050 #[test]
15054 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
15055 let path = path("part-range-certain");
15056 let mut writer =
15057 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15058 .expect("new file");
15059 let parts = STRIPE_PARTS + 3;
15060 let per_part = 128;
15061 for part in 0..parts {
15062 let held: Vec<Value> = (0..per_part)
15063 .map(|row| {
15064 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15065 })
15066 .collect();
15067 let chunk =
15068 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15069 .expect("one column");
15070 writer.append(&chunk).expect("one part");
15071 }
15072 writer.finish().expect("commit");
15073
15074 let reader = Reader::open(&path).expect("reopen from disk");
15075 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15076 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
15077 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
15078 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
15081 fs::remove_file(path).expect("remove scratch file");
15082 }
15083
15084 #[test]
15087 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
15088 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
15089 let path = path("part-range-page");
15090 let mut writer =
15091 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15092 .expect("new file");
15093 for part in 0..parts {
15094 let held: Vec<Value> = (0..128)
15095 .map(|row| {
15096 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
15097 })
15098 .collect();
15099 let chunk = Chunk::new(vec![
15100 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
15101 ])
15102 .expect("one column");
15103 writer.append(&chunk).expect("one part");
15104 }
15105 writer.finish().expect("commit");
15106 let reader = Reader::open(&path).expect("reopen from disk");
15107 let bytes = reader.layout().columns[0].part_ranges;
15108 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
15109 fs::remove_file(path).expect("remove scratch file");
15110 }
15111 }
15112
15113 #[test]
15116 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
15117 let long = vec![b'a'; PART_BOUND_BYTES * 2];
15118 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
15119 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
15120 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
15121 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
15122 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
15123 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
15124 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
15125 }
15126
15127 #[test]
15130 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
15131 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
15132 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
15133 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
15134 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
15135 }
15136
15137 #[test]
15149 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
15150 let parts = 4;
15151 let per_part = 1024;
15152 let rows = parts * per_part;
15153 let written = |name: &str, keys: &[i64]| {
15154 let path = path(name);
15155 let fields = vec![Field::required("key", LogicalType::BigInt)];
15156 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
15157 for part in 0..parts {
15158 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
15159 .iter()
15160 .map(|key| Value::BigInt(*key))
15161 .collect();
15162 let chunk = Chunk::new(vec![
15163 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
15164 ])
15165 .expect("one column");
15166 writer.append(&chunk).expect("one part");
15167 }
15168 writer.finish().expect("commit");
15169 path
15170 };
15171 let climbing = |step: &dyn Fn(usize) -> i64| {
15174 let mut key = 0;
15175 (0..rows)
15176 .map(|row| {
15177 key += step(row);
15178 key
15179 })
15180 .collect::<Vec<i64>>()
15181 };
15182 let ascending = climbing(&|row| (row % 3) as i64);
15183 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
15187 let near_path = written("stored-near", &ascending);
15188 let far_path = written("stored-far", &sparse);
15189
15190 let one = Reader::open(&near_path).expect("reopen from disk");
15191 let other = Reader::open(&far_path).expect("reopen from disk");
15192 let near = one.stored(0).expect("the column is stored");
15193 let far = other.stored(0).expect("the column is stored");
15194 assert_eq!(near.len(), parts, "one row per part");
15195 assert_eq!(far.len(), parts);
15196 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
15199 assert_eq!(total(&near), one.layout().columns[0].pages);
15200 assert_eq!(total(&far), other.layout().columns[0].pages);
15201 assert!(
15202 total(&near) * 2 < total(&far),
15203 "the sparse keys cost more, {} against {}",
15204 total(&far),
15205 total(&near)
15206 );
15207 for (at, part) in near.iter().enumerate() {
15209 assert_eq!(part.part, at);
15210 assert_eq!(part.row, at * per_part);
15211 assert_eq!(part.rows, per_part);
15212 let held = &ascending[at * per_part..(at + 1) * per_part];
15213 assert_eq!(part.low, Some(Value::BigInt(held[0])));
15214 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
15215 assert_eq!(part.nulls, Some(0));
15216 }
15217 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
15220 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
15221 assert_ne!(near[0].encoding, far[0].encoding);
15222 fs::remove_file(near_path).expect("remove scratch file");
15223 fs::remove_file(far_path).expect("remove scratch file");
15224 }
15225
15226 #[test]
15236 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
15237 let path = path("sieve-pays");
15238 let fields = vec![
15239 Field::required("spread", LogicalType::BigInt),
15240 Field::required("repeated", LogicalType::BigInt),
15241 ];
15242 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
15243 let parts = 3;
15244 let per_part = 1024;
15245 for part in 0..parts {
15246 let base = (part * per_part) as i64;
15247 let spread: Vec<Value> =
15248 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
15249 let repeated: Vec<Value> =
15250 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
15251 let chunk = Chunk::new(vec![
15252 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
15253 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
15254 ])
15255 .expect("two columns");
15256 writer.append(&chunk).expect("one part");
15257 }
15258 writer.finish().expect("commit");
15259
15260 let reader = Reader::open(&path).expect("reopen from disk");
15261 let layout = reader.layout();
15262 let spread = &layout.columns[0];
15263 let repeated = &layout.columns[1];
15264 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
15265 assert_eq!(
15266 repeated.sieves, 0,
15267 "a column whose filter costs more than its parts keeps none"
15268 );
15269 for column in &layout.columns {
15272 assert!(
15273 column.sieves < column.pages,
15274 "{} spends {} on sieves over {} of data",
15275 column.name,
15276 column.sieves,
15277 column.pages
15278 );
15279 }
15280 let absent = [Probe {
15282 column: 0,
15283 op: Op::Equal,
15284 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
15285 }];
15286 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
15287 fs::remove_file(path).expect("remove scratch file");
15288 }
15289
15290 #[test]
15296 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
15297 let path = path("sieve-damaged");
15298 let mut writer =
15299 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
15300 .expect("new file");
15301 let rows = 128;
15302 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
15303 let chunk =
15304 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15305 .expect("one column");
15306 writer.append(&chunk).expect("one part");
15307 writer.finish().expect("commit");
15308
15309 let page = Reader::open(&path).expect("reopen").table.stripes[0]
15310 .sieves
15311 .get(0)
15312 .expect("a sieve page");
15313 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
15314 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
15315 file.write_all(&[0xff]).expect("damage one byte");
15316 drop(file);
15317
15318 let reader = Reader::open(&path).expect("reopen the damaged file");
15319 let absent =
15320 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
15321 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
15322 assert_eq!(
15323 reader.read(0, &[0]).expect("the rows are untouched").len(),
15324 usize::try_from(rows).expect("a small count")
15325 );
15326 fs::remove_file(path).expect("remove scratch file");
15327 }
15328
15329 #[test]
15335 fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
15336 let path = path("asked-twice");
15337 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15338 let mut writer =
15339 Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
15340 .expect("new file");
15341 for part in 0..parts {
15342 let chunk = Chunk::new(vec![
15343 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15344 .expect("integers"),
15345 ])
15346 .expect("matching rows");
15347 writer.append(&chunk).expect("one part");
15348 }
15349 writer.finish().expect("commit");
15350
15351 let pool = PagePool::new(usize::MAX);
15352 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15353 let a = catalog.table("a").expect("a");
15354 let stripes = a.table().stripes().len();
15355 for part in 0..parts {
15356 for _ in 0..2 {
15357 let chunk = a.read(part, &[0]).expect("a part");
15358 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15359 }
15360 }
15361 assert_eq!(
15362 a.pages.load(Atomic::Relaxed),
15363 stripes,
15364 "a page a stripe, read on the second ask"
15365 );
15366 assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
15367 let column = a.cache.columns[0].lock().expect("the column");
15368 assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
15369 drop(column);
15370 drop((a, catalog));
15371 fs::remove_file(path).expect("remove scratch file");
15372 }
15373
15374 #[test]
15385 fn workers_that_want_the_same_stripe_read_it_once() {
15386 let path = path("single-flight");
15387 let mut writer =
15388 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15389 .expect("new file");
15390 for part in 0..STRIPE_PARTS {
15391 let id = part as i32;
15392 let chunk = Chunk::new(vec![
15393 Vector::from_values(
15394 LogicalType::Integer,
15395 &[Value::Integer(id), Value::Integer(-id)],
15396 )
15397 .expect("integers"),
15398 ])
15399 .expect("matching rows");
15400 writer.append(&chunk).expect("one part");
15401 }
15402 writer.finish().expect("commit");
15403
15404 let reader = Reader::open(&path).expect("reopen from disk");
15405 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
15406 for part in 0..STRIPE_PARTS {
15409 reader.read(part, &[0]).expect("a part");
15410 }
15411 assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
15412 let barrier = std::sync::Barrier::new(8);
15413 std::thread::scope(|scope| {
15414 for worker in 0..8 {
15415 let reader = &reader;
15416 let barrier = &barrier;
15417 scope.spawn(move || {
15418 barrier.wait();
15419 for part in (worker..STRIPE_PARTS).step_by(8) {
15420 let chunk = reader.read(part, &[0]).expect("a whole page read");
15421 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15422 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15423 }
15424 });
15425 }
15426 });
15427 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15428 fs::remove_file(path).expect("remove scratch file");
15429 }
15430
15431 #[test]
15444 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15445 let opened = |label: &str, rows_per_part: i32| {
15446 let path = path(label);
15447 let mut writer =
15448 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15449 .expect("new file");
15450 for part in 0..STRIPE_PARTS * 3 {
15451 let values = (0..rows_per_part)
15455 .map(|row| {
15456 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15457 })
15458 .collect::<Vec<_>>();
15459 let chunk = Chunk::new(vec![
15460 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15461 ])
15462 .expect("matching rows");
15463 writer.append(&chunk).expect("one part");
15464 }
15465 writer.finish().expect("commit");
15466 let reader = Reader::open(&path).expect("reopen from disk");
15467 let size = fs::metadata(&path).expect("the file is there").len();
15468 let out = (reader.reads(), reader.table().stripes().len(), size);
15469 fs::remove_file(path).expect("remove scratch file");
15470 out
15471 };
15472
15473 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15474 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15475 assert_eq!(
15476 thin_stripes, fat_stripes,
15477 "the same stripe count is what makes this a fair ask"
15478 );
15479 assert!(
15480 fat_size > thin_size * 50,
15481 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15482 );
15483
15484 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15485 assert_eq!(thin.pages, 0, "opening read a page");
15486 assert_eq!(fat.pages, 0, "opening read a page");
15487 assert_eq!(thin.indexes, 0, "opening read an index");
15488 assert_eq!(fat.indexes, 0, "opening read an index");
15489 assert!(
15492 fat.opening.bytes < thin.opening.bytes * 2,
15493 "opening the thin file read {} bytes and the fat one read {}",
15494 thin.opening.bytes,
15495 fat.opening.bytes
15496 );
15497 }
15498
15499 #[test]
15507 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15508 let path = path("open-twice");
15509 let mut writer =
15510 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15511 .expect("new file");
15512 for part in 0..STRIPE_PARTS * 3 {
15513 let chunk = Chunk::new(vec![
15514 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15515 .expect("integers"),
15516 ])
15517 .expect("matching rows");
15518 writer.append(&chunk).expect("one part");
15519 }
15520 writer.finish().expect("commit");
15521
15522 let first = Reader::open(&path).expect("open");
15523 for part in 0..first.parts() {
15526 first.read(part, &[0]).expect("a part");
15527 }
15528 assert!(first.reads().indexes > 0, "the scan has to have read something");
15529 let second = Reader::open(&path).expect("open again");
15530
15531 assert_eq!(first.reads().opening, second.reads().opening);
15532 assert_eq!(
15533 second.reads().pages,
15534 0,
15535 "the second open read a page off the back of the first"
15536 );
15537 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15538 fs::remove_file(path).expect("remove scratch file");
15539 }
15540
15541 #[test]
15549 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15550 let path = path("index-cache");
15551 let mut writer =
15552 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15553 .expect("new file");
15554 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15555 for part in 0..parts {
15556 let id = part as i32;
15557 let chunk = Chunk::new(vec![
15558 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15559 ])
15560 .expect("matching rows");
15561 writer.append(&chunk).expect("one part");
15562 }
15563 writer.finish().expect("commit");
15564
15565 let reader = Reader::open(&path).expect("reopen from disk");
15566 let stripes = reader.table().stripes().len();
15567 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15568 for _ in 0..3 {
15571 for part in 0..parts {
15572 let chunk = reader.read(part, &[0]).expect("a part");
15573 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15574 }
15575 }
15576 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15577 assert!(
15578 reader.pages.load(Atomic::Relaxed) > stripes,
15579 "the pages are the ones that get read again, which is what makes the index count mean \
15580 something"
15581 );
15582 fs::remove_file(path).expect("remove scratch file");
15583 }
15584
15585 #[test]
15592 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15593 let path = path("page-pool");
15594 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15595 let fields = || vec![Field::required("id", LogicalType::Integer)];
15596 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15597 for table in ["a", "b"] {
15598 if table == "b" {
15599 writer = writer.next("b".to_string(), fields()).expect("a second table");
15600 }
15601 for part in 0..parts {
15602 let chunk = Chunk::new(vec![
15603 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15604 .expect("integers"),
15605 ])
15606 .expect("matching rows");
15607 writer.append(&chunk).expect("one part");
15608 }
15609 }
15610 writer.finish().expect("commit");
15611
15612 let pool = PagePool::new(usize::MAX);
15613 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15614 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15615 let stripes = a.table().stripes().len();
15616 assert!(
15617 stripes > CACHED_STRIPES_PER_COLUMN * 2,
15618 "the floor has to be smaller than a table"
15619 );
15620 let scan = |reader: &Reader| {
15621 for part in 0..parts {
15622 let chunk = reader.read(part, &[0]).expect("a part");
15623 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15624 }
15625 };
15626 scan(&a);
15629 assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15630 assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15631 scan(&a);
15632 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15633 scan(&a);
15634 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15635 let one = pool.bytes();
15636 assert!(one > 0, "the pool counts what the reader holds");
15637
15638 pool.budget.store(one, Atomic::Relaxed);
15640 scan(&b);
15641 scan(&b);
15642 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15643 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15644 let column = a.cache.columns[0].lock().expect("the column");
15645 let held = column.pages.iter().flatten().count();
15646 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15647 drop(column);
15648
15649 drop((a, b, catalog));
15651 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15652 scan(&c);
15653 scan(&c);
15654 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15655 fs::remove_file(path).expect("remove scratch file");
15656 }
15657
15658 #[test]
15667 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15668 let workers = CACHED_STRIPES_PER_COLUMN + 4;
15669 let path = path("stripe-per-worker");
15670 let mut writer =
15671 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15672 .expect("new file");
15673 for part in 0..STRIPE_PARTS * workers {
15674 let chunk = Chunk::new(vec![
15675 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15676 .expect("integers"),
15677 ])
15678 .expect("matching rows");
15679 writer.append(&chunk).expect("one part");
15680 }
15681 writer.finish().expect("commit");
15682
15683 let read = |told: bool| {
15684 let reader = Reader::open(&path).expect("reopen from disk");
15685 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15686 if told {
15687 reader.keep_stripes(workers);
15688 }
15689 for part in 0..reader.parts() {
15691 reader.read(part, &[0]).expect("a part");
15692 }
15693 let barrier = std::sync::Barrier::new(workers);
15694 std::thread::scope(|scope| {
15695 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15696 let reader = &reader;
15697 let barrier = &barrier;
15698 scope.spawn(move || {
15699 for part in run {
15700 barrier.wait();
15701 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15702 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15703 }
15704 assert!(worker < workers);
15705 });
15706 }
15707 });
15708 reader.pages.load(Atomic::Relaxed)
15709 };
15710
15711 assert_eq!(read(true), workers, "one page read per stripe and no more");
15712 assert!(read(false) > workers, "a cache that small is read again on every part");
15713 fs::remove_file(path).expect("remove scratch file");
15714 }
15715
15716 #[test]
15721 fn a_damaged_index_page_is_an_error() {
15722 let path = path("damaged-index");
15723 let mut writer =
15724 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15725 .expect("new file");
15726 writer.append(&sample_ids()).expect("first part");
15727 writer.append(&sample_ids()).expect("second part");
15728 writer.finish().expect("commit");
15729
15730 let reader = Reader::open(&path).expect("valid directory");
15731 let index = reader.table.stripes[0].index;
15732 let mut byte = [0; 1];
15733 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15734 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15735 file.seek(SeekFrom::Start(index.offset)).expect("index start");
15736 file.write_all(&[!byte[0]]).expect("damage the first part length");
15737 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15738 assert!(error.message().contains("index page section checksum differs"), "{error}");
15739 fs::remove_file(path).expect("remove scratch file");
15740 }
15741
15742 #[test]
15749 fn every_integer_width_round_trips_through_a_page() {
15750 let path = path("integer-widths");
15751 let columns = [
15752 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15753 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15754 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15755 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15756 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15757 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15758 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15759 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15760 ];
15761 let fields = columns
15762 .iter()
15763 .enumerate()
15764 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15765 .collect::<Vec<_>>();
15766 let vectors = columns
15767 .iter()
15768 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15769 .collect::<Vec<_>>();
15770 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15771 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15772 writer.finish().expect("commit");
15773
15774 let reader = Reader::open(&path).expect("reopen from disk");
15775 let wanted = (0..columns.len()).collect::<Vec<_>>();
15776 let read = reader.read(0, &wanted).expect("every column");
15777 assert_eq!(read.len(), 2);
15778 for (at, (ty, values)) in columns.iter().enumerate() {
15780 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15781 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15782 }
15783 fs::remove_file(path).expect("remove scratch file");
15784 }
15785
15786 #[test]
15797 fn every_other_type_the_format_knows_round_trips_through_a_page() {
15798 let path = path("other-types");
15799 let columns = [
15800 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15801 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15802 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15803 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15804 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15805 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15806 (
15807 LogicalType::TimestampTz,
15808 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15809 ),
15810 (
15811 LogicalType::Interval,
15812 vec![
15813 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15814 Value::Interval { months: 13, days: -1, micros: 1 },
15815 ],
15816 ),
15817 (
15818 LogicalType::Blob,
15819 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15820 ),
15821 ];
15822 let fields = columns
15823 .iter()
15824 .enumerate()
15825 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15826 .collect::<Vec<_>>();
15827 let vectors = columns
15828 .iter()
15829 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15830 .collect::<Vec<_>>();
15831 let mut writer = Writer::create(&path, "others", fields).expect("new file");
15832 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15833 writer.finish().expect("commit");
15834
15835 let reader = Reader::open(&path).expect("reopen from disk");
15836 let wanted = (0..columns.len()).collect::<Vec<_>>();
15837 let read = reader.read(0, &wanted).expect("every column");
15838 assert_eq!(read.len(), 2);
15839 for (at, (ty, values)) in columns.iter().enumerate() {
15840 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15841 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15842 }
15843 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15846 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15847
15848 fs::remove_file(path).expect("remove scratch file");
15849 }
15850
15851 #[test]
15857 fn a_nan_survives_being_written_down() {
15858 let path = path("nan");
15859 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15860 .expect("a NaN vector");
15861 let mut writer =
15862 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15863 .expect("new file");
15864 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15865 writer.finish().expect("commit");
15866 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15867 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15868 assert!(back.is_nan(), "a NaN came back as {back}");
15869 fs::remove_file(path).expect("remove scratch file");
15870 }
15871
15872 #[test]
15879 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15880 let path = path("uuid-and-bit");
15881 let uuids = vec![0_i128, i128::MIN, -1];
15882 let mut bits = StringColumn::new();
15883 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15884 bits.push_bytes(value);
15885 }
15886 let expected = bits.clone();
15887 let fields =
15888 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15889 let vectors = vec![
15890 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15891 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15892 ];
15893 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15894 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15895 writer.finish().expect("commit");
15896
15897 let reader = Reader::open(&path).expect("reopen from disk");
15898 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15899 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15900 panic!("a uuid column is the 128 bit lane")
15901 };
15902 assert_eq!(back.as_slice(), uuids.as_slice());
15903 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15904 panic!("a bit column is bytes")
15905 };
15906 for row in 0..expected.len() {
15907 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15908 }
15909 fs::remove_file(path).expect("remove scratch file");
15910 }
15911
15912 #[test]
15915 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15916 let mut rows: Vec<Option<u64>> = Vec::new();
15917 let mut state = 0x2545_f491_4f6c_dd1d_u64;
15918 for index in 0..400_000_u64 {
15919 state ^= state << 13;
15920 state ^= state >> 7;
15921 state ^= state << 17;
15922 let times = 1 + (state % 7) as usize;
15923 let bits = match state % 11 {
15924 0 => None,
15925 1..=3 => Some(state % 16),
15926 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15927 };
15928 rows.extend(std::iter::repeat_n(bits, times));
15929 }
15930 let mut by_row = Candidates::default();
15931 for &bits in &rows {
15932 by_row.add(bits, 1);
15933 }
15934 let mut by_run = Candidates::default();
15935 let mut run = Run::default();
15936 let mut runs = 0_usize;
15937 for &bits in &rows {
15938 if let Some((bits, times)) = run.push(bits) {
15939 by_run.add(bits, times);
15940 runs += 1;
15941 }
15942 }
15943 if let Some((bits, times)) = run.take() {
15944 by_run.add(bits, times);
15945 }
15946 assert!(runs < rows.len() / 2, "the rows came in runs");
15947 assert!(by_row.decrements > 0, "the table filled and turned values away");
15948 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15949 assert_eq!(by_run.nulls, by_row.nulls);
15950 assert_eq!(by_run.decrements, by_row.decrements);
15951 }
15952
15953 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15954 let mut pairs = candidates.pairs().collect::<Vec<_>>();
15955 pairs.sort_unstable();
15956 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15957 pairs
15958 }
15959
15960 #[derive(Default)]
15963 struct MapCandidates {
15964 counts: HashMap<u64, u32>,
15965 nulls: u32,
15966 decrements: u64,
15967 }
15968
15969 impl MapCandidates {
15970 fn add(&mut self, bits: Option<u64>, mut times: u32) {
15971 while times > 0 {
15972 let held = match bits {
15973 Some(bits) => self.counts.get_mut(&bits),
15974 None if self.nulls != 0 => Some(&mut self.nulls),
15975 None => None,
15976 };
15977 if let Some(count) = held {
15978 *count = count.saturating_add(times);
15979 return;
15980 }
15981 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15982 match bits {
15983 Some(bits) => {
15984 self.counts.insert(bits, times);
15985 }
15986 None => self.nulls = times,
15987 }
15988 return;
15989 }
15990 self.counts.retain(|_, count| {
15991 *count -= 1;
15992 *count != 0
15993 });
15994 self.nulls = self.nulls.saturating_sub(1);
15995 self.decrements = self.decrements.saturating_add(1);
15996 times -= 1;
15997 }
15998 }
15999 }
16000
16001 #[test]
16005 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
16006 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
16007 let mut table = Candidates::default();
16008 let mut oracle = MapCandidates::default();
16009 let mut state = seed;
16010 for index in 0..300_000_u64 {
16011 state ^= state << 13;
16012 state ^= state >> 7;
16013 state ^= state << 17;
16014 let bits = match state % 13 {
16015 0 => None,
16016 1..=4 => Some(state % 40),
16017 5 => Some((index % 1000) * 1_000_000),
16018 _ => Some(state),
16019 };
16020 let times = 1 + (state >> 60) as u32 % 3;
16021 table.add(bits, times);
16022 oracle.add(bits, times);
16023 if index % 50_000 == 0 {
16024 let mut expected =
16025 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16026 expected.sort_unstable();
16027 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
16028 }
16029 }
16030 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16031 expected.sort_unstable();
16032 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
16033 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
16034 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
16035 assert!(table.decrements > 0, "seed {seed} never filled the table");
16036 for &(bits, _) in &expected {
16037 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
16038 }
16039 }
16040 }
16041
16042 #[test]
16043 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
16044 let path = path("frequency-ordinals");
16045 let mut writer =
16046 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
16047 .expect("new file");
16048 let mut values = Vec::new();
16049 for leader in 0..10_i64 {
16050 values.extend(std::iter::repeat_n(leader, 100));
16051 }
16052 values.extend(1_000_i64..41_000);
16053 for part in values.chunks(1_024) {
16054 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
16055 .expect("big integers");
16056 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
16057 }
16058 writer.finish().expect("commit");
16059
16060 let reader = Reader::open(&path).expect("reopen from disk");
16061 let occurrences =
16062 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
16063 assert!(occurrences.omitted_max < 100);
16064 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
16065 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
16066 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
16067 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
16068 assert_eq!(
16069 &occurrences.anchor_indices[..1_000]
16070 .iter()
16071 .map(|&entry| occurrences.anchors[entry as usize].clone())
16072 .collect::<Vec<_>>(),
16073 &(0_i64..10)
16074 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
16075 .collect::<Vec<_>>()
16076 );
16077 fs::remove_file(path).expect("remove scratch file");
16078 }
16079
16080 #[test]
16081 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
16082 let path = path("frequency-bits");
16087 let mut writer = Writer::create(
16088 &path,
16089 "items",
16090 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
16091 )
16092 .expect("new file");
16093 let mut rows = Vec::new();
16094 let mut leaders = Vec::new();
16095 for leader in 0..10_u64 {
16096 let count = 300 - leader * 10;
16097 let (unsigned, signed) = if leader == 0 {
16098 (Value::Null, Value::Null)
16099 } else {
16100 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
16101 };
16102 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
16103 leaders.push(((unsigned, count), (signed, count)));
16104 }
16105 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
16106 for part in rows.chunks(1_024) {
16107 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
16108 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
16109 let chunk = Chunk::new(vec![
16110 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
16111 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
16112 ])
16113 .expect("matching columns");
16114 writer.append(&chunk).expect("rows");
16115 }
16116 writer.finish().expect("commit");
16117
16118 let reader = Reader::open(&path).expect("reopen from disk");
16119 for column in 0..2 {
16120 let prefix =
16121 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16122 let wanted = leaders
16123 .iter()
16124 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
16125 .cloned()
16126 .collect::<Vec<_>>();
16127 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
16128 assert!(prefix.omitted_max < 210, "column {column}");
16129 assert_eq!(
16130 reader.distinct_values(column).expect("valid metadata"),
16131 Some(9 + 40_000),
16132 "column {column}"
16133 );
16134 }
16135 fs::remove_file(path).expect("remove scratch file");
16136 }
16137
16138 #[test]
16139 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
16140 let path = path("frequency-tally");
16146 let types = [
16147 LogicalType::TinyInt,
16148 LogicalType::UInteger,
16149 LogicalType::Date,
16150 LogicalType::Timestamp,
16151 ];
16152 let value = |ty: &LogicalType, at: i64| match ty {
16153 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
16154 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
16155 LogicalType::Date => Value::Date(19_000 - at as i32),
16156 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
16157 };
16158 let fields = types
16159 .iter()
16160 .enumerate()
16161 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
16162 .collect::<Vec<_>>();
16163 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16164 let mut rows = Vec::new();
16165 for at in 0..250_i64 {
16166 for _ in 0..=(at % 37) {
16167 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
16168 }
16169 }
16170 for part in rows.chunks(1_000) {
16171 let columns = types
16172 .iter()
16173 .map(|ty| {
16174 let values = part
16175 .iter()
16176 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
16177 .collect::<Vec<_>>();
16178 Vector::from_values(ty.clone(), &values).expect("a column")
16179 })
16180 .collect();
16181 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
16182 }
16183 writer.finish().expect("commit");
16184
16185 let reader = Reader::open(&path).expect("reopen from disk");
16186 for (column, ty) in types.iter().enumerate() {
16187 let mut counts = HashMap::<Option<i64>, u64>::new();
16188 for row in &rows {
16189 *counts.entry(*row).or_default() += 1;
16190 }
16191 let wanted = counts
16192 .into_iter()
16193 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
16194 .collect::<Vec<_>>();
16195 let prefix =
16196 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16197 assert_eq!(prefix.entries.len(), 2, "column {column}");
16198 assert!(prefix.omitted_max > 0, "column {column}");
16199 for (value, count) in &prefix.entries {
16200 let held =
16201 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
16202 assert_eq!(held, Some(count), "column {column} value {value:?}");
16203 }
16204 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
16205 assert_eq!(
16206 reader.distinct_values(column).expect("valid metadata"),
16207 Some(wanted.len() as u64 - 1),
16208 "column {column}"
16209 );
16210 }
16211 fs::remove_file(path).expect("remove scratch file");
16212 }
16213
16214 #[test]
16215 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
16216 let edge = FREQUENCY_CANDIDATES as i64;
16221 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
16222 for with_null in [false, true] {
16223 let path = path("distinct-edge");
16224 let mut writer =
16225 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
16226 .expect("new file");
16227 let mut values = Vec::new();
16228 for round in 0..2 {
16229 for value in 0..distinct {
16230 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
16231 values.extend(std::iter::repeat_n(
16232 Value::BigInt(value * 7_919 % distinct),
16233 repeat,
16234 ));
16235 if with_null && value % 1_000 == 0 {
16236 values.push(Value::Null);
16237 }
16238 }
16239 }
16240 if with_null {
16241 values.push(Value::Null);
16242 }
16243 for part in values.chunks(1_024) {
16244 let chunk = Chunk::new(vec![
16245 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
16246 ])
16247 .expect("one column");
16248 writer.append(&chunk).expect("rows");
16249 }
16250 writer.finish().expect("commit");
16251 let reader = Reader::open(&path).expect("reopen from disk");
16252 assert_eq!(
16253 reader.distinct_values(0).expect("valid metadata"),
16254 Some(distinct as u64),
16255 "{distinct} values, null {with_null}"
16256 );
16257 fs::remove_file(path).expect("remove scratch file");
16258 }
16259 }
16260 }
16261
16262 #[test]
16263 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
16264 let path = path("quick-nonzero");
16265 let mut writer = Writer::create(
16266 &path,
16267 "items",
16268 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
16269 )
16270 .expect("create");
16271 for ids in [
16272 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
16273 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
16274 ] {
16275 let labels = vec![Value::Varchar("same".into()); ids.len()];
16276 writer
16277 .append(
16278 &Chunk::new(vec![
16279 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
16280 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
16281 ])
16282 .expect("chunk"),
16283 )
16284 .expect("append");
16285 }
16286 writer.finish().expect("finish");
16287 let catalog = Catalog::open(&path).expect("catalog");
16288 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
16289 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
16290 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
16291 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
16292 let prefix = catalog
16293 .table("items")
16294 .expect("reader")
16295 .frequency_prefix(1)
16296 .expect("valid metadata")
16297 .expect("partial frequencies");
16298 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
16299 assert_eq!(prefix.omitted_max, 1);
16300 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
16301 assert_eq!(
16302 catalog.integer_extremes("items", 1).expect("extremes"),
16303 Some(IntegerExtremes::Values { low: 0, high: 7 })
16304 );
16305 assert_eq!(
16306 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
16307 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16308 );
16309 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
16310 let mut legacy = catalog.clone();
16311 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
16312 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
16313 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
16314 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
16315 Writer::certify_counts(&path).expect("recertify");
16316 assert_eq!(
16317 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
16318 Some(2)
16319 );
16320 assert_eq!(
16321 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
16322 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16323 );
16324 assert_eq!(
16325 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
16326 Some(3)
16327 );
16328 assert_eq!(
16329 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
16330 Some(IntegerExtremes::Values { low: 0, high: 7 })
16331 );
16332 assert_eq!(
16333 Catalog::open(&path)
16334 .expect("reopen")
16335 .exact_numeric_frequencies("items", 1)
16336 .expect("frequencies"),
16337 None
16338 );
16339 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
16340 fs::remove_file(path).expect("remove scratch file");
16341 }
16342
16343 #[test]
16344 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
16345 let path = path("pair-frequencies");
16346 let mut pairs = Vec::new();
16347 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
16348 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
16349 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
16350 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
16351 let mut writer = Writer::create(
16352 &path,
16353 "items",
16354 vec![
16355 Field::required("id", LogicalType::BigInt),
16356 Field::required("phrase", LogicalType::Varchar),
16357 ],
16358 )
16359 .expect("new file");
16360 for part in pairs.chunks(1_024) {
16361 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
16362 let phrases =
16363 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
16364 writer
16365 .append(
16366 &Chunk::new(vec![
16367 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
16368 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
16369 ])
16370 .expect("matching columns"),
16371 )
16372 .expect("rows");
16373 }
16374 writer.finish().expect("commit");
16375
16376 let reader = Reader::open(&path).expect("reopen from disk");
16377 assert!(
16378 reader.table.pair_frequencies.is_empty(),
16379 "no query-specific pair result is stored"
16380 );
16381 fs::remove_file(path).expect("remove scratch file");
16382 }
16383
16384 #[test]
16385 fn legacy_group_answers_are_ignored() {
16386 let path = path("legacy-group-answers");
16387 let mut writer = Writer::create(
16388 &path,
16389 "items",
16390 vec![
16391 Field::required("id", LogicalType::BigInt),
16392 Field::required("text", LogicalType::Varchar),
16393 ],
16394 )
16395 .expect("new file");
16396 writer
16397 .append(
16398 &Chunk::new(vec![
16399 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
16400 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
16401 .expect("text"),
16402 ])
16403 .expect("row"),
16404 )
16405 .expect("append");
16406 writer.finish().expect("commit");
16407 let mut reader = Reader::open(&path).expect("reopen");
16408 let table = Arc::make_mut(&mut reader.table);
16409 table.pair_frequencies.push(PairFrequencySummary {
16410 first: 0,
16411 second: 1,
16412 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
16413 omitted_max: 0,
16414 });
16415 table.host_groups = Some(host::HostSummary {
16416 column: 1,
16417 omitted_max: 0,
16418 entries: vec![host::HostEntry {
16419 host: "fake.test".into(),
16420 count: 999,
16421 bytes_sum: 999,
16422 minimum: "x".into(),
16423 }],
16424 });
16425 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16426 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16427 fs::remove_file(path).expect("remove scratch file");
16428 }
16429
16430 #[test]
16436 fn a_file_from_another_format_says_which_format_it_is() {
16437 let older = path("older-format");
16438 let mut writer =
16439 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16440 .expect("new file");
16441 let chunk = Chunk::new(vec![
16442 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16443 .expect("integers"),
16444 ])
16445 .expect("chunk");
16446 writer.append(&chunk).expect("page written");
16447 writer.finish().expect("commit");
16448
16449 let unreadable =
16453 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16454 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16455 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16456 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16457 drop(file);
16458 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16459 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16460 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16461
16462 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16463 file.seek(SeekFrom::Start(0)).expect("the magic is first");
16464 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16465 drop(file);
16466 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16467 assert!(complaint.contains("magic"), "{complaint}");
16468 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16469 fs::remove_file(older).expect("remove scratch file");
16470 }
16471
16472 #[test]
16473 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16474 let unfinished = path("unfinished");
16475 let mut writer =
16476 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16477 .expect("new file");
16478 let chunk = Chunk::new(vec![
16479 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16480 .expect("integers"),
16481 ])
16482 .expect("chunk");
16483 writer.append(&chunk).expect("page written");
16484 drop(writer);
16485 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16486 fs::remove_file(unfinished).expect("remove scratch file");
16487
16488 let damaged = path("damaged");
16489 let mut writer =
16490 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16491 .expect("new file");
16492 writer.append(&chunk).expect("page written");
16493 writer.finish().expect("commit");
16494 let reader = Reader::open(&damaged).expect("valid directory");
16495 let mut file =
16496 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16497 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16498 file.write_all(&[255]).expect("damage one byte");
16499 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16500 fs::remove_file(damaged).expect("remove scratch file");
16501 }
16502
16503 #[test]
16504 fn damaged_lazy_dictionary_payload_is_an_error() {
16505 let path = path("damaged-dictionary");
16506 let mut writer = Writer::create(
16507 &path,
16508 "items",
16509 vec![
16510 Field::required("id", LogicalType::Integer),
16511 Field::new("text", LogicalType::Varchar),
16512 ],
16513 )
16514 .expect("new file");
16515 writer.append(&sample()).expect("stripe written");
16516 writer.finish().expect("commit");
16517
16518 let reader = Reader::open(&path).expect("valid directory");
16519 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16520 let mut header = [0; DICTIONARY_HEADER];
16523 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16524 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16527 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16528 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16529 let bits = (width & !DICTIONARY_FLAGS) as usize;
16530 let mut start = [0; 8];
16531 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16532 read_at(&reader.file, at, &mut start).expect("the first block's start");
16533 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16534 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16535 file.write_all(&[255]).expect("damage dictionary payload");
16536
16537 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16538 let error =
16539 chunk.validate_external().expect_err("payload corruption must reach the caller");
16540 assert!(error.message().contains("payload checksum differs"), "{error}");
16541 fs::remove_file(path).expect("remove scratch file");
16542 }
16543
16544 #[test]
16554 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16555 let path = path("dictionary-decide");
16556 let rows = 20_000;
16557 let unique =
16559 |row: usize| format!("{row:09} a value that appears exactly once in the table");
16560 let repeated = |row: usize| unique(row / 40);
16562 let mut writer = Writer::create(
16563 &path,
16564 "items",
16565 vec![
16566 Field::required("unique", LogicalType::Varchar),
16567 Field::required("repeated", LogicalType::Varchar),
16568 ],
16569 )
16570 .expect("new file");
16571 for part in (0..rows).step_by(1_000) {
16572 let span = part..(part + 1_000).min(rows);
16573 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16574 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16575 writer
16576 .append(
16577 &Chunk::new(vec![
16578 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16579 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16580 ])
16581 .expect("two columns"),
16582 )
16583 .expect("a part");
16584 }
16585 writer.finish().expect("commit");
16586
16587 let reader = Reader::open(&path).expect("reopen from disk");
16588 assert!(
16589 reader.table.dictionaries[0].is_none(),
16590 "a column with no repeats has nothing to say twice"
16591 );
16592 assert!(
16593 reader.table.dictionaries[1].is_some(),
16594 "a column whose values come round again keeps its dictionary"
16595 );
16596 let mut first = 0;
16597 for part in 0..reader.parts() {
16598 let chunk = reader.read(part, &[0, 1]).expect("a part");
16599 for row in 0..chunk.len() {
16600 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16601 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16602 }
16603 first += chunk.len();
16604 }
16605 assert_eq!(first, rows, "every row was read back");
16606 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16607 let size = fs::metadata(&path).expect("the file is there").len() as usize;
16608 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16609 fs::remove_file(path).expect("remove scratch file");
16610 }
16611
16612 #[test]
16625 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16626 let path = path("dictionary-blocks");
16627 let value = |row: usize| {
16628 let row = row.saturating_sub(8_000);
16629 format!("{row:07} a value long enough to be worth a payload block")
16630 };
16631 let parts = 40;
16632 let per_part = 1000;
16633 let mut writer =
16634 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16635 .expect("new file");
16636 for part in 0..parts {
16637 let values = (0..per_part)
16638 .map(|row| Value::Varchar(value(part * per_part + row)))
16639 .collect::<Vec<_>>();
16640 let chunk = Chunk::new(vec![
16641 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16642 ])
16643 .expect("matching rows");
16644 writer.append(&chunk).expect("a part");
16645 }
16646 writer.finish().expect("commit");
16647
16648 let reader = Reader::open(&path).expect("reopen from disk");
16649 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16650 assert!(
16651 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16652 "the dictionary has to be several blocks for this to be testing anything"
16653 );
16654 for part in [0, parts - 1] {
16655 let chunk = reader.read(part, &[0]).expect("a part");
16656 chunk.validate_external().expect("every payload block checks out");
16657 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16658 }
16659
16660 let mut header = [0; DICTIONARY_HEADER];
16662 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16663 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16664 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16665 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16666 let bits = (width & !DICTIONARY_FLAGS) as usize;
16667 let mut place = [0; 16];
16668 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16669 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16670 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16671 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16672 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16673 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16674 file.write_all(&[255]).expect("damage the last payload block");
16675 let reader = Reader::open(&path).expect("the directory and the index are untouched");
16676 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16677 let error = chunk.validate_external().expect_err("the damage must reach the caller");
16678 assert!(error.message().contains("payload checksum differs"), "{error}");
16679 fs::remove_file(path).expect("remove scratch file");
16680 }
16681
16682 #[test]
16696 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16697 let path = path("dictionary-offsets");
16698 let value = |row: usize| {
16699 let row = row % 5_000;
16700 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16701 };
16702 let rows = 6_000;
16703 let mut writer =
16704 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16705 .expect("new file");
16706 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16707 for part in values.chunks(1_000) {
16708 let chunk =
16709 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16710 .expect("matching rows");
16711 writer.append(&chunk).expect("a part");
16712 }
16713 writer.finish().expect("commit");
16714
16715 let reader = Reader::open(&path).expect("reopen from disk");
16716 assert!(
16717 rows > TEXT_PAYLOAD_VALUES * 4,
16718 "the dictionary has to be several blocks for this to be testing anything"
16719 );
16720 for part in 0..rows / 1_000 {
16721 let chunk = reader.read(part, &[0]).expect("a part");
16722 for row in 0..1_000 {
16723 let row = part * 1_000 + row;
16724 assert_eq!(
16725 chunk.value_at(row % 1_000, 0),
16726 Value::Varchar(value(row)),
16727 "value {row}"
16728 );
16729 }
16730 }
16731 for _ in 0..2 {
16734 for part in 0..rows / 1_000 {
16735 let chunk = reader.read(part, &[0]).expect("a part");
16736 let mut lens = vec![0_i64; 1_000];
16737 let column = chunk.column(0).expect("one column");
16738 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16739 for (row, &len) in lens.iter().enumerate() {
16740 let row = part * 1_000 + row;
16741 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16742 }
16743 }
16744 }
16745 fs::remove_file(path).expect("remove scratch file");
16746 }
16747
16748 #[test]
16750 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16751 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16752 ends.extend([3, 3, 10]);
16753 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16754 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16755 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16756 let long = [5, 70_005, 70_006];
16758 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16759 assert_eq!(lens, [5, 70_000, 1]);
16760 let mut read = Vec::new();
16761 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16762 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16763 ends.push(9);
16764 assert!(lengths_of(&ends).is_none());
16765 }
16766
16767 #[test]
16779 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16780 let path = path("dictionary-once");
16781 let parts = 8;
16782 let per_part = 500;
16783 let value =
16784 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16785 let mut writer =
16786 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16787 .expect("new file");
16788 for part in 0..parts {
16789 let values = (0..per_part)
16790 .map(|row| Value::Varchar(value(part * per_part + row)))
16791 .collect::<Vec<_>>();
16792 let chunk = Chunk::new(vec![
16793 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16794 ])
16795 .expect("matching rows");
16796 writer.append(&chunk).expect("a part");
16797 }
16798 writer.finish().expect("commit");
16799
16800 let reader = Reader::open(&path).expect("reopen from disk");
16801 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16802 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16803
16804 let workers = 16;
16805 let gate = std::sync::Barrier::new(workers);
16806 std::thread::scope(|scope| {
16807 for worker in 0..workers {
16808 let reader = reader.clone();
16809 let gate = &gate;
16810 scope.spawn(move || {
16811 gate.wait();
16812 let chunk = reader.read(worker % parts, &[0]).expect("a part");
16813 assert_eq!(
16814 chunk.value_at(0, 0),
16815 Value::Varchar(value((worker % parts) * per_part))
16816 );
16817 });
16818 }
16819 });
16820
16821 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16822 fs::remove_file(path).expect("remove scratch file");
16823 }
16824
16825 #[test]
16830 fn a_damaged_sorted_order_is_an_error() {
16831 let path = path("damaged-order");
16832 let mut writer = Writer::create(
16833 &path,
16834 "items",
16835 vec![
16836 Field::required("id", LogicalType::Integer),
16837 Field::new("text", LogicalType::Varchar),
16838 ],
16839 )
16840 .expect("new file");
16841 writer.append(&sample()).expect("stripe written");
16842 writer.finish().expect("commit");
16843
16844 let reader = Reader::open(&path).expect("valid directory");
16845 let page = reader.table.dictionaries[1].expect("string dictionary page");
16846 let mut header = [0; DICTIONARY_HEADER];
16847 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16848 let index_len = dictionary_index_len(&header);
16849 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16850 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16851 file.write_all(&[255]).expect("damage the order");
16852
16853 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16854 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16855 assert!(error.message().contains("rank checksum differs"), "{error}");
16856 fs::remove_file(path).expect("remove scratch file");
16857 }
16858
16859 #[test]
16863 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16864 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16867 let path = path("dictionary-order");
16868 let mut writer =
16869 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16870 .expect("new file");
16871 writer
16872 .append(
16873 &Chunk::new(vec![
16874 Vector::from_values(
16875 LogicalType::Varchar,
16876 &spellings.map(|text| Value::Varchar(text.into())),
16877 )
16878 .expect("strings"),
16879 ])
16880 .expect("one column"),
16881 )
16882 .expect("stripe written");
16883 writer.finish().expect("commit");
16884
16885 let reader = Reader::open(&path).expect("valid directory");
16886 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16887 let count = dictionary.ranks().expect("a v10 file stores one");
16888 assert_eq!(count, spellings.len(), "every distinct value has a rank");
16889 let order = (0..count)
16890 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16891 .collect::<Vec<_>>();
16892 let mut seen = order.clone();
16893 seen.sort_unstable();
16894 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16895
16896 let ranked = order
16897 .iter()
16898 .map(|&code| {
16899 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16900 })
16901 .collect::<Vec<_>>();
16902 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16903 expected.sort();
16904 assert_eq!(ranked, expected, "rank order is value order");
16905
16906 for (rank, value) in expected.iter().enumerate() {
16909 assert_eq!(
16910 dictionary.compare_rank(rank, value).expect("compare"),
16911 Ordering::Equal,
16912 "rank {rank} is its own value"
16913 );
16914 if rank > 0 {
16915 assert_eq!(
16916 dictionary.compare_rank(rank - 1, value).expect("compare"),
16917 Ordering::Less,
16918 "rank {rank} follows the one before it"
16919 );
16920 }
16921 }
16922 fs::remove_file(path).expect("remove scratch file");
16923 }
16924
16925 #[test]
16932 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16933 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16934 let path = path("dictionaries-at-once");
16935 let fields = (0..sizes.len())
16936 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16937 .collect::<Vec<_>>();
16938 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16939 let rows = 10_000_usize;
16940 for start in (0..rows).step_by(1_024) {
16941 let columns = sizes
16942 .iter()
16943 .enumerate()
16944 .map(|(column, &size)| {
16945 let values = (start..(start + 1_024).min(rows))
16946 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16947 .collect::<Vec<_>>();
16948 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16949 })
16950 .collect::<Vec<_>>();
16951 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16952 }
16953 writer.finish().expect("commit");
16954
16955 let reader = Reader::open(&path).expect("valid directory");
16956 for (column, &size) in sizes.iter().enumerate() {
16957 let dictionary =
16958 reader.dictionary(column).expect("read").expect("a string column has one");
16959 let count = dictionary.ranks().expect("a v10 file stores one");
16960 assert_eq!(count, size, "column {column} has its own distinct count");
16961 let ranked = (0..count)
16962 .map(|rank| {
16963 let code = dictionary.code_at_rank(rank).expect("a code");
16964 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16965 })
16966 .collect::<Vec<_>>();
16967 let expected = (0..size)
16968 .map(|value| format!("c{column}-{value:05}").into_bytes())
16969 .collect::<Vec<_>>();
16970 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16971 }
16972 fs::remove_file(path).expect("remove scratch file");
16973 }
16974
16975 #[test]
16983 fn a_large_dictionary_ranks_in_value_order() {
16984 let path = path("dictionary-large-rank");
16985 let value = |row: u64| {
16986 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16987 match row % 3 {
16988 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16989 1 => format!("{mixed}"),
16990 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16991 }
16992 };
16993 let distinct = 70_000;
16994 let parts = 4 * distinct / 1000;
16995 let per_part = 1000;
16996 let mut writer =
16997 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16998 .expect("new file");
16999 for part in 0..parts {
17000 let values = (0..per_part)
17001 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
17002 .collect::<Vec<_>>();
17003 let chunk = Chunk::new(vec![
17004 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
17005 ])
17006 .expect("matching rows");
17007 writer.append(&chunk).expect("a part");
17008 }
17009 writer.finish().expect("commit");
17010
17011 let reader = Reader::open(&path).expect("reopen from disk");
17012 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17013 let count = dictionary.ranks().expect("a ranked dictionary");
17014 assert_eq!(count, distinct as usize, "every distinct value has a rank");
17015 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
17016 let ranked = (0..count)
17017 .map(|rank| {
17018 let code = dictionary.code_at_rank(rank).expect("a code");
17019 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
17020 })
17021 .collect::<Vec<_>>();
17022 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
17023 expected.sort();
17024 assert_eq!(ranked, expected, "rank order is value order");
17025 fs::remove_file(path).expect("remove scratch file");
17026 }
17027
17028 #[test]
17041 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
17042 let path = path("windowed-directory");
17043 let fields = vec![
17044 Field::required("id", LogicalType::BigInt),
17045 Field::required("word", LogicalType::Varchar),
17046 Field::new("score", LogicalType::Double),
17047 ];
17048 let mut writer = Writer::create(&path, "items", fields).expect("new file");
17049 for part in 0..70_i64 {
17050 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
17051 let words = (0..100)
17052 .map(|row| Value::Varchar(format!("word {}", row % 13)))
17053 .collect::<Vec<_>>();
17054 let scores = (0..100)
17055 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
17056 .collect::<Vec<_>>();
17057 let chunk = Chunk::new(vec![
17058 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
17059 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
17060 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
17061 ])
17062 .expect("three columns");
17063 writer.append(&chunk).expect("a part");
17064 }
17065 writer.finish().expect("commit");
17066
17067 let catalog = Catalog::open(&path).expect("reopen");
17068 let entry = catalog.entries.first().expect("one table").directory;
17069 let (offset, length) = (entry.offset, entry.length as usize);
17070 let mut bytes = vec![0; length];
17071 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
17072 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
17073 let whole = decode_directory(&bytes, catalog.size).expect("whole");
17074 assert!(whole.stripes.len() > 1, "the table should span stripes");
17075 for size in [1, 7, 33, 4_096] {
17076 let mut cursor = Cursor::over(&catalog.file, offset, length);
17077 cursor.window.as_mut().expect("a window").size = size;
17078 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
17079 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
17080 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
17081 let mut stored = 0;
17082 for (column, (left, held)) in
17083 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
17084 {
17085 match (left, held) {
17086 (None, None) => {}
17087 (
17088 Some(super::Frequencies::Stored { span, values, entries }),
17089 Some(super::Frequencies::Held(summary)),
17090 ) => {
17091 let mut one = vec![0; span.length as usize];
17092 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
17093 let read = decode_summary(
17094 &mut Cursor::new(&one),
17095 &whole.fields[column],
17096 whole.rows,
17097 *values,
17098 )
17099 .expect("a valid synopsis")
17100 .expect("one is there");
17101 assert_eq!(*entries, read.entries.len());
17102 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
17103 stored += 1;
17104 }
17105 other => panic!("column {column} came back as {other:?}"),
17106 }
17107 }
17108 assert!(stored >= 2, "only {stored} synopses were left in the file");
17109 }
17110 let reader = catalog.table("items").expect("the table");
17111 assert!(reader.frequency_heads[1].get().is_none());
17112 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
17113 let first = reader.frequency_heads[1].get().expect("decoded synopsis");
17114 let clone = reader.clone();
17115 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
17116 assert!(Arc::ptr_eq(first, clone.frequency_heads[1].get().expect("same synopsis")));
17117 fs::remove_file(path).expect("remove scratch file");
17118 }
17119
17120 #[test]
17121 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
17122 let path = path("file-checksum");
17123 let bytes = (0..200_000_u32)
17124 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
17125 .collect::<Vec<_>>();
17126 fs::write(&path, &bytes).expect("scratch file");
17127 let file = File::open(&path).expect("open");
17128 for (offset, length) in [
17129 (0, 0),
17130 (3, 1),
17131 (5, 31),
17132 (0, 32),
17133 (9, 33),
17134 (1, 65_536),
17135 (7, 65_567),
17136 (0, 200_000),
17137 (11, 131_101),
17138 ] {
17139 let whole = checksum(&bytes[offset..offset + length]);
17140 assert_eq!(
17141 file_checksum(&file, offset as u64, length).expect("read"),
17142 whole,
17143 "{offset} {length}"
17144 );
17145 }
17146 fs::remove_file(path).expect("remove scratch file");
17147 }
17148
17149 #[test]
17150 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
17151 let path = path("synopsis-keeps-no-block");
17152 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
17153 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
17154 for _ in 0..3 {
17155 values.extend((0..3_000).step_by(5).map(spelled));
17156 }
17157 let mut writer =
17158 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17159 .expect("new file");
17160 for part in values.chunks(1_024) {
17161 writer
17162 .append(
17163 &Chunk::new(vec![
17164 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17165 ])
17166 .expect("one column"),
17167 )
17168 .expect("a part");
17169 }
17170 writer.finish().expect("commit");
17171
17172 let reader = Reader::open(&path).expect("reopen from disk");
17173 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17174 let resting = dictionary.footprint();
17175 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17176 assert_eq!(prefix.entries.len(), 512);
17177 for (value, count) in &prefix.entries {
17178 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
17179 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
17180 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
17181 }
17182 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
17183 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17184 assert_eq!(again.entries, prefix.entries);
17185 fs::remove_file(path).expect("remove scratch file");
17186 }
17187
17188 #[test]
17195 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
17196 let path = path("character-lengths");
17197 let spellings = (0..2_500)
17198 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
17199 .collect::<Vec<_>>();
17200 let mut writer =
17201 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17202 .expect("new file");
17203 for part in spellings.chunks(1_024) {
17204 writer
17205 .append(
17206 &Chunk::new(vec![
17207 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17208 ])
17209 .expect("one column"),
17210 )
17211 .expect("a part");
17212 }
17213 writer.finish().expect("commit");
17214
17215 let reader = Reader::open(&path).expect("reopen from disk");
17216 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17217 let resting = dictionary.footprint();
17218 let mut lens = Vec::new();
17219 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
17220 let counted = dictionary.footprint() - resting;
17221 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17222 assert!(
17223 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17224 "counting kept {counted} bytes, more than a count a value"
17225 );
17226 let expected = (0..dictionary.len())
17227 .map(|code| {
17228 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
17229 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
17230 .expect("small")
17231 })
17232 .collect::<Vec<_>>();
17233 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
17234 let mut again = Vec::new();
17235 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
17236 assert_eq!(again, lens, "the kept counts answer the second time");
17237 fs::remove_file(path).expect("remove scratch file");
17238 }
17239
17240 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
17242 let path = path(label);
17243 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
17244 let mut writer =
17245 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17246 .expect("new file");
17247 for part in values.chunks(1_024) {
17248 writer
17249 .append(
17250 &Chunk::new(vec![
17251 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17252 ])
17253 .expect("one column"),
17254 )
17255 .expect("a part");
17256 }
17257 writer.finish().expect("commit");
17258 let reader = Reader::open(&path).expect("reopen from disk");
17259 (path, reader)
17260 }
17261
17262 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
17268 let codes = (0..len)
17269 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
17270 .collect::<Vec<_>>();
17271 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
17272 (codes, valid)
17273 }
17274
17275 #[test]
17282 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
17283 let spellings = (0..2_500)
17284 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
17285 .collect::<Vec<_>>();
17286 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
17287 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17288 let (codes, valid) = scattered_rows(spellings.len());
17289 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
17290 .expect("every code is inside")
17291 .with_validity(Validity::from_run(&valid));
17292
17293 let resting = dictionary.footprint();
17294 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
17295 .expect("length reads");
17296 let counted = dictionary.footprint() - resting;
17297 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17298 assert!(
17299 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17300 "length over a vector with nulls kept {counted} bytes, more than a count a value"
17301 );
17302 let expected = (0..rows.len())
17303 .map(|row| match valid[row] {
17304 true => Value::BigInt(
17305 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
17306 ),
17307 false => Value::Null,
17308 })
17309 .collect::<Vec<_>>();
17310 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
17311 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
17312 fs::remove_file(path).expect("remove scratch file");
17313 }
17314
17315 #[test]
17325 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
17326 let spellings = (0..2_500)
17327 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
17328 .collect::<Vec<_>>();
17329 let (path, reader) = stored_spellings("string-kernels", &spellings);
17330 let page = reader.table.dictionaries[0].expect("a string column has one");
17331 let starved =
17332 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
17333 .expect("a dictionary opens whatever it may keep");
17334 let starved = Arc::new(starved);
17335 let (codes, valid) = scattered_rows(spellings.len());
17336 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
17337 .expect("every code is inside")
17338 .with_validity(Validity::from_run(&valid));
17339 let expected = |each: &dyn Fn(&str) -> String| {
17340 (0..rows.len())
17341 .map(|row| match valid[row] {
17342 true => Value::Varchar(each(&spellings[codes[row] as usize])),
17343 false => Value::Null,
17344 })
17345 .collect::<Vec<_>>()
17346 };
17347 let answers =
17348 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
17349
17350 let resting = starved.footprint();
17353 let ends = spellings.len() * size_of::<u32>();
17354 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
17355 .expect("lower reads");
17356 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
17357 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
17358
17359 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
17360 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
17361 let cut =
17362 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
17363 .expect("substring reads");
17364 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
17365 assert_eq!(answers(&cut), expected(&cut_of), "substring");
17366 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
17367
17368 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17371 .expect("upper reads");
17372 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
17373 let payload = spellings.iter().map(String::len).sum::<usize>();
17374 assert!(
17375 starved.footprint() >= resting + payload,
17376 "a visit that has dropped a column's worth of blocks keeps what it reads"
17377 );
17378 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17379 .expect("upper reads kept blocks");
17380 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
17381 fs::remove_file(path).expect("remove scratch file");
17382 }
17383
17384 #[test]
17394 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
17395 let path = path("dictionary-sweep");
17396 let spellings = (0..2_500)
17399 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17400 .collect::<Vec<_>>();
17401 let mut writer =
17402 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17403 .expect("new file");
17404 for part in spellings.chunks(1_024) {
17407 writer
17408 .append(
17409 &Chunk::new(vec![
17410 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17411 ])
17412 .expect("one column"),
17413 )
17414 .expect("stripe written");
17415 }
17416 writer.finish().expect("commit");
17417
17418 let reader = Reader::open(&path).expect("valid directory");
17419 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17420 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17421 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17422 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17423 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17424 }
17425
17426 let resting = dictionary.footprint();
17427 let sweep = || {
17428 let mut swept: Vec<Vec<u8>> = Vec::new();
17429 let mut at = 0;
17430 let mut calls = 0;
17431 while at < dictionary.len() {
17432 let stopped = dictionary
17433 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17434 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17435 swept.push(text.to_vec());
17436 Ok(())
17437 })
17438 .expect("a sweep reads");
17439 assert!(stopped > at, "a sweep moves");
17440 at = stopped;
17441 calls += 1;
17442 }
17443 assert_eq!(calls, 3, "a sweep hands over one block at a time");
17444 swept
17445 };
17446 let swept = sweep();
17447 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17448 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17449 let after = dictionary.footprint();
17450 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17451
17452 let read = (0..dictionary.len())
17453 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17454 .collect::<Vec<_>>();
17455 assert_eq!(swept, read, "a sweep answers what a point read answers");
17456 let grown = dictionary.footprint() - after;
17460 assert!(
17461 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17462 "a point read of a kept block decodes nothing, and {grown} bytes grew"
17463 );
17464 fs::remove_file(path).expect("remove scratch file");
17465 }
17466
17467 #[test]
17468 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17469 let path = path("narrow-substring-signature");
17470 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17471 let mut grams = Vec::new();
17472 for text in blocks {
17473 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17474 for gram in text.windows(4) {
17475 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17476 bits[bit / 8] |= 1 << (bit % 8);
17477 }
17478 }
17479 grams.extend(bits);
17480 }
17481 fs::write(&path, &grams).expect("scratch file");
17482 let file = File::open(&path).expect("open scratch file");
17483 let signatures = NativeGrams {
17484 start: 0,
17485 length: grams.len(),
17486 width: NARROW_GRAM_BYTES,
17487 hash: checksum(&grams),
17488 verdicts: Mutex::new(Vec::new()),
17489 };
17490 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17491 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17492 assert!(signatures.footprint() > 0, "a verdict is remembered");
17493 let again = signatures.verdicts(&file, b"google").expect("remembered");
17494 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17495
17496 let damaged = NativeGrams {
17497 hash: signatures.hash ^ 1,
17498 verdicts: Mutex::new(Vec::new()),
17499 ..signatures
17500 };
17501 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17502 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17503 fs::remove_file(path).expect("remove scratch file");
17504 }
17505
17506 #[test]
17507 fn a_damaged_substring_signature_is_checked_only_when_used() {
17508 let path = path("damaged-substring-signature");
17509 let mut writer =
17510 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17511 .expect("new file");
17512 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17513 writer
17514 .append(
17515 &Chunk::new(vec![
17516 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17517 ])
17518 .expect("one column"),
17519 )
17520 .expect("stripe written");
17521 writer.finish().expect("commit");
17522
17523 let reader = Reader::open(&path).expect("valid directory");
17524 let page = reader.table.dictionaries[0].expect("string dictionary page");
17525 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17526 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17527 .expect("last signature byte");
17528 file.write_all(&[255]).expect("damage signature");
17529 let reader = Reader::open(&path).expect("the directory is still valid");
17530 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17531 let error = dictionary
17532 .text_block_might_contain(0, b"goog")
17533 .expect_err("a used signature checks its own checksum");
17534 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17535 fs::remove_file(path).expect("remove scratch file");
17536 }
17537
17538 #[test]
17549 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17550 let path = path("dictionary-sweep-short-run");
17551 let spellings = (0..2_800)
17552 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17553 .collect::<Vec<_>>();
17554 let mut writer =
17555 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17556 .expect("new file");
17557 for part in spellings.chunks(1_024) {
17558 writer
17559 .append(
17560 &Chunk::new(vec![
17561 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17562 ])
17563 .expect("one column"),
17564 )
17565 .expect("stripe written");
17566 }
17567 writer.finish().expect("commit");
17568
17569 let reader = Reader::open(&path).expect("valid directory");
17570 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17571 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17572 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17573 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17574 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17575
17576 let mut swept: Vec<Vec<u8>> = Vec::new();
17577 let mut at = 0;
17578 while at < dictionary.len() {
17579 let stopped = dictionary
17580 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17581 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17582 swept.push(text.to_vec());
17583 Ok(())
17584 })
17585 .expect("a sweep reads");
17586 assert!(stopped > at, "a sweep moves");
17587 at = stopped;
17588 }
17589 let read = (0..dictionary.len())
17590 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17591 .collect::<Vec<_>>();
17592 assert_eq!(swept, read, "a sweep answers what a point read answers");
17593 fs::remove_file(path).expect("remove scratch file");
17594 }
17595
17596 #[test]
17605 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17606 let path = path("dictionary-unpacked-ends");
17607 let spellings = (0..2_800)
17608 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17609 .collect::<Vec<_>>();
17610 let mut writer =
17611 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17612 .expect("new file");
17613 for part in spellings.chunks(1_024) {
17614 writer
17615 .append(
17616 &Chunk::new(vec![
17617 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17618 ])
17619 .expect("one column"),
17620 )
17621 .expect("stripe written");
17622 }
17623 writer.finish().expect("commit");
17624
17625 let reader = Reader::open(&path).expect("valid directory");
17626 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17627 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17628 let wanted = (0..spellings.len())
17629 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17630 .collect::<Vec<_>>();
17631
17632 let pass = |what: &str| {
17633 for (index, value) in wanted.iter().enumerate() {
17634 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17635 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17636 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17637 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17638 }
17639 };
17640 pass("the first pass");
17641 pass("the second pass");
17642
17643 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17647 let mut whole = vec![0i64; wanted.len()];
17648 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17649 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17650 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17651 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17652 let mut through = vec![0i64; codes.len()];
17653 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17654 for (row, &code) in codes.iter().enumerate() {
17655 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17656 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17657 assert_eq!(through[row], one as i64, "row {row} a row at a time");
17658 }
17659
17660 let fresh = Reader::open(&path).expect("valid directory");
17663 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17664 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17665 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17666 let mut short = vec![0i64; few.len()];
17667 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17668 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17669 assert_eq!(short, expected, "the packed ends answer what the table answers");
17670 fs::remove_file(path).expect("remove scratch file");
17671 }
17672
17673 #[test]
17688 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17689 let spellings = (0..3_000)
17690 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17691 .collect::<Vec<_>>();
17692 let mut read = Vec::new();
17693 for layout in ["outside", "inside", "behind"] {
17694 let mut dictionary = GlobalDictionary::new();
17695 for text in &spellings {
17696 dictionary.code(text).expect("a code for every spelling");
17697 }
17698 dictionary.finish_blocks().expect("the last block encodes");
17699 let order = dictionary.ranked(None).expect("a sorted order");
17700 let laid = |from: u64| {
17702 let mut at = from;
17703 dictionary
17704 .blocks
17705 .iter()
17706 .map(|block| {
17707 let place =
17708 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17709 at += block.len() as u64;
17710 place
17711 })
17712 .collect::<Vec<_>>()
17713 };
17714 let payload = dictionary.blocks.concat();
17715 let scattered = layout != "behind";
17716 let (bytes, encoded, offset, length) = if layout == "outside" {
17717 let mut bytes = vec![0; HEADER as usize];
17718 bytes.extend_from_slice(&payload);
17719 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17720 .expect("an encoding");
17721 let offset = bytes.len() as u64;
17722 bytes.extend_from_slice(&encoded.index);
17723 bytes.extend_from_slice(&encoded.ranks);
17724 bytes.extend_from_slice(&encoded.grams);
17725 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17726 (bytes, encoded, offset, length)
17727 } else {
17728 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17731 .expect("an encoding");
17732 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17733 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17734 .expect("an encoding");
17735 let mut bytes = encoded.index.clone();
17736 bytes.extend_from_slice(&encoded.ranks);
17737 bytes.extend_from_slice(&encoded.grams);
17738 bytes.extend_from_slice(&payload);
17739 let length = bytes.len();
17740 (bytes, encoded, 0, length)
17741 };
17742 let path = path(&format!("blocks-{layout}"));
17743 fs::write(&path, &bytes).expect("the dictionary is written on its own");
17744 let file = Arc::new(File::open(&path).expect("it opens again"));
17745 let page = Page {
17746 offset,
17747 length: u32::try_from(length).expect("a test dictionary is small"),
17748 hash: checksum(&encoded.index),
17749 };
17750 let opened =
17751 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17752 .expect("a dictionary laid out either way opens");
17753 let mut swept: Vec<Vec<u8>> = Vec::new();
17754 let mut at = 0;
17755 while at < opened.len() {
17756 at = opened
17757 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17758 swept.push(text.to_vec());
17759 Ok(())
17760 })
17761 .expect("a sweep reads");
17762 }
17763 fs::remove_file(&path).expect("clean up");
17764 read.push(swept);
17765 }
17766 let wanted =
17767 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17768 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17769 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17770 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17771 }
17772
17773 #[test]
17781 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17782 let path = path("dictionary-budget");
17783 let spellings = (0..2_500)
17784 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17785 .collect::<Vec<_>>();
17786 let mut writer =
17787 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17788 .expect("new file");
17789 for part in spellings.chunks(1_024) {
17790 writer
17791 .append(
17792 &Chunk::new(vec![
17793 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17794 ])
17795 .expect("one column"),
17796 )
17797 .expect("stripe written");
17798 }
17799 writer.finish().expect("commit");
17800
17801 let reader = Reader::open(&path).expect("valid directory");
17802 let page = reader.table.dictionaries[0].expect("a string column has one");
17803 let file = Arc::clone(&reader.file);
17804 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17805 .expect("a dictionary opens whatever it may keep");
17806
17807 let resting = starved.footprint();
17808 let mut swept: Vec<Vec<u8>> = Vec::new();
17809 let mut at = 0;
17810 while at < starved.len() {
17811 at = starved
17812 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17813 swept.push(text.to_vec());
17814 Ok(())
17815 })
17816 .expect("a sweep reads");
17817 }
17818 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17819 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17820
17821 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17822 let read = (0..generous.len())
17823 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17824 .collect::<Vec<_>>();
17825 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17826 fs::remove_file(path).expect("remove scratch file");
17827 }
17828
17829 #[test]
17832 fn a_part_is_checked_once_per_open_reader() {
17833 let path = path("checked-once");
17834 let mut writer = Writer::create(
17835 &path,
17836 "items",
17837 vec![
17838 Field::required("id", LogicalType::Integer),
17839 Field::new("text", LogicalType::Varchar),
17840 ],
17841 )
17842 .expect("new file");
17843 writer.append(&sample()).expect("stripe written");
17844 writer.finish().expect("commit");
17845
17846 let reader = Reader::open(&path).expect("valid directory");
17847 let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17848 assert!(reader.is_verified(0), "the part is remembered as checked");
17849 let page = reader.table.stripes[0].pages[0];
17850 let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17851 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17852 file.write_all(&[0xa5]).expect("damage page");
17853 if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17854 assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17855 }
17856 let fresh = Reader::open(&path).expect("valid directory");
17857 let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17858 assert!(error.message().contains("column page checksum differs"), "{error}");
17859 assert_eq!(first.len(), 2);
17860 fs::remove_file(path).expect("remove scratch file");
17861 }
17862
17863 #[test]
17864 fn damaged_membership_cannot_skip_a_string_page() {
17865 let path = path("damaged-membership");
17866 let mut writer = Writer::create(
17867 &path,
17868 "items",
17869 vec![
17870 Field::required("id", LogicalType::Integer),
17871 Field::new("text", LogicalType::Varchar),
17872 ],
17873 )
17874 .expect("new file");
17875 writer.append(&sample()).expect("stripe written");
17876 writer.finish().expect("commit");
17877
17878 let reader = Reader::open(&path).expect("valid directory");
17879 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17880 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17881 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17882 file.write_all(&[255]).expect("damage membership");
17883 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17884 assert!(error.message().contains("membership page checksum differs"), "{error}");
17885 fs::remove_file(path).expect("remove scratch file");
17886 }
17887
17888 #[test]
17889 fn membership_delta_stream_is_sorted_exact_and_bounded() {
17890 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17891 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17892 let encoded = encode_membership(&unique);
17893 assert_eq!(
17894 decode_membership(&encoded).expect("valid membership"),
17895 [4, 9, 72, 900, u32::MAX]
17896 );
17897 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17900 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17901 assert_eq!(
17902 decode_membership(&encode_membership(&merged)).expect("valid membership"),
17903 unique
17904 );
17905 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17906 assert!(
17907 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17908 "a value past u32 is invalid"
17909 );
17910 }
17911
17912 #[test]
17913 fn a_global_dictionary_may_be_larger_than_one_column_page() {
17914 let dictionary = Page {
17915 offset: HEADER,
17916 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17917 hash: 0,
17918 };
17919 let table = Table {
17920 name: "items".to_owned(),
17921 fields: vec![Field::new("text", LogicalType::Varchar)],
17922 stripes: Vec::new(),
17923 rows: 0,
17924 dictionaries: vec![Some(dictionary)],
17925 dictionary_payloads: Vec::new(),
17926 demoted: Vec::new(),
17927 distincts: vec![None],
17928 frequencies: vec![None],
17929 ordinal_bounds: Vec::new(),
17930 pair_frequencies: Vec::new(),
17931 frequency_texts: Vec::new(),
17932 host_groups: None,
17933 clustering: None,
17934 constraints: Constraints::default(),
17935 generation: 1,
17936 sections: Vec::new(),
17937 };
17938 let directory = encode_directory(&table).expect("directory");
17939 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17940
17941 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17942 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17943 }
17944
17945 #[test]
17946 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17947 let path = path("constant-codes");
17948 let mut writer =
17949 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17950 .expect("new file");
17951 let empty = vec![Value::Varchar(String::new()); 1024];
17952 for _ in 0..4 {
17953 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17954 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17955 }
17956 writer.finish().expect("commit");
17957
17958 let reader = Reader::open(&path).expect("valid directory");
17959 let pages = reader.layout().columns.first().expect("one column").pages;
17960 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17964 let read = reader.read(3, &[0]).expect("the last part back");
17965 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17966 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17967 fs::remove_file(path).expect("remove scratch file");
17968 }
17969
17970 #[test]
17971 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17972 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17975 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17976 assert!(format!("{error}").contains("not of its type"), "{error}");
17977 let low = integer::encode(&[i64::MIN]).expect("a chunk");
17978 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17979 let zero = integer::encode(&[0]).expect("a chunk");
17980 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17981 }
17982
17983 #[test]
17984 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17985 let mut state: u32 = 0x9e37_79b9;
17989 let spread: Vec<u32> = (0..1024)
17990 .map(|_| {
17991 state ^= state << 13;
17992 state ^= state >> 17;
17993 state ^= state << 5;
17994 state
17995 })
17996 .collect();
17997 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17998 let near: Vec<u32> = (0..1024).collect();
17999 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
18000 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
18001 }
18002
18003 #[test]
18009 fn two_writes_of_the_same_rows_give_the_same_bytes() {
18010 fn written(path: &PathBuf) {
18011 let fields = (0..40)
18012 .map(|column| {
18013 let ty =
18014 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
18015 Field::new(format!("c{column}"), ty)
18016 })
18017 .collect::<Vec<_>>();
18018 let mut writer = Writer::create(path, "wide", fields).expect("new file");
18019 for part in 0..70_u64 {
18020 let columns = (0..40)
18021 .map(|column| {
18022 let values = (0..64_u64)
18023 .map(|row| {
18024 let seed = part.wrapping_mul(31).wrapping_add(row);
18025 if column % 4 == 0 {
18026 Value::Varchar(format!("v{}", seed % 17))
18027 } else {
18028 Value::BigInt(i64::try_from(seed % 97).expect("small"))
18029 }
18030 })
18031 .collect::<Vec<_>>();
18032 let ty = if column % 4 == 0 {
18033 LogicalType::Varchar
18034 } else {
18035 LogicalType::BigInt
18036 };
18037 Vector::from_values(ty, &values).expect("a column")
18038 })
18039 .collect::<Vec<_>>();
18040 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
18041 }
18042 writer.finish().expect("commit");
18043 }
18044
18045 let first = path("repeatable-one");
18046 let second = path("repeatable-two");
18047 written(&first);
18048 written(&second);
18049 let left = fs::read(&first).expect("the first file");
18050 let right = fs::read(&second).expect("the second file");
18051 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
18052 assert!(left == right, "two writes of the same rows differ in their bytes");
18053
18054 let reader = Reader::open(&first).expect("valid directory");
18057 assert_eq!(reader.table().rows(), 70 * 64);
18058 let read = reader.read(0, &[0, 1]).expect("the first part back");
18059 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
18060 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
18061 fs::remove_file(first).expect("remove scratch file");
18062 fs::remove_file(second).expect("remove scratch file");
18063 }
18064
18065 fn three_tables(path: &PathBuf) {
18067 let writer = Writer::create(
18068 path,
18069 "region",
18070 vec![
18071 Field::new("r_key", LogicalType::Integer),
18072 Field::new("r_name", LogicalType::Varchar),
18073 ],
18074 )
18075 .expect("new file");
18076 let mut writer = writer;
18077 writer
18078 .append(
18079 &Chunk::new(vec![
18080 Vector::from_values(
18081 LogicalType::Integer,
18082 &[Value::Integer(0), Value::Integer(1)],
18083 )
18084 .expect("keys"),
18085 Vector::from_values(
18086 LogicalType::Varchar,
18087 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
18088 )
18089 .expect("names"),
18090 ])
18091 .expect("two columns"),
18092 )
18093 .expect("a part");
18094 let mut writer = writer
18095 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
18096 .expect("a second table");
18097 writer
18098 .append(
18099 &Chunk::new(vec![
18100 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
18101 ])
18102 .expect("one column"),
18103 )
18104 .expect("a part");
18105 let mut writer =
18106 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
18107 for part in 0..70_i64 {
18108 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
18109 writer
18110 .append(
18111 &Chunk::new(vec![
18112 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
18113 ])
18114 .expect("one column"),
18115 )
18116 .expect("a part");
18117 }
18118 writer.finish().expect("commit");
18119 }
18120
18121 #[test]
18122 fn three_tables_in_one_file_read_back_by_name() {
18123 let file = path("three-tables");
18124 three_tables(&file);
18125 let catalog = Catalog::open(&file).expect("a committed catalog");
18126 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
18127
18128 let region = catalog.table("region").expect("the first table");
18129 assert_eq!(region.table().rows(), 2);
18130 assert_eq!(
18131 region.read(0, &[1]).expect("names").value_at(1, 0),
18132 Value::Varchar("ASIA".to_owned())
18133 );
18134
18135 let wide = catalog.table("wide").expect("the third table");
18136 assert_eq!(wide.table().rows(), 70 * 64);
18137 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
18138
18139 let empty = catalog.table("empty").expect("the second table");
18142 assert_eq!(empty.table().rows(), 1);
18143 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
18144
18145 fs::remove_file(file).expect("remove scratch file");
18146 }
18147
18148 #[test]
18149 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
18150 let file = path("three-tables-missing");
18151 three_tables(&file);
18152 let catalog = Catalog::open(&file).expect("a committed catalog");
18153 let error = catalog.table("nation").expect_err("no such table");
18154 assert!(error.message().contains("nation"), "{}", error.message());
18155 fs::remove_file(file).expect("remove scratch file");
18156 }
18157
18158 #[test]
18159 fn a_file_of_three_tables_will_not_open_as_one() {
18160 let file = path("three-tables-unnamed");
18161 three_tables(&file);
18162 let error = Reader::open(&file).expect_err("more than one table");
18163 assert!(error.message().contains("more than one table"), "{}", error.message());
18164 fs::remove_file(file).expect("remove scratch file");
18165 }
18166
18167 #[test]
18169 fn decimals_of_every_storage_width_round_trip() {
18170 let file = path("decimals");
18171 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
18172 let fields = widths
18173 .iter()
18174 .enumerate()
18175 .map(|(index, (width, scale))| {
18176 Field::new(
18177 format!("d{index}"),
18178 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18179 )
18180 })
18181 .collect::<Vec<_>>();
18182 let mut writer = Writer::create(&file, "money", fields).expect("new file");
18183 let rows: [i128; 3] = [-1234, 0, 999];
18184 let columns = widths
18185 .iter()
18186 .map(|(width, scale)| {
18187 let values = rows
18188 .iter()
18189 .map(|unscaled| Value::Decimal {
18190 unscaled: *unscaled,
18191 width: *width,
18192 scale: *scale,
18193 })
18194 .collect::<Vec<_>>();
18195 Vector::from_values(
18196 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18197 &values,
18198 )
18199 .expect("a decimal column")
18200 })
18201 .collect::<Vec<_>>();
18202 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
18203 writer.finish().expect("commit");
18204
18205 let reader = Reader::open(&file).expect("a committed file");
18206 for (index, (width, scale)) in widths.iter().enumerate() {
18207 assert_eq!(
18208 reader.table().fields()[index].ty,
18209 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18210 "column {index} came back as another type"
18211 );
18212 let column = reader.read(0, &[index]).expect("the column");
18213 for (row, unscaled) in rows.iter().enumerate() {
18214 assert_eq!(
18215 column.value_at(row, 0),
18216 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
18217 "column {index} row {row}"
18218 );
18219 }
18220 }
18221 fs::remove_file(file).expect("remove scratch file");
18222 }
18223
18224 #[test]
18225 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
18226 let file = path("two-of-a-name");
18227 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
18228 .expect("new file");
18229 let error = writer
18230 .next("t", vec![Field::new("a", LogicalType::BigInt)])
18231 .expect_err("the same name twice");
18232 assert!(error.message().contains("same name"), "{}", error.message());
18233 fs::remove_file(file).expect("remove scratch file");
18234 }
18235
18236 #[test]
18237 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
18238 let file = path("integer-tally");
18239 let mut writer =
18240 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
18241 .expect("new file");
18242 let mut values = vec![Value::SmallInt(0); 1024];
18243 values[7] = Value::SmallInt(3);
18244 values[99] = Value::SmallInt(-2);
18245 values[1001] = Value::SmallInt(3);
18246 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
18247 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
18248 values[0] = Value::Null;
18249 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
18250 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
18251 writer.finish().expect("commit");
18252
18253 let reader = Reader::open(&file).expect("read file");
18254 assert_eq!(
18255 reader.integer_tally(0, 0).expect("valid part"),
18256 Some(vec![(-2, 1), (0, 1021), (3, 2)])
18257 );
18258 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
18259 let catalog = Catalog::open(&file).expect("catalog");
18260 assert_eq!(
18261 catalog.integer_tally("events", 0).expect("nullable column"),
18262 Some(vec![(-2, 2), (0, 2041), (3, 4)])
18263 );
18264 fs::remove_file(file).expect("remove scratch file");
18265 }
18266
18267 #[test]
18268 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
18269 let file = path("catalog-integer-tally");
18270 let mut writer = Writer::create(
18271 &file,
18272 "events",
18273 vec![
18274 Field::new("noise", LogicalType::SmallInt),
18275 Field::new("source", LogicalType::SmallInt),
18276 ],
18277 )
18278 .expect("new file");
18279 let noise = vec![Value::SmallInt(9); 1024];
18280 let mut source = vec![Value::SmallInt(0); 1024];
18281 source[7] = Value::SmallInt(3);
18282 source[99] = Value::SmallInt(-2);
18283 let chunk = Chunk::new(vec![
18284 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
18285 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
18286 ])
18287 .expect("two columns");
18288 writer.append(&chunk).expect("append");
18289 writer.finish().expect("commit");
18290
18291 let catalog = Catalog::open(&file).expect("catalog");
18292 assert_eq!(
18293 catalog.integer_tally("events", 1).expect("selected column"),
18294 Some(vec![(-2, 1), (0, 1022), (3, 1)])
18295 );
18296 assert_eq!(
18297 catalog.integer_tally("events", 0).expect("other column"),
18298 Some(vec![(9, 1024)])
18299 );
18300 fs::remove_file(file).expect("remove scratch file");
18301 }
18302
18303 #[test]
18304 fn opening_the_catalog_reads_no_table_directory() {
18305 let file = path("catalog-only");
18306 three_tables(&file);
18307 let catalog = Catalog::open(&file).expect("a committed catalog");
18308 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
18311 assert_eq!(catalog.names().len(), 3);
18312 fs::remove_file(file).expect("remove scratch file");
18313 }
18314
18315 #[test]
18326 fn the_checksum_answers_what_it_has_always_answered() {
18327 let bytes: Vec<u8> =
18328 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
18329 for (length, expected) in [
18330 (0, 0xef46_db37_51d8_e999),
18331 (1, 0xa96c_7f0c_e858_bbb7),
18332 (3, 0x56e6_9576_32a4_87f9),
18333 (4, 0xc60d_15b1_e3ff_8f04),
18334 (5, 0x8088_1585_8624_dd4e),
18335 (7, 0xafbe_fc3d_6c6f_9a8e),
18336 (8, 0x3da5_c7aa_2696_83e0),
18337 (9, 0x465e_c429_b13c_3892),
18338 (15, 0xdee8_9d8a_065a_6233),
18339 (16, 0x1330_489a_7767_9c80),
18340 (31, 0x3391_303d_485e_846e),
18341 (32, 0x40b7_aff7_5d45_bbc8),
18342 (33, 0x4997_cae4_951c_17a5),
18343 (39, 0x5807_28fd_5c14_5739),
18344 (40, 0xf95c_f6f5_c08a_3d3b),
18345 (63, 0x2944_b4da_fc69_b206),
18346 (64, 0xbb76_f6ef_19bd_5a1b),
18347 (65, 0x814e_0c65_4a9f_d640),
18348 (127, 0x00de_aab1_31cf_f89b),
18349 (1000, 0x9e33_00c1_cde3_c58d),
18350 ] {
18351 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
18352 }
18353 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
18354 }
18355 #[test]
18362 fn a_declared_order_comes_back_out_of_the_file() {
18363 let path = path("clustered");
18364 let shipped = vec![
18365 Field::new("key", LogicalType::BigInt),
18366 Field::new("line", LogicalType::Integer),
18367 Field::new("shipdate", LogicalType::Date),
18368 ];
18369 let plain = vec![Field::new("a", LogicalType::Integer)];
18370 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
18371
18372 let mut writer = Writer::create(&path, "lineitem", shipped)
18373 .expect("new file")
18374 .declare(stage_zero.clone())
18375 .expect("the columns are the table's");
18376 let column = |ty: LogicalType, values: &[Value]| {
18377 Vector::from_values(ty, values).expect("the values match the type")
18378 };
18379 writer
18380 .append(
18381 &Chunk::new(vec![
18382 column(
18383 LogicalType::BigInt,
18384 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
18385 ),
18386 column(
18387 LogicalType::Integer,
18388 &[
18389 Value::Integer(1),
18390 Value::Integer(1),
18391 Value::Integer(1),
18392 Value::Integer(1),
18393 ],
18394 ),
18395 column(
18396 LogicalType::Date,
18397 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
18398 ),
18399 ])
18400 .expect("three columns"),
18401 )
18402 .expect("four rows");
18403 let mut writer = writer.next("nation", plain).expect("a second table");
18404 writer
18405 .append(
18406 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
18407 .expect("one column"),
18408 )
18409 .expect("one row");
18410 writer.finish().expect("commit");
18411
18412 let catalog = Catalog::open(&path).expect("reopen");
18413 let lineitem = catalog.table("lineitem").expect("the clustered table");
18414 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
18415 let nation = catalog.table("nation").expect("the plain table");
18416 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18417
18418 assert_eq!(lineitem.table().rows(), 4);
18421 assert_eq!(nation.table().rows(), 1);
18422 fs::remove_file(&path).ok();
18423 }
18424
18425 #[test]
18427 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18428 let path = path("clustered-bad");
18429 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18430 .expect("new file");
18431 let four =
18432 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18433 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18434 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18435 fs::remove_file(&path).ok();
18436 }
18437
18438 #[test]
18444 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18445 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18446 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18447 .collect::<Vec<_>>();
18448 let filled = || {
18449 let mut dictionary = GlobalDictionary::new();
18450 for value in &values {
18451 dictionary.code(value).expect("a code for every value");
18452 }
18453 dictionary.settle().expect("a shape");
18454 dictionary
18455 };
18456 let mut in_place = filled();
18457 in_place.finish_blocks().expect("every block encodes");
18458
18459 let mut handed = filled();
18460 let out = handed.hand_out(3);
18461 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18462 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18463 for job in out.iter().rev() {
18464 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18465 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18466 }
18467 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18468 handed.finish_blocks().expect("the last block encodes");
18469
18470 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18471 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18472 }
18473
18474 #[test]
18476 fn a_block_given_back_twice_is_refused() {
18477 let mut dictionary = GlobalDictionary::new();
18478 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18479 dictionary.code(&format!("value {at}")).expect("a code");
18480 }
18481 dictionary.settle().expect("a shape");
18482 let out = dictionary.hand_out(0);
18483 let last = out.last().expect("blocks went out");
18484 let at = last.place().1;
18485 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18486 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18487 }
18488
18489 #[test]
18495 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18496 let mut values = vec![String::new(), "http://".to_owned()];
18497 for host in 0..7 {
18498 for path in 0..30 {
18499 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18500 values.push(format!("http://example{host}.test/page/{path:04}"));
18501 }
18502 }
18503 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18504
18505 let mut dictionary = GlobalDictionary::new();
18506 for value in &values {
18507 dictionary.code(value).expect("a code for every value");
18508 }
18509 dictionary.finish_blocks().expect("the last block encodes");
18510 let ranked = dictionary.ranked(None).expect("a sorted order");
18511 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18512
18513 let spellings = dictionary_values(&dictionary);
18514 let seen = ranked
18515 .iter()
18516 .map(|&(_, code)| {
18517 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18518 })
18519 .collect::<Vec<_>>();
18520 let mut wanted = values.clone();
18521 wanted.sort_unstable();
18522 assert_eq!(seen, wanted, "the order is the order the bytes give");
18523
18524 for &(carried, code) in &ranked {
18525 let value = &spellings[code as usize];
18526 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18527 }
18528 }
18529
18530 #[test]
18535 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18536 let entry =
18537 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18538 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18539 .map(|code| entry(code, u64::from(code % 7) + 1))
18540 .collect::<Vec<_>>();
18541 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18542
18543 let mut sorted = all.clone();
18544 sorted.sort_unstable_by(|left, right| {
18545 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18546 });
18547 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18548 sorted.truncate(FREQUENCY_ENTRIES);
18549
18550 let mut picked = all.clone();
18551 let omitted = keep_most_frequent(&mut picked);
18552 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18553 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18554 assert!(
18555 picked
18556 .iter()
18557 .zip(&sorted)
18558 .all(|(one, two)| one.value == two.value && one.count == two.count),
18559 "the same entries in the same order"
18560 );
18561
18562 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18563 let omitted = keep_most_frequent(&mut short);
18564 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18565 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18566 }
18567
18568 #[test]
18570 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18571 let empty = GlobalDictionary::new();
18572 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18573
18574 let mut dictionary = GlobalDictionary::new();
18575 for value in ["pear", "apple", "", "apples", "app"] {
18576 dictionary.code(value).expect("a code for every value");
18577 }
18578 dictionary.finish_blocks().expect("the one block encodes");
18579 let spellings = dictionary_values(&dictionary);
18580 let seen = dictionary
18581 .ranked(None)
18582 .expect("a sorted order")
18583 .iter()
18584 .map(|&(_, code)| spellings[code as usize].clone())
18585 .collect::<Vec<_>>();
18586 let wanted: Vec<Vec<u8>> =
18587 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18588 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18589 }
18590
18591 #[test]
18594 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18595 let profile = LoadProfile::begin("demoted");
18596 let mut dictionary = GlobalDictionary::new();
18597 for value in 0..50_000 {
18598 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18599 }
18600 let (_, grown) = dictionary.recharge(Some(&profile));
18601 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18602
18603 dictionary.demote();
18604 let (before, after) = dictionary.recharge(Some(&profile));
18605 assert_eq!(before, grown);
18606 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18609 assert_eq!(profile.held(), after, "the profile was told about the drop");
18610 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18611
18612 dictionary.demote();
18613 assert_eq!(
18614 dictionary.recharge(Some(&profile)),
18615 (after, after),
18616 "demoting twice is a no-op"
18617 );
18618 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18619 }
18620}