1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU32, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, Mapped, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod anchor;
58mod distinct;
59pub mod grams;
60pub mod graph;
61pub mod host;
62mod prepare;
63mod projection;
64mod run_projection;
65use prepare::Lent;
66pub mod section;
67pub mod stats;
68mod zones;
69
70pub use anchor::{LaneStart, LogAnchor};
71pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
72pub use projection::build_sorted_projection;
73pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
74pub use section::Section;
75pub use zones::{Common, Stripes, ascending, distincts, widths};
76
77const MAGIC: &[u8; 8] = b"RUDBNV10";
78const DIRECTORY: &[u8; 8] = b"RUDBDI10";
79const CATALOG: &[u8; 8] = b"RUDBCA10";
80const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
81const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
82const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
83const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
84const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
85const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
86const MAX_CATALOG_FREQUENCIES: usize = 64;
87const FORMAT: u32 = 30;
88
89const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
124
125const HEADER: u64 = 80;
126const SLOT_BYTES: usize = 28;
127const MAX_PAGE: usize = 256 * 1024 * 1024;
128const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
129const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
130const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
131const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
132const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
140const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
142const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
148const ORDINAL_BOUNDS: &[u8; 8] = b"RUDBFO1\0";
157const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
172const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
192const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
200const KEYS: &[u8; 8] = b"RUDBKY1\0";
207const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
215
216const MAX_SECTIONS: usize = 4096;
223const FREQUENCY_CANDIDATES: usize = 32_768;
224const FREQUENCY_ENTRIES: usize = 512;
225const FREQUENCY_BUILD_RANK: usize = 10;
226const FREQUENCY_ORDINALS: usize = 131_072;
227const MAX_PAIR_FREQUENCIES: usize = 1024;
228const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
233const MAX_FREQUENCY_WORKERS: usize = 32;
240
241fn close_workers() -> usize {
243 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
244}
245
246const CLOSE_BYTES: usize = 1 << 30;
257
258const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
261
262const MAX_ENCODE_WORKERS: usize = 32;
269
270const WRITEBACK_STRETCH: u64 = 32 << 20;
278
279const SIEVE_BUDGET: usize = 8 * 1024;
287
288const PART_BOUND_BYTES: usize = 24;
297
298fn io(error: std::io::Error) -> Error {
299 Error::io(error.to_string())
300}
301
302fn invalid(message: &str) -> Error {
303 Error::invalid_input(format!("invalid rudb native file: {message}"))
304}
305
306fn sum(counts: impl Iterator<Item = u64>) -> u64 {
308 counts.fold(0, u64::saturating_add)
309}
310
311fn span_bytes(spans: &[Span], at: usize) -> u64 {
313 spans.get(at).map_or(0, |span| u64::from(span.length))
314}
315
316fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
318 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
319}
320
321fn dictionary_bytes(table: &Table, at: usize) -> u64 {
323 page_bytes(&table.dictionaries, at)
324 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
325}
326
327fn checksum(bytes: &[u8]) -> u64 {
337 seeded_checksum(bytes, 0)
338}
339
340#[must_use]
347pub fn content_name(bytes: &[u8]) -> u128 {
348 let seed = u64::from(FORMAT);
349 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
350}
351
352#[derive(Debug, Clone)]
358pub struct ContentNamer {
359 seeds: [u64; 2],
360 lanes: [[u64; 4]; 2],
361 held: [u8; 32],
362 filled: usize,
363 length: u64,
364}
365
366impl Default for ContentNamer {
367 fn default() -> Self {
368 let seed = u64::from(FORMAT);
369 let seeds = [seed, !seed];
370 let lanes = seeds.map(|seed| {
371 [
372 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
373 seed.wrapping_add(XXH_P2),
374 seed,
375 seed.wrapping_sub(XXH_P1),
376 ]
377 });
378 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
379 }
380}
381
382impl ContentNamer {
383 pub fn update(&mut self, mut bytes: &[u8]) {
385 self.length += bytes.len() as u64;
386 if self.filled > 0 {
387 let take = (32 - self.filled).min(bytes.len());
388 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
389 self.filled += take;
390 bytes = &bytes[take..];
391 if self.filled < 32 {
392 return;
393 }
394 let block = self.held;
395 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
396 self.filled = 0;
397 }
398 let mut blocks = bytes.chunks_exact(32);
399 for block in blocks.by_ref() {
400 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
401 }
402 let rest = blocks.remainder();
403 self.held[..rest.len()].copy_from_slice(rest);
404 self.filled = rest.len();
405 }
406
407 #[must_use]
409 pub fn finish(&self) -> u128 {
410 let rest = &self.held[..self.filled];
411 let [first, second] = [0, 1].map(|at| {
412 if self.length < 32 {
413 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
414 } else {
415 finish_checksum(self.lanes[at], rest, self.length)
416 }
417 });
418 u128::from(first) << 64 | u128::from(second)
419 }
420}
421
422fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
431 let mut blocks = bytes.chunks_exact(32);
434 let rest = blocks.remainder();
435 if bytes.len() < 32 {
436 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
437 }
438 let mut lanes = [
439 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
440 seed.wrapping_add(XXH_P2),
441 seed,
442 seed.wrapping_sub(XXH_P1),
443 ];
444 for block in blocks.by_ref() {
445 checksum_block(&mut lanes, block);
446 }
447 finish_checksum(lanes, rest, bytes.len() as u64)
448}
449
450const XXH_P1: u64 = 11_400_714_785_074_694_791;
451const XXH_P2: u64 = 14_029_467_366_897_019_727;
452const XXH_P3: u64 = 1_609_587_929_392_839_161;
453const XXH_P4: u64 = 9_650_029_242_287_828_579;
454const XXH_P5: u64 = 2_870_177_450_012_600_261;
455
456fn checksum_round(state: u64, word: u64) -> u64 {
457 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
458}
459
460fn checksum_word(chunk: &[u8]) -> u64 {
461 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
462}
463
464fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
466 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
467 *lane = checksum_round(*lane, checksum_word(chunk));
468 }
469}
470
471fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
473 let merge = |state: u64, lane: u64| {
474 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
475 };
476 let [one, two, three, four] = lanes;
477 let combined = one
478 .rotate_left(1)
479 .wrapping_add(two.rotate_left(7))
480 .wrapping_add(three.rotate_left(12))
481 .wrapping_add(four.rotate_left(18));
482 let hash = merge(merge(merge(merge(combined, one), two), three), four);
483 checksum_tail(hash.wrapping_add(length), rest)
484}
485
486fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
488 let mut words = rest.chunks_exact(8);
489 for chunk in words.by_ref() {
490 hash ^= checksum_round(0, checksum_word(chunk));
491 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
492 }
493 rest = words.remainder();
494 if rest.len() >= 4 {
495 let (head, tail) = rest.split_at(4);
496 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
497 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
498 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
499 rest = tail;
500 }
501 for &byte in rest {
502 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
503 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
504 }
505 hash ^= hash >> 33;
506 hash = hash.wrapping_mul(XXH_P2);
507 hash ^= hash >> 29;
508 hash = hash.wrapping_mul(XXH_P3);
509 hash ^ (hash >> 32)
510}
511
512fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
518 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
519}
520
521fn walk_checksummed(
527 file: &File,
528 offset: u64,
529 length: usize,
530 window: usize,
531 mut each: impl FnMut(&[u8]) -> Result<()>,
532) -> Result<u64> {
533 debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
534 if length < 32 {
535 let mut bytes = vec![0; length];
536 read_at(file, offset, &mut bytes)?;
537 each(&bytes)?;
538 return Ok(checksum(&bytes));
539 }
540 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
541 let mut buffer = vec![0; window.min(length)];
542 let mut read = 0;
543 let (mut whole, mut filled) = (0, 0);
544 while read < length {
545 filled = buffer.len().min(length - read);
546 read_at(file, offset + read as u64, &mut buffer[..filled])?;
547 read += filled;
548 each(&buffer[..filled])?;
549 whole = filled / 32 * 32;
550 for block in buffer[..whole].chunks_exact(32) {
551 checksum_block(&mut lanes, block);
552 }
553 }
554 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
555}
556
557#[derive(Debug, Clone, Copy)]
558struct Slot {
559 offset: u64,
560 length: u32,
561 generation: u64,
562 hash: u64,
563}
564
565impl Slot {
566 fn bytes(self) -> [u8; SLOT_BYTES] {
567 let mut result = [0; SLOT_BYTES];
568 result[..8].copy_from_slice(&self.offset.to_le_bytes());
569 result[8..12].copy_from_slice(&self.length.to_le_bytes());
570 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
571 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
572 result
573 }
574
575 fn read(bytes: &[u8]) -> Self {
576 Self {
577 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
578 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
579 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
580 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
581 }
582 }
583}
584
585#[derive(Debug, Clone, Copy)]
586struct Page {
587 offset: u64,
588 length: u32,
589 hash: u64,
590}
591
592impl Page {
593 fn bytes(&self) -> u64 {
595 u64::from(self.length)
596 }
597}
598
599#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
600enum FrequencyValue {
601 Null,
602 Integer(i128),
603 Code(u32),
604}
605
606type FrequencyMap<V> = HashMap<u64, V, Spread>;
612
613#[derive(Debug)]
627struct Candidates {
628 slots: Vec<Candidate>,
631 held: usize,
632 nulls: u32,
633 decrements: u64,
634 survivors: Vec<Candidate>,
636}
637
638#[derive(Debug, Default, Clone, Copy)]
640struct Candidate {
641 bits: u64,
642 count: u32,
643}
644
645const FIRST_CANDIDATE_SLOTS: usize = 64;
647
648impl Default for Candidates {
649 fn default() -> Self {
650 Self {
651 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
652 held: 0,
653 nulls: 0,
654 decrements: 0,
655 survivors: Vec::new(),
656 }
657 }
658}
659
660impl Candidates {
661 fn add(&mut self, bits: Option<u64>, mut times: u32) {
668 while times > 0 {
669 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
670 match bits {
671 Some(bits) => {
672 let (at, found) = self.find(bits);
673 if found {
674 self.slots[at].count = self.slots[at].count.saturating_add(times);
675 return;
676 }
677 if room {
678 self.place(at, bits, times);
679 return;
680 }
681 }
682 None if self.nulls != 0 => {
683 self.nulls = self.nulls.saturating_add(times);
684 return;
685 }
686 None if room => {
687 self.nulls = times;
688 return;
689 }
690 None => {}
691 }
692 self.decrement();
693 times -= 1;
694 }
695 }
696
697 fn find(&self, bits: u64) -> (usize, bool) {
699 let mask = self.slots.len() - 1;
700 let mut at = home(bits, self.slots.len());
701 loop {
702 let slot = self.slots[at];
703 if slot.count == 0 {
704 return (at, false);
705 }
706 if slot.bits == bits {
707 return (at, true);
708 }
709 at = (at + 1) & mask;
710 }
711 }
712
713 fn position(&self, bits: u64) -> Option<usize> {
715 match self.find(bits) {
716 (at, true) => Some(at),
717 (_, false) => None,
718 }
719 }
720
721 fn place(&mut self, at: usize, bits: u64, count: u32) {
724 let at = if (self.held + 1) * 2 > self.slots.len() {
725 let wider = self.slots.len() * 2;
726 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
727 for slot in old.into_iter().filter(|slot| slot.count != 0) {
728 let (to, _) = self.find(slot.bits);
729 self.slots[to] = slot;
730 }
731 self.find(bits).0
732 } else {
733 at
734 };
735 self.slots[at] = Candidate { bits, count };
736 self.held += 1;
737 }
738
739 fn decrement(&mut self) {
741 let mut survivors = std::mem::take(&mut self.survivors);
742 survivors.clear();
743 survivors.extend(
744 self.slots
745 .iter()
746 .filter(|slot| slot.count > 1)
747 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
748 );
749 self.slots.fill(Candidate::default());
750 self.held = survivors.len();
751 for &slot in &survivors {
752 let (at, _) = self.find(slot.bits);
753 self.slots[at] = slot;
754 }
755 self.survivors = survivors;
756 self.nulls = self.nulls.saturating_sub(1);
757 self.decrements = self.decrements.saturating_add(1);
758 }
759
760 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
762 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
763 }
764}
765
766fn home(bits: u64, slots: usize) -> usize {
771 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
772}
773
774#[derive(Debug, Default)]
776struct Run {
777 bits: Option<u64>,
778 times: u32,
779}
780
781impl Run {
782 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
784 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
785 self.times += 1;
786 return None;
787 }
788 let ended = self.take();
789 self.bits = bits;
790 self.times = 1;
791 ended
792 }
793
794 fn take(&mut self) -> Option<(Option<u64>, u32)> {
796 let times = std::mem::take(&mut self.times);
797 (times != 0).then_some((self.bits, times))
798 }
799}
800
801#[derive(Debug, Default, Clone, Copy)]
803struct Spread;
804
805impl std::hash::BuildHasher for Spread {
806 type Hasher = SpreadHasher;
807
808 fn build_hasher(&self) -> SpreadHasher {
809 SpreadHasher(0)
810 }
811}
812
813#[derive(Debug)]
820struct SpreadHasher(u64);
821
822impl SpreadHasher {
823 fn mix(&mut self, word: u64) {
824 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
825 self.0 = (product as u64) ^ ((product >> 64) as u64);
826 }
827}
828
829impl std::hash::Hasher for SpreadHasher {
830 fn write(&mut self, bytes: &[u8]) {
831 for part in bytes.chunks(8) {
832 let mut word = [0; 8];
833 word[..part.len()].copy_from_slice(part);
834 self.mix(u64::from_le_bytes(word));
835 }
836 }
837
838 fn write_u32(&mut self, value: u32) {
839 self.mix(u64::from(value));
840 }
841
842 fn write_u64(&mut self, value: u64) {
843 self.mix(value);
844 }
845
846 fn write_i128(&mut self, value: i128) {
847 self.mix(value as u64);
848 self.mix((value >> 64) as u64);
849 }
850
851 fn write_isize(&mut self, value: isize) {
852 self.mix(value as u64);
853 }
854
855 fn finish(&self) -> u64 {
856 self.0
857 }
858}
859
860#[derive(Debug, Clone)]
861struct FrequencyEntry {
862 value: FrequencyValue,
863 count: u64,
864}
865
866#[derive(Debug, Clone)]
871struct FrequencySummary {
872 entries: Vec<FrequencyEntry>,
873 omitted_max: u64,
874 ordinals: Vec<u64>,
875 ordinal_entries: Vec<u16>,
876 ordinal_bound: u64,
879}
880
881#[derive(Debug, Clone)]
882struct PairFrequencyEntry {
883 first_entry: u16,
884 second: Option<u32>,
885 count: u64,
886}
887
888#[derive(Debug, Clone)]
894struct PairFrequencySummary {
895 first: u16,
896 second: u16,
897 entries: Vec<PairFrequencyEntry>,
898 omitted_max: u64,
899}
900
901type FrequencyHead = (Vec<FrequencyEntry>, u64);
903
904#[derive(Debug, Clone)]
912enum Frequencies {
913 Held(FrequencySummary),
914 Stored {
917 span: Span,
918 values: bool,
919 entries: usize,
920 },
921}
922
923#[derive(Debug, Clone)]
928pub struct FrequencyPrefix {
929 pub entries: Vec<(Value, u64)>,
931 pub omitted_max: u64,
933}
934
935#[derive(Debug, Clone, PartialEq)]
937pub struct FrequencyOccurrences {
938 pub omitted_max: u64,
940 pub ordinals: Vec<u64>,
942 pub anchors: Vec<Value>,
944 pub anchor_indices: Vec<u16>,
946}
947
948pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
950
951#[derive(Debug, Clone, Copy, Default)]
958struct Span {
959 offset: u64,
960 length: u32,
961}
962
963#[derive(Debug, Clone, Default)]
971struct Pages {
972 columns: usize,
973 held: Box<[StripePage]>,
974}
975
976#[derive(Debug, Clone, Copy)]
978struct StripePage {
979 offset: u64,
980 hash: u64,
981 length: u32,
982 column: u32,
983}
984
985impl Pages {
986 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
988 let mut held = Vec::with_capacity(slots.iter().flatten().count());
989 for (column, page) in slots.iter().enumerate() {
990 if let Some(page) = page {
991 let column =
992 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
993 held.push(StripePage {
994 offset: page.offset,
995 hash: page.hash,
996 length: page.length,
997 column,
998 });
999 }
1000 }
1001 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1002 }
1003
1004 fn get(&self, column: usize) -> Option<Page> {
1006 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1007 let placed = self.held[at];
1008 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1009 }
1010
1011 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1013 (0..self.columns).map(|column| self.get(column))
1014 }
1015
1016 fn bytes(&self, column: usize) -> u64 {
1018 self.get(column).map_or(0, |page| page.bytes())
1019 }
1020}
1021
1022#[derive(Debug, Clone)]
1024pub struct Stripe {
1025 rows: usize,
1026 parts: Vec<u32>,
1029 index: Span,
1033 pages: Vec<Span>,
1034 memberships: Pages,
1035 sieves: Pages,
1038 part_ranges: Pages,
1049 zone: Zone,
1050}
1051
1052impl Stripe {
1053 #[must_use]
1055 pub fn rows(&self) -> usize {
1056 self.rows
1057 }
1058
1059 #[must_use]
1061 pub fn parts(&self) -> usize {
1062 self.parts.len()
1063 }
1064
1065 #[must_use]
1071 pub fn zone(&self) -> &Zone {
1072 &self.zone
1073 }
1074}
1075
1076#[derive(Debug, Clone)]
1078pub struct Table {
1079 name: String,
1080 fields: Vec<Field>,
1081 stripes: Vec<Stripe>,
1082 rows: usize,
1083 dictionaries: Vec<Option<Page>>,
1084 dictionary_payloads: Vec<u64>,
1090 demoted: Vec<bool>,
1096 frequencies: Vec<Option<Frequencies>>,
1097 ordinal_bounds: Vec<u64>,
1100 pair_frequencies: Vec<PairFrequencySummary>,
1101 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1106 host_groups: Option<host::HostSummary>,
1108 distincts: Vec<Option<u64>>,
1118 clustering: Option<Clustering>,
1126 generation: u64,
1140 sections: Vec<Section>,
1147 constraints: Constraints,
1150}
1151
1152#[derive(Debug, Clone, Default, PartialEq, Eq)]
1157pub struct Constraints {
1158 pub keys: Vec<(Vec<u16>, bool)>,
1160 pub foreign: Vec<StoredForeign>,
1162}
1163
1164impl Constraints {
1165 #[must_use]
1167 pub fn is_empty(&self) -> bool {
1168 self.keys.is_empty() && self.foreign.is_empty()
1169 }
1170}
1171
1172#[derive(Debug, Clone, PartialEq, Eq)]
1174pub struct StoredForeign {
1175 pub columns: Vec<u16>,
1177 pub table: String,
1179 pub referenced: Vec<u16>,
1181}
1182
1183impl Table {
1184 #[must_use]
1186 pub fn name(&self) -> &str {
1187 &self.name
1188 }
1189
1190 #[must_use]
1192 pub fn fields(&self) -> &[Field] {
1193 &self.fields
1194 }
1195
1196 #[must_use]
1198 pub fn rows(&self) -> usize {
1199 self.rows
1200 }
1201
1202 #[must_use]
1204 pub fn stripes(&self) -> &[Stripe] {
1205 &self.stripes
1206 }
1207
1208 #[must_use]
1210 pub fn clustering(&self) -> Option<&Clustering> {
1211 self.clustering.as_ref()
1212 }
1213
1214 #[must_use]
1216 pub fn constraints(&self) -> &Constraints {
1217 &self.constraints
1218 }
1219
1220 #[must_use]
1225 pub fn generation(&self) -> u64 {
1226 self.generation
1227 }
1228
1229 #[must_use]
1236 pub fn sections(&self) -> &[Section] {
1237 &self.sections
1238 }
1239}
1240
1241#[derive(Debug, Clone)]
1253struct Entry {
1254 name: String,
1255 fields: Vec<Field>,
1256 rows: usize,
1257 directory: Page,
1259 nonzero: Vec<Option<u64>>,
1262 aggregates: Vec<Option<(i128, u64)>>,
1264 distincts: Vec<Option<u64>>,
1266 extremes: Vec<StoredIntegerExtremes>,
1268 frequencies: Vec<StoredNumericFrequencies>,
1270}
1271
1272type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1273type StoredNumericFrequencies = Option<NumericFrequencies>;
1274
1275#[derive(Debug, Clone, PartialEq, Eq)]
1288pub struct ViewEntry {
1289 pub name: String,
1291 pub sql: String,
1293 pub statement: String,
1295 pub aliases: Vec<String>,
1297 pub columns: Vec<Field>,
1299}
1300
1301#[derive(Debug, Clone)]
1303pub struct ColumnLayout {
1304 pub name: String,
1306 pub kind: String,
1308 pub pages: u64,
1310 pub memberships: u64,
1312 pub sieves: u64,
1314 pub part_ranges: u64,
1316 pub dictionary: u64,
1318}
1319
1320impl ColumnLayout {
1321 #[must_use]
1323 pub fn total(&self) -> u64 {
1324 self.pages
1325 .saturating_add(self.memberships)
1326 .saturating_add(self.sieves)
1327 .saturating_add(self.part_ranges)
1328 .saturating_add(self.dictionary)
1329 }
1330}
1331
1332#[derive(Debug, Clone)]
1343pub struct Layout {
1344 pub file: u64,
1346 pub rows: usize,
1348 pub stripes: usize,
1350 pub parts: usize,
1352 pub columns: Vec<ColumnLayout>,
1354 pub indexes: u64,
1357 pub directory: u64,
1359 pub header: u64,
1361}
1362
1363impl Layout {
1364 #[must_use]
1366 pub fn columns_total(&self) -> u64 {
1367 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1368 }
1369
1370 #[must_use]
1376 pub fn unaccounted(&self) -> u64 {
1377 self.file
1378 .saturating_sub(self.columns_total())
1379 .saturating_sub(self.indexes)
1380 .saturating_sub(self.directory)
1381 .saturating_sub(self.header)
1382 }
1383}
1384
1385#[derive(Debug, Clone)]
1396pub struct StoredPart {
1397 pub stripe: usize,
1399 pub part: usize,
1401 pub row: usize,
1403 pub rows: usize,
1405 pub encoding: String,
1407 pub bytes: u64,
1409 pub page: u64,
1411 pub offset: u64,
1413 pub low: Option<Value>,
1415 pub high: Option<Value>,
1417 pub nulls: Option<usize>,
1419}
1420
1421const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1428
1429#[derive(Debug)]
1454struct GlobalDictionary {
1455 primary: HashMap<u64, u32, Spread>,
1459 collisions: HashMap<u64, Vec<u32>, Spread>,
1460 checks: Vec<u64>,
1462 ends: Vec<u32>,
1464 counts: Vec<u64>,
1465 nulls: u64,
1466 filling: Vec<u8>,
1468 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1474 waiting: Vec<(usize, Vec<u8>)>,
1479 sample: Vec<(usize, Vec<u8>)>,
1485 stride: usize,
1487 shape: Option<chooser::Settled>,
1489 settled: usize,
1491 blocks: Vec<Vec<u8>>,
1496 early: BTreeMap<usize, EncodedBlock>,
1502 placed: Vec<Placed>,
1504 charged: u64,
1507 demoted: bool,
1509}
1510
1511#[derive(Debug, Clone, Copy)]
1513struct Placed {
1514 start: u64,
1515 length: u64,
1516 hash: u64,
1517}
1518
1519type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1521
1522impl GlobalDictionary {
1523 fn new() -> Self {
1524 Self {
1525 primary: HashMap::default(),
1526 collisions: HashMap::default(),
1527 checks: Vec::new(),
1528 ends: Vec::new(),
1529 counts: Vec::new(),
1530 nulls: 0,
1531 filling: Vec::new(),
1532 grams: Vec::new(),
1533 waiting: Vec::new(),
1534 sample: Vec::new(),
1535 stride: 1,
1536 shape: None,
1537 settled: 0,
1538 blocks: Vec::new(),
1539 early: BTreeMap::new(),
1540 placed: Vec::new(),
1541 charged: 0,
1542 demoted: false,
1543 }
1544 }
1545
1546 fn values(&self) -> usize {
1548 self.ends.len()
1549 }
1550
1551 fn closing_bytes(&self) -> usize {
1554 let values = self.values();
1555 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1556 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1557 .sum::<usize>();
1558 let beside = size_of::<u32>().max(size_of::<(u64, Option<u32>)>());
1562 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + beside))
1563 }
1564
1565 fn held_bytes(&self) -> u64 {
1571 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1572 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1573 }
1574 fn spilled<T>(values: &Vec<T>) -> usize {
1575 values.capacity() * size_of::<T>()
1576 }
1577 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1578 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1579 };
1580 let bytes = table(&self.primary)
1581 + table(&self.collisions)
1582 + self.collisions.values().map(spilled).sum::<usize>()
1583 + spilled(&self.checks)
1584 + spilled(&self.ends)
1585 + spilled(&self.counts)
1586 + self.filling.capacity()
1587 + spilled(&self.grams)
1588 + raw(&self.waiting)
1589 + raw(&self.sample)
1590 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1591 + spilled(&self.placed);
1592 bytes as u64
1593 }
1594
1595 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1598 let before = self.charged;
1599 let now = self.held_bytes();
1600 if let Some(profile) = profile {
1601 if now >= before {
1602 profile.hold(now - before);
1603 } else {
1604 profile.release(before - now);
1605 }
1606 }
1607 self.charged = now;
1608 (before, now)
1609 }
1610
1611 fn demote(&mut self) {
1619 if self.demoted {
1620 return;
1621 }
1622 self.seal_rest();
1623 self.release_lookup();
1624 self.demoted = true;
1625 }
1626
1627 fn release_lookup(&mut self) {
1634 self.primary = HashMap::default();
1635 self.collisions = HashMap::default();
1636 self.checks = Vec::new();
1637 self.sample = Vec::new();
1638 self.filling = Vec::new();
1639 }
1640
1641 fn encoded(&self) -> usize {
1643 self.placed.len() + self.blocks.len()
1644 }
1645
1646 #[cfg(test)]
1647 fn code(&mut self, text: &str) -> Result<u32> {
1648 let bytes = text.as_bytes();
1649 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1650 }
1651
1652 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1658 if let Some(&code) = self.primary.get(&hash) {
1659 if self.checks.get(code as usize) == Some(&check) {
1660 return Ok(code);
1661 }
1662 if let Some(codes) = self.collisions.get(&hash)
1663 && let Some(code) =
1664 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1665 {
1666 return Ok(code);
1667 }
1668 let code = self.insert(text, check)?;
1669 self.collisions.entry(hash).or_default().push(code);
1670 return Ok(code);
1671 }
1672 let code = self.insert(text, check)?;
1673 self.primary.insert(hash, code);
1674 Ok(code)
1675 }
1676
1677 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1678 if self.demoted {
1679 return Err(Error::internal("a value was coded against a demoted dictionary"));
1680 }
1681 let code = u32::try_from(self.ends.len())
1682 .map_err(|_| invalid("global dictionary has too many values"))?;
1683 self.filling.extend_from_slice(text);
1684 self.ends.push(
1685 u32::try_from(self.filling.len())
1686 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1687 );
1688 self.checks.push(check);
1689 self.counts.push(0);
1690 if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1691 self.seal();
1692 }
1693 Ok(code)
1694 }
1695
1696 fn seal(&mut self) {
1702 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1703 let bytes = std::mem::take(&mut self.filling);
1704 if at.is_multiple_of(self.stride) {
1705 self.sample.push((at, bytes.clone()));
1706 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1707 self.stride *= 2;
1708 let stride = self.stride;
1709 self.sample.retain(|(at, _)| at % stride == 0);
1710 }
1711 }
1712 self.waiting.push((at, bytes));
1713 }
1714
1715 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1717 block_values(self.block_ends(at), bytes)
1718 }
1719
1720 fn block_ends(&self, at: usize) -> &[u32] {
1722 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1723 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1724 &self.ends[first..last]
1725 }
1726
1727 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1734 let Some(shape) = &self.shape else { return Vec::new() };
1735 let waiting = std::mem::take(&mut self.waiting);
1736 waiting
1737 .into_iter()
1738 .map(|(at, bytes)| Unencoded {
1739 column,
1740 at,
1741 ends: self.block_ends(at).to_vec(),
1742 bytes,
1743 shape: shape.clone(),
1744 })
1745 .collect()
1746 }
1747
1748 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1751 if at < self.encoded() || self.early.insert(at, block).is_some() {
1752 return Err(Error::internal("a dictionary block came back twice"));
1753 }
1754 while let Some(block) = self.early.remove(&self.encoded()) {
1755 self.push_block(block);
1756 }
1757 Ok(())
1758 }
1759
1760 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1762 self.blocks.push(bytes);
1763 self.grams.push(*grams);
1764 }
1765
1766 fn settle(&mut self) -> Result<()> {
1774 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1775 return Ok(());
1776 }
1777 self.settle_on_sample()
1778 }
1779
1780 fn settle_rest(&mut self) -> Result<()> {
1788 if self.shape.is_some() || self.sample.is_empty() {
1789 return Ok(());
1790 }
1791 self.settle_on_sample()
1792 }
1793
1794 fn settle_on_sample(&mut self) -> Result<()> {
1795 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1796 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1797 return Ok(());
1798 }
1799 let sample =
1800 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1801 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1802 self.settled = complete;
1803 Ok(())
1804 }
1805
1806 fn seal_rest(&mut self) {
1808 if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1812 self.seal();
1813 }
1814 }
1815
1816 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1819 let (block, bytes) = &self.waiting[at];
1820 let values = self.slices(*block, bytes);
1821 let encoded = match &self.shape {
1822 Some(shape) => string::encode_with(&values, shape)?,
1823 None => string::encode(&values)?,
1824 };
1825 Ok((encoded, block_grams(&values)))
1826 }
1827
1828 #[cfg(test)]
1830 fn finish_blocks(&mut self) -> Result<()> {
1831 self.seal_rest();
1832 let made = (0..self.waiting.len())
1833 .map(|at| self.encode_waiting(at))
1834 .collect::<Result<Vec<_>>>()?;
1835 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1836 if self.encoded() != at {
1837 return Err(Error::internal("a dictionary block was encoded out of order"));
1838 }
1839 self.push_block(block);
1840 }
1841 Ok(())
1842 }
1843
1844 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1862 let count = self.placed.len() + self.blocks.len();
1863 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1864 return Err(invalid("global dictionary blocks do not cover its values"));
1865 }
1866 let mut bases = Vec::with_capacity(count);
1867 let mut total = 0_usize;
1868 for block in 0..count {
1869 bases.push(total as u64);
1870 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1871 total = total
1872 .checked_add(self.ends[last] as usize)
1873 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1874 }
1875 let mut flat = vec![0_u8; total];
1876 let mut outs = Vec::with_capacity(count);
1877 let mut rest = flat.as_mut_slice();
1878 for block in 0..count {
1879 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1880 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1881 outs.push((block, out));
1882 rest = after;
1883 }
1884 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1885 let mut stored = Vec::new();
1886 for (block, out) in run {
1887 let encoded = match self.placed.get(*block) {
1888 Some(place) => {
1889 let file = file.ok_or_else(|| {
1890 Error::internal("a written dictionary block has no file")
1891 })?;
1892 let length = usize::try_from(place.length).map_err(|_| {
1893 invalid("global dictionary block does not fit in memory")
1894 })?;
1895 stored.resize(length, 0);
1896 read_at(file, place.start, &mut stored)?;
1897 if checksum(&stored) != place.hash {
1898 return Err(invalid(
1899 "a global dictionary block did not read back as written",
1900 ));
1901 }
1902 stored.as_slice()
1903 }
1904 None => &self.blocks[*block - self.placed.len()],
1905 };
1906 let decoded = string::decode_flat(encoded)?;
1907 if decoded.bytes().len() != out.len() {
1908 return Err(invalid(
1909 "a global dictionary block is not the length its ends say",
1910 ));
1911 }
1912 out.copy_from_slice(decoded.bytes());
1913 }
1914 Ok(())
1915 };
1916 let workers = close_workers().min(count / 16).max(1);
1919 if workers <= 1 {
1920 one(&mut outs)?;
1921 } else {
1922 let per = count.div_ceil(workers);
1923 std::thread::scope(|scope| {
1924 outs.chunks_mut(per)
1925 .map(|run| scope.spawn(|| one(run)))
1926 .collect::<Vec<_>>()
1927 .into_iter()
1928 .try_for_each(|handle| {
1929 handle.join().map_err(|_| {
1930 Error::internal("a global dictionary decode worker panicked")
1931 })?
1932 })
1933 })?;
1934 }
1935 drop(outs);
1936 Ok((flat, bases))
1937 }
1938
1939 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1944 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1945 let Some(&end) = ends.get(code) else { return (0, 0) };
1946 let base = base as usize;
1947 let from =
1948 if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1949 (base + from, base + end as usize)
1950 }
1951
1952 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1972 let (flat, bases) = self.decoded(file)?;
1973 let value = |code: u32| {
1974 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1975 flat.get(from..to).unwrap_or_default()
1976 };
1977 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1978 sort_by_value_across(&mut codes, value, close_workers());
1979 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1980 Ok((order, flat, bases))
1981 }
1982
1983 #[cfg(test)]
1984 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1985 self.ranked_with_values(file).map(|(order, _, _)| order)
1986 }
1987}
1988
1989#[derive(Debug)]
1997pub struct Writer {
1998 file: Box<dyn rudb_io::File>,
2001 at: u64,
2009 written_back: u64,
2011 table: Table,
2012 generation: u64,
2013 order: Vec<((u64, u64), (u64, u64))>,
2016 next_order: u64,
2017 dictionaries: Vec<Option<GlobalDictionary>>,
2018 coded: Arc<prepare::Coding>,
2021 gathers: Vec<Option<stats::Gather>>,
2027 lent: Option<Arc<Lent>>,
2030 pending: Vec<PendingChunk>,
2031 closed: Vec<Entry>,
2033 views: Vec<ViewEntry>,
2038 card: Option<KeptCard>,
2040 anchor: Option<LogAnchor>,
2043 profile: Option<Arc<LoadProfile>>,
2049}
2050
2051#[derive(Debug)]
2059struct PendingChunk {
2060 order: (u64, u64),
2061 chunk: Chunk,
2062}
2063
2064#[derive(Debug, Clone, Copy)]
2070struct Part {
2071 order: (u64, u64),
2072 rows: usize,
2073 footprint: usize,
2074}
2075
2076impl Part {
2077 fn of(pending: &PendingChunk) -> Self {
2078 Self {
2079 order: pending.order,
2080 rows: pending.chunk.len(),
2081 footprint: pending.chunk.footprint(),
2082 }
2083 }
2084}
2085
2086#[derive(Debug, Default)]
2092struct ColumnStripe {
2093 pages: Vec<Vec<u8>>,
2094 sums: Vec<u64>,
2097 codes: Vec<Option<Vec<u32>>>,
2098 sieves: Vec<Option<Sieve>>,
2099 ranges: Vec<Range>,
2100}
2101
2102fn coded_type(ty: &LogicalType) -> bool {
2110 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2111}
2112
2113fn dictionary_tag(ty: &LogicalType) -> u8 {
2120 if ty == &LogicalType::Blob { 2 } else { 1 }
2121}
2122
2123fn weight(ty: &LogicalType) -> usize {
2131 match ty {
2132 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2133 LogicalType::HugeInt
2134 | LogicalType::UHugeInt
2135 | LogicalType::Uuid
2136 | LogicalType::Interval => 16,
2137 LogicalType::BigInt
2138 | LogicalType::UBigInt
2139 | LogicalType::Timestamp
2140 | LogicalType::Time
2141 | LogicalType::TimeTz
2142 | LogicalType::TimestampTz
2143 | LogicalType::TimestampS
2144 | LogicalType::TimestampMs
2145 | LogicalType::TimestampNs
2146 | LogicalType::Double
2147 | LogicalType::Decimal { .. } => 8,
2148 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2149 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2150 _ => 1,
2151 }
2152}
2153
2154pub const STRIPE_PARTS: usize = 64;
2161
2162const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2170
2171const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2187
2188const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2190
2191fn index_section(parts: usize) -> Result<usize> {
2193 parts
2194 .checked_mul(INDEX_ENTRY)
2195 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2196 .ok_or_else(|| invalid("index page length overflow"))
2197}
2198
2199impl Writer {
2200 pub fn open(
2219 path: impl AsRef<Path>,
2220 name: impl Into<String>,
2221 fields: Vec<Field>,
2222 ) -> Result<Self> {
2223 Self::open_in(&RealFilesystem::new(), path, name, fields)
2224 }
2225
2226 pub fn open_in(
2233 fs: &dyn Filesystem,
2234 path: impl AsRef<Path>,
2235 name: impl Into<String>,
2236 fields: Vec<Field>,
2237 ) -> Result<Self> {
2238 for field in &fields {
2239 type_tag(&field.ty)?;
2240 }
2241 let name = name.into();
2242 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2243 let size = file.len()?;
2244 let (slot, bytes, _) = committed_slot(&*file, size)?;
2245 let (mut closed, views, card, anchor) = decode_catalog(&bytes, size)?;
2246 let card = card_for(path.as_ref(), card);
2247 if let Some(at) = closed.iter().position(|held| held.name == name) {
2258 if closed[at].rows > 0 {
2259 return Err(invalid("two tables in one native file have the same name"));
2260 }
2261 closed.remove(at);
2262 }
2263 let generation = slot
2268 .generation
2269 .checked_add(1)
2270 .ok_or_else(|| invalid("native file generation overflow"))?;
2271 Ok(Self {
2272 file,
2273 at: size,
2276 written_back: size,
2277 dictionaries: fields
2278 .iter()
2279 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2280 .collect(),
2281 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2282 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2283 lent: None,
2284 table: Table {
2285 name,
2286 dictionaries: vec![None; fields.len()],
2287 dictionary_payloads: Vec::new(),
2288 demoted: Vec::new(),
2289 distincts: vec![None; fields.len()],
2290 fields,
2291 stripes: Vec::new(),
2292 rows: 0,
2293 frequencies: Vec::new(),
2294 ordinal_bounds: Vec::new(),
2295 pair_frequencies: Vec::new(),
2296 frequency_texts: Vec::new(),
2297 host_groups: None,
2298 clustering: None,
2299 constraints: Constraints::default(),
2300 generation,
2301 sections: Vec::new(),
2302 },
2303 generation,
2304 order: Vec::new(),
2305 next_order: 0,
2306 pending: Vec::with_capacity(STRIPE_PARTS),
2307 closed,
2308 views,
2309 card,
2310 anchor,
2311 profile: None,
2312 })
2313 }
2314
2315 pub fn create(
2321 path: impl AsRef<Path>,
2322 name: impl Into<String>,
2323 fields: Vec<Field>,
2324 ) -> Result<Self> {
2325 Self::create_in(&RealFilesystem::new(), path, name, fields)
2326 }
2327
2328 pub fn create_in(
2338 fs: &dyn Filesystem,
2339 path: impl AsRef<Path>,
2340 name: impl Into<String>,
2341 fields: Vec<Field>,
2342 ) -> Result<Self> {
2343 for field in &fields {
2344 type_tag(&field.ty)?;
2345 }
2346 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2347 let mut header = [0; HEADER as usize];
2348 header[..8].copy_from_slice(MAGIC);
2349 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2350 file.write_at(0, &header)?;
2351 Ok(Self {
2352 file,
2353 at: HEADER,
2354 written_back: HEADER,
2355 dictionaries: fields
2356 .iter()
2357 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2358 .collect(),
2359 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2360 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2361 lent: None,
2362 table: Table {
2363 name: name.into(),
2364 dictionaries: vec![None; fields.len()],
2365 dictionary_payloads: Vec::new(),
2366 demoted: Vec::new(),
2367 distincts: vec![None; fields.len()],
2368 fields,
2369 stripes: Vec::new(),
2370 rows: 0,
2371 frequencies: Vec::new(),
2372 ordinal_bounds: Vec::new(),
2373 pair_frequencies: Vec::new(),
2374 frequency_texts: Vec::new(),
2375 host_groups: None,
2376 clustering: None,
2377 constraints: Constraints::default(),
2378 generation: 1,
2379 sections: Vec::new(),
2380 },
2381 generation: 1,
2382 order: Vec::new(),
2383 next_order: 0,
2384 pending: Vec::with_capacity(STRIPE_PARTS),
2385 closed: Vec::new(),
2386 views: Vec::new(),
2387 card: card_for(path.as_ref(), None),
2388 anchor: None,
2389 profile: None,
2390 })
2391 }
2392
2393 pub fn empty(
2417 path: impl AsRef<Path>,
2418 views: &[ViewEntry],
2419 anchor: Option<&LogAnchor>,
2420 ) -> Result<()> {
2421 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2422 let mut header = [0; HEADER as usize];
2423 header[..8].copy_from_slice(MAGIC);
2424 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2425 file.write_at(0, &header)?;
2426 let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref(), anchor)?;
2427 file.write_at(HEADER, &catalog)?;
2428 file.sync()?;
2432 let slot = Slot {
2433 offset: HEADER,
2434 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2435 generation: 1,
2436 hash: checksum(&catalog),
2437 };
2438 file.write_at(slot_offset(1), &slot.bytes())?;
2439 file.sync()?;
2440 Ok(())
2441 }
2442
2443 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2454 for field in &fields {
2455 type_tag(&field.ty)?;
2456 }
2457 let name = name.into();
2458 let entry = self.close()?;
2459 if entry.name == name {
2460 return Err(invalid("two tables in one native file have the same name"));
2461 }
2462 if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2466 if self.closed[at].rows > 0 {
2467 return Err(invalid("two tables in one native file have the same name"));
2468 }
2469 self.closed.remove(at);
2470 }
2471 let Self { file, at, generation, mut closed, views, card, anchor, .. } = self;
2472 closed.push(entry);
2473 Ok(Self {
2474 file,
2475 written_back: at,
2476 at,
2477 generation,
2478 closed,
2479 views,
2480 card,
2481 anchor,
2482 profile: None,
2483 dictionaries: fields
2484 .iter()
2485 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2486 .collect(),
2487 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2488 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2489 lent: None,
2490 table: Table {
2491 name,
2492 dictionaries: vec![None; fields.len()],
2493 dictionary_payloads: Vec::new(),
2494 demoted: Vec::new(),
2495 distincts: vec![None; fields.len()],
2496 fields,
2497 stripes: Vec::new(),
2498 rows: 0,
2499 frequencies: Vec::new(),
2500 ordinal_bounds: Vec::new(),
2501 pair_frequencies: Vec::new(),
2502 frequency_texts: Vec::new(),
2503 host_groups: None,
2504 clustering: None,
2505 constraints: Constraints::default(),
2506 generation,
2507 sections: Vec::new(),
2508 },
2509 order: Vec::new(),
2510 next_order: 0,
2511 pending: Vec::with_capacity(STRIPE_PARTS),
2512 })
2513 }
2514
2515 #[must_use]
2525 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2526 self.views = views;
2527 self
2528 }
2529
2530 #[must_use]
2533 pub fn with_log_anchor(mut self, anchor: LogAnchor) -> Self {
2534 self.anchor = Some(anchor);
2535 self
2536 }
2537
2538 #[must_use]
2544 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2545 self.profile = Some(profile);
2546 self
2547 }
2548
2549 #[must_use]
2553 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2554 self.coded.cap(bytes);
2555 self
2556 }
2557
2558 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2573 self.table.clustering = Some(Clustering::new(
2576 clustering.columns().to_vec(),
2577 clustering.width(),
2578 &self.table.fields,
2579 )?);
2580 Ok(self)
2581 }
2582
2583 pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2591 let width = self.table.fields.len();
2592 let fits = |columns: &[u16]| {
2593 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2594 };
2595 if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2596 || !constraints.foreign.iter().all(|foreign| {
2597 fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2598 })
2599 {
2600 return Err(invalid("a constraint names a column the table does not have"));
2601 }
2602 self.table.constraints = constraints;
2603 Ok(self)
2604 }
2605
2606 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2611 self.file.write_at(self.at, bytes)?;
2612 self.at = self
2613 .at
2614 .checked_add(bytes.len() as u64)
2615 .ok_or_else(|| invalid("native file length overflow"))?;
2616 if self.at - self.written_back >= WRITEBACK_STRETCH {
2617 self.file.start_writeback(self.written_back, self.at - self.written_back);
2618 self.written_back = self.at;
2619 }
2620 Ok(())
2621 }
2622
2623 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2629 let order = (self.next_order, 0);
2630 self.next_order = self.next_order.saturating_add(1);
2631 self.append_at(order, chunk)
2632 }
2633
2634 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2645 if chunk.is_empty() {
2646 return Ok(());
2647 }
2648 self.admit(chunk)?;
2649 if self.pending.last().is_some_and(|last| last.order > order) {
2650 self.flush_pending()?;
2651 }
2652 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2657 if self.pending.len() == STRIPE_PARTS {
2658 self.flush_pending()?;
2659 }
2660 Ok(())
2661 }
2662
2663 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2679 if parts.len() > STRIPE_PARTS {
2680 return Err(invalid("a stripe was handed more parts than it holds"));
2681 }
2682 self.flush_pending()?;
2685 for (order, chunk) in parts {
2686 if chunk.is_empty() {
2687 continue;
2688 }
2689 self.admit(&chunk)?;
2690 self.pending.push(PendingChunk { order, chunk });
2691 }
2692 self.flush_pending()
2693 }
2694
2695 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2697 if chunk.width() != self.table.fields.len() {
2698 return Err(invalid("chunk width differs from table schema"));
2699 }
2700 for (index, field) in self.table.fields.iter().enumerate() {
2701 if chunk.column(index)?.logical_type() != &field.ty {
2702 return Err(invalid("chunk type differs from table schema"));
2703 }
2704 }
2705 self.table.rows = self
2706 .table
2707 .rows
2708 .checked_add(chunk.len())
2709 .ok_or_else(|| invalid("row count overflow"))?;
2710 Ok(())
2711 }
2712
2713 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2715 let mut stripe = ColumnStripe {
2716 pages: Vec::with_capacity(columns.len()),
2717 sums: Vec::with_capacity(columns.len()),
2718 codes: Vec::with_capacity(columns.len()),
2719 sieves: Vec::with_capacity(columns.len()),
2720 ranges: Vec::with_capacity(columns.len()),
2721 };
2722 let mut settling = Settling::default();
2723 for &column in columns {
2724 Self::encode_page(&mut stripe, &mut settling, column)?;
2725 }
2726 Ok(stripe)
2727 }
2728
2729 fn encode_page(
2732 stripe: &mut ColumnStripe,
2733 settling: &mut Settling,
2734 column: &Vector,
2735 ) -> Result<()> {
2736 let bytes = encode(column, settling)?;
2737 if bytes.len() > MAX_PAGE {
2738 return Err(invalid("column page exceeds the configured bound"));
2739 }
2740 let range = Range::of(column);
2743 let sieve =
2754 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2755 stripe.sums.push(checksum(&bytes));
2756 stripe.pages.push(bytes);
2757 stripe.codes.push(None);
2758 stripe.sieves.push(sieve);
2759 stripe.ranges.push(range);
2760 Ok(())
2761 }
2762
2763 fn place_blocks(&mut self) -> Result<()> {
2768 if let Some(lent) = self.lent.clone() {
2769 return self.place_lent_blocks(&lent);
2770 }
2771 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2772 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2773 for block in std::mem::take(&mut dictionary.blocks) {
2774 let start = self.at;
2775 self.put(&block)?;
2776 dictionary.placed.push(Placed {
2777 start,
2778 length: block.len() as u64,
2779 hash: checksum(&block),
2780 });
2781 }
2782 Ok(())
2783 });
2784 self.dictionaries = dictionaries;
2785 placed
2786 }
2787
2788 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2794 for column in lent.columns() {
2795 let Ok(mut held) = column.try_lock() else { continue };
2796 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2797 for block in std::mem::take(&mut dictionary.blocks) {
2798 let start = self.at;
2799 self.put(&block)?;
2800 dictionary.placed.push(Placed {
2801 start,
2802 length: block.len() as u64,
2803 hash: checksum(&block),
2804 });
2805 }
2806 }
2807 Ok(())
2808 }
2809
2810 fn reclaim(&mut self) -> Result<()> {
2814 let Some(lent) = self.lent.take() else { return Ok(()) };
2815 let (dictionaries, gathers) = lent.reclaim()?;
2816 self.dictionaries = dictionaries;
2817 self.gathers = gathers;
2818 Ok(())
2819 }
2820
2821 fn flush_pending(&mut self) -> Result<()> {
2826 if self.pending.is_empty() {
2827 return Ok(());
2828 }
2829 let held = std::mem::take(&mut self.pending);
2830 let prepared = self.preparer().prepare_held(held)?;
2831 let merged = self.merge_held(prepared)?;
2832 let paged = merged.pages()?;
2833 self.write_paged(paged)
2834 }
2835
2836 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2838 let width = self.table.fields.len();
2839 let parts = held.len();
2840 if encoded.len() != width {
2841 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2842 }
2843 let profile = self.profile.clone();
2844 if let Some(profile) = &profile {
2845 let rows = held.iter().map(|part| part.rows as u64).sum();
2846 let raw = held.iter().map(|part| part.footprint as u64).sum();
2847 let pages =
2848 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2849 profile.moved(Stage::Pages, raw, pages, rows);
2850 }
2851 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2854 let before = self.at;
2855 self.place_blocks()?;
2856 drop(timing);
2857 if let Some(profile) = &profile {
2858 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2859 }
2860 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2861 let before = self.at;
2862 let mut pages = Vec::with_capacity(width);
2863 let mut memberships = vec![None; width];
2864 let mut ranges = Vec::with_capacity(width);
2865 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2866 let start = self.at;
2869 let mut out = Vec::with_capacity(width.saturating_mul(parts));
2870 for stripe in &encoded {
2871 let offset = self.at;
2872 let section = index.len();
2873 let mut length = 0_usize;
2874 if stripe.sums.len() != stripe.pages.len() {
2875 return Err(Error::internal("a stripe's pages came without their checksums"));
2876 }
2877 for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2878 put_u32(
2879 &mut index,
2880 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2881 );
2882 put_u64(&mut index, sum);
2883 out.push(bytes.as_slice());
2884 length = length
2885 .checked_add(bytes.len())
2886 .ok_or_else(|| invalid("column page length overflow"))?;
2887 }
2888 let hash = checksum(&index[section..]);
2889 put_u64(&mut index, hash);
2890 if length > MAX_PAGE {
2891 return Err(invalid("column page exceeds the configured bound"));
2892 }
2893 self.at = self
2894 .at
2895 .checked_add(length as u64)
2896 .ok_or_else(|| invalid("native file length overflow"))?;
2897 pages.push(Span {
2898 offset,
2899 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2900 });
2901 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2902 }
2903 self.file.write_parts_at(start, &out)?;
2904 drop(out);
2905 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2906 if stripe.codes.iter().all(Option::is_none) {
2907 continue;
2908 }
2909 let lists = stripe
2910 .codes
2911 .iter()
2912 .map(|codes| codes.clone().unwrap_or_default())
2913 .collect::<Vec<_>>();
2914 let bytes = encode_membership(&merged_codes(lists));
2915 let offset = self.at;
2916 self.put(&bytes)?;
2917 *membership = Some(Page {
2918 offset,
2919 length: u32::try_from(bytes.len())
2920 .map_err(|_| invalid("membership page length overflow"))?,
2921 hash: checksum(&bytes),
2922 });
2923 }
2924 let mut sieves = vec![None; width];
2925 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2926 if stripe.sieves.iter().all(Option::is_none) {
2927 continue;
2928 }
2929 let bytes = encode_sieves(stripe.sieves.iter())?;
2930 let offset = self.at;
2931 self.put(&bytes)?;
2932 *page = Some(Page {
2933 offset,
2934 length: u32::try_from(bytes.len())
2935 .map_err(|_| invalid("sieve page length overflow"))?,
2936 hash: checksum(&bytes),
2937 });
2938 }
2939 let mut part_ranges = vec![None; width];
2945 if parts > 1 {
2946 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2947 let bytes = encode_part_ranges(&stripe.ranges)?;
2948 if bytes.len() >= span.length as usize {
2949 continue;
2950 }
2951 let offset = self.at;
2952 self.put(&bytes)?;
2953 *page = Some(Page {
2954 offset,
2955 length: u32::try_from(bytes.len())
2956 .map_err(|_| invalid("part range page length overflow"))?,
2957 hash: checksum(&bytes),
2958 });
2959 }
2960 }
2961 let offset = self.at;
2962 self.put(&index)?;
2963 let index = Span {
2964 offset,
2965 length: u32::try_from(index.len())
2966 .map_err(|_| invalid("index page length overflow"))?,
2967 };
2968 let mut rows = 0_usize;
2969 let mut lengths = Vec::with_capacity(parts);
2970 let mut span = None;
2971 for part in held {
2972 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2973 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2974 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2975 }
2976 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2977 self.table.stripes.push(Stripe {
2978 rows,
2979 parts: lengths,
2980 index,
2981 pages,
2982 memberships: Pages::from_slots(memberships)?,
2983 sieves: Pages::from_slots(sieves)?,
2984 part_ranges: Pages::from_slots(part_ranges)?,
2985 zone: Zone::from_ranges(ranges),
2986 });
2987 drop(timing);
2988 if let Some(profile) = &profile {
2989 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2990 }
2991 Ok(())
2992 }
2993
2994 fn numeric_frequency(
3014 &self,
3015 column: usize,
3016 counted: bool,
3017 dense: Option<(u64, usize)>,
3018 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
3019 let signed = match self.table.fields[column].ty {
3020 LogicalType::TinyInt
3021 | LogicalType::SmallInt
3022 | LogicalType::Integer
3023 | LogicalType::BigInt
3024 | LogicalType::Date
3025 | LogicalType::Timestamp => true,
3026 LogicalType::UTinyInt
3027 | LogicalType::USmallInt
3028 | LogicalType::UInteger
3029 | LogicalType::UBigInt => false,
3030 _ => return Ok((None, None)),
3031 };
3032 let value_of = |bits: Option<u64>| match bits {
3033 None => FrequencyValue::Null,
3034 Some(bits) => integer_value(bits, signed),
3035 };
3036 let tallied = self
3041 .gathers
3042 .get(column)
3043 .and_then(Option::as_ref)
3044 .filter(|gather| gather.rows() == self.table.rows as u64)
3045 .and_then(stats::Gather::frequencies)
3046 .and_then(|(values, nulls)| {
3047 let entries = values
3048 .iter()
3049 .map(|(value, count)| {
3050 let value = value_of(Some(frequency_bits(value)?));
3051 Some(FrequencyEntry { value, count: *count })
3052 })
3053 .chain((nulls != 0).then_some(Some(FrequencyEntry {
3054 value: FrequencyValue::Null,
3055 count: nulls,
3056 })))
3057 .collect::<Option<Vec<_>>>()?;
3058 Some((entries, values.len() as u64))
3059 });
3060 let exact = match (&tallied, counted) {
3064 (None, true) => self.exact_frequency(column, signed, dense)?,
3065 _ => None,
3066 };
3067 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3068 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3069 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3070 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3071 (None, None) => {
3072 let mut first = Candidates::default();
3076 let mut run = Run::default();
3077 self.visit_numeric(column, signed, |_, bits| {
3078 if let Some((ended, times)) = run.push(bits) {
3079 first.add(ended, times);
3080 }
3081 })?;
3082 if let Some((bits, times)) = run.take() {
3083 first.add(bits, times);
3084 }
3085 let (nulls, decrements) = (first.nulls, first.decrements);
3088 let distinct_count = (decrements == 0).then_some(first.held as u64);
3089 let (exact, null_count) = if decrements == 0 {
3090 let exact = first
3091 .pairs()
3092 .map(|(bits, count)| (bits, u64::from(count)))
3093 .collect::<FrequencyMap<_>>();
3094 (exact, (nulls != 0).then_some(u64::from(nulls)))
3095 } else {
3096 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3097 if nulls != 0 {
3098 lower.push(nulls);
3099 }
3100 lower.sort_unstable_by(|left, right| right.cmp(left));
3101 if lower.len() < FREQUENCY_BUILD_RANK
3102 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3103 {
3104 return Ok((None, distinct_count));
3105 }
3106 let mut recounts = vec![0_u64; first.slots.len()];
3109 let mut null_count = (nulls != 0).then_some(0_u64);
3110 let mut recount = |bits: Option<u64>, times: u32| {
3111 let held = match bits {
3112 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3113 None => null_count.as_mut(),
3114 };
3115 if let Some(count) = held {
3116 *count = count.saturating_add(u64::from(times));
3117 }
3118 };
3119 let mut run = Run::default();
3120 self.visit_numeric(column, signed, |_, bits| {
3121 if let Some((bits, times)) = run.push(bits) {
3122 recount(bits, times);
3123 }
3124 })?;
3125 if let Some((bits, times)) = run.take() {
3126 recount(bits, times);
3127 }
3128 let exact = first
3129 .slots
3130 .iter()
3131 .zip(&recounts)
3132 .filter(|(slot, _)| slot.count != 0)
3133 .map(|(slot, &count)| (slot.bits, count))
3134 .collect::<FrequencyMap<_>>();
3135 (exact, null_count)
3136 };
3137 let entries = exact
3138 .into_iter()
3139 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3140 .chain(
3141 null_count
3142 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3143 )
3144 .collect::<Vec<_>>();
3145 (entries, decrements, distinct_count)
3146 }
3147 };
3148 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3149 if omitted_max == 0 && entries.len() > 1 {
3153 let retained = entries.len().saturating_sub(1).min(2);
3154 omitted_max = entries[retained].count;
3155 entries.truncate(retained);
3156 }
3157 let mut covered = 0;
3161 let mut kept_rows = 0_u64;
3162 for entry in &entries {
3163 match kept_rows.checked_add(entry.count) {
3164 Some(total) if total <= FREQUENCY_ORDINALS as u64 => kept_rows = total,
3165 _ => break,
3166 }
3167 covered += 1;
3168 }
3169 let ordinal_bound = entries.get(covered).map_or(0, |entry| entry.count);
3170 let worth_keeping = covered == entries.len()
3171 || (covered >= FREQUENCY_BUILD_RANK
3172 && entries[FREQUENCY_BUILD_RANK - 1].count > ordinal_bound.max(omitted_max));
3173 let mut ordinals = Vec::new();
3174 let mut ordinal_entries = Vec::new();
3175 if worth_keeping {
3176 let mut kept = FrequencyMap::default();
3177 let mut null_kept = None;
3178 for (at, entry) in entries.iter().enumerate().take(covered) {
3179 let at = u16::try_from(at)
3180 .map_err(|_| invalid("too many retained frequency entries"))?;
3181 match entry.value {
3182 FrequencyValue::Integer(value) => {
3183 kept.insert(value as u64, at);
3184 }
3185 FrequencyValue::Null => null_kept = Some(at),
3186 FrequencyValue::Code(_) => {}
3187 }
3188 }
3189 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3190 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3191 self.visit_numeric(column, signed, |ordinal, bits| {
3192 let held = match bits {
3193 Some(bits) => kept.get(&bits).copied(),
3194 None => null_kept,
3195 };
3196 if let Some(entry) = held {
3197 ordinals.push(ordinal);
3198 ordinal_entries.push(entry);
3199 }
3200 })?;
3201 }
3202 Ok((
3203 Some(FrequencySummary {
3204 entries,
3205 omitted_max,
3206 ordinals,
3207 ordinal_entries,
3208 ordinal_bound: if worth_keeping { ordinal_bound } else { 0 },
3209 }),
3210 distinct_count,
3211 ))
3212 }
3213
3214 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3221 let rows = self.table.rows;
3222 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3223 return None;
3224 }
3225 let (low, high) = gather.span()?;
3226 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3227 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3228 let bits = low as u64;
3229 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3230 }
3231
3232 fn exact_frequency(
3246 &self,
3247 column: usize,
3248 signed: bool,
3249 dense: Option<(u64, usize)>,
3250 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3251 if let Some((low, len)) = dense {
3254 let mut counts = distinct::DenseCounts::new(low, len);
3255 let nulls =
3256 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3257 if let Some(distinct) = counts.count() {
3258 let Some(distinct) = distinct else { return Ok(None) };
3259 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3260 counts.visit(visit);
3261 })));
3262 }
3263 }
3264 let mut set = distinct::ExactCounts::new();
3265 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3266 let Some(distinct) = set.count() else {
3267 return Ok(None);
3268 };
3269 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3270 set.visit(visit);
3271 })))
3272 }
3273
3274 fn count_numeric(
3277 &self,
3278 column: usize,
3279 signed: bool,
3280 mut add: impl FnMut(u64, u32),
3281 ) -> Result<u64> {
3282 let mut nulls = 0_u64;
3283 let mut run = Run::default();
3284 let mut take = |bits: Option<u64>, times: u32| match bits {
3285 Some(bits) => add(bits, times),
3286 None => nulls += u64::from(times),
3287 };
3288 self.visit_numeric(column, signed, |_, bits| {
3289 if let Some((bits, times)) = run.push(bits) {
3290 take(bits, times);
3291 }
3292 })?;
3293 if let Some((bits, times)) = run.take() {
3294 take(bits, times);
3295 }
3296 Ok(nulls)
3297 }
3298
3299 fn frequent_entries(
3302 &self,
3303 signed: bool,
3304 distinct: u64,
3305 nulls: u64,
3306 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3307 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3308 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3310 let mut rank = |count: u64| {
3311 if top.len() <= FREQUENCY_ENTRIES {
3312 top.push(Reverse(count));
3313 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3314 top.pop();
3315 top.push(Reverse(count));
3316 }
3317 };
3318 visit(&mut |_, count| rank(count));
3319 if nulls != 0 {
3320 rank(nulls);
3321 }
3322 let top = top.into_sorted_vec();
3323 let values = distinct + u64::from(nulls != 0);
3324 if values > FREQUENCY_CANDIDATES as u64 {
3325 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3326 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3327 return (None, distinct);
3328 }
3329 }
3330 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3331 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3332 visit(&mut |bits, count| {
3333 if count >= least {
3334 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3335 }
3336 });
3337 if nulls != 0 && nulls >= least {
3338 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3339 }
3340 (Some(entries), distinct)
3341 }
3342
3343 fn visit_numeric(
3350 &self,
3351 column: usize,
3352 signed: bool,
3353 mut visit: impl FnMut(u64, Option<u64>),
3354 ) -> Result<()> {
3355 let ty = &self.table.fields[column].ty;
3356 let mut start = 0_u64;
3357 let mut block = Vec::new();
3358 for stripe in &self.table.stripes {
3359 let spans = read_index(&self.file, stripe, column)?;
3360 let page = stripe.pages[column];
3361 let mut bytes = vec![0; page.length as usize];
3362 read_at(&self.file, page.offset, &mut bytes)?;
3363 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3364 let part = part_bytes(&bytes, *span)?;
3365 if checksum(part) != span.hash {
3366 return Err(invalid("column page checksum differs while building frequencies"));
3367 }
3368 let rows = rows as usize;
3369 let vector = decode(ty, rows, part, None)?;
3370 if signed && vector.signed_block(&mut block) && block.len() == rows {
3374 if vector.none_null() {
3375 for (row, &value) in block.iter().enumerate() {
3376 visit(start.saturating_add(row as u64), Some(value as u64));
3377 }
3378 } else {
3379 for (row, &value) in block.iter().enumerate() {
3380 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3381 visit(start.saturating_add(row as u64), bits);
3382 }
3383 }
3384 start = start.saturating_add(rows as u64);
3385 continue;
3386 }
3387 for row in 0..rows {
3389 let bits = if vector.is_null_at(row) {
3390 None
3391 } else {
3392 let widened = match vector.signed_at(row) {
3396 Some(value) => Some(value as u64),
3397 None => match vector.value_at(row) {
3398 Value::UTinyInt(value) => Some(u64::from(value)),
3399 Value::USmallInt(value) => Some(u64::from(value)),
3400 Value::UInteger(value) => Some(u64::from(value)),
3401 Value::UBigInt(value) => Some(value),
3402 _ => None,
3403 },
3404 };
3405 Some(widened.ok_or_else(|| {
3406 invalid("numeric frequency page did not contain an integer value")
3407 })?)
3408 };
3409 visit(start.saturating_add(row as u64), bits);
3410 }
3411 start = start.saturating_add(rows as u64);
3412 }
3413 }
3414 Ok(())
3415 }
3416
3417 fn numeric_columns(&self) -> Vec<usize> {
3419 self.table
3420 .fields
3421 .iter()
3422 .enumerate()
3423 .filter_map(|(column, field)| {
3424 matches!(
3425 field.ty,
3426 LogicalType::TinyInt
3427 | LogicalType::SmallInt
3428 | LogicalType::Integer
3429 | LogicalType::BigInt
3430 | LogicalType::UTinyInt
3431 | LogicalType::USmallInt
3432 | LogicalType::UInteger
3433 | LogicalType::UBigInt
3434 | LogicalType::Date
3435 | LogicalType::Timestamp
3436 )
3437 .then_some(column)
3438 })
3439 .collect()
3440 }
3441
3442 #[allow(dead_code)]
3444 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3445 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3446 return Ok(None);
3447 }
3448 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3449 return Err(invalid("frequency ordinals are not sorted and unique"));
3450 }
3451 let mut out = Vec::with_capacity(ordinals.len());
3452 let mut wanted = 0;
3453 let mut stripe_start = 0_u64;
3454 for stripe in &self.table.stripes {
3455 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3456 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3457 stripe_start = stripe_end;
3458 continue;
3459 }
3460 let spans = read_index(&self.file, stripe, column)?;
3461 let page = stripe.pages[column];
3462 let mut bytes = vec![0; page.length as usize];
3463 read_at(&self.file, page.offset, &mut bytes)?;
3464 let mut part_start = stripe_start;
3465 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3466 let part_end = part_start.saturating_add(u64::from(rows));
3467 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3468 let part = part_bytes(&bytes, *span)?;
3469 if checksum(part) != span.hash {
3470 return Err(invalid(
3471 "column page checksum differs while building pair frequencies",
3472 ));
3473 }
3474 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3475 let positions = ordinals[wanted..upto]
3476 .iter()
3477 .map(|&ordinal| {
3478 usize::try_from(ordinal.saturating_sub(part_start))
3479 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3480 })
3481 .collect::<Result<Vec<_>>>()?;
3482 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3483 return Ok(None);
3484 }
3485 wanted = upto;
3486 }
3487 part_start = part_end;
3488 }
3489 stripe_start = stripe_end;
3490 }
3491 if wanted != ordinals.len() {
3492 return Err(invalid("frequency ordinal is outside the table"));
3493 }
3494 Ok(Some(out))
3495 }
3496
3497 #[allow(dead_code)]
3499 fn pair_frequencies(
3500 &self,
3501 frequencies: &[Option<Frequencies>],
3502 ) -> Result<Vec<PairFrequencySummary>> {
3503 let anchors = frequencies
3504 .iter()
3505 .enumerate()
3506 .filter_map(|(column, summary)| {
3507 match summary {
3509 Some(Frequencies::Held(summary)) => Some(summary),
3510 _ => None,
3511 }
3512 .filter(|summary| {
3513 !summary.ordinals.is_empty()
3514 && summary.ordinal_entries.len() == summary.ordinals.len()
3515 && summary.ordinal_bound == 0
3516 })
3517 .cloned()
3518 .map(|summary| (column, summary))
3519 })
3520 .collect::<Vec<_>>();
3521 let strings = self
3522 .dictionaries
3523 .iter()
3524 .enumerate()
3525 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3526 .collect::<Vec<_>>();
3527 let mut summaries = Vec::new();
3528 for (first, anchors) in anchors {
3529 for &second in &strings {
3530 if summaries.len() == MAX_PAIR_FREQUENCIES {
3531 return Ok(summaries);
3532 }
3533 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3534 continue;
3535 };
3536 if codes.len() != anchors.ordinal_entries.len() {
3537 return Err(invalid("pair frequency columns have different lengths"));
3538 }
3539 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3540 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3541 *counts.entry((anchor, code)).or_default() += 1;
3542 }
3543 let mut entries = counts
3544 .into_iter()
3545 .map(|((first_entry, second), count)| PairFrequencyEntry {
3546 first_entry,
3547 second,
3548 count,
3549 })
3550 .collect::<Vec<_>>();
3551 entries.sort_unstable_by(|left, right| {
3552 right
3553 .count
3554 .cmp(&left.count)
3555 .then_with(|| left.first_entry.cmp(&right.first_entry))
3556 .then_with(|| left.second.cmp(&right.second))
3557 });
3558 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3559 entries.truncate(FREQUENCY_ENTRIES);
3560 summaries.push(PairFrequencySummary {
3561 first: u16::try_from(first)
3562 .map_err(|_| invalid("pair frequency column index overflows"))?,
3563 second: u16::try_from(second)
3564 .map_err(|_| invalid("pair frequency column index overflows"))?,
3565 entries,
3566 omitted_max: anchors.omitted_max.max(pair_omitted),
3567 });
3568 }
3569 }
3570 Ok(summaries)
3571 }
3572
3573 fn close(&mut self) -> Result<Entry> {
3584 self.reclaim()?;
3585 self.flush_pending()?;
3586 let profile = self.profile.clone();
3590 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3591 let before = self.at;
3592 let mut stripes = std::mem::take(&mut self.order)
3593 .into_iter()
3594 .zip(std::mem::take(&mut self.table.stripes))
3595 .collect::<Vec<_>>();
3596 stripes.sort_by_key(|(order, _)| order.0);
3597 let mut previous: Option<(u64, u64)> = None;
3598 for ((first, last), _) in &stripes {
3599 if previous.is_some_and(|previous| previous >= *first) {
3600 return Err(invalid("chunks did not arrive in source order"));
3601 }
3602 previous = Some(*last);
3603 }
3604 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3605 drop(timing);
3606 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3607 let placing = self.at;
3608 finish_dictionaries(&mut self.dictionaries)?;
3609 self.place_blocks()?;
3610 for dictionary in self.dictionaries.iter_mut().flatten() {
3611 dictionary.release_lookup();
3612 dictionary.recharge(profile.as_deref());
3613 }
3614 let (numeric, closed) = self.close_columns()?;
3615 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3616 numeric.into_iter().unzip();
3617 let frequencies =
3618 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3619 let pairs = Vec::new();
3621 self.table.frequencies = frequencies;
3622 self.table.distincts = distincts;
3623 self.table.pair_frequencies = pairs;
3624 if let Some(profile) = &profile {
3625 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3626 }
3627 self.table.demoted = self
3628 .dictionaries
3629 .iter()
3630 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3631 .collect();
3632 if !self.table.demoted.contains(&true) {
3633 self.table.demoted = Vec::new();
3634 }
3635 self.dictionaries = Vec::new();
3636 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3637 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3638 self.table.host_groups = None;
3639 for (index, closed) in closed.into_iter().enumerate() {
3640 let Some(closed) = closed else { continue };
3641 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3642 self.table.distincts[index] = distinct;
3643 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3644 self.table.frequency_texts[index] = texts;
3645 if hosts.is_some() {
3646 self.table.host_groups = hosts;
3647 }
3648 let offset = self.at;
3649 self.put(&encoded.index)?;
3650 self.put(&encoded.ranks)?;
3651 self.put(&encoded.grams)?;
3652 self.table.dictionary_payloads[index] = payload;
3653 let length = encoded
3654 .index
3655 .len()
3656 .checked_add(encoded.ranks.len())
3657 .and_then(|len| len.checked_add(encoded.grams.len()))
3658 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3659 self.table.dictionaries[index] = Some(Page {
3660 offset,
3661 length: u32::try_from(length)
3662 .map_err(|_| invalid("dictionary page length overflow"))?,
3663 hash: checksum(&encoded.index),
3664 });
3665 }
3666 drop(timing);
3667 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3668 let placed = self.at - placing;
3669 self.write_stats()?;
3670 let directory = encode_directory(&self.table)?;
3671 if directory.len() > MAX_DIRECTORY {
3672 return Err(invalid("directory exceeds the configured bound"));
3673 }
3674 let offset = self.at;
3675 self.put(&directory)?;
3676 drop(timing);
3677 if let Some(profile) = &profile {
3678 profile.moved(Stage::Dictionary, 0, placed, 0);
3679 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3680 }
3681 Ok(Entry {
3682 name: self.table.name.clone(),
3683 fields: self.table.fields.clone(),
3684 rows: self.table.rows,
3685 nonzero: vec![None; self.table.fields.len()],
3686 aggregates: table_aggregate_sums(&self.table),
3687 distincts: self.table.distincts.clone(),
3688 extremes: table_integer_extremes(&self.table),
3689 frequencies: table_complete_numeric_frequencies(&self.table),
3690 directory: Page {
3691 offset,
3692 length: u32::try_from(directory.len())
3693 .map_err(|_| invalid("directory length overflow"))?,
3694 hash: checksum(&directory),
3695 },
3696 })
3697 }
3698
3699 #[allow(clippy::type_complexity)]
3716 fn close_columns(
3717 &self,
3718 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3719 let numeric = self.numeric_columns().into_iter().map(|column| {
3720 let gather = self.gathers.get(column).and_then(Option::as_ref);
3721 let estimate = gather.and_then(stats::Gather::distinct);
3722 let counted = !estimate.is_some_and(distinct::beyond);
3723 let set =
3724 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3725 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3726 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3727 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3728 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3729 });
3730 let dictionaries =
3731 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3732 let dictionary = dictionary.as_ref()?;
3733 let bytes = dictionary.closing_bytes();
3734 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3735 });
3736 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3737 jobs.sort_by_key(|&(_, _, cost)| cost);
3738 let columns = self.table.fields.len();
3739 let mut frequencies = vec![(None, None); columns];
3740 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3741 let profile = self.profile.as_deref();
3742 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3743 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3744 let closed = match job {
3745 Closing::Numeric { column, counted, dense } => {
3746 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3747 Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3748 }
3749 Closing::Dictionary { index, dictionary } => {
3750 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3751 Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3752 }
3753 };
3754 rudb_common::heap::release();
3757 Ok(closed)
3758 };
3759 let workers = close_workers().min(jobs.len());
3760 let pieces = if workers <= 1 {
3761 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3762 } else {
3763 let state = Mutex::new((jobs, 0_usize));
3765 let finished = Condvar::new();
3766 std::thread::scope(|scope| {
3767 (0..workers)
3768 .map(|_| {
3769 scope.spawn(|| {
3770 let mut mine = Vec::new();
3771 loop {
3772 let mut held = state.lock().map_err(|_| {
3773 Error::internal("a native close worker panicked")
3774 })?;
3775 let (job, bytes) = loop {
3776 let (jobs, busy) = &mut *held;
3777 if jobs.is_empty() {
3778 return Ok(mine);
3779 }
3780 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3781 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3782 });
3783 if let Some(at) = fits {
3784 let (job, bytes, _) = jobs.remove(at);
3785 *busy += bytes;
3786 break (job, bytes);
3787 }
3788 held = finished.wait(held).map_err(|_| {
3789 Error::internal("a native close worker panicked")
3790 })?;
3791 };
3792 drop(held);
3793 let _room = Room { state: &state, finished: &finished, bytes };
3796 mine.push(run(job, bytes)?);
3797 }
3798 })
3799 })
3800 .collect::<Vec<_>>()
3801 .into_iter()
3802 .map(|handle| {
3803 handle
3804 .join()
3805 .map_err(|_| Error::internal("a native close worker panicked"))?
3806 })
3807 .collect::<Result<Vec<_>>>()
3808 })?
3809 .into_iter()
3810 .flatten()
3811 .collect()
3812 };
3813 for piece in pieces {
3814 match piece {
3815 Closed::Numeric(column, summary) => frequencies[column] = summary,
3816 Closed::Dictionary(index, one) => closed[index] = Some(one),
3817 }
3818 }
3819 Ok((frequencies, closed))
3820 }
3821
3822 fn close_dictionary(
3829 &self,
3830 _index: usize,
3831 dictionary: &GlobalDictionary,
3832 ) -> Result<ClosedDictionary> {
3833 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3834 let (distinct, frequencies, texts) = if dictionary.demoted {
3839 (None, None, Vec::new())
3840 } else {
3841 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3842 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3843 (Some(distinct), Some(frequencies), texts)
3844 };
3845 let hosts = None;
3847 drop(flat);
3848 drop(bases);
3849 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3850 let payload = dictionary
3851 .placed
3852 .iter()
3853 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3854 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3855 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3856 }
3857
3858 fn write_stats(&mut self) -> Result<()> {
3870 let gathers = std::mem::take(&mut self.gathers);
3871 let rows = self.table.rows as u64;
3872 let mut payloads = Vec::new();
3873 for (column, gather) in gathers.into_iter().enumerate() {
3874 let Some(gather) = gather else { continue };
3875 if gather.rows() != rows {
3881 continue;
3882 }
3883 let Some(stats) = gather.finish() else { continue };
3884 let mut summary = Vec::new();
3885 stats.summary.encode(&mut summary)?;
3886 let mut sketches = Vec::new();
3887 stats.sketches.encode(&mut sketches)?;
3888 payloads.push((column, summary, sketches));
3889 }
3890 if payloads.is_empty() {
3891 return Ok(());
3892 }
3893 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3894 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3895 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3896 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3899 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3900 if !built {
3901 continue;
3902 }
3903 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3904 let sections = [
3905 (*section::SUMMARY, summary, summary.len() as u32),
3908 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3909 ];
3910 let wanted = 1 + usize::from(sketched);
3911 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3912 let written = write_section(
3913 &*self.file,
3914 &mut self.at,
3915 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3916 self.generation,
3917 )?;
3918 self.table.sections.push(written);
3919 }
3920 }
3921 if self.table.sections.len() > MAX_SECTIONS {
3922 return Err(invalid("the table would name more sections than the bound allows"));
3923 }
3924 Ok(())
3925 }
3926
3927 pub fn finish(mut self) -> Result<Table> {
3937 let entry = self.close()?;
3938 let profile = self.profile.take();
3939 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3940 let mut tables = std::mem::take(&mut self.closed);
3941 tables.push(entry);
3942 let catalog =
3943 encode_catalog(&tables, &self.views, self.card.as_ref(), self.anchor.as_ref())?;
3944 if catalog.len() > MAX_DIRECTORY {
3945 return Err(invalid("catalog exceeds the configured bound"));
3946 }
3947 let offset = self.at;
3948 self.put(&catalog)?;
3949 if let Some(profile) = &profile {
3950 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3951 }
3952 synced(&*self.file, profile.as_deref())?;
3956 let slot = Slot {
3957 offset,
3958 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3959 generation: self.generation,
3960 hash: checksum(&catalog),
3961 };
3962 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3967 synced(&*self.file, profile.as_deref())?;
3968 Ok(self.table)
3969 }
3970
3971 pub fn restate(
3990 path: impl AsRef<Path>,
3991 views: &[ViewEntry],
3992 anchor: Option<&LogAnchor>,
3993 ) -> Result<()> {
3994 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3995 let size = file.len()?;
3996 let (slot, bytes, _) = committed_slot(&*file, size)?;
3997 let (closed, _, card, held) = decode_catalog(&bytes, size)?;
3998 let anchor = anchor.cloned().or(held);
3999 let generation = slot
4000 .generation
4001 .checked_add(1)
4002 .ok_or_else(|| invalid("native file generation overflow"))?;
4003 let catalog = encode_catalog(
4004 &closed,
4005 views,
4006 card_for(path.as_ref(), card).as_ref(),
4007 anchor.as_ref(),
4008 )?;
4009 if catalog.len() > MAX_DIRECTORY {
4010 return Err(invalid("catalog exceeds the configured bound"));
4011 }
4012 file.write_at(size, &catalog)?;
4013 file.sync()?;
4014 let slot = Slot {
4015 offset: size,
4016 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4017 generation,
4018 hash: checksum(&catalog),
4019 };
4020 file.write_at(slot_offset(generation), &slot.bytes())?;
4021 file.sync()?;
4022 Ok(())
4023 }
4024
4025 pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
4036 let path = path.as_ref();
4037 let (_, size, _, bytes, _) = slot_bytes(path)?;
4038 let (_, views, held, _) = decode_catalog(&bytes, size)?;
4039 if card_for(path, held.clone()) == held {
4040 return Ok(());
4041 }
4042 Self::restate(path, &views, None)
4043 }
4044
4045 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
4048 let path = path.as_ref();
4049 let (_, size, slot, bytes, _) = slot_bytes(path)?;
4050 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4051 let native = Catalog::open(path)?;
4052 for entry in &mut entries {
4053 let reader = native.table(&entry.name)?;
4054 entry.nonzero.fill(None);
4055 entry.aggregates = reader_aggregate_sums(&reader)?;
4056 entry.distincts = (0..entry.fields.len())
4057 .map(|column| reader.distinct_values(column))
4058 .collect::<Result<Vec<_>>>()?;
4059 entry.extremes = reader_integer_extremes(&reader)?;
4060 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
4061 }
4062 let generation = slot
4063 .generation
4064 .checked_add(1)
4065 .ok_or_else(|| invalid("native file generation overflow"))?;
4066 let catalog =
4067 encode_catalog(&entries, &views, card_for(path, card).as_ref(), anchor.as_ref())?;
4068 if catalog.len() > MAX_DIRECTORY {
4069 return Err(invalid("catalog exceeds the configured bound"));
4070 }
4071 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
4072 file.write_at(size, &catalog)?;
4073 file.sync()?;
4074 let slot = Slot {
4075 offset: size,
4076 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4077 generation,
4078 hash: checksum(&catalog),
4079 };
4080 file.write_at(slot_offset(generation), &slot.bytes())?;
4081 file.sync()?;
4082 Ok(())
4083 }
4084
4085 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4087 Self::certify_summaries(path)
4088 }
4089}
4090
4091fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4097 let offset = *at;
4098 file.write_at(offset, bytes)?;
4099 *at =
4100 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4101 Ok(offset)
4102}
4103
4104fn write_section(
4110 file: &dyn rudb_io::File,
4111 at: &mut u64,
4112 one: §ion::Attachment<'_>,
4113 generation: u64,
4114) -> Result<Section> {
4115 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4119 return Err(invalid("a section's header is longer than its payload"));
4120 }
4121 let mut extents = Vec::new();
4122 let mut first = 0_u64;
4123 let extent_size =
4124 if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4125 run_projection::RLE_PAGE_BYTES
4126 } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4127 1 << 19
4128 } else {
4129 section::MAX_EXTENT as usize
4130 };
4131 for chunk in one.bytes.chunks(extent_size) {
4132 let offset = append(file, at, chunk)?;
4133 extents.push(section::Extent {
4134 offset,
4135 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4136 hash: checksum(chunk),
4137 first,
4138 });
4139 first += chunk.len() as u64;
4140 }
4141 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4142 section::encode_extents(&extents, &mut table)?;
4143 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4147 Ok(Section {
4148 kind: one.kind,
4149 id: one.id,
4150 generation,
4151 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4152 extent_page,
4153 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4154 hash: checksum(&table),
4155 flags: one.flags,
4156 header_bytes: one.header_bytes,
4157 })
4158}
4159
4160pub fn attach(
4184 path: impl AsRef<Path>,
4185 table: &str,
4186 attachments: &[section::Attachment<'_>],
4187) -> Result<Table> {
4188 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4189 let file = &*file;
4190 let size = file.len()?;
4191 let (slot, bytes, _) = committed_slot(file, size)?;
4192 let (mut entries, views, card, anchor) = decode_catalog(&bytes, size)?;
4193 let card = card_for(path.as_ref(), card);
4194 let at = entries
4195 .iter()
4196 .position(|entry| entry.name == table)
4197 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4198 let mut version = [0; 4];
4199 read_at(file, 8, &mut version)?;
4200 let version = u32::from_le_bytes(version);
4201 if version != FORMAT {
4207 return Err(invalid(&format!(
4208 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4209 to be written again"
4210 )));
4211 }
4212 let mut directory = vec![0; entries[at].directory.length as usize];
4213 read_at(file, entries[at].directory.offset, &mut directory)?;
4214 if checksum(&directory) != entries[at].directory.hash {
4215 return Err(invalid(&format!("the directory of table {table} does not checksum")));
4216 }
4217 let mut held = decode_directory(&directory, size)?;
4218 let mut cursor = size;
4219 for one in attachments {
4220 let written = write_section(file, &mut cursor, one, held.generation)?;
4221 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4222 held.sections.push(written);
4223 }
4224 if held.sections.len() > MAX_SECTIONS {
4225 return Err(invalid("the table would name more sections than the bound allows"));
4226 }
4227 let encoded = encode_directory(&held)?;
4228 if encoded.len() > MAX_DIRECTORY {
4229 return Err(invalid("directory exceeds the configured bound"));
4230 }
4231 let offset = append(file, &mut cursor, &encoded)?;
4232 entries[at].directory = Page {
4233 offset,
4234 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4235 hash: checksum(&encoded),
4236 };
4237 let catalog = encode_catalog(&entries, &views, card.as_ref(), anchor.as_ref())?;
4240 if catalog.len() > MAX_DIRECTORY {
4241 return Err(invalid("catalog exceeds the configured bound"));
4242 }
4243 let offset = append(file, &mut cursor, &catalog)?;
4244 file.sync()?;
4245 let generation =
4246 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4247 let committed = Slot {
4248 offset,
4249 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4250 generation,
4251 hash: checksum(&catalog),
4252 };
4253 file.write_at(slot_offset(generation), &committed.bytes())?;
4254 file.sync()?;
4255 Ok(held)
4256}
4257
4258type Synopsis = Arc<Vec<(Value, u64)>>;
4261
4262#[derive(Debug, Clone)]
4264pub struct Reader {
4265 file: Arc<File>,
4266 map: Option<Arc<Mapped>>,
4268 table: Arc<Table>,
4269 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4270 loading: Arc<Vec<Mutex<()>>>,
4279 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4282 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4286 frequency_heads: Arc<Vec<OnceLock<Arc<FrequencyHead>>>>,
4290 summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4292 opened: Arc<AtomicUsize>,
4296 sieves: Arc<Vec<OnceLock<Box<[SieveSlot]>>>>,
4301 part_ranges: Arc<Vec<OnceLock<Box<[RangeSlot]>>>>,
4304 places: Arc<Vec<Place>>,
4306 cache: Arc<Shelf>,
4307 pool: PagePool,
4309 pages: Arc<AtomicUsize>,
4312 indexes: Arc<AtomicUsize>,
4315 verified: Arc<Vec<AtomicU64>>,
4325 unreleased: Arc<Vec<AtomicU32>>,
4333 text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4336 firsts: Arc<Vec<usize>>,
4338 graph: Arc<graph::Decoded>,
4341 size: u64,
4343 directory: u64,
4345 opening: Opening,
4347}
4348
4349#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4361pub struct Opening {
4362 pub reads: u32,
4365 pub bytes: u64,
4367}
4368
4369#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4371pub struct Reads {
4372 pub opening: Opening,
4374 pub pages: usize,
4376 pub indexes: usize,
4378 pub dictionaries: usize,
4381}
4382
4383#[derive(Debug, Clone, Copy)]
4385struct Place {
4386 stripe: u32,
4387 part: u32,
4388 rows: u32,
4389}
4390
4391#[derive(Debug, Clone, Copy)]
4393struct PartSpan {
4394 start: usize,
4395 length: usize,
4396 hash: u64,
4397}
4398
4399#[derive(Debug, Clone)]
4405struct CachedColumn {
4406 stripe: usize,
4407 index: Arc<Vec<PartSpan>>,
4408 page: Option<Arc<HeldPage>>,
4409}
4410
4411#[derive(Debug)]
4418struct HeldPage {
4419 bytes: PageBytes,
4420 checked: Vec<AtomicBool>,
4421}
4422
4423#[derive(Debug)]
4425enum PageBytes {
4426 Read(Vec<u8>),
4427 Mapped { map: Arc<Mapped>, offset: u64, length: usize },
4428}
4429
4430impl PageBytes {
4431 fn bytes(&self) -> &[u8] {
4432 match self {
4433 Self::Read(bytes) => bytes,
4434 Self::Mapped { map, offset, length } => map.get(*offset, *length).unwrap_or_default(),
4436 }
4437 }
4438}
4439
4440impl HeldPage {
4441 fn bytes(&self) -> &[u8] {
4442 self.bytes.bytes()
4443 }
4444
4445 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4447 let bytes = part_bytes(self.bytes(), span)?;
4448 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4449 if !checked.load(Atomic::Relaxed) {
4450 verify_part(bytes, span)?;
4451 checked.store(true, Atomic::Relaxed);
4452 }
4453 Ok(bytes)
4454 }
4455}
4456
4457fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4459 let got = checksum(bytes);
4460 if got != span.hash {
4461 return Err(invalid(&format!(
4462 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4463 span.start, span.length, span.hash,
4464 )));
4465 }
4466 Ok(())
4467}
4468
4469#[derive(Debug, Default)]
4509struct Cached {
4510 pages: Vec<Option<Resident>>,
4511 loading: Vec<usize>,
4512 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4513 touched: Vec<Vec<u64>>,
4514 passing: VecDeque<usize>,
4515}
4516
4517#[derive(Debug, Clone)]
4519struct Resident {
4520 page: Arc<HeldPage>,
4521 used: Arc<AtomicBool>,
4522}
4523
4524#[derive(Debug)]
4526struct Shelf {
4527 columns: Vec<Mutex<Cached>>,
4528 held: Vec<AtomicUsize>,
4531 kept: AtomicUsize,
4534}
4535
4536#[derive(Debug, Clone, Default)]
4555pub struct PagePool {
4556 ring: Arc<Mutex<Ring>>,
4557 budget: Arc<AtomicUsize>,
4558}
4559
4560#[derive(Debug, Default)]
4561struct Ring {
4562 held: VecDeque<Held>,
4563 bytes: usize,
4564}
4565
4566#[derive(Debug)]
4571struct Held {
4572 shelf: Weak<Shelf>,
4573 column: usize,
4574 stripe: usize,
4575 bytes: usize,
4576 used: Arc<AtomicBool>,
4577}
4578
4579impl PagePool {
4580 #[must_use]
4582 pub fn new(budget: usize) -> Self {
4583 let pool = Self::default();
4584 pool.budget.store(budget, Atomic::Relaxed);
4585 pool
4586 }
4587
4588 #[must_use]
4594 pub fn bytes(&self) -> usize {
4595 self.ring.lock().map_or(0, |ring| ring.bytes)
4596 }
4597
4598 fn admit(&self, held: Held) {
4604 let budget = self.budget.load(Atomic::Relaxed);
4605 let mut gone = Vec::new();
4606 {
4607 let Ok(mut ring) = self.ring.lock() else { return };
4608 ring.bytes += held.bytes;
4609 ring.held.push_back(held);
4610 let mut looked = 0;
4613 let limit = ring.held.len();
4614 while ring.bytes > budget && looked < limit {
4615 looked += 1;
4616 let Some(entry) = ring.held.pop_front() else { break };
4617 let Some(shelf) = entry.shelf.upgrade() else {
4618 ring.bytes -= entry.bytes;
4619 continue;
4620 };
4621 if entry.used.swap(false, Atomic::Relaxed) {
4622 ring.held.push_back(entry);
4623 continue;
4624 }
4625 let count = &shelf.held[entry.column];
4626 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4627 ring.held.push_back(entry);
4628 continue;
4629 }
4630 count.fetch_sub(1, Atomic::Relaxed);
4631 ring.bytes -= entry.bytes;
4632 gone.push((shelf, entry));
4633 }
4634 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4637 if let Some(entry) = ring.held.pop_front() {
4638 ring.bytes -= entry.bytes;
4639 }
4640 }
4641 }
4642 for (shelf, entry) in gone {
4643 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4644 if let Some(slot) = cached.pages.get_mut(entry.stripe)
4645 && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4646 {
4647 *slot = None;
4648 }
4649 }
4650 }
4651}
4652
4653const CACHED_STRIPES_PER_COLUMN: usize = 4;
4665
4666type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4668
4669type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4670
4671#[derive(Debug)]
4672struct NativeText {
4673 file: Arc<File>,
4674 values: usize,
4676 offsets: Vec<u8>,
4688 offset_bits: usize,
4691 value_ends: OnceLock<Option<Vec<u32>>>,
4704 value_lens: OnceLock<Option<Lengths>>,
4714 ends_asked: AtomicUsize,
4720 ranks: usize,
4722 rank_at: u64,
4726 rank_ends: Vec<u64>,
4730 rank_hashes: Vec<u64>,
4731 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4732 code_bits: usize,
4735 code_ranks: OnceLock<Option<Vec<u32>>>,
4742 starts: Vec<u64>,
4749 lengths: Vec<u64>,
4750 hashes: Vec<u64>,
4751 grams: Option<NativeGrams>,
4753 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4755 char_lens: Vec<OnceLock<Box<[u32]>>>,
4764 keep_budget: usize,
4767 payload_kept: AtomicUsize,
4775 swept: Vec<AtomicBool>,
4783 visit_dropped: AtomicUsize,
4798 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4815}
4816
4817#[derive(Debug)]
4818struct NativeGrams {
4819 start: u64,
4820 length: usize,
4821 width: usize,
4823 hash: u64,
4824 verdicts: Mutex<Vec<Verdict>>,
4831}
4832
4833type Verdict = (Vec<u8>, Arc<[bool]>);
4835
4836const GRAM_VERDICTS: usize = 8;
4838
4839impl NativeGrams {
4840 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4845 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4846 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4847 return Ok(Arc::clone(verdict));
4848 }
4849 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4850 let mut verdict = Vec::with_capacity(self.length / self.width);
4851 let window = GRAM_WINDOW / self.width * self.width;
4852 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4853 verdict.extend(bytes.chunks(self.width).map(|bits| {
4854 wanted
4855 .iter()
4856 .flatten()
4857 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4858 }));
4859 Ok(())
4860 })?;
4861 if hash != self.hash {
4862 return Err(invalid("global dictionary substring signatures checksum differs"));
4863 }
4864 let verdict: Arc<[bool]> = verdict.into();
4865 if held.len() >= GRAM_VERDICTS {
4866 held.remove(0);
4867 }
4868 held.push((literal.to_vec(), Arc::clone(&verdict)));
4869 Ok(verdict)
4870 }
4871
4872 fn footprint(&self) -> usize {
4873 self.verdicts.lock().map_or(0, |held| {
4874 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4875 })
4876 }
4877}
4878
4879const TEXT_SEARCH_MEMO: usize = 64;
4884
4885const TEXT_PAYLOAD_VALUES: usize = 1024;
4901
4902const TEXT_GRAM_BYTES: usize = 8192;
4913
4914const NARROW_GRAM_BYTES: usize = 2048;
4916
4917const GRAM_WINDOW: usize = 256 << 10;
4919
4920fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4923 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4924 let mut first = original ^ (original >> 16);
4925 first = first.wrapping_mul(0x7feb_352d);
4926 first ^= first >> 15;
4927 let mut second = original ^ (original >> 17);
4928 second = second.wrapping_mul(0x846c_a68b);
4929 second ^= second >> 16;
4930 let mask = width * 8 - 1;
4931 [(first as usize) & mask, (second as usize) & mask]
4932}
4933
4934const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4955
4956#[derive(Debug)]
4963enum Lengths {
4964 Narrow(Vec<u16>),
4966 Wide(Vec<u32>),
4968}
4969
4970impl Lengths {
4971 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4974 match self {
4975 Lengths::Narrow(lens) => into.extend(
4976 indices
4977 .iter()
4978 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4979 ),
4980 Lengths::Wide(lens) => into.extend(
4981 indices
4982 .iter()
4983 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4984 ),
4985 }
4986 }
4987
4988 fn footprint(&self) -> usize {
4990 match self {
4991 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4992 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4993 }
4994 }
4995}
4996
4997fn lengths_of(ends: &[u32]) -> Option<Lengths> {
5006 match lengths_as::<u16>(ends)? {
5007 Some(narrow) => Some(Lengths::Narrow(narrow)),
5008 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
5009 }
5010}
5011
5012fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
5015 let mut lens = Vec::with_capacity(ends.len());
5016 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
5017 let mut start = 0;
5018 for &end in block {
5019 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
5020 return Some(None);
5021 };
5022 lens.push(len);
5023 start = end;
5024 }
5025 }
5026 Some(Some(lens))
5027}
5028
5029const TEXT_OFFSET_RUN: usize = 512;
5036
5037const DICTIONARY_HEADER: usize = 16;
5040
5041const DICTIONARY_SCATTERED: u32 = 1 << 31;
5055const DICTIONARY_GRAMS: u32 = 1 << 30;
5057const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
5060const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
5062
5063const TEXT_RANK_BLOCK: usize = 512;
5074
5075const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
5089
5090impl NativeText {
5091 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
5098 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
5099 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
5100 Ok(Some(bytes.as_slice()))
5101 }
5102
5103 fn block_chars(&self, block: usize) -> Result<&[u32]> {
5110 let slot = self
5111 .char_lens
5112 .get(block)
5113 .ok_or_else(|| invalid("a block past the global dictionary"))?;
5114 if let Some(lens) = slot.get() {
5115 return Ok(lens);
5116 }
5117 let decoded;
5118 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5119 Some(Ok(kept)) => kept,
5120 _ => {
5121 decoded = self.decode_block(block)?;
5122 &decoded
5123 }
5124 };
5125 let first = block * TEXT_PAYLOAD_VALUES;
5126 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5127 let ends = self.ends_within(first, last)?;
5128 if ends.len() != last - first {
5129 return Err(invalid("global dictionary offsets are short"));
5130 }
5131 let mut lens = Vec::with_capacity(ends.len());
5132 let mut start = u64::from(self.start_within(first)?);
5133 for &end in &ends {
5134 let value = usize::try_from(start)
5135 .ok()
5136 .zip(usize::try_from(end).ok())
5137 .and_then(|(from, to)| bytes.get(from..to))
5138 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5139 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5142 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5143 start = end;
5144 }
5145 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5146 }
5147
5148 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5153 let len = self.lengths[block];
5154 let mut stored = vec![
5155 0;
5156 usize::try_from(len).map_err(|_| invalid(
5157 "global dictionary block does not fit in memory"
5158 ))?
5159 ];
5160 read_at(&self.file, self.starts[block], &mut stored)?;
5161 if checksum(&stored) != self.hashes[block] {
5162 return Err(invalid("global dictionary payload checksum differs"));
5163 }
5164 let first = block * TEXT_PAYLOAD_VALUES;
5165 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5166 let want = self.end_within(last - 1)? as usize;
5167 let values = string::decode_flat(&stored)?;
5168 if values.len() != last - first {
5169 return Err(invalid("global dictionary block holds the wrong value count"));
5170 }
5171 let bytes = values.into_bytes();
5172 if bytes.len() != want {
5173 return Err(invalid("global dictionary block decodes to the wrong length"));
5174 }
5175 Ok(bytes)
5176 }
5177
5178 fn loaned_block<'a>(
5187 &'a self,
5188 block: usize,
5189 decoded: &'a mut Vec<u8>,
5190 scattered: bool,
5191 ) -> Result<&'a [u8]> {
5192 let kept = self.blocks.get(block).and_then(OnceLock::get);
5193 if let Some(Ok(kept)) = kept {
5194 return Ok(kept);
5195 }
5196 let again = kept.is_none()
5197 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5198 let keep = again
5199 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5200 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5201 if keep {
5202 let kept = self
5203 .payload_block(block)?
5204 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5205 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5206 return Ok(kept);
5207 }
5208 *decoded = self.decode_block(block)?;
5209 if scattered && again {
5210 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5211 }
5212 Ok(decoded)
5213 }
5214
5215 fn ends_worth_unpacking(&self) -> usize {
5232 self.values.max(TEXT_PAYLOAD_VALUES)
5233 }
5234
5235 fn value_ends(&self) -> Option<&[u32]> {
5237 if let Some(built) = self.value_ends.get() {
5238 return built.as_deref();
5239 }
5240 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5241 return None;
5242 }
5243 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5244 }
5245
5246 fn unpack_ends(&self) -> Option<Vec<u32>> {
5252 let mut ends = vec![0u32; self.values];
5253 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5254 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5255 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5256 u32::try_from(bits).unwrap_or(u32::MAX)
5257 })
5258 .ok()?;
5259 }
5260 if ends.contains(&u32::MAX) { None } else { Some(ends) }
5263 }
5264
5265 fn packed(&self) -> &[u8] {
5267 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5268 }
5269
5270 fn end_within(&self, index: usize) -> Result<u32> {
5272 if let Some(ends) = self.value_ends() {
5273 return ends
5274 .get(index)
5275 .copied()
5276 .ok_or_else(|| invalid("global dictionary offsets are short"));
5277 }
5278 self.packed_end(index)
5279 }
5280
5281 fn packed_end(&self, index: usize) -> Result<u32> {
5283 let run = index / TEXT_OFFSET_RUN;
5284 let bytes = self
5285 .packed()
5286 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5287 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5288 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5289 .map_err(|_| invalid("global dictionary offsets are short"))?;
5290 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5291 }
5292
5293 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5311 let mut ends = vec![0u64; last.saturating_sub(first)];
5312 let mut scratch = Vec::new();
5313 let mut at = first;
5314 while at < last {
5315 let run = at / TEXT_OFFSET_RUN;
5316 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5317 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5318 let bytes = self
5319 .packed()
5320 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5321 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5322 let from = at % TEXT_OFFSET_RUN;
5323 let upto = stop - run * TEXT_OFFSET_RUN;
5324 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5325 return Err(invalid("global dictionary offsets are short"));
5326 }
5327 let into = &mut ends[at - first..stop - first];
5328 if from == 0 {
5329 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5330 .map_err(|_| invalid("global dictionary offsets are short"))?;
5331 } else {
5332 scratch.resize(held, 0);
5333 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5334 .map_err(|_| invalid("global dictionary offsets are short"))?;
5335 into.copy_from_slice(&scratch[from..upto]);
5336 }
5337 at = stop;
5338 }
5339 Ok(ends)
5340 }
5341
5342 fn start_within(&self, index: usize) -> Result<u32> {
5345 if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5346 }
5347
5348 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5356 if let Some(ends) = self.value_ends() {
5357 let end =
5358 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5359 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5362 if start > end {
5363 return Err(invalid("global dictionary value ends before it starts"));
5364 }
5365 return Ok((start, end));
5366 }
5367 self.packed_span(index)
5368 }
5369
5370 fn packed_span(&self, index: usize) -> Result<(u32, u32)> {
5372 let within = index % TEXT_OFFSET_RUN;
5373 let (start, end) = if within == 0 {
5374 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) {
5375 0
5376 } else {
5377 self.packed_end(index - 1)?
5378 };
5379 (start, self.packed_end(index)?)
5380 } else {
5381 let run = index / TEXT_OFFSET_RUN;
5382 let bytes = self
5383 .packed()
5384 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5385 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5386 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5387 .map_err(|_| invalid("global dictionary offsets are short"))?;
5388 let ends = u32::try_from(end)
5389 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5390 let starts = u32::try_from(start)
5391 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5392 (starts, ends)
5393 };
5394 if start > end {
5395 return Err(invalid("global dictionary value ends before it starts"));
5396 }
5397 Ok((start, end))
5398 }
5399
5400 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5407 let slot = self
5408 .rank_blocks
5409 .get(rank / TEXT_RANK_BLOCK)
5410 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5411 let block = slot
5412 .get_or_init(|| {
5413 let mut bytes = Vec::new();
5414 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5415 Ok(bytes)
5416 })
5417 .as_ref()
5418 .map_err(Clone::clone)?;
5419 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5420 }
5421
5422 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5425 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5426 let end = self.rank_ends[which];
5427 bytes.clear();
5428 bytes.resize((end - start) as usize, 0);
5429 read_at(&self.file, self.rank_at + start, bytes)?;
5430 let expected = self
5431 .rank_hashes
5432 .get(which)
5433 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5434 if checksum(bytes) != *expected {
5435 return Err(invalid("global dictionary rank checksum differs"));
5436 }
5437 Ok(())
5438 }
5439
5440 fn head_at(&self, rank: usize) -> Result<u64> {
5442 let (block, within) = self.rank_parts(rank)?;
5443 let (base, width, packed) = rank_heads(block)?;
5444 let above = bitpack::tail_at(packed, width, within)
5445 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5446 Ok(base.wrapping_add(above))
5447 }
5448
5449 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5451 let (_, width, packed) = rank_heads(block)?;
5452 packed
5453 .get(bitpack::tail_len(count, width)..)
5454 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5455 }
5456
5457 fn rank_block_len(&self, rank: usize) -> usize {
5459 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5460 TEXT_RANK_BLOCK.min(self.ranks - first)
5461 }
5462}
5463
5464fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5466 let header = block
5467 .get(..RANK_BLOCK_HEADER)
5468 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5469 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5470 let width = header[8] as usize;
5471 if width > 64 {
5472 return Err(invalid("global dictionary rank block packs heads past a word"));
5473 }
5474 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5475}
5476
5477fn offset_width(ends: &[u32]) -> usize {
5484 let span = ends.iter().copied().max().unwrap_or(0);
5488 (u32::BITS - span.leading_zeros()) as usize
5489}
5490
5491fn offset_bytes(values: usize, bits: usize) -> usize {
5494 let full = values / TEXT_OFFSET_RUN;
5495 let rest = values % TEXT_OFFSET_RUN;
5496 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5497}
5498
5499fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5503 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5504 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5505 run.clear();
5506 run.extend(chunk.iter().map(|&end| u64::from(end)));
5507 bitpack::pack_tail(&run, bits, out)
5508 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5509 }
5510 Ok(())
5511}
5512
5513fn code_width(values: usize) -> usize {
5515 match u64::try_from(values).unwrap_or(u64::MAX) {
5516 0 | 1 => 0,
5517 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5518 }
5519}
5520
5521impl TextSource for NativeText {
5522 fn len(&self) -> usize {
5523 self.values
5524 }
5525
5526 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5527 let Some(grams) = &self.grams else { return Ok(true) };
5528 if literal.len() < 4 || first >= self.values {
5529 return Ok(true);
5530 }
5531 let verdict = grams.verdicts(&self.file, literal)?;
5532 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5533 }
5534
5535 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5536 if index >= self.values {
5537 return Ok(None);
5538 }
5539 let (start, end) = self.span_within(index)?;
5540 if start == end {
5541 return Ok(Some(&[]));
5542 }
5543 let block = index / TEXT_PAYLOAD_VALUES;
5546 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5547 Ok(bytes.get(start as usize..end as usize))
5548 }
5549
5550 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5551 if index >= self.values {
5552 return Ok(None);
5553 }
5554 let (start, end) = self.span_within(index)?;
5555 Ok(Some((end - start) as usize))
5556 }
5557
5558 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5565 into.reserve(indices.len());
5566 if let Some(Some(lens)) = self.value_lens.get() {
5569 lens.extend_at(indices, into);
5570 return Ok(());
5571 }
5572 let asked = self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed) + indices.len();
5577 let lens = match self.value_ends.get() {
5578 Some(Some(ends)) => self.value_lens.get_or_init(|| lengths_of(ends)).as_ref(),
5579 _ if asked >= self.ends_worth_unpacking() => self
5580 .value_lens
5581 .get_or_init(|| self.unpack_ends().and_then(|ends| lengths_of(&ends)))
5582 .as_ref(),
5583 _ => None,
5584 };
5585 if let Some(lens) = lens {
5586 lens.extend_at(indices, into);
5587 return Ok(());
5588 }
5589 for &index in indices {
5590 let index = index as usize;
5591 if index >= self.values {
5593 into.push(0);
5594 continue;
5595 }
5596 let (start, end) = self.packed_span(index)?;
5597 into.push(i64::from(end - start));
5598 }
5599 Ok(())
5600 }
5601
5602 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5605 into.reserve(indices.len());
5606 for &index in indices {
5607 let index = index as usize;
5608 if index >= self.values {
5610 into.push(0);
5611 continue;
5612 }
5613 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5614 let len = lens
5615 .get(index % TEXT_PAYLOAD_VALUES)
5616 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5617 into.push(i64::from(*len));
5618 }
5619 Ok(())
5620 }
5621
5622 fn sweep(
5635 &self,
5636 first: usize,
5637 limit: usize,
5638 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5639 ) -> Result<usize> {
5640 let limit = limit.min(self.values);
5641 if first >= limit {
5642 return Ok(first);
5643 }
5644 let block = first / TEXT_PAYLOAD_VALUES;
5645 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5646 let mut decoded = Vec::new();
5647 let bytes = self.loaned_block(block, &mut decoded, false)?;
5648 let ends = self.ends_within(first, last)?;
5649 if ends.len() != last - first {
5650 return Err(invalid("global dictionary offsets are short"));
5651 }
5652 let mut start = u64::from(self.start_within(first)?);
5653 for (index, &end) in (first..last).zip(&ends) {
5656 let value = usize::try_from(start)
5657 .ok()
5658 .zip(usize::try_from(end).ok())
5659 .and_then(|(from, to)| bytes.get(from..to))
5660 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5661 body(index, value)?;
5662 start = end;
5663 }
5664 Ok(last)
5665 }
5666
5667 fn visit_at(
5676 &self,
5677 indices: &[u32],
5678 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5679 ) -> Result<()> {
5680 let mut order = (0..indices.len()).collect::<Vec<_>>();
5681 order.sort_unstable_by_key(|&at| indices[at]);
5682 let block_of = |at: usize| {
5683 let index = indices[at] as usize;
5684 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5685 };
5686 let mut decoded = Vec::new();
5687 let mut run = 0;
5688 while run < order.len() {
5689 let Some(block) = block_of(order[run]) else {
5690 for &at in &order[run..] {
5692 body(at, &[])?;
5693 }
5694 break;
5695 };
5696 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5697 let bytes = self.loaned_block(block, &mut decoded, true)?;
5698 for &at in &order[run..upto] {
5699 let (start, end) = self.span_within(indices[at] as usize)?;
5700 let value = bytes
5701 .get(start as usize..end as usize)
5702 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5703 body(at, value)?;
5704 }
5705 run = upto;
5706 }
5707 Ok(())
5708 }
5709
5710 fn visit(
5716 &self,
5717 indices: &[usize],
5718 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5719 ) -> Result<()> {
5720 let mut at = 0;
5721 while at < indices.len() {
5722 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5723 let upto =
5724 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5725 let wanted = &indices[at..upto];
5726 if wanted.iter().any(|&index| index >= self.values) {
5727 return Err(invalid("a visited value is past the global dictionary"));
5728 }
5729 let decoded;
5730 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5731 Some(Ok(kept)) => kept,
5732 _ => {
5733 decoded = self.decode_block(block)?;
5734 &decoded
5735 }
5736 };
5737 for (offset, &index) in wanted.iter().enumerate() {
5738 let (start, end) = self.span_within(index)?;
5739 let value = bytes
5740 .get(start as usize..end as usize)
5741 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5742 body(at + offset, value)?;
5743 }
5744 at = upto;
5745 }
5746 Ok(())
5747 }
5748
5749 fn ranks(&self) -> Option<usize> {
5750 (self.ranks > 0).then_some(self.ranks)
5751 }
5752
5753 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5761 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5762 if let Some(&answer) = memo.get(wanted) {
5763 return Ok(answer);
5764 }
5765 let answer = search_below(self, ranks, wanted)?;
5766 if memo.len() >= TEXT_SEARCH_MEMO {
5767 memo.clear();
5768 }
5769 memo.insert(wanted.to_vec(), answer);
5770 Ok(answer)
5771 }
5772
5773 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5774 let settled = self.head_at(rank)?.cmp(&head(wanted));
5778 if settled != Ordering::Equal {
5779 return Ok(settled);
5780 }
5781 let code = self.code_at_rank(rank)?;
5782 let bytes = self
5783 .bytes_at(code as usize)?
5784 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5785 Ok(bytes.cmp(wanted))
5786 }
5787
5788 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5789 let (block, within) = self.rank_parts(rank)?;
5790 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5791 let code = bitpack::tail_at(codes, self.code_bits, within)
5792 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5793 let code = u32::try_from(code)
5794 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5795 if code as usize >= self.len() {
5796 return Err(invalid("global dictionary order names a code it does not have"));
5797 }
5798 Ok(code)
5799 }
5800
5801 fn code_ranks(&self) -> Option<&[u32]> {
5802 if self.ranks == 0 || self.ranks != self.len() {
5806 return None;
5807 }
5808 self.code_ranks
5809 .get_or_init(|| {
5810 let mut ranks = vec![u32::MAX; self.ranks];
5811 let mut scratch = Vec::new();
5819 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5820 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5821 let which = first / TEXT_RANK_BLOCK;
5822 let block = match self.rank_blocks.get(which)?.get() {
5823 Some(kept) => kept.as_ref().ok()?.as_slice(),
5824 None => {
5825 self.read_rank_block(which, &mut scratch).ok()?;
5826 scratch.as_slice()
5827 }
5828 };
5829 let count = self.rank_block_len(first);
5830 let packed = self.rank_codes(block, count).ok()?;
5831 let codes = codes.get_mut(..count)?;
5832 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5833 for (within, &code) in codes.iter().enumerate() {
5834 let code = usize::try_from(code).ok()?;
5835 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5836 }
5837 }
5838 if ranks.contains(&u32::MAX) {
5839 return None;
5840 }
5841 Some(ranks)
5842 })
5843 .as_deref()
5844 }
5845
5846 fn footprint(&self) -> usize {
5847 self.offsets.capacity()
5848 + self
5849 .value_ends
5850 .get()
5851 .and_then(Option::as_ref)
5852 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5853 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5854 + self
5855 .code_ranks
5856 .get()
5857 .and_then(Option::as_ref)
5858 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5859 + self.rank_hashes.capacity() * size_of::<u64>()
5860 + self.rank_ends.capacity() * size_of::<u64>()
5861 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5862 + self
5863 .rank_blocks
5864 .iter()
5865 .filter_map(OnceLock::get)
5866 .filter_map(|result| result.as_ref().ok())
5867 .map(Vec::capacity)
5868 .sum::<usize>()
5869 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5870 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5871 + self
5872 .char_lens
5873 .iter()
5874 .filter_map(OnceLock::get)
5875 .map(|lens| lens.len() * size_of::<u32>())
5876 .sum::<usize>()
5877 + self.hashes.capacity() * size_of::<u64>()
5878 + self.starts.capacity() * size_of::<u64>()
5879 + self.lengths.capacity() * size_of::<u64>()
5880 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5881 + self
5882 .blocks
5883 .iter()
5884 .filter_map(OnceLock::get)
5885 .filter_map(|result| result.as_ref().ok())
5886 .map(Vec::capacity)
5887 .sum::<usize>()
5888 }
5889}
5890
5891fn places(table: &Table) -> Result<Vec<Place>> {
5893 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5894 for (at, stripe) in table.stripes.iter().enumerate() {
5895 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5896 for (part, &rows) in stripe.parts.iter().enumerate() {
5897 places.push(Place {
5898 stripe: index,
5899 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5900 rows,
5901 });
5902 }
5903 }
5904 Ok(places)
5905}
5906
5907fn read_index<F: Positional + ?Sized>(
5912 file: &F,
5913 stripe: &Stripe,
5914 column: usize,
5915) -> Result<Vec<PartSpan>> {
5916 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5917 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5918}
5919
5920fn read_index_span<F: Positional + ?Sized>(
5921 file: &F,
5922 index: Span,
5923 page: Span,
5924 parts: usize,
5925 column: usize,
5926) -> Result<Vec<PartSpan>> {
5927 let section = index_section(parts)?;
5928 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5929 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5930 if end > index.length as usize {
5931 return Err(invalid("index page is shorter than its columns"));
5932 }
5933 let mut bytes = vec![0; section];
5934 let offset =
5935 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5936 read_at(file, offset, &mut bytes)?;
5937 let entries = section - size_of::<u64>();
5938 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5939 if checksum(&bytes[..entries]) != stored {
5940 return Err(invalid(&format!(
5943 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5944 wanted {stored:016x} and got {:016x}",
5945 checksum(&bytes[..entries]),
5946 )));
5947 }
5948 let mut spans = Vec::with_capacity(parts);
5949 let mut start = 0_usize;
5950 for part in 0..parts {
5951 let at = part * INDEX_ENTRY;
5952 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5953 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5954 spans.push(PartSpan { start, length, hash });
5955 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5956 }
5957 if start != page.length as usize {
5958 return Err(invalid("column page length differs from its index"));
5959 }
5960 Ok(spans)
5961}
5962
5963fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5965 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5966 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5967}
5968
5969fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5972 if bits.is_empty() {
5973 bits.resize(parts.div_ceil(64).max(1), 0);
5974 }
5975 let (word, bit) = (part / 64, 1_u64 << (part % 64));
5976 let Some(held) = bits.get_mut(word) else { return (false, false) };
5977 let again = *held & bit != 0;
5978 *held |= bit;
5979 let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5980 (again, again && through)
5981}
5982
5983fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5989 if let Some(slot) = cached.index.get_mut(held.stripe)
5990 && slot.is_none()
5991 {
5992 *slot = Some(Arc::clone(&held.index));
5993 }
5994 let page = held.page.clone()?;
5995 let slot = cached.pages.get_mut(held.stripe)?;
5996 if slot.is_some() {
5997 return None;
5998 }
5999 let bytes = page.bytes().len();
6000 let used = Arc::new(AtomicBool::new(true));
6003 *slot = Some(Resident { page, used: Arc::clone(&used) });
6004 Some((bytes, used))
6005}
6006
6007#[derive(Debug, Clone)]
6016pub struct Catalog {
6017 file: Arc<File>,
6018 size: u64,
6019 map: Option<Arc<Mapped>>,
6022 entries: Arc<Vec<Entry>>,
6023 views: Arc<Vec<ViewEntry>>,
6025 anchor: Option<Arc<LogAnchor>>,
6027 opening: Opening,
6028 pool: PagePool,
6030}
6031
6032#[derive(Debug, Clone, PartialEq, Eq)]
6034pub struct CertifiedSums {
6035 pub columns: Vec<(i128, u64)>,
6036 pub rows: u64,
6037}
6038
6039#[derive(Debug, Clone, Copy, PartialEq, Eq)]
6041pub enum IntegerExtremes {
6042 Null,
6043 Values { low: i128, high: i128 },
6044}
6045
6046pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
6048
6049impl Catalog {
6050 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6059 Self::open_in(path, &PagePool::default())
6060 }
6061
6062 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
6068 let path = path.as_ref();
6069 let (file, size, _, bytes, opening) = slot_bytes(path)?;
6070 let (entries, views, card, anchor) = decode_catalog(&bytes, size)?;
6071 remember_card(path, card.as_ref());
6072 let map = Mapped::open(&file, size).map(Arc::new);
6073 Ok(Self {
6074 anchor: anchor.map(Arc::new),
6075 file: Arc::new(file),
6076 size,
6077 map,
6078 entries: Arc::new(entries),
6079 views: Arc::new(views),
6080 opening,
6081 pool: pool.clone(),
6082 })
6083 }
6084
6085 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
6087 self.entries.iter().map(|entry| entry.name.as_str())
6088 }
6089
6090 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
6097 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
6098 }
6099
6100 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
6106 self.views.iter()
6107 }
6108
6109 #[must_use]
6111 pub fn log_anchor(&self) -> Option<&LogAnchor> {
6112 self.anchor.as_deref()
6113 }
6114
6115 #[must_use]
6117 pub fn len(&self) -> usize {
6118 self.entries.len()
6119 }
6120
6121 #[must_use]
6124 pub fn is_empty(&self) -> bool {
6125 self.entries.is_empty()
6126 }
6127
6128 pub fn table(&self, name: &str) -> Result<Reader> {
6134 let entry = self
6135 .entries
6136 .iter()
6137 .find(|entry| entry.name == name)
6138 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6139 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6143 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6144 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6145 }
6146 let mut opening = self.opening;
6147 opening.reads += 1;
6148 opening.bytes += u64::from(entry.directory.length);
6149 Reader::build(
6150 Arc::clone(&self.file),
6151 self.map.clone(),
6152 self.size,
6153 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6154 u64::from(entry.directory.length),
6155 opening,
6156 self.pool.clone(),
6157 )
6158 }
6159
6160 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6168 let mut counts = BTreeMap::<i64, u64>::new();
6169 let Some(()) = self.integer_fold(name, column, |value, count| {
6170 let held = counts.entry(value).or_default();
6171 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6172 Ok(())
6173 })?
6174 else {
6175 return Ok(None);
6176 };
6177 Ok(Some(counts.into_iter().collect()))
6178 }
6179
6180 pub fn integer_fold(
6187 &self,
6188 name: &str,
6189 column: usize,
6190 mut emit: impl FnMut(i64, u64) -> Result<()>,
6191 ) -> Result<Option<()>> {
6192 let entry = self
6193 .entries
6194 .iter()
6195 .find(|entry| entry.name == name)
6196 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6197 let field =
6198 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6199 if !signed_integer(&field.ty) {
6200 return Ok(None);
6201 }
6202 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6203 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6204 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6205 }
6206 quick_integer_fold(
6207 &self.file,
6208 Cursor::over(&self.file, offset, length),
6209 entry,
6210 self.size,
6211 column,
6212 &mut emit,
6213 )?;
6214 Ok(Some(()))
6215 }
6216
6217 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6222 let entry = self
6223 .entries
6224 .iter()
6225 .find(|entry| entry.name == name)
6226 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6227 let Some(field) = entry.fields.get(column) else {
6228 return Err(invalid("frequency column index out of range"));
6229 };
6230 if !matches!(
6231 field.ty,
6232 LogicalType::TinyInt
6233 | LogicalType::SmallInt
6234 | LogicalType::Integer
6235 | LogicalType::BigInt
6236 | LogicalType::UTinyInt
6237 | LogicalType::USmallInt
6238 | LogicalType::UInteger
6239 | LogicalType::UBigInt
6240 ) {
6241 return Ok(None);
6242 }
6243 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6244 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6245 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6246 }
6247 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6248 return frequencies
6249 .iter()
6250 .filter(|(value, _)| value.is_some_and(|value| value != 0))
6251 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6252 .map(Some)
6253 .ok_or_else(|| invalid("numeric frequency count overflow"));
6254 }
6255 quick_nonzero(
6256 Cursor::over(&self.file, offset, length),
6257 &entry.name,
6258 &entry.fields,
6259 entry.rows,
6260 column,
6261 )
6262 }
6263
6264 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6267 let entry = self
6268 .entries
6269 .iter()
6270 .find(|entry| entry.name == name)
6271 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6272 let mut sums = Vec::with_capacity(columns.len());
6273 for &column in columns {
6274 let Some(field) = entry.fields.get(column) else {
6275 return Err(invalid("aggregate column index out of range"));
6276 };
6277 if !signed_integer(&field.ty) {
6278 return Ok(None);
6279 }
6280 let Some(sum) = entry.aggregates[column] else {
6281 return Ok(None);
6282 };
6283 sums.push(sum);
6284 }
6285 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6286 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6287 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6288 }
6289 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6290 }
6291
6292 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6294 let entry = self
6295 .entries
6296 .iter()
6297 .find(|entry| entry.name == name)
6298 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6299 let Some(count) = entry.distincts.get(column).copied() else {
6300 return Err(invalid("distinct column index out of range"));
6301 };
6302 let Some(count) = count else { return Ok(None) };
6303 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6304 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6305 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6306 }
6307 Ok(Some(count))
6308 }
6309
6310 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6312 let entry = self
6313 .entries
6314 .iter()
6315 .find(|entry| entry.name == name)
6316 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6317 let Some(extremes) = entry.extremes.get(column).copied() else {
6318 return Err(invalid("extremes column index out of range"));
6319 };
6320 let Some(extremes) = extremes else { return Ok(None) };
6321 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6322 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6323 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6324 }
6325 Ok(Some(match extremes {
6326 None => IntegerExtremes::Null,
6327 Some((low, high)) => IntegerExtremes::Values { low, high },
6328 }))
6329 }
6330
6331 pub fn exact_numeric_frequencies(
6333 &self,
6334 name: &str,
6335 column: usize,
6336 ) -> Result<Option<NumericFrequencies>> {
6337 let entry = self
6338 .entries
6339 .iter()
6340 .find(|entry| entry.name == name)
6341 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6342 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6343 return Err(invalid("numeric frequency column index out of range"));
6344 };
6345 let Some(frequencies) = frequencies else { return Ok(None) };
6346 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6347 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6348 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6349 }
6350 Ok(Some(frequencies))
6351 }
6352
6353 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6355 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6356 }
6357}
6358
6359fn slot_offset(generation: u64) -> u64 {
6364 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6365}
6366
6367fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6372 let file = File::open(path).map_err(io)?;
6373 let size = file.metadata().map_err(io)?.len();
6374 let (slot, bytes, opening) = committed_slot(&file, size)?;
6375 Ok((file, size, slot, bytes, opening))
6376}
6377
6378fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6384 if size < HEADER {
6385 return Err(invalid("file is shorter than its header"));
6386 }
6387 let mut header = [0; HEADER as usize];
6388 read_at(file, 0, &mut header)?;
6389 let mut opening = Opening { reads: 1, bytes: HEADER };
6390 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6391 if &header[..8] != MAGIC {
6396 return Err(invalid("the header does not begin with a rudb native magic"));
6397 }
6398 if !READABLE.contains(&version) {
6399 return Err(invalid(&format!(
6400 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6401 be written again"
6402 )));
6403 }
6404 let mut selected = None;
6405 for start in [16, 16 + SLOT_BYTES] {
6406 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6407 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6408 continue;
6409 }
6410 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6411 if slot.offset < HEADER || end > size {
6412 continue;
6413 }
6414 let mut bytes = vec![0; slot.length as usize];
6415 read_at(file, slot.offset, &mut bytes)?;
6416 opening.reads += 1;
6417 opening.bytes += u64::from(slot.length);
6418 if checksum(&bytes) == slot.hash
6419 && selected
6420 .as_ref()
6421 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6422 {
6423 selected = Some((slot, bytes));
6424 }
6425 }
6426 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6427 Ok((slot, bytes, opening))
6428}
6429
6430impl Reader {
6431 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6438 let catalog = Catalog::open(path)?;
6439 let mut names = catalog.names();
6440 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6441 if names.next().is_some() {
6442 return Err(invalid(
6443 "the file holds more than one table, so it has to be opened by name",
6444 ));
6445 }
6446 catalog.table(&name)
6447 }
6448
6449 fn build(
6451 file: Arc<File>,
6452 map: Option<Arc<Mapped>>,
6453 size: u64,
6454 table: Table,
6455 directory: u64,
6456 opening: Opening,
6457 pool: PagePool,
6458 ) -> Result<Self> {
6459 let places = places(&table)?;
6460 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6461 let table_fields = table.fields.len();
6462 let columns = (0..table_fields).map(|_| Mutex::new(Cached::default())).collect::<Vec<_>>();
6463 let cache = Shelf {
6464 columns,
6465 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6466 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6467 };
6468 let sieves = (0..table_fields).map(|_| OnceLock::new()).collect();
6469 let part_ranges = (0..table_fields).map(|_| OnceLock::new()).collect();
6470 let verified = (places.len() * table_fields).div_ceil(64);
6471 let unreleased = table
6472 .stripes
6473 .iter()
6474 .flat_map(|stripe| {
6475 let parts = u32::try_from(stripe.parts.len()).unwrap_or(u32::MAX);
6476 (0..table_fields).map(move |_| AtomicU32::new(parts))
6477 })
6478 .collect();
6479 let firsts = places
6480 .iter()
6481 .scan(0, |first, place| {
6482 let at = *first;
6483 *first += place.rows as usize;
6484 Some(at)
6485 })
6486 .collect();
6487 Ok(Self {
6488 file,
6489 map,
6490 table: Arc::new(table),
6491 dictionaries: Arc::new(dictionaries),
6492 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6493 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6494 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6495 frequency_heads: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6496 summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6497 opened: Arc::new(AtomicUsize::new(0)),
6498 sieves: Arc::new(sieves),
6499 part_ranges: Arc::new(part_ranges),
6500 places: Arc::new(places),
6501 cache: Arc::new(cache),
6502 pool,
6503 pages: Arc::new(AtomicUsize::new(0)),
6504 indexes: Arc::new(AtomicUsize::new(0)),
6505 verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6506 unreleased: Arc::new(unreleased),
6507 text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6508 firsts: Arc::new(firsts),
6509 graph: Arc::default(),
6510 size,
6511 directory,
6512 opening,
6513 })
6514 }
6515
6516 #[must_use]
6523 pub fn reads(&self) -> Reads {
6524 Reads {
6525 opening: self.opening,
6526 pages: self.pages.load(Atomic::Relaxed),
6527 indexes: self.indexes.load(Atomic::Relaxed),
6528 dictionaries: self.opened.load(Atomic::Relaxed),
6529 }
6530 }
6531
6532 #[must_use]
6537 pub fn layout(&self) -> Layout {
6538 let table = &self.table;
6539 let stripes = table.stripes.as_slice();
6540 let columns = table
6541 .fields
6542 .iter()
6543 .enumerate()
6544 .map(|(at, field)| ColumnLayout {
6545 name: field.name.clone(),
6546 kind: field.ty.to_string(),
6547 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6548 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6549 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6550 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6551 dictionary: dictionary_bytes(table, at),
6552 })
6553 .collect();
6554 Layout {
6555 file: self.size,
6556 rows: table.rows,
6557 stripes: stripes.len(),
6558 parts: self.places.len(),
6559 columns,
6560 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6561 directory: self.directory,
6562 header: HEADER,
6563 }
6564 }
6565
6566 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6583 let field = self
6584 .table
6585 .fields
6586 .get(column)
6587 .ok_or_else(|| invalid("stored column index out of range"))?;
6588 let mut stored = Vec::with_capacity(self.places.len());
6589 let mut row = 0;
6590 for (at, stripe) in self.table.stripes.iter().enumerate() {
6591 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6592 let index = read_index(&self.file, stripe, column)?;
6593 let mut bytes = vec![0; page.length as usize];
6594 read_at(&self.file, page.offset, &mut bytes)?;
6595 let ranges = self.stripe_part_ranges(at, column);
6596 for (part, &rows) in stripe.parts.iter().enumerate() {
6597 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6598 let held = part_bytes(&bytes, span)?;
6599 let range = ranges.and_then(|held| held.get(part));
6600 stored.push(StoredPart {
6601 stripe: at,
6602 part,
6603 row,
6604 rows: rows as usize,
6605 encoding: page_encoding(&field.ty, rows as usize, held),
6606 bytes: span.length as u64,
6607 page: page.offset,
6608 offset: span.start as u64,
6609 low: range
6610 .and_then(|range| range.low.clone())
6611 .and_then(|bound| bound.into_value(&field.ty)),
6612 high: range
6613 .and_then(|range| range.high.clone())
6614 .and_then(|bound| bound.into_value(&field.ty)),
6615 nulls: range.map(|range| range.nulls),
6616 });
6617 row += rows as usize;
6618 }
6619 }
6620 Ok(stored)
6621 }
6622
6623 #[must_use]
6625 pub fn parts(&self) -> usize {
6626 self.places.len()
6627 }
6628
6629 #[must_use]
6636 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6637 let mut runs = Vec::with_capacity(self.table.stripes.len());
6638 let mut start = 0;
6639 for stripe in &self.table.stripes {
6640 let end = start + stripe.parts.len();
6641 runs.push(start..end);
6642 start = end;
6643 }
6644 runs
6645 }
6646
6647 #[must_use]
6652 pub fn stripe_rows(&self, stripe: usize) -> usize {
6653 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6654 }
6655
6656 pub fn keep_stripes(&self, stripes: usize) {
6663 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6664 }
6665
6666 #[must_use]
6668 pub fn part_rows(&self, at: usize) -> usize {
6669 self.places.get(at).map_or(0, |place| place.rows as usize)
6670 }
6671
6672 #[must_use]
6674 pub fn table(&self) -> &Table {
6675 &self.table
6676 }
6677
6678 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6687 let field = self
6688 .table
6689 .fields
6690 .get(column)
6691 .ok_or_else(|| invalid("frequency column index out of range"))?;
6692 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6693 return Ok(None);
6694 };
6695 if top == 0 || entries.len() < top {
6696 return Ok(None);
6697 }
6698 let boundary = entries[top - 1].count;
6699 if boundary <= omitted_max {
6700 return Ok(None);
6701 }
6702 self.decode_frequencies(column, &field.ty, &entries).map(|values| Some(Vec::clone(&values)))
6703 }
6704
6705 pub fn top_pair_frequencies(
6713 &self,
6714 first: usize,
6715 second: usize,
6716 _top: usize,
6717 ) -> Result<Option<PairFrequencyCounts>> {
6718 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6719 return Err(invalid("pair frequency column index out of range"));
6720 }
6721 Ok(None)
6722 }
6723
6724 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6744 let Some(prefix) = self.frequency_prefix(column)? else {
6745 return Ok(None);
6746 };
6747 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6748 }
6749
6750 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6773 Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6774 entries: Vec::clone(&entries),
6775 omitted_max,
6776 }))
6777 }
6778
6779 pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6785 let field = self
6786 .table
6787 .fields
6788 .get(column)
6789 .ok_or_else(|| invalid("frequency column index out of range"))?;
6790 let Some((entries, omitted_max)) = self.frequency_head(column)? else {
6791 return Ok(None);
6792 };
6793 let entries = self.decode_frequencies(column, &field.ty, &entries)?;
6794 Ok(Some((entries, omitted_max)))
6795 }
6796
6797 fn frequency_head(&self, column: usize) -> Result<Option<(Cow<'_, [FrequencyEntry]>, u64)>> {
6800 let (span, count) = match self.table.frequencies.get(column) {
6801 None | Some(None) => return Ok(None),
6802 Some(Some(Frequencies::Held(summary))) => {
6803 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6804 }
6805 Some(Some(Frequencies::Stored { span, entries, .. })) => (span, *entries),
6806 };
6807 if let Some(summary) = self.frequency_summaries.get(column).and_then(OnceLock::get) {
6808 return Ok(Some((Cow::Borrowed(&summary.entries), summary.omitted_max)));
6809 }
6810 let slot = self
6811 .frequency_heads
6812 .get(column)
6813 .ok_or_else(|| invalid("frequency column index out of range"))?;
6814 if slot.get().is_none() {
6815 let field = self
6816 .table
6817 .fields
6818 .get(column)
6819 .ok_or_else(|| invalid("frequency column index out of range"))?;
6820 let length = (span.length as usize).min(13 + count * 25);
6823 let mut bytes = vec![0; length];
6824 read_at(&self.file, span.offset, &mut bytes)?;
6825 let head = decode_summary_head(&mut Cursor::new(&bytes), field, self.table.rows)?
6826 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
6827 if head.0.len() != count {
6828 return Err(invalid("a stored synopsis differs from its directory span"));
6829 }
6830 let _ = slot.set(Arc::new(head));
6831 }
6832 let (entries, omitted_max) = slot.get().expect("the synopsis head was stored").as_ref();
6833 Ok(Some((Cow::Borrowed(entries), *omitted_max)))
6834 }
6835
6836 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6838 Ok(match self.table.frequencies.get(column) {
6839 None | Some(None) => None,
6840 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6841 Some(Some(Frequencies::Stored { span, values, entries })) => {
6842 let slot = self
6843 .frequency_summaries
6844 .get(column)
6845 .ok_or_else(|| invalid("frequency column index out of range"))?;
6846 if let Some(summary) = slot.get() {
6847 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6848 }
6849 let field = self
6850 .table
6851 .fields
6852 .get(column)
6853 .ok_or_else(|| invalid("frequency column index out of range"))?;
6854 let mut bytes = vec![0; span.length as usize];
6855 read_at(&self.file, span.offset, &mut bytes)?;
6856 let mut cur = Cursor::new(&bytes);
6857 let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6858 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6859 if !cur.done() || summary.entries.len() != *entries {
6860 return Err(invalid("a stored synopsis differs from its directory span"));
6861 }
6862 let _ = slot.set(Arc::new(summary));
6863 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6864 }
6865 })
6866 }
6867
6868 fn decode_frequencies(
6876 &self,
6877 column: usize,
6878 ty: &LogicalType,
6879 entries: &[FrequencyEntry],
6880 ) -> Result<Synopsis> {
6881 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6882 return Ok(Arc::clone(values));
6883 }
6884 let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6885 if let Some(slot) = self.frequency_values.get(column) {
6886 let _ = slot.set(Arc::clone(&values));
6887 }
6888 Ok(values)
6889 }
6890
6891 fn decode_frequencies_once(
6892 &self,
6893 column: usize,
6894 ty: &LogicalType,
6895 entries: &[FrequencyEntry],
6896 ) -> Result<Vec<(Value, u64)>> {
6897 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6898 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6899 return Err(invalid("frequency text count differs from its synopsis"));
6900 }
6901 let dictionary =
6902 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6903 let mut codes = entries
6904 .iter()
6905 .filter_map(|entry| match entry.value {
6906 FrequencyValue::Code(code) => Some(code as usize),
6907 _ => None,
6908 })
6909 .collect::<Vec<_>>();
6910 codes.sort_unstable();
6911 codes.dedup();
6912 let texts = match &dictionary {
6913 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6914 _ => Vec::new(),
6915 };
6916 let mut out = Vec::with_capacity(entries.len());
6917 for (entry_at, entry) in entries.iter().enumerate() {
6918 let value = match entry.value {
6919 FrequencyValue::Null => {
6920 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6921 return Err(invalid("a null frequency entry has text"));
6922 }
6923 Value::Null
6924 }
6925 FrequencyValue::Integer(value) => match *ty {
6926 LogicalType::TinyInt => Value::TinyInt(
6927 i8::try_from(value)
6928 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6929 ),
6930 LogicalType::UTinyInt => Value::UTinyInt(
6931 u8::try_from(value)
6932 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6933 ),
6934 LogicalType::USmallInt => Value::USmallInt(
6935 u16::try_from(value)
6936 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6937 ),
6938 LogicalType::UInteger => Value::UInteger(
6939 u32::try_from(value)
6940 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6941 ),
6942 LogicalType::UBigInt => Value::UBigInt(
6943 u64::try_from(value)
6944 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6945 ),
6946 LogicalType::SmallInt => Value::SmallInt(
6947 i16::try_from(value)
6948 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6949 ),
6950 LogicalType::Integer => Value::Integer(
6951 i32::try_from(value)
6952 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6953 ),
6954 LogicalType::BigInt => Value::BigInt(
6955 i64::try_from(value)
6956 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6957 ),
6958 LogicalType::Date => Value::Date(
6959 i32::try_from(value)
6960 .map_err(|_| invalid("frequency DATE is out of range"))?,
6961 ),
6962 LogicalType::Timestamp => Value::Timestamp(
6963 i64::try_from(value)
6964 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6965 ),
6966 _ => return Err(invalid("integer frequency belongs to another type")),
6967 },
6968 FrequencyValue::Code(code) => {
6969 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6970 if *ty == LogicalType::Blob {
6971 Value::Blob(text.clone())
6972 } else {
6973 Value::Varchar(
6974 String::from_utf8(text.clone())
6975 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6976 )
6977 }
6978 } else {
6979 if dictionary.is_none() {
6980 return Err(invalid("frequency code has no dictionary or stored text"));
6981 }
6982 let at = codes
6983 .binary_search(&(code as usize))
6984 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6985 texts[at].clone()
6986 }
6987 }
6988 };
6989 out.push((value, entry.count));
6990 }
6991 Ok(out)
6992 }
6993
6994 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
7004 let field = self
7005 .table
7006 .fields
7007 .get(column)
7008 .ok_or_else(|| invalid("frequency column index out of range"))?;
7009 let Some(summary) = self.frequency_summary(column)? else {
7010 return Ok(None);
7011 };
7012 if summary.ordinals.is_empty() {
7013 return Ok(None);
7014 }
7015 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
7016 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
7017 (
7018 entries.iter().map(|(value, _)| value.clone()).collect(),
7019 summary.ordinal_entries.clone(),
7020 )
7021 } else {
7022 (Vec::new(), Vec::new())
7023 };
7024 let stored = self.table.ordinal_bounds.get(column).copied().unwrap_or(0);
7027 Ok(Some(FrequencyOccurrences {
7028 omitted_max: summary.omitted_max.max(summary.ordinal_bound).max(stored),
7029 ordinals: summary.ordinals.clone(),
7030 anchors,
7031 anchor_indices,
7032 }))
7033 }
7034
7035 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
7061 self.table
7062 .distincts
7063 .get(column)
7064 .copied()
7065 .ok_or_else(|| invalid("distinct column index out of range"))
7066 }
7067
7068 pub fn null_count(&self, column: usize) -> Result<u64> {
7079 if column >= self.table.fields.len() {
7080 return Err(invalid("null count column index out of range"));
7081 }
7082 let mut nulls = 0_u64;
7083 for stripe in &self.table.stripes {
7084 let range = stripe
7085 .zone
7086 .column(column)
7087 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7088 nulls = nulls
7089 .checked_add(range.nulls as u64)
7090 .ok_or_else(|| invalid("null count overflow"))?;
7091 }
7092 Ok(nulls)
7093 }
7094
7095 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
7110 if self.null_count(column)? > 0 || self.demoted(column) {
7111 return Ok(None);
7112 }
7113 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
7114 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
7115 if ranks == 0 {
7116 return Ok(None);
7117 }
7118 let low = text_at_rank(&dictionary, 0)?;
7119 let high = text_at_rank(&dictionary, ranks - 1)?;
7120 Ok(Some((low, high)))
7121 }
7122
7123 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
7146 if column >= self.table.fields.len() {
7147 return Err(invalid("extremes column index out of range"));
7148 }
7149 let mut low: Option<Bound> = None;
7150 let mut high: Option<Bound> = None;
7151 for stripe in &self.table.stripes {
7152 let range = stripe
7153 .zone
7154 .column(column)
7155 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7156 if !range.exact {
7157 return Ok(None);
7158 }
7159 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
7164 if stripe.rows > range.nulls {
7165 return Ok(None);
7166 }
7167 continue;
7168 };
7169 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
7170 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
7171 }
7172 Ok(low.zip(high))
7173 }
7174
7175 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
7188 if column >= self.table.fields.len() {
7189 return Err(invalid("sum column index out of range"));
7190 }
7191 let mut total = 0_i128;
7192 let mut rows = 0_u64;
7193 for stripe in &self.table.stripes {
7194 let range = stripe
7195 .zone
7196 .column(column)
7197 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7198 let Some(part) = range.sum else { return Ok(None) };
7199 let Some(sum) = total.checked_add(part) else { return Ok(None) };
7200 total = sum;
7201 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7202 }
7203 Ok(Some((total, rows)))
7204 }
7205
7206 pub fn host_groups(
7208 &self,
7209 column: usize,
7210 _minimum_count: u64,
7211 ) -> Result<Option<Vec<host::HostEntry>>> {
7212 if column >= self.table.fields.len() {
7213 return Err(invalid("host group column index out of range"));
7214 }
7215 Ok(None)
7216 }
7217
7218 #[must_use]
7222 pub fn demoted(&self, column: usize) -> bool {
7223 self.table.demoted.get(column).copied().unwrap_or(false)
7224 }
7225
7226 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7235 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7236 if let Some(dictionary) = self.dictionaries[column].get() {
7237 return Ok(Some(Arc::clone(dictionary)));
7238 }
7239 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7240 if let Some(dictionary) = self.dictionaries[column].get() {
7241 return Ok(Some(Arc::clone(dictionary)));
7242 }
7243 self.opened.fetch_add(1, Atomic::Relaxed);
7244 let dictionary = Arc::new(open_global_dictionary(
7245 Arc::clone(&self.file),
7246 page,
7247 &self.table.fields[column].ty,
7248 TEXT_KEEP_BUDGET,
7249 )?);
7250 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7251 Ok(Some(dictionary))
7252 }
7253
7254 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7261 if of.extent_bytes == 0 {
7262 return Ok(Vec::new());
7263 }
7264 let mut bytes = vec![0; of.extent_bytes as usize];
7265 read_at(&self.file, of.extent_page, &mut bytes)?;
7266 if checksum(&bytes) != of.hash {
7267 return Err(invalid("a section's extent table does not checksum"));
7268 }
7269 let extents = section::decode_extents(&bytes)?;
7270 if extents.len() != of.extents as usize {
7271 return Err(invalid("a section's extent table is not the length the entry says"));
7272 }
7273 Ok(extents)
7274 }
7275
7276 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
7286 let mut bytes = Vec::new();
7287 self.extent_into(of, &mut bytes)?;
7288 Ok(bytes)
7289 }
7290
7291 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7293 bytes.resize(of.length as usize, 0);
7294 self.extent_in_place(of, bytes)
7295 }
7296
7297 fn extent_in_place(&self, of: §ion::Extent, bytes: &mut [u8]) -> Result<()> {
7299 let end = of
7300 .offset
7301 .checked_add(u64::from(of.length))
7302 .ok_or_else(|| invalid("an extent overflows the file"))?;
7303 if of.offset < HEADER || end > self.size || bytes.len() != of.length as usize {
7304 return Err(invalid("an extent is outside the file"));
7305 }
7306 read_at(&self.file, of.offset, bytes)?;
7307 if checksum(bytes) != of.hash {
7308 return Err(invalid("an extent does not checksum"));
7309 }
7310 Ok(())
7311 }
7312
7313 pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7326 let extents = self.extents(of)?;
7327 let Some(first) = extents.first() else { return Ok(Vec::new()) };
7328 let end = first
7329 .offset
7330 .checked_add(u64::from(first.length))
7331 .ok_or_else(|| invalid("an extent overflows the file"))?;
7332 if first.offset < HEADER || end > self.size {
7333 return Err(invalid("an extent is outside the file"));
7334 }
7335 let mut bytes = vec![0; len.min(first.length as usize)];
7336 read_at(&self.file, first.offset, &mut bytes)?;
7337 Ok(bytes)
7338 }
7339
7340 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7353 let extents = self.extents(of)?;
7354 let total = usize::try_from(sum(extents.iter().map(|one| u64::from(one.length))))
7355 .map_err(|_| invalid("a section longer than fits in memory"))?;
7356 let mut bytes = vec![0; total];
7357 let mut at = 0;
7358 for one in &extents {
7359 if one.first != at as u64 {
7360 return Err(invalid("a section's extents do not join up"));
7361 }
7362 let end = at + one.length as usize;
7363 self.extent_in_place(one, &mut bytes[at..end])?;
7364 at = end;
7365 }
7366 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7369 return Err(invalid("a section's header is longer than its payload"));
7370 }
7371 Ok(bytes)
7372 }
7373
7374 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7385 self.read_impl(part, columns, true, None)
7386 }
7387
7388 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7398 self.read_impl(part, columns, false, None)
7399 }
7400
7401 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7409 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7410 let field =
7411 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7412 if !matches!(
7413 field.ty,
7414 LogicalType::TinyInt
7415 | LogicalType::SmallInt
7416 | LogicalType::Integer
7417 | LogicalType::BigInt
7418 ) {
7419 return Ok(None);
7420 }
7421 let (rows, counts) = match self.with_part(place, column, |bytes| {
7422 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7423 return Ok(None);
7424 }
7425 integer::tally(&bytes[2..]).map(Some)
7426 })? {
7427 Some(tallied) => tallied,
7428 None => return Ok(None),
7429 };
7430 if rows != place.rows as usize {
7431 return Err(invalid("encoded integer part holds the wrong number of rows"));
7432 }
7433 for &(value, _) in &counts {
7434 let fits = match field.ty {
7435 LogicalType::TinyInt => i8::try_from(value).is_ok(),
7436 LogicalType::SmallInt => i16::try_from(value).is_ok(),
7437 LogicalType::Integer => i32::try_from(value).is_ok(),
7438 LogicalType::BigInt => true,
7439 _ => false,
7440 };
7441 if !fits {
7442 return Err(invalid("encoded integer value is outside its column type"));
7443 }
7444 }
7445 Ok(Some(counts))
7446 }
7447
7448 pub fn rows_holding(
7460 &self,
7461 part: usize,
7462 column: usize,
7463 sequence: &Sequence,
7464 negated: bool,
7465 ) -> Result<Option<Vec<u32>>> {
7466 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7467 let field =
7468 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7469 if field.ty != LogicalType::Varchar {
7470 return Ok(None);
7471 }
7472 let rows = place.rows as usize;
7473 self.with_part(place, column, |bytes| {
7474 if bytes.first() != Some(&6) {
7475 return Ok(None);
7476 }
7477 let mut cur = Cursor::new(bytes);
7478 cur.u8()?;
7479 let mask = match cur.u8()? {
7480 0 => None,
7481 1 => return Ok(Some(Vec::new())),
7482 2 => {
7483 let from = cur.at;
7484 cur.take(rows.div_ceil(8))?;
7485 Some(&bytes[from..cur.at])
7486 }
7487 _ => return Err(invalid("page validity tag differs")),
7488 };
7489 let needs = sequence.needs();
7492 let first = self.firsts.get(part).copied().unwrap_or_default();
7493 let sketch = self
7494 .text_grams
7495 .get(column)
7496 .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7497 .and_then(|words| words.get(first..first + rows));
7498 let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7499 let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7500 return Ok(None);
7501 };
7502 if held.len() != rows {
7503 return Err(invalid("compressed text page holds the wrong number of rows"));
7504 }
7505 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7506 Ok(Some(
7507 (0..rows)
7508 .filter(|&row| held[row] != negated && valid(row))
7509 .map(|row| row as u32)
7510 .collect(),
7511 ))
7512 })
7513 }
7514
7515 fn is_verified(&self, bit: usize) -> bool {
7518 self.verified
7519 .get(bit / 64)
7520 .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7521 }
7522
7523 fn set_verified(&self, bit: usize) {
7525 if let Some(word) = self.verified.get(bit / 64) {
7526 word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7527 }
7528 }
7529
7530 fn with_part<T>(
7533 &self,
7534 place: Place,
7535 column: usize,
7536 read: impl FnOnce(&[u8]) -> Result<T>,
7537 ) -> Result<T> {
7538 let stripe_index = place.stripe as usize;
7539 let stripe = self
7540 .table
7541 .stripes
7542 .get(stripe_index)
7543 .ok_or_else(|| invalid("stripe index out of range"))?;
7544 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7545 let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7546 let span = *held
7547 .index
7548 .get(place.part as usize)
7549 .ok_or_else(|| invalid("part index out of range"))?;
7550 match &held.page {
7551 Some(page) => read(page.part(place.part as usize, span)?),
7552 None => {
7553 let offset = page
7554 .offset
7555 .checked_add(span.start as u64)
7556 .ok_or_else(|| invalid("part range overflow"))?;
7557 let mut bytes = vec![0; span.length];
7558 read_at(&self.file, offset, &mut bytes)?;
7559 verify_part(&bytes, span)?;
7560 read(&bytes)
7561 }
7562 }
7563 }
7564
7565 pub fn read_rows(
7578 &self,
7579 part: usize,
7580 columns: &[usize],
7581 positions: &[u32],
7582 whole: bool,
7583 ) -> Result<Chunk> {
7584 self.read_impl(part, columns, whole, Some(positions))
7585 }
7586
7587 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7594 if self.demoted(column) {
7597 return Ok(false);
7598 }
7599 if candidates.is_empty() {
7600 return Ok(true);
7601 }
7602 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7603 return Err(Error::internal("native code candidates are not sorted and unique"));
7604 }
7605 let stripe = self.stripe_of(part)?;
7606 let Some(page) = stripe.memberships.get(column) else {
7607 return Ok(false);
7608 };
7609 let mut bytes = vec![0; page.length as usize];
7610 read_at(&self.file, page.offset, &mut bytes)?;
7611 if checksum(&bytes) != page.hash {
7612 return Err(invalid("membership page checksum differs"));
7613 }
7614 let codes = decode_membership(&bytes)?;
7615 let mut left = 0;
7616 let mut right = 0;
7617 while left < codes.len() && right < candidates.len() {
7618 match codes[left].cmp(&candidates[right]) {
7619 Ordering::Less => left += 1,
7620 Ordering::Greater => right += 1,
7621 Ordering::Equal => return Ok(false),
7622 }
7623 }
7624 Ok(true)
7625 }
7626
7627 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7628 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7629 self.table
7630 .stripes
7631 .get(place.stripe as usize)
7632 .ok_or_else(|| invalid("stripe index out of range"))
7633 }
7634
7635 fn held(
7652 &self,
7653 at: usize,
7654 part: usize,
7655 stripe: &Stripe,
7656 column: usize,
7657 whole: bool,
7658 ) -> Result<CachedColumn> {
7659 let cache =
7660 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7661 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7662 if cached.index.is_empty() {
7663 let stripes = self.table.stripes.len();
7664 cached.pages = (0..stripes).map(|_| None).collect();
7665 cached.index = vec![None; stripes];
7666 cached.touched = vec![Vec::new(); stripes];
7667 }
7668 let known = cached.index.get(at).and_then(Clone::clone);
7669 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7670 slot.used.store(true, Atomic::Relaxed);
7671 Arc::clone(&slot.page)
7672 });
7673 let (again, through) = match cached.touched.get_mut(at) {
7675 Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7676 _ => (false, false),
7677 };
7678 let whole = whole && again;
7679 if let Some(index) = known.clone()
7680 && (!whole || page.is_some())
7681 {
7682 return Ok(CachedColumn { stripe: at, index, page });
7683 }
7684 if cached.loading.contains(&at) {
7685 drop(cached);
7686 if let Some(index) = known {
7690 return Ok(CachedColumn { stripe: at, index, page: None });
7691 }
7692 let held = self.page_of(stripe, column, at, false, None)?;
7693 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7694 remember(&mut cached, &held);
7695 return Ok(held);
7696 }
7697 cached.loading.push(at);
7698 drop(cached);
7699
7700 let read = self.page_of(stripe, column, at, whole, known);
7701
7702 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7706 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7707 cached.loading.remove(position);
7708 }
7709 let held = read?;
7710 let taken = remember(&mut cached, &held);
7711 if taken.is_some() && !through {
7712 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7713 cached.passing.push_back(at);
7714 while cached.passing.len() > floor {
7715 let Some(old) = cached.passing.pop_front() else { break };
7716 if let Some(slot) = cached.pages.get_mut(old) {
7717 *slot = None;
7718 }
7719 }
7720 return Ok(held);
7721 }
7722 drop(cached);
7723 if let Some((bytes, used)) = taken {
7724 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7725 self.pool.admit(Held {
7726 shelf: Arc::downgrade(&self.cache),
7727 column,
7728 stripe: at,
7729 bytes,
7730 used,
7731 });
7732 }
7733 Ok(held)
7734 }
7735
7736 fn page_of(
7742 &self,
7743 stripe: &Stripe,
7744 column: usize,
7745 at: usize,
7746 whole: bool,
7747 known: Option<Arc<Vec<PartSpan>>>,
7748 ) -> Result<CachedColumn> {
7749 let index = match known {
7750 Some(index) => index,
7751 None => {
7752 self.indexes.fetch_add(1, Atomic::Relaxed);
7753 Arc::new(read_index(&self.file, stripe, column)?)
7754 }
7755 };
7756 let page = if whole {
7757 self.pages.fetch_add(1, Atomic::Relaxed);
7758 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7759 let length = span.length as usize;
7760 let bytes = match &self.map {
7761 Some(map) if map.get(span.offset, length).is_some() => {
7762 PageBytes::Mapped { map: Arc::clone(map), offset: span.offset, length }
7763 }
7764 _ => {
7765 let mut bytes = vec![0; length];
7766 read_at(&self.file, span.offset, &mut bytes)?;
7767 PageBytes::Read(bytes)
7768 }
7769 };
7770 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7771 Some(Arc::new(HeldPage { bytes, checked }))
7772 } else {
7773 None
7774 };
7775 Ok(CachedColumn { stripe: at, index, page })
7776 }
7777
7778 fn read_impl(
7779 &self,
7780 at: usize,
7781 columns: &[usize],
7782 whole: bool,
7783 positions: Option<&[u32]>,
7784 ) -> Result<Chunk> {
7785 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7786 let index = place.stripe as usize;
7787 let stripe =
7788 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7789 let rows = place.rows as usize;
7790 let mut picked = Vec::with_capacity(columns.len());
7791 for &column in columns {
7792 let field = self
7793 .table
7794 .fields
7795 .get(column)
7796 .ok_or_else(|| invalid("column index out of range"))?;
7797 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7798 let held = self.held(index, place.part as usize, stripe, column, whole)?;
7799 let span = *held
7800 .index
7801 .get(place.part as usize)
7802 .ok_or_else(|| invalid("part index out of range"))?;
7803 let owned;
7804 let mut mapped = false;
7805 let bit = at * self.table.fields.len() + column;
7806 let bytes = match &held.page {
7807 Some(held) if self.is_verified(bit) => part_bytes(held.bytes(), span),
7808 Some(held) => {
7809 held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7810 }
7811 None => {
7812 let offset = page
7813 .offset
7814 .checked_add(span.start as u64)
7815 .ok_or_else(|| invalid("part range overflow"))?;
7816 let bytes =
7817 match self.map.as_deref().and_then(|map| map.get(offset, span.length)) {
7818 Some(bytes) => {
7819 mapped = true;
7820 bytes
7821 }
7822 None => {
7823 let mut bytes = vec![0; span.length];
7824 read_at(&self.file, offset, &mut bytes)?;
7825 owned = bytes;
7826 owned.as_slice()
7827 }
7828 };
7829 if self.is_verified(bit) {
7830 Ok(bytes)
7831 } else {
7832 verify_part(bytes, span).map(|()| {
7833 self.set_verified(bit);
7834 bytes
7835 })
7836 }
7837 }
7838 }
7839 .map_err(|error| {
7840 invalid(&format!(
7841 "{}, column {column} part {} of the page at {}",
7842 error.message(),
7843 place.part,
7844 page.offset,
7845 ))
7846 })?;
7847 let dictionary = self.dictionary(column)?;
7848 let mut vector = match positions {
7854 None => decode(&field.ty, rows, bytes, dictionary)?,
7855 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7856 };
7857 if mapped
7861 && let Some(map) = self.map.as_deref()
7862 && let Some(left) = self.unreleased.get(index * self.table.fields.len() + column)
7863 && left.fetch_update(Atomic::Relaxed, Atomic::Relaxed, |left| left.checked_sub(1))
7864 == Ok(1)
7865 {
7866 map.release(page.offset, page.length as usize);
7867 }
7868 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7872 vector = vector.flatten()?;
7873 }
7874 picked.push(vector.into_pages());
7875 }
7876 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7877 }
7878
7879 #[must_use]
7895 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7896 let Some(place) = self.places.get(part).copied() else { return false };
7897 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7898 if stripe.zone.skips(probes) {
7899 return true;
7900 }
7901 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7902 }
7903
7904 #[must_use]
7911 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7912 let Some(place) = self.places.get(part).copied() else { return false };
7913 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7914 if stripe.zone.column(column).is_some_and(&rule) {
7915 return true;
7916 }
7917 self.stripe_part_ranges(place.stripe as usize, column)
7918 .and_then(|ranges| ranges.get(place.part as usize))
7919 .is_some_and(rule)
7920 }
7921
7922 #[must_use]
7925 pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7926 let place = self.places.get(part).copied()?;
7927 let own = self
7928 .stripe_part_ranges(place.stripe as usize, column)
7929 .and_then(|ranges| ranges.get(place.part as usize));
7930 own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7931 }
7932
7933 #[must_use]
7935 pub fn stripe_ruled_by(
7936 &self,
7937 stripe: usize,
7938 column: usize,
7939 rule: impl Fn(&Range) -> bool,
7940 ) -> bool {
7941 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7942 }
7943
7944 fn outside(&self, place: Place, probe: &Probe) -> bool {
7950 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7951 Some(ranges) => ranges
7952 .get(place.part as usize)
7953 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7954 None => false,
7955 }
7956 }
7957
7958 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7964 let slot = self
7965 .part_ranges
7966 .get(column)?
7967 .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
7968 .get(stripe)?;
7969 if let Some(held) = slot.get() {
7970 return Some(held);
7971 }
7972 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7973 let mut bytes = vec![0; page.length as usize];
7974 read_at(&self.file, page.offset, &mut bytes).ok()?;
7975 if checksum(&bytes) != page.hash {
7976 return None;
7977 }
7978 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7979 let _ = slot.set(ranges);
7980 slot.get().map(|held| held.as_slice())
7981 }
7982
7983 #[must_use]
8000 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
8001 let Some(place) = self.places.get(part).copied() else { return false };
8002 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
8003 if stripe.zone.certain(probes) {
8004 return true;
8005 }
8006 probes
8007 .iter()
8008 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
8009 }
8010
8011 fn inside(&self, place: Place, probe: &Probe) -> bool {
8017 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
8018 Some(ranges) => ranges
8019 .get(place.part as usize)
8020 .is_some_and(|range| range.certain(probe.op, &probe.value)),
8021 None => false,
8022 }
8023 }
8024
8025 #[must_use]
8036 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
8037 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
8038 }
8039
8040 fn sifted(&self, place: Place, probe: &Probe) -> bool {
8046 if probe.op != Op::Equal {
8047 return false;
8048 }
8049 match self.stripe_sieves(place.stripe as usize, probe.column) {
8050 Some(sieves) => sieves
8051 .get(place.part as usize)
8052 .and_then(Option::as_ref)
8053 .is_some_and(|sieve| sieve.excludes(&probe.value)),
8054 None => false,
8055 }
8056 }
8057
8058 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
8065 let slot = self
8066 .sieves
8067 .get(column)?
8068 .get_or_init(|| self.table.stripes.iter().map(|_| OnceLock::new()).collect())
8069 .get(stripe)?;
8070 if let Some(held) = slot.get() {
8071 return Some(held);
8072 }
8073 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
8074 let mut bytes = vec![0; page.length as usize];
8075 read_at(&self.file, page.offset, &mut bytes).ok()?;
8076 if checksum(&bytes) != page.hash {
8077 return None;
8078 }
8079 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
8080 let _ = slot.set(sieves);
8081 slot.get().map(|held| held.as_slice())
8082 }
8083}
8084
8085fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
8087 let code = dictionary.code_at_rank(rank)? as usize;
8088 if dictionary.logical_type() == &LogicalType::Blob {
8089 let bytes = dictionary
8090 .try_bytes_at(code)?
8091 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8092 return Ok(Value::Blob(bytes.to_vec()));
8093 }
8094 let text = dictionary
8095 .try_text_at(code)?
8096 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
8097 Ok(Value::Varchar(text.into()))
8098}
8099
8100fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
8110 file.fill_at(offset, bytes)
8111}
8112
8113trait Positional {
8121 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
8126}
8127
8128impl<T: Positional + ?Sized> Positional for &T {
8129 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8130 (**self).fill_at(offset, bytes)
8131 }
8132}
8133
8134impl<T: Positional + ?Sized> Positional for Arc<T> {
8135 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8136 (**self).fill_at(offset, bytes)
8137 }
8138}
8139
8140impl<T: Positional + ?Sized> Positional for Box<T> {
8141 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8142 (**self).fill_at(offset, bytes)
8143 }
8144}
8145
8146impl Positional for dyn rudb_io::File + '_ {
8147 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8148 while !bytes.is_empty() {
8149 let read = self.read_at(offset, bytes)?;
8150 if read == 0 {
8151 return Err(invalid("column page ends before its declared length"));
8152 }
8153 offset += read as u64;
8154 bytes = &mut bytes[read..];
8155 }
8156 Ok(())
8157 }
8158}
8159
8160impl Positional for File {
8161 #[cfg(unix)]
8162 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8163 use std::os::unix::fs::FileExt;
8164 while !bytes.is_empty() {
8165 let read = self.read_at(bytes, offset).map_err(io)?;
8166 if read == 0 {
8167 return Err(invalid("column page ends before its declared length"));
8168 }
8169 offset += read as u64;
8170 bytes = &mut bytes[read..];
8171 }
8172 Ok(())
8173 }
8174
8175 #[cfg(windows)]
8181 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
8182 use std::os::windows::fs::FileExt;
8183 while !bytes.is_empty() {
8184 let read = self.seek_read(bytes, offset).map_err(io)?;
8185 if read == 0 {
8186 return Err(invalid("column page ends before its declared length"));
8187 }
8188 offset += read as u64;
8189 bytes = &mut bytes[read..];
8190 }
8191 Ok(())
8192 }
8193
8194 #[cfg(not(any(unix, windows)))]
8199 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
8200 use std::io::{Read, Seek, SeekFrom};
8201 let mut file = self.try_clone().map_err(io)?;
8202 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8203 file.read_exact(bytes).map_err(io)
8204 }
8205}
8206
8207#[cfg(test)]
8212fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
8213 use std::io::{Seek, SeekFrom, Write};
8214 let mut file = file;
8215 file.seek(SeekFrom::Start(offset)).map_err(io)?;
8216 file.write_all(bytes).map_err(io)
8217}
8218
8219fn type_tag(ty: &LogicalType) -> Result<u8> {
8226 match ty {
8227 LogicalType::SmallInt => Ok(1),
8228 LogicalType::Integer => Ok(2),
8229 LogicalType::BigInt => Ok(3),
8230 LogicalType::Varchar => Ok(4),
8231 LogicalType::Date => Ok(5),
8232 LogicalType::Timestamp => Ok(6),
8233 LogicalType::Boolean => Ok(7),
8234 LogicalType::TinyInt => Ok(8),
8235 LogicalType::UTinyInt => Ok(9),
8236 LogicalType::USmallInt => Ok(10),
8237 LogicalType::UInteger => Ok(11),
8238 LogicalType::UBigInt => Ok(12),
8239 LogicalType::Decimal { .. } => Ok(13),
8240 LogicalType::Float => Ok(14),
8241 LogicalType::Double => Ok(15),
8242 LogicalType::HugeInt => Ok(16),
8243 LogicalType::UHugeInt => Ok(17),
8244 LogicalType::Time => Ok(18),
8245 LogicalType::TimeTz => Ok(19),
8246 LogicalType::TimestampTz => Ok(20),
8247 LogicalType::Interval => Ok(21),
8248 LogicalType::Uuid => Ok(22),
8249 LogicalType::Blob => Ok(23),
8250 LogicalType::Bit => Ok(24),
8251 LogicalType::TimestampS => Ok(25),
8252 LogicalType::TimestampMs => Ok(26),
8253 LogicalType::TimestampNs => Ok(27),
8254 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8255 }
8256}
8257
8258fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8264 out.push(type_tag(ty)?);
8265 if let LogicalType::Decimal { width, scale } = ty {
8266 out.push(*width);
8267 out.push(*scale);
8268 }
8269 Ok(())
8270}
8271
8272fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8274 let tag = cur.u8()?;
8275 if tag == 13 {
8276 let width = cur.u8()?;
8277 let scale = cur.u8()?;
8278 return LogicalType::decimal(width, scale)
8279 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8280 }
8281 tag_type(tag)
8282}
8283
8284fn tag_type(tag: u8) -> Result<LogicalType> {
8285 match tag {
8286 1 => Ok(LogicalType::SmallInt),
8287 2 => Ok(LogicalType::Integer),
8288 3 => Ok(LogicalType::BigInt),
8289 4 => Ok(LogicalType::Varchar),
8290 5 => Ok(LogicalType::Date),
8291 6 => Ok(LogicalType::Timestamp),
8292 7 => Ok(LogicalType::Boolean),
8293 8 => Ok(LogicalType::TinyInt),
8294 9 => Ok(LogicalType::UTinyInt),
8295 10 => Ok(LogicalType::USmallInt),
8296 11 => Ok(LogicalType::UInteger),
8297 12 => Ok(LogicalType::UBigInt),
8298 14 => Ok(LogicalType::Float),
8299 15 => Ok(LogicalType::Double),
8300 16 => Ok(LogicalType::HugeInt),
8301 17 => Ok(LogicalType::UHugeInt),
8302 18 => Ok(LogicalType::Time),
8303 19 => Ok(LogicalType::TimeTz),
8304 20 => Ok(LogicalType::TimestampTz),
8305 21 => Ok(LogicalType::Interval),
8306 22 => Ok(LogicalType::Uuid),
8307 23 => Ok(LogicalType::Blob),
8308 24 => Ok(LogicalType::Bit),
8309 25 => Ok(LogicalType::TimestampS),
8310 26 => Ok(LogicalType::TimestampMs),
8311 27 => Ok(LogicalType::TimestampNs),
8312 _ => Err(invalid("column type tag is unknown")),
8313 }
8314}
8315
8316fn put_u16(out: &mut Vec<u8>, value: u16) {
8317 out.extend_from_slice(&value.to_le_bytes());
8318}
8319fn put_u32(out: &mut Vec<u8>, value: u32) {
8320 out.extend_from_slice(&value.to_le_bytes());
8321}
8322fn put_u64(out: &mut Vec<u8>, value: u64) {
8323 out.extend_from_slice(&value.to_le_bytes());
8324}
8325fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8326 while value >= 0x80 {
8327 out.push((value as u8 & 0x7f) | 0x80);
8328 value >>= 7;
8329 }
8330 out.push(value as u8);
8331}
8332
8333fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8334 match (left, right) {
8335 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8336 (FrequencyValue::Null, _) => Ordering::Less,
8337 (_, FrequencyValue::Null) => Ordering::Greater,
8338 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8339 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8340 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8341 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8342 }
8343}
8344
8345fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8358 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8359 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8360 };
8361 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8362 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8363 let omitted_max = next.count;
8364 entries.truncate(FREQUENCY_ENTRIES);
8365 entries.shrink_to_fit();
8368 omitted_max
8369 } else {
8370 0
8371 };
8372 entries.sort_unstable_by(order);
8373 omitted_max
8374}
8375
8376fn code_frequency(
8377 dictionary: &GlobalDictionary,
8378 flat: &[u8],
8379 bases: &[u64],
8380) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8381 let seen = dictionary.counts.iter().filter(|count| **count != 0).count();
8386 let mut candidates = Vec::with_capacity(seen + usize::from(dictionary.nulls != 0));
8387 candidates.extend(
8388 dictionary
8389 .counts
8390 .iter()
8391 .enumerate()
8392 .filter(|(_, count)| **count != 0)
8393 .map(|(code, &count)| (count, Some(code as u32))),
8394 );
8395 if dictionary.nulls != 0 {
8396 candidates.push((dictionary.nulls, None));
8397 }
8398 let order = |left: &(u64, Option<u32>), right: &(u64, Option<u32>)| {
8400 right.0.cmp(&left.0).then_with(|| left.1.cmp(&right.1))
8401 };
8402 let omitted_max = if candidates.len() > FREQUENCY_ENTRIES {
8403 let (_, next, _) = candidates.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8404 let omitted_max = next.0;
8405 candidates.truncate(FREQUENCY_ENTRIES);
8406 omitted_max
8407 } else {
8408 0
8409 };
8410 candidates.sort_unstable_by(order);
8411 let entries = candidates
8412 .into_iter()
8413 .map(|(count, code)| FrequencyEntry {
8414 value: code.map_or(FrequencyValue::Null, FrequencyValue::Code),
8415 count,
8416 })
8417 .collect::<Vec<_>>();
8418 let mut spans = Vec::with_capacity(entries.len());
8419 let mut text_bytes = 0_usize;
8420 for entry in &entries {
8421 let span = match entry.value {
8422 FrequencyValue::Code(code) => {
8423 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8424 let bytes = flat
8425 .get(span.0..span.1)
8426 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8427 text_bytes = text_bytes.saturating_add(bytes.len());
8428 Some(span)
8429 }
8430 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8431 };
8432 spans.push(span);
8433 }
8434 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8435 Vec::new()
8436 } else {
8437 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8438 };
8439 Ok((
8440 FrequencySummary {
8441 entries,
8442 omitted_max,
8443 ordinals: Vec::new(),
8444 ordinal_entries: Vec::new(),
8445 ordinal_bound: 0,
8446 },
8447 texts,
8448 ))
8449}
8450
8451fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8452 let mut out = DIRECTORY.to_vec();
8453 let name = table.name.as_bytes();
8454 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8455 out.extend_from_slice(name);
8456 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8457 for field in &table.fields {
8458 let name = field.name.as_bytes();
8459 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8460 out.extend_from_slice(name);
8461 put_type(&mut out, &field.ty)?;
8462 out.push(u8::from(field.not_null));
8463 }
8464 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8465 match dictionary {
8466 None => out.push(0),
8467 Some(page) => {
8468 out.push(dictionary_tag(&field.ty));
8469 put_u64(&mut out, page.offset);
8470 put_u32(&mut out, page.length);
8471 put_u64(&mut out, page.hash);
8472 }
8473 }
8474 }
8475 for distinct in &table.distincts {
8476 match distinct {
8477 None => out.push(0),
8478 Some(count) => {
8479 out.push(1);
8480 put_u64(&mut out, *count);
8481 }
8482 }
8483 }
8484 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8485 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8486 for stripe in &table.stripes {
8487 put_u32(
8488 &mut out,
8489 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8490 );
8491 for &rows in &stripe.parts {
8492 put_u32(&mut out, rows);
8493 }
8494 put_u64(&mut out, stripe.index.offset);
8495 put_u32(&mut out, stripe.index.length);
8496 for page in &stripe.pages {
8497 put_u64(&mut out, page.offset);
8498 put_u32(&mut out, page.length);
8499 }
8500 for (column, ((field, dictionary), membership)) in
8505 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8506 {
8507 if !coded_type(&field.ty) || dictionary.is_none() {
8508 continue;
8509 }
8510 let page = match membership {
8511 Some(page) => page,
8512 None if table.demoted.get(column).copied().unwrap_or(false) => {
8513 Page { offset: HEADER, length: 0, hash: 0 }
8514 }
8515 None => return Err(invalid("string page has no code membership index")),
8516 };
8517 put_u64(&mut out, page.offset);
8518 put_u32(&mut out, page.length);
8519 put_u64(&mut out, page.hash);
8520 }
8521 for sieve in stripe.sieves.slots() {
8522 match sieve {
8523 None => out.push(0),
8524 Some(page) => {
8525 out.push(1);
8526 put_u64(&mut out, page.offset);
8527 put_u32(&mut out, page.length);
8528 put_u64(&mut out, page.hash);
8529 }
8530 }
8531 }
8532 for held in stripe.part_ranges.slots() {
8533 match held {
8534 None => out.push(0),
8535 Some(page) => {
8536 out.push(1);
8537 put_u64(&mut out, page.offset);
8538 put_u32(&mut out, page.length);
8539 put_u64(&mut out, page.hash);
8540 }
8541 }
8542 }
8543 for range in stripe.zone.columns() {
8544 put_bound(&mut out, range.low.as_ref())?;
8545 put_bound(&mut out, range.high.as_ref())?;
8546 put_u32(
8547 &mut out,
8548 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8549 );
8550 out.push(u8::from(range.exact));
8551 match range.sum {
8552 None => out.push(0),
8553 Some(total) => {
8554 out.push(1);
8555 out.extend_from_slice(&total.to_le_bytes());
8556 }
8557 }
8558 }
8559 }
8560 out.extend_from_slice(FREQUENCIES_SPANS);
8561 put_u16(
8562 &mut out,
8563 u16::try_from(table.frequencies.len())
8564 .map_err(|_| invalid("too many frequency columns"))?,
8565 );
8566 for summary in &table.frequencies {
8567 let summary = match summary {
8568 None => {
8569 put_u32(&mut out, 0);
8570 put_u32(&mut out, 0);
8571 continue;
8572 }
8573 Some(Frequencies::Held(summary)) => summary,
8574 Some(Frequencies::Stored { .. }) => {
8576 return Err(invalid("a synopsis left in the file cannot be written back"));
8577 }
8578 };
8579 let length_at = out.len();
8580 put_u32(&mut out, 0);
8581 put_u32(
8582 &mut out,
8583 u32::try_from(summary.entries.len())
8584 .map_err(|_| invalid("too many frequency entries"))?,
8585 );
8586 let start = out.len();
8587 out.push(1);
8588 put_u64(&mut out, summary.omitted_max);
8589 put_u32(
8590 &mut out,
8591 u32::try_from(summary.entries.len())
8592 .map_err(|_| invalid("too many frequency entries"))?,
8593 );
8594 for entry in &summary.entries {
8595 match entry.value {
8596 FrequencyValue::Null => out.push(0),
8597 FrequencyValue::Integer(value) => {
8598 out.push(1);
8599 out.extend_from_slice(&value.to_le_bytes());
8600 }
8601 FrequencyValue::Code(value) => {
8602 out.push(2);
8603 put_u32(&mut out, value);
8604 }
8605 }
8606 put_u64(&mut out, entry.count);
8607 }
8608 put_u32(
8609 &mut out,
8610 u32::try_from(summary.ordinals.len())
8611 .map_err(|_| invalid("too many frequency ordinals"))?,
8612 );
8613 let mut previous = 0_u64;
8614 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8615 let delta = if at == 0 {
8616 ordinal
8617 } else {
8618 ordinal
8619 .checked_sub(previous)
8620 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8621 };
8622 if at != 0 && delta == 0 {
8623 return Err(invalid("frequency ordinals are not unique"));
8624 }
8625 put_var_u64(&mut out, delta);
8626 previous = ordinal;
8627 }
8628 if summary.ordinal_entries.len() != summary.ordinals.len() {
8629 return Err(invalid("frequency ordinal values have a different length"));
8630 }
8631 for &entry in &summary.ordinal_entries {
8632 if entry as usize >= summary.entries.len() {
8633 return Err(invalid("frequency ordinal value is outside its entries"));
8634 }
8635 put_u16(&mut out, entry);
8636 }
8637 let length = u32::try_from(out.len() - start)
8638 .map_err(|_| invalid("a frequency synopsis is too long"))?;
8639 out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8640 }
8641 let bounds = table
8642 .frequencies
8643 .iter()
8644 .enumerate()
8645 .filter_map(|(column, summary)| match summary {
8646 Some(Frequencies::Held(summary)) if summary.ordinal_bound != 0 => {
8647 Some((column, summary.ordinal_bound))
8648 }
8649 _ => None,
8650 })
8651 .collect::<Vec<_>>();
8652 if !bounds.is_empty() {
8653 out.extend_from_slice(ORDINAL_BOUNDS);
8654 put_u16(&mut out, u16::try_from(bounds.len()).map_err(|_| invalid("too many bounds"))?);
8655 for (column, bound) in bounds {
8656 put_u16(
8657 &mut out,
8658 u16::try_from(column).map_err(|_| invalid("bound column overflows"))?,
8659 );
8660 put_u64(&mut out, bound);
8661 }
8662 }
8663 if !table.pair_frequencies.is_empty() {
8664 out.extend_from_slice(PAIR_FREQUENCIES);
8665 put_u16(
8666 &mut out,
8667 u16::try_from(table.pair_frequencies.len())
8668 .map_err(|_| invalid("too many pair frequency summaries"))?,
8669 );
8670 for summary in &table.pair_frequencies {
8671 put_u16(&mut out, summary.first);
8672 put_u16(&mut out, summary.second);
8673 put_u64(&mut out, summary.omitted_max);
8674 put_u16(
8675 &mut out,
8676 u16::try_from(summary.entries.len())
8677 .map_err(|_| invalid("too many pair frequency entries"))?,
8678 );
8679 for entry in &summary.entries {
8680 put_u16(&mut out, entry.first_entry);
8681 match entry.second {
8682 None => out.push(0),
8683 Some(code) => {
8684 out.push(1);
8685 put_u32(&mut out, code);
8686 }
8687 }
8688 put_u64(&mut out, entry.count);
8689 }
8690 }
8691 }
8692 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8693 if text_columns != 0 {
8694 out.extend_from_slice(FREQUENCY_TEXTS);
8695 put_u16(
8696 &mut out,
8697 u16::try_from(text_columns)
8698 .map_err(|_| invalid("too many string frequency columns"))?,
8699 );
8700 for (column, texts) in table.frequency_texts.iter().enumerate() {
8701 if texts.is_empty() {
8702 continue;
8703 }
8704 put_u16(
8705 &mut out,
8706 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8707 );
8708 put_u16(
8709 &mut out,
8710 u16::try_from(texts.len())
8711 .map_err(|_| invalid("too many frequency text entries"))?,
8712 );
8713 for text in texts {
8714 match text {
8715 None => out.push(0),
8716 Some(text) => {
8717 out.push(1);
8718 put_u32(
8719 &mut out,
8720 u32::try_from(text.len())
8721 .map_err(|_| invalid("frequency text is too long"))?,
8722 );
8723 out.extend_from_slice(text);
8724 }
8725 }
8726 }
8727 }
8728 }
8729 if let Some(summary) = &table.host_groups {
8730 out.extend_from_slice(HOST_GROUPS);
8731 put_u16(
8732 &mut out,
8733 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8734 );
8735 put_u64(&mut out, summary.omitted_max);
8736 put_u16(
8737 &mut out,
8738 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8739 );
8740 for entry in &summary.entries {
8741 put_u32(
8742 &mut out,
8743 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8744 );
8745 out.extend_from_slice(entry.host.as_bytes());
8746 put_u64(&mut out, entry.count);
8747 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8748 put_u32(
8749 &mut out,
8750 u32::try_from(entry.minimum.len())
8751 .map_err(|_| invalid("host minimum is too long"))?,
8752 );
8753 out.extend_from_slice(entry.minimum.as_bytes());
8754 }
8755 }
8756 if let Some(clustering) = &table.clustering {
8759 out.extend_from_slice(CLUSTERING);
8760 out.push(clustering.width().tag());
8761 put_u16(
8762 &mut out,
8763 u16::try_from(clustering.columns().len())
8764 .map_err(|_| invalid("too many clustering columns"))?,
8765 );
8766 for &column in clustering.columns() {
8767 put_u16(
8768 &mut out,
8769 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8770 );
8771 }
8772 }
8773 let demoted = (0..table.fields.len())
8774 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8775 .collect::<Vec<_>>();
8776 if !demoted.is_empty() {
8777 out.extend_from_slice(DEMOTED);
8778 put_u16(
8779 &mut out,
8780 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8781 );
8782 for column in demoted {
8783 put_u16(
8784 &mut out,
8785 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8786 );
8787 }
8788 }
8789 if !table.constraints.is_empty() {
8790 out.extend_from_slice(KEYS);
8791 put_count(&mut out, table.constraints.keys.len())?;
8792 for (columns, primary) in &table.constraints.keys {
8793 out.push(u8::from(*primary));
8794 put_columns(&mut out, columns)?;
8795 }
8796 put_count(&mut out, table.constraints.foreign.len())?;
8797 for foreign in &table.constraints.foreign {
8798 put_columns(&mut out, &foreign.columns)?;
8799 put_columns(&mut out, &foreign.referenced)?;
8800 put_u32(
8801 &mut out,
8802 u32::try_from(foreign.table.len())
8803 .map_err(|_| invalid("table name is too long"))?,
8804 );
8805 out.extend_from_slice(foreign.table.as_bytes());
8806 }
8807 }
8808 out.extend_from_slice(SECTIONS);
8814 put_u64(&mut out, table.generation);
8815 put_u16(
8816 &mut out,
8817 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8818 );
8819 for held in &table.sections {
8820 held.encode(&mut out)?;
8821 }
8822 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8823 out.extend_from_slice(DICTIONARY_PAYLOADS);
8824 put_u16(
8825 &mut out,
8826 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8827 );
8828 for at in 0..table.fields.len() {
8829 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8830 }
8831 }
8832 Ok(out)
8833}
8834
8835fn signed_integer(ty: &LogicalType) -> bool {
8844 matches!(
8845 ty,
8846 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8847 )
8848}
8849
8850fn integer_or_date(ty: &LogicalType) -> bool {
8851 matches!(
8852 ty,
8853 LogicalType::TinyInt
8854 | LogicalType::SmallInt
8855 | LogicalType::Integer
8856 | LogicalType::BigInt
8857 | LogicalType::UTinyInt
8858 | LogicalType::USmallInt
8859 | LogicalType::UInteger
8860 | LogicalType::UBigInt
8861 | LogicalType::Date
8862 )
8863}
8864
8865fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8866 table
8867 .fields
8868 .iter()
8869 .enumerate()
8870 .map(|(column, field)| {
8871 if !integer_or_date(&field.ty) {
8872 return None;
8873 }
8874 let mut low: Option<i128> = None;
8875 let mut high: Option<i128> = None;
8876 for stripe in &table.stripes {
8877 let range = stripe.zone.column(column)?;
8878 if !range.exact {
8879 return None;
8880 }
8881 match (range.low.as_ref(), range.high.as_ref()) {
8882 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8883 low = Some(low.map_or(*small, |held| held.min(*small)));
8884 high = Some(high.map_or(*large, |held| held.max(*large)));
8885 }
8886 (None, None) if stripe.rows == range.nulls => {}
8887 _ => return None,
8888 }
8889 }
8890 Some(low.zip(high))
8891 })
8892 .collect()
8893}
8894
8895fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8896 reader
8897 .table
8898 .fields
8899 .iter()
8900 .enumerate()
8901 .map(|(column, field)| {
8902 if !integer_or_date(&field.ty) {
8903 return Ok(None);
8904 }
8905 match reader.exact_extremes(column)? {
8906 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8907 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8908 _ => Ok(None),
8909 }
8910 })
8911 .collect()
8912}
8913
8914fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8915 table
8916 .fields
8917 .iter()
8918 .enumerate()
8919 .map(|(column, field)| {
8920 if !integer_or_date(&field.ty) {
8921 return None;
8922 }
8923 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8924 return None;
8925 };
8926 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8927 return None;
8928 }
8929 let entries = summary
8930 .entries
8931 .iter()
8932 .map(|entry| {
8933 let value = match entry.value {
8934 FrequencyValue::Null => None,
8935 FrequencyValue::Integer(value) => Some(value),
8936 FrequencyValue::Code(_) => return None,
8937 };
8938 Some((value, entry.count))
8939 })
8940 .collect::<Option<Vec<_>>>()?;
8941 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8942 (rows == table.rows as u64).then_some(entries)
8943 })
8944 .collect()
8945}
8946
8947fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8954 if signed {
8955 FrequencyValue::Integer(i128::from(bits as i64))
8956 } else {
8957 FrequencyValue::Integer(i128::from(bits))
8958 }
8959}
8960
8961fn frequency_bits(value: &Value) -> Option<u64> {
8962 Some(match value {
8963 Value::TinyInt(value) => i64::from(*value) as u64,
8964 Value::SmallInt(value) => i64::from(*value) as u64,
8965 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8966 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8967 Value::UTinyInt(value) => u64::from(*value),
8968 Value::USmallInt(value) => u64::from(*value),
8969 Value::UInteger(value) => u64::from(*value),
8970 Value::UBigInt(value) => *value,
8971 _ => return None,
8972 })
8973}
8974
8975fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8976 Some(match value {
8977 Value::Null => None,
8978 Value::TinyInt(value) => Some(i128::from(*value)),
8979 Value::SmallInt(value) => Some(i128::from(*value)),
8980 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8981 Value::BigInt(value) => Some(i128::from(*value)),
8982 Value::UTinyInt(value) => Some(i128::from(*value)),
8983 Value::USmallInt(value) => Some(i128::from(*value)),
8984 Value::UInteger(value) => Some(i128::from(*value)),
8985 Value::UBigInt(value) => Some(i128::from(*value)),
8986 _ => return None,
8987 })
8988}
8989
8990fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8991 reader
8992 .table
8993 .fields
8994 .iter()
8995 .enumerate()
8996 .map(|(column, field)| {
8997 if !integer_or_date(&field.ty) {
8998 return Ok(None);
8999 }
9000 let Some((entries, omitted_max)) = reader.frequency_head(column)? else {
9001 return Ok(None);
9002 };
9003 if omitted_max != 0 || entries.len() > MAX_CATALOG_FREQUENCIES {
9004 return Ok(None);
9005 }
9006 let entries = reader.decode_frequencies(column, &field.ty, &entries)?;
9007 let Some(entries) = entries
9008 .iter()
9009 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
9010 .collect::<Option<Vec<_>>>()
9011 else {
9012 return Ok(None);
9013 };
9014 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
9015 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
9016 })
9017 .collect()
9018}
9019
9020fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
9021 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
9022 let range = stripe.zone.column(column)?;
9023 let sum = sum.checked_add(range.sum?)?;
9024 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
9025 Some((sum, count.checked_add(nonnull)?))
9026 })
9027}
9028
9029fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
9030 table
9031 .fields
9032 .iter()
9033 .enumerate()
9034 .map(|(column, field)| {
9035 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
9036 })
9037 .collect()
9038}
9039
9040fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
9041 reader
9042 .table
9043 .fields
9044 .iter()
9045 .enumerate()
9046 .map(
9047 |(column, field)| {
9048 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
9049 },
9050 )
9051 .collect()
9052}
9053
9054fn encode_catalog(
9055 entries: &[Entry],
9056 views: &[ViewEntry],
9057 card: Option<&KeptCard>,
9058 anchor: Option<&LogAnchor>,
9059) -> Result<Vec<u8>> {
9060 let mut out = CATALOG.to_vec();
9061 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
9062 for entry in entries {
9063 let name = entry.name.as_bytes();
9064 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
9065 out.extend_from_slice(name);
9066 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
9067 put_u16(
9068 &mut out,
9069 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
9070 );
9071 for field in &entry.fields {
9072 let name = field.name.as_bytes();
9073 put_u16(
9074 &mut out,
9075 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9076 );
9077 out.extend_from_slice(name);
9078 put_type(&mut out, &field.ty)?;
9079 out.push(u8::from(field.not_null));
9080 }
9081 put_u64(&mut out, entry.directory.offset);
9082 put_u32(&mut out, entry.directory.length);
9083 put_u64(&mut out, entry.directory.hash);
9084 }
9085 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
9086 for view in views {
9087 let name = view.name.as_bytes();
9088 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
9089 out.extend_from_slice(name);
9090 put_long_text(&mut out, &view.sql, "view body")?;
9091 put_long_text(&mut out, &view.statement, "view statement")?;
9092 put_u16(
9093 &mut out,
9094 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
9095 );
9096 for alias in &view.aliases {
9097 let alias = alias.as_bytes();
9098 put_u16(
9099 &mut out,
9100 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
9101 );
9102 out.extend_from_slice(alias);
9103 }
9104 put_u16(
9105 &mut out,
9106 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
9107 );
9108 for field in &view.columns {
9109 let name = field.name.as_bytes();
9110 put_u16(
9111 &mut out,
9112 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
9113 );
9114 out.extend_from_slice(name);
9115 put_type(&mut out, &field.ty)?;
9116 out.push(u8::from(field.not_null));
9117 }
9118 }
9119 out.extend_from_slice(NONZERO_COUNTS);
9120 for entry in entries {
9121 if entry.nonzero.len() != entry.fields.len() {
9122 return Err(invalid("nonzero count width differs from schema"));
9123 }
9124 for count in &entry.nonzero {
9125 match count {
9126 None => out.push(0),
9127 Some(count) => {
9128 out.push(1);
9129 put_u64(&mut out, *count);
9130 }
9131 }
9132 }
9133 }
9134 out.extend_from_slice(AGGREGATE_SUMS);
9135 for entry in entries {
9136 if entry.aggregates.len() != entry.fields.len() {
9137 return Err(invalid("aggregate sum width differs from schema"));
9138 }
9139 for summary in &entry.aggregates {
9140 match summary {
9141 None => out.push(0),
9142 Some((sum, count)) => {
9143 out.push(1);
9144 out.extend_from_slice(&sum.to_le_bytes());
9145 put_u64(&mut out, *count);
9146 }
9147 }
9148 }
9149 }
9150 out.extend_from_slice(DISTINCT_COUNTS);
9151 for entry in entries {
9152 if entry.distincts.len() != entry.fields.len() {
9153 return Err(invalid("distinct count width differs from schema"));
9154 }
9155 for count in &entry.distincts {
9156 match count {
9157 None => out.push(0),
9158 Some(count) => {
9159 if *count > entry.rows as u64 {
9160 return Err(invalid("distinct count exceeds table rows"));
9161 }
9162 out.push(1);
9163 put_u64(&mut out, *count);
9164 }
9165 }
9166 }
9167 }
9168 out.extend_from_slice(INTEGER_EXTREMES);
9169 for entry in entries {
9170 if entry.extremes.len() != entry.fields.len() {
9171 return Err(invalid("integer extremes width differs from schema"));
9172 }
9173 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
9174 match extremes {
9175 None => out.push(0),
9176 Some(None) if integer_or_date(&field.ty) => out.push(1),
9177 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
9178 out.push(2);
9179 out.extend_from_slice(&low.to_le_bytes());
9180 out.extend_from_slice(&high.to_le_bytes());
9181 }
9182 _ => return Err(invalid("integer extremes type or range differs")),
9183 }
9184 }
9185 }
9186 out.extend_from_slice(COMPLETE_FREQUENCIES);
9187 for entry in entries {
9188 if entry.frequencies.len() != entry.fields.len() {
9189 return Err(invalid("numeric frequency width differs from schema"));
9190 }
9191 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
9192 match frequencies {
9193 None => out.push(0),
9194 Some(entries)
9195 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
9196 {
9197 let mut total = 0_u64;
9198 for (at, (value, count)) in entries.iter().enumerate() {
9199 if entries[..at].iter().any(|(held, _)| held == value) {
9200 return Err(invalid("numeric frequency value repeats"));
9201 }
9202 total = total
9203 .checked_add(*count)
9204 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9205 }
9206 if total != entry.rows as u64 {
9207 return Err(invalid("numeric frequencies do not cover table rows"));
9208 }
9209 out.push(1);
9210 out.push(entries.len() as u8);
9211 for (value, count) in entries {
9212 match value {
9213 None => out.push(0),
9214 Some(value) => {
9215 out.push(1);
9216 out.extend_from_slice(&value.to_le_bytes());
9217 }
9218 }
9219 put_u64(&mut out, *count);
9220 }
9221 }
9222 _ => return Err(invalid("numeric frequency type or width differs")),
9223 }
9224 }
9225 }
9226 if let Some(card) = card {
9227 out.extend_from_slice(DEVICE_CARD);
9228 let device = card.device.as_bytes();
9229 put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
9230 out.extend_from_slice(device);
9231 put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
9232 out.extend_from_slice(&card.bytes);
9233 }
9234 if let Some(anchor) = anchor {
9235 anchor.encode(&mut out)?;
9236 }
9237 Ok(out)
9238}
9239
9240fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
9242 let bytes = text.as_bytes();
9243 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
9244 out.extend_from_slice(bytes);
9245 Ok(())
9246}
9247
9248fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
9251 let mut cur = Cursor::new(bytes);
9252 if cur.take(8)? != CATALOG {
9253 return Err(invalid("catalog magic differs"));
9254 }
9255 let count = cur.u32()? as usize;
9256 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
9257 for _ in 0..count {
9258 let name = cur.text()?;
9259 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9260 let width = cur.u16()? as usize;
9261 let mut fields = Vec::with_capacity(width);
9262 for _ in 0..width {
9263 let name = cur.text()?;
9264 let ty = read_type(&mut cur)?;
9265 let not_null = match cur.u8()? {
9266 0 => false,
9267 1 => true,
9268 _ => return Err(invalid("nullability flag differs")),
9269 };
9270 fields.push(Field { name, ty, not_null });
9271 }
9272 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9273 let end = directory
9274 .offset
9275 .checked_add(u64::from(directory.length))
9276 .ok_or_else(|| invalid("table directory offset overflow"))?;
9277 if directory.offset < HEADER
9278 || end > size
9279 || directory.length as usize > MAX_DIRECTORY
9280 || directory.length == 0
9281 {
9282 return Err(invalid("table directory range is outside the file"));
9283 }
9284 if entries.iter().any(|held| held.name == name) {
9285 return Err(invalid("two tables in the catalog have the same name"));
9286 }
9287 let nonzero = vec![None; fields.len()];
9288 let aggregates = vec![None; fields.len()];
9289 let distincts = vec![None; fields.len()];
9290 let extremes = vec![None; fields.len()];
9291 let frequencies = vec![None; fields.len()];
9292 entries.push(Entry {
9293 name,
9294 fields,
9295 rows,
9296 directory,
9297 nonzero,
9298 aggregates,
9299 distincts,
9300 extremes,
9301 frequencies,
9302 });
9303 }
9304 let count = if cur.done() { 0 } else { cur.u32()? as usize };
9309 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9310 for _ in 0..count {
9311 let name = cur.text()?;
9312 let sql = cur.long_text()?;
9313 let statement = cur.long_text()?;
9314 let width = cur.u16()? as usize;
9315 let mut aliases = Vec::with_capacity(width);
9316 for _ in 0..width {
9317 aliases.push(cur.text()?);
9318 }
9319 let width = cur.u16()? as usize;
9320 let mut columns = Vec::with_capacity(width);
9321 for _ in 0..width {
9322 let name = cur.text()?;
9323 let ty = read_type(&mut cur)?;
9324 let not_null = match cur.u8()? {
9325 0 => false,
9326 1 => true,
9327 _ => return Err(invalid("nullability flag differs")),
9328 };
9329 columns.push(Field { name, ty, not_null });
9330 }
9331 if views.iter().any(|held| held.name == name) {
9335 return Err(invalid("two views in the catalog have the same name"));
9336 }
9337 if entries.iter().any(|held| held.name == name) {
9338 return Err(invalid("a table and a view in the catalog have the same name"));
9339 }
9340 views.push(ViewEntry { name, sql, statement, aliases, columns });
9341 }
9342 if !cur.done() {
9343 if cur.take(8)? != NONZERO_COUNTS {
9344 return Err(invalid("catalog extension magic differs"));
9345 }
9346 for entry in &mut entries {
9347 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9348 *count = match cur.u8()? {
9349 0 => None,
9350 1 if matches!(
9351 field.ty,
9352 LogicalType::TinyInt
9353 | LogicalType::SmallInt
9354 | LogicalType::Integer
9355 | LogicalType::BigInt
9356 | LogicalType::UTinyInt
9357 | LogicalType::USmallInt
9358 | LogicalType::UInteger
9359 | LogicalType::UBigInt
9360 ) =>
9361 {
9362 let value = cur.u64()?;
9363 if value > entry.rows as u64 {
9364 return Err(invalid("nonzero count exceeds rows"));
9365 }
9366 Some(value)
9367 }
9368 _ => return Err(invalid("nonzero count tag or column type differs")),
9369 };
9370 }
9371 }
9372 }
9373 if !cur.done() {
9374 if cur.take(8)? != AGGREGATE_SUMS {
9375 return Err(invalid("aggregate catalog extension magic differs"));
9376 }
9377 for entry in &mut entries {
9378 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9379 *summary = match cur.u8()? {
9380 0 => None,
9381 1 if signed_integer(&field.ty) => {
9382 let sum = i128::from_le_bytes(
9383 cur.take(16)?
9384 .try_into()
9385 .map_err(|_| invalid("aggregate sum is truncated"))?,
9386 );
9387 let count = cur.u64()?;
9388 if count > entry.rows as u64 {
9389 return Err(invalid("aggregate count exceeds table rows"));
9390 }
9391 Some((sum, count))
9392 }
9393 _ => return Err(invalid("aggregate sum tag or column type differs")),
9394 };
9395 }
9396 }
9397 }
9398 if !cur.done() {
9399 if cur.take(8)? != DISTINCT_COUNTS {
9400 return Err(invalid("distinct catalog extension magic differs"));
9401 }
9402 for entry in &mut entries {
9403 for count in &mut entry.distincts {
9404 *count = match cur.u8()? {
9405 0 => None,
9406 1 => {
9407 let value = cur.u64()?;
9408 if value > entry.rows as u64 {
9409 return Err(invalid("distinct count exceeds table rows"));
9410 }
9411 Some(value)
9412 }
9413 _ => return Err(invalid("distinct count tag differs")),
9414 };
9415 }
9416 }
9417 }
9418 if !cur.done() {
9419 if cur.take(8)? != INTEGER_EXTREMES {
9420 return Err(invalid("integer extremes catalog extension magic differs"));
9421 }
9422 for entry in &mut entries {
9423 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9424 *extremes = match cur.u8()? {
9425 0 => None,
9426 1 if integer_or_date(&field.ty) => Some(None),
9427 2 if integer_or_date(&field.ty) => {
9428 let low = i128::from_le_bytes(
9429 cur.take(16)?
9430 .try_into()
9431 .map_err(|_| invalid("minimum is truncated"))?,
9432 );
9433 let high = i128::from_le_bytes(
9434 cur.take(16)?
9435 .try_into()
9436 .map_err(|_| invalid("maximum is truncated"))?,
9437 );
9438 if low > high {
9439 return Err(invalid("integer extremes are reversed"));
9440 }
9441 Some(Some((low, high)))
9442 }
9443 _ => return Err(invalid("integer extremes tag or type differs")),
9444 };
9445 }
9446 }
9447 }
9448 if !cur.done() {
9449 if cur.take(8)? != COMPLETE_FREQUENCIES {
9450 return Err(invalid("numeric frequency catalog extension magic differs"));
9451 }
9452 for entry in &mut entries {
9453 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9454 *frequencies = match cur.u8()? {
9455 0 => None,
9456 1 if integer_or_date(&field.ty) => {
9457 let len = cur.u8()? as usize;
9458 if len > MAX_CATALOG_FREQUENCIES {
9459 return Err(invalid("too many catalog numeric frequencies"));
9460 }
9461 let mut values = Vec::with_capacity(len);
9462 let mut total = 0_u64;
9463 for _ in 0..len {
9464 let value = match cur.u8()? {
9465 0 => None,
9466 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9467 |_| invalid("numeric frequency value is truncated"),
9468 )?)),
9469 _ => return Err(invalid("numeric frequency value tag differs")),
9470 };
9471 if values.iter().any(|(held, _)| *held == value) {
9472 return Err(invalid("numeric frequency value repeats"));
9473 }
9474 let count = cur.u64()?;
9475 total = total
9476 .checked_add(count)
9477 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9478 values.push((value, count));
9479 }
9480 if total != entry.rows as u64 {
9481 return Err(invalid("numeric frequencies do not cover table rows"));
9482 }
9483 Some(values)
9484 }
9485 _ => return Err(invalid("numeric frequency tag or type differs")),
9486 };
9487 }
9488 }
9489 }
9490 let mut card = None;
9491 let mut anchor = None;
9492 while !cur.done() {
9495 let tag = cur.take(8)?;
9496 if tag == DEVICE_CARD && card.is_none() && anchor.is_none() {
9497 let device = cur.text()?;
9498 let len = cur.u32()? as usize;
9499 if len > MAX_CARD {
9500 return Err(invalid("device card is longer than any card"));
9501 }
9502 card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9503 } else if tag == anchor::LOG_ANCHOR && anchor.is_none() {
9504 anchor = Some(LogAnchor::decode(&mut cur)?);
9505 } else {
9506 return Err(invalid("catalog extension magic differs or repeats"));
9507 }
9508 }
9509 Ok((entries, views, card, anchor))
9510}
9511
9512type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>, Option<LogAnchor>);
9514
9515const MAX_CARD: usize = 64 << 10;
9517
9518#[derive(Debug, Clone, PartialEq, Eq)]
9525struct KeptCard {
9526 device: String,
9527 bytes: Vec<u8>,
9528}
9529
9530fn directory_of(path: &Path) -> &Path {
9532 path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9533}
9534
9535fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9542 let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9543 return held;
9544 };
9545 match rudb_io::device::kept(&device) {
9546 Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9547 None => held,
9548 }
9549}
9550
9551fn remember_card(path: &Path, card: Option<&KeptCard>) {
9553 let Some(card) = card else { return };
9554 let dir = directory_of(path);
9555 let Ok(device) = rudb_io::device::device_key(dir) else { return };
9556 if device != card.device {
9557 return;
9558 }
9559 if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9560 rudb_io::device::remember(&device, decoded);
9561 }
9562}
9563
9564struct Cursor<'a> {
9572 bytes: &'a [u8],
9573 at: usize,
9574 window: Option<Window<'a>>,
9575}
9576
9577struct Window<'a> {
9579 file: &'a File,
9580 offset: u64,
9581 length: usize,
9582 start: usize,
9584 held: Vec<u8>,
9585 size: usize,
9587}
9588
9589const DIRECTORY_WINDOW: usize = 64 << 10;
9591
9592impl<'a> Cursor<'a> {
9593 fn new(bytes: &'a [u8]) -> Self {
9594 Self { bytes, at: 0, window: None }
9595 }
9596
9597 fn over(file: &'a File, offset: u64, length: usize) -> Self {
9599 let window =
9600 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9601 Self { bytes: &[], at: 0, window: Some(window) }
9602 }
9603
9604 fn len(&self) -> usize {
9606 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9607 }
9608
9609 fn ensure(&mut self, len: usize) -> Result<()> {
9611 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9612 if end > self.len() {
9613 return Err(invalid("directory is truncated"));
9614 }
9615 let Some(window) = &mut self.window else { return Ok(()) };
9616 if self.at < window.start || end > window.start + window.held.len() {
9617 let want = len.max(window.size).min(window.length - self.at);
9618 window.start = self.at;
9619 window.held.resize(want, 0);
9620 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9621 }
9622 Ok(())
9623 }
9624
9625 fn held(&self, at: usize, len: usize) -> &[u8] {
9627 match &self.window {
9628 Some(window) => &window.held[at - window.start..at - window.start + len],
9629 None => &self.bytes[at..at + len],
9630 }
9631 }
9632
9633 #[inline]
9635 fn peek(&mut self, len: usize) -> Result<&[u8]> {
9636 if self.window.is_none() {
9637 let bytes = self.bytes;
9638 return Ok(&bytes[self.at..self.end(len)?]);
9639 }
9640 self.ensure(len)?;
9641 Ok(self.held(self.at, len))
9642 }
9643
9644 #[inline]
9650 fn take(&mut self, len: usize) -> Result<&[u8]> {
9651 if self.window.is_none() {
9652 let bytes = self.bytes;
9653 let (at, end) = (self.at, self.end(len)?);
9654 self.at = end;
9655 return Ok(&bytes[at..end]);
9656 }
9657 self.take_windowed(len)
9658 }
9659
9660 fn skip(&mut self, len: usize) -> Result<()> {
9662 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9663 if end > self.len() {
9664 return Err(invalid("directory is truncated"));
9665 }
9666 self.at = end;
9667 Ok(())
9668 }
9669
9670 fn skip_bound(&mut self) -> Result<()> {
9671 match self.u8()? {
9672 0 => Ok(()),
9673 1 => self.skip(16),
9674 2 => self.skip(8),
9675 3 => {
9676 let length = self.u32()? as usize;
9677 self.skip(length)
9678 }
9679 4 => self.skip(17),
9680 _ => Err(invalid("a stored bound has an unknown tag")),
9681 }
9682 }
9683
9684 #[inline]
9686 fn end(&self, len: usize) -> Result<usize> {
9687 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9688 if end > self.bytes.len() {
9689 return Err(invalid("directory is truncated"));
9690 }
9691 Ok(end)
9692 }
9693
9694 #[inline(never)]
9696 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9697 self.ensure(len)?;
9698 self.at += len;
9699 Ok(self.held(self.at - len, len))
9700 }
9701 #[inline]
9702 fn u8(&mut self) -> Result<u8> {
9703 Ok(self.take(1)?[0])
9704 }
9705 #[inline]
9706 fn u16(&mut self) -> Result<u16> {
9707 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9708 }
9709 #[inline]
9710 fn u32(&mut self) -> Result<u32> {
9711 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9712 }
9713 #[inline]
9714 fn u64(&mut self) -> Result<u64> {
9715 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9716 }
9717 fn var_u64(&mut self) -> Result<u64> {
9718 let mut value = 0_u64;
9719 for shift in (0..=63).step_by(7) {
9720 let byte = self.u8()?;
9721 let part = u64::from(byte & 0x7f);
9722 if shift == 63 && part > 1 {
9723 return Err(invalid("frequency ordinal varint overflows"));
9724 }
9725 value |= part << shift;
9726 if byte & 0x80 == 0 {
9727 return Ok(value);
9728 }
9729 }
9730 Err(invalid("frequency ordinal varint is too long"))
9731 }
9732 fn bound(&mut self) -> Result<Option<Bound>> {
9741 let rest = self.len().saturating_sub(self.at);
9742 let mut want = 32;
9743 loop {
9744 let offered = self.peek(want.min(rest))?;
9745 let mut used = 0;
9746 match bounds::get(offered, &mut used) {
9747 Ok(bound) => {
9748 self.at += used;
9749 return Ok(bound);
9750 }
9751 Err(_) if want < rest => want *= 2,
9752 Err(error) => return Err(error),
9753 }
9754 }
9755 }
9756 fn text(&mut self) -> Result<String> {
9757 let len = self.u16()? as usize;
9758 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9759 }
9760 fn done(&self) -> bool {
9763 self.at >= self.len()
9764 }
9765 fn long_text(&mut self) -> Result<String> {
9772 let len = self.u32()? as usize;
9773 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9774 }
9775}
9776
9777fn decode_summary(
9779 cur: &mut Cursor<'_>,
9780 field: &Field,
9781 rows: usize,
9782 values: bool,
9783) -> Result<Option<FrequencySummary>> {
9784 let Some((entries, omitted_max)) = decode_summary_head(cur, field, rows)? else {
9785 return Ok(None);
9786 };
9787 let ordinals = {
9788 let ordinal_count = cur.u32()? as usize;
9789 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9790 return Err(invalid("frequency ordinal count exceeds its bound"));
9791 }
9792 let mut ordinals = Vec::with_capacity(ordinal_count);
9793 let mut previous = 0_u64;
9794 for at in 0..ordinal_count {
9795 let delta = cur.var_u64()?;
9796 if at != 0 && delta == 0 {
9797 return Err(invalid("frequency ordinals are not increasing"));
9798 }
9799 let ordinal = if at == 0 {
9800 delta
9801 } else {
9802 previous.checked_add(delta).ok_or_else(|| invalid("frequency ordinal overflows"))?
9803 };
9804 if ordinal >= rows as u64 {
9805 return Err(invalid("frequency ordinal is outside the table"));
9806 }
9807 ordinals.push(ordinal);
9808 previous = ordinal;
9809 }
9810 ordinals
9811 };
9812 let ordinal_entries = if values {
9813 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9814 for _ in 0..ordinals.len() {
9815 let entry = cur.u16()?;
9816 if entry as usize >= entries.len() {
9817 return Err(invalid("frequency ordinal value is outside its entries"));
9818 }
9819 ordinal_entries.push(entry);
9820 }
9821 ordinal_entries
9822 } else {
9823 Vec::new()
9824 };
9825 Ok(Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries, ordinal_bound: 0 }))
9826}
9827
9828fn decode_summary_head(
9831 cur: &mut Cursor<'_>,
9832 field: &Field,
9833 rows: usize,
9834) -> Result<Option<(Vec<FrequencyEntry>, u64)>> {
9835 Ok(match cur.u8()? {
9836 0 => None,
9837 1 => {
9838 let omitted_max = cur.u64()?;
9839 let count = cur.u32()? as usize;
9840 if count > FREQUENCY_ENTRIES {
9841 return Err(invalid("frequency entry count exceeds its bound"));
9842 }
9843 let mut entries = Vec::with_capacity(count);
9844 for _ in 0..count {
9846 let value = match cur.u8()? {
9847 0 => FrequencyValue::Null,
9848 1 => FrequencyValue::Integer(i128::from_le_bytes(
9849 cur.take(16)?.try_into().expect("sixteen bytes"),
9850 )),
9851 2 => FrequencyValue::Code(cur.u32()?),
9852 _ => return Err(invalid("frequency value tag differs")),
9853 };
9854 let valid = matches!(
9855 (&field.ty, value),
9856 (_, FrequencyValue::Null)
9857 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9858 | (
9859 LogicalType::TinyInt
9860 | LogicalType::SmallInt
9861 | LogicalType::Integer
9862 | LogicalType::BigInt
9863 | LogicalType::UTinyInt
9864 | LogicalType::USmallInt
9865 | LogicalType::UInteger
9866 | LogicalType::UBigInt
9867 | LogicalType::Date
9868 | LogicalType::Timestamp,
9869 FrequencyValue::Integer(_),
9870 )
9871 );
9872 if !valid {
9873 return Err(invalid("frequency value does not match its column"));
9874 }
9875 let count = cur.u64()?;
9876 if count == 0 || count > rows as u64 {
9877 return Err(invalid("frequency count is outside the table"));
9878 }
9879 entries.push(FrequencyEntry { value, count });
9880 }
9881 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9882 return Err(invalid("frequency entries are not descending"));
9883 }
9884 Some((entries, omitted_max))
9885 }
9886 _ => return Err(invalid("frequency summary tag differs")),
9887 })
9888}
9889
9890fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9892 let length = cur.u32()? as usize;
9893 let entries = cur.u32()? as usize;
9894 if entries > FREQUENCY_ENTRIES {
9895 return Err(invalid("frequency entry count exceeds its bound"));
9896 }
9897 if length == 0 {
9898 if entries != 0 {
9899 return Err(invalid("missing frequency synopsis has entries"));
9900 }
9901 return Ok(None);
9902 }
9903 if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9904 return Err(invalid("frequency synopsis span is outside the directory"));
9905 }
9906 Ok(Some((length, entries)))
9907}
9908
9909fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9912 match cur.u8()? {
9913 0 => Ok(()),
9914 1 => {
9915 cur.skip(8)?;
9916 let entries = cur.u32()? as usize;
9917 if entries > FREQUENCY_ENTRIES {
9918 return Err(invalid("frequency entry count exceeds its bound"));
9919 }
9920 for _ in 0..entries {
9921 match cur.u8()? {
9922 0 => {}
9923 1 => cur.skip(16)?,
9924 2 => cur.skip(4)?,
9925 _ => return Err(invalid("frequency value tag differs")),
9926 }
9927 cur.skip(8)?;
9928 }
9929 let ordinals = cur.u32()? as usize;
9930 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9931 return Err(invalid("frequency ordinal count exceeds its bound"));
9932 }
9933 for _ in 0..ordinals {
9934 cur.var_u64()?;
9935 }
9936 if values {
9937 cur.skip(ordinals * 2)?;
9938 }
9939 Ok(())
9940 }
9941 _ => Err(invalid("frequency summary tag differs")),
9942 }
9943}
9944
9945fn quick_nonzero(
9949 mut cur: Cursor<'_>,
9950 name: &str,
9951 fields: &[Field],
9952 rows: usize,
9953 wanted: usize,
9954) -> Result<Option<u64>> {
9955 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9956 return Err(invalid("table directory differs from the catalog"));
9957 }
9958 let width = cur.u16()? as usize;
9959 if width != fields.len() {
9960 return Err(invalid("table directory width differs from the catalog"));
9961 }
9962 for field in fields {
9963 let stored =
9964 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9965 if &stored != field {
9966 return Err(invalid("table directory schema differs from the catalog"));
9967 }
9968 }
9969 let mut dictionaries = Vec::with_capacity(width);
9970 for field in fields {
9971 let held = match cur.u8()? {
9972 0 => false,
9973 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9974 cur.skip(20)?;
9975 true
9976 }
9977 _ => return Err(invalid("dictionary page tag differs")),
9978 };
9979 dictionaries.push(held);
9980 }
9981 for _ in 0..width {
9982 match cur.u8()? {
9983 0 => {}
9984 1 => cur.skip(8)?,
9985 _ => return Err(invalid("distinct count tag differs")),
9986 }
9987 }
9988 if cur.u64()? != rows as u64 {
9989 return Err(invalid("table row count differs from the catalog"));
9990 }
9991 let stripes = cur.u32()? as usize;
9992 let mut total = 0_usize;
9993 let mut nulls = 0_u64;
9994 for _ in 0..stripes {
9995 let parts = cur.u32()? as usize;
9996 if parts == 0 || parts > STRIPE_PARTS {
9997 return Err(invalid("stripe part count is outside its bound"));
9998 }
9999 let mut stripe_rows = 0_usize;
10000 for _ in 0..parts {
10001 stripe_rows = stripe_rows
10002 .checked_add(cur.u32()? as usize)
10003 .ok_or_else(|| invalid("stripe row count overflow"))?;
10004 }
10005 total =
10006 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10007 cur.skip(12 + width * 12)?;
10008 for (field, held) in fields.iter().zip(&dictionaries) {
10009 if coded_type(&field.ty) && *held {
10010 cur.skip(20)?;
10011 }
10012 }
10013 for _ in 0..width * 2 {
10014 match cur.u8()? {
10015 0 => {}
10016 1 => cur.skip(20)?,
10017 _ => return Err(invalid("stripe page tag differs")),
10018 }
10019 }
10020 for column in 0..width {
10021 cur.skip_bound()?;
10022 cur.skip_bound()?;
10023 let count = cur.u32()? as u64;
10024 if count > stripe_rows as u64 {
10025 return Err(invalid("null count exceeds stripe rows"));
10026 }
10027 if column == wanted {
10028 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
10029 }
10030 cur.skip(1)?;
10031 match cur.u8()? {
10032 0 => {}
10033 1 => cur.skip(16)?,
10034 _ => return Err(invalid("a stripe sum has an unknown tag")),
10035 }
10036 }
10037 }
10038 if total != rows {
10039 return Err(invalid("table row count differs from stripes"));
10040 }
10041 if cur.done() {
10042 return Ok(None);
10043 }
10044 let magic = cur.take(8)?;
10045 let spanned = magic == FREQUENCIES_SPANS;
10046 let values = magic == FREQUENCIES || spanned;
10047 if !values && magic != FREQUENCIES_V2 {
10048 return Err(invalid("directory extension magic differs"));
10049 }
10050 if cur.u16()? as usize != width {
10051 return Err(invalid("frequency column count differs"));
10052 }
10053 for _ in 0..wanted {
10054 if spanned {
10055 if let Some((length, _)) = summary_span(&mut cur)? {
10056 cur.skip(length)?;
10057 }
10058 } else {
10059 skip_summary(&mut cur, values, rows)?;
10060 }
10061 }
10062 let summary = if spanned {
10063 let Some((length, entries)) = summary_span(&mut cur)? else {
10064 return Ok(None);
10065 };
10066 let start = cur.at;
10067 let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
10068 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10069 if cur.at - start != length || summary.entries.len() != entries {
10070 return Err(invalid("a stored synopsis differs from its directory span"));
10071 }
10072 summary
10073 } else {
10074 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
10075 return Ok(None);
10076 };
10077 summary
10078 };
10079 let zero = summary
10080 .entries
10081 .iter()
10082 .find(|entry| entry.value == FrequencyValue::Integer(0))
10083 .map(|entry| entry.count)
10084 .or_else(|| (summary.omitted_max == 0).then_some(0));
10085 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
10086}
10087
10088fn quick_integer_fold(
10091 file: &File,
10092 mut cur: Cursor<'_>,
10093 entry: &Entry,
10094 size: u64,
10095 wanted: usize,
10096 emit: &mut impl FnMut(i64, u64) -> Result<()>,
10097) -> Result<()> {
10098 let name = &entry.name;
10099 let fields = &entry.fields;
10100 let rows = entry.rows;
10101 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
10102 return Err(invalid("table directory differs from the catalog"));
10103 }
10104 let width = cur.u16()? as usize;
10105 if width != fields.len() {
10106 return Err(invalid("table directory width differs from the catalog"));
10107 }
10108 for field in fields {
10109 let stored =
10110 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
10111 if &stored != field {
10112 return Err(invalid("table directory schema differs from the catalog"));
10113 }
10114 }
10115 let mut dictionaries = Vec::with_capacity(width);
10116 for field in fields {
10117 dictionaries.push(match cur.u8()? {
10118 0 => false,
10119 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
10120 cur.skip(20)?;
10121 true
10122 }
10123 _ => return Err(invalid("dictionary page tag differs")),
10124 });
10125 }
10126 for _ in 0..width {
10127 match cur.u8()? {
10128 0 => {}
10129 1 => cur.skip(8)?,
10130 _ => return Err(invalid("distinct count tag differs")),
10131 }
10132 }
10133 if cur.u64()? != rows as u64 {
10134 return Err(invalid("table row count differs from the catalog"));
10135 }
10136 let stripes = cur.u32()? as usize;
10137 let mut total = 0_usize;
10138 let mut bytes = Vec::new();
10139 for _ in 0..stripes {
10140 let parts = cur.u32()? as usize;
10141 if parts == 0 || parts > STRIPE_PARTS {
10142 return Err(invalid("stripe part count is outside its bound"));
10143 }
10144 let mut part_rows = Vec::with_capacity(parts);
10145 for _ in 0..parts {
10146 let count = cur.u32()? as usize;
10147 if count == 0 {
10148 return Err(invalid("empty part"));
10149 }
10150 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
10151 part_rows.push(count);
10152 }
10153 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10154 let section = index_section(parts)?;
10155 let index_length =
10156 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
10157 if index.offset < HEADER
10158 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
10159 || index.length as usize != index_length
10160 {
10161 return Err(invalid("index page range is outside the file"));
10162 }
10163 cur.skip(wanted * 12)?;
10164 let page = Span { offset: cur.u64()?, length: cur.u32()? };
10165 if page.offset < HEADER
10166 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
10167 || page.length as usize > MAX_PAGE
10168 {
10169 return Err(invalid("column page range is outside the file"));
10170 }
10171 cur.skip((width - wanted - 1) * 12)?;
10172 for (field, held) in fields.iter().zip(&dictionaries) {
10173 if coded_type(&field.ty) && *held {
10174 cur.skip(20)?;
10175 }
10176 }
10177 for _ in 0..width * 2 {
10178 match cur.u8()? {
10179 0 => {}
10180 1 => cur.skip(20)?,
10181 _ => return Err(invalid("stripe page tag differs")),
10182 }
10183 }
10184 for _ in 0..width {
10185 cur.skip_bound()?;
10186 cur.skip_bound()?;
10187 cur.skip(5)?;
10188 match cur.u8()? {
10189 0 => {}
10190 1 => cur.skip(16)?,
10191 _ => return Err(invalid("a stripe sum has an unknown tag")),
10192 }
10193 }
10194 let spans = read_index_span(file, index, page, parts, wanted)?;
10195 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
10196 bytes.resize(span.length, 0);
10197 let at = page
10198 .offset
10199 .checked_add(span.start as u64)
10200 .ok_or_else(|| invalid("part range overflow"))?;
10201 read_at(file, at, &mut bytes)?;
10202 if checksum(&bytes) != span.hash {
10203 return Err(invalid("integer part checksum differs"));
10204 }
10205 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
10206 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
10207 check_integer_tally_value(value, &fields[wanted].ty)?;
10208 emit(value, count)
10209 })?;
10210 if decoded_rows != expected_rows {
10211 return Err(invalid("encoded integer part holds the wrong number of rows"));
10212 }
10213 } else {
10214 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
10215 if let Some(packed) = column.packed_parts() {
10216 let validity = column.validity();
10217 let all_valid = column.none_null();
10218 let base = packed.base();
10219 let mut codes = [0_u64; 64];
10220 for from in (0..expected_rows).step_by(codes.len()) {
10221 let count = (expected_rows - from).min(codes.len());
10222 packed.unpack(from, &mut codes[..count]);
10223 for (offset, &code) in codes[..count].iter().enumerate() {
10224 if all_valid || validity.is_valid(from + offset) {
10225 emit((base + i128::from(code)) as i64, 1)?;
10227 }
10228 }
10229 }
10230 continue;
10231 }
10232 let column = column.into_flat()?;
10233 let validity = column.validity();
10234 macro_rules! count_decoded {
10235 ($values:expr) => {
10236 for (row, &value) in $values.as_slice().iter().enumerate() {
10237 if validity.is_valid(row) {
10238 emit(i64::from(value), 1)?;
10239 }
10240 }
10241 };
10242 }
10243 match column.data() {
10244 Some(Data::Int8(values)) => count_decoded!(values),
10245 Some(Data::Int16(values)) => count_decoded!(values),
10246 Some(Data::Int32(values)) => count_decoded!(values),
10247 Some(Data::Int64(values)) => count_decoded!(values),
10248 _ => return Err(invalid("decoded integer part has the wrong type")),
10249 }
10250 }
10251 }
10252 }
10253 if total != rows {
10254 return Err(invalid("table row count differs from stripes"));
10255 }
10256 Ok(())
10257}
10258
10259fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
10260 let fits = match ty {
10261 LogicalType::TinyInt => i8::try_from(value).is_ok(),
10262 LogicalType::SmallInt => i16::try_from(value).is_ok(),
10263 LogicalType::Integer => i32::try_from(value).is_ok(),
10264 LogicalType::BigInt => true,
10265 _ => false,
10266 };
10267 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
10268}
10269
10270fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
10271 read_directory(Cursor::new(bytes), size, None)
10272}
10273
10274fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
10279 if cur.take(8)? != DIRECTORY {
10280 return Err(invalid("directory magic differs"));
10281 }
10282 let name = cur.text()?;
10283 let width = cur.u16()? as usize;
10284 let mut fields = Vec::with_capacity(width);
10285 for _ in 0..width {
10286 let name = cur.text()?;
10287 let ty = read_type(&mut cur)?;
10288 let not_null = match cur.u8()? {
10289 0 => false,
10290 1 => true,
10291 _ => return Err(invalid("nullability flag differs")),
10292 };
10293 fields.push(Field { name, ty, not_null });
10294 }
10295 let mut dictionaries = Vec::with_capacity(width);
10296 for field in &fields {
10297 dictionaries.push(match cur.u8()? {
10298 0 => None,
10299 tag if tag == dictionary_tag(&field.ty) => {
10300 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10301 let end = page
10302 .offset
10303 .checked_add(u64::from(page.length))
10304 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
10305 if page.offset < HEADER || end > size {
10310 return Err(invalid("dictionary page range is outside the file"));
10311 }
10312 Some(page)
10313 }
10314 _ => return Err(invalid("dictionary page tag differs")),
10315 });
10316 }
10317 let mut distincts = Vec::with_capacity(width);
10318 for _ in 0..width {
10319 distincts.push(match cur.u8()? {
10320 0 => None,
10321 1 => Some(cur.u64()?),
10322 _ => return Err(invalid("distinct count tag differs")),
10323 });
10324 }
10325 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10326 let count = cur.u32()? as usize;
10327 let mut stripes = Vec::with_capacity(count);
10328 let mut total = 0_usize;
10329 for _ in 0..count {
10330 let count = cur.u32()? as usize;
10331 if count == 0 || count > STRIPE_PARTS {
10332 return Err(invalid("stripe part count is outside its bound"));
10333 }
10334 let mut parts = Vec::with_capacity(count);
10335 let mut stripe_rows = 0_usize;
10336 for _ in 0..count {
10337 let rows = cur.u32()?;
10338 if rows == 0 {
10339 return Err(invalid("empty part"));
10340 }
10341 parts.push(rows);
10342 stripe_rows = stripe_rows
10343 .checked_add(rows as usize)
10344 .ok_or_else(|| invalid("stripe row count overflow"))?;
10345 }
10346 total =
10347 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10348 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10349 let section = index_section(count)?;
10350 let wanted = section
10351 .checked_mul(width)
10352 .and_then(|bytes| u32::try_from(bytes).ok())
10353 .ok_or_else(|| invalid("index page length overflow"))?;
10354 let end = index
10355 .offset
10356 .checked_add(u64::from(index.length))
10357 .ok_or_else(|| invalid("index page offset overflow"))?;
10358 if index.offset < HEADER || end > size || index.length != wanted {
10359 return Err(invalid("index page range is outside the file"));
10360 }
10361 let mut pages = Vec::with_capacity(width);
10362 for _ in 0..width {
10363 let offset = cur.u64()?;
10364 let length = cur.u32()?;
10365 let end = offset
10366 .checked_add(u64::from(length))
10367 .ok_or_else(|| invalid("page offset overflow"))?;
10368 if offset < HEADER || end > size || length as usize > MAX_PAGE {
10369 return Err(invalid("page range is outside the file"));
10370 }
10371 pages.push(Span { offset, length });
10372 }
10373 let mut memberships = vec![None; width];
10374 for (column, field) in fields.iter().enumerate() {
10375 if !coded_type(&field.ty) || dictionaries[column].is_none() {
10376 continue;
10377 }
10378 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10379 let end = page
10380 .offset
10381 .checked_add(u64::from(page.length))
10382 .ok_or_else(|| invalid("membership page offset overflow"))?;
10383 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10384 return Err(invalid("membership page range is outside the file"));
10385 }
10386 if page.length != 0 {
10389 memberships[column] = Some(page);
10390 }
10391 }
10392 let mut sieves = vec![None; width];
10393 for sieve in sieves.iter_mut().take(width) {
10394 match cur.u8()? {
10395 0 => continue,
10396 1 => {}
10397 _ => return Err(invalid("a sieve page has an unknown tag")),
10398 }
10399 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10400 let end = page
10401 .offset
10402 .checked_add(u64::from(page.length))
10403 .ok_or_else(|| invalid("sieve page offset overflow"))?;
10404 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10405 return Err(invalid("sieve page range is outside the file"));
10406 }
10407 *sieve = Some(page);
10408 }
10409 let mut part_ranges = vec![None; width];
10410 for held in part_ranges.iter_mut().take(width) {
10411 match cur.u8()? {
10412 0 => continue,
10413 1 => {}
10414 _ => return Err(invalid("a part range page has an unknown tag")),
10415 }
10416 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10417 let end = page
10418 .offset
10419 .checked_add(u64::from(page.length))
10420 .ok_or_else(|| invalid("part range page offset overflow"))?;
10421 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10422 return Err(invalid("part range page range is outside the file"));
10423 }
10424 *held = Some(page);
10425 }
10426 let mut ranges = Vec::with_capacity(width);
10427 for column in 0..width {
10428 let low = cur.bound()?;
10429 let high = cur.bound()?;
10430 let nulls = cur.u32()? as usize;
10431 if nulls > stripe_rows {
10432 return Err(invalid("null count exceeds stripe rows"));
10433 }
10434 let exact = cur.u8()? != 0;
10435 let sum = match cur.u8()? {
10436 0 => None,
10437 1 => Some(i128::from_le_bytes(
10438 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10439 )),
10440 _ => return Err(invalid("a stripe sum has an unknown tag")),
10441 };
10442 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10448 let low = low.map(|bound| scaled_as(bound, ty));
10449 let high = high.map(|bound| scaled_as(bound, ty));
10450 ranges.push(Range { low, high, nulls, exact, sum });
10451 }
10452 stripes.push(Stripe {
10453 rows: stripe_rows,
10454 parts,
10455 index,
10456 pages,
10457 memberships: Pages::from_slots(memberships)?,
10458 sieves: Pages::from_slots(sieves)?,
10459 part_ranges: Pages::from_slots(part_ranges)?,
10460 zone: Zone::from_ranges(ranges),
10461 });
10462 }
10463 if total != rows {
10464 return Err(invalid("table row count differs from stripes"));
10465 }
10466 let mut entry_counts = vec![0; width];
10469 let frequencies = if cur.done() {
10470 vec![None; width]
10471 } else {
10472 let frequency_magic = cur.take(8)?;
10473 let spanned = frequency_magic == FREQUENCIES_SPANS;
10474 let frequency_values = frequency_magic == FREQUENCIES || spanned;
10475 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10476 return Err(invalid("directory extension magic differs"));
10477 }
10478 if cur.u16()? as usize != width {
10479 return Err(invalid("frequency column count differs"));
10480 }
10481 let mut frequencies = Vec::with_capacity(width);
10482 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10483 if spanned {
10484 let Some((length, entries)) = summary_span(&mut cur)? else {
10485 frequencies.push(None);
10486 continue;
10487 };
10488 *entry_count = entries;
10489 let start = cur.at;
10490 if let Some(offset) = stored_at {
10491 cur.skip(length)?;
10492 frequencies.push(Some(Frequencies::Stored {
10493 span: Span {
10494 offset: offset
10495 .checked_add(start as u64)
10496 .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10497 length: u32::try_from(length)
10498 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10499 },
10500 values: true,
10501 entries,
10502 }));
10503 } else {
10504 let summary = decode_summary(&mut cur, field, rows, true)?
10505 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10506 if cur.at - start != length || summary.entries.len() != entries {
10507 return Err(invalid("a stored synopsis differs from its directory span"));
10508 }
10509 frequencies.push(Some(Frequencies::Held(summary)));
10510 }
10511 continue;
10512 }
10513 let start = cur.at;
10514 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10515 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10516 frequencies.push(match (summary, stored_at) {
10517 (None, _) => None,
10518 (Some(summary), None) => Some(Frequencies::Held(summary)),
10519 (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10520 span: Span {
10521 offset: offset + start as u64,
10522 length: u32::try_from(cur.at - start)
10523 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10524 },
10525 values: frequency_values,
10526 entries: summary.entries.len(),
10527 }),
10528 });
10529 }
10530 frequencies
10531 };
10532 let mut clustering = None;
10542 let mut sections = Vec::new();
10543 let mut pair_frequencies = Vec::new();
10544 let mut seen_pair_frequencies = false;
10545 let mut ordinal_bounds = Vec::new();
10546 let mut seen_ordinal_bounds = false;
10547 let mut frequency_texts = vec![Vec::new(); width];
10548 let mut seen_frequency_texts = false;
10549 let mut host_groups = None;
10550 let mut demoted = Vec::new();
10551 let mut seen_sections = false;
10552 let mut dictionary_payloads = Vec::new();
10553 let mut seen_payloads = false;
10554 let mut constraints = Constraints::default();
10555 let mut generation = 0;
10558 while !cur.done() {
10559 let mut tag = [0u8; 8];
10560 tag.copy_from_slice(cur.take(8)?);
10561 if &tag == PAIR_FREQUENCIES {
10562 if seen_pair_frequencies {
10563 return Err(invalid("directory names two pair frequency blocks"));
10564 }
10565 seen_pair_frequencies = true;
10566 let count = cur.u16()? as usize;
10567 if count > MAX_PAIR_FREQUENCIES {
10568 return Err(invalid("pair frequency count exceeds its bound"));
10569 }
10570 pair_frequencies = Vec::with_capacity(count);
10571 for _ in 0..count {
10572 let first = cur.u16()?;
10573 let second = cur.u16()?;
10574 let first_at = first as usize;
10575 let second_at = second as usize;
10576 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10577 return Err(invalid("pair frequency first column has no synopsis"));
10578 }
10579 let first_entries = entry_counts[first_at];
10580 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10581 || dictionaries.get(second_at).copied().flatten().is_none()
10582 {
10583 return Err(invalid("pair frequency second column has no stable dictionary"));
10584 }
10585 if pair_frequencies
10586 .iter()
10587 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10588 {
10589 return Err(invalid("directory repeats a pair frequency summary"));
10590 }
10591 let omitted_max = cur.u64()?;
10592 if omitted_max > rows as u64 {
10593 return Err(invalid("pair frequency omitted count exceeds the table"));
10594 }
10595 let entries_count = cur.u16()? as usize;
10596 if entries_count > FREQUENCY_ENTRIES {
10597 return Err(invalid("pair frequency entry count exceeds its bound"));
10598 }
10599 let mut entries = Vec::with_capacity(entries_count);
10600 for _ in 0..entries_count {
10601 let first_entry = cur.u16()?;
10602 if first_entry as usize >= first_entries {
10603 return Err(invalid("pair frequency anchor is outside its synopsis"));
10604 }
10605 let second = match cur.u8()? {
10606 0 => None,
10607 1 => Some(cur.u32()?),
10608 _ => return Err(invalid("pair frequency string tag differs")),
10609 };
10610 let count = cur.u64()?;
10611 if count == 0 || count > rows as u64 {
10612 return Err(invalid("pair frequency count is outside the table"));
10613 }
10614 entries.push(PairFrequencyEntry { first_entry, second, count });
10615 }
10616 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10617 return Err(invalid("pair frequency entries are not descending"));
10618 }
10619 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10620 }
10621 } else if &tag == ORDINAL_BOUNDS {
10622 if seen_ordinal_bounds {
10623 return Err(invalid("directory names two ordinal bound blocks"));
10624 }
10625 seen_ordinal_bounds = true;
10626 ordinal_bounds = vec![0; width];
10627 let count = cur.u16()? as usize;
10628 if count > width {
10629 return Err(invalid("ordinal bound count exceeds the columns"));
10630 }
10631 for _ in 0..count {
10632 let column = cur.u16()? as usize;
10633 let bound = cur.u64()?;
10634 if column >= width || frequencies.get(column).and_then(Option::as_ref).is_none() {
10635 return Err(invalid("ordinal bound names a column with no synopsis"));
10636 }
10637 if bound == 0 || bound > rows as u64 || ordinal_bounds[column] != 0 {
10638 return Err(invalid("ordinal bound is outside the table or repeated"));
10639 }
10640 ordinal_bounds[column] = bound;
10641 }
10642 } else if &tag == FREQUENCY_TEXTS {
10643 if seen_frequency_texts {
10644 return Err(invalid("directory names two frequency text blocks"));
10645 }
10646 seen_frequency_texts = true;
10647 let columns = cur.u16()? as usize;
10648 if columns > width {
10649 return Err(invalid("frequency text column count exceeds the schema"));
10650 }
10651 for _ in 0..columns {
10652 let column = cur.u16()? as usize;
10653 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10654 return Err(invalid("frequency text column is repeated or out of range"));
10655 }
10656 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10657 || dictionaries.get(column).copied().flatten().is_none()
10658 || frequencies.get(column).and_then(Option::as_ref).is_none()
10659 {
10660 return Err(invalid("frequency texts belong to a non-string synopsis"));
10661 }
10662 let count = cur.u16()? as usize;
10663 if count == 0 || count != entry_counts[column] {
10664 return Err(invalid("frequency text count differs from its synopsis"));
10665 }
10666 let mut texts = Vec::with_capacity(count);
10667 for _ in 0..count {
10668 texts.push(match cur.u8()? {
10669 0 => None,
10670 1 => {
10671 let length = cur.u32()? as usize;
10672 let bytes = cur.take(length)?.to_vec();
10673 if fields[column].ty == LogicalType::Varchar {
10674 std::str::from_utf8(&bytes)
10675 .map_err(|_| invalid("frequency text is not UTF-8"))?;
10676 }
10677 Some(bytes)
10678 }
10679 _ => return Err(invalid("frequency text tag differs")),
10680 });
10681 }
10682 frequency_texts[column] = texts;
10683 }
10684 } else if &tag == HOST_GROUPS {
10685 if host_groups.is_some() {
10686 return Err(invalid("directory names two host group blocks"));
10687 }
10688 let column = cur.u16()? as usize;
10689 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10690 || dictionaries.get(column).copied().flatten().is_none()
10691 {
10692 return Err(invalid("host groups belong to a non-string dictionary"));
10693 }
10694 let omitted_max = cur.u64()?;
10695 if omitted_max > rows as u64 {
10696 return Err(invalid("host group bound exceeds the table"));
10697 }
10698 let count = cur.u16()? as usize;
10699 if count > host::CAPACITY {
10700 return Err(invalid("host group count exceeds its bound"));
10701 }
10702 let mut entries = Vec::with_capacity(count);
10703 let mut bytes = 0_usize;
10704 for _ in 0..count {
10705 let host_len = cur.u32()? as usize;
10706 bytes =
10707 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10708 if bytes > host::BYTE_BUDGET {
10709 return Err(invalid("host groups exceed their byte budget"));
10710 }
10711 let host = std::str::from_utf8(cur.take(host_len)?)
10712 .map_err(|_| invalid("host is not UTF-8"))?
10713 .to_owned();
10714 let count = cur.u64()?;
10715 if count == 0 || count > rows as u64 {
10716 return Err(invalid("host group count exceeds the table"));
10717 }
10718 let bytes_sum = i128::from_le_bytes(
10719 cur.take(16)?
10720 .try_into()
10721 .map_err(|_| invalid("host length sum is truncated"))?,
10722 );
10723 if bytes_sum < 0 {
10724 return Err(invalid("host length sum is negative"));
10725 }
10726 let minimum_len = cur.u32()? as usize;
10727 bytes =
10728 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10729 if bytes > host::BYTE_BUDGET {
10730 return Err(invalid("host groups exceed their byte budget"));
10731 }
10732 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10733 .map_err(|_| invalid("host minimum is not UTF-8"))?
10734 .to_owned();
10735 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10736 }
10737 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10738 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10739 {
10740 return Err(invalid("host groups are not in certified order"));
10741 }
10742 host_groups = Some(host::HostSummary { column, omitted_max, entries });
10743 } else if &tag == CLUSTERING {
10744 if clustering.is_some() {
10745 return Err(invalid("directory names two clustering declarations"));
10746 }
10747 let bucket = Width::from_tag(cur.u8()?)
10748 .ok_or_else(|| invalid("clustering width tag differs"))?;
10749 let count = cur.u16()? as usize;
10750 let mut columns = Vec::with_capacity(count.min(fields.len()));
10751 for _ in 0..count {
10752 columns.push(u32::from(cur.u16()?));
10753 }
10754 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10757 invalid("stored clustering declaration does not match the table it is on")
10758 })?);
10759 } else if &tag == DEMOTED {
10760 if !demoted.is_empty() {
10761 return Err(invalid("directory names two demoted column blocks"));
10762 }
10763 let count = cur.u16()? as usize;
10764 if count == 0 || count > width {
10765 return Err(invalid("demoted column count is outside the schema"));
10766 }
10767 demoted = vec![false; width];
10768 for _ in 0..count {
10769 let column = cur.u16()? as usize;
10770 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10771 return Err(invalid("a demoted column is repeated or has no dictionary"));
10772 }
10773 demoted[column] = true;
10774 }
10775 } else if &tag == SECTIONS {
10776 if seen_sections {
10777 return Err(invalid("directory names two section tables"));
10778 }
10779 seen_sections = true;
10780 generation = cur.u64()?;
10781 let count = cur.u16()? as usize;
10782 if count > MAX_SECTIONS {
10783 return Err(invalid("section count exceeds its bound"));
10784 }
10785 sections = Vec::with_capacity(count);
10786 for _ in 0..count {
10789 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10790 }
10791 for held in §ions {
10792 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10793 return Err(invalid("a section's extent table overflows the file"));
10794 };
10795 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10799 return Err(invalid("a section's extent table is outside the file"));
10800 }
10801 if held.extents == 0 && held.extent_bytes != 0 {
10802 return Err(invalid("a section with no extents names an extent table"));
10803 }
10804 }
10805 } else if &tag == DICTIONARY_PAYLOADS {
10806 if seen_payloads {
10807 return Err(invalid("directory names two dictionary payload blocks"));
10808 }
10809 seen_payloads = true;
10810 let count = cur.u16()? as usize;
10811 if count != fields.len() {
10812 return Err(invalid("dictionary payload block does not match the table's columns"));
10813 }
10814 dictionary_payloads = Vec::with_capacity(count);
10815 for _ in 0..count {
10816 let bytes = cur.u64()?;
10817 if bytes > size {
10818 return Err(invalid("a dictionary payload is larger than the file"));
10819 }
10820 dictionary_payloads.push(bytes);
10821 }
10822 } else if &tag == KEYS {
10823 if !constraints.is_empty() {
10824 return Err(invalid("directory names two key blocks"));
10825 }
10826 let fits = |columns: &[u16]| {
10827 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10828 };
10829 let count = cur.u16()? as usize;
10830 for _ in 0..count {
10831 let primary = cur.u8()? != 0;
10832 let columns = columns_of(&mut cur)?;
10833 if !fits(&columns) {
10834 return Err(invalid("a stored key names a column the table does not have"));
10835 }
10836 constraints.keys.push((columns, primary));
10837 }
10838 let count = cur.u16()? as usize;
10839 for _ in 0..count {
10840 let columns = columns_of(&mut cur)?;
10841 let referenced = columns_of(&mut cur)?;
10842 let len = cur.u32()? as usize;
10843 let table = std::str::from_utf8(cur.take(len)?)
10844 .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10845 .to_owned();
10846 if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10847 return Err(invalid("a stored foreign key does not match its table"));
10848 }
10849 constraints.foreign.push(StoredForeign { columns, table, referenced });
10850 }
10851 if constraints.is_empty() {
10852 return Err(invalid("a key block holds no key"));
10853 }
10854 } else {
10855 return Err(invalid("directory extension magic differs"));
10856 }
10857 }
10858 if !cur.done() {
10859 return Err(invalid("directory has trailing bytes"));
10860 }
10861 for stripe in &stripes {
10862 for (column, field) in fields.iter().enumerate() {
10863 if coded_type(&field.ty)
10864 && dictionaries[column].is_some()
10865 && stripe.memberships.get(column).is_none()
10866 && !demoted.get(column).copied().unwrap_or(false)
10867 {
10868 return Err(invalid("string page has no code membership index"));
10869 }
10870 }
10871 }
10872 Ok(Table {
10873 name,
10874 fields,
10875 stripes,
10876 rows,
10877 dictionaries,
10878 dictionary_payloads,
10879 demoted,
10880 distincts,
10881 frequencies,
10882 ordinal_bounds,
10883 pair_frequencies,
10884 frequency_texts,
10885 host_groups,
10886 clustering,
10887 generation,
10888 sections,
10889 constraints,
10890 })
10891}
10892
10893fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10895 put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10896 Ok(())
10897}
10898
10899fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10901 put_count(out, columns.len())?;
10902 for &column in columns {
10903 put_u16(out, column);
10904 }
10905 Ok(())
10906}
10907
10908fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10910 let count = cur.u16()? as usize;
10911 (0..count).map(|_| cur.u16()).collect()
10912}
10913
10914fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10916 bounds::put(out, bound)
10917}
10918
10919#[derive(Debug)]
10936struct Codes;
10937
10938impl chooser::Chooser for Codes {
10939 fn name(&self) -> &'static str {
10940 "codes"
10941 }
10942
10943 fn narrow_strings(
10944 &self,
10945 _values: &[&[u8]],
10946 offered: &[string::Kind],
10947 _depth: u8,
10948 ) -> Vec<string::Kind> {
10949 offered.to_vec()
10952 }
10953
10954 fn narrow_integers(
10955 &self,
10956 _values: &[i64],
10957 offered: &[integer::Kind],
10958 depth: u8,
10959 ) -> Vec<integer::Kind> {
10960 narrowed_to(Codes::keep(depth), offered)
10963 }
10964
10965 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10966 Codes::keep(depth).contains(&kind)
10967 }
10968}
10969
10970impl Codes {
10971 fn keep(depth: u8) -> &'static [integer::Kind] {
10972 if depth == 0 {
10973 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10974 } else {
10975 &[integer::Kind::Constant, integer::Kind::Packed]
10976 }
10977 }
10978}
10979
10980fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10988 let narrowed: Vec<integer::Kind> =
10989 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10990 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10991}
10992
10993#[derive(Debug)]
11005struct Fixed;
11006
11007impl chooser::Chooser for Fixed {
11008 fn name(&self) -> &'static str {
11009 "fixed"
11010 }
11011
11012 fn narrow_strings(
11013 &self,
11014 _values: &[&[u8]],
11015 offered: &[string::Kind],
11016 _depth: u8,
11017 ) -> Vec<string::Kind> {
11018 offered.to_vec()
11019 }
11020
11021 fn narrow_integers(
11022 &self,
11023 _values: &[i64],
11024 offered: &[integer::Kind],
11025 depth: u8,
11026 ) -> Vec<integer::Kind> {
11027 narrowed_to(Fixed::keep(depth), offered)
11028 }
11029
11030 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
11031 Fixed::keep(depth).contains(&kind)
11032 }
11033}
11034
11035impl Fixed {
11036 fn keep(depth: u8) -> &'static [integer::Kind] {
11037 if depth == 0 {
11038 &[
11039 integer::Kind::Constant,
11040 integer::Kind::Packed,
11041 integer::Kind::Delta,
11042 integer::Kind::Rle,
11043 integer::Kind::Sparse,
11044 integer::Kind::Strided,
11045 ]
11046 } else {
11047 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
11048 }
11049 }
11050}
11051
11052fn widened(data: &Data) -> Option<Vec<i64>> {
11059 match data {
11060 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11061 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11062 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11063 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11064 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11065 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
11066 Data::Int64(values) => Some(values.to_vec()),
11067 _ => None,
11068 }
11069}
11070
11071fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
11077 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
11078 let values = integer::decode_as::<T>(bytes)
11079 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
11080 if values.len() != rows {
11081 return Err(invalid("cascade page holds the wrong number of rows"));
11082 }
11083 Ok(values)
11084 }
11085 Ok(match ty {
11086 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
11087 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
11088 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11089 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
11090 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11091 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
11092 LogicalType::BigInt
11093 | LogicalType::Timestamp
11094 | LogicalType::Time
11095 | LogicalType::TimeTz
11096 | LogicalType::TimestampTz
11097 | LogicalType::TimestampS
11098 | LogicalType::TimestampMs
11099 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11100 LogicalType::Decimal { .. } => match ty.physical() {
11103 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
11104 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
11105 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
11106 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
11107 },
11108 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
11109 })
11110}
11111
11112fn plain_width(ty: &LogicalType) -> Option<usize> {
11115 Some(match ty {
11116 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
11117 LogicalType::SmallInt | LogicalType::USmallInt => 2,
11118 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
11119 LogicalType::BigInt
11120 | LogicalType::Timestamp
11121 | LogicalType::Time
11122 | LogicalType::TimeTz
11123 | LogicalType::TimestampTz
11124 | LogicalType::TimestampS
11125 | LogicalType::TimestampMs
11126 | LogicalType::TimestampNs => 8,
11127 LogicalType::Decimal { .. } => match ty.physical() {
11128 PhysicalType::Int16 => 2,
11129 PhysicalType::Int32 => 4,
11130 PhysicalType::Int64 => 8,
11131 _ => return None,
11134 },
11135 _ => return None,
11136 })
11137}
11138
11139fn cascaded(
11145 flat: &Vector,
11146 ty: &LogicalType,
11147 packed: Option<&Packed<'_>>,
11148 settling: &mut Settling,
11149) -> Result<Option<Vec<u8>>> {
11150 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
11151 let Some(values) = widened(data) else { return Ok(None) };
11152 let plain = values.len().saturating_mul(width);
11153 let best = match packed {
11154 Some(packed) => plain.min(21 + size_of_val(packed.words())),
11156 None => plain,
11157 };
11158 let out = settling.encode(&values)?;
11159 Ok((out.len() < best).then_some(out))
11160}
11161
11162const SEARCH_EVERY: usize = 16;
11169
11170#[derive(Debug, Default)]
11176struct Settling {
11177 shape: Option<Shape>,
11180 since: usize,
11182 symbols: Option<Symbols>,
11184}
11185
11186#[derive(Debug)]
11189struct Symbols {
11190 shape: chooser::Settled,
11191 len: usize,
11194 payload: usize,
11195 since: usize,
11196}
11197
11198impl Settling {
11199 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
11207 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
11208 {
11209 let out = string::encode_fsst(values, &symbols.shape)?;
11210 let held = match &out {
11212 None => symbols.len == 0,
11213 Some(out) => {
11214 (out.len() as u128) * (symbols.payload as u128) * 4
11215 <= (symbols.len as u128) * (payload as u128) * 5
11216 }
11217 };
11218 if held {
11219 symbols.since += 1;
11220 return Ok(out);
11221 }
11222 }
11223 let shape = string::fsst_shape(values);
11224 let out = string::encode_fsst(values, &shape)?;
11225 let len = out.as_ref().map_or(0, Vec::len);
11226 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
11227 Ok(out)
11228 }
11229
11230 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
11237 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
11238 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
11239 let out = integer::encode_with(values, &replay)?;
11240 if !replay.held() {
11241 self.settle(&out, values.len(), replay.first_offered())?;
11242 return Ok(out);
11243 }
11244 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
11245 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
11246 self.since += 1;
11247 return Ok(out);
11248 }
11249 }
11250 let search = chooser::Replay::new(&[], &Fixed);
11252 let out = integer::encode_with(values, &search)?;
11253 self.settle(&out, values.len(), search.first_offered())?;
11254 Ok(out)
11255 }
11256
11257 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
11258 let kinds = integer::shape(out)?;
11259 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
11260 self.since = 0;
11261 Ok(())
11262 }
11263}
11264
11265#[derive(Debug)]
11267struct Shape {
11268 kinds: Vec<integer::Kind>,
11269 offered: Vec<integer::Kind>,
11270 len: usize,
11271 rows: usize,
11272}
11273
11274fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
11315 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
11316 let mut payload = 0_usize;
11317 for row in 0..flat.len() {
11318 let text = flat.bytes_at(row).unwrap_or(b"");
11321 payload = payload.saturating_add(text.len());
11322 values.push(text);
11323 }
11324 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
11326 let Some(out) = settling.text(&values, payload)? else {
11327 return Ok(None);
11328 };
11329 Ok((out.len() < plain).then_some(out))
11330}
11331
11332fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
11333 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
11334 let coded = integer::encode_with(&wide, &Codes)?;
11335 let plain = codes.len().saturating_mul(size_of::<u32>());
11336 Ok((coded.len() < plain).then_some(coded))
11337}
11338
11339fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
11342 let flag = match flat.validity() {
11343 Validity::AllValid => 0,
11344 Validity::AllInvalid => 1,
11345 Validity::Mask(_) => 2,
11346 };
11347 out.push(flag);
11348 if flag == 2 {
11349 for group in (0..flat.len()).step_by(8) {
11350 let mut bits = 0_u8;
11351 for bit in 0..8 {
11352 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11353 bits |= 1 << bit;
11354 }
11355 }
11356 out.push(bits);
11357 }
11358 }
11359}
11360
11361fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11368 let coded = encoded_codes(codes)?;
11369 let mut out = Vec::with_capacity(
11370 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11371 );
11372 out.push(if coded.is_some() { 4 } else { 3 });
11373 out.extend_from_slice(validity);
11374 match coded {
11375 Some(coded) => out.extend_from_slice(&coded),
11376 None => {
11377 for &code in codes {
11378 put_u32(&mut out, code);
11379 }
11380 }
11381 }
11382 Ok(out)
11383}
11384
11385fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11388 let ty = vector.logical_type();
11389 let flat = vector.flatten()?;
11391 let mut out = Vec::new();
11392 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11393 let compressed_text = if dictionary.is_none() && coded_type(ty) {
11394 text_compressed(&flat, settling)?
11395 } else {
11396 None
11397 };
11398 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11399 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11400 let cascade =
11404 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11405 out.push(if cascade.is_some() {
11406 5
11407 } else if dictionary.is_some() {
11408 1
11409 } else if compressed_text.is_some() {
11410 6
11411 } else if packed.is_some() {
11412 2
11413 } else {
11414 0
11415 });
11416 push_validity(&mut out, &flat);
11417 if let Some(cascade) = cascade {
11418 out.extend_from_slice(&cascade);
11419 return Ok(out);
11420 }
11421 if let Some(dictionary) = dictionary {
11422 out.extend_from_slice(&dictionary);
11423 return Ok(out);
11424 }
11425 if let Some(compressed_text) = compressed_text {
11426 out.extend_from_slice(&compressed_text);
11427 return Ok(out);
11428 }
11429 if let Some(packed) = packed {
11430 if packed.offset() != 0 {
11431 return Err(invalid("writer received a sliced packed vector"));
11432 }
11433 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11434 out.extend_from_slice(&packed.base().to_le_bytes());
11435 put_u32(
11436 &mut out,
11437 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11438 );
11439 for word in packed.words() {
11440 put_u64(&mut out, *word);
11441 }
11442 return Ok(out);
11443 }
11444 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11445 match (ty, data) {
11446 (LogicalType::TinyInt, Data::Int8(values)) => {
11447 for value in &**values {
11448 out.extend_from_slice(&value.to_le_bytes());
11449 }
11450 }
11451 (LogicalType::UTinyInt, Data::UInt8(values)) => {
11452 for value in &**values {
11453 out.extend_from_slice(&value.to_le_bytes());
11454 }
11455 }
11456 (LogicalType::SmallInt, Data::Int16(values)) => {
11457 for value in &**values {
11458 out.extend_from_slice(&value.to_le_bytes());
11459 }
11460 }
11461 (LogicalType::USmallInt, Data::UInt16(values)) => {
11462 for value in &**values {
11463 out.extend_from_slice(&value.to_le_bytes());
11464 }
11465 }
11466 (LogicalType::UInteger, Data::UInt32(values)) => {
11467 for value in &**values {
11468 out.extend_from_slice(&value.to_le_bytes());
11469 }
11470 }
11471 (LogicalType::UBigInt, Data::UInt64(values)) => {
11472 for value in &**values {
11473 out.extend_from_slice(&value.to_le_bytes());
11474 }
11475 }
11476 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11477 for value in &**values {
11478 out.extend_from_slice(&value.to_le_bytes());
11479 }
11480 }
11481 (
11482 LogicalType::BigInt
11483 | LogicalType::Timestamp
11484 | LogicalType::Time
11485 | LogicalType::TimeTz
11486 | LogicalType::TimestampTz
11487 | LogicalType::TimestampS
11488 | LogicalType::TimestampMs
11489 | LogicalType::TimestampNs,
11490 Data::Int64(values),
11491 ) => {
11492 for value in &**values {
11493 out.extend_from_slice(&value.to_le_bytes());
11494 }
11495 }
11496 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11499 for value in &**values {
11500 out.extend_from_slice(&value.to_le_bytes());
11501 }
11502 }
11503 (LogicalType::UHugeInt, Data::UInt128(values)) => {
11504 for value in &**values {
11505 out.extend_from_slice(&value.to_le_bytes());
11506 }
11507 }
11508 (LogicalType::Float, Data::Float32(values)) => {
11511 for value in &**values {
11512 out.extend_from_slice(&value.to_le_bytes());
11513 }
11514 }
11515 (LogicalType::Double, Data::Float64(values)) => {
11516 for value in &**values {
11517 out.extend_from_slice(&value.to_le_bytes());
11518 }
11519 }
11520 (LogicalType::Interval, Data::Interval(values)) => {
11524 for (months, days, micros) in &**values {
11525 out.extend_from_slice(&months.to_le_bytes());
11526 out.extend_from_slice(&days.to_le_bytes());
11527 out.extend_from_slice(µs.to_le_bytes());
11528 }
11529 }
11530 (LogicalType::Boolean, Data::Bool(values)) => {
11531 for value in &**values {
11532 out.push(u8::from(*value));
11533 }
11534 }
11535 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11538 for value in &**values {
11539 out.extend_from_slice(&value.to_le_bytes());
11540 }
11541 }
11542 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11543 for value in &**values {
11544 out.extend_from_slice(&value.to_le_bytes());
11545 }
11546 }
11547 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11548 for value in &**values {
11549 out.extend_from_slice(&value.to_le_bytes());
11550 }
11551 }
11552 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11553 for value in &**values {
11554 out.extend_from_slice(&value.to_le_bytes());
11555 }
11556 }
11557 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11562 let mut bytes = Vec::new();
11563 put_u32(&mut out, 0);
11564 for row in 0..vector.len() {
11565 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11566 bytes.extend_from_slice(value);
11567 put_u32(
11568 &mut out,
11569 u32::try_from(bytes.len())
11570 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11571 );
11572 }
11573 out.extend_from_slice(&bytes);
11574 }
11575 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11576 }
11577 Ok(out)
11578}
11579
11580fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11581 while value >= 0x80 {
11582 out.push((value as u8 & 0x7f) | 0x80);
11583 value >>= 7;
11584 }
11585 out.push(value as u8);
11586}
11587
11588fn unique_codes(codes: &[u32]) -> Vec<u32> {
11590 let mut unique = codes.to_vec();
11591 unique.sort_unstable();
11592 unique.dedup();
11593 unique
11594}
11595
11596fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11602 let mut lists = lists;
11603 while lists.len() > 1 {
11604 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11605 for pair in lists.chunks(2) {
11606 match pair {
11607 [left, right] => next.push(merged_pair(left, right)),
11608 [only] => next.push(only.clone()),
11609 _ => {}
11610 }
11611 }
11612 lists = next;
11613 }
11614 lists.pop().unwrap_or_default()
11615}
11616
11617fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11618 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11619 let mut at = 0;
11620 let mut to = 0;
11621 while at < left.len() && to < right.len() {
11622 match left[at].cmp(&right[to]) {
11623 Ordering::Less => {
11624 out.push(left[at]);
11625 at += 1;
11626 }
11627 Ordering::Greater => {
11628 out.push(right[to]);
11629 to += 1;
11630 }
11631 Ordering::Equal => {
11632 out.push(left[at]);
11633 at += 1;
11634 to += 1;
11635 }
11636 }
11637 }
11638 out.extend_from_slice(&left[at..]);
11639 out.extend_from_slice(&right[to..]);
11640 out
11641}
11642
11643fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11648 let mut merged = Range::default();
11649 let mut first = true;
11650 for range in ranges {
11651 merged.nulls = merged.nulls.saturating_add(range.nulls);
11652 merged.sum = match (merged.sum.take(), range.sum) {
11656 (Some(held), Some(next)) if !first => held.checked_add(next),
11657 (_, next) if first => next,
11658 _ => None,
11659 };
11660 merged.exact = if first { range.exact } else { merged.exact && range.exact };
11661 if first {
11662 merged.low = range.low;
11663 merged.high = range.high;
11664 first = false;
11665 continue;
11666 }
11667 merged.low = match (merged.low.take(), range.low) {
11668 (Some(held), Some(next)) => Some(held.smaller(next)),
11669 _ => None,
11670 };
11671 merged.high = match (merged.high.take(), range.high) {
11672 (Some(held), Some(next)) => Some(held.larger(next)),
11673 _ => None,
11674 };
11675 }
11676 merged
11677}
11678
11679fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11692 match bound {
11693 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11694 value.truncate(PART_BOUND_BYTES);
11695 if !high {
11696 return Some(Bound::Bytes(value));
11697 }
11698 while let Some(last) = value.pop() {
11699 if last < u8::MAX {
11700 value.push(last + 1);
11701 return Some(Bound::Bytes(value));
11702 }
11703 }
11704 None
11705 }
11706 other => other,
11707 }
11708}
11709
11710fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11718 let mut out = Vec::new();
11719 put_u32(
11720 &mut out,
11721 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11722 );
11723 for range in ranges {
11724 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11725 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11726 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11727 }
11728 Ok(out)
11729}
11730
11731fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11733 let mut cur = Cursor::new(bytes);
11734 let parts = cur.u32()? as usize;
11735 let mut out = Vec::new();
11736 for _ in 0..parts {
11737 let low = cur.bound()?;
11738 let high = cur.bound()?;
11739 let nulls = cur.u32()? as usize;
11740 out.push(Range { low, high, nulls, exact: false, sum: None });
11741 }
11742 Ok(out)
11743}
11744
11745fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11746 let held: Vec<&Option<Sieve>> = sieves.collect();
11747 let mut out = Vec::new();
11748 put_u32(
11749 &mut out,
11750 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11751 );
11752 for sieve in &held {
11753 let length = sieve.as_ref().map_or(0, Sieve::len);
11754 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11755 }
11756 for sieve in held.into_iter().flatten() {
11758 out.extend_from_slice(&sieve.to_bytes());
11759 }
11760 Ok(out)
11761}
11762
11763fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11769 let parts = u32::from_le_bytes(
11770 bytes
11771 .get(..4)
11772 .ok_or_else(|| invalid("sieve page is truncated"))?
11773 .try_into()
11774 .map_err(|_| invalid("sieve page is truncated"))?,
11775 ) as usize;
11776 let mut lengths = Vec::with_capacity(parts);
11777 for part in 0..parts {
11778 let at = 4 + part * 4;
11779 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11780 lengths.push(u32::from_le_bytes(
11781 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11782 ) as usize);
11783 }
11784 let mut at = 4 + parts * 4;
11785 let mut out = Vec::with_capacity(parts);
11786 for length in lengths {
11787 if length == 0 {
11788 out.push(None);
11789 continue;
11790 }
11791 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11792 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11793 out.push(Sieve::from_bytes(field));
11794 at = end;
11795 }
11796 if at != bytes.len() {
11797 return Err(invalid("sieve page has trailing bytes"));
11798 }
11799 Ok(out)
11800}
11801
11802fn encode_membership(unique: &[u32]) -> Vec<u8> {
11808 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11809 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11810 let mut previous = 0;
11811 for (at, &code) in unique.iter().enumerate() {
11812 put_varint(&mut out, if at == 0 { code } else { code - previous });
11813 previous = code;
11814 }
11815 out
11816}
11817
11818fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11819 let mut value = 0_u32;
11820 for shift in (0..35).step_by(7) {
11821 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11822 *at += 1;
11823 let part = u32::from(byte & 0x7f);
11824 if shift == 28 && part > 0x0f {
11825 return Err(invalid("membership varint overflow"));
11826 }
11827 value = value
11828 .checked_add(
11829 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11830 )
11831 .ok_or_else(|| invalid("membership varint overflow"))?;
11832 if byte & 0x80 == 0 {
11833 return Ok(value);
11834 }
11835 }
11836 Err(invalid("membership varint is too long"))
11837}
11838
11839fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11840 let mut at = 0;
11841 let count = take_varint(bytes, &mut at)? as usize;
11842 let mut codes = Vec::with_capacity(count);
11843 let mut previous = 0_u32;
11844 for index in 0..count {
11845 let delta = take_varint(bytes, &mut at)?;
11846 let code = if index == 0 {
11847 delta
11848 } else {
11849 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11850 };
11851 if index > 0 && code <= previous {
11852 return Err(invalid("membership codes are not increasing"));
11853 }
11854 codes.push(code);
11855 previous = code;
11856 }
11857 if at != bytes.len() {
11858 return Err(invalid("membership page has trailing bytes"));
11859 }
11860 Ok(codes)
11861}
11862
11863fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11871 let mut by_text: HashMap<&[u8], u32, Spread> =
11872 HashMap::with_capacity_and_hasher(vector.len(), Spread);
11873 let mut values = Vec::new();
11874 let mut codes = Vec::with_capacity(vector.len());
11875 let mut plain_bytes = 0_usize;
11876 for row in 0..vector.len() {
11877 let text = vector.bytes_at(row).unwrap_or(b"");
11878 plain_bytes = plain_bytes.saturating_add(text.len());
11879 let code = match by_text.get(text) {
11880 Some(&code) => code,
11881 None => {
11882 let code = u32::try_from(values.len())
11883 .map_err(|_| invalid("too many dictionary values"))?;
11884 by_text.insert(text, code);
11885 values.push(text);
11886 code
11887 }
11888 };
11889 codes.push(code);
11890 }
11891 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11892 let encoded = 8_usize
11893 .saturating_add((values.len() + 1).saturating_mul(4))
11894 .saturating_add(dictionary_bytes)
11895 .saturating_add(codes.len().saturating_mul(4));
11896 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11897 if encoded >= plain {
11898 return Ok(None);
11899 }
11900 let mut out = Vec::with_capacity(encoded);
11901 put_u32(
11902 &mut out,
11903 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11904 );
11905 put_u32(
11906 &mut out,
11907 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11908 );
11909 let mut offset = 0_u32;
11910 put_u32(&mut out, offset);
11911 for value in &values {
11912 offset = offset
11913 .checked_add(
11914 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11915 )
11916 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11917 put_u32(&mut out, offset);
11918 }
11919 for value in values {
11920 out.extend_from_slice(value);
11921 }
11922 for code in codes {
11923 put_u32(&mut out, code);
11924 }
11925 Ok(Some(out))
11926}
11927
11928struct Room<'a, T> {
11930 state: &'a Mutex<(T, usize)>,
11931 finished: &'a Condvar,
11932 bytes: usize,
11933}
11934
11935impl<T> Drop for Room<'_, T> {
11936 fn drop(&mut self) {
11937 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11938 held.1 -= self.bytes;
11939 drop(held);
11940 self.finished.notify_all();
11941 }
11942}
11943
11944enum Closing<'a> {
11946 Numeric {
11949 column: usize,
11950 counted: bool,
11951 dense: Option<(u64, usize)>,
11952 },
11953 Dictionary {
11954 index: usize,
11955 dictionary: &'a GlobalDictionary,
11956 },
11957}
11958
11959enum Closed {
11961 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11962 Dictionary(usize, ClosedDictionary),
11963}
11964
11965struct ClosedDictionary {
11967 distinct: Option<u64>,
11969 frequencies: Option<FrequencySummary>,
11970 texts: Vec<Option<Vec<u8>>>,
11971 hosts: Option<host::HostSummary>,
11972 encoded: EncodedDictionary,
11973 payload: u64,
11975}
11976
11977struct EncodedDictionary {
11978 index: Vec<u8>,
11979 ranks: Vec<u8>,
11980 grams: Vec<u8>,
11981}
11982
11983fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
12024 let mut work = vec![(0, codes.len(), 0)];
12025 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
12026 while let Some((from, to, depth)) = work.pop() {
12027 let part = &mut codes[from..to];
12028 keyed.clear();
12029 keyed.extend(part.iter().map(|&code| {
12030 let value = values(code);
12031 let rest = value.get(depth..).unwrap_or_default();
12032 (head(rest), rest.len().min(8) as u8, code)
12033 }));
12034 keyed.sort_unstable();
12035 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
12036 *slot = entry.2;
12037 }
12038 let mut start = 0;
12039 while start < keyed.len() {
12040 let (key, taken, _) = keyed[start];
12041 let mut end = start + 1;
12042 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
12043 end += 1;
12044 }
12045 if taken == 8 && end - start > 1 {
12046 work.push((from + start, from + end, depth + 8));
12047 }
12048 start = end;
12049 }
12050 }
12051}
12052
12053const PARALLEL_SORT_MIN: usize = 1 << 16;
12055
12056const BUCKETS_PER_WORKER: usize = 4;
12059
12060const SAMPLES_PER_BUCKET: usize = 32;
12062
12063fn sort_by_value_across<'a>(
12081 codes: &mut [u32],
12082 values: impl Fn(u32) -> &'a [u8] + Sync,
12083 workers: usize,
12084) {
12085 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
12086 sort_by_value(codes, values);
12087 return;
12088 }
12089 let buckets = workers * BUCKETS_PER_WORKER;
12090 let wanted = buckets * SAMPLES_PER_BUCKET;
12091 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
12092 sort_by_value(&mut sample, &values);
12093 let splitters =
12094 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
12095 let values = &values;
12096 let splitters = &splitters;
12097 let per = codes.len().div_ceil(workers);
12098 let places = std::thread::scope(|scope| {
12100 codes
12101 .chunks(per)
12102 .map(|run| {
12103 scope.spawn(move || {
12104 run.iter()
12105 .map(|&code| {
12106 let value = values(code);
12107 splitters.partition_point(|splitter| *splitter <= value) as u32
12108 })
12109 .collect::<Vec<_>>()
12110 })
12111 })
12112 .collect::<Vec<_>>()
12113 .into_iter()
12114 .flat_map(|handle| {
12115 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
12116 })
12117 .collect::<Vec<_>>()
12118 });
12119 let mut starts = vec![0_usize; buckets + 1];
12120 for &place in &places {
12121 starts[place as usize + 1] += 1;
12122 }
12123 for bucket in 0..buckets {
12124 starts[bucket + 1] += starts[bucket];
12125 }
12126 let mut laid = vec![0_u32; codes.len()];
12127 let mut next = starts.clone();
12128 for (&code, &place) in codes.iter().zip(&places) {
12129 laid[next[place as usize]] = code;
12130 next[place as usize] += 1;
12131 }
12132 drop(places);
12133 let mut runs = Vec::with_capacity(buckets);
12134 let mut rest = laid.as_mut_slice();
12135 for bucket in 0..buckets {
12136 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
12137 runs.push(run);
12138 rest = after;
12139 }
12140 runs.sort_by_key(|run| run.len());
12142 let queue = Mutex::new(runs);
12143 std::thread::scope(|scope| {
12144 for _ in 0..workers {
12145 scope.spawn(|| {
12146 loop {
12147 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
12148 let Some(run) = taken else { break };
12149 sort_by_value(run, values);
12150 }
12151 });
12152 }
12153 });
12154 codes.copy_from_slice(&laid);
12155}
12156
12157fn head(bytes: &[u8]) -> u64 {
12165 if let Some(word) = bytes.first_chunk::<8>() {
12166 return u64::from_be_bytes(*word);
12167 }
12168 let len = bytes.len();
12169 if len >= 4 {
12170 let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
12171 let back = &bytes[len - 4..];
12172 let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
12173 return (front << 32) | (back << (8 * (8 - len)));
12174 }
12175 bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
12176}
12177
12178fn encode_global_dictionary(
12189 dictionary: &GlobalDictionary,
12190 order: &[(u64, u32)],
12191 places: &[Placed],
12192 scattered: bool,
12193) -> Result<EncodedDictionary> {
12194 let values = dictionary.values();
12195 if order.len() != values {
12196 return Err(invalid("global dictionary order does not cover its values"));
12197 }
12198 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
12199 if places.len() != blocks {
12200 return Err(invalid("global dictionary payload is not the blocks it says it is"));
12201 }
12202 if dictionary.grams.len() != blocks {
12203 return Err(invalid("global dictionary signatures do not cover its blocks"));
12204 }
12205 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
12206 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
12207 let offset_bits = offset_width(&dictionary.ends);
12208 let payload_words = if scattered { 3 } else { 2 };
12209 let index_len = DICTIONARY_HEADER
12210 .checked_add(offset_bytes(values, offset_bits))
12211 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
12212 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12213 .and_then(|len| len.checked_add(8))
12214 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
12215 let mut index = Vec::with_capacity(index_len);
12216 put_u32(
12217 &mut index,
12218 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
12219 );
12220 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
12221 put_u32(
12222 &mut index,
12223 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
12224 );
12225 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
12226 | DICTIONARY_GRAMS
12227 | DICTIONARY_WIDE_GRAMS;
12228 put_u32(&mut index, offset_bits as u32 | flag);
12229 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
12230 let mut end = 0_u64;
12235 for place in places {
12236 if scattered {
12237 put_u64(&mut index, place.start);
12238 put_u64(&mut index, place.length);
12239 } else {
12240 end = end
12241 .checked_add(place.length)
12242 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
12243 put_u64(&mut index, end);
12244 }
12245 }
12246 for place in places {
12247 put_u64(&mut index, place.hash);
12248 }
12249 if rank_ends.len() != rank_blocks {
12252 return Err(invalid("global dictionary order is not the blocks it says it is"));
12253 }
12254 for end in &rank_ends {
12255 put_u64(&mut index, *end);
12256 }
12257 let mut at = 0_usize;
12258 for end in &rank_ends {
12259 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
12260 put_u64(&mut index, checksum(&ranks[at..end]));
12261 at = end;
12262 }
12263 let gram_len = blocks
12264 .checked_mul(TEXT_GRAM_BYTES)
12265 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
12266 let mut grams = Vec::with_capacity(gram_len);
12267 for block in &dictionary.grams {
12268 grams.extend_from_slice(block);
12269 }
12270 put_u64(&mut index, checksum(&grams));
12271 if index.len() != index_len {
12272 return Err(invalid("global dictionary index is not the length it was laid out for"));
12273 }
12274 Ok(EncodedDictionary { index, ranks, grams })
12275}
12276
12277const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
12284
12285fn payload_shapes() -> Vec<chooser::Settled> {
12311 let integers = vec![integer::Kind::Packed];
12312 [
12313 vec![string::Kind::Front, string::Kind::Lz],
12314 vec![string::Kind::Lz, string::Kind::Fsst],
12315 vec![string::Kind::Lz, string::Kind::Plain],
12316 vec![string::Kind::Fsst],
12317 vec![string::Kind::Plain],
12318 ]
12319 .into_iter()
12320 .map(|strings| chooser::Settled::new(strings, integers.clone()))
12321 .collect()
12322}
12323
12324fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
12331 let started = profile.map(|_| std::time::Instant::now());
12332 file.sync()?;
12333 if let (Some(profile), Some(started)) = (profile, started) {
12334 profile.waited(
12335 Stage::Publish,
12336 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
12337 );
12338 }
12339 Ok(())
12340}
12341
12342#[derive(Debug)]
12347pub(crate) struct Unencoded {
12348 column: usize,
12349 at: usize,
12350 ends: Vec<u32>,
12351 bytes: Vec<u8>,
12352 shape: chooser::Settled,
12353}
12354
12355impl Unencoded {
12356 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12358 let values = block_values(&self.ends, &self.bytes);
12359 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12360 }
12361
12362 pub(crate) fn place(&self) -> (usize, usize) {
12364 (self.column, self.at)
12365 }
12366}
12367
12368pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12372
12373fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12375 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12376 for value in values {
12377 for gram in value.windows(4) {
12378 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12379 grams[bit / 8] |= 1 << (bit % 8);
12380 }
12381 }
12382 }
12383 grams
12384}
12385
12386fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12388 let mut out = Vec::with_capacity(ends.len());
12389 let mut from = 0;
12390 for &to in ends {
12391 out.push(&bytes[from..to as usize]);
12392 from = to as usize;
12393 }
12394 out
12395}
12396
12397fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12404 for dictionary in dictionaries.iter_mut().flatten() {
12405 if !dictionary.early.is_empty() {
12406 return Err(Error::internal("a dictionary block handed out never came back"));
12407 }
12408 dictionary.seal_rest();
12409 dictionary.settle_rest()?;
12410 }
12411 encode_waiting(dictionaries)?;
12412 if dictionaries
12415 .iter()
12416 .flatten()
12417 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12418 {
12419 return Err(Error::internal("a dictionary block handed out never came back"));
12420 }
12421 Ok(())
12422}
12423
12424fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12427 let jobs = dictionaries
12428 .iter()
12429 .enumerate()
12430 .flat_map(|(column, held)| {
12431 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12432 })
12433 .collect::<Vec<_>>();
12434 if jobs.is_empty() {
12435 return Ok(());
12436 }
12437 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12438 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12439 Ok((column, at, held.encode_waiting(at)?))
12440 };
12441 let workers = std::thread::available_parallelism()
12442 .map_or(1, usize::from)
12443 .min(MAX_FREQUENCY_WORKERS)
12444 .min(jobs.len());
12445 let made = if workers <= 1 {
12446 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12447 } else {
12448 let next = AtomicUsize::new(0);
12449 let jobs = &jobs;
12450 let pieces = std::thread::scope(|scope| {
12451 (0..workers)
12452 .map(|_| {
12453 scope.spawn(|| {
12454 let mut mine = Vec::new();
12455 loop {
12456 let job = next.fetch_add(1, Atomic::Relaxed);
12457 let Some(&(column, at)) = jobs.get(job) else { break };
12458 mine.push(one(column, at)?);
12459 }
12460 Ok(mine)
12461 })
12462 })
12463 .collect::<Vec<_>>()
12464 .into_iter()
12465 .map(|handle| {
12466 handle
12467 .join()
12468 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12469 })
12470 .collect::<Result<Vec<_>>>()
12471 })?;
12472 pieces.into_iter().flatten().collect()
12473 };
12474 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12475 (0..dictionaries.len()).map(|_| Vec::new()).collect();
12476 for (column, at, bytes) in made {
12477 done[column].push((at, bytes));
12478 }
12479 for (column, mut made) in done.into_iter().enumerate() {
12480 if made.is_empty() {
12481 continue;
12482 }
12483 let Some(held) = dictionaries[column].as_mut() else { continue };
12484 made.sort_by_key(|(at, _)| *at);
12485 let waiting = std::mem::take(&mut held.waiting);
12486 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12487 if held.encoded() != at {
12488 return Err(Error::internal("a dictionary block was encoded out of order"));
12489 }
12490 held.push_block(block);
12491 }
12492 }
12493 Ok(())
12494}
12495
12496fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12506 let mut best: Option<(chooser::Settled, usize)> = None;
12507 for shape in payload_shapes() {
12508 let mut size = 0;
12509 for block in sample {
12510 size += string::encode_with(block, &shape)?.len();
12511 }
12512 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12513 best = Some((shape, size));
12514 }
12515 }
12516 best.map(|(shape, _)| shape)
12517 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12518}
12519
12520fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12527 let mut out = Vec::with_capacity(order.len() * 4);
12528 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12529 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12530 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12531 for block in order.chunks(TEXT_RANK_BLOCK) {
12532 let base = block.first().map_or(0, |&(head, _)| head);
12535 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12536 let width = (u64::BITS - span.leading_zeros()) as usize;
12537 heads.clear();
12538 codes.clear();
12539 for &(head, code) in block {
12540 heads.push(head.wrapping_sub(base));
12541 codes.push(u64::from(code));
12542 }
12543 put_u64(&mut out, base);
12544 out.push(width as u8);
12545 bitpack::pack_tail(&heads, width, &mut out)
12546 .map_err(|_| invalid("global dictionary heads do not pack"))?;
12547 bitpack::pack_tail(&codes, code_bits, &mut out)
12548 .map_err(|_| invalid("global dictionary codes do not pack"))?;
12549 ends.push(out.len() as u64);
12550 }
12551 Ok((out, ends))
12552}
12553
12554fn open_global_dictionary(
12561 file: Arc<File>,
12562 page: Page,
12563 ty: &LogicalType,
12564 keep_budget: usize,
12565) -> Result<Vector> {
12566 if !coded_type(ty) {
12567 return Err(invalid("global dictionary belongs to a non-string column"));
12568 }
12569 let mut header = [0; DICTIONARY_HEADER];
12570 read_at(&file, page.offset, &mut header)?;
12571 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12572 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12573 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12574 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12575 let scattered = width & DICTIONARY_SCATTERED != 0;
12576 let has_grams = width & DICTIONARY_GRAMS != 0;
12577 let gram_width =
12578 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12579 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12580 if per_block != TEXT_PAYLOAD_VALUES {
12581 return Err(invalid("global dictionary block width differs"));
12582 }
12583 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12584 return Err(invalid("global dictionary block count differs from its value count"));
12585 }
12586 if offset_bits > u32::BITS as usize {
12587 return Err(invalid("global dictionary packs offsets past a payload"));
12588 }
12589 let offset_len = offset_bytes(count, offset_bits);
12590 let ranks = count;
12595 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12596 let payload_words = if scattered { 3 } else { 2 };
12600 let hash_len = blocks
12601 .checked_mul(payload_words * 8)
12602 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12603 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12604 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12605 let gram_len = if has_grams {
12606 blocks
12607 .checked_mul(gram_width)
12608 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12609 } else {
12610 0
12611 };
12612 let index_len = DICTIONARY_HEADER
12613 .checked_add(offset_len)
12614 .and_then(|len| len.checked_add(hash_len))
12615 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12616 if index_len > page.length as usize {
12617 return Err(invalid("global dictionary offset index exceeds its page"));
12618 }
12619 let mut index = vec![0; index_len];
12620 index[..DICTIONARY_HEADER].copy_from_slice(&header);
12621 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12622 if checksum(&index) != page.hash {
12623 return Err(invalid("global dictionary index checksum differs"));
12624 }
12625 let word_end = index_len - usize::from(has_grams) * 8;
12626 let gram_hash = has_grams
12627 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12628 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12629 .chunks_exact(8)
12630 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12631 .collect::<Vec<_>>();
12632 let mut rest = words.split_off(blocks * payload_words);
12633 let rank_hashes = rest.split_off(rank_blocks);
12634 let rank_ends = rest;
12635 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12638 return Err(invalid("global dictionary order blocks do not rise"));
12639 }
12640 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12641 .map_err(|_| invalid("global dictionary rank overflow"))?;
12642 let body_len = index_len
12643 .checked_add(rank_len)
12644 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12645 if body_len > page.length as usize {
12646 return Err(invalid("global dictionary order exceeds its page"));
12647 }
12648 let gram_end = body_len
12649 .checked_add(gram_len)
12650 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12651 if gram_end > page.length as usize {
12652 return Err(invalid("global dictionary signatures exceed their page"));
12653 }
12654 let grams = gram_hash.map(|hash| NativeGrams {
12655 start: page.offset + body_len as u64,
12656 length: gram_len,
12657 width: gram_width,
12658 hash,
12659 verdicts: Mutex::new(Vec::new()),
12660 });
12661 let mut offsets = index;
12665 offsets.truncate(DICTIONARY_HEADER + offset_len);
12666 let hashes = words.split_off(blocks * (payload_words - 1));
12667 let (starts, lengths) = if scattered {
12668 let mut starts = Vec::with_capacity(blocks);
12669 let mut lengths = Vec::with_capacity(blocks);
12670 for pair in words.chunks_exact(2) {
12671 starts.push(pair[0]);
12672 lengths.push(pair[1]);
12673 }
12674 (starts, lengths)
12675 } else {
12676 let base = page.offset + gram_end as u64;
12680 let mut starts = Vec::with_capacity(blocks);
12681 let mut lengths = Vec::with_capacity(blocks);
12682 let mut at = 0_u64;
12683 for &end in &words {
12684 let len = end
12685 .checked_sub(at)
12686 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12687 starts.push(base + at);
12688 lengths.push(len);
12689 at = end;
12690 }
12691 (starts, lengths)
12692 };
12693 let stored_len = page.length as u64 - gram_end as u64;
12699 if scattered && stored_len == 0 {
12700 let size = file.metadata().map_err(io)?.len();
12701 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12702 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12703 });
12704 if !inside {
12705 return Err(invalid("global dictionary block lies outside the file"));
12706 }
12707 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12708 return Err(invalid("global dictionary blocks do not bound the payload"));
12709 }
12710 Vector::external_text(
12711 ty.clone(),
12712 Arc::new(NativeText {
12713 file,
12714 values: count,
12715 offsets,
12716 offset_bits,
12717 value_ends: OnceLock::new(),
12718 value_lens: OnceLock::new(),
12719 ends_asked: AtomicUsize::new(0),
12720 ranks,
12721 rank_at: page.offset + index_len as u64,
12722 rank_ends,
12723 rank_hashes,
12724 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12725 code_bits: code_width(count),
12726 code_ranks: OnceLock::new(),
12727 starts,
12728 lengths,
12729 hashes,
12730 grams,
12731 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12732 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12733 keep_budget,
12734 payload_kept: AtomicUsize::new(0),
12735 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12736 visit_dropped: AtomicUsize::new(0),
12737 searched: Mutex::new(HashMap::new()),
12738 }),
12739 )
12740}
12741
12742fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12755 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12757 let mut cur = Cursor::new(bytes);
12758 let codec = cur.u8()?;
12759 if cur.u8()? == 2 {
12760 cur.take(rows.div_ceil(8))?;
12761 }
12762 Ok((codec, cur.at))
12763 }
12764 let Ok((codec, at)) = cascade_at(rows, bytes) else {
12765 return "UNREADABLE".to_string();
12766 };
12767 let tail = &bytes[at..];
12768 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12769 match codec {
12770 0 => match ty {
12771 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12772 _ => "FIXED".to_string(),
12773 },
12774 1 => "DICT(PLAIN)".to_string(),
12775 2 => "FOR+BITPACK".to_string(),
12776 3 => "TABLE DICT".to_string(),
12777 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12778 5 => described(integer::describe(tail)),
12779 6 => described(string::describe(tail)),
12780 other => format!("CODEC {other}"),
12781 }
12782}
12783
12784fn decode_selected_stable_codes(
12789 rows: usize,
12790 bytes: &[u8],
12791 positions: &[usize],
12792 out: &mut Vec<Option<u32>>,
12793) -> Result<bool> {
12794 if positions.windows(2).any(|pair| pair[0] >= pair[1])
12795 || positions.last().is_some_and(|&position| position >= rows)
12796 {
12797 return Err(invalid("selected code positions are not sorted and in range"));
12798 }
12799 let mut cur = Cursor::new(bytes);
12800 let codec = cur.u8()?;
12801 if codec != 3 && codec != 4 {
12802 return Ok(false);
12803 }
12804 let flag = cur.u8()?;
12805 let mask = match flag {
12806 0 | 1 => None,
12807 2 => {
12808 let at = cur.at;
12809 let len = rows.div_ceil(8);
12810 cur.take(len)?;
12811 Some((at, len))
12812 }
12813 _ => return Err(invalid("page validity tag differs")),
12814 };
12815 let valid = |row: usize| match flag {
12816 0 => true,
12817 1 => false,
12818 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12819 _ => unreachable!("the validity tag was checked"),
12820 };
12821 if codec == 4 {
12822 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12823 for (&row, code) in positions.iter().zip(wide) {
12824 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12825 out.push(valid(row).then_some(code));
12826 }
12827 return Ok(true);
12828 }
12829 let codes_at = cur.at;
12830 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12831 cur.take(codes_len)?;
12832 if cur.at != bytes.len() {
12833 return Err(invalid("global code page has trailing bytes"));
12834 }
12835 let codes = &bytes[codes_at..codes_at + codes_len];
12836 for &row in positions {
12837 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12838 let code = u32::from_le_bytes(
12839 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12840 );
12841 out.push(valid(row).then_some(code));
12842 }
12843 Ok(true)
12844}
12845
12846fn decode_at(
12852 ty: &LogicalType,
12853 rows: usize,
12854 bytes: &[u8],
12855 global: Option<Arc<Vector>>,
12856 positions: &[u32],
12857) -> Result<Vector> {
12858 if positions.last().is_some_and(|&last| last as usize >= rows) {
12859 return Err(invalid("a position is past the end of the part"));
12860 }
12861 if bytes.first() == Some(&5)
12864 && positions.len().saturating_mul(8) <= rows
12865 && bytes
12867 .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12868 .is_some_and(integer::pointed)
12869 {
12870 return cascade_at(ty, rows, bytes, positions);
12871 }
12872 if bytes.first() != Some(&6) {
12873 return decode(ty, rows, bytes, global)?.gather(positions);
12874 }
12875 if !coded_type(ty) {
12876 return Err(invalid("compressed text codec belongs to a non-string page"));
12877 }
12878 let mut cur = Cursor::new(bytes);
12879 cur.u8()?;
12880 let validity = match cur.u8()? {
12881 0 => Validity::AllValid,
12882 1 => Validity::AllInvalid,
12883 2 => {
12884 let mask = cur.take(rows.div_ceil(8))?;
12885 Validity::from_iter(positions.len(), |at| {
12886 let row = positions[at] as usize;
12887 mask[row / 8] >> (row % 8) & 1 == 1
12888 })
12889 }
12890 _ => return Err(invalid("page validity tag differs")),
12891 };
12892 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12893 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12894 push_values(&mut values, ty, &ends)?;
12895 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12896}
12897
12898fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12904 fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12905 values
12906 .iter()
12907 .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12908 .collect()
12909 }
12910 let mut cur = Cursor::new(bytes);
12911 cur.u8()?;
12912 let validity = match cur.u8()? {
12913 0 => Validity::AllValid,
12914 1 => Validity::AllInvalid,
12915 2 => {
12916 let mask = cur.take(rows.div_ceil(8))?;
12917 Validity::from_iter(positions.len(), |at| {
12918 let row = positions[at] as usize;
12919 mask[row / 8] >> (row % 8) & 1 == 1
12920 })
12921 }
12922 _ => return Err(invalid("page validity tag differs")),
12923 };
12924 let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12925 let values = integer::decode_selected(&bytes[cur.at..], &at)
12926 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12927 if values.len() != positions.len() {
12928 return Err(invalid("cascade page holds the wrong number of rows"));
12929 }
12930 let data = match ty {
12931 LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12932 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12933 LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12934 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12935 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12936 LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12937 LogicalType::BigInt
12938 | LogicalType::Timestamp
12939 | LogicalType::Time
12940 | LogicalType::TimeTz
12941 | LogicalType::TimestampTz
12942 | LogicalType::TimestampS
12943 | LogicalType::TimestampMs
12944 | LogicalType::TimestampNs => Data::Int64(values.into()),
12945 LogicalType::Decimal { .. } => match ty.physical() {
12946 PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12947 PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12948 PhysicalType::Int64 => Data::Int64(values.into()),
12949 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12950 },
12951 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12952 };
12953 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12954}
12955
12956fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12960 if ty == &LogicalType::Varchar {
12961 return values.push_run_in_place(0, ends);
12962 }
12963 let mut start = 0;
12964 for &end in ends {
12965 let len = end
12966 .checked_sub(start)
12967 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12968 values.push_bytes_in_place(start, len)?;
12969 start = end;
12970 }
12971 Ok(())
12972}
12973
12974fn decode(
12975 ty: &LogicalType,
12976 rows: usize,
12977 bytes: &[u8],
12978 global: Option<Arc<Vector>>,
12979) -> Result<Vector> {
12980 let mut cur = Cursor::new(bytes);
12981 let codec = cur.u8()?;
12982 let flag = cur.u8()?;
12983 let validity = match flag {
12984 0 => Validity::AllValid,
12985 1 => Validity::AllInvalid,
12986 2 => {
12987 let mask = cur.take(rows.div_ceil(8))?;
12988 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12989 }
12990 _ => return Err(invalid("page validity tag differs")),
12991 };
12992 if codec == 1 {
12993 if !coded_type(ty) {
12994 return Err(invalid("dictionary codec belongs to a non-string page"));
12995 }
12996 let count = cur.u32()? as usize;
12997 let payload_len = cur.u32()? as usize;
12998 let offset_bytes = cur.take(
12999 (count + 1)
13000 .checked_mul(4)
13001 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
13002 )?;
13003 let offsets = offset_bytes
13004 .chunks_exact(4)
13005 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13006 .collect::<Vec<_>>();
13007 let payload = cur.take(payload_len)?.to_vec();
13008 if offsets.first() != Some(&0)
13009 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13010 || offsets.windows(2).any(|pair| pair[0] > pair[1])
13011 {
13012 return Err(invalid("dictionary offsets do not bound the payload"));
13013 }
13014 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
13017 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13018 push_values(&mut strings, ty, &ends)?;
13019 let mut codes = Vec::with_capacity(rows);
13020 for _ in 0..rows {
13021 codes.push(cur.u32()?);
13022 }
13023 if codes.iter().any(|code| *code as usize >= count) {
13024 return Err(invalid("dictionary code is out of range"));
13025 }
13026 if cur.at != bytes.len() {
13027 return Err(invalid("dictionary page has trailing bytes"));
13028 }
13029 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
13030 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
13031 }
13032 if codec == 3 || codec == 4 {
13033 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
13034 let codes = if codec == 4 {
13035 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
13040 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
13041 if codes.len() != rows {
13042 return Err(invalid("encoded code page holds the wrong number of rows"));
13043 }
13044 codes
13045 } else {
13046 let mut codes = Vec::with_capacity(rows);
13047 for _ in 0..rows {
13048 codes.push(cur.u32()?);
13049 }
13050 if cur.at != bytes.len() {
13051 return Err(invalid("global code page has trailing bytes"));
13052 }
13053 codes
13054 };
13055 let highest = codes.iter().copied().max();
13056 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
13057 .with_validity(validity));
13058 }
13059 if codec == 6 {
13060 if !coded_type(ty) {
13061 return Err(invalid("compressed text codec belongs to a non-string page"));
13062 }
13063 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
13067 if ends.len() != rows {
13068 return Err(invalid("compressed text page holds the wrong number of rows"));
13069 }
13070 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13073 push_values(&mut values, ty, &ends)?;
13074 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
13075 }
13076 if codec == 5 {
13077 let data = cascade(ty, &bytes[cur.at..], rows)?;
13079 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
13080 }
13081 if codec == 2 {
13082 let width = u32::from(cur.u8()?);
13083 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
13084 let count = cur.u32()? as usize;
13085 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
13086 let words: Vec<u64> = cur
13087 .take(length)?
13088 .chunks_exact(8)
13089 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
13090 .collect();
13091 if cur.at != bytes.len() {
13092 return Err(invalid("packed page has trailing bytes"));
13093 }
13094 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
13095 }
13096 if codec != 0 {
13097 return Err(invalid("page codec is unknown"));
13098 }
13099 let data = match ty {
13100 LogicalType::TinyInt => {
13101 let values = cur.take(rows)?;
13102 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
13103 }
13104 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
13105 LogicalType::SmallInt => {
13106 let values =
13107 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13108 Data::Int16(
13109 values
13110 .chunks_exact(2)
13111 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13112 .collect::<Vec<_>>()
13113 .into(),
13114 )
13115 }
13116 LogicalType::USmallInt => {
13117 let values =
13118 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13119 Data::UInt16(
13120 values
13121 .chunks_exact(2)
13122 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
13123 .collect::<Vec<_>>()
13124 .into(),
13125 )
13126 }
13127 LogicalType::UInteger => {
13128 let values =
13129 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13130 Data::UInt32(
13131 values
13132 .chunks_exact(4)
13133 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
13134 .collect::<Vec<_>>()
13135 .into(),
13136 )
13137 }
13138 LogicalType::UBigInt => {
13139 let values =
13140 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13141 Data::UInt64(
13142 values
13143 .chunks_exact(8)
13144 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
13145 .collect::<Vec<_>>()
13146 .into(),
13147 )
13148 }
13149 LogicalType::Integer | LogicalType::Date => {
13150 let values =
13151 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13152 Data::Int32(
13153 values
13154 .chunks_exact(4)
13155 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13156 .collect::<Vec<_>>()
13157 .into(),
13158 )
13159 }
13160 LogicalType::BigInt
13161 | LogicalType::Timestamp
13162 | LogicalType::Time
13163 | LogicalType::TimeTz
13164 | LogicalType::TimestampTz
13165 | LogicalType::TimestampS
13166 | LogicalType::TimestampMs
13167 | LogicalType::TimestampNs => {
13168 let values =
13169 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13170 Data::Int64(
13171 values
13172 .chunks_exact(8)
13173 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13174 .collect::<Vec<_>>()
13175 .into(),
13176 )
13177 }
13178 LogicalType::HugeInt | LogicalType::Uuid => {
13179 let values =
13180 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13181 Data::Int128(
13182 values
13183 .chunks_exact(16)
13184 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13185 .collect::<Vec<_>>()
13186 .into(),
13187 )
13188 }
13189 LogicalType::UHugeInt => {
13190 let values =
13191 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13192 Data::UInt128(
13193 values
13194 .chunks_exact(16)
13195 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13196 .collect::<Vec<_>>()
13197 .into(),
13198 )
13199 }
13200 LogicalType::Float => {
13201 let values =
13202 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13203 Data::Float32(
13204 values
13205 .chunks_exact(4)
13206 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
13207 .collect::<Vec<_>>()
13208 .into(),
13209 )
13210 }
13211 LogicalType::Double => {
13212 let values =
13213 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13214 Data::Float64(
13215 values
13216 .chunks_exact(8)
13217 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
13218 .collect::<Vec<_>>()
13219 .into(),
13220 )
13221 }
13222 LogicalType::Interval => {
13223 let values =
13224 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13225 Data::Interval(
13226 values
13227 .chunks_exact(16)
13228 .map(|item| {
13229 (
13230 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
13231 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
13232 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
13233 )
13234 })
13235 .collect::<Vec<_>>()
13236 .into(),
13237 )
13238 }
13239 LogicalType::Boolean => {
13240 let values = cur.take(rows)?;
13241 if values.iter().any(|value| *value > 1) {
13242 return Err(invalid("boolean page has another value"));
13243 }
13244 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
13245 }
13246 LogicalType::Decimal { .. } => match ty.physical() {
13249 PhysicalType::Int16 => {
13250 let values =
13251 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
13252 Data::Int16(
13253 values
13254 .chunks_exact(2)
13255 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
13256 .collect::<Vec<_>>()
13257 .into(),
13258 )
13259 }
13260 PhysicalType::Int32 => {
13261 let values =
13262 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
13263 Data::Int32(
13264 values
13265 .chunks_exact(4)
13266 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
13267 .collect::<Vec<_>>()
13268 .into(),
13269 )
13270 }
13271 PhysicalType::Int64 => {
13272 let values =
13273 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
13274 Data::Int64(
13275 values
13276 .chunks_exact(8)
13277 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
13278 .collect::<Vec<_>>()
13279 .into(),
13280 )
13281 }
13282 _ => {
13283 let values =
13284 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
13285 Data::Int128(
13286 values
13287 .chunks_exact(16)
13288 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
13289 .collect::<Vec<_>>()
13290 .into(),
13291 )
13292 }
13293 },
13294 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
13295 let offset_bytes = cur
13296 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
13297 let offsets = offset_bytes
13298 .chunks_exact(4)
13299 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
13300 .collect::<Vec<_>>();
13301 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
13302 if offsets.first() != Some(&0)
13303 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
13304 || offsets.windows(2).any(|pair| pair[0] > pair[1])
13305 {
13306 return Err(invalid("string offsets do not bound the payload"));
13307 }
13308 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
13316 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
13317 push_values(&mut values, ty, &ends)?;
13318 Data::Varlen(values)
13319 }
13320 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
13321 };
13322 if cur.at != bytes.len() {
13323 return Err(invalid("page has trailing bytes"));
13324 }
13325 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
13326}
13327
13328#[cfg(test)]
13329mod tests {
13330 use std::fs::{self, OpenOptions};
13331 use std::io::{Seek, SeekFrom, Write};
13332 use std::path::PathBuf;
13333 use std::time::{SystemTime, UNIX_EPOCH};
13334
13335 use rudb_common::Stat;
13336 use rudb_common::Value;
13337 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
13338 use rudb_common::stat::Provenance;
13339
13340 use super::*;
13341
13342 #[test]
13343 fn head_is_the_value_padded_to_eight_bytes() {
13344 let bytes: Vec<u8> = (1..=12).collect();
13345 for len in 0..=bytes.len() {
13346 let value = &bytes[..len];
13347 let mut word = [0; 8];
13348 let take = len.min(8);
13349 word[..take].copy_from_slice(&value[..take]);
13350 assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13351 }
13352 assert!(head(b"ab") < head(b"ab\x01"));
13353 assert!(head(b"abcd") < head(b"abce"));
13354 }
13355
13356 #[test]
13357 fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13358 for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13359 let mut bytes = Vec::new();
13360 put_u32(&mut bytes, length);
13361 put_u32(&mut bytes, entries);
13362 bytes.push(1);
13363 assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13364 }
13365 let mut bytes = Vec::new();
13366 put_u32(&mut bytes, 1);
13367 put_u32(&mut bytes, 0);
13368 bytes.push(1);
13369 assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13370 }
13371
13372 #[test]
13373 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13374 let bytes: Vec<u8> =
13375 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13376 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13377 let whole = content_name(&bytes[..length]);
13378 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13379 let mut namer = ContentNamer::default();
13380 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13381 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13382 }
13383 }
13384 }
13385
13386 #[derive(Debug)]
13389 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13390
13391 impl chooser::Chooser for TestsEverything<'_> {
13392 fn name(&self) -> &'static str {
13393 "tests everything"
13394 }
13395
13396 fn narrow_strings(
13397 &self,
13398 values: &[&[u8]],
13399 offered: &[string::Kind],
13400 depth: u8,
13401 ) -> Vec<string::Kind> {
13402 self.0.narrow_strings(values, offered, depth)
13403 }
13404
13405 fn narrow_integers(
13406 &self,
13407 values: &[i64],
13408 offered: &[integer::Kind],
13409 depth: u8,
13410 ) -> Vec<integer::Kind> {
13411 self.0.narrow_integers(values, offered, depth)
13412 }
13413 }
13414
13415 #[test]
13416 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13417 let columns: Vec<Vec<i64>> = vec![
13418 vec![],
13419 vec![5; 1000],
13420 (0..1000).collect(),
13421 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13422 (0..1000).map(|row| row / 50).collect(),
13423 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13424 (0..1000).map(|row| (row * 7919) % 13).collect(),
13425 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13426 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13427 (0..1000).map(|row| i64::MIN + row % 3).collect(),
13428 ];
13429 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13430 for column in &columns {
13431 for chooser in choosers {
13432 let quick = integer::encode_with(column, chooser).unwrap();
13433 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13434 assert_eq!(
13435 quick,
13436 full,
13437 "{} on {:?}",
13438 chooser.name(),
13439 &column[..column.len().min(8)]
13440 );
13441 }
13442 }
13443 }
13444
13445 #[test]
13448 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13449 let mut settling = Settling::default();
13450 for part in 0..STRIPE_PARTS as i64 {
13451 let values: Vec<i64> = (0..2048)
13452 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13453 .collect();
13454 let searched = integer::encode_with(&values, &Fixed).unwrap();
13455 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13456 }
13457 }
13458
13459 #[test]
13463 fn text_pages_share_a_table_until_the_text_changes() {
13464 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13465 let english: Vec<Vec<u8>> = (0..1024)
13466 .map(|row: usize| {
13467 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13468 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13469 })
13470 .collect();
13471 let digits: Vec<Vec<u8>> =
13472 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13473 let mut settling = Settling::default();
13474 for page in 0..8 {
13475 let values: Vec<&[u8]> =
13476 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13477 let payload = values.iter().map(|value| value.len()).sum();
13478 let out = settling.text(&values, payload).unwrap().unwrap();
13479 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13480 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13481 assert!(
13482 out.len() * 4 <= alone.len() * 5,
13483 "page {page}: {} against {}",
13484 out.len(),
13485 alone.len()
13486 );
13487 let since = settling.symbols.as_ref().unwrap().since;
13488 assert_eq!(since, page % 4, "page {page}");
13489 }
13490 }
13491
13492 #[test]
13496 fn a_column_that_changes_under_the_shape_is_searched_again() {
13497 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13498 let mut noise = move || {
13499 state ^= state << 13;
13500 state ^= state >> 7;
13501 state ^= state << 17;
13502 (state % 1_000_000) as i64
13503 };
13504 let mut settling = Settling::default();
13505 for part in 0..STRIPE_PARTS as i64 {
13506 let values: Vec<i64> = match part / 16 {
13507 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13508 1 => (0..2048).map(|_| noise()).collect(),
13509 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13510 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13511 };
13512 let settled = settling.encode(&values).unwrap();
13513 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13514 let searched = integer::encode_with(&values, &Fixed).unwrap();
13515 assert!(
13516 settled.len() * 4 <= searched.len() * 5,
13517 "part {part}: {} settled against {} searched, {} against {}",
13518 settled.len(),
13519 searched.len(),
13520 integer::describe(&settled).unwrap(),
13521 integer::describe(&searched).unwrap(),
13522 );
13523 }
13524 }
13525
13526 #[test]
13527 fn checksum_matches_fixed_vectors() {
13528 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13529 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13530 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13531 }
13532
13533 #[test]
13534 fn sorting_across_threads_matches_sorting_on_one() {
13535 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13536 let mut next = move || {
13537 state ^= state << 13;
13538 state ^= state >> 7;
13539 state ^= state << 17;
13540 state
13541 };
13542 let mut values = Vec::new();
13543 for at in 0..150_000_u64 {
13544 let value = match next() % 6 {
13545 0 => Vec::new(),
13546 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13547 2 => format!("https://example.com/path/{at}").into_bytes(),
13548 3 => b"same".to_vec(),
13549 4 => vec![0xff; (next() % 12) as usize],
13550 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13551 };
13552 values.push(value);
13553 }
13554 let value = |code: u32| values[code as usize].as_slice();
13555 for workers in [1, 2, 3, 8, 32] {
13556 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13557 let mut across = one.clone();
13558 sort_by_value(&mut one, value);
13559 sort_by_value_across(&mut across, value, workers);
13560 assert_eq!(one, across, "{workers} workers");
13561 }
13562 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13563 sort_by_value_across(&mut sorted, value, 8);
13564 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13565 }
13566
13567 fn path(label: &str) -> PathBuf {
13568 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13569 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13570 }
13571
13572 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13577 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13578 (0..dictionary.values())
13579 .map(|code| {
13580 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13581 flat[from..to].to_vec()
13582 })
13583 .collect()
13584 }
13585
13586 fn attached(table: &Table) -> Vec<&Section> {
13593 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13594 }
13595
13596 #[test]
13598 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13599 const SPANS: usize = 64;
13600 const SPAN: usize = 512;
13601 let path = path("positional");
13602 let content: Vec<u8> =
13603 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13604 fs::write(&path, &content).expect("the file is written");
13605 let file = Arc::new(File::open(&path).expect("the file opens"));
13606 std::thread::scope(|scope| {
13607 for _ in 0..8 {
13608 let file = Arc::clone(&file);
13609 scope.spawn(move || {
13610 for _ in 0..64 {
13611 for span in 0..SPANS {
13612 let mut bytes = [0_u8; SPAN];
13613 read_at(&file, (span * SPAN) as u64, &mut bytes)
13614 .expect("the span reads");
13615 assert!(
13616 bytes.iter().all(|byte| *byte == span as u8),
13617 "span {span} came back as {}",
13618 bytes[0],
13619 );
13620 }
13621 }
13622 });
13623 }
13624 });
13625 let mut past = [0_u8; SPAN];
13626 let end = (SPANS * SPAN) as u64;
13627 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13628 assert!(error.message().contains("ends before its declared length"), "{error}");
13629 drop(file);
13630 let _ = fs::remove_file(&path);
13631 }
13632
13633 #[test]
13640 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13641 let path = path("cursor");
13642 let mut writer = Writer::create(
13643 &path,
13644 "items",
13645 vec![
13646 Field::required("id", LogicalType::Integer),
13647 Field::new("text", LogicalType::Varchar),
13648 ],
13649 )
13650 .expect("new file");
13651 writer.append(&sample()).expect("first part");
13652 writer.append(&sample()).expect("second part");
13653 writer.finish().expect("commit");
13654 let reader = Reader::open(&path).expect("reopen from disk");
13655 assert_eq!(reader.table().rows(), 6);
13656 let ids = reader.read(0, &[0]).expect("the integer page reads back");
13657 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13658 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13659 let text = reader.read(1, &[1]).expect("the text page reads back");
13660 assert_eq!(text.value_at(1, 0), Value::Null);
13661 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13662 let end = reader.table().stripes().iter().flat_map(|stripe| {
13665 stripe
13666 .pages
13667 .iter()
13668 .map(|page| page.offset + u64::from(page.length))
13669 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13670 });
13671 let last = end.fold(HEADER, u64::max);
13672 let directory = fs::metadata(&path).expect("the file is there").len();
13673 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13674 fs::remove_file(path).expect("remove scratch file");
13675 }
13676
13677 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13683 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13684 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13685 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13686 let bits = (width & !DICTIONARY_FLAGS) as usize;
13687 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13688 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13689 DICTIONARY_HEADER as u64
13690 + offset_bytes(count as usize, bits) as u64
13691 + blocks * payload_words * 8
13692 + rank_blocks * 16
13693 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13694 }
13695
13696 fn sample() -> Chunk {
13697 Chunk::new(vec![
13698 Vector::from_values(
13699 LogicalType::Integer,
13700 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13701 )
13702 .expect("integers"),
13703 Vector::from_values(
13704 LogicalType::Varchar,
13705 &[
13706 Value::Varchar("alpha".into()),
13707 Value::Null,
13708 Value::Varchar("long text after a slash".into()),
13709 ],
13710 )
13711 .expect("strings"),
13712 ])
13713 .expect("matching rows")
13714 }
13715
13716 fn sample_ids() -> Chunk {
13717 Chunk::new(vec![
13718 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13719 .expect("integers"),
13720 ])
13721 .expect("one column")
13722 }
13723
13724 #[test]
13725 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13726 let path = path("nulls_for_the_planner");
13729 let mut writer =
13730 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13731 .expect("new file");
13732 let rows = Chunk::new(vec![
13733 Vector::from_values(
13734 LogicalType::Integer,
13735 &[
13736 Value::Integer(4),
13737 Value::Null,
13738 Value::Integer(9),
13739 Value::Null,
13740 Value::Integer(1),
13741 Value::Integer(2),
13742 ],
13743 )
13744 .expect("integers"),
13745 ])
13746 .expect("one column");
13747 writer.append(&rows).expect("the only part");
13748 writer.finish().expect("commit");
13749 let reader = Reader::open(&path).expect("reopen from disk");
13750 let stripes = Stripes::new(reader);
13751 let column = stripes.column("a").expect("the file has that column");
13752 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13753 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13756 fs::remove_file(&path).expect("clean up");
13757 }
13758
13759 #[test]
13760 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13761 let path = path("frequencies_for_the_planner");
13764 let mut writer =
13765 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13766 .expect("new file");
13767 let rows = Chunk::new(vec![
13768 Vector::from_values(
13769 LogicalType::Integer,
13770 &[
13771 Value::Integer(4),
13772 Value::Integer(4),
13773 Value::Integer(4),
13774 Value::Integer(9),
13775 Value::Integer(9),
13776 Value::Integer(1),
13777 ],
13778 )
13779 .expect("integers"),
13780 ])
13781 .expect("one column");
13782 writer.append(&rows).expect("the only part");
13783 writer.finish().expect("commit");
13784 let reader = Reader::open(&path).expect("reopen from disk");
13785 let common = Common::new(reader);
13786 assert_eq!(common.rows(), 6);
13787 let column = common.column("id").expect("the file has that column");
13788 assert_eq!(common.column("nothing"), None);
13789 assert_eq!(
13790 common.rows_with(column, &Bound::Int(4)),
13791 Stat::exact(3, Provenance::FrequencySynopsis)
13792 );
13793 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13795 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13798 assert!(common.remainder(column).is_some());
13799 fs::remove_file(&path).expect("clean up");
13800 }
13801
13802 #[test]
13803 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13804 let path = path("string_frequencies_for_the_planner");
13805 let mut writer =
13806 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13807 .expect("new file");
13808 let rows = Chunk::new(vec![
13809 Vector::from_values(
13810 LogicalType::Varchar,
13811 &[
13812 Value::Varchar(String::new()),
13813 Value::Varchar("alpha".into()),
13814 Value::Varchar(String::new()),
13815 Value::Varchar("beta".into()),
13816 Value::Varchar(String::new()),
13817 ],
13818 )
13819 .expect("strings"),
13820 ])
13821 .expect("one column");
13822 writer.append(&rows).expect("the only part");
13823 writer.finish().expect("commit");
13824
13825 let reader = Reader::open(&path).expect("reopen from disk");
13826 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13827 let common = Common::new(reader.clone());
13828 let column = common.column("text").expect("the file has that column");
13829 assert_eq!(
13830 common.rows_with(column, &Bound::Bytes(Vec::new())),
13831 Stat::exact(3, Provenance::FrequencySynopsis)
13832 );
13833 assert_eq!(
13834 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13835 Stat::exact(0, Provenance::FrequencySynopsis)
13836 );
13837 assert_eq!(
13838 reader.reads().dictionaries,
13839 0,
13840 "the bounded spellings answer without opening the dictionary index"
13841 );
13842 fs::remove_file(&path).expect("clean up");
13843 }
13844
13845 #[test]
13846 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13847 let path = path("certified_host_groups");
13848 let mut writer =
13849 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13850 .expect("new file");
13851 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13852 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13853 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13854 values.push(Value::Varchar(String::new()));
13855 for part in values.chunks(512) {
13856 writer
13857 .append(
13858 &Chunk::new(vec![
13859 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13860 ])
13861 .expect("one column"),
13862 )
13863 .expect("part written");
13864 }
13865 writer.finish().expect("commit");
13866 let reader = Reader::open(&path).expect("reopen");
13867 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13868 fs::remove_file(&path).expect("clean up");
13869 }
13870
13871 fn bare_table(sections: Vec<Section>) -> Table {
13876 Table {
13877 name: "linked".to_owned(),
13878 fields: vec![Field::required("id", LogicalType::Integer)],
13879 stripes: Vec::new(),
13880 rows: 0,
13881 dictionaries: vec![None],
13882 dictionary_payloads: Vec::new(),
13883 demoted: Vec::new(),
13884 distincts: vec![None],
13885 frequencies: vec![None],
13886 ordinal_bounds: Vec::new(),
13887 pair_frequencies: Vec::new(),
13888 frequency_texts: Vec::new(),
13889 host_groups: None,
13890 clustering: None,
13891 constraints: Constraints::default(),
13892 generation: 1,
13893 sections,
13894 }
13895 }
13896
13897 fn a_key_map_section() -> Section {
13898 Section {
13899 kind: *section::KEY_MAP,
13900 id: 1,
13901 generation: 3,
13902 extents: 1,
13903 extent_page: HEADER,
13904 extent_bytes: section::EXTENT_BYTES as u32,
13905 hash: 0x1234_5678_9abc_def0,
13906 flags: 0,
13907 header_bytes: 24,
13908 }
13909 }
13910
13911 #[test]
13912 fn a_section_table_round_trips_through_a_directory() {
13913 let mut later = a_key_map_section();
13914 later.kind = *b"RUDBZZ9\0";
13915 later.id = 2;
13916 let table = bare_table(vec![a_key_map_section(), later]);
13917 let directory = encode_directory(&table).expect("directory");
13918 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13919 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13920 assert!(decoded.sections()[0].known());
13924 assert!(!decoded.sections()[1].known());
13925 }
13926
13927 #[test]
13928 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13929 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13933 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13934 let older = &directory[..directory.len() - block];
13935 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13936 assert!(decoded.sections().is_empty());
13937 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13938 assert_eq!(decoded.name(), "linked");
13939 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13940 }
13941
13942 #[test]
13943 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13944 let path = path("format_twenty_two");
13951 let mut writer =
13952 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13953 .expect("new file");
13954 let rows = Chunk::new(vec![
13955 Vector::from_values(
13956 LogicalType::Integer,
13957 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13958 )
13959 .expect("integers"),
13960 ])
13961 .expect("one column");
13962 writer.append(&rows).expect("the only part");
13963 writer.finish().expect("commit");
13964
13965 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13966 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13967 drop(file);
13968
13969 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13970 assert_eq!(reader.table().rows(), 3);
13971 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13976
13977 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13980 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13981 drop(file);
13982 let error = Reader::open(&path).expect_err("format 21 is not readable");
13983 assert!(error.to_string().contains("format 21"), "{error}");
13984
13985 fs::remove_file(&path).expect("clean up");
13986 }
13987
13988 #[test]
13989 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13990 let mut past = a_key_map_section();
13995 past.extent_page = 1 << 30;
13996 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13997 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13998 assert!(error.to_string().contains("outside the file"), "{error}");
13999
14000 let mut inside_the_header = a_key_map_section();
14001 inside_the_header.extent_page = 8;
14002 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
14003 assert!(
14004 decode_directory(&directory, 1 << 20).is_err(),
14005 "a section may not overlap a header"
14006 );
14007 }
14008
14009 #[test]
14010 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
14011 let not_built = Section {
14015 kind: *section::FORWARD_LINK,
14016 id: 9,
14017 generation: 3,
14018 extents: 0,
14019 extent_page: 0,
14020 extent_bytes: 0,
14021 hash: 0,
14022 flags: 0,
14023 header_bytes: 0,
14024 };
14025 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
14026 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
14027 assert_eq!(decoded.sections(), &[not_built]);
14028
14029 let mut incoherent = not_built;
14032 incoherent.extent_bytes = 28;
14033 incoherent.extent_page = HEADER;
14034 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
14035 assert!(decode_directory(&directory, 1 << 20).is_err());
14036 }
14037
14038 #[test]
14039 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
14040 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
14041 let mut torn = directory.clone();
14042 let count_at = torn.len() - size_of::<u16>();
14043 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
14044 assert!(decode_directory(&torn, 1 << 20).is_err());
14047 }
14048
14049 fn linked_file(label: &str, rows: i32) -> PathBuf {
14051 let path = path(label);
14052 let mut writer =
14053 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14054 .expect("new file");
14055 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
14056 let chunk =
14057 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
14058 .expect("one column");
14059 writer.append(&chunk).expect("the only part");
14060 writer.finish().expect("commit");
14061 path
14062 }
14063
14064 fn a_key_map_payload() -> Vec<u8> {
14065 (0..512_u32).flat_map(u32::to_le_bytes).collect()
14068 }
14069
14070 #[test]
14071 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
14072 let path = linked_file("attach", 64);
14073 let payload = a_key_map_payload();
14074 let table = attach(
14075 &path,
14076 "items",
14077 &[section::Attachment {
14078 kind: *section::KEY_MAP,
14079 id: 0,
14080 flags: 2,
14081 header_bytes: 40,
14082 bytes: &payload,
14083 }],
14084 )
14085 .expect("attach a key map");
14086 assert_eq!(attached(&table).len(), 1);
14087
14088 let reader = Reader::open(&path).expect("reopen after the attach");
14089 let held = attached(reader.table());
14090 assert_eq!(held.len(), 1);
14091 assert_eq!(held[0].kind, *section::KEY_MAP);
14092 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
14093 assert_eq!(held[0].header_bytes, 40);
14094 assert_eq!(held[0].generation, 1);
14098 assert!(held[0].usable(reader.table().generation()));
14099 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
14100 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
14101
14102 fs::remove_file(&path).expect("clean up");
14103 }
14104
14105 #[test]
14106 fn attaching_a_section_answers_every_row_exactly_as_before() {
14107 let path = linked_file("attach_changes_nothing", 300);
14112 let before = Reader::open(&path).expect("open before");
14113 let rows = before.table().rows();
14114 let first = before.read(0, &[0]).expect("read before");
14115 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
14116 let layout = before.layout().columns_total();
14117 drop(before);
14118
14119 let payload = a_key_map_payload();
14120 attach(
14121 &path,
14122 "items",
14123 &[section::Attachment {
14124 kind: *section::KEY_MAP,
14125 id: 0,
14126 flags: 0,
14127 header_bytes: 0,
14128 bytes: &payload,
14129 }],
14130 )
14131 .expect("attach");
14132
14133 let after = Reader::open(&path).expect("open after");
14134 assert_eq!(after.table().rows(), rows);
14135 let read = after.read(0, &[0]).expect("read after");
14136 for (at, value) in values.iter().enumerate() {
14137 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
14138 }
14139 assert_eq!(
14140 after.layout().columns_total(),
14141 layout,
14142 "an attach appends and does not rewrite a column page"
14143 );
14144
14145 fs::remove_file(&path).expect("clean up");
14146 }
14147
14148 #[test]
14149 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
14150 let path = linked_file("attach_twice", 32);
14154 let one = a_key_map_payload();
14155 let two = vec![7_u8; 1024];
14156 let entry = |bytes| section::Attachment {
14157 kind: *section::KEY_MAP,
14158 id: 4,
14159 flags: 1,
14160 header_bytes: 0,
14161 bytes,
14162 };
14163 attach(&path, "items", &[entry(&one)]).expect("first build");
14164 attach(&path, "items", &[entry(&two)]).expect("rebuild");
14165
14166 let reader = Reader::open(&path).expect("reopen");
14167 let held = attached(reader.table());
14168 assert_eq!(held.len(), 1, "one map per column and not one per build");
14169 assert_eq!(reader.payload(held[0]).expect("payload"), two);
14170
14171 fs::remove_file(&path).expect("clean up");
14172 }
14173
14174 #[test]
14175 fn an_attach_carries_through_a_kind_it_does_not_know() {
14176 let path = linked_file("attach_unknown", 16);
14180 let payload = vec![3_u8; 96];
14181 attach(
14182 &path,
14183 "items",
14184 &[section::Attachment {
14185 kind: *b"RUDBZZ9\0",
14186 id: 1,
14187 flags: 0,
14188 header_bytes: 0,
14189 bytes: &payload,
14190 }],
14191 )
14192 .expect("a kind this build does not know still writes");
14193 let key_map = a_key_map_payload();
14194 attach(
14195 &path,
14196 "items",
14197 &[section::Attachment {
14198 kind: *section::KEY_MAP,
14199 id: 0,
14200 flags: 0,
14201 header_bytes: 0,
14202 bytes: &key_map,
14203 }],
14204 )
14205 .expect("attach beside it");
14206
14207 let reader = Reader::open(&path).expect("reopen");
14208 let held = attached(reader.table());
14209 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
14210 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
14211 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
14212
14213 fs::remove_file(&path).expect("clean up");
14214 }
14215
14216 #[test]
14217 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
14218 let path = linked_file("attach_not_built", 8);
14219 attach(
14220 &path,
14221 "items",
14222 &[section::Attachment {
14223 kind: *section::FORWARD_LINK,
14224 id: 2,
14225 flags: 0,
14226 header_bytes: 0,
14227 bytes: &[],
14228 }],
14229 )
14230 .expect("record a link that did not fit the budget");
14231
14232 let reader = Reader::open(&path).expect("reopen");
14233 let held = attached(reader.table());
14234 assert_eq!(held.len(), 1);
14235 assert_eq!(held[0].extents, 0);
14236 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
14237 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
14238 assert!(reader.payload(held[0]).expect("no payload").is_empty());
14239
14240 fs::remove_file(&path).expect("clean up");
14241 }
14242
14243 #[test]
14244 fn a_payload_past_one_extent_is_split_and_joined_back() {
14245 let path = linked_file("attach_two_extents", 8);
14249 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
14250 attach(
14251 &path,
14252 "items",
14253 &[section::Attachment {
14254 kind: *section::KEY_MAP,
14255 id: 0,
14256 flags: 0,
14257 header_bytes: 0,
14258 bytes: &payload,
14259 }],
14260 )
14261 .expect("attach a payload past the bound");
14262
14263 let reader = Reader::open(&path).expect("reopen");
14264 let held = attached(reader.table());
14265 let extents = reader.extents(held[0]).expect("extent table");
14266 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
14267 assert_eq!(extents[0].length, section::MAX_EXTENT);
14268 assert_eq!(extents[1].length, 1);
14269 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
14270 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
14272 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
14273
14274 fs::remove_file(&path).expect("clean up");
14275 }
14276
14277 #[test]
14278 fn a_torn_extent_is_refused_rather_than_decoded() {
14279 let path = linked_file("attach_torn", 8);
14280 let payload = a_key_map_payload();
14281 attach(
14282 &path,
14283 "items",
14284 &[section::Attachment {
14285 kind: *section::KEY_MAP,
14286 id: 0,
14287 flags: 0,
14288 header_bytes: 0,
14289 bytes: &payload,
14290 }],
14291 )
14292 .expect("attach");
14293
14294 let reader = Reader::open(&path).expect("reopen");
14295 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
14296 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
14297 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
14298 drop(file);
14299
14300 let reader = Reader::open(&path).expect("the table still opens");
14301 let error = reader
14302 .payload(&reader.table().sections()[0])
14303 .expect_err("a corrupt payload is not handed out");
14304 assert!(error.to_string().contains("checksum"), "{error}");
14305 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
14308
14309 fs::remove_file(&path).expect("clean up");
14310 }
14311
14312 #[test]
14313 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
14314 let path = linked_file("attach_old_format", 8);
14317 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
14318 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
14319 drop(file);
14320
14321 let payload = a_key_map_payload();
14322 let error = attach(
14323 &path,
14324 "items",
14325 &[section::Attachment {
14326 kind: *section::KEY_MAP,
14327 id: 0,
14328 flags: 0,
14329 header_bytes: 0,
14330 bytes: &payload,
14331 }],
14332 )
14333 .expect_err("format 22 cannot gain a section");
14334 assert!(error.to_string().contains("format 22"), "{error}");
14335 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
14336
14337 fs::remove_file(&path).expect("clean up");
14338 }
14339
14340 #[test]
14341 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
14342 let path = linked_file("attach_bad_header", 8);
14343 let error = attach(
14344 &path,
14345 "items",
14346 &[section::Attachment {
14347 kind: *section::KEY_MAP,
14348 id: 0,
14349 flags: 0,
14350 header_bytes: 40,
14351 bytes: &[1, 2, 3],
14352 }],
14353 )
14354 .expect_err("a writer's bug stops at the write");
14355 assert!(error.to_string().contains("header is longer"), "{error}");
14356
14357 fs::remove_file(&path).expect("clean up");
14358 }
14359
14360 #[test]
14361 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14362 let path = linked_file("attach_wrong_name", 8);
14363 let error = attach(&path, "orders", &[]).expect_err("no such table");
14364 assert!(error.to_string().contains("orders"), "{error}");
14365 fs::remove_file(&path).expect("clean up");
14366 }
14367
14368 #[test]
14369 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14370 let path = path("frequency_prefix_for_the_planner");
14377 let mut writer =
14378 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14379 .expect("new file");
14380 let mut values = vec![Value::Integer(1); 10_000];
14381 for _ in 0..10 {
14382 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14383 }
14384 for part in values.chunks(8_000) {
14387 let rows = Chunk::new(vec![
14388 Vector::from_values(LogicalType::Integer, part).expect("integers"),
14389 ])
14390 .expect("one column");
14391 writer.append(&rows).expect("a part");
14392 }
14393 writer.finish().expect("commit");
14394 let reader = Reader::open(&path).expect("reopen from disk");
14395 let prefix =
14396 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14397 assert_eq!(prefix.entries.len(), 512);
14400 assert_eq!(prefix.omitted_max, 10);
14401 let common = Common::new(reader);
14402 assert_eq!(common.rows(), 16_000);
14403 let column = common.column("id").expect("the file has that column");
14404 assert_eq!(
14405 common.rows_with(column, &Bound::Int(1)),
14406 Stat::exact(10_000, Provenance::FrequencySynopsis)
14407 );
14408 assert_eq!(
14410 common.rows_with(column, &Bound::Int(1_100)),
14411 Stat::exact(10, Provenance::FrequencySynopsis)
14412 );
14413 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14416 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14419 let remainder = common.remainder(column).expect("the list is a prefix");
14423 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14424 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14425 fs::remove_file(&path).expect("clean up");
14426 }
14427
14428 #[test]
14430 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14431 let path = path("empty");
14432 Writer::empty(&path, &[], None).expect("a file with nothing in it");
14433 let catalog = Catalog::open(&path).expect("the empty file opens");
14434 assert_eq!(catalog.len(), 0);
14435 assert!(catalog.is_empty());
14436 assert_eq!(catalog.names().count(), 0);
14437 let mut writer =
14440 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14441 .expect("a table goes into the empty file");
14442 writer.append(&sample_ids()).expect("rows");
14443 writer.finish().expect("commit");
14444 let catalog = Catalog::open(&path).expect("the file opens again");
14445 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14446 fs::remove_file(&path).expect("clean up");
14447 }
14448
14449 #[test]
14459 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14460 let path = path("empty-name");
14461 let field = || vec![Field::required("id", LogicalType::Integer)];
14462 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14463 let catalog = Catalog::open(&path).expect("the file opens");
14464 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14465
14466 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14467 writer.append(&sample_ids()).expect("rows");
14468 writer.finish().expect("commit");
14469 let catalog = Catalog::open(&path).expect("the file opens again");
14470 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14472 let held = catalog.rows().collect::<Vec<_>>();
14473 assert_eq!(held.len(), 1);
14474 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14475
14476 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14478 assert!(error.to_string().contains("same name"), "{error}");
14479 fs::remove_file(&path).expect("clean up");
14480 }
14481
14482 #[test]
14483 fn a_log_anchor_rides_the_catalog_after_the_card_and_goes_forward_with_every_commit() {
14484 let entry = || Entry {
14485 name: "items".to_string(),
14486 fields: vec![Field::required("id", LogicalType::Integer)],
14487 rows: 1,
14488 directory: Page { offset: HEADER, length: 8, hash: 0 },
14489 nonzero: vec![None],
14490 aggregates: vec![None],
14491 distincts: vec![None],
14492 extremes: vec![None],
14493 frequencies: vec![None],
14494 };
14495 let anchor = LogAnchor {
14496 database: 0xfeed,
14497 durable: 41,
14498 lanes: vec![LaneStart { sequence: 3, offset: 4096 }],
14499 voids: vec![43, 47],
14500 };
14501 assert!(!anchor.replays(41) && anchor.replays(42) && !anchor.replays(43));
14502 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14503 for card in [None, Some(&card)] {
14504 let bytes = encode_catalog(&[entry()], &[], card, Some(&anchor)).expect("encodes");
14505 let (_, _, kept, held) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14506 assert_eq!((kept.as_ref(), held.as_ref()), (card, Some(&anchor)));
14507 }
14508 let mut twice = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14509 anchor.encode(&mut twice).expect("encodes");
14510 assert!(decode_catalog(&twice, HEADER + 8).is_err(), "a second anchor");
14511 let mut after = encode_catalog(&[entry()], &[], None, Some(&anchor)).expect("encodes");
14512 after.extend_from_slice(DEVICE_CARD);
14513 assert!(decode_catalog(&after, HEADER + 8).is_err(), "a card after the anchor");
14514 let under = LogAnchor { voids: vec![40], ..anchor.clone() };
14515 let bytes = encode_catalog(&[entry()], &[], None, Some(&under)).expect("encodes");
14516 assert!(decode_catalog(&bytes, HEADER + 8).is_err(), "a void under the cut");
14517
14518 let path = path("anchored");
14519 let mut writer =
14520 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14521 .expect("new file");
14522 writer.append(&sample_ids()).expect("rows");
14523 writer.with_log_anchor(anchor.clone()).finish().expect("commit");
14524 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14525 Writer::restate(&path, &[sample_view("v")], None).expect("a view");
14526 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14527 let mut writer =
14528 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14529 .expect("a second table");
14530 writer.append(&sample_ids()).expect("rows");
14531 writer.finish().expect("commit");
14532 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&anchor));
14533 let next = LogAnchor { durable: 90, voids: Vec::new(), ..anchor };
14534 Writer::restate(&path, &[], Some(&next)).expect("a new cut");
14535 assert_eq!(Catalog::open(&path).expect("reopen").log_anchor(), Some(&next));
14536 fs::remove_file(&path).expect("clean up");
14537 let empty = self::path("anchoredempty");
14538 Writer::empty(&empty, &[], Some(&next)).expect("an empty file");
14539 assert_eq!(Catalog::open(&empty).expect("reopen").log_anchor(), Some(&next));
14540 fs::remove_file(&empty).expect("clean up");
14541 }
14542
14543 #[test]
14544 fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14545 let entry = || Entry {
14546 name: "items".to_string(),
14547 fields: vec![Field::required("id", LogicalType::Integer)],
14548 rows: 1,
14549 directory: Page { offset: HEADER, length: 8, hash: 0 },
14550 nonzero: vec![None],
14551 aggregates: vec![None],
14552 distincts: vec![None],
14553 extremes: vec![None],
14554 frequencies: vec![None],
14555 };
14556 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14557 let bytes = encode_catalog(&[entry()], &[], Some(&card), None).expect("encodes");
14558 let (entries, views, kept, _) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14559 assert_eq!((entries.len(), views.len()), (1, 0));
14560 assert_eq!(kept, Some(card));
14561 let bytes = encode_catalog(&[entry()], &[], None, None).expect("encodes");
14562 assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14563 }
14564
14565 fn sample_view(name: &str) -> ViewEntry {
14567 ViewEntry {
14568 name: name.to_string(),
14569 sql: "SELECT id FROM items WHERE id > 0".to_string(),
14570 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14571 aliases: vec!["n".to_string()],
14572 columns: vec![Field::new("n", LogicalType::Integer)],
14573 }
14574 }
14575
14576 #[test]
14577 fn a_view_written_into_the_catalog_comes_back_whole() {
14578 let path = path("views");
14579 let mut writer =
14580 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14581 .expect("new file");
14582 writer.append(&sample_ids()).expect("rows");
14583 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14584 let catalog = Catalog::open(&path).expect("reopen");
14585 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14586 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14589 fs::remove_file(&path).expect("clean up");
14590 }
14591
14592 #[test]
14594 fn appending_a_table_carries_the_views_forward() {
14595 let path = path("viewscarry");
14596 let mut writer =
14597 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14598 .expect("new file");
14599 writer.append(&sample_ids()).expect("rows");
14600 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14601 let mut writer =
14602 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14603 .expect("a second table");
14604 writer.append(&sample_ids()).expect("rows");
14605 writer.finish().expect("commit");
14606 let catalog = Catalog::open(&path).expect("reopen");
14607 assert_eq!(catalog.views().count(), 1);
14608 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14609 fs::remove_file(&path).expect("clean up");
14610 }
14611
14612 #[test]
14614 fn restating_the_views_leaves_every_table_where_it_was() {
14615 let path = path("restate");
14616 let mut writer =
14617 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14618 .expect("new file");
14619 writer.append(&sample_ids()).expect("rows");
14620 writer.finish().expect("commit");
14621 let before = fs::metadata(&path).expect("the file is there").len();
14622 Writer::restate(&path, &[sample_view("v"), sample_view("w")], None).expect("two views");
14623 let catalog = Catalog::open(&path).expect("reopen");
14624 assert_eq!(catalog.views().count(), 2);
14625 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14626 let after = fs::metadata(&path).expect("the file is there").len();
14629 assert!(after > before, "a generation was written");
14630 assert!(after - before < before, "the table was not written again");
14631 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14634 assert_eq!(reader.table().rows, 3);
14635 Writer::restate(&path, &[], None).expect("no views at all");
14638 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14639 fs::remove_file(&path).expect("clean up");
14640 }
14641
14642 #[test]
14644 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14645 let bytes = encode_catalog(
14646 &[Entry {
14647 name: "items".to_string(),
14648 fields: vec![Field::required("id", LogicalType::Integer)],
14649 rows: 1,
14650 directory: Page { offset: HEADER, length: 8, hash: 0 },
14651 nonzero: vec![None],
14652 aggregates: vec![None],
14653 distincts: vec![None],
14654 extremes: vec![None],
14655 frequencies: vec![None],
14656 }],
14657 &[sample_view("items")],
14658 None,
14659 None,
14660 )
14661 .expect("it encodes, because encoding does not look");
14662 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14663 assert!(error.to_string().contains("same name"), "{error}");
14664 }
14665
14666 #[test]
14669 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14670 let rows: usize = 300;
14671 let text: Vec<String> =
14672 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14673 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14674 let mut page = vec![6, 2];
14675 page.extend((0..rows.div_ceil(8)).map(|byte| {
14676 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14677 }));
14678 let compressed = string::encode_only(string::Kind::Fsst, &values)
14679 .expect("encoded")
14680 .expect("text this repetitive compresses");
14681 page.extend_from_slice(&compressed);
14682 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14683 let positions = [0_u32, 3, 8, 13, 200, 299];
14684 let some =
14685 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14686 assert_eq!(some.len(), positions.len());
14687 for (at, &row) in positions.iter().enumerate() {
14688 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14689 }
14690 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14691 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14692 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14693 }
14694
14695 #[test]
14698 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14699 let path = path("rows");
14700 let mut writer = Writer::create(
14701 &path,
14702 "items",
14703 vec![
14704 Field::required("id", LogicalType::Integer),
14705 Field::new("text", LogicalType::Varchar),
14706 ],
14707 )
14708 .expect("new file");
14709 let rows = 2_000;
14710 let chunk = Chunk::new(vec![
14711 Vector::from_values(
14712 LogicalType::Integer,
14713 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14714 )
14715 .expect("integers"),
14716 Vector::from_values(
14717 LogicalType::Varchar,
14718 &(0..rows)
14719 .map(|row| {
14720 if row % 7 == 2 {
14721 Value::Null
14722 } else {
14723 Value::Varchar(format!("a comment about order {}", row * 13))
14724 }
14725 })
14726 .collect::<Vec<_>>(),
14727 )
14728 .expect("strings"),
14729 ])
14730 .expect("matching rows");
14731 writer.append(&chunk).expect("one part");
14732 writer.finish().expect("commit");
14733 let reader = Reader::open(&path).expect("reopen from disk");
14734 let positions = [1_u32, 2, 9, 1_000, 1_999];
14735 for whole in [true, false] {
14736 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14737 let all = reader.read(0, &[0, 1]).expect("the whole part");
14738 assert_eq!(some.len(), positions.len());
14739 for column in 0..2 {
14740 for (at, &row) in positions.iter().enumerate() {
14741 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14742 }
14743 }
14744 }
14745 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14746 }
14747
14748 #[test]
14749 fn committed_file_reopens_and_reads_only_requested_columns() {
14750 let path = path("reopen");
14751 let mut writer = Writer::create(
14752 &path,
14753 "items",
14754 vec![
14755 Field::required("id", LogicalType::Integer),
14756 Field::new("text", LogicalType::Varchar),
14757 ],
14758 )
14759 .expect("new file");
14760 writer.append(&sample()).expect("first part");
14761 writer.append(&sample()).expect("second part");
14762 writer.finish().expect("commit");
14763 let reader = Reader::open(&path).expect("reopen from disk");
14764 assert_eq!(reader.table().rows(), 6);
14765 assert_eq!(reader.table().stripes().len(), 1);
14768 assert_eq!(reader.parts(), 2);
14769 assert_eq!(reader.part_rows(0), 3);
14770 assert_eq!(reader.part_rows(1), 3);
14771 let text = reader.read(1, &[1]).expect("only text page");
14772 assert_eq!(text.width(), 1);
14773 assert_eq!(text.value_at(1, 0), Value::Null);
14774 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14775 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14776 assert_eq!(sparse.width(), 1);
14777 assert_eq!(sparse.value_at(1, 0), Value::Null);
14778 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14779 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14780 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14781 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14782 let count = reader.read(0, &[]).expect("no page is needed for count");
14783 assert_eq!(count.len(), 3);
14784 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14785 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14786 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14787 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14788 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14789 assert_eq!(integers.omitted_max, 2);
14790 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14791 assert_eq!(strings.len(), 3);
14792 assert!(strings.contains(&(Value::Null, 2)));
14793 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14794 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14795 fs::remove_file(path).expect("remove scratch file");
14796 }
14797
14798 #[test]
14806 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14807 let path = path("interleaved-runs");
14808 let mut writer =
14809 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14810 .expect("new file");
14811 for morsel in [2_u64, 0, 3, 1] {
14812 let parts = (0..4_u64)
14813 .map(|chunk| {
14814 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14815 let values =
14816 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14817 let column =
14818 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14819 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14820 })
14821 .collect::<Vec<_>>();
14822 writer.append_stripe(parts).expect("a stripe");
14823 }
14824 writer.finish().expect("commit");
14825
14826 let reader = Reader::open(&path).expect("valid directory");
14827 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14828 assert_eq!(reader.table().rows(), 128);
14829 for part in 0..16_usize {
14830 let read = reader.read(part, &[0]).expect("a part back");
14831 for row in 0..8_usize {
14832 let want = i64::try_from(part * 8 + row).expect("small");
14833 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14834 }
14835 }
14836 fs::remove_file(path).expect("remove scratch file");
14837 }
14838
14839 #[test]
14842 fn runs_that_overlap_each_other_are_refused_at_commit() {
14843 let path = path("overlapping-runs");
14844 let mut writer =
14845 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14846 .expect("new file");
14847 let one = |order: (u64, u64)| {
14848 let column =
14849 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14850 (order, Chunk::new(vec![column]).expect("one column"))
14851 };
14852 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14855 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14856 let error = writer.finish().expect_err("the runs overlap");
14857 assert!(error.message().contains("source order"), "{error}");
14858 fs::remove_file(path).expect("remove scratch file");
14859 }
14860
14861 #[test]
14864 fn a_run_longer_than_a_stripe_is_refused() {
14865 let path = path("overlong-run");
14866 let mut writer =
14867 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14868 .expect("new file");
14869 let parts = (0..=STRIPE_PARTS)
14870 .map(|at| {
14871 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14872 .expect("a column");
14873 let chunk = Chunk::new(vec![column]).expect("one column");
14874 ((0, u64::try_from(at).expect("small")), chunk)
14875 })
14876 .collect::<Vec<_>>();
14877 let error = writer.append_stripe(parts).expect_err("one part too many");
14878 assert!(error.message().contains("more parts than it holds"), "{error}");
14879 fs::remove_file(path).expect("remove scratch file");
14880 }
14881
14882 #[test]
14888 fn parts_past_the_stripe_bound_start_a_new_stripe() {
14889 let path = path("stripe-bound");
14890 let mut writer = Writer::create(
14891 &path,
14892 "items",
14893 vec![
14894 Field::required("id", LogicalType::Integer),
14895 Field::new("text", LogicalType::Varchar),
14896 ],
14897 )
14898 .expect("new file");
14899 let parts = STRIPE_PARTS * 2 + 3;
14900 for part in 0..parts {
14901 let id = part as i32;
14902 let chunk = Chunk::new(vec![
14903 Vector::from_values(
14904 LogicalType::Integer,
14905 &[Value::Integer(id), Value::Integer(-id)],
14906 )
14907 .expect("integers"),
14908 Vector::from_values(
14909 LogicalType::Varchar,
14910 &[Value::Varchar(format!("value {part}")), Value::Null],
14911 )
14912 .expect("strings"),
14913 ])
14914 .expect("matching rows");
14915 writer.append(&chunk).expect("one part");
14916 }
14917 writer.finish().expect("commit");
14918
14919 let reader = Reader::open(&path).expect("reopen from disk");
14920 assert_eq!(reader.parts(), parts);
14921 assert_eq!(reader.table().rows(), parts * 2);
14922 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14923 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14924 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14925 assert_eq!(reader.table().stripes()[2].parts(), 3);
14926 for part in (0..parts).rev() {
14929 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14930 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14931 for chunk in [&dense, &sparse] {
14932 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14933 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14934 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14935 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14936 assert_eq!(chunk.value_at(1, 1), Value::Null);
14937 }
14938 }
14939 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14942 assert!(reader.skips(0, &above), "the first stripe stops at 63");
14943 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14944 fs::remove_file(path).expect("remove scratch file");
14945 }
14946
14947 fn scattered(n: i64) -> i64 {
14949 n.wrapping_mul(-7_046_029_254_386_353_131)
14950 }
14951
14952 #[test]
14958 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14959 let path = path("sieve-skip");
14960 let mut writer =
14961 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14962 .expect("new file");
14963 let parts = STRIPE_PARTS + 3;
14964 let per_part = 128;
14968 for part in 0..parts {
14969 let held: Vec<Value> = (0..per_part)
14970 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14971 .collect();
14972 let chunk =
14973 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14974 .expect("one column");
14975 writer.append(&chunk).expect("one part");
14976 }
14977 writer.finish().expect("commit");
14978
14979 let reader = Reader::open(&path).expect("reopen from disk");
14980 let probe = |value: i64| Probe {
14981 column: 0,
14982 op: Op::Equal,
14983 value: Bound::Int(i128::from(scattered(value))),
14984 };
14985 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14986 let tests = [probe(wanted)];
14987 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14988 let home = wanted as usize / per_part;
14989 assert!(kept.contains(&home), "the part holding {wanted} is read");
14990 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14994 }
14995 let absent = [probe((parts * per_part) as i64 + 1)];
14996 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
14997 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
14998 let tests = [probe(0)];
15001 assert!(
15002 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
15003 "the bounds rule out no stripe at all"
15004 );
15005 fs::remove_file(path).expect("remove scratch file");
15006 }
15007
15008 #[test]
15014 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
15015 let path = path("part-range-skip");
15016 let mut writer =
15017 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15018 .expect("new file");
15019 let parts = STRIPE_PARTS + 3;
15020 let per_part = 128;
15021 for part in 0..parts {
15022 let held: Vec<Value> = (0..per_part)
15026 .map(|row| {
15027 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15028 })
15029 .collect();
15030 let chunk =
15031 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15032 .expect("one column");
15033 writer.append(&chunk).expect("one part");
15034 }
15035 writer.finish().expect("commit");
15036
15037 let reader = Reader::open(&path).expect("reopen from disk");
15038 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15039 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
15040 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
15041 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
15043 fs::remove_file(path).expect("remove scratch file");
15044 }
15045
15046 #[test]
15050 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
15051 let path = path("part-range-certain");
15052 let mut writer =
15053 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15054 .expect("new file");
15055 let parts = STRIPE_PARTS + 3;
15056 let per_part = 128;
15057 for part in 0..parts {
15058 let held: Vec<Value> = (0..per_part)
15059 .map(|row| {
15060 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
15061 })
15062 .collect();
15063 let chunk =
15064 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15065 .expect("one column");
15066 writer.append(&chunk).expect("one part");
15067 }
15068 writer.finish().expect("commit");
15069
15070 let reader = Reader::open(&path).expect("reopen from disk");
15071 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
15072 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
15073 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
15074 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
15077 fs::remove_file(path).expect("remove scratch file");
15078 }
15079
15080 #[test]
15083 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
15084 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
15085 let path = path("part-range-page");
15086 let mut writer =
15087 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
15088 .expect("new file");
15089 for part in 0..parts {
15090 let held: Vec<Value> = (0..128)
15091 .map(|row| {
15092 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
15093 })
15094 .collect();
15095 let chunk = Chunk::new(vec![
15096 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
15097 ])
15098 .expect("one column");
15099 writer.append(&chunk).expect("one part");
15100 }
15101 writer.finish().expect("commit");
15102 let reader = Reader::open(&path).expect("reopen from disk");
15103 let bytes = reader.layout().columns[0].part_ranges;
15104 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
15105 fs::remove_file(path).expect("remove scratch file");
15106 }
15107 }
15108
15109 #[test]
15112 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
15113 let long = vec![b'a'; PART_BOUND_BYTES * 2];
15114 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
15115 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
15116 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
15117 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
15118 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
15119 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
15120 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
15121 }
15122
15123 #[test]
15126 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
15127 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
15128 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
15129 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
15130 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
15131 }
15132
15133 #[test]
15145 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
15146 let parts = 4;
15147 let per_part = 1024;
15148 let rows = parts * per_part;
15149 let written = |name: &str, keys: &[i64]| {
15150 let path = path(name);
15151 let fields = vec![Field::required("key", LogicalType::BigInt)];
15152 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
15153 for part in 0..parts {
15154 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
15155 .iter()
15156 .map(|key| Value::BigInt(*key))
15157 .collect();
15158 let chunk = Chunk::new(vec![
15159 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
15160 ])
15161 .expect("one column");
15162 writer.append(&chunk).expect("one part");
15163 }
15164 writer.finish().expect("commit");
15165 path
15166 };
15167 let climbing = |step: &dyn Fn(usize) -> i64| {
15170 let mut key = 0;
15171 (0..rows)
15172 .map(|row| {
15173 key += step(row);
15174 key
15175 })
15176 .collect::<Vec<i64>>()
15177 };
15178 let ascending = climbing(&|row| (row % 3) as i64);
15179 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
15183 let near_path = written("stored-near", &ascending);
15184 let far_path = written("stored-far", &sparse);
15185
15186 let one = Reader::open(&near_path).expect("reopen from disk");
15187 let other = Reader::open(&far_path).expect("reopen from disk");
15188 let near = one.stored(0).expect("the column is stored");
15189 let far = other.stored(0).expect("the column is stored");
15190 assert_eq!(near.len(), parts, "one row per part");
15191 assert_eq!(far.len(), parts);
15192 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
15195 assert_eq!(total(&near), one.layout().columns[0].pages);
15196 assert_eq!(total(&far), other.layout().columns[0].pages);
15197 assert!(
15198 total(&near) * 2 < total(&far),
15199 "the sparse keys cost more, {} against {}",
15200 total(&far),
15201 total(&near)
15202 );
15203 for (at, part) in near.iter().enumerate() {
15205 assert_eq!(part.part, at);
15206 assert_eq!(part.row, at * per_part);
15207 assert_eq!(part.rows, per_part);
15208 let held = &ascending[at * per_part..(at + 1) * per_part];
15209 assert_eq!(part.low, Some(Value::BigInt(held[0])));
15210 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
15211 assert_eq!(part.nulls, Some(0));
15212 }
15213 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
15216 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
15217 assert_ne!(near[0].encoding, far[0].encoding);
15218 fs::remove_file(near_path).expect("remove scratch file");
15219 fs::remove_file(far_path).expect("remove scratch file");
15220 }
15221
15222 #[test]
15232 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
15233 let path = path("sieve-pays");
15234 let fields = vec![
15235 Field::required("spread", LogicalType::BigInt),
15236 Field::required("repeated", LogicalType::BigInt),
15237 ];
15238 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
15239 let parts = 3;
15240 let per_part = 1024;
15241 for part in 0..parts {
15242 let base = (part * per_part) as i64;
15243 let spread: Vec<Value> =
15244 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
15245 let repeated: Vec<Value> =
15246 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
15247 let chunk = Chunk::new(vec![
15248 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
15249 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
15250 ])
15251 .expect("two columns");
15252 writer.append(&chunk).expect("one part");
15253 }
15254 writer.finish().expect("commit");
15255
15256 let reader = Reader::open(&path).expect("reopen from disk");
15257 let layout = reader.layout();
15258 let spread = &layout.columns[0];
15259 let repeated = &layout.columns[1];
15260 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
15261 assert_eq!(
15262 repeated.sieves, 0,
15263 "a column whose filter costs more than its parts keeps none"
15264 );
15265 for column in &layout.columns {
15268 assert!(
15269 column.sieves < column.pages,
15270 "{} spends {} on sieves over {} of data",
15271 column.name,
15272 column.sieves,
15273 column.pages
15274 );
15275 }
15276 let absent = [Probe {
15278 column: 0,
15279 op: Op::Equal,
15280 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
15281 }];
15282 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
15283 fs::remove_file(path).expect("remove scratch file");
15284 }
15285
15286 #[test]
15292 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
15293 let path = path("sieve-damaged");
15294 let mut writer =
15295 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
15296 .expect("new file");
15297 let rows = 128;
15298 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
15299 let chunk =
15300 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
15301 .expect("one column");
15302 writer.append(&chunk).expect("one part");
15303 writer.finish().expect("commit");
15304
15305 let page = Reader::open(&path).expect("reopen").table.stripes[0]
15306 .sieves
15307 .get(0)
15308 .expect("a sieve page");
15309 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
15310 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
15311 file.write_all(&[0xff]).expect("damage one byte");
15312 drop(file);
15313
15314 let reader = Reader::open(&path).expect("reopen the damaged file");
15315 let absent =
15316 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
15317 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
15318 assert_eq!(
15319 reader.read(0, &[0]).expect("the rows are untouched").len(),
15320 usize::try_from(rows).expect("a small count")
15321 );
15322 fs::remove_file(path).expect("remove scratch file");
15323 }
15324
15325 #[test]
15331 fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
15332 let path = path("asked-twice");
15333 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15334 let mut writer =
15335 Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
15336 .expect("new file");
15337 for part in 0..parts {
15338 let chunk = Chunk::new(vec![
15339 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15340 .expect("integers"),
15341 ])
15342 .expect("matching rows");
15343 writer.append(&chunk).expect("one part");
15344 }
15345 writer.finish().expect("commit");
15346
15347 let pool = PagePool::new(usize::MAX);
15348 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15349 let a = catalog.table("a").expect("a");
15350 let stripes = a.table().stripes().len();
15351 for part in 0..parts {
15352 for _ in 0..2 {
15353 let chunk = a.read(part, &[0]).expect("a part");
15354 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15355 }
15356 }
15357 assert_eq!(
15358 a.pages.load(Atomic::Relaxed),
15359 stripes,
15360 "a page a stripe, read on the second ask"
15361 );
15362 assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
15363 let column = a.cache.columns[0].lock().expect("the column");
15364 assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
15365 drop(column);
15366 drop((a, catalog));
15367 fs::remove_file(path).expect("remove scratch file");
15368 }
15369
15370 #[test]
15381 fn workers_that_want_the_same_stripe_read_it_once() {
15382 let path = path("single-flight");
15383 let mut writer =
15384 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15385 .expect("new file");
15386 for part in 0..STRIPE_PARTS {
15387 let id = part as i32;
15388 let chunk = Chunk::new(vec![
15389 Vector::from_values(
15390 LogicalType::Integer,
15391 &[Value::Integer(id), Value::Integer(-id)],
15392 )
15393 .expect("integers"),
15394 ])
15395 .expect("matching rows");
15396 writer.append(&chunk).expect("one part");
15397 }
15398 writer.finish().expect("commit");
15399
15400 let reader = Reader::open(&path).expect("reopen from disk");
15401 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
15402 for part in 0..STRIPE_PARTS {
15405 reader.read(part, &[0]).expect("a part");
15406 }
15407 assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
15408 let barrier = std::sync::Barrier::new(8);
15409 std::thread::scope(|scope| {
15410 for worker in 0..8 {
15411 let reader = &reader;
15412 let barrier = &barrier;
15413 scope.spawn(move || {
15414 barrier.wait();
15415 for part in (worker..STRIPE_PARTS).step_by(8) {
15416 let chunk = reader.read(part, &[0]).expect("a whole page read");
15417 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15418 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15419 }
15420 });
15421 }
15422 });
15423 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15424 fs::remove_file(path).expect("remove scratch file");
15425 }
15426
15427 #[test]
15440 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15441 let opened = |label: &str, rows_per_part: i32| {
15442 let path = path(label);
15443 let mut writer =
15444 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15445 .expect("new file");
15446 for part in 0..STRIPE_PARTS * 3 {
15447 let values = (0..rows_per_part)
15451 .map(|row| {
15452 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15453 })
15454 .collect::<Vec<_>>();
15455 let chunk = Chunk::new(vec![
15456 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15457 ])
15458 .expect("matching rows");
15459 writer.append(&chunk).expect("one part");
15460 }
15461 writer.finish().expect("commit");
15462 let reader = Reader::open(&path).expect("reopen from disk");
15463 let size = fs::metadata(&path).expect("the file is there").len();
15464 let out = (reader.reads(), reader.table().stripes().len(), size);
15465 fs::remove_file(path).expect("remove scratch file");
15466 out
15467 };
15468
15469 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15470 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15471 assert_eq!(
15472 thin_stripes, fat_stripes,
15473 "the same stripe count is what makes this a fair ask"
15474 );
15475 assert!(
15476 fat_size > thin_size * 50,
15477 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15478 );
15479
15480 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15481 assert_eq!(thin.pages, 0, "opening read a page");
15482 assert_eq!(fat.pages, 0, "opening read a page");
15483 assert_eq!(thin.indexes, 0, "opening read an index");
15484 assert_eq!(fat.indexes, 0, "opening read an index");
15485 assert!(
15488 fat.opening.bytes < thin.opening.bytes * 2,
15489 "opening the thin file read {} bytes and the fat one read {}",
15490 thin.opening.bytes,
15491 fat.opening.bytes
15492 );
15493 }
15494
15495 #[test]
15503 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15504 let path = path("open-twice");
15505 let mut writer =
15506 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15507 .expect("new file");
15508 for part in 0..STRIPE_PARTS * 3 {
15509 let chunk = Chunk::new(vec![
15510 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15511 .expect("integers"),
15512 ])
15513 .expect("matching rows");
15514 writer.append(&chunk).expect("one part");
15515 }
15516 writer.finish().expect("commit");
15517
15518 let first = Reader::open(&path).expect("open");
15519 for part in 0..first.parts() {
15522 first.read(part, &[0]).expect("a part");
15523 }
15524 assert!(first.reads().indexes > 0, "the scan has to have read something");
15525 let second = Reader::open(&path).expect("open again");
15526
15527 assert_eq!(first.reads().opening, second.reads().opening);
15528 assert_eq!(
15529 second.reads().pages,
15530 0,
15531 "the second open read a page off the back of the first"
15532 );
15533 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15534 fs::remove_file(path).expect("remove scratch file");
15535 }
15536
15537 #[test]
15545 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15546 let path = path("index-cache");
15547 let mut writer =
15548 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15549 .expect("new file");
15550 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15551 for part in 0..parts {
15552 let id = part as i32;
15553 let chunk = Chunk::new(vec![
15554 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15555 ])
15556 .expect("matching rows");
15557 writer.append(&chunk).expect("one part");
15558 }
15559 writer.finish().expect("commit");
15560
15561 let reader = Reader::open(&path).expect("reopen from disk");
15562 let stripes = reader.table().stripes().len();
15563 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15564 for _ in 0..3 {
15567 for part in 0..parts {
15568 let chunk = reader.read(part, &[0]).expect("a part");
15569 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15570 }
15571 }
15572 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15573 assert!(
15574 reader.pages.load(Atomic::Relaxed) > stripes,
15575 "the pages are the ones that get read again, which is what makes the index count mean \
15576 something"
15577 );
15578 fs::remove_file(path).expect("remove scratch file");
15579 }
15580
15581 #[test]
15588 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15589 let path = path("page-pool");
15590 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15591 let fields = || vec![Field::required("id", LogicalType::Integer)];
15592 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15593 for table in ["a", "b"] {
15594 if table == "b" {
15595 writer = writer.next("b".to_string(), fields()).expect("a second table");
15596 }
15597 for part in 0..parts {
15598 let chunk = Chunk::new(vec![
15599 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15600 .expect("integers"),
15601 ])
15602 .expect("matching rows");
15603 writer.append(&chunk).expect("one part");
15604 }
15605 }
15606 writer.finish().expect("commit");
15607
15608 let pool = PagePool::new(usize::MAX);
15609 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15610 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15611 let stripes = a.table().stripes().len();
15612 assert!(
15613 stripes > CACHED_STRIPES_PER_COLUMN * 2,
15614 "the floor has to be smaller than a table"
15615 );
15616 let scan = |reader: &Reader| {
15617 for part in 0..parts {
15618 let chunk = reader.read(part, &[0]).expect("a part");
15619 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15620 }
15621 };
15622 scan(&a);
15625 assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15626 assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15627 scan(&a);
15628 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15629 scan(&a);
15630 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15631 let one = pool.bytes();
15632 assert!(one > 0, "the pool counts what the reader holds");
15633
15634 pool.budget.store(one, Atomic::Relaxed);
15636 scan(&b);
15637 scan(&b);
15638 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15639 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15640 let column = a.cache.columns[0].lock().expect("the column");
15641 let held = column.pages.iter().flatten().count();
15642 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15643 drop(column);
15644
15645 drop((a, b, catalog));
15647 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15648 scan(&c);
15649 scan(&c);
15650 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15651 fs::remove_file(path).expect("remove scratch file");
15652 }
15653
15654 #[test]
15663 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15664 let workers = CACHED_STRIPES_PER_COLUMN + 4;
15665 let path = path("stripe-per-worker");
15666 let mut writer =
15667 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15668 .expect("new file");
15669 for part in 0..STRIPE_PARTS * workers {
15670 let chunk = Chunk::new(vec![
15671 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15672 .expect("integers"),
15673 ])
15674 .expect("matching rows");
15675 writer.append(&chunk).expect("one part");
15676 }
15677 writer.finish().expect("commit");
15678
15679 let read = |told: bool| {
15680 let reader = Reader::open(&path).expect("reopen from disk");
15681 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15682 if told {
15683 reader.keep_stripes(workers);
15684 }
15685 for part in 0..reader.parts() {
15687 reader.read(part, &[0]).expect("a part");
15688 }
15689 let barrier = std::sync::Barrier::new(workers);
15690 std::thread::scope(|scope| {
15691 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15692 let reader = &reader;
15693 let barrier = &barrier;
15694 scope.spawn(move || {
15695 for part in run {
15696 barrier.wait();
15697 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15698 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15699 }
15700 assert!(worker < workers);
15701 });
15702 }
15703 });
15704 reader.pages.load(Atomic::Relaxed)
15705 };
15706
15707 assert_eq!(read(true), workers, "one page read per stripe and no more");
15708 assert!(read(false) > workers, "a cache that small is read again on every part");
15709 fs::remove_file(path).expect("remove scratch file");
15710 }
15711
15712 #[test]
15717 fn a_damaged_index_page_is_an_error() {
15718 let path = path("damaged-index");
15719 let mut writer =
15720 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15721 .expect("new file");
15722 writer.append(&sample_ids()).expect("first part");
15723 writer.append(&sample_ids()).expect("second part");
15724 writer.finish().expect("commit");
15725
15726 let reader = Reader::open(&path).expect("valid directory");
15727 let index = reader.table.stripes[0].index;
15728 let mut byte = [0; 1];
15729 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15730 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15731 file.seek(SeekFrom::Start(index.offset)).expect("index start");
15732 file.write_all(&[!byte[0]]).expect("damage the first part length");
15733 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15734 assert!(error.message().contains("index page section checksum differs"), "{error}");
15735 fs::remove_file(path).expect("remove scratch file");
15736 }
15737
15738 #[test]
15745 fn every_integer_width_round_trips_through_a_page() {
15746 let path = path("integer-widths");
15747 let columns = [
15748 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15749 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15750 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15751 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15752 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15753 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15754 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15755 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15756 ];
15757 let fields = columns
15758 .iter()
15759 .enumerate()
15760 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15761 .collect::<Vec<_>>();
15762 let vectors = columns
15763 .iter()
15764 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15765 .collect::<Vec<_>>();
15766 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15767 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15768 writer.finish().expect("commit");
15769
15770 let reader = Reader::open(&path).expect("reopen from disk");
15771 let wanted = (0..columns.len()).collect::<Vec<_>>();
15772 let read = reader.read(0, &wanted).expect("every column");
15773 assert_eq!(read.len(), 2);
15774 for (at, (ty, values)) in columns.iter().enumerate() {
15776 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15777 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15778 }
15779 fs::remove_file(path).expect("remove scratch file");
15780 }
15781
15782 #[test]
15793 fn every_other_type_the_format_knows_round_trips_through_a_page() {
15794 let path = path("other-types");
15795 let columns = [
15796 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15797 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15798 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15799 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15800 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15801 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15802 (
15803 LogicalType::TimestampTz,
15804 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15805 ),
15806 (
15807 LogicalType::Interval,
15808 vec![
15809 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15810 Value::Interval { months: 13, days: -1, micros: 1 },
15811 ],
15812 ),
15813 (
15814 LogicalType::Blob,
15815 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15816 ),
15817 ];
15818 let fields = columns
15819 .iter()
15820 .enumerate()
15821 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15822 .collect::<Vec<_>>();
15823 let vectors = columns
15824 .iter()
15825 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15826 .collect::<Vec<_>>();
15827 let mut writer = Writer::create(&path, "others", fields).expect("new file");
15828 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15829 writer.finish().expect("commit");
15830
15831 let reader = Reader::open(&path).expect("reopen from disk");
15832 let wanted = (0..columns.len()).collect::<Vec<_>>();
15833 let read = reader.read(0, &wanted).expect("every column");
15834 assert_eq!(read.len(), 2);
15835 for (at, (ty, values)) in columns.iter().enumerate() {
15836 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15837 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15838 }
15839 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15842 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15843
15844 fs::remove_file(path).expect("remove scratch file");
15845 }
15846
15847 #[test]
15853 fn a_nan_survives_being_written_down() {
15854 let path = path("nan");
15855 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15856 .expect("a NaN vector");
15857 let mut writer =
15858 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15859 .expect("new file");
15860 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15861 writer.finish().expect("commit");
15862 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15863 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15864 assert!(back.is_nan(), "a NaN came back as {back}");
15865 fs::remove_file(path).expect("remove scratch file");
15866 }
15867
15868 #[test]
15875 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15876 let path = path("uuid-and-bit");
15877 let uuids = vec![0_i128, i128::MIN, -1];
15878 let mut bits = StringColumn::new();
15879 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15880 bits.push_bytes(value);
15881 }
15882 let expected = bits.clone();
15883 let fields =
15884 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15885 let vectors = vec![
15886 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15887 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15888 ];
15889 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15890 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15891 writer.finish().expect("commit");
15892
15893 let reader = Reader::open(&path).expect("reopen from disk");
15894 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15895 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15896 panic!("a uuid column is the 128 bit lane")
15897 };
15898 assert_eq!(back.as_slice(), uuids.as_slice());
15899 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15900 panic!("a bit column is bytes")
15901 };
15902 for row in 0..expected.len() {
15903 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15904 }
15905 fs::remove_file(path).expect("remove scratch file");
15906 }
15907
15908 #[test]
15911 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15912 let mut rows: Vec<Option<u64>> = Vec::new();
15913 let mut state = 0x2545_f491_4f6c_dd1d_u64;
15914 for index in 0..400_000_u64 {
15915 state ^= state << 13;
15916 state ^= state >> 7;
15917 state ^= state << 17;
15918 let times = 1 + (state % 7) as usize;
15919 let bits = match state % 11 {
15920 0 => None,
15921 1..=3 => Some(state % 16),
15922 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15923 };
15924 rows.extend(std::iter::repeat_n(bits, times));
15925 }
15926 let mut by_row = Candidates::default();
15927 for &bits in &rows {
15928 by_row.add(bits, 1);
15929 }
15930 let mut by_run = Candidates::default();
15931 let mut run = Run::default();
15932 let mut runs = 0_usize;
15933 for &bits in &rows {
15934 if let Some((bits, times)) = run.push(bits) {
15935 by_run.add(bits, times);
15936 runs += 1;
15937 }
15938 }
15939 if let Some((bits, times)) = run.take() {
15940 by_run.add(bits, times);
15941 }
15942 assert!(runs < rows.len() / 2, "the rows came in runs");
15943 assert!(by_row.decrements > 0, "the table filled and turned values away");
15944 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15945 assert_eq!(by_run.nulls, by_row.nulls);
15946 assert_eq!(by_run.decrements, by_row.decrements);
15947 }
15948
15949 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15950 let mut pairs = candidates.pairs().collect::<Vec<_>>();
15951 pairs.sort_unstable();
15952 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15953 pairs
15954 }
15955
15956 #[derive(Default)]
15959 struct MapCandidates {
15960 counts: HashMap<u64, u32>,
15961 nulls: u32,
15962 decrements: u64,
15963 }
15964
15965 impl MapCandidates {
15966 fn add(&mut self, bits: Option<u64>, mut times: u32) {
15967 while times > 0 {
15968 let held = match bits {
15969 Some(bits) => self.counts.get_mut(&bits),
15970 None if self.nulls != 0 => Some(&mut self.nulls),
15971 None => None,
15972 };
15973 if let Some(count) = held {
15974 *count = count.saturating_add(times);
15975 return;
15976 }
15977 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15978 match bits {
15979 Some(bits) => {
15980 self.counts.insert(bits, times);
15981 }
15982 None => self.nulls = times,
15983 }
15984 return;
15985 }
15986 self.counts.retain(|_, count| {
15987 *count -= 1;
15988 *count != 0
15989 });
15990 self.nulls = self.nulls.saturating_sub(1);
15991 self.decrements = self.decrements.saturating_add(1);
15992 times -= 1;
15993 }
15994 }
15995 }
15996
15997 #[test]
16001 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
16002 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
16003 let mut table = Candidates::default();
16004 let mut oracle = MapCandidates::default();
16005 let mut state = seed;
16006 for index in 0..300_000_u64 {
16007 state ^= state << 13;
16008 state ^= state >> 7;
16009 state ^= state << 17;
16010 let bits = match state % 13 {
16011 0 => None,
16012 1..=4 => Some(state % 40),
16013 5 => Some((index % 1000) * 1_000_000),
16014 _ => Some(state),
16015 };
16016 let times = 1 + (state >> 60) as u32 % 3;
16017 table.add(bits, times);
16018 oracle.add(bits, times);
16019 if index % 50_000 == 0 {
16020 let mut expected =
16021 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16022 expected.sort_unstable();
16023 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
16024 }
16025 }
16026 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
16027 expected.sort_unstable();
16028 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
16029 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
16030 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
16031 assert!(table.decrements > 0, "seed {seed} never filled the table");
16032 for &(bits, _) in &expected {
16033 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
16034 }
16035 }
16036 }
16037
16038 #[test]
16039 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
16040 let path = path("frequency-ordinals");
16041 let mut writer =
16042 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
16043 .expect("new file");
16044 let mut values = Vec::new();
16045 for leader in 0..10_i64 {
16046 values.extend(std::iter::repeat_n(leader, 100));
16047 }
16048 values.extend(1_000_i64..41_000);
16049 for part in values.chunks(1_024) {
16050 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
16051 .expect("big integers");
16052 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
16053 }
16054 writer.finish().expect("commit");
16055
16056 let reader = Reader::open(&path).expect("reopen from disk");
16057 let occurrences =
16058 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
16059 assert!(occurrences.omitted_max < 100);
16060 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
16061 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
16062 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
16063 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
16064 assert_eq!(
16065 &occurrences.anchor_indices[..1_000]
16066 .iter()
16067 .map(|&entry| occurrences.anchors[entry as usize].clone())
16068 .collect::<Vec<_>>(),
16069 &(0_i64..10)
16070 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
16071 .collect::<Vec<_>>()
16072 );
16073 fs::remove_file(path).expect("remove scratch file");
16074 }
16075
16076 #[test]
16077 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
16078 let path = path("frequency-bits");
16083 let mut writer = Writer::create(
16084 &path,
16085 "items",
16086 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
16087 )
16088 .expect("new file");
16089 let mut rows = Vec::new();
16090 let mut leaders = Vec::new();
16091 for leader in 0..10_u64 {
16092 let count = 300 - leader * 10;
16093 let (unsigned, signed) = if leader == 0 {
16094 (Value::Null, Value::Null)
16095 } else {
16096 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
16097 };
16098 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
16099 leaders.push(((unsigned, count), (signed, count)));
16100 }
16101 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
16102 for part in rows.chunks(1_024) {
16103 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
16104 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
16105 let chunk = Chunk::new(vec![
16106 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
16107 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
16108 ])
16109 .expect("matching columns");
16110 writer.append(&chunk).expect("rows");
16111 }
16112 writer.finish().expect("commit");
16113
16114 let reader = Reader::open(&path).expect("reopen from disk");
16115 for column in 0..2 {
16116 let prefix =
16117 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16118 let wanted = leaders
16119 .iter()
16120 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
16121 .cloned()
16122 .collect::<Vec<_>>();
16123 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
16124 assert!(prefix.omitted_max < 210, "column {column}");
16125 assert_eq!(
16126 reader.distinct_values(column).expect("valid metadata"),
16127 Some(9 + 40_000),
16128 "column {column}"
16129 );
16130 }
16131 fs::remove_file(path).expect("remove scratch file");
16132 }
16133
16134 #[test]
16135 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
16136 let path = path("frequency-tally");
16142 let types = [
16143 LogicalType::TinyInt,
16144 LogicalType::UInteger,
16145 LogicalType::Date,
16146 LogicalType::Timestamp,
16147 ];
16148 let value = |ty: &LogicalType, at: i64| match ty {
16149 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
16150 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
16151 LogicalType::Date => Value::Date(19_000 - at as i32),
16152 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
16153 };
16154 let fields = types
16155 .iter()
16156 .enumerate()
16157 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
16158 .collect::<Vec<_>>();
16159 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16160 let mut rows = Vec::new();
16161 for at in 0..250_i64 {
16162 for _ in 0..=(at % 37) {
16163 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
16164 }
16165 }
16166 for part in rows.chunks(1_000) {
16167 let columns = types
16168 .iter()
16169 .map(|ty| {
16170 let values = part
16171 .iter()
16172 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
16173 .collect::<Vec<_>>();
16174 Vector::from_values(ty.clone(), &values).expect("a column")
16175 })
16176 .collect();
16177 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
16178 }
16179 writer.finish().expect("commit");
16180
16181 let reader = Reader::open(&path).expect("reopen from disk");
16182 for (column, ty) in types.iter().enumerate() {
16183 let mut counts = HashMap::<Option<i64>, u64>::new();
16184 for row in &rows {
16185 *counts.entry(*row).or_default() += 1;
16186 }
16187 let wanted = counts
16188 .into_iter()
16189 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
16190 .collect::<Vec<_>>();
16191 let prefix =
16192 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
16193 assert_eq!(prefix.entries.len(), 2, "column {column}");
16194 assert!(prefix.omitted_max > 0, "column {column}");
16195 for (value, count) in &prefix.entries {
16196 let held =
16197 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
16198 assert_eq!(held, Some(count), "column {column} value {value:?}");
16199 }
16200 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
16201 assert_eq!(
16202 reader.distinct_values(column).expect("valid metadata"),
16203 Some(wanted.len() as u64 - 1),
16204 "column {column}"
16205 );
16206 }
16207 fs::remove_file(path).expect("remove scratch file");
16208 }
16209
16210 #[test]
16211 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
16212 let edge = FREQUENCY_CANDIDATES as i64;
16217 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
16218 for with_null in [false, true] {
16219 let path = path("distinct-edge");
16220 let mut writer =
16221 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
16222 .expect("new file");
16223 let mut values = Vec::new();
16224 for round in 0..2 {
16225 for value in 0..distinct {
16226 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
16227 values.extend(std::iter::repeat_n(
16228 Value::BigInt(value * 7_919 % distinct),
16229 repeat,
16230 ));
16231 if with_null && value % 1_000 == 0 {
16232 values.push(Value::Null);
16233 }
16234 }
16235 }
16236 if with_null {
16237 values.push(Value::Null);
16238 }
16239 for part in values.chunks(1_024) {
16240 let chunk = Chunk::new(vec![
16241 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
16242 ])
16243 .expect("one column");
16244 writer.append(&chunk).expect("rows");
16245 }
16246 writer.finish().expect("commit");
16247 let reader = Reader::open(&path).expect("reopen from disk");
16248 assert_eq!(
16249 reader.distinct_values(0).expect("valid metadata"),
16250 Some(distinct as u64),
16251 "{distinct} values, null {with_null}"
16252 );
16253 fs::remove_file(path).expect("remove scratch file");
16254 }
16255 }
16256 }
16257
16258 #[test]
16259 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
16260 let path = path("quick-nonzero");
16261 let mut writer = Writer::create(
16262 &path,
16263 "items",
16264 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
16265 )
16266 .expect("create");
16267 for ids in [
16268 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
16269 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
16270 ] {
16271 let labels = vec![Value::Varchar("same".into()); ids.len()];
16272 writer
16273 .append(
16274 &Chunk::new(vec![
16275 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
16276 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
16277 ])
16278 .expect("chunk"),
16279 )
16280 .expect("append");
16281 }
16282 writer.finish().expect("finish");
16283 let catalog = Catalog::open(&path).expect("catalog");
16284 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
16285 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
16286 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
16287 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
16288 let prefix = catalog
16289 .table("items")
16290 .expect("reader")
16291 .frequency_prefix(1)
16292 .expect("valid metadata")
16293 .expect("partial frequencies");
16294 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
16295 assert_eq!(prefix.omitted_max, 1);
16296 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
16297 assert_eq!(
16298 catalog.integer_extremes("items", 1).expect("extremes"),
16299 Some(IntegerExtremes::Values { low: 0, high: 7 })
16300 );
16301 assert_eq!(
16302 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
16303 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16304 );
16305 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
16306 let mut legacy = catalog.clone();
16307 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
16308 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
16309 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
16310 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
16311 Writer::certify_counts(&path).expect("recertify");
16312 assert_eq!(
16313 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
16314 Some(2)
16315 );
16316 assert_eq!(
16317 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
16318 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
16319 );
16320 assert_eq!(
16321 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
16322 Some(3)
16323 );
16324 assert_eq!(
16325 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
16326 Some(IntegerExtremes::Values { low: 0, high: 7 })
16327 );
16328 assert_eq!(
16329 Catalog::open(&path)
16330 .expect("reopen")
16331 .exact_numeric_frequencies("items", 1)
16332 .expect("frequencies"),
16333 None
16334 );
16335 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
16336 fs::remove_file(path).expect("remove scratch file");
16337 }
16338
16339 #[test]
16340 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
16341 let path = path("pair-frequencies");
16342 let mut pairs = Vec::new();
16343 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
16344 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
16345 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
16346 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
16347 let mut writer = Writer::create(
16348 &path,
16349 "items",
16350 vec![
16351 Field::required("id", LogicalType::BigInt),
16352 Field::required("phrase", LogicalType::Varchar),
16353 ],
16354 )
16355 .expect("new file");
16356 for part in pairs.chunks(1_024) {
16357 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
16358 let phrases =
16359 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
16360 writer
16361 .append(
16362 &Chunk::new(vec![
16363 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
16364 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
16365 ])
16366 .expect("matching columns"),
16367 )
16368 .expect("rows");
16369 }
16370 writer.finish().expect("commit");
16371
16372 let reader = Reader::open(&path).expect("reopen from disk");
16373 assert!(
16374 reader.table.pair_frequencies.is_empty(),
16375 "no query-specific pair result is stored"
16376 );
16377 fs::remove_file(path).expect("remove scratch file");
16378 }
16379
16380 #[test]
16381 fn legacy_group_answers_are_ignored() {
16382 let path = path("legacy-group-answers");
16383 let mut writer = Writer::create(
16384 &path,
16385 "items",
16386 vec![
16387 Field::required("id", LogicalType::BigInt),
16388 Field::required("text", LogicalType::Varchar),
16389 ],
16390 )
16391 .expect("new file");
16392 writer
16393 .append(
16394 &Chunk::new(vec![
16395 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
16396 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
16397 .expect("text"),
16398 ])
16399 .expect("row"),
16400 )
16401 .expect("append");
16402 writer.finish().expect("commit");
16403 let mut reader = Reader::open(&path).expect("reopen");
16404 let table = Arc::make_mut(&mut reader.table);
16405 table.pair_frequencies.push(PairFrequencySummary {
16406 first: 0,
16407 second: 1,
16408 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
16409 omitted_max: 0,
16410 });
16411 table.host_groups = Some(host::HostSummary {
16412 column: 1,
16413 omitted_max: 0,
16414 entries: vec![host::HostEntry {
16415 host: "fake.test".into(),
16416 count: 999,
16417 bytes_sum: 999,
16418 minimum: "x".into(),
16419 }],
16420 });
16421 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16422 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16423 fs::remove_file(path).expect("remove scratch file");
16424 }
16425
16426 #[test]
16432 fn a_file_from_another_format_says_which_format_it_is() {
16433 let older = path("older-format");
16434 let mut writer =
16435 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16436 .expect("new file");
16437 let chunk = Chunk::new(vec![
16438 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16439 .expect("integers"),
16440 ])
16441 .expect("chunk");
16442 writer.append(&chunk).expect("page written");
16443 writer.finish().expect("commit");
16444
16445 let unreadable =
16449 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16450 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16451 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16452 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16453 drop(file);
16454 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16455 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16456 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16457
16458 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16459 file.seek(SeekFrom::Start(0)).expect("the magic is first");
16460 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16461 drop(file);
16462 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16463 assert!(complaint.contains("magic"), "{complaint}");
16464 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16465 fs::remove_file(older).expect("remove scratch file");
16466 }
16467
16468 #[test]
16469 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16470 let unfinished = path("unfinished");
16471 let mut writer =
16472 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16473 .expect("new file");
16474 let chunk = Chunk::new(vec![
16475 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16476 .expect("integers"),
16477 ])
16478 .expect("chunk");
16479 writer.append(&chunk).expect("page written");
16480 drop(writer);
16481 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16482 fs::remove_file(unfinished).expect("remove scratch file");
16483
16484 let damaged = path("damaged");
16485 let mut writer =
16486 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16487 .expect("new file");
16488 writer.append(&chunk).expect("page written");
16489 writer.finish().expect("commit");
16490 let reader = Reader::open(&damaged).expect("valid directory");
16491 let mut file =
16492 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16493 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16494 file.write_all(&[255]).expect("damage one byte");
16495 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16496 fs::remove_file(damaged).expect("remove scratch file");
16497 }
16498
16499 #[test]
16500 fn damaged_lazy_dictionary_payload_is_an_error() {
16501 let path = path("damaged-dictionary");
16502 let mut writer = Writer::create(
16503 &path,
16504 "items",
16505 vec![
16506 Field::required("id", LogicalType::Integer),
16507 Field::new("text", LogicalType::Varchar),
16508 ],
16509 )
16510 .expect("new file");
16511 writer.append(&sample()).expect("stripe written");
16512 writer.finish().expect("commit");
16513
16514 let reader = Reader::open(&path).expect("valid directory");
16515 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16516 let mut header = [0; DICTIONARY_HEADER];
16519 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16520 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16523 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16524 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16525 let bits = (width & !DICTIONARY_FLAGS) as usize;
16526 let mut start = [0; 8];
16527 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16528 read_at(&reader.file, at, &mut start).expect("the first block's start");
16529 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16530 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16531 file.write_all(&[255]).expect("damage dictionary payload");
16532
16533 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16534 let error =
16535 chunk.validate_external().expect_err("payload corruption must reach the caller");
16536 assert!(error.message().contains("payload checksum differs"), "{error}");
16537 fs::remove_file(path).expect("remove scratch file");
16538 }
16539
16540 #[test]
16550 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16551 let path = path("dictionary-decide");
16552 let rows = 20_000;
16553 let unique =
16555 |row: usize| format!("{row:09} a value that appears exactly once in the table");
16556 let repeated = |row: usize| unique(row / 40);
16558 let mut writer = Writer::create(
16559 &path,
16560 "items",
16561 vec![
16562 Field::required("unique", LogicalType::Varchar),
16563 Field::required("repeated", LogicalType::Varchar),
16564 ],
16565 )
16566 .expect("new file");
16567 for part in (0..rows).step_by(1_000) {
16568 let span = part..(part + 1_000).min(rows);
16569 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16570 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16571 writer
16572 .append(
16573 &Chunk::new(vec![
16574 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16575 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16576 ])
16577 .expect("two columns"),
16578 )
16579 .expect("a part");
16580 }
16581 writer.finish().expect("commit");
16582
16583 let reader = Reader::open(&path).expect("reopen from disk");
16584 assert!(
16585 reader.table.dictionaries[0].is_none(),
16586 "a column with no repeats has nothing to say twice"
16587 );
16588 assert!(
16589 reader.table.dictionaries[1].is_some(),
16590 "a column whose values come round again keeps its dictionary"
16591 );
16592 let mut first = 0;
16593 for part in 0..reader.parts() {
16594 let chunk = reader.read(part, &[0, 1]).expect("a part");
16595 for row in 0..chunk.len() {
16596 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16597 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16598 }
16599 first += chunk.len();
16600 }
16601 assert_eq!(first, rows, "every row was read back");
16602 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16603 let size = fs::metadata(&path).expect("the file is there").len() as usize;
16604 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16605 fs::remove_file(path).expect("remove scratch file");
16606 }
16607
16608 #[test]
16621 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16622 let path = path("dictionary-blocks");
16623 let value = |row: usize| {
16624 let row = row.saturating_sub(8_000);
16625 format!("{row:07} a value long enough to be worth a payload block")
16626 };
16627 let parts = 40;
16628 let per_part = 1000;
16629 let mut writer =
16630 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16631 .expect("new file");
16632 for part in 0..parts {
16633 let values = (0..per_part)
16634 .map(|row| Value::Varchar(value(part * per_part + row)))
16635 .collect::<Vec<_>>();
16636 let chunk = Chunk::new(vec![
16637 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16638 ])
16639 .expect("matching rows");
16640 writer.append(&chunk).expect("a part");
16641 }
16642 writer.finish().expect("commit");
16643
16644 let reader = Reader::open(&path).expect("reopen from disk");
16645 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16646 assert!(
16647 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16648 "the dictionary has to be several blocks for this to be testing anything"
16649 );
16650 for part in [0, parts - 1] {
16651 let chunk = reader.read(part, &[0]).expect("a part");
16652 chunk.validate_external().expect("every payload block checks out");
16653 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16654 }
16655
16656 let mut header = [0; DICTIONARY_HEADER];
16658 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16659 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16660 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16661 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16662 let bits = (width & !DICTIONARY_FLAGS) as usize;
16663 let mut place = [0; 16];
16664 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16665 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16666 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16667 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16668 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16669 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16670 file.write_all(&[255]).expect("damage the last payload block");
16671 let reader = Reader::open(&path).expect("the directory and the index are untouched");
16672 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16673 let error = chunk.validate_external().expect_err("the damage must reach the caller");
16674 assert!(error.message().contains("payload checksum differs"), "{error}");
16675 fs::remove_file(path).expect("remove scratch file");
16676 }
16677
16678 #[test]
16692 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16693 let path = path("dictionary-offsets");
16694 let value = |row: usize| {
16695 let row = row % 5_000;
16696 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16697 };
16698 let rows = 6_000;
16699 let mut writer =
16700 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16701 .expect("new file");
16702 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16703 for part in values.chunks(1_000) {
16704 let chunk =
16705 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16706 .expect("matching rows");
16707 writer.append(&chunk).expect("a part");
16708 }
16709 writer.finish().expect("commit");
16710
16711 let reader = Reader::open(&path).expect("reopen from disk");
16712 assert!(
16713 rows > TEXT_PAYLOAD_VALUES * 4,
16714 "the dictionary has to be several blocks for this to be testing anything"
16715 );
16716 for part in 0..rows / 1_000 {
16717 let chunk = reader.read(part, &[0]).expect("a part");
16718 for row in 0..1_000 {
16719 let row = part * 1_000 + row;
16720 assert_eq!(
16721 chunk.value_at(row % 1_000, 0),
16722 Value::Varchar(value(row)),
16723 "value {row}"
16724 );
16725 }
16726 }
16727 for _ in 0..2 {
16730 for part in 0..rows / 1_000 {
16731 let chunk = reader.read(part, &[0]).expect("a part");
16732 let mut lens = vec![0_i64; 1_000];
16733 let column = chunk.column(0).expect("one column");
16734 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16735 for (row, &len) in lens.iter().enumerate() {
16736 let row = part * 1_000 + row;
16737 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16738 }
16739 }
16740 }
16741 fs::remove_file(path).expect("remove scratch file");
16742 }
16743
16744 #[test]
16746 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16747 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16748 ends.extend([3, 3, 10]);
16749 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16750 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16751 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16752 let long = [5, 70_005, 70_006];
16754 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16755 assert_eq!(lens, [5, 70_000, 1]);
16756 let mut read = Vec::new();
16757 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16758 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16759 ends.push(9);
16760 assert!(lengths_of(&ends).is_none());
16761 }
16762
16763 #[test]
16775 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16776 let path = path("dictionary-once");
16777 let parts = 8;
16778 let per_part = 500;
16779 let value =
16780 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16781 let mut writer =
16782 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16783 .expect("new file");
16784 for part in 0..parts {
16785 let values = (0..per_part)
16786 .map(|row| Value::Varchar(value(part * per_part + row)))
16787 .collect::<Vec<_>>();
16788 let chunk = Chunk::new(vec![
16789 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16790 ])
16791 .expect("matching rows");
16792 writer.append(&chunk).expect("a part");
16793 }
16794 writer.finish().expect("commit");
16795
16796 let reader = Reader::open(&path).expect("reopen from disk");
16797 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16798 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16799
16800 let workers = 16;
16801 let gate = std::sync::Barrier::new(workers);
16802 std::thread::scope(|scope| {
16803 for worker in 0..workers {
16804 let reader = reader.clone();
16805 let gate = &gate;
16806 scope.spawn(move || {
16807 gate.wait();
16808 let chunk = reader.read(worker % parts, &[0]).expect("a part");
16809 assert_eq!(
16810 chunk.value_at(0, 0),
16811 Value::Varchar(value((worker % parts) * per_part))
16812 );
16813 });
16814 }
16815 });
16816
16817 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16818 fs::remove_file(path).expect("remove scratch file");
16819 }
16820
16821 #[test]
16826 fn a_damaged_sorted_order_is_an_error() {
16827 let path = path("damaged-order");
16828 let mut writer = Writer::create(
16829 &path,
16830 "items",
16831 vec![
16832 Field::required("id", LogicalType::Integer),
16833 Field::new("text", LogicalType::Varchar),
16834 ],
16835 )
16836 .expect("new file");
16837 writer.append(&sample()).expect("stripe written");
16838 writer.finish().expect("commit");
16839
16840 let reader = Reader::open(&path).expect("valid directory");
16841 let page = reader.table.dictionaries[1].expect("string dictionary page");
16842 let mut header = [0; DICTIONARY_HEADER];
16843 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16844 let index_len = dictionary_index_len(&header);
16845 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16846 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16847 file.write_all(&[255]).expect("damage the order");
16848
16849 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16850 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16851 assert!(error.message().contains("rank checksum differs"), "{error}");
16852 fs::remove_file(path).expect("remove scratch file");
16853 }
16854
16855 #[test]
16859 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16860 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16863 let path = path("dictionary-order");
16864 let mut writer =
16865 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16866 .expect("new file");
16867 writer
16868 .append(
16869 &Chunk::new(vec![
16870 Vector::from_values(
16871 LogicalType::Varchar,
16872 &spellings.map(|text| Value::Varchar(text.into())),
16873 )
16874 .expect("strings"),
16875 ])
16876 .expect("one column"),
16877 )
16878 .expect("stripe written");
16879 writer.finish().expect("commit");
16880
16881 let reader = Reader::open(&path).expect("valid directory");
16882 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16883 let count = dictionary.ranks().expect("a v10 file stores one");
16884 assert_eq!(count, spellings.len(), "every distinct value has a rank");
16885 let order = (0..count)
16886 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16887 .collect::<Vec<_>>();
16888 let mut seen = order.clone();
16889 seen.sort_unstable();
16890 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16891
16892 let ranked = order
16893 .iter()
16894 .map(|&code| {
16895 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16896 })
16897 .collect::<Vec<_>>();
16898 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16899 expected.sort();
16900 assert_eq!(ranked, expected, "rank order is value order");
16901
16902 for (rank, value) in expected.iter().enumerate() {
16905 assert_eq!(
16906 dictionary.compare_rank(rank, value).expect("compare"),
16907 Ordering::Equal,
16908 "rank {rank} is its own value"
16909 );
16910 if rank > 0 {
16911 assert_eq!(
16912 dictionary.compare_rank(rank - 1, value).expect("compare"),
16913 Ordering::Less,
16914 "rank {rank} follows the one before it"
16915 );
16916 }
16917 }
16918 fs::remove_file(path).expect("remove scratch file");
16919 }
16920
16921 #[test]
16928 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16929 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16930 let path = path("dictionaries-at-once");
16931 let fields = (0..sizes.len())
16932 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16933 .collect::<Vec<_>>();
16934 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16935 let rows = 10_000_usize;
16936 for start in (0..rows).step_by(1_024) {
16937 let columns = sizes
16938 .iter()
16939 .enumerate()
16940 .map(|(column, &size)| {
16941 let values = (start..(start + 1_024).min(rows))
16942 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16943 .collect::<Vec<_>>();
16944 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16945 })
16946 .collect::<Vec<_>>();
16947 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16948 }
16949 writer.finish().expect("commit");
16950
16951 let reader = Reader::open(&path).expect("valid directory");
16952 for (column, &size) in sizes.iter().enumerate() {
16953 let dictionary =
16954 reader.dictionary(column).expect("read").expect("a string column has one");
16955 let count = dictionary.ranks().expect("a v10 file stores one");
16956 assert_eq!(count, size, "column {column} has its own distinct count");
16957 let ranked = (0..count)
16958 .map(|rank| {
16959 let code = dictionary.code_at_rank(rank).expect("a code");
16960 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16961 })
16962 .collect::<Vec<_>>();
16963 let expected = (0..size)
16964 .map(|value| format!("c{column}-{value:05}").into_bytes())
16965 .collect::<Vec<_>>();
16966 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16967 }
16968 fs::remove_file(path).expect("remove scratch file");
16969 }
16970
16971 #[test]
16979 fn a_large_dictionary_ranks_in_value_order() {
16980 let path = path("dictionary-large-rank");
16981 let value = |row: u64| {
16982 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16983 match row % 3 {
16984 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16985 1 => format!("{mixed}"),
16986 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16987 }
16988 };
16989 let distinct = 70_000;
16990 let parts = 4 * distinct / 1000;
16991 let per_part = 1000;
16992 let mut writer =
16993 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16994 .expect("new file");
16995 for part in 0..parts {
16996 let values = (0..per_part)
16997 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
16998 .collect::<Vec<_>>();
16999 let chunk = Chunk::new(vec![
17000 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
17001 ])
17002 .expect("matching rows");
17003 writer.append(&chunk).expect("a part");
17004 }
17005 writer.finish().expect("commit");
17006
17007 let reader = Reader::open(&path).expect("reopen from disk");
17008 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17009 let count = dictionary.ranks().expect("a ranked dictionary");
17010 assert_eq!(count, distinct as usize, "every distinct value has a rank");
17011 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
17012 let ranked = (0..count)
17013 .map(|rank| {
17014 let code = dictionary.code_at_rank(rank).expect("a code");
17015 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
17016 })
17017 .collect::<Vec<_>>();
17018 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
17019 expected.sort();
17020 assert_eq!(ranked, expected, "rank order is value order");
17021 fs::remove_file(path).expect("remove scratch file");
17022 }
17023
17024 #[test]
17037 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
17038 let path = path("windowed-directory");
17039 let fields = vec![
17040 Field::required("id", LogicalType::BigInt),
17041 Field::required("word", LogicalType::Varchar),
17042 Field::new("score", LogicalType::Double),
17043 ];
17044 let mut writer = Writer::create(&path, "items", fields).expect("new file");
17045 for part in 0..70_i64 {
17046 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
17047 let words = (0..100)
17048 .map(|row| Value::Varchar(format!("word {}", row % 13)))
17049 .collect::<Vec<_>>();
17050 let scores = (0..100)
17051 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
17052 .collect::<Vec<_>>();
17053 let chunk = Chunk::new(vec![
17054 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
17055 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
17056 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
17057 ])
17058 .expect("three columns");
17059 writer.append(&chunk).expect("a part");
17060 }
17061 writer.finish().expect("commit");
17062
17063 let catalog = Catalog::open(&path).expect("reopen");
17064 let entry = catalog.entries.first().expect("one table").directory;
17065 let (offset, length) = (entry.offset, entry.length as usize);
17066 let mut bytes = vec![0; length];
17067 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
17068 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
17069 let whole = decode_directory(&bytes, catalog.size).expect("whole");
17070 assert!(whole.stripes.len() > 1, "the table should span stripes");
17071 for size in [1, 7, 33, 4_096] {
17072 let mut cursor = Cursor::over(&catalog.file, offset, length);
17073 cursor.window.as_mut().expect("a window").size = size;
17074 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
17075 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
17076 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
17077 let mut stored = 0;
17078 for (column, (left, held)) in
17079 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
17080 {
17081 match (left, held) {
17082 (None, None) => {}
17083 (
17084 Some(super::Frequencies::Stored { span, values, entries }),
17085 Some(super::Frequencies::Held(summary)),
17086 ) => {
17087 let mut one = vec![0; span.length as usize];
17088 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
17089 let read = decode_summary(
17090 &mut Cursor::new(&one),
17091 &whole.fields[column],
17092 whole.rows,
17093 *values,
17094 )
17095 .expect("a valid synopsis")
17096 .expect("one is there");
17097 assert_eq!(*entries, read.entries.len());
17098 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
17099 stored += 1;
17100 }
17101 other => panic!("column {column} came back as {other:?}"),
17102 }
17103 }
17104 assert!(stored >= 2, "only {stored} synopses were left in the file");
17105 }
17106 let reader = catalog.table("items").expect("the table");
17107 assert!(reader.frequency_heads[1].get().is_none());
17108 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
17109 let first = reader.frequency_heads[1].get().expect("decoded synopsis");
17110 let clone = reader.clone();
17111 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
17112 assert!(Arc::ptr_eq(first, clone.frequency_heads[1].get().expect("same synopsis")));
17113 fs::remove_file(path).expect("remove scratch file");
17114 }
17115
17116 #[test]
17117 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
17118 let path = path("file-checksum");
17119 let bytes = (0..200_000_u32)
17120 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
17121 .collect::<Vec<_>>();
17122 fs::write(&path, &bytes).expect("scratch file");
17123 let file = File::open(&path).expect("open");
17124 for (offset, length) in [
17125 (0, 0),
17126 (3, 1),
17127 (5, 31),
17128 (0, 32),
17129 (9, 33),
17130 (1, 65_536),
17131 (7, 65_567),
17132 (0, 200_000),
17133 (11, 131_101),
17134 ] {
17135 let whole = checksum(&bytes[offset..offset + length]);
17136 assert_eq!(
17137 file_checksum(&file, offset as u64, length).expect("read"),
17138 whole,
17139 "{offset} {length}"
17140 );
17141 }
17142 fs::remove_file(path).expect("remove scratch file");
17143 }
17144
17145 #[test]
17146 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
17147 let path = path("synopsis-keeps-no-block");
17148 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
17149 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
17150 for _ in 0..3 {
17151 values.extend((0..3_000).step_by(5).map(spelled));
17152 }
17153 let mut writer =
17154 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17155 .expect("new file");
17156 for part in values.chunks(1_024) {
17157 writer
17158 .append(
17159 &Chunk::new(vec![
17160 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17161 ])
17162 .expect("one column"),
17163 )
17164 .expect("a part");
17165 }
17166 writer.finish().expect("commit");
17167
17168 let reader = Reader::open(&path).expect("reopen from disk");
17169 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17170 let resting = dictionary.footprint();
17171 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17172 assert_eq!(prefix.entries.len(), 512);
17173 for (value, count) in &prefix.entries {
17174 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
17175 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
17176 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
17177 }
17178 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
17179 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
17180 assert_eq!(again.entries, prefix.entries);
17181 fs::remove_file(path).expect("remove scratch file");
17182 }
17183
17184 #[test]
17191 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
17192 let path = path("character-lengths");
17193 let spellings = (0..2_500)
17194 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
17195 .collect::<Vec<_>>();
17196 let mut writer =
17197 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
17198 .expect("new file");
17199 for part in spellings.chunks(1_024) {
17200 writer
17201 .append(
17202 &Chunk::new(vec![
17203 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17204 ])
17205 .expect("one column"),
17206 )
17207 .expect("a part");
17208 }
17209 writer.finish().expect("commit");
17210
17211 let reader = Reader::open(&path).expect("reopen from disk");
17212 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17213 let resting = dictionary.footprint();
17214 let mut lens = Vec::new();
17215 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
17216 let counted = dictionary.footprint() - resting;
17217 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17218 assert!(
17219 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17220 "counting kept {counted} bytes, more than a count a value"
17221 );
17222 let expected = (0..dictionary.len())
17223 .map(|code| {
17224 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
17225 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
17226 .expect("small")
17227 })
17228 .collect::<Vec<_>>();
17229 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
17230 let mut again = Vec::new();
17231 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
17232 assert_eq!(again, lens, "the kept counts answer the second time");
17233 fs::remove_file(path).expect("remove scratch file");
17234 }
17235
17236 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
17238 let path = path(label);
17239 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
17240 let mut writer =
17241 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17242 .expect("new file");
17243 for part in values.chunks(1_024) {
17244 writer
17245 .append(
17246 &Chunk::new(vec![
17247 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17248 ])
17249 .expect("one column"),
17250 )
17251 .expect("a part");
17252 }
17253 writer.finish().expect("commit");
17254 let reader = Reader::open(&path).expect("reopen from disk");
17255 (path, reader)
17256 }
17257
17258 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
17264 let codes = (0..len)
17265 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
17266 .collect::<Vec<_>>();
17267 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
17268 (codes, valid)
17269 }
17270
17271 #[test]
17278 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
17279 let spellings = (0..2_500)
17280 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
17281 .collect::<Vec<_>>();
17282 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
17283 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17284 let (codes, valid) = scattered_rows(spellings.len());
17285 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
17286 .expect("every code is inside")
17287 .with_validity(Validity::from_run(&valid));
17288
17289 let resting = dictionary.footprint();
17290 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
17291 .expect("length reads");
17292 let counted = dictionary.footprint() - resting;
17293 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
17294 assert!(
17295 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
17296 "length over a vector with nulls kept {counted} bytes, more than a count a value"
17297 );
17298 let expected = (0..rows.len())
17299 .map(|row| match valid[row] {
17300 true => Value::BigInt(
17301 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
17302 ),
17303 false => Value::Null,
17304 })
17305 .collect::<Vec<_>>();
17306 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
17307 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
17308 fs::remove_file(path).expect("remove scratch file");
17309 }
17310
17311 #[test]
17321 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
17322 let spellings = (0..2_500)
17323 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
17324 .collect::<Vec<_>>();
17325 let (path, reader) = stored_spellings("string-kernels", &spellings);
17326 let page = reader.table.dictionaries[0].expect("a string column has one");
17327 let starved =
17328 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
17329 .expect("a dictionary opens whatever it may keep");
17330 let starved = Arc::new(starved);
17331 let (codes, valid) = scattered_rows(spellings.len());
17332 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
17333 .expect("every code is inside")
17334 .with_validity(Validity::from_run(&valid));
17335 let expected = |each: &dyn Fn(&str) -> String| {
17336 (0..rows.len())
17337 .map(|row| match valid[row] {
17338 true => Value::Varchar(each(&spellings[codes[row] as usize])),
17339 false => Value::Null,
17340 })
17341 .collect::<Vec<_>>()
17342 };
17343 let answers =
17344 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
17345
17346 let resting = starved.footprint();
17349 let ends = spellings.len() * size_of::<u32>();
17350 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
17351 .expect("lower reads");
17352 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
17353 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
17354
17355 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
17356 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
17357 let cut =
17358 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
17359 .expect("substring reads");
17360 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
17361 assert_eq!(answers(&cut), expected(&cut_of), "substring");
17362 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
17363
17364 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17367 .expect("upper reads");
17368 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
17369 let payload = spellings.iter().map(String::len).sum::<usize>();
17370 assert!(
17371 starved.footprint() >= resting + payload,
17372 "a visit that has dropped a column's worth of blocks keeps what it reads"
17373 );
17374 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
17375 .expect("upper reads kept blocks");
17376 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
17377 fs::remove_file(path).expect("remove scratch file");
17378 }
17379
17380 #[test]
17390 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
17391 let path = path("dictionary-sweep");
17392 let spellings = (0..2_500)
17395 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17396 .collect::<Vec<_>>();
17397 let mut writer =
17398 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17399 .expect("new file");
17400 for part in spellings.chunks(1_024) {
17403 writer
17404 .append(
17405 &Chunk::new(vec![
17406 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17407 ])
17408 .expect("one column"),
17409 )
17410 .expect("stripe written");
17411 }
17412 writer.finish().expect("commit");
17413
17414 let reader = Reader::open(&path).expect("valid directory");
17415 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17416 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17417 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17418 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17419 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17420 }
17421
17422 let resting = dictionary.footprint();
17423 let sweep = || {
17424 let mut swept: Vec<Vec<u8>> = Vec::new();
17425 let mut at = 0;
17426 let mut calls = 0;
17427 while at < dictionary.len() {
17428 let stopped = dictionary
17429 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17430 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17431 swept.push(text.to_vec());
17432 Ok(())
17433 })
17434 .expect("a sweep reads");
17435 assert!(stopped > at, "a sweep moves");
17436 at = stopped;
17437 calls += 1;
17438 }
17439 assert_eq!(calls, 3, "a sweep hands over one block at a time");
17440 swept
17441 };
17442 let swept = sweep();
17443 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17444 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17445 let after = dictionary.footprint();
17446 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17447
17448 let read = (0..dictionary.len())
17449 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17450 .collect::<Vec<_>>();
17451 assert_eq!(swept, read, "a sweep answers what a point read answers");
17452 let grown = dictionary.footprint() - after;
17456 assert!(
17457 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17458 "a point read of a kept block decodes nothing, and {grown} bytes grew"
17459 );
17460 fs::remove_file(path).expect("remove scratch file");
17461 }
17462
17463 #[test]
17464 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17465 let path = path("narrow-substring-signature");
17466 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17467 let mut grams = Vec::new();
17468 for text in blocks {
17469 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17470 for gram in text.windows(4) {
17471 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17472 bits[bit / 8] |= 1 << (bit % 8);
17473 }
17474 }
17475 grams.extend(bits);
17476 }
17477 fs::write(&path, &grams).expect("scratch file");
17478 let file = File::open(&path).expect("open scratch file");
17479 let signatures = NativeGrams {
17480 start: 0,
17481 length: grams.len(),
17482 width: NARROW_GRAM_BYTES,
17483 hash: checksum(&grams),
17484 verdicts: Mutex::new(Vec::new()),
17485 };
17486 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17487 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17488 assert!(signatures.footprint() > 0, "a verdict is remembered");
17489 let again = signatures.verdicts(&file, b"google").expect("remembered");
17490 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17491
17492 let damaged = NativeGrams {
17493 hash: signatures.hash ^ 1,
17494 verdicts: Mutex::new(Vec::new()),
17495 ..signatures
17496 };
17497 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17498 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17499 fs::remove_file(path).expect("remove scratch file");
17500 }
17501
17502 #[test]
17503 fn a_damaged_substring_signature_is_checked_only_when_used() {
17504 let path = path("damaged-substring-signature");
17505 let mut writer =
17506 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17507 .expect("new file");
17508 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17509 writer
17510 .append(
17511 &Chunk::new(vec![
17512 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17513 ])
17514 .expect("one column"),
17515 )
17516 .expect("stripe written");
17517 writer.finish().expect("commit");
17518
17519 let reader = Reader::open(&path).expect("valid directory");
17520 let page = reader.table.dictionaries[0].expect("string dictionary page");
17521 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17522 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17523 .expect("last signature byte");
17524 file.write_all(&[255]).expect("damage signature");
17525 let reader = Reader::open(&path).expect("the directory is still valid");
17526 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17527 let error = dictionary
17528 .text_block_might_contain(0, b"goog")
17529 .expect_err("a used signature checks its own checksum");
17530 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17531 fs::remove_file(path).expect("remove scratch file");
17532 }
17533
17534 #[test]
17545 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17546 let path = path("dictionary-sweep-short-run");
17547 let spellings = (0..2_800)
17548 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17549 .collect::<Vec<_>>();
17550 let mut writer =
17551 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17552 .expect("new file");
17553 for part in spellings.chunks(1_024) {
17554 writer
17555 .append(
17556 &Chunk::new(vec![
17557 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17558 ])
17559 .expect("one column"),
17560 )
17561 .expect("stripe written");
17562 }
17563 writer.finish().expect("commit");
17564
17565 let reader = Reader::open(&path).expect("valid directory");
17566 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17567 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17568 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17569 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17570 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17571
17572 let mut swept: Vec<Vec<u8>> = Vec::new();
17573 let mut at = 0;
17574 while at < dictionary.len() {
17575 let stopped = dictionary
17576 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17577 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17578 swept.push(text.to_vec());
17579 Ok(())
17580 })
17581 .expect("a sweep reads");
17582 assert!(stopped > at, "a sweep moves");
17583 at = stopped;
17584 }
17585 let read = (0..dictionary.len())
17586 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17587 .collect::<Vec<_>>();
17588 assert_eq!(swept, read, "a sweep answers what a point read answers");
17589 fs::remove_file(path).expect("remove scratch file");
17590 }
17591
17592 #[test]
17601 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17602 let path = path("dictionary-unpacked-ends");
17603 let spellings = (0..2_800)
17604 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17605 .collect::<Vec<_>>();
17606 let mut writer =
17607 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17608 .expect("new file");
17609 for part in spellings.chunks(1_024) {
17610 writer
17611 .append(
17612 &Chunk::new(vec![
17613 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17614 ])
17615 .expect("one column"),
17616 )
17617 .expect("stripe written");
17618 }
17619 writer.finish().expect("commit");
17620
17621 let reader = Reader::open(&path).expect("valid directory");
17622 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17623 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17624 let wanted = (0..spellings.len())
17625 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17626 .collect::<Vec<_>>();
17627
17628 let pass = |what: &str| {
17629 for (index, value) in wanted.iter().enumerate() {
17630 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17631 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17632 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17633 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17634 }
17635 };
17636 pass("the first pass");
17637 pass("the second pass");
17638
17639 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17643 let mut whole = vec![0i64; wanted.len()];
17644 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17645 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17646 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17647 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17648 let mut through = vec![0i64; codes.len()];
17649 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17650 for (row, &code) in codes.iter().enumerate() {
17651 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17652 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17653 assert_eq!(through[row], one as i64, "row {row} a row at a time");
17654 }
17655
17656 let fresh = Reader::open(&path).expect("valid directory");
17659 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17660 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17661 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17662 let mut short = vec![0i64; few.len()];
17663 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17664 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17665 assert_eq!(short, expected, "the packed ends answer what the table answers");
17666 fs::remove_file(path).expect("remove scratch file");
17667 }
17668
17669 #[test]
17684 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17685 let spellings = (0..3_000)
17686 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17687 .collect::<Vec<_>>();
17688 let mut read = Vec::new();
17689 for layout in ["outside", "inside", "behind"] {
17690 let mut dictionary = GlobalDictionary::new();
17691 for text in &spellings {
17692 dictionary.code(text).expect("a code for every spelling");
17693 }
17694 dictionary.finish_blocks().expect("the last block encodes");
17695 let order = dictionary.ranked(None).expect("a sorted order");
17696 let laid = |from: u64| {
17698 let mut at = from;
17699 dictionary
17700 .blocks
17701 .iter()
17702 .map(|block| {
17703 let place =
17704 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17705 at += block.len() as u64;
17706 place
17707 })
17708 .collect::<Vec<_>>()
17709 };
17710 let payload = dictionary.blocks.concat();
17711 let scattered = layout != "behind";
17712 let (bytes, encoded, offset, length) = if layout == "outside" {
17713 let mut bytes = vec![0; HEADER as usize];
17714 bytes.extend_from_slice(&payload);
17715 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17716 .expect("an encoding");
17717 let offset = bytes.len() as u64;
17718 bytes.extend_from_slice(&encoded.index);
17719 bytes.extend_from_slice(&encoded.ranks);
17720 bytes.extend_from_slice(&encoded.grams);
17721 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17722 (bytes, encoded, offset, length)
17723 } else {
17724 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17727 .expect("an encoding");
17728 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17729 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17730 .expect("an encoding");
17731 let mut bytes = encoded.index.clone();
17732 bytes.extend_from_slice(&encoded.ranks);
17733 bytes.extend_from_slice(&encoded.grams);
17734 bytes.extend_from_slice(&payload);
17735 let length = bytes.len();
17736 (bytes, encoded, 0, length)
17737 };
17738 let path = path(&format!("blocks-{layout}"));
17739 fs::write(&path, &bytes).expect("the dictionary is written on its own");
17740 let file = Arc::new(File::open(&path).expect("it opens again"));
17741 let page = Page {
17742 offset,
17743 length: u32::try_from(length).expect("a test dictionary is small"),
17744 hash: checksum(&encoded.index),
17745 };
17746 let opened =
17747 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17748 .expect("a dictionary laid out either way opens");
17749 let mut swept: Vec<Vec<u8>> = Vec::new();
17750 let mut at = 0;
17751 while at < opened.len() {
17752 at = opened
17753 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17754 swept.push(text.to_vec());
17755 Ok(())
17756 })
17757 .expect("a sweep reads");
17758 }
17759 fs::remove_file(&path).expect("clean up");
17760 read.push(swept);
17761 }
17762 let wanted =
17763 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17764 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17765 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17766 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17767 }
17768
17769 #[test]
17777 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17778 let path = path("dictionary-budget");
17779 let spellings = (0..2_500)
17780 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17781 .collect::<Vec<_>>();
17782 let mut writer =
17783 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17784 .expect("new file");
17785 for part in spellings.chunks(1_024) {
17786 writer
17787 .append(
17788 &Chunk::new(vec![
17789 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17790 ])
17791 .expect("one column"),
17792 )
17793 .expect("stripe written");
17794 }
17795 writer.finish().expect("commit");
17796
17797 let reader = Reader::open(&path).expect("valid directory");
17798 let page = reader.table.dictionaries[0].expect("a string column has one");
17799 let file = Arc::clone(&reader.file);
17800 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17801 .expect("a dictionary opens whatever it may keep");
17802
17803 let resting = starved.footprint();
17804 let mut swept: Vec<Vec<u8>> = Vec::new();
17805 let mut at = 0;
17806 while at < starved.len() {
17807 at = starved
17808 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17809 swept.push(text.to_vec());
17810 Ok(())
17811 })
17812 .expect("a sweep reads");
17813 }
17814 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17815 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17816
17817 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17818 let read = (0..generous.len())
17819 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17820 .collect::<Vec<_>>();
17821 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17822 fs::remove_file(path).expect("remove scratch file");
17823 }
17824
17825 #[test]
17828 fn a_part_is_checked_once_per_open_reader() {
17829 let path = path("checked-once");
17830 let mut writer = Writer::create(
17831 &path,
17832 "items",
17833 vec![
17834 Field::required("id", LogicalType::Integer),
17835 Field::new("text", LogicalType::Varchar),
17836 ],
17837 )
17838 .expect("new file");
17839 writer.append(&sample()).expect("stripe written");
17840 writer.finish().expect("commit");
17841
17842 let reader = Reader::open(&path).expect("valid directory");
17843 let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17844 assert!(reader.is_verified(0), "the part is remembered as checked");
17845 let page = reader.table.stripes[0].pages[0];
17846 let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17847 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17848 file.write_all(&[0xa5]).expect("damage page");
17849 if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17850 assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17851 }
17852 let fresh = Reader::open(&path).expect("valid directory");
17853 let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17854 assert!(error.message().contains("column page checksum differs"), "{error}");
17855 assert_eq!(first.len(), 2);
17856 fs::remove_file(path).expect("remove scratch file");
17857 }
17858
17859 #[test]
17860 fn damaged_membership_cannot_skip_a_string_page() {
17861 let path = path("damaged-membership");
17862 let mut writer = Writer::create(
17863 &path,
17864 "items",
17865 vec![
17866 Field::required("id", LogicalType::Integer),
17867 Field::new("text", LogicalType::Varchar),
17868 ],
17869 )
17870 .expect("new file");
17871 writer.append(&sample()).expect("stripe written");
17872 writer.finish().expect("commit");
17873
17874 let reader = Reader::open(&path).expect("valid directory");
17875 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17876 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17877 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17878 file.write_all(&[255]).expect("damage membership");
17879 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17880 assert!(error.message().contains("membership page checksum differs"), "{error}");
17881 fs::remove_file(path).expect("remove scratch file");
17882 }
17883
17884 #[test]
17885 fn membership_delta_stream_is_sorted_exact_and_bounded() {
17886 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17887 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17888 let encoded = encode_membership(&unique);
17889 assert_eq!(
17890 decode_membership(&encoded).expect("valid membership"),
17891 [4, 9, 72, 900, u32::MAX]
17892 );
17893 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17896 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17897 assert_eq!(
17898 decode_membership(&encode_membership(&merged)).expect("valid membership"),
17899 unique
17900 );
17901 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17902 assert!(
17903 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17904 "a value past u32 is invalid"
17905 );
17906 }
17907
17908 #[test]
17909 fn a_global_dictionary_may_be_larger_than_one_column_page() {
17910 let dictionary = Page {
17911 offset: HEADER,
17912 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17913 hash: 0,
17914 };
17915 let table = Table {
17916 name: "items".to_owned(),
17917 fields: vec![Field::new("text", LogicalType::Varchar)],
17918 stripes: Vec::new(),
17919 rows: 0,
17920 dictionaries: vec![Some(dictionary)],
17921 dictionary_payloads: Vec::new(),
17922 demoted: Vec::new(),
17923 distincts: vec![None],
17924 frequencies: vec![None],
17925 ordinal_bounds: Vec::new(),
17926 pair_frequencies: Vec::new(),
17927 frequency_texts: Vec::new(),
17928 host_groups: None,
17929 clustering: None,
17930 constraints: Constraints::default(),
17931 generation: 1,
17932 sections: Vec::new(),
17933 };
17934 let directory = encode_directory(&table).expect("directory");
17935 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17936
17937 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17938 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17939 }
17940
17941 #[test]
17942 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17943 let path = path("constant-codes");
17944 let mut writer =
17945 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17946 .expect("new file");
17947 let empty = vec![Value::Varchar(String::new()); 1024];
17948 for _ in 0..4 {
17949 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17950 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17951 }
17952 writer.finish().expect("commit");
17953
17954 let reader = Reader::open(&path).expect("valid directory");
17955 let pages = reader.layout().columns.first().expect("one column").pages;
17956 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17960 let read = reader.read(3, &[0]).expect("the last part back");
17961 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17962 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17963 fs::remove_file(path).expect("remove scratch file");
17964 }
17965
17966 #[test]
17967 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17968 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17971 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17972 assert!(format!("{error}").contains("not of its type"), "{error}");
17973 let low = integer::encode(&[i64::MIN]).expect("a chunk");
17974 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17975 let zero = integer::encode(&[0]).expect("a chunk");
17976 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17977 }
17978
17979 #[test]
17980 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17981 let mut state: u32 = 0x9e37_79b9;
17985 let spread: Vec<u32> = (0..1024)
17986 .map(|_| {
17987 state ^= state << 13;
17988 state ^= state >> 17;
17989 state ^= state << 5;
17990 state
17991 })
17992 .collect();
17993 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17994 let near: Vec<u32> = (0..1024).collect();
17995 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
17996 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
17997 }
17998
17999 #[test]
18005 fn two_writes_of_the_same_rows_give_the_same_bytes() {
18006 fn written(path: &PathBuf) {
18007 let fields = (0..40)
18008 .map(|column| {
18009 let ty =
18010 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
18011 Field::new(format!("c{column}"), ty)
18012 })
18013 .collect::<Vec<_>>();
18014 let mut writer = Writer::create(path, "wide", fields).expect("new file");
18015 for part in 0..70_u64 {
18016 let columns = (0..40)
18017 .map(|column| {
18018 let values = (0..64_u64)
18019 .map(|row| {
18020 let seed = part.wrapping_mul(31).wrapping_add(row);
18021 if column % 4 == 0 {
18022 Value::Varchar(format!("v{}", seed % 17))
18023 } else {
18024 Value::BigInt(i64::try_from(seed % 97).expect("small"))
18025 }
18026 })
18027 .collect::<Vec<_>>();
18028 let ty = if column % 4 == 0 {
18029 LogicalType::Varchar
18030 } else {
18031 LogicalType::BigInt
18032 };
18033 Vector::from_values(ty, &values).expect("a column")
18034 })
18035 .collect::<Vec<_>>();
18036 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
18037 }
18038 writer.finish().expect("commit");
18039 }
18040
18041 let first = path("repeatable-one");
18042 let second = path("repeatable-two");
18043 written(&first);
18044 written(&second);
18045 let left = fs::read(&first).expect("the first file");
18046 let right = fs::read(&second).expect("the second file");
18047 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
18048 assert!(left == right, "two writes of the same rows differ in their bytes");
18049
18050 let reader = Reader::open(&first).expect("valid directory");
18053 assert_eq!(reader.table().rows(), 70 * 64);
18054 let read = reader.read(0, &[0, 1]).expect("the first part back");
18055 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
18056 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
18057 fs::remove_file(first).expect("remove scratch file");
18058 fs::remove_file(second).expect("remove scratch file");
18059 }
18060
18061 fn three_tables(path: &PathBuf) {
18063 let writer = Writer::create(
18064 path,
18065 "region",
18066 vec![
18067 Field::new("r_key", LogicalType::Integer),
18068 Field::new("r_name", LogicalType::Varchar),
18069 ],
18070 )
18071 .expect("new file");
18072 let mut writer = writer;
18073 writer
18074 .append(
18075 &Chunk::new(vec![
18076 Vector::from_values(
18077 LogicalType::Integer,
18078 &[Value::Integer(0), Value::Integer(1)],
18079 )
18080 .expect("keys"),
18081 Vector::from_values(
18082 LogicalType::Varchar,
18083 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
18084 )
18085 .expect("names"),
18086 ])
18087 .expect("two columns"),
18088 )
18089 .expect("a part");
18090 let mut writer = writer
18091 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
18092 .expect("a second table");
18093 writer
18094 .append(
18095 &Chunk::new(vec![
18096 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
18097 ])
18098 .expect("one column"),
18099 )
18100 .expect("a part");
18101 let mut writer =
18102 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
18103 for part in 0..70_i64 {
18104 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
18105 writer
18106 .append(
18107 &Chunk::new(vec![
18108 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
18109 ])
18110 .expect("one column"),
18111 )
18112 .expect("a part");
18113 }
18114 writer.finish().expect("commit");
18115 }
18116
18117 #[test]
18118 fn three_tables_in_one_file_read_back_by_name() {
18119 let file = path("three-tables");
18120 three_tables(&file);
18121 let catalog = Catalog::open(&file).expect("a committed catalog");
18122 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
18123
18124 let region = catalog.table("region").expect("the first table");
18125 assert_eq!(region.table().rows(), 2);
18126 assert_eq!(
18127 region.read(0, &[1]).expect("names").value_at(1, 0),
18128 Value::Varchar("ASIA".to_owned())
18129 );
18130
18131 let wide = catalog.table("wide").expect("the third table");
18132 assert_eq!(wide.table().rows(), 70 * 64);
18133 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
18134
18135 let empty = catalog.table("empty").expect("the second table");
18138 assert_eq!(empty.table().rows(), 1);
18139 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
18140
18141 fs::remove_file(file).expect("remove scratch file");
18142 }
18143
18144 #[test]
18145 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
18146 let file = path("three-tables-missing");
18147 three_tables(&file);
18148 let catalog = Catalog::open(&file).expect("a committed catalog");
18149 let error = catalog.table("nation").expect_err("no such table");
18150 assert!(error.message().contains("nation"), "{}", error.message());
18151 fs::remove_file(file).expect("remove scratch file");
18152 }
18153
18154 #[test]
18155 fn a_file_of_three_tables_will_not_open_as_one() {
18156 let file = path("three-tables-unnamed");
18157 three_tables(&file);
18158 let error = Reader::open(&file).expect_err("more than one table");
18159 assert!(error.message().contains("more than one table"), "{}", error.message());
18160 fs::remove_file(file).expect("remove scratch file");
18161 }
18162
18163 #[test]
18165 fn decimals_of_every_storage_width_round_trip() {
18166 let file = path("decimals");
18167 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
18168 let fields = widths
18169 .iter()
18170 .enumerate()
18171 .map(|(index, (width, scale))| {
18172 Field::new(
18173 format!("d{index}"),
18174 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18175 )
18176 })
18177 .collect::<Vec<_>>();
18178 let mut writer = Writer::create(&file, "money", fields).expect("new file");
18179 let rows: [i128; 3] = [-1234, 0, 999];
18180 let columns = widths
18181 .iter()
18182 .map(|(width, scale)| {
18183 let values = rows
18184 .iter()
18185 .map(|unscaled| Value::Decimal {
18186 unscaled: *unscaled,
18187 width: *width,
18188 scale: *scale,
18189 })
18190 .collect::<Vec<_>>();
18191 Vector::from_values(
18192 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18193 &values,
18194 )
18195 .expect("a decimal column")
18196 })
18197 .collect::<Vec<_>>();
18198 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
18199 writer.finish().expect("commit");
18200
18201 let reader = Reader::open(&file).expect("a committed file");
18202 for (index, (width, scale)) in widths.iter().enumerate() {
18203 assert_eq!(
18204 reader.table().fields()[index].ty,
18205 LogicalType::decimal(*width, *scale).expect("a decimal type"),
18206 "column {index} came back as another type"
18207 );
18208 let column = reader.read(0, &[index]).expect("the column");
18209 for (row, unscaled) in rows.iter().enumerate() {
18210 assert_eq!(
18211 column.value_at(row, 0),
18212 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
18213 "column {index} row {row}"
18214 );
18215 }
18216 }
18217 fs::remove_file(file).expect("remove scratch file");
18218 }
18219
18220 #[test]
18221 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
18222 let file = path("two-of-a-name");
18223 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
18224 .expect("new file");
18225 let error = writer
18226 .next("t", vec![Field::new("a", LogicalType::BigInt)])
18227 .expect_err("the same name twice");
18228 assert!(error.message().contains("same name"), "{}", error.message());
18229 fs::remove_file(file).expect("remove scratch file");
18230 }
18231
18232 #[test]
18233 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
18234 let file = path("integer-tally");
18235 let mut writer =
18236 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
18237 .expect("new file");
18238 let mut values = vec![Value::SmallInt(0); 1024];
18239 values[7] = Value::SmallInt(3);
18240 values[99] = Value::SmallInt(-2);
18241 values[1001] = Value::SmallInt(3);
18242 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
18243 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
18244 values[0] = Value::Null;
18245 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
18246 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
18247 writer.finish().expect("commit");
18248
18249 let reader = Reader::open(&file).expect("read file");
18250 assert_eq!(
18251 reader.integer_tally(0, 0).expect("valid part"),
18252 Some(vec![(-2, 1), (0, 1021), (3, 2)])
18253 );
18254 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
18255 let catalog = Catalog::open(&file).expect("catalog");
18256 assert_eq!(
18257 catalog.integer_tally("events", 0).expect("nullable column"),
18258 Some(vec![(-2, 2), (0, 2041), (3, 4)])
18259 );
18260 fs::remove_file(file).expect("remove scratch file");
18261 }
18262
18263 #[test]
18264 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
18265 let file = path("catalog-integer-tally");
18266 let mut writer = Writer::create(
18267 &file,
18268 "events",
18269 vec![
18270 Field::new("noise", LogicalType::SmallInt),
18271 Field::new("source", LogicalType::SmallInt),
18272 ],
18273 )
18274 .expect("new file");
18275 let noise = vec![Value::SmallInt(9); 1024];
18276 let mut source = vec![Value::SmallInt(0); 1024];
18277 source[7] = Value::SmallInt(3);
18278 source[99] = Value::SmallInt(-2);
18279 let chunk = Chunk::new(vec![
18280 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
18281 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
18282 ])
18283 .expect("two columns");
18284 writer.append(&chunk).expect("append");
18285 writer.finish().expect("commit");
18286
18287 let catalog = Catalog::open(&file).expect("catalog");
18288 assert_eq!(
18289 catalog.integer_tally("events", 1).expect("selected column"),
18290 Some(vec![(-2, 1), (0, 1022), (3, 1)])
18291 );
18292 assert_eq!(
18293 catalog.integer_tally("events", 0).expect("other column"),
18294 Some(vec![(9, 1024)])
18295 );
18296 fs::remove_file(file).expect("remove scratch file");
18297 }
18298
18299 #[test]
18300 fn opening_the_catalog_reads_no_table_directory() {
18301 let file = path("catalog-only");
18302 three_tables(&file);
18303 let catalog = Catalog::open(&file).expect("a committed catalog");
18304 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
18307 assert_eq!(catalog.names().len(), 3);
18308 fs::remove_file(file).expect("remove scratch file");
18309 }
18310
18311 #[test]
18322 fn the_checksum_answers_what_it_has_always_answered() {
18323 let bytes: Vec<u8> =
18324 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
18325 for (length, expected) in [
18326 (0, 0xef46_db37_51d8_e999),
18327 (1, 0xa96c_7f0c_e858_bbb7),
18328 (3, 0x56e6_9576_32a4_87f9),
18329 (4, 0xc60d_15b1_e3ff_8f04),
18330 (5, 0x8088_1585_8624_dd4e),
18331 (7, 0xafbe_fc3d_6c6f_9a8e),
18332 (8, 0x3da5_c7aa_2696_83e0),
18333 (9, 0x465e_c429_b13c_3892),
18334 (15, 0xdee8_9d8a_065a_6233),
18335 (16, 0x1330_489a_7767_9c80),
18336 (31, 0x3391_303d_485e_846e),
18337 (32, 0x40b7_aff7_5d45_bbc8),
18338 (33, 0x4997_cae4_951c_17a5),
18339 (39, 0x5807_28fd_5c14_5739),
18340 (40, 0xf95c_f6f5_c08a_3d3b),
18341 (63, 0x2944_b4da_fc69_b206),
18342 (64, 0xbb76_f6ef_19bd_5a1b),
18343 (65, 0x814e_0c65_4a9f_d640),
18344 (127, 0x00de_aab1_31cf_f89b),
18345 (1000, 0x9e33_00c1_cde3_c58d),
18346 ] {
18347 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
18348 }
18349 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
18350 }
18351 #[test]
18358 fn a_declared_order_comes_back_out_of_the_file() {
18359 let path = path("clustered");
18360 let shipped = vec![
18361 Field::new("key", LogicalType::BigInt),
18362 Field::new("line", LogicalType::Integer),
18363 Field::new("shipdate", LogicalType::Date),
18364 ];
18365 let plain = vec![Field::new("a", LogicalType::Integer)];
18366 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
18367
18368 let mut writer = Writer::create(&path, "lineitem", shipped)
18369 .expect("new file")
18370 .declare(stage_zero.clone())
18371 .expect("the columns are the table's");
18372 let column = |ty: LogicalType, values: &[Value]| {
18373 Vector::from_values(ty, values).expect("the values match the type")
18374 };
18375 writer
18376 .append(
18377 &Chunk::new(vec![
18378 column(
18379 LogicalType::BigInt,
18380 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
18381 ),
18382 column(
18383 LogicalType::Integer,
18384 &[
18385 Value::Integer(1),
18386 Value::Integer(1),
18387 Value::Integer(1),
18388 Value::Integer(1),
18389 ],
18390 ),
18391 column(
18392 LogicalType::Date,
18393 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
18394 ),
18395 ])
18396 .expect("three columns"),
18397 )
18398 .expect("four rows");
18399 let mut writer = writer.next("nation", plain).expect("a second table");
18400 writer
18401 .append(
18402 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
18403 .expect("one column"),
18404 )
18405 .expect("one row");
18406 writer.finish().expect("commit");
18407
18408 let catalog = Catalog::open(&path).expect("reopen");
18409 let lineitem = catalog.table("lineitem").expect("the clustered table");
18410 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
18411 let nation = catalog.table("nation").expect("the plain table");
18412 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18413
18414 assert_eq!(lineitem.table().rows(), 4);
18417 assert_eq!(nation.table().rows(), 1);
18418 fs::remove_file(&path).ok();
18419 }
18420
18421 #[test]
18423 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18424 let path = path("clustered-bad");
18425 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18426 .expect("new file");
18427 let four =
18428 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18429 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18430 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18431 fs::remove_file(&path).ok();
18432 }
18433
18434 #[test]
18440 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18441 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18442 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18443 .collect::<Vec<_>>();
18444 let filled = || {
18445 let mut dictionary = GlobalDictionary::new();
18446 for value in &values {
18447 dictionary.code(value).expect("a code for every value");
18448 }
18449 dictionary.settle().expect("a shape");
18450 dictionary
18451 };
18452 let mut in_place = filled();
18453 in_place.finish_blocks().expect("every block encodes");
18454
18455 let mut handed = filled();
18456 let out = handed.hand_out(3);
18457 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18458 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18459 for job in out.iter().rev() {
18460 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18461 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18462 }
18463 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18464 handed.finish_blocks().expect("the last block encodes");
18465
18466 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18467 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18468 }
18469
18470 #[test]
18472 fn a_block_given_back_twice_is_refused() {
18473 let mut dictionary = GlobalDictionary::new();
18474 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18475 dictionary.code(&format!("value {at}")).expect("a code");
18476 }
18477 dictionary.settle().expect("a shape");
18478 let out = dictionary.hand_out(0);
18479 let last = out.last().expect("blocks went out");
18480 let at = last.place().1;
18481 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18482 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18483 }
18484
18485 #[test]
18491 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18492 let mut values = vec![String::new(), "http://".to_owned()];
18493 for host in 0..7 {
18494 for path in 0..30 {
18495 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18496 values.push(format!("http://example{host}.test/page/{path:04}"));
18497 }
18498 }
18499 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18500
18501 let mut dictionary = GlobalDictionary::new();
18502 for value in &values {
18503 dictionary.code(value).expect("a code for every value");
18504 }
18505 dictionary.finish_blocks().expect("the last block encodes");
18506 let ranked = dictionary.ranked(None).expect("a sorted order");
18507 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18508
18509 let spellings = dictionary_values(&dictionary);
18510 let seen = ranked
18511 .iter()
18512 .map(|&(_, code)| {
18513 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18514 })
18515 .collect::<Vec<_>>();
18516 let mut wanted = values.clone();
18517 wanted.sort_unstable();
18518 assert_eq!(seen, wanted, "the order is the order the bytes give");
18519
18520 for &(carried, code) in &ranked {
18521 let value = &spellings[code as usize];
18522 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18523 }
18524 }
18525
18526 #[test]
18531 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18532 let entry =
18533 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18534 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18535 .map(|code| entry(code, u64::from(code % 7) + 1))
18536 .collect::<Vec<_>>();
18537 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18538
18539 let mut sorted = all.clone();
18540 sorted.sort_unstable_by(|left, right| {
18541 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18542 });
18543 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18544 sorted.truncate(FREQUENCY_ENTRIES);
18545
18546 let mut picked = all.clone();
18547 let omitted = keep_most_frequent(&mut picked);
18548 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18549 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18550 assert!(
18551 picked
18552 .iter()
18553 .zip(&sorted)
18554 .all(|(one, two)| one.value == two.value && one.count == two.count),
18555 "the same entries in the same order"
18556 );
18557
18558 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18559 let omitted = keep_most_frequent(&mut short);
18560 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18561 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18562 }
18563
18564 #[test]
18566 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18567 let empty = GlobalDictionary::new();
18568 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18569
18570 let mut dictionary = GlobalDictionary::new();
18571 for value in ["pear", "apple", "", "apples", "app"] {
18572 dictionary.code(value).expect("a code for every value");
18573 }
18574 dictionary.finish_blocks().expect("the one block encodes");
18575 let spellings = dictionary_values(&dictionary);
18576 let seen = dictionary
18577 .ranked(None)
18578 .expect("a sorted order")
18579 .iter()
18580 .map(|&(_, code)| spellings[code as usize].clone())
18581 .collect::<Vec<_>>();
18582 let wanted: Vec<Vec<u8>> =
18583 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18584 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18585 }
18586
18587 #[test]
18590 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18591 let profile = LoadProfile::begin("demoted");
18592 let mut dictionary = GlobalDictionary::new();
18593 for value in 0..50_000 {
18594 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18595 }
18596 let (_, grown) = dictionary.recharge(Some(&profile));
18597 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18598
18599 dictionary.demote();
18600 let (before, after) = dictionary.recharge(Some(&profile));
18601 assert_eq!(before, grown);
18602 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18605 assert_eq!(profile.held(), after, "the profile was told about the drop");
18606 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18607
18608 dictionary.demote();
18609 assert_eq!(
18610 dictionary.recharge(Some(&profile)),
18611 (after, after),
18612 "demoting twice is a no-op"
18613 );
18614 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18615 }
18616}