1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicU64, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod distinct;
58pub mod grams;
59pub mod graph;
60pub mod host;
61mod prepare;
62mod projection;
63mod run_projection;
64use prepare::Lent;
65pub mod section;
66pub mod stats;
67mod zones;
68
69pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
70pub use projection::build_sorted_projection;
71pub use run_projection::{RunProjectionPart, RunProjectionScan, build_run_projection};
72pub use section::Section;
73pub use zones::{Common, Stripes, ascending, distincts, widths};
74
75const MAGIC: &[u8; 8] = b"RUDBNV10";
76const DIRECTORY: &[u8; 8] = b"RUDBDI10";
77const CATALOG: &[u8; 8] = b"RUDBCA10";
78const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
79const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
80const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
81const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
82const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
83const DEVICE_CARD: &[u8; 8] = b"RUDBDV10";
84const MAX_CATALOG_FREQUENCIES: usize = 64;
85const FORMAT: u32 = 30;
86
87const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, 29, FORMAT];
122
123const HEADER: u64 = 80;
124const SLOT_BYTES: usize = 28;
125const MAX_PAGE: usize = 256 * 1024 * 1024;
126const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
127const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
128const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
129const FREQUENCIES_SPANS: &[u8; 8] = b"RUDBFQ4\0";
130const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
138const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
140const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
146const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
161const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
181const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
189const KEYS: &[u8; 8] = b"RUDBKY1\0";
196const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
204
205const MAX_SECTIONS: usize = 4096;
212const FREQUENCY_CANDIDATES: usize = 32_768;
213const FREQUENCY_ENTRIES: usize = 512;
214const FREQUENCY_BUILD_RANK: usize = 10;
215const FREQUENCY_ORDINALS: usize = 131_072;
216const MAX_PAIR_FREQUENCIES: usize = 1024;
217const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
222const MAX_FREQUENCY_WORKERS: usize = 32;
229
230fn close_workers() -> usize {
232 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
233}
234
235const CLOSE_BYTES: usize = 1 << 30;
246
247const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
250
251const MAX_ENCODE_WORKERS: usize = 32;
258
259const WRITEBACK_STRETCH: u64 = 32 << 20;
267
268const SIEVE_BUDGET: usize = 8 * 1024;
276
277const PART_BOUND_BYTES: usize = 24;
286
287fn io(error: std::io::Error) -> Error {
288 Error::io(error.to_string())
289}
290
291fn invalid(message: &str) -> Error {
292 Error::invalid_input(format!("invalid rudb native file: {message}"))
293}
294
295fn sum(counts: impl Iterator<Item = u64>) -> u64 {
297 counts.fold(0, u64::saturating_add)
298}
299
300fn span_bytes(spans: &[Span], at: usize) -> u64 {
302 spans.get(at).map_or(0, |span| u64::from(span.length))
303}
304
305fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
307 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
308}
309
310fn dictionary_bytes(table: &Table, at: usize) -> u64 {
312 page_bytes(&table.dictionaries, at)
313 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
314}
315
316fn checksum(bytes: &[u8]) -> u64 {
326 seeded_checksum(bytes, 0)
327}
328
329#[must_use]
336pub fn content_name(bytes: &[u8]) -> u128 {
337 let seed = u64::from(FORMAT);
338 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
339}
340
341#[derive(Debug, Clone)]
347pub struct ContentNamer {
348 seeds: [u64; 2],
349 lanes: [[u64; 4]; 2],
350 held: [u8; 32],
351 filled: usize,
352 length: u64,
353}
354
355impl Default for ContentNamer {
356 fn default() -> Self {
357 let seed = u64::from(FORMAT);
358 let seeds = [seed, !seed];
359 let lanes = seeds.map(|seed| {
360 [
361 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
362 seed.wrapping_add(XXH_P2),
363 seed,
364 seed.wrapping_sub(XXH_P1),
365 ]
366 });
367 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
368 }
369}
370
371impl ContentNamer {
372 pub fn update(&mut self, mut bytes: &[u8]) {
374 self.length += bytes.len() as u64;
375 if self.filled > 0 {
376 let take = (32 - self.filled).min(bytes.len());
377 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
378 self.filled += take;
379 bytes = &bytes[take..];
380 if self.filled < 32 {
381 return;
382 }
383 let block = self.held;
384 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
385 self.filled = 0;
386 }
387 let mut blocks = bytes.chunks_exact(32);
388 for block in blocks.by_ref() {
389 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
390 }
391 let rest = blocks.remainder();
392 self.held[..rest.len()].copy_from_slice(rest);
393 self.filled = rest.len();
394 }
395
396 #[must_use]
398 pub fn finish(&self) -> u128 {
399 let rest = &self.held[..self.filled];
400 let [first, second] = [0, 1].map(|at| {
401 if self.length < 32 {
402 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
403 } else {
404 finish_checksum(self.lanes[at], rest, self.length)
405 }
406 });
407 u128::from(first) << 64 | u128::from(second)
408 }
409}
410
411fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
420 let mut blocks = bytes.chunks_exact(32);
423 let rest = blocks.remainder();
424 if bytes.len() < 32 {
425 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
426 }
427 let mut lanes = [
428 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
429 seed.wrapping_add(XXH_P2),
430 seed,
431 seed.wrapping_sub(XXH_P1),
432 ];
433 for block in blocks.by_ref() {
434 checksum_block(&mut lanes, block);
435 }
436 finish_checksum(lanes, rest, bytes.len() as u64)
437}
438
439const XXH_P1: u64 = 11_400_714_785_074_694_791;
440const XXH_P2: u64 = 14_029_467_366_897_019_727;
441const XXH_P3: u64 = 1_609_587_929_392_839_161;
442const XXH_P4: u64 = 9_650_029_242_287_828_579;
443const XXH_P5: u64 = 2_870_177_450_012_600_261;
444
445fn checksum_round(state: u64, word: u64) -> u64 {
446 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
447}
448
449fn checksum_word(chunk: &[u8]) -> u64 {
450 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
451}
452
453fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
455 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
456 *lane = checksum_round(*lane, checksum_word(chunk));
457 }
458}
459
460fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
462 let merge = |state: u64, lane: u64| {
463 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
464 };
465 let [one, two, three, four] = lanes;
466 let combined = one
467 .rotate_left(1)
468 .wrapping_add(two.rotate_left(7))
469 .wrapping_add(three.rotate_left(12))
470 .wrapping_add(four.rotate_left(18));
471 let hash = merge(merge(merge(merge(combined, one), two), three), four);
472 checksum_tail(hash.wrapping_add(length), rest)
473}
474
475fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
477 let mut words = rest.chunks_exact(8);
478 for chunk in words.by_ref() {
479 hash ^= checksum_round(0, checksum_word(chunk));
480 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
481 }
482 rest = words.remainder();
483 if rest.len() >= 4 {
484 let (head, tail) = rest.split_at(4);
485 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
486 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
487 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
488 rest = tail;
489 }
490 for &byte in rest {
491 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
492 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
493 }
494 hash ^= hash >> 33;
495 hash = hash.wrapping_mul(XXH_P2);
496 hash ^= hash >> 29;
497 hash = hash.wrapping_mul(XXH_P3);
498 hash ^ (hash >> 32)
499}
500
501fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
507 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
508}
509
510fn walk_checksummed(
516 file: &File,
517 offset: u64,
518 length: usize,
519 window: usize,
520 mut each: impl FnMut(&[u8]) -> Result<()>,
521) -> Result<u64> {
522 debug_assert!(window.is_multiple_of(32) && window > 0, "a window is whole blocks of the hash");
523 if length < 32 {
524 let mut bytes = vec![0; length];
525 read_at(file, offset, &mut bytes)?;
526 each(&bytes)?;
527 return Ok(checksum(&bytes));
528 }
529 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
530 let mut buffer = vec![0; window.min(length)];
531 let mut read = 0;
532 let (mut whole, mut filled) = (0, 0);
533 while read < length {
534 filled = buffer.len().min(length - read);
535 read_at(file, offset + read as u64, &mut buffer[..filled])?;
536 read += filled;
537 each(&buffer[..filled])?;
538 whole = filled / 32 * 32;
539 for block in buffer[..whole].chunks_exact(32) {
540 checksum_block(&mut lanes, block);
541 }
542 }
543 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
544}
545
546#[derive(Debug, Clone, Copy)]
547struct Slot {
548 offset: u64,
549 length: u32,
550 generation: u64,
551 hash: u64,
552}
553
554impl Slot {
555 fn bytes(self) -> [u8; SLOT_BYTES] {
556 let mut result = [0; SLOT_BYTES];
557 result[..8].copy_from_slice(&self.offset.to_le_bytes());
558 result[8..12].copy_from_slice(&self.length.to_le_bytes());
559 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
560 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
561 result
562 }
563
564 fn read(bytes: &[u8]) -> Self {
565 Self {
566 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
567 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
568 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
569 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
570 }
571 }
572}
573
574#[derive(Debug, Clone, Copy)]
575struct Page {
576 offset: u64,
577 length: u32,
578 hash: u64,
579}
580
581impl Page {
582 fn bytes(&self) -> u64 {
584 u64::from(self.length)
585 }
586}
587
588#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
589enum FrequencyValue {
590 Null,
591 Integer(i128),
592 Code(u32),
593}
594
595type FrequencyMap<V> = HashMap<u64, V, Spread>;
601
602#[derive(Debug)]
616struct Candidates {
617 slots: Vec<Candidate>,
620 held: usize,
621 nulls: u32,
622 decrements: u64,
623 survivors: Vec<Candidate>,
625}
626
627#[derive(Debug, Default, Clone, Copy)]
629struct Candidate {
630 bits: u64,
631 count: u32,
632}
633
634const FIRST_CANDIDATE_SLOTS: usize = 64;
636
637impl Default for Candidates {
638 fn default() -> Self {
639 Self {
640 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
641 held: 0,
642 nulls: 0,
643 decrements: 0,
644 survivors: Vec::new(),
645 }
646 }
647}
648
649impl Candidates {
650 fn add(&mut self, bits: Option<u64>, mut times: u32) {
657 while times > 0 {
658 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
659 match bits {
660 Some(bits) => {
661 let (at, found) = self.find(bits);
662 if found {
663 self.slots[at].count = self.slots[at].count.saturating_add(times);
664 return;
665 }
666 if room {
667 self.place(at, bits, times);
668 return;
669 }
670 }
671 None if self.nulls != 0 => {
672 self.nulls = self.nulls.saturating_add(times);
673 return;
674 }
675 None if room => {
676 self.nulls = times;
677 return;
678 }
679 None => {}
680 }
681 self.decrement();
682 times -= 1;
683 }
684 }
685
686 fn find(&self, bits: u64) -> (usize, bool) {
688 let mask = self.slots.len() - 1;
689 let mut at = home(bits, self.slots.len());
690 loop {
691 let slot = self.slots[at];
692 if slot.count == 0 {
693 return (at, false);
694 }
695 if slot.bits == bits {
696 return (at, true);
697 }
698 at = (at + 1) & mask;
699 }
700 }
701
702 fn position(&self, bits: u64) -> Option<usize> {
704 match self.find(bits) {
705 (at, true) => Some(at),
706 (_, false) => None,
707 }
708 }
709
710 fn place(&mut self, at: usize, bits: u64, count: u32) {
713 let at = if (self.held + 1) * 2 > self.slots.len() {
714 let wider = self.slots.len() * 2;
715 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
716 for slot in old.into_iter().filter(|slot| slot.count != 0) {
717 let (to, _) = self.find(slot.bits);
718 self.slots[to] = slot;
719 }
720 self.find(bits).0
721 } else {
722 at
723 };
724 self.slots[at] = Candidate { bits, count };
725 self.held += 1;
726 }
727
728 fn decrement(&mut self) {
730 let mut survivors = std::mem::take(&mut self.survivors);
731 survivors.clear();
732 survivors.extend(
733 self.slots
734 .iter()
735 .filter(|slot| slot.count > 1)
736 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
737 );
738 self.slots.fill(Candidate::default());
739 self.held = survivors.len();
740 for &slot in &survivors {
741 let (at, _) = self.find(slot.bits);
742 self.slots[at] = slot;
743 }
744 self.survivors = survivors;
745 self.nulls = self.nulls.saturating_sub(1);
746 self.decrements = self.decrements.saturating_add(1);
747 }
748
749 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
751 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
752 }
753}
754
755fn home(bits: u64, slots: usize) -> usize {
760 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
761}
762
763#[derive(Debug, Default)]
765struct Run {
766 bits: Option<u64>,
767 times: u32,
768}
769
770impl Run {
771 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
773 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
774 self.times += 1;
775 return None;
776 }
777 let ended = self.take();
778 self.bits = bits;
779 self.times = 1;
780 ended
781 }
782
783 fn take(&mut self) -> Option<(Option<u64>, u32)> {
785 let times = std::mem::take(&mut self.times);
786 (times != 0).then_some((self.bits, times))
787 }
788}
789
790#[derive(Debug, Default, Clone, Copy)]
792struct Spread;
793
794impl std::hash::BuildHasher for Spread {
795 type Hasher = SpreadHasher;
796
797 fn build_hasher(&self) -> SpreadHasher {
798 SpreadHasher(0)
799 }
800}
801
802#[derive(Debug)]
809struct SpreadHasher(u64);
810
811impl SpreadHasher {
812 fn mix(&mut self, word: u64) {
813 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
814 self.0 = (product as u64) ^ ((product >> 64) as u64);
815 }
816}
817
818impl std::hash::Hasher for SpreadHasher {
819 fn write(&mut self, bytes: &[u8]) {
820 for part in bytes.chunks(8) {
821 let mut word = [0; 8];
822 word[..part.len()].copy_from_slice(part);
823 self.mix(u64::from_le_bytes(word));
824 }
825 }
826
827 fn write_u32(&mut self, value: u32) {
828 self.mix(u64::from(value));
829 }
830
831 fn write_u64(&mut self, value: u64) {
832 self.mix(value);
833 }
834
835 fn write_i128(&mut self, value: i128) {
836 self.mix(value as u64);
837 self.mix((value >> 64) as u64);
838 }
839
840 fn write_isize(&mut self, value: isize) {
841 self.mix(value as u64);
842 }
843
844 fn finish(&self) -> u64 {
845 self.0
846 }
847}
848
849#[derive(Debug, Clone)]
850struct FrequencyEntry {
851 value: FrequencyValue,
852 count: u64,
853}
854
855#[derive(Debug, Clone)]
860struct FrequencySummary {
861 entries: Vec<FrequencyEntry>,
862 omitted_max: u64,
863 ordinals: Vec<u64>,
864 ordinal_entries: Vec<u16>,
865}
866
867#[derive(Debug, Clone)]
868struct PairFrequencyEntry {
869 first_entry: u16,
870 second: Option<u32>,
871 count: u64,
872}
873
874#[derive(Debug, Clone)]
880struct PairFrequencySummary {
881 first: u16,
882 second: u16,
883 entries: Vec<PairFrequencyEntry>,
884 omitted_max: u64,
885}
886
887#[derive(Debug, Clone)]
895enum Frequencies {
896 Held(FrequencySummary),
897 Stored {
900 span: Span,
901 values: bool,
902 entries: usize,
903 },
904}
905
906#[derive(Debug, Clone)]
911pub struct FrequencyPrefix {
912 pub entries: Vec<(Value, u64)>,
914 pub omitted_max: u64,
916}
917
918#[derive(Debug, Clone, PartialEq)]
920pub struct FrequencyOccurrences {
921 pub omitted_max: u64,
923 pub ordinals: Vec<u64>,
925 pub anchors: Vec<Value>,
927 pub anchor_indices: Vec<u16>,
929}
930
931pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
933
934#[derive(Debug, Clone, Copy, Default)]
941struct Span {
942 offset: u64,
943 length: u32,
944}
945
946#[derive(Debug, Clone, Default)]
954struct Pages {
955 columns: usize,
956 held: Box<[StripePage]>,
957}
958
959#[derive(Debug, Clone, Copy)]
961struct StripePage {
962 offset: u64,
963 hash: u64,
964 length: u32,
965 column: u32,
966}
967
968impl Pages {
969 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
971 let mut held = Vec::with_capacity(slots.iter().flatten().count());
972 for (column, page) in slots.iter().enumerate() {
973 if let Some(page) = page {
974 let column =
975 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
976 held.push(StripePage {
977 offset: page.offset,
978 hash: page.hash,
979 length: page.length,
980 column,
981 });
982 }
983 }
984 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
985 }
986
987 fn get(&self, column: usize) -> Option<Page> {
989 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
990 let placed = self.held[at];
991 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
992 }
993
994 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
996 (0..self.columns).map(|column| self.get(column))
997 }
998
999 fn bytes(&self, column: usize) -> u64 {
1001 self.get(column).map_or(0, |page| page.bytes())
1002 }
1003}
1004
1005#[derive(Debug, Clone)]
1007pub struct Stripe {
1008 rows: usize,
1009 parts: Vec<u32>,
1012 index: Span,
1016 pages: Vec<Span>,
1017 memberships: Pages,
1018 sieves: Pages,
1021 part_ranges: Pages,
1032 zone: Zone,
1033}
1034
1035impl Stripe {
1036 #[must_use]
1038 pub fn rows(&self) -> usize {
1039 self.rows
1040 }
1041
1042 #[must_use]
1044 pub fn parts(&self) -> usize {
1045 self.parts.len()
1046 }
1047
1048 #[must_use]
1054 pub fn zone(&self) -> &Zone {
1055 &self.zone
1056 }
1057}
1058
1059#[derive(Debug, Clone)]
1061pub struct Table {
1062 name: String,
1063 fields: Vec<Field>,
1064 stripes: Vec<Stripe>,
1065 rows: usize,
1066 dictionaries: Vec<Option<Page>>,
1067 dictionary_payloads: Vec<u64>,
1073 demoted: Vec<bool>,
1079 frequencies: Vec<Option<Frequencies>>,
1080 pair_frequencies: Vec<PairFrequencySummary>,
1081 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1086 host_groups: Option<host::HostSummary>,
1088 distincts: Vec<Option<u64>>,
1098 clustering: Option<Clustering>,
1106 generation: u64,
1120 sections: Vec<Section>,
1127 constraints: Constraints,
1130}
1131
1132#[derive(Debug, Clone, Default, PartialEq, Eq)]
1137pub struct Constraints {
1138 pub keys: Vec<(Vec<u16>, bool)>,
1140 pub foreign: Vec<StoredForeign>,
1142}
1143
1144impl Constraints {
1145 #[must_use]
1147 pub fn is_empty(&self) -> bool {
1148 self.keys.is_empty() && self.foreign.is_empty()
1149 }
1150}
1151
1152#[derive(Debug, Clone, PartialEq, Eq)]
1154pub struct StoredForeign {
1155 pub columns: Vec<u16>,
1157 pub table: String,
1159 pub referenced: Vec<u16>,
1161}
1162
1163impl Table {
1164 #[must_use]
1166 pub fn name(&self) -> &str {
1167 &self.name
1168 }
1169
1170 #[must_use]
1172 pub fn fields(&self) -> &[Field] {
1173 &self.fields
1174 }
1175
1176 #[must_use]
1178 pub fn rows(&self) -> usize {
1179 self.rows
1180 }
1181
1182 #[must_use]
1184 pub fn stripes(&self) -> &[Stripe] {
1185 &self.stripes
1186 }
1187
1188 #[must_use]
1190 pub fn clustering(&self) -> Option<&Clustering> {
1191 self.clustering.as_ref()
1192 }
1193
1194 #[must_use]
1196 pub fn constraints(&self) -> &Constraints {
1197 &self.constraints
1198 }
1199
1200 #[must_use]
1205 pub fn generation(&self) -> u64 {
1206 self.generation
1207 }
1208
1209 #[must_use]
1216 pub fn sections(&self) -> &[Section] {
1217 &self.sections
1218 }
1219}
1220
1221#[derive(Debug, Clone)]
1233struct Entry {
1234 name: String,
1235 fields: Vec<Field>,
1236 rows: usize,
1237 directory: Page,
1239 nonzero: Vec<Option<u64>>,
1242 aggregates: Vec<Option<(i128, u64)>>,
1244 distincts: Vec<Option<u64>>,
1246 extremes: Vec<StoredIntegerExtremes>,
1248 frequencies: Vec<StoredNumericFrequencies>,
1250}
1251
1252type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1253type StoredNumericFrequencies = Option<NumericFrequencies>;
1254
1255#[derive(Debug, Clone, PartialEq, Eq)]
1268pub struct ViewEntry {
1269 pub name: String,
1271 pub sql: String,
1273 pub statement: String,
1275 pub aliases: Vec<String>,
1277 pub columns: Vec<Field>,
1279}
1280
1281#[derive(Debug, Clone)]
1283pub struct ColumnLayout {
1284 pub name: String,
1286 pub kind: String,
1288 pub pages: u64,
1290 pub memberships: u64,
1292 pub sieves: u64,
1294 pub part_ranges: u64,
1296 pub dictionary: u64,
1298}
1299
1300impl ColumnLayout {
1301 #[must_use]
1303 pub fn total(&self) -> u64 {
1304 self.pages
1305 .saturating_add(self.memberships)
1306 .saturating_add(self.sieves)
1307 .saturating_add(self.part_ranges)
1308 .saturating_add(self.dictionary)
1309 }
1310}
1311
1312#[derive(Debug, Clone)]
1323pub struct Layout {
1324 pub file: u64,
1326 pub rows: usize,
1328 pub stripes: usize,
1330 pub parts: usize,
1332 pub columns: Vec<ColumnLayout>,
1334 pub indexes: u64,
1337 pub directory: u64,
1339 pub header: u64,
1341}
1342
1343impl Layout {
1344 #[must_use]
1346 pub fn columns_total(&self) -> u64 {
1347 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1348 }
1349
1350 #[must_use]
1356 pub fn unaccounted(&self) -> u64 {
1357 self.file
1358 .saturating_sub(self.columns_total())
1359 .saturating_sub(self.indexes)
1360 .saturating_sub(self.directory)
1361 .saturating_sub(self.header)
1362 }
1363}
1364
1365#[derive(Debug, Clone)]
1376pub struct StoredPart {
1377 pub stripe: usize,
1379 pub part: usize,
1381 pub row: usize,
1383 pub rows: usize,
1385 pub encoding: String,
1387 pub bytes: u64,
1389 pub page: u64,
1391 pub offset: u64,
1393 pub low: Option<Value>,
1395 pub high: Option<Value>,
1397 pub nulls: Option<usize>,
1399}
1400
1401const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1408
1409#[derive(Debug)]
1434struct GlobalDictionary {
1435 primary: HashMap<u64, u32, Spread>,
1439 collisions: HashMap<u64, Vec<u32>, Spread>,
1440 checks: Vec<u64>,
1442 ends: Vec<u32>,
1444 counts: Vec<u64>,
1445 nulls: u64,
1446 filling: Vec<u8>,
1448 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1454 waiting: Vec<(usize, Vec<u8>)>,
1459 sample: Vec<(usize, Vec<u8>)>,
1465 stride: usize,
1467 shape: Option<chooser::Settled>,
1469 settled: usize,
1471 blocks: Vec<Vec<u8>>,
1476 early: BTreeMap<usize, EncodedBlock>,
1482 placed: Vec<Placed>,
1484 charged: u64,
1487 demoted: bool,
1489}
1490
1491#[derive(Debug, Clone, Copy)]
1493struct Placed {
1494 start: u64,
1495 length: u64,
1496 hash: u64,
1497}
1498
1499type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1501
1502impl GlobalDictionary {
1503 fn new() -> Self {
1504 Self {
1505 primary: HashMap::default(),
1506 collisions: HashMap::default(),
1507 checks: Vec::new(),
1508 ends: Vec::new(),
1509 counts: Vec::new(),
1510 nulls: 0,
1511 filling: Vec::new(),
1512 grams: Vec::new(),
1513 waiting: Vec::new(),
1514 sample: Vec::new(),
1515 stride: 1,
1516 shape: None,
1517 settled: 0,
1518 blocks: Vec::new(),
1519 early: BTreeMap::new(),
1520 placed: Vec::new(),
1521 charged: 0,
1522 demoted: false,
1523 }
1524 }
1525
1526 fn values(&self) -> usize {
1528 self.ends.len()
1529 }
1530
1531 fn closing_bytes(&self) -> usize {
1534 let values = self.values();
1535 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1536 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1537 .sum::<usize>();
1538 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1539 }
1540
1541 fn held_bytes(&self) -> u64 {
1547 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1548 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1549 }
1550 fn spilled<T>(values: &Vec<T>) -> usize {
1551 values.capacity() * size_of::<T>()
1552 }
1553 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1554 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1555 };
1556 let bytes = table(&self.primary)
1557 + table(&self.collisions)
1558 + self.collisions.values().map(spilled).sum::<usize>()
1559 + spilled(&self.checks)
1560 + spilled(&self.ends)
1561 + spilled(&self.counts)
1562 + self.filling.capacity()
1563 + spilled(&self.grams)
1564 + raw(&self.waiting)
1565 + raw(&self.sample)
1566 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1567 + spilled(&self.placed);
1568 bytes as u64
1569 }
1570
1571 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1574 let before = self.charged;
1575 let now = self.held_bytes();
1576 if let Some(profile) = profile {
1577 if now >= before {
1578 profile.hold(now - before);
1579 } else {
1580 profile.release(before - now);
1581 }
1582 }
1583 self.charged = now;
1584 (before, now)
1585 }
1586
1587 fn demote(&mut self) {
1595 if self.demoted {
1596 return;
1597 }
1598 self.seal_rest();
1599 self.release_lookup();
1600 self.demoted = true;
1601 }
1602
1603 fn release_lookup(&mut self) {
1610 self.primary = HashMap::default();
1611 self.collisions = HashMap::default();
1612 self.checks = Vec::new();
1613 self.sample = Vec::new();
1614 self.filling = Vec::new();
1615 }
1616
1617 fn encoded(&self) -> usize {
1619 self.placed.len() + self.blocks.len()
1620 }
1621
1622 #[cfg(test)]
1623 fn code(&mut self, text: &str) -> Result<u32> {
1624 let bytes = text.as_bytes();
1625 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1626 }
1627
1628 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1634 if let Some(&code) = self.primary.get(&hash) {
1635 if self.checks.get(code as usize) == Some(&check) {
1636 return Ok(code);
1637 }
1638 if let Some(codes) = self.collisions.get(&hash)
1639 && let Some(code) =
1640 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1641 {
1642 return Ok(code);
1643 }
1644 let code = self.insert(text, check)?;
1645 self.collisions.entry(hash).or_default().push(code);
1646 return Ok(code);
1647 }
1648 let code = self.insert(text, check)?;
1649 self.primary.insert(hash, code);
1650 Ok(code)
1651 }
1652
1653 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1654 if self.demoted {
1655 return Err(Error::internal("a value was coded against a demoted dictionary"));
1656 }
1657 let code = u32::try_from(self.ends.len())
1658 .map_err(|_| invalid("global dictionary has too many values"))?;
1659 self.filling.extend_from_slice(text);
1660 self.ends.push(
1661 u32::try_from(self.filling.len())
1662 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1663 );
1664 self.checks.push(check);
1665 self.counts.push(0);
1666 if self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1667 self.seal();
1668 }
1669 Ok(code)
1670 }
1671
1672 fn seal(&mut self) {
1678 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1679 let bytes = std::mem::take(&mut self.filling);
1680 if at.is_multiple_of(self.stride) {
1681 self.sample.push((at, bytes.clone()));
1682 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1683 self.stride *= 2;
1684 let stride = self.stride;
1685 self.sample.retain(|(at, _)| at % stride == 0);
1686 }
1687 }
1688 self.waiting.push((at, bytes));
1689 }
1690
1691 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1693 block_values(self.block_ends(at), bytes)
1694 }
1695
1696 fn block_ends(&self, at: usize) -> &[u32] {
1698 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1699 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1700 &self.ends[first..last]
1701 }
1702
1703 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1710 let Some(shape) = &self.shape else { return Vec::new() };
1711 let waiting = std::mem::take(&mut self.waiting);
1712 waiting
1713 .into_iter()
1714 .map(|(at, bytes)| Unencoded {
1715 column,
1716 at,
1717 ends: self.block_ends(at).to_vec(),
1718 bytes,
1719 shape: shape.clone(),
1720 })
1721 .collect()
1722 }
1723
1724 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1727 if at < self.encoded() || self.early.insert(at, block).is_some() {
1728 return Err(Error::internal("a dictionary block came back twice"));
1729 }
1730 while let Some(block) = self.early.remove(&self.encoded()) {
1731 self.push_block(block);
1732 }
1733 Ok(())
1734 }
1735
1736 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1738 self.blocks.push(bytes);
1739 self.grams.push(*grams);
1740 }
1741
1742 fn settle(&mut self) -> Result<()> {
1750 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1751 return Ok(());
1752 }
1753 self.settle_on_sample()
1754 }
1755
1756 fn settle_rest(&mut self) -> Result<()> {
1764 if self.shape.is_some() || self.sample.is_empty() {
1765 return Ok(());
1766 }
1767 self.settle_on_sample()
1768 }
1769
1770 fn settle_on_sample(&mut self) -> Result<()> {
1771 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1772 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1773 return Ok(());
1774 }
1775 let sample =
1776 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1777 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1778 self.settled = complete;
1779 Ok(())
1780 }
1781
1782 fn seal_rest(&mut self) {
1784 if !self.demoted && !self.ends.len().is_multiple_of(TEXT_PAYLOAD_VALUES) {
1788 self.seal();
1789 }
1790 }
1791
1792 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1795 let (block, bytes) = &self.waiting[at];
1796 let values = self.slices(*block, bytes);
1797 let encoded = match &self.shape {
1798 Some(shape) => string::encode_with(&values, shape)?,
1799 None => string::encode(&values)?,
1800 };
1801 Ok((encoded, block_grams(&values)))
1802 }
1803
1804 #[cfg(test)]
1806 fn finish_blocks(&mut self) -> Result<()> {
1807 self.seal_rest();
1808 let made = (0..self.waiting.len())
1809 .map(|at| self.encode_waiting(at))
1810 .collect::<Result<Vec<_>>>()?;
1811 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1812 if self.encoded() != at {
1813 return Err(Error::internal("a dictionary block was encoded out of order"));
1814 }
1815 self.push_block(block);
1816 }
1817 Ok(())
1818 }
1819
1820 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1838 let count = self.placed.len() + self.blocks.len();
1839 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1840 return Err(invalid("global dictionary blocks do not cover its values"));
1841 }
1842 let mut bases = Vec::with_capacity(count);
1843 let mut total = 0_usize;
1844 for block in 0..count {
1845 bases.push(total as u64);
1846 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1847 total = total
1848 .checked_add(self.ends[last] as usize)
1849 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1850 }
1851 let mut flat = vec![0_u8; total];
1852 let mut outs = Vec::with_capacity(count);
1853 let mut rest = flat.as_mut_slice();
1854 for block in 0..count {
1855 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1856 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1857 outs.push((block, out));
1858 rest = after;
1859 }
1860 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1861 let mut stored = Vec::new();
1862 for (block, out) in run {
1863 let encoded = match self.placed.get(*block) {
1864 Some(place) => {
1865 let file = file.ok_or_else(|| {
1866 Error::internal("a written dictionary block has no file")
1867 })?;
1868 let length = usize::try_from(place.length).map_err(|_| {
1869 invalid("global dictionary block does not fit in memory")
1870 })?;
1871 stored.resize(length, 0);
1872 read_at(file, place.start, &mut stored)?;
1873 if checksum(&stored) != place.hash {
1874 return Err(invalid(
1875 "a global dictionary block did not read back as written",
1876 ));
1877 }
1878 stored.as_slice()
1879 }
1880 None => &self.blocks[*block - self.placed.len()],
1881 };
1882 let decoded = string::decode_flat(encoded)?;
1883 if decoded.bytes().len() != out.len() {
1884 return Err(invalid(
1885 "a global dictionary block is not the length its ends say",
1886 ));
1887 }
1888 out.copy_from_slice(decoded.bytes());
1889 }
1890 Ok(())
1891 };
1892 let workers = close_workers().min(count / 16).max(1);
1895 if workers <= 1 {
1896 one(&mut outs)?;
1897 } else {
1898 let per = count.div_ceil(workers);
1899 std::thread::scope(|scope| {
1900 outs.chunks_mut(per)
1901 .map(|run| scope.spawn(|| one(run)))
1902 .collect::<Vec<_>>()
1903 .into_iter()
1904 .try_for_each(|handle| {
1905 handle.join().map_err(|_| {
1906 Error::internal("a global dictionary decode worker panicked")
1907 })?
1908 })
1909 })?;
1910 }
1911 drop(outs);
1912 Ok((flat, bases))
1913 }
1914
1915 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1920 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1921 let Some(&end) = ends.get(code) else { return (0, 0) };
1922 let base = base as usize;
1923 let from =
1924 if code.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[code - 1] as usize };
1925 (base + from, base + end as usize)
1926 }
1927
1928 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1948 let (flat, bases) = self.decoded(file)?;
1949 let value = |code: u32| {
1950 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1951 flat.get(from..to).unwrap_or_default()
1952 };
1953 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1954 sort_by_value_across(&mut codes, value, close_workers());
1955 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1956 Ok((order, flat, bases))
1957 }
1958
1959 #[cfg(test)]
1960 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1961 self.ranked_with_values(file).map(|(order, _, _)| order)
1962 }
1963}
1964
1965#[derive(Debug)]
1973pub struct Writer {
1974 file: Box<dyn rudb_io::File>,
1977 at: u64,
1985 written_back: u64,
1987 table: Table,
1988 generation: u64,
1989 order: Vec<((u64, u64), (u64, u64))>,
1992 next_order: u64,
1993 dictionaries: Vec<Option<GlobalDictionary>>,
1994 coded: Arc<prepare::Coding>,
1997 gathers: Vec<Option<stats::Gather>>,
2003 lent: Option<Arc<Lent>>,
2006 pending: Vec<PendingChunk>,
2007 closed: Vec<Entry>,
2009 views: Vec<ViewEntry>,
2014 card: Option<KeptCard>,
2016 profile: Option<Arc<LoadProfile>>,
2022}
2023
2024#[derive(Debug)]
2032struct PendingChunk {
2033 order: (u64, u64),
2034 chunk: Chunk,
2035}
2036
2037#[derive(Debug, Clone, Copy)]
2043struct Part {
2044 order: (u64, u64),
2045 rows: usize,
2046 footprint: usize,
2047}
2048
2049impl Part {
2050 fn of(pending: &PendingChunk) -> Self {
2051 Self {
2052 order: pending.order,
2053 rows: pending.chunk.len(),
2054 footprint: pending.chunk.footprint(),
2055 }
2056 }
2057}
2058
2059#[derive(Debug, Default)]
2065struct ColumnStripe {
2066 pages: Vec<Vec<u8>>,
2067 sums: Vec<u64>,
2070 codes: Vec<Option<Vec<u32>>>,
2071 sieves: Vec<Option<Sieve>>,
2072 ranges: Vec<Range>,
2073}
2074
2075fn coded_type(ty: &LogicalType) -> bool {
2083 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2084}
2085
2086fn dictionary_tag(ty: &LogicalType) -> u8 {
2093 if ty == &LogicalType::Blob { 2 } else { 1 }
2094}
2095
2096fn weight(ty: &LogicalType) -> usize {
2104 match ty {
2105 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2106 LogicalType::HugeInt
2107 | LogicalType::UHugeInt
2108 | LogicalType::Uuid
2109 | LogicalType::Interval => 16,
2110 LogicalType::BigInt
2111 | LogicalType::UBigInt
2112 | LogicalType::Timestamp
2113 | LogicalType::Time
2114 | LogicalType::TimeTz
2115 | LogicalType::TimestampTz
2116 | LogicalType::TimestampS
2117 | LogicalType::TimestampMs
2118 | LogicalType::TimestampNs
2119 | LogicalType::Double
2120 | LogicalType::Decimal { .. } => 8,
2121 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2122 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2123 _ => 1,
2124 }
2125}
2126
2127pub const STRIPE_PARTS: usize = 64;
2134
2135const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2143
2144const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2160
2161const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2163
2164fn index_section(parts: usize) -> Result<usize> {
2166 parts
2167 .checked_mul(INDEX_ENTRY)
2168 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2169 .ok_or_else(|| invalid("index page length overflow"))
2170}
2171
2172impl Writer {
2173 pub fn open(
2192 path: impl AsRef<Path>,
2193 name: impl Into<String>,
2194 fields: Vec<Field>,
2195 ) -> Result<Self> {
2196 Self::open_in(&RealFilesystem::new(), path, name, fields)
2197 }
2198
2199 pub fn open_in(
2206 fs: &dyn Filesystem,
2207 path: impl AsRef<Path>,
2208 name: impl Into<String>,
2209 fields: Vec<Field>,
2210 ) -> Result<Self> {
2211 for field in &fields {
2212 type_tag(&field.ty)?;
2213 }
2214 let name = name.into();
2215 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2216 let size = file.len()?;
2217 let (slot, bytes, _) = committed_slot(&*file, size)?;
2218 let (mut closed, views, card) = decode_catalog(&bytes, size)?;
2219 let card = card_for(path.as_ref(), card);
2220 if let Some(at) = closed.iter().position(|held| held.name == name) {
2231 if closed[at].rows > 0 {
2232 return Err(invalid("two tables in one native file have the same name"));
2233 }
2234 closed.remove(at);
2235 }
2236 let generation = slot
2241 .generation
2242 .checked_add(1)
2243 .ok_or_else(|| invalid("native file generation overflow"))?;
2244 Ok(Self {
2245 file,
2246 at: size,
2249 written_back: size,
2250 dictionaries: fields
2251 .iter()
2252 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2253 .collect(),
2254 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2255 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2256 lent: None,
2257 table: Table {
2258 name,
2259 dictionaries: vec![None; fields.len()],
2260 dictionary_payloads: Vec::new(),
2261 demoted: Vec::new(),
2262 distincts: vec![None; fields.len()],
2263 fields,
2264 stripes: Vec::new(),
2265 rows: 0,
2266 frequencies: Vec::new(),
2267 pair_frequencies: Vec::new(),
2268 frequency_texts: Vec::new(),
2269 host_groups: None,
2270 clustering: None,
2271 constraints: Constraints::default(),
2272 generation,
2273 sections: Vec::new(),
2274 },
2275 generation,
2276 order: Vec::new(),
2277 next_order: 0,
2278 pending: Vec::with_capacity(STRIPE_PARTS),
2279 closed,
2280 views,
2281 card,
2282 profile: None,
2283 })
2284 }
2285
2286 pub fn create(
2292 path: impl AsRef<Path>,
2293 name: impl Into<String>,
2294 fields: Vec<Field>,
2295 ) -> Result<Self> {
2296 Self::create_in(&RealFilesystem::new(), path, name, fields)
2297 }
2298
2299 pub fn create_in(
2309 fs: &dyn Filesystem,
2310 path: impl AsRef<Path>,
2311 name: impl Into<String>,
2312 fields: Vec<Field>,
2313 ) -> Result<Self> {
2314 for field in &fields {
2315 type_tag(&field.ty)?;
2316 }
2317 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2318 let mut header = [0; HEADER as usize];
2319 header[..8].copy_from_slice(MAGIC);
2320 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2321 file.write_at(0, &header)?;
2322 Ok(Self {
2323 file,
2324 at: HEADER,
2325 written_back: HEADER,
2326 dictionaries: fields
2327 .iter()
2328 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2329 .collect(),
2330 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2331 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2332 lent: None,
2333 table: Table {
2334 name: name.into(),
2335 dictionaries: vec![None; fields.len()],
2336 dictionary_payloads: Vec::new(),
2337 demoted: Vec::new(),
2338 distincts: vec![None; fields.len()],
2339 fields,
2340 stripes: Vec::new(),
2341 rows: 0,
2342 frequencies: Vec::new(),
2343 pair_frequencies: Vec::new(),
2344 frequency_texts: Vec::new(),
2345 host_groups: None,
2346 clustering: None,
2347 constraints: Constraints::default(),
2348 generation: 1,
2349 sections: Vec::new(),
2350 },
2351 generation: 1,
2352 order: Vec::new(),
2353 next_order: 0,
2354 pending: Vec::with_capacity(STRIPE_PARTS),
2355 closed: Vec::new(),
2356 views: Vec::new(),
2357 card: card_for(path.as_ref(), None),
2358 profile: None,
2359 })
2360 }
2361
2362 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2384 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2385 let mut header = [0; HEADER as usize];
2386 header[..8].copy_from_slice(MAGIC);
2387 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2388 file.write_at(0, &header)?;
2389 let catalog = encode_catalog(&[], views, card_for(path.as_ref(), None).as_ref())?;
2390 file.write_at(HEADER, &catalog)?;
2391 file.sync()?;
2395 let slot = Slot {
2396 offset: HEADER,
2397 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2398 generation: 1,
2399 hash: checksum(&catalog),
2400 };
2401 file.write_at(slot_offset(1), &slot.bytes())?;
2402 file.sync()?;
2403 Ok(())
2404 }
2405
2406 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2417 for field in &fields {
2418 type_tag(&field.ty)?;
2419 }
2420 let name = name.into();
2421 let entry = self.close()?;
2422 if entry.name == name {
2423 return Err(invalid("two tables in one native file have the same name"));
2424 }
2425 if let Some(at) = self.closed.iter().position(|held| held.name == name) {
2429 if self.closed[at].rows > 0 {
2430 return Err(invalid("two tables in one native file have the same name"));
2431 }
2432 self.closed.remove(at);
2433 }
2434 let Self { file, at, generation, mut closed, views, card, .. } = self;
2435 closed.push(entry);
2436 Ok(Self {
2437 file,
2438 written_back: at,
2439 at,
2440 generation,
2441 closed,
2442 views,
2443 card,
2444 profile: None,
2445 dictionaries: fields
2446 .iter()
2447 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2448 .collect(),
2449 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2450 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2451 lent: None,
2452 table: Table {
2453 name,
2454 dictionaries: vec![None; fields.len()],
2455 dictionary_payloads: Vec::new(),
2456 demoted: Vec::new(),
2457 distincts: vec![None; fields.len()],
2458 fields,
2459 stripes: Vec::new(),
2460 rows: 0,
2461 frequencies: Vec::new(),
2462 pair_frequencies: Vec::new(),
2463 frequency_texts: Vec::new(),
2464 host_groups: None,
2465 clustering: None,
2466 constraints: Constraints::default(),
2467 generation,
2468 sections: Vec::new(),
2469 },
2470 order: Vec::new(),
2471 next_order: 0,
2472 pending: Vec::with_capacity(STRIPE_PARTS),
2473 })
2474 }
2475
2476 #[must_use]
2486 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2487 self.views = views;
2488 self
2489 }
2490
2491 #[must_use]
2497 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2498 self.profile = Some(profile);
2499 self
2500 }
2501
2502 #[must_use]
2506 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2507 self.coded.cap(bytes);
2508 self
2509 }
2510
2511 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2526 self.table.clustering = Some(Clustering::new(
2529 clustering.columns().to_vec(),
2530 clustering.width(),
2531 &self.table.fields,
2532 )?);
2533 Ok(self)
2534 }
2535
2536 pub fn constrain(mut self, constraints: Constraints) -> Result<Self> {
2544 let width = self.table.fields.len();
2545 let fits = |columns: &[u16]| {
2546 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
2547 };
2548 if !constraints.keys.iter().all(|(columns, _)| fits(columns))
2549 || !constraints.foreign.iter().all(|foreign| {
2550 fits(&foreign.columns) && foreign.referenced.len() == foreign.columns.len()
2551 })
2552 {
2553 return Err(invalid("a constraint names a column the table does not have"));
2554 }
2555 self.table.constraints = constraints;
2556 Ok(self)
2557 }
2558
2559 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2564 self.file.write_at(self.at, bytes)?;
2565 self.at = self
2566 .at
2567 .checked_add(bytes.len() as u64)
2568 .ok_or_else(|| invalid("native file length overflow"))?;
2569 if self.at - self.written_back >= WRITEBACK_STRETCH {
2570 self.file.start_writeback(self.written_back, self.at - self.written_back);
2571 self.written_back = self.at;
2572 }
2573 Ok(())
2574 }
2575
2576 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2582 let order = (self.next_order, 0);
2583 self.next_order = self.next_order.saturating_add(1);
2584 self.append_at(order, chunk)
2585 }
2586
2587 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2598 if chunk.is_empty() {
2599 return Ok(());
2600 }
2601 self.admit(chunk)?;
2602 if self.pending.last().is_some_and(|last| last.order > order) {
2603 self.flush_pending()?;
2604 }
2605 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2610 if self.pending.len() == STRIPE_PARTS {
2611 self.flush_pending()?;
2612 }
2613 Ok(())
2614 }
2615
2616 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2632 if parts.len() > STRIPE_PARTS {
2633 return Err(invalid("a stripe was handed more parts than it holds"));
2634 }
2635 self.flush_pending()?;
2638 for (order, chunk) in parts {
2639 if chunk.is_empty() {
2640 continue;
2641 }
2642 self.admit(&chunk)?;
2643 self.pending.push(PendingChunk { order, chunk });
2644 }
2645 self.flush_pending()
2646 }
2647
2648 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2650 if chunk.width() != self.table.fields.len() {
2651 return Err(invalid("chunk width differs from table schema"));
2652 }
2653 for (index, field) in self.table.fields.iter().enumerate() {
2654 if chunk.column(index)?.logical_type() != &field.ty {
2655 return Err(invalid("chunk type differs from table schema"));
2656 }
2657 }
2658 self.table.rows = self
2659 .table
2660 .rows
2661 .checked_add(chunk.len())
2662 .ok_or_else(|| invalid("row count overflow"))?;
2663 Ok(())
2664 }
2665
2666 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2668 let mut stripe = ColumnStripe {
2669 pages: Vec::with_capacity(columns.len()),
2670 sums: Vec::with_capacity(columns.len()),
2671 codes: Vec::with_capacity(columns.len()),
2672 sieves: Vec::with_capacity(columns.len()),
2673 ranges: Vec::with_capacity(columns.len()),
2674 };
2675 let mut settling = Settling::default();
2676 for &column in columns {
2677 Self::encode_page(&mut stripe, &mut settling, column)?;
2678 }
2679 Ok(stripe)
2680 }
2681
2682 fn encode_page(
2685 stripe: &mut ColumnStripe,
2686 settling: &mut Settling,
2687 column: &Vector,
2688 ) -> Result<()> {
2689 let bytes = encode(column, settling)?;
2690 if bytes.len() > MAX_PAGE {
2691 return Err(invalid("column page exceeds the configured bound"));
2692 }
2693 let range = Range::of(column);
2696 let sieve =
2707 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2708 stripe.sums.push(checksum(&bytes));
2709 stripe.pages.push(bytes);
2710 stripe.codes.push(None);
2711 stripe.sieves.push(sieve);
2712 stripe.ranges.push(range);
2713 Ok(())
2714 }
2715
2716 fn place_blocks(&mut self) -> Result<()> {
2721 if let Some(lent) = self.lent.clone() {
2722 return self.place_lent_blocks(&lent);
2723 }
2724 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2725 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2726 for block in std::mem::take(&mut dictionary.blocks) {
2727 let start = self.at;
2728 self.put(&block)?;
2729 dictionary.placed.push(Placed {
2730 start,
2731 length: block.len() as u64,
2732 hash: checksum(&block),
2733 });
2734 }
2735 Ok(())
2736 });
2737 self.dictionaries = dictionaries;
2738 placed
2739 }
2740
2741 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2747 for column in lent.columns() {
2748 let Ok(mut held) = column.try_lock() else { continue };
2749 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2750 for block in std::mem::take(&mut dictionary.blocks) {
2751 let start = self.at;
2752 self.put(&block)?;
2753 dictionary.placed.push(Placed {
2754 start,
2755 length: block.len() as u64,
2756 hash: checksum(&block),
2757 });
2758 }
2759 }
2760 Ok(())
2761 }
2762
2763 fn reclaim(&mut self) -> Result<()> {
2767 let Some(lent) = self.lent.take() else { return Ok(()) };
2768 let (dictionaries, gathers) = lent.reclaim()?;
2769 self.dictionaries = dictionaries;
2770 self.gathers = gathers;
2771 Ok(())
2772 }
2773
2774 fn flush_pending(&mut self) -> Result<()> {
2779 if self.pending.is_empty() {
2780 return Ok(());
2781 }
2782 let held = std::mem::take(&mut self.pending);
2783 let prepared = self.preparer().prepare_held(held)?;
2784 let merged = self.merge_held(prepared)?;
2785 let paged = merged.pages()?;
2786 self.write_paged(paged)
2787 }
2788
2789 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2791 let width = self.table.fields.len();
2792 let parts = held.len();
2793 if encoded.len() != width {
2794 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2795 }
2796 let profile = self.profile.clone();
2797 if let Some(profile) = &profile {
2798 let rows = held.iter().map(|part| part.rows as u64).sum();
2799 let raw = held.iter().map(|part| part.footprint as u64).sum();
2800 let pages =
2801 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2802 profile.moved(Stage::Pages, raw, pages, rows);
2803 }
2804 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2807 let before = self.at;
2808 self.place_blocks()?;
2809 drop(timing);
2810 if let Some(profile) = &profile {
2811 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2812 }
2813 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2814 let before = self.at;
2815 let mut pages = Vec::with_capacity(width);
2816 let mut memberships = vec![None; width];
2817 let mut ranges = Vec::with_capacity(width);
2818 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2819 let start = self.at;
2822 let mut out = Vec::with_capacity(width.saturating_mul(parts));
2823 for stripe in &encoded {
2824 let offset = self.at;
2825 let section = index.len();
2826 let mut length = 0_usize;
2827 if stripe.sums.len() != stripe.pages.len() {
2828 return Err(Error::internal("a stripe's pages came without their checksums"));
2829 }
2830 for (bytes, &sum) in stripe.pages.iter().zip(&stripe.sums) {
2831 put_u32(
2832 &mut index,
2833 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2834 );
2835 put_u64(&mut index, sum);
2836 out.push(bytes.as_slice());
2837 length = length
2838 .checked_add(bytes.len())
2839 .ok_or_else(|| invalid("column page length overflow"))?;
2840 }
2841 let hash = checksum(&index[section..]);
2842 put_u64(&mut index, hash);
2843 if length > MAX_PAGE {
2844 return Err(invalid("column page exceeds the configured bound"));
2845 }
2846 self.at = self
2847 .at
2848 .checked_add(length as u64)
2849 .ok_or_else(|| invalid("native file length overflow"))?;
2850 pages.push(Span {
2851 offset,
2852 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2853 });
2854 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2855 }
2856 self.file.write_parts_at(start, &out)?;
2857 drop(out);
2858 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2859 if stripe.codes.iter().all(Option::is_none) {
2860 continue;
2861 }
2862 let lists = stripe
2863 .codes
2864 .iter()
2865 .map(|codes| codes.clone().unwrap_or_default())
2866 .collect::<Vec<_>>();
2867 let bytes = encode_membership(&merged_codes(lists));
2868 let offset = self.at;
2869 self.put(&bytes)?;
2870 *membership = Some(Page {
2871 offset,
2872 length: u32::try_from(bytes.len())
2873 .map_err(|_| invalid("membership page length overflow"))?,
2874 hash: checksum(&bytes),
2875 });
2876 }
2877 let mut sieves = vec![None; width];
2878 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2879 if stripe.sieves.iter().all(Option::is_none) {
2880 continue;
2881 }
2882 let bytes = encode_sieves(stripe.sieves.iter())?;
2883 let offset = self.at;
2884 self.put(&bytes)?;
2885 *page = Some(Page {
2886 offset,
2887 length: u32::try_from(bytes.len())
2888 .map_err(|_| invalid("sieve page length overflow"))?,
2889 hash: checksum(&bytes),
2890 });
2891 }
2892 let mut part_ranges = vec![None; width];
2898 if parts > 1 {
2899 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2900 let bytes = encode_part_ranges(&stripe.ranges)?;
2901 if bytes.len() >= span.length as usize {
2902 continue;
2903 }
2904 let offset = self.at;
2905 self.put(&bytes)?;
2906 *page = Some(Page {
2907 offset,
2908 length: u32::try_from(bytes.len())
2909 .map_err(|_| invalid("part range page length overflow"))?,
2910 hash: checksum(&bytes),
2911 });
2912 }
2913 }
2914 let offset = self.at;
2915 self.put(&index)?;
2916 let index = Span {
2917 offset,
2918 length: u32::try_from(index.len())
2919 .map_err(|_| invalid("index page length overflow"))?,
2920 };
2921 let mut rows = 0_usize;
2922 let mut lengths = Vec::with_capacity(parts);
2923 let mut span = None;
2924 for part in held {
2925 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2926 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2927 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2928 }
2929 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2930 self.table.stripes.push(Stripe {
2931 rows,
2932 parts: lengths,
2933 index,
2934 pages,
2935 memberships: Pages::from_slots(memberships)?,
2936 sieves: Pages::from_slots(sieves)?,
2937 part_ranges: Pages::from_slots(part_ranges)?,
2938 zone: Zone::from_ranges(ranges),
2939 });
2940 drop(timing);
2941 if let Some(profile) = &profile {
2942 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2943 }
2944 Ok(())
2945 }
2946
2947 fn numeric_frequency(
2967 &self,
2968 column: usize,
2969 counted: bool,
2970 dense: Option<(u64, usize)>,
2971 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2972 let signed = match self.table.fields[column].ty {
2973 LogicalType::TinyInt
2974 | LogicalType::SmallInt
2975 | LogicalType::Integer
2976 | LogicalType::BigInt
2977 | LogicalType::Date
2978 | LogicalType::Timestamp => true,
2979 LogicalType::UTinyInt
2980 | LogicalType::USmallInt
2981 | LogicalType::UInteger
2982 | LogicalType::UBigInt => false,
2983 _ => return Ok((None, None)),
2984 };
2985 let value_of = |bits: Option<u64>| match bits {
2986 None => FrequencyValue::Null,
2987 Some(bits) => integer_value(bits, signed),
2988 };
2989 let tallied = self
2994 .gathers
2995 .get(column)
2996 .and_then(Option::as_ref)
2997 .filter(|gather| gather.rows() == self.table.rows as u64)
2998 .and_then(stats::Gather::frequencies)
2999 .and_then(|(values, nulls)| {
3000 let entries = values
3001 .iter()
3002 .map(|(value, count)| {
3003 let value = value_of(Some(frequency_bits(value)?));
3004 Some(FrequencyEntry { value, count: *count })
3005 })
3006 .chain((nulls != 0).then_some(Some(FrequencyEntry {
3007 value: FrequencyValue::Null,
3008 count: nulls,
3009 })))
3010 .collect::<Option<Vec<_>>>()?;
3011 Some((entries, values.len() as u64))
3012 });
3013 let exact = match (&tallied, counted) {
3017 (None, true) => self.exact_frequency(column, signed, dense)?,
3018 _ => None,
3019 };
3020 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
3021 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
3022 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
3023 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
3024 (None, None) => {
3025 let mut first = Candidates::default();
3029 let mut run = Run::default();
3030 self.visit_numeric(column, signed, |_, bits| {
3031 if let Some((ended, times)) = run.push(bits) {
3032 first.add(ended, times);
3033 }
3034 })?;
3035 if let Some((bits, times)) = run.take() {
3036 first.add(bits, times);
3037 }
3038 let (nulls, decrements) = (first.nulls, first.decrements);
3041 let distinct_count = (decrements == 0).then_some(first.held as u64);
3042 let (exact, null_count) = if decrements == 0 {
3043 let exact = first
3044 .pairs()
3045 .map(|(bits, count)| (bits, u64::from(count)))
3046 .collect::<FrequencyMap<_>>();
3047 (exact, (nulls != 0).then_some(u64::from(nulls)))
3048 } else {
3049 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
3050 if nulls != 0 {
3051 lower.push(nulls);
3052 }
3053 lower.sort_unstable_by(|left, right| right.cmp(left));
3054 if lower.len() < FREQUENCY_BUILD_RANK
3055 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
3056 {
3057 return Ok((None, distinct_count));
3058 }
3059 let mut recounts = vec![0_u64; first.slots.len()];
3062 let mut null_count = (nulls != 0).then_some(0_u64);
3063 let mut recount = |bits: Option<u64>, times: u32| {
3064 let held = match bits {
3065 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
3066 None => null_count.as_mut(),
3067 };
3068 if let Some(count) = held {
3069 *count = count.saturating_add(u64::from(times));
3070 }
3071 };
3072 let mut run = Run::default();
3073 self.visit_numeric(column, signed, |_, bits| {
3074 if let Some((bits, times)) = run.push(bits) {
3075 recount(bits, times);
3076 }
3077 })?;
3078 if let Some((bits, times)) = run.take() {
3079 recount(bits, times);
3080 }
3081 let exact = first
3082 .slots
3083 .iter()
3084 .zip(&recounts)
3085 .filter(|(slot, _)| slot.count != 0)
3086 .map(|(slot, &count)| (slot.bits, count))
3087 .collect::<FrequencyMap<_>>();
3088 (exact, null_count)
3089 };
3090 let entries = exact
3091 .into_iter()
3092 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3093 .chain(
3094 null_count
3095 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
3096 )
3097 .collect::<Vec<_>>();
3098 (entries, decrements, distinct_count)
3099 }
3100 };
3101 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
3102 if omitted_max == 0 && entries.len() > 1 {
3106 let retained = entries.len().saturating_sub(1).min(2);
3107 omitted_max = entries[retained].count;
3108 entries.truncate(retained);
3109 }
3110 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3111 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3112 });
3113 let mut ordinals = Vec::new();
3114 let mut ordinal_entries = Vec::new();
3115 if let Some(kept_rows) = kept_rows {
3116 let mut kept = FrequencyMap::default();
3117 let mut null_kept = None;
3118 for (at, entry) in entries.iter().enumerate() {
3119 let at = u16::try_from(at)
3120 .map_err(|_| invalid("too many retained frequency entries"))?;
3121 match entry.value {
3122 FrequencyValue::Integer(value) => {
3123 kept.insert(value as u64, at);
3124 }
3125 FrequencyValue::Null => null_kept = Some(at),
3126 FrequencyValue::Code(_) => {}
3127 }
3128 }
3129 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3130 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3131 self.visit_numeric(column, signed, |ordinal, bits| {
3132 let held = match bits {
3133 Some(bits) => kept.get(&bits).copied(),
3134 None => null_kept,
3135 };
3136 if let Some(entry) = held {
3137 ordinals.push(ordinal);
3138 ordinal_entries.push(entry);
3139 }
3140 })?;
3141 }
3142 Ok((
3143 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3144 distinct_count,
3145 ))
3146 }
3147
3148 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3155 let rows = self.table.rows;
3156 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3157 return None;
3158 }
3159 let (low, high) = gather.span()?;
3160 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3161 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3162 let bits = low as u64;
3163 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3164 }
3165
3166 fn exact_frequency(
3180 &self,
3181 column: usize,
3182 signed: bool,
3183 dense: Option<(u64, usize)>,
3184 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3185 if let Some((low, len)) = dense {
3188 let mut counts = distinct::DenseCounts::new(low, len);
3189 let nulls =
3190 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3191 if let Some(distinct) = counts.count() {
3192 let Some(distinct) = distinct else { return Ok(None) };
3193 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3194 counts.visit(visit);
3195 })));
3196 }
3197 }
3198 let mut set = distinct::ExactCounts::new();
3199 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3200 let Some(distinct) = set.count() else {
3201 return Ok(None);
3202 };
3203 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3204 set.visit(visit);
3205 })))
3206 }
3207
3208 fn count_numeric(
3211 &self,
3212 column: usize,
3213 signed: bool,
3214 mut add: impl FnMut(u64, u32),
3215 ) -> Result<u64> {
3216 let mut nulls = 0_u64;
3217 let mut run = Run::default();
3218 let mut take = |bits: Option<u64>, times: u32| match bits {
3219 Some(bits) => add(bits, times),
3220 None => nulls += u64::from(times),
3221 };
3222 self.visit_numeric(column, signed, |_, bits| {
3223 if let Some((bits, times)) = run.push(bits) {
3224 take(bits, times);
3225 }
3226 })?;
3227 if let Some((bits, times)) = run.take() {
3228 take(bits, times);
3229 }
3230 Ok(nulls)
3231 }
3232
3233 fn frequent_entries(
3236 &self,
3237 signed: bool,
3238 distinct: u64,
3239 nulls: u64,
3240 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3241 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3242 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3244 let mut rank = |count: u64| {
3245 if top.len() <= FREQUENCY_ENTRIES {
3246 top.push(Reverse(count));
3247 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3248 top.pop();
3249 top.push(Reverse(count));
3250 }
3251 };
3252 visit(&mut |_, count| rank(count));
3253 if nulls != 0 {
3254 rank(nulls);
3255 }
3256 let top = top.into_sorted_vec();
3257 let values = distinct + u64::from(nulls != 0);
3258 if values > FREQUENCY_CANDIDATES as u64 {
3259 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3260 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3261 return (None, distinct);
3262 }
3263 }
3264 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3265 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3266 visit(&mut |bits, count| {
3267 if count >= least {
3268 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3269 }
3270 });
3271 if nulls != 0 && nulls >= least {
3272 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3273 }
3274 (Some(entries), distinct)
3275 }
3276
3277 fn visit_numeric(
3284 &self,
3285 column: usize,
3286 signed: bool,
3287 mut visit: impl FnMut(u64, Option<u64>),
3288 ) -> Result<()> {
3289 let ty = &self.table.fields[column].ty;
3290 let mut start = 0_u64;
3291 let mut block = Vec::new();
3292 for stripe in &self.table.stripes {
3293 let spans = read_index(&self.file, stripe, column)?;
3294 let page = stripe.pages[column];
3295 let mut bytes = vec![0; page.length as usize];
3296 read_at(&self.file, page.offset, &mut bytes)?;
3297 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3298 let part = part_bytes(&bytes, *span)?;
3299 if checksum(part) != span.hash {
3300 return Err(invalid("column page checksum differs while building frequencies"));
3301 }
3302 let rows = rows as usize;
3303 let vector = decode(ty, rows, part, None)?;
3304 if signed && vector.signed_block(&mut block) && block.len() == rows {
3308 if vector.none_null() {
3309 for (row, &value) in block.iter().enumerate() {
3310 visit(start.saturating_add(row as u64), Some(value as u64));
3311 }
3312 } else {
3313 for (row, &value) in block.iter().enumerate() {
3314 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3315 visit(start.saturating_add(row as u64), bits);
3316 }
3317 }
3318 start = start.saturating_add(rows as u64);
3319 continue;
3320 }
3321 for row in 0..rows {
3323 let bits = if vector.is_null_at(row) {
3324 None
3325 } else {
3326 let widened = match vector.signed_at(row) {
3330 Some(value) => Some(value as u64),
3331 None => match vector.value_at(row) {
3332 Value::UTinyInt(value) => Some(u64::from(value)),
3333 Value::USmallInt(value) => Some(u64::from(value)),
3334 Value::UInteger(value) => Some(u64::from(value)),
3335 Value::UBigInt(value) => Some(value),
3336 _ => None,
3337 },
3338 };
3339 Some(widened.ok_or_else(|| {
3340 invalid("numeric frequency page did not contain an integer value")
3341 })?)
3342 };
3343 visit(start.saturating_add(row as u64), bits);
3344 }
3345 start = start.saturating_add(rows as u64);
3346 }
3347 }
3348 Ok(())
3349 }
3350
3351 fn numeric_columns(&self) -> Vec<usize> {
3353 self.table
3354 .fields
3355 .iter()
3356 .enumerate()
3357 .filter_map(|(column, field)| {
3358 matches!(
3359 field.ty,
3360 LogicalType::TinyInt
3361 | LogicalType::SmallInt
3362 | LogicalType::Integer
3363 | LogicalType::BigInt
3364 | LogicalType::UTinyInt
3365 | LogicalType::USmallInt
3366 | LogicalType::UInteger
3367 | LogicalType::UBigInt
3368 | LogicalType::Date
3369 | LogicalType::Timestamp
3370 )
3371 .then_some(column)
3372 })
3373 .collect()
3374 }
3375
3376 #[allow(dead_code)]
3378 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3379 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3380 return Ok(None);
3381 }
3382 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3383 return Err(invalid("frequency ordinals are not sorted and unique"));
3384 }
3385 let mut out = Vec::with_capacity(ordinals.len());
3386 let mut wanted = 0;
3387 let mut stripe_start = 0_u64;
3388 for stripe in &self.table.stripes {
3389 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3390 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3391 stripe_start = stripe_end;
3392 continue;
3393 }
3394 let spans = read_index(&self.file, stripe, column)?;
3395 let page = stripe.pages[column];
3396 let mut bytes = vec![0; page.length as usize];
3397 read_at(&self.file, page.offset, &mut bytes)?;
3398 let mut part_start = stripe_start;
3399 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3400 let part_end = part_start.saturating_add(u64::from(rows));
3401 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3402 let part = part_bytes(&bytes, *span)?;
3403 if checksum(part) != span.hash {
3404 return Err(invalid(
3405 "column page checksum differs while building pair frequencies",
3406 ));
3407 }
3408 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3409 let positions = ordinals[wanted..upto]
3410 .iter()
3411 .map(|&ordinal| {
3412 usize::try_from(ordinal.saturating_sub(part_start))
3413 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3414 })
3415 .collect::<Result<Vec<_>>>()?;
3416 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3417 return Ok(None);
3418 }
3419 wanted = upto;
3420 }
3421 part_start = part_end;
3422 }
3423 stripe_start = stripe_end;
3424 }
3425 if wanted != ordinals.len() {
3426 return Err(invalid("frequency ordinal is outside the table"));
3427 }
3428 Ok(Some(out))
3429 }
3430
3431 #[allow(dead_code)]
3433 fn pair_frequencies(
3434 &self,
3435 frequencies: &[Option<Frequencies>],
3436 ) -> Result<Vec<PairFrequencySummary>> {
3437 let anchors = frequencies
3438 .iter()
3439 .enumerate()
3440 .filter_map(|(column, summary)| {
3441 match summary {
3443 Some(Frequencies::Held(summary)) => Some(summary),
3444 _ => None,
3445 }
3446 .filter(|summary| {
3447 !summary.ordinals.is_empty()
3448 && summary.ordinal_entries.len() == summary.ordinals.len()
3449 })
3450 .cloned()
3451 .map(|summary| (column, summary))
3452 })
3453 .collect::<Vec<_>>();
3454 let strings = self
3455 .dictionaries
3456 .iter()
3457 .enumerate()
3458 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3459 .collect::<Vec<_>>();
3460 let mut summaries = Vec::new();
3461 for (first, anchors) in anchors {
3462 for &second in &strings {
3463 if summaries.len() == MAX_PAIR_FREQUENCIES {
3464 return Ok(summaries);
3465 }
3466 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3467 continue;
3468 };
3469 if codes.len() != anchors.ordinal_entries.len() {
3470 return Err(invalid("pair frequency columns have different lengths"));
3471 }
3472 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3473 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3474 *counts.entry((anchor, code)).or_default() += 1;
3475 }
3476 let mut entries = counts
3477 .into_iter()
3478 .map(|((first_entry, second), count)| PairFrequencyEntry {
3479 first_entry,
3480 second,
3481 count,
3482 })
3483 .collect::<Vec<_>>();
3484 entries.sort_unstable_by(|left, right| {
3485 right
3486 .count
3487 .cmp(&left.count)
3488 .then_with(|| left.first_entry.cmp(&right.first_entry))
3489 .then_with(|| left.second.cmp(&right.second))
3490 });
3491 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3492 entries.truncate(FREQUENCY_ENTRIES);
3493 summaries.push(PairFrequencySummary {
3494 first: u16::try_from(first)
3495 .map_err(|_| invalid("pair frequency column index overflows"))?,
3496 second: u16::try_from(second)
3497 .map_err(|_| invalid("pair frequency column index overflows"))?,
3498 entries,
3499 omitted_max: anchors.omitted_max.max(pair_omitted),
3500 });
3501 }
3502 }
3503 Ok(summaries)
3504 }
3505
3506 fn close(&mut self) -> Result<Entry> {
3517 self.reclaim()?;
3518 self.flush_pending()?;
3519 let profile = self.profile.clone();
3523 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3524 let before = self.at;
3525 let mut stripes = std::mem::take(&mut self.order)
3526 .into_iter()
3527 .zip(std::mem::take(&mut self.table.stripes))
3528 .collect::<Vec<_>>();
3529 stripes.sort_by_key(|(order, _)| order.0);
3530 let mut previous: Option<(u64, u64)> = None;
3531 for ((first, last), _) in &stripes {
3532 if previous.is_some_and(|previous| previous >= *first) {
3533 return Err(invalid("chunks did not arrive in source order"));
3534 }
3535 previous = Some(*last);
3536 }
3537 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3538 drop(timing);
3539 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3540 let placing = self.at;
3541 finish_dictionaries(&mut self.dictionaries)?;
3542 self.place_blocks()?;
3543 for dictionary in self.dictionaries.iter_mut().flatten() {
3544 dictionary.release_lookup();
3545 dictionary.recharge(profile.as_deref());
3546 }
3547 let (numeric, closed) = self.close_columns()?;
3548 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3549 numeric.into_iter().unzip();
3550 let frequencies =
3551 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3552 let pairs = Vec::new();
3554 self.table.frequencies = frequencies;
3555 self.table.distincts = distincts;
3556 self.table.pair_frequencies = pairs;
3557 if let Some(profile) = &profile {
3558 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3559 }
3560 self.table.demoted = self
3561 .dictionaries
3562 .iter()
3563 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3564 .collect();
3565 if !self.table.demoted.contains(&true) {
3566 self.table.demoted = Vec::new();
3567 }
3568 self.dictionaries = Vec::new();
3569 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3570 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3571 self.table.host_groups = None;
3572 for (index, closed) in closed.into_iter().enumerate() {
3573 let Some(closed) = closed else { continue };
3574 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3575 self.table.distincts[index] = distinct;
3576 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3577 self.table.frequency_texts[index] = texts;
3578 if hosts.is_some() {
3579 self.table.host_groups = hosts;
3580 }
3581 let offset = self.at;
3582 self.put(&encoded.index)?;
3583 self.put(&encoded.ranks)?;
3584 self.put(&encoded.grams)?;
3585 self.table.dictionary_payloads[index] = payload;
3586 let length = encoded
3587 .index
3588 .len()
3589 .checked_add(encoded.ranks.len())
3590 .and_then(|len| len.checked_add(encoded.grams.len()))
3591 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3592 self.table.dictionaries[index] = Some(Page {
3593 offset,
3594 length: u32::try_from(length)
3595 .map_err(|_| invalid("dictionary page length overflow"))?,
3596 hash: checksum(&encoded.index),
3597 });
3598 }
3599 drop(timing);
3600 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3601 let placed = self.at - placing;
3602 self.write_stats()?;
3603 let directory = encode_directory(&self.table)?;
3604 if directory.len() > MAX_DIRECTORY {
3605 return Err(invalid("directory exceeds the configured bound"));
3606 }
3607 let offset = self.at;
3608 self.put(&directory)?;
3609 drop(timing);
3610 if let Some(profile) = &profile {
3611 profile.moved(Stage::Dictionary, 0, placed, 0);
3612 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3613 }
3614 Ok(Entry {
3615 name: self.table.name.clone(),
3616 fields: self.table.fields.clone(),
3617 rows: self.table.rows,
3618 nonzero: vec![None; self.table.fields.len()],
3619 aggregates: table_aggregate_sums(&self.table),
3620 distincts: self.table.distincts.clone(),
3621 extremes: table_integer_extremes(&self.table),
3622 frequencies: table_complete_numeric_frequencies(&self.table),
3623 directory: Page {
3624 offset,
3625 length: u32::try_from(directory.len())
3626 .map_err(|_| invalid("directory length overflow"))?,
3627 hash: checksum(&directory),
3628 },
3629 })
3630 }
3631
3632 #[allow(clippy::type_complexity)]
3649 fn close_columns(
3650 &self,
3651 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3652 let numeric = self.numeric_columns().into_iter().map(|column| {
3653 let gather = self.gathers.get(column).and_then(Option::as_ref);
3654 let estimate = gather.and_then(stats::Gather::distinct);
3655 let counted = !estimate.is_some_and(distinct::beyond);
3656 let set =
3657 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3658 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3659 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3660 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3661 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3662 });
3663 let dictionaries =
3664 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3665 let dictionary = dictionary.as_ref()?;
3666 let bytes = dictionary.closing_bytes();
3667 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3668 });
3669 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3670 jobs.sort_by_key(|&(_, _, cost)| cost);
3671 let columns = self.table.fields.len();
3672 let mut frequencies = vec![(None, None); columns];
3673 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3674 let profile = self.profile.as_deref();
3675 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3676 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3677 let closed = match job {
3678 Closing::Numeric { column, counted, dense } => {
3679 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3680 Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?)
3681 }
3682 Closing::Dictionary { index, dictionary } => {
3683 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3684 Closed::Dictionary(index, self.close_dictionary(index, dictionary)?)
3685 }
3686 };
3687 rudb_common::heap::release();
3690 Ok(closed)
3691 };
3692 let workers = close_workers().min(jobs.len());
3693 let pieces = if workers <= 1 {
3694 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3695 } else {
3696 let state = Mutex::new((jobs, 0_usize));
3698 let finished = Condvar::new();
3699 std::thread::scope(|scope| {
3700 (0..workers)
3701 .map(|_| {
3702 scope.spawn(|| {
3703 let mut mine = Vec::new();
3704 loop {
3705 let mut held = state.lock().map_err(|_| {
3706 Error::internal("a native close worker panicked")
3707 })?;
3708 let (job, bytes) = loop {
3709 let (jobs, busy) = &mut *held;
3710 if jobs.is_empty() {
3711 return Ok(mine);
3712 }
3713 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3714 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3715 });
3716 if let Some(at) = fits {
3717 let (job, bytes, _) = jobs.remove(at);
3718 *busy += bytes;
3719 break (job, bytes);
3720 }
3721 held = finished.wait(held).map_err(|_| {
3722 Error::internal("a native close worker panicked")
3723 })?;
3724 };
3725 drop(held);
3726 let _room = Room { state: &state, finished: &finished, bytes };
3729 mine.push(run(job, bytes)?);
3730 }
3731 })
3732 })
3733 .collect::<Vec<_>>()
3734 .into_iter()
3735 .map(|handle| {
3736 handle
3737 .join()
3738 .map_err(|_| Error::internal("a native close worker panicked"))?
3739 })
3740 .collect::<Result<Vec<_>>>()
3741 })?
3742 .into_iter()
3743 .flatten()
3744 .collect()
3745 };
3746 for piece in pieces {
3747 match piece {
3748 Closed::Numeric(column, summary) => frequencies[column] = summary,
3749 Closed::Dictionary(index, one) => closed[index] = Some(one),
3750 }
3751 }
3752 Ok((frequencies, closed))
3753 }
3754
3755 fn close_dictionary(
3762 &self,
3763 _index: usize,
3764 dictionary: &GlobalDictionary,
3765 ) -> Result<ClosedDictionary> {
3766 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3767 let (distinct, frequencies, texts) = if dictionary.demoted {
3772 (None, None, Vec::new())
3773 } else {
3774 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3775 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3776 (Some(distinct), Some(frequencies), texts)
3777 };
3778 let hosts = None;
3780 drop(flat);
3781 drop(bases);
3782 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3783 let payload = dictionary
3784 .placed
3785 .iter()
3786 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3787 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3788 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3789 }
3790
3791 fn write_stats(&mut self) -> Result<()> {
3803 let gathers = std::mem::take(&mut self.gathers);
3804 let rows = self.table.rows as u64;
3805 let mut payloads = Vec::new();
3806 for (column, gather) in gathers.into_iter().enumerate() {
3807 let Some(gather) = gather else { continue };
3808 if gather.rows() != rows {
3814 continue;
3815 }
3816 let Some(stats) = gather.finish() else { continue };
3817 let mut summary = Vec::new();
3818 stats.summary.encode(&mut summary)?;
3819 let mut sketches = Vec::new();
3820 stats.sketches.encode(&mut sketches)?;
3821 payloads.push((column, summary, sketches));
3822 }
3823 if payloads.is_empty() {
3824 return Ok(());
3825 }
3826 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3827 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3828 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3829 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3832 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3833 if !built {
3834 continue;
3835 }
3836 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3837 let sections = [
3838 (*section::SUMMARY, summary, summary.len() as u32),
3841 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3842 ];
3843 let wanted = 1 + usize::from(sketched);
3844 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3845 let written = write_section(
3846 &*self.file,
3847 &mut self.at,
3848 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3849 self.generation,
3850 )?;
3851 self.table.sections.push(written);
3852 }
3853 }
3854 if self.table.sections.len() > MAX_SECTIONS {
3855 return Err(invalid("the table would name more sections than the bound allows"));
3856 }
3857 Ok(())
3858 }
3859
3860 pub fn finish(mut self) -> Result<Table> {
3870 let entry = self.close()?;
3871 let profile = self.profile.take();
3872 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3873 let mut tables = std::mem::take(&mut self.closed);
3874 tables.push(entry);
3875 let catalog = encode_catalog(&tables, &self.views, self.card.as_ref())?;
3876 if catalog.len() > MAX_DIRECTORY {
3877 return Err(invalid("catalog exceeds the configured bound"));
3878 }
3879 let offset = self.at;
3880 self.put(&catalog)?;
3881 if let Some(profile) = &profile {
3882 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3883 }
3884 synced(&*self.file, profile.as_deref())?;
3888 let slot = Slot {
3889 offset,
3890 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3891 generation: self.generation,
3892 hash: checksum(&catalog),
3893 };
3894 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3899 synced(&*self.file, profile.as_deref())?;
3900 Ok(self.table)
3901 }
3902
3903 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3920 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3921 let size = file.len()?;
3922 let (slot, bytes, _) = committed_slot(&*file, size)?;
3923 let (closed, _, card) = decode_catalog(&bytes, size)?;
3924 let generation = slot
3925 .generation
3926 .checked_add(1)
3927 .ok_or_else(|| invalid("native file generation overflow"))?;
3928 let catalog = encode_catalog(&closed, views, card_for(path.as_ref(), card).as_ref())?;
3929 if catalog.len() > MAX_DIRECTORY {
3930 return Err(invalid("catalog exceeds the configured bound"));
3931 }
3932 file.write_at(size, &catalog)?;
3933 file.sync()?;
3934 let slot = Slot {
3935 offset: size,
3936 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3937 generation,
3938 hash: checksum(&catalog),
3939 };
3940 file.write_at(slot_offset(generation), &slot.bytes())?;
3941 file.sync()?;
3942 Ok(())
3943 }
3944
3945 pub fn keep_device_card(path: impl AsRef<Path>) -> Result<()> {
3956 let path = path.as_ref();
3957 let (_, size, _, bytes, _) = slot_bytes(path)?;
3958 let (_, views, held) = decode_catalog(&bytes, size)?;
3959 if card_for(path, held.clone()) == held {
3960 return Ok(());
3961 }
3962 Self::restate(path, &views)
3963 }
3964
3965 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3968 let path = path.as_ref();
3969 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3970 let (mut entries, views, card) = decode_catalog(&bytes, size)?;
3971 let native = Catalog::open(path)?;
3972 for entry in &mut entries {
3973 let reader = native.table(&entry.name)?;
3974 entry.nonzero.fill(None);
3975 entry.aggregates = reader_aggregate_sums(&reader)?;
3976 entry.distincts = (0..entry.fields.len())
3977 .map(|column| reader.distinct_values(column))
3978 .collect::<Result<Vec<_>>>()?;
3979 entry.extremes = reader_integer_extremes(&reader)?;
3980 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3981 }
3982 let generation = slot
3983 .generation
3984 .checked_add(1)
3985 .ok_or_else(|| invalid("native file generation overflow"))?;
3986 let catalog = encode_catalog(&entries, &views, card_for(path, card).as_ref())?;
3987 if catalog.len() > MAX_DIRECTORY {
3988 return Err(invalid("catalog exceeds the configured bound"));
3989 }
3990 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3991 file.write_at(size, &catalog)?;
3992 file.sync()?;
3993 let slot = Slot {
3994 offset: size,
3995 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3996 generation,
3997 hash: checksum(&catalog),
3998 };
3999 file.write_at(slot_offset(generation), &slot.bytes())?;
4000 file.sync()?;
4001 Ok(())
4002 }
4003
4004 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
4006 Self::certify_summaries(path)
4007 }
4008}
4009
4010fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
4016 let offset = *at;
4017 file.write_at(offset, bytes)?;
4018 *at =
4019 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
4020 Ok(offset)
4021}
4022
4023fn write_section(
4029 file: &dyn rudb_io::File,
4030 at: &mut u64,
4031 one: §ion::Attachment<'_>,
4032 generation: u64,
4033) -> Result<Section> {
4034 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
4038 return Err(invalid("a section's header is longer than its payload"));
4039 }
4040 let mut extents = Vec::new();
4041 let mut first = 0_u64;
4042 let extent_size =
4043 if one.kind == *section::RUN_PROJECTION && one.flags == run_projection::RLE_PAGES {
4044 run_projection::RLE_PAGE_BYTES
4045 } else if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
4046 1 << 19
4047 } else {
4048 section::MAX_EXTENT as usize
4049 };
4050 for chunk in one.bytes.chunks(extent_size) {
4051 let offset = append(file, at, chunk)?;
4052 extents.push(section::Extent {
4053 offset,
4054 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
4055 hash: checksum(chunk),
4056 first,
4057 });
4058 first += chunk.len() as u64;
4059 }
4060 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
4061 section::encode_extents(&extents, &mut table)?;
4062 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
4066 Ok(Section {
4067 kind: one.kind,
4068 id: one.id,
4069 generation,
4070 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
4071 extent_page,
4072 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
4073 hash: checksum(&table),
4074 flags: one.flags,
4075 header_bytes: one.header_bytes,
4076 })
4077}
4078
4079pub fn attach(
4103 path: impl AsRef<Path>,
4104 table: &str,
4105 attachments: &[section::Attachment<'_>],
4106) -> Result<Table> {
4107 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
4108 let file = &*file;
4109 let size = file.len()?;
4110 let (slot, bytes, _) = committed_slot(file, size)?;
4111 let (mut entries, views, card) = decode_catalog(&bytes, size)?;
4112 let card = card_for(path.as_ref(), card);
4113 let at = entries
4114 .iter()
4115 .position(|entry| entry.name == table)
4116 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
4117 let mut version = [0; 4];
4118 read_at(file, 8, &mut version)?;
4119 let version = u32::from_le_bytes(version);
4120 if version != FORMAT {
4126 return Err(invalid(&format!(
4127 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
4128 to be written again"
4129 )));
4130 }
4131 let mut directory = vec![0; entries[at].directory.length as usize];
4132 read_at(file, entries[at].directory.offset, &mut directory)?;
4133 if checksum(&directory) != entries[at].directory.hash {
4134 return Err(invalid(&format!("the directory of table {table} does not checksum")));
4135 }
4136 let mut held = decode_directory(&directory, size)?;
4137 let mut cursor = size;
4138 for one in attachments {
4139 let written = write_section(file, &mut cursor, one, held.generation)?;
4140 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4141 held.sections.push(written);
4142 }
4143 if held.sections.len() > MAX_SECTIONS {
4144 return Err(invalid("the table would name more sections than the bound allows"));
4145 }
4146 let encoded = encode_directory(&held)?;
4147 if encoded.len() > MAX_DIRECTORY {
4148 return Err(invalid("directory exceeds the configured bound"));
4149 }
4150 let offset = append(file, &mut cursor, &encoded)?;
4151 entries[at].directory = Page {
4152 offset,
4153 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4154 hash: checksum(&encoded),
4155 };
4156 let catalog = encode_catalog(&entries, &views, card.as_ref())?;
4159 if catalog.len() > MAX_DIRECTORY {
4160 return Err(invalid("catalog exceeds the configured bound"));
4161 }
4162 let offset = append(file, &mut cursor, &catalog)?;
4163 file.sync()?;
4164 let generation =
4165 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4166 let committed = Slot {
4167 offset,
4168 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4169 generation,
4170 hash: checksum(&catalog),
4171 };
4172 file.write_at(slot_offset(generation), &committed.bytes())?;
4173 file.sync()?;
4174 Ok(held)
4175}
4176
4177type Synopsis = Arc<Vec<(Value, u64)>>;
4180
4181#[derive(Debug, Clone)]
4183pub struct Reader {
4184 file: Arc<File>,
4185 table: Arc<Table>,
4186 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4187 loading: Arc<Vec<Mutex<()>>>,
4196 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4199 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4203 summaries: Arc<Vec<OnceLock<Option<Arc<rudb_stats::Summary>>>>>,
4205 opened: Arc<AtomicUsize>,
4209 sieves: Arc<Vec<Vec<SieveSlot>>>,
4213 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4216 places: Arc<Vec<Place>>,
4218 cache: Arc<Shelf>,
4219 pool: PagePool,
4221 pages: Arc<AtomicUsize>,
4224 indexes: Arc<AtomicUsize>,
4227 verified: Arc<Vec<AtomicU64>>,
4237 text_grams: Arc<Vec<OnceLock<Option<Vec<u64>>>>>,
4240 firsts: Arc<Vec<usize>>,
4242 size: u64,
4244 directory: u64,
4246 opening: Opening,
4248}
4249
4250#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4262pub struct Opening {
4263 pub reads: u32,
4266 pub bytes: u64,
4268}
4269
4270#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4272pub struct Reads {
4273 pub opening: Opening,
4275 pub pages: usize,
4277 pub indexes: usize,
4279 pub dictionaries: usize,
4282}
4283
4284#[derive(Debug, Clone, Copy)]
4286struct Place {
4287 stripe: u32,
4288 part: u32,
4289 rows: u32,
4290}
4291
4292#[derive(Debug, Clone, Copy)]
4294struct PartSpan {
4295 start: usize,
4296 length: usize,
4297 hash: u64,
4298}
4299
4300#[derive(Debug, Clone)]
4306struct CachedColumn {
4307 stripe: usize,
4308 index: Arc<Vec<PartSpan>>,
4309 page: Option<Arc<HeldPage>>,
4310}
4311
4312#[derive(Debug)]
4319struct HeldPage {
4320 bytes: Vec<u8>,
4321 checked: Vec<AtomicBool>,
4322}
4323
4324impl HeldPage {
4325 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4327 let bytes = part_bytes(&self.bytes, span)?;
4328 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4329 if !checked.load(Atomic::Relaxed) {
4330 verify_part(bytes, span)?;
4331 checked.store(true, Atomic::Relaxed);
4332 }
4333 Ok(bytes)
4334 }
4335}
4336
4337fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4339 let got = checksum(bytes);
4340 if got != span.hash {
4341 return Err(invalid(&format!(
4342 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4343 span.start, span.length, span.hash,
4344 )));
4345 }
4346 Ok(())
4347}
4348
4349#[derive(Debug, Default)]
4385struct Cached {
4386 pages: Vec<Option<Resident>>,
4387 loading: Vec<usize>,
4388 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4389 touched: Vec<Vec<u64>>,
4390 passing: VecDeque<usize>,
4391}
4392
4393#[derive(Debug, Clone)]
4395struct Resident {
4396 page: Arc<HeldPage>,
4397 used: Arc<AtomicBool>,
4398}
4399
4400#[derive(Debug)]
4402struct Shelf {
4403 columns: Vec<Mutex<Cached>>,
4404 held: Vec<AtomicUsize>,
4407 kept: AtomicUsize,
4410}
4411
4412#[derive(Debug, Clone, Default)]
4431pub struct PagePool {
4432 ring: Arc<Mutex<Ring>>,
4433 budget: Arc<AtomicUsize>,
4434}
4435
4436#[derive(Debug, Default)]
4437struct Ring {
4438 held: VecDeque<Held>,
4439 bytes: usize,
4440}
4441
4442#[derive(Debug)]
4447struct Held {
4448 shelf: Weak<Shelf>,
4449 column: usize,
4450 stripe: usize,
4451 bytes: usize,
4452 used: Arc<AtomicBool>,
4453}
4454
4455impl PagePool {
4456 #[must_use]
4458 pub fn new(budget: usize) -> Self {
4459 let pool = Self::default();
4460 pool.budget.store(budget, Atomic::Relaxed);
4461 pool
4462 }
4463
4464 #[must_use]
4470 pub fn bytes(&self) -> usize {
4471 self.ring.lock().map_or(0, |ring| ring.bytes)
4472 }
4473
4474 fn admit(&self, held: Held) {
4480 let budget = self.budget.load(Atomic::Relaxed);
4481 let mut gone = Vec::new();
4482 {
4483 let Ok(mut ring) = self.ring.lock() else { return };
4484 ring.bytes += held.bytes;
4485 ring.held.push_back(held);
4486 let mut looked = 0;
4489 let limit = ring.held.len();
4490 while ring.bytes > budget && looked < limit {
4491 looked += 1;
4492 let Some(entry) = ring.held.pop_front() else { break };
4493 let Some(shelf) = entry.shelf.upgrade() else {
4494 ring.bytes -= entry.bytes;
4495 continue;
4496 };
4497 if entry.used.swap(false, Atomic::Relaxed) {
4498 ring.held.push_back(entry);
4499 continue;
4500 }
4501 let count = &shelf.held[entry.column];
4502 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4503 ring.held.push_back(entry);
4504 continue;
4505 }
4506 count.fetch_sub(1, Atomic::Relaxed);
4507 ring.bytes -= entry.bytes;
4508 gone.push((shelf, entry));
4509 }
4510 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4513 if let Some(entry) = ring.held.pop_front() {
4514 ring.bytes -= entry.bytes;
4515 }
4516 }
4517 }
4518 for (shelf, entry) in gone {
4519 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4520 if let Some(slot) = cached.pages.get_mut(entry.stripe)
4521 && slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used))
4522 {
4523 *slot = None;
4524 }
4525 }
4526 }
4527}
4528
4529const CACHED_STRIPES_PER_COLUMN: usize = 4;
4541
4542type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4544
4545type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4546
4547#[derive(Debug)]
4548struct NativeText {
4549 file: Arc<File>,
4550 values: usize,
4552 offsets: Vec<u8>,
4564 offset_bits: usize,
4567 value_ends: OnceLock<Option<Vec<u32>>>,
4580 value_lens: OnceLock<Option<Lengths>>,
4590 ends_asked: AtomicUsize,
4596 ranks: usize,
4598 rank_at: u64,
4602 rank_ends: Vec<u64>,
4606 rank_hashes: Vec<u64>,
4607 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4608 code_bits: usize,
4611 code_ranks: OnceLock<Option<Vec<u32>>>,
4618 starts: Vec<u64>,
4625 lengths: Vec<u64>,
4626 hashes: Vec<u64>,
4627 grams: Option<NativeGrams>,
4629 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4631 char_lens: Vec<OnceLock<Box<[u32]>>>,
4640 keep_budget: usize,
4643 payload_kept: AtomicUsize,
4651 swept: Vec<AtomicBool>,
4659 visit_dropped: AtomicUsize,
4674 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4691}
4692
4693#[derive(Debug)]
4694struct NativeGrams {
4695 start: u64,
4696 length: usize,
4697 width: usize,
4699 hash: u64,
4700 verdicts: Mutex<Vec<Verdict>>,
4707}
4708
4709type Verdict = (Vec<u8>, Arc<[bool]>);
4711
4712const GRAM_VERDICTS: usize = 8;
4714
4715impl NativeGrams {
4716 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4721 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4722 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4723 return Ok(Arc::clone(verdict));
4724 }
4725 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4726 let mut verdict = Vec::with_capacity(self.length / self.width);
4727 let window = GRAM_WINDOW / self.width * self.width;
4728 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4729 verdict.extend(bytes.chunks(self.width).map(|bits| {
4730 wanted
4731 .iter()
4732 .flatten()
4733 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4734 }));
4735 Ok(())
4736 })?;
4737 if hash != self.hash {
4738 return Err(invalid("global dictionary substring signatures checksum differs"));
4739 }
4740 let verdict: Arc<[bool]> = verdict.into();
4741 if held.len() >= GRAM_VERDICTS {
4742 held.remove(0);
4743 }
4744 held.push((literal.to_vec(), Arc::clone(&verdict)));
4745 Ok(verdict)
4746 }
4747
4748 fn footprint(&self) -> usize {
4749 self.verdicts.lock().map_or(0, |held| {
4750 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4751 })
4752 }
4753}
4754
4755const TEXT_SEARCH_MEMO: usize = 64;
4760
4761const TEXT_PAYLOAD_VALUES: usize = 1024;
4777
4778const TEXT_GRAM_BYTES: usize = 8192;
4789
4790const NARROW_GRAM_BYTES: usize = 2048;
4792
4793const GRAM_WINDOW: usize = 256 << 10;
4795
4796fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4799 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4800 let mut first = original ^ (original >> 16);
4801 first = first.wrapping_mul(0x7feb_352d);
4802 first ^= first >> 15;
4803 let mut second = original ^ (original >> 17);
4804 second = second.wrapping_mul(0x846c_a68b);
4805 second ^= second >> 16;
4806 let mask = width * 8 - 1;
4807 [(first as usize) & mask, (second as usize) & mask]
4808}
4809
4810const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4831
4832#[derive(Debug)]
4839enum Lengths {
4840 Narrow(Vec<u16>),
4842 Wide(Vec<u32>),
4844}
4845
4846impl Lengths {
4847 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4850 match self {
4851 Lengths::Narrow(lens) => into.extend(
4852 indices
4853 .iter()
4854 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4855 ),
4856 Lengths::Wide(lens) => into.extend(
4857 indices
4858 .iter()
4859 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4860 ),
4861 }
4862 }
4863
4864 fn footprint(&self) -> usize {
4866 match self {
4867 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4868 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4869 }
4870 }
4871}
4872
4873fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4882 match lengths_as::<u16>(ends)? {
4883 Some(narrow) => Some(Lengths::Narrow(narrow)),
4884 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4885 }
4886}
4887
4888fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4891 let mut lens = Vec::with_capacity(ends.len());
4892 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4893 let mut start = 0;
4894 for &end in block {
4895 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4896 return Some(None);
4897 };
4898 lens.push(len);
4899 start = end;
4900 }
4901 }
4902 Some(Some(lens))
4903}
4904
4905const TEXT_OFFSET_RUN: usize = 512;
4912
4913const DICTIONARY_HEADER: usize = 16;
4916
4917const DICTIONARY_SCATTERED: u32 = 1 << 31;
4931const DICTIONARY_GRAMS: u32 = 1 << 30;
4933const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4936const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4938
4939const TEXT_RANK_BLOCK: usize = 512;
4950
4951const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4965
4966impl NativeText {
4967 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4974 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4975 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4976 Ok(Some(bytes.as_slice()))
4977 }
4978
4979 fn block_chars(&self, block: usize) -> Result<&[u32]> {
4986 let slot = self
4987 .char_lens
4988 .get(block)
4989 .ok_or_else(|| invalid("a block past the global dictionary"))?;
4990 if let Some(lens) = slot.get() {
4991 return Ok(lens);
4992 }
4993 let decoded;
4994 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4995 Some(Ok(kept)) => kept,
4996 _ => {
4997 decoded = self.decode_block(block)?;
4998 &decoded
4999 }
5000 };
5001 let first = block * TEXT_PAYLOAD_VALUES;
5002 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5003 let ends = self.ends_within(first, last)?;
5004 if ends.len() != last - first {
5005 return Err(invalid("global dictionary offsets are short"));
5006 }
5007 let mut lens = Vec::with_capacity(ends.len());
5008 let mut start = u64::from(self.start_within(first)?);
5009 for &end in &ends {
5010 let value = usize::try_from(start)
5011 .ok()
5012 .zip(usize::try_from(end).ok())
5013 .and_then(|(from, to)| bytes.get(from..to))
5014 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5015 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
5018 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
5019 start = end;
5020 }
5021 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
5022 }
5023
5024 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
5029 let len = self.lengths[block];
5030 let mut stored = vec![
5031 0;
5032 usize::try_from(len).map_err(|_| invalid(
5033 "global dictionary block does not fit in memory"
5034 ))?
5035 ];
5036 read_at(&self.file, self.starts[block], &mut stored)?;
5037 if checksum(&stored) != self.hashes[block] {
5038 return Err(invalid("global dictionary payload checksum differs"));
5039 }
5040 let first = block * TEXT_PAYLOAD_VALUES;
5041 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
5042 let want = self.end_within(last - 1)? as usize;
5043 let values = string::decode_flat(&stored)?;
5044 if values.len() != last - first {
5045 return Err(invalid("global dictionary block holds the wrong value count"));
5046 }
5047 let bytes = values.into_bytes();
5048 if bytes.len() != want {
5049 return Err(invalid("global dictionary block decodes to the wrong length"));
5050 }
5051 Ok(bytes)
5052 }
5053
5054 fn loaned_block<'a>(
5063 &'a self,
5064 block: usize,
5065 decoded: &'a mut Vec<u8>,
5066 scattered: bool,
5067 ) -> Result<&'a [u8]> {
5068 let kept = self.blocks.get(block).and_then(OnceLock::get);
5069 if let Some(Ok(kept)) = kept {
5070 return Ok(kept);
5071 }
5072 let again = kept.is_none()
5073 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
5074 let keep = again
5075 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
5076 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
5077 if keep {
5078 let kept = self
5079 .payload_block(block)?
5080 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
5081 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
5082 return Ok(kept);
5083 }
5084 *decoded = self.decode_block(block)?;
5085 if scattered && again {
5086 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
5087 }
5088 Ok(decoded)
5089 }
5090
5091 fn ends_worth_unpacking(&self) -> usize {
5108 self.values.max(TEXT_PAYLOAD_VALUES)
5109 }
5110
5111 fn value_ends(&self) -> Option<&[u32]> {
5113 if let Some(built) = self.value_ends.get() {
5114 return built.as_deref();
5115 }
5116 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
5117 return None;
5118 }
5119 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
5120 }
5121
5122 fn unpack_ends(&self) -> Option<Vec<u32>> {
5128 let mut ends = vec![0u32; self.values];
5129 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
5130 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
5131 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
5132 u32::try_from(bits).unwrap_or(u32::MAX)
5133 })
5134 .ok()?;
5135 }
5136 if ends.contains(&u32::MAX) { None } else { Some(ends) }
5139 }
5140
5141 fn packed(&self) -> &[u8] {
5143 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
5144 }
5145
5146 fn end_within(&self, index: usize) -> Result<u32> {
5148 if let Some(ends) = self.value_ends() {
5149 return ends
5150 .get(index)
5151 .copied()
5152 .ok_or_else(|| invalid("global dictionary offsets are short"));
5153 }
5154 let run = index / TEXT_OFFSET_RUN;
5155 let bytes = self
5156 .packed()
5157 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5158 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5159 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
5160 .map_err(|_| invalid("global dictionary offsets are short"))?;
5161 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5162 }
5163
5164 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5182 let mut ends = vec![0u64; last.saturating_sub(first)];
5183 let mut scratch = Vec::new();
5184 let mut at = first;
5185 while at < last {
5186 let run = at / TEXT_OFFSET_RUN;
5187 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5188 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5189 let bytes = self
5190 .packed()
5191 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5192 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5193 let from = at % TEXT_OFFSET_RUN;
5194 let upto = stop - run * TEXT_OFFSET_RUN;
5195 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5196 return Err(invalid("global dictionary offsets are short"));
5197 }
5198 let into = &mut ends[at - first..stop - first];
5199 if from == 0 {
5200 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5201 .map_err(|_| invalid("global dictionary offsets are short"))?;
5202 } else {
5203 scratch.resize(held, 0);
5204 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5205 .map_err(|_| invalid("global dictionary offsets are short"))?;
5206 into.copy_from_slice(&scratch[from..upto]);
5207 }
5208 at = stop;
5209 }
5210 Ok(ends)
5211 }
5212
5213 fn start_within(&self, index: usize) -> Result<u32> {
5216 if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { Ok(0) } else { self.end_within(index - 1) }
5217 }
5218
5219 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5227 if let Some(ends) = self.value_ends() {
5228 let end =
5229 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5230 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5233 if start > end {
5234 return Err(invalid("global dictionary value ends before it starts"));
5235 }
5236 return Ok((start, end));
5237 }
5238 let within = index % TEXT_OFFSET_RUN;
5239 let (start, end) = if within == 0 {
5240 (self.start_within(index)?, self.end_within(index)?)
5241 } else {
5242 let run = index / TEXT_OFFSET_RUN;
5243 let bytes = self
5244 .packed()
5245 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5246 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5247 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5248 .map_err(|_| invalid("global dictionary offsets are short"))?;
5249 let ends = u32::try_from(end)
5250 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5251 let starts = u32::try_from(start)
5252 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5253 (starts, ends)
5254 };
5255 if start > end {
5256 return Err(invalid("global dictionary value ends before it starts"));
5257 }
5258 Ok((start, end))
5259 }
5260
5261 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5268 let slot = self
5269 .rank_blocks
5270 .get(rank / TEXT_RANK_BLOCK)
5271 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5272 let block = slot
5273 .get_or_init(|| {
5274 let mut bytes = Vec::new();
5275 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5276 Ok(bytes)
5277 })
5278 .as_ref()
5279 .map_err(Clone::clone)?;
5280 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5281 }
5282
5283 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5286 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5287 let end = self.rank_ends[which];
5288 bytes.clear();
5289 bytes.resize((end - start) as usize, 0);
5290 read_at(&self.file, self.rank_at + start, bytes)?;
5291 let expected = self
5292 .rank_hashes
5293 .get(which)
5294 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5295 if checksum(bytes) != *expected {
5296 return Err(invalid("global dictionary rank checksum differs"));
5297 }
5298 Ok(())
5299 }
5300
5301 fn head_at(&self, rank: usize) -> Result<u64> {
5303 let (block, within) = self.rank_parts(rank)?;
5304 let (base, width, packed) = rank_heads(block)?;
5305 let above = bitpack::tail_at(packed, width, within)
5306 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5307 Ok(base.wrapping_add(above))
5308 }
5309
5310 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5312 let (_, width, packed) = rank_heads(block)?;
5313 packed
5314 .get(bitpack::tail_len(count, width)..)
5315 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5316 }
5317
5318 fn rank_block_len(&self, rank: usize) -> usize {
5320 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5321 TEXT_RANK_BLOCK.min(self.ranks - first)
5322 }
5323}
5324
5325fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5327 let header = block
5328 .get(..RANK_BLOCK_HEADER)
5329 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5330 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5331 let width = header[8] as usize;
5332 if width > 64 {
5333 return Err(invalid("global dictionary rank block packs heads past a word"));
5334 }
5335 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5336}
5337
5338fn offset_width(ends: &[u32]) -> usize {
5345 let span = ends.iter().copied().max().unwrap_or(0);
5349 (u32::BITS - span.leading_zeros()) as usize
5350}
5351
5352fn offset_bytes(values: usize, bits: usize) -> usize {
5355 let full = values / TEXT_OFFSET_RUN;
5356 let rest = values % TEXT_OFFSET_RUN;
5357 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5358}
5359
5360fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5364 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5365 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5366 run.clear();
5367 run.extend(chunk.iter().map(|&end| u64::from(end)));
5368 bitpack::pack_tail(&run, bits, out)
5369 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5370 }
5371 Ok(())
5372}
5373
5374fn code_width(values: usize) -> usize {
5376 match u64::try_from(values).unwrap_or(u64::MAX) {
5377 0 | 1 => 0,
5378 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5379 }
5380}
5381
5382impl TextSource for NativeText {
5383 fn len(&self) -> usize {
5384 self.values
5385 }
5386
5387 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5388 let Some(grams) = &self.grams else { return Ok(true) };
5389 if literal.len() < 4 || first >= self.values {
5390 return Ok(true);
5391 }
5392 let verdict = grams.verdicts(&self.file, literal)?;
5393 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5394 }
5395
5396 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5397 if index >= self.values {
5398 return Ok(None);
5399 }
5400 let (start, end) = self.span_within(index)?;
5401 if start == end {
5402 return Ok(Some(&[]));
5403 }
5404 let block = index / TEXT_PAYLOAD_VALUES;
5407 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5408 Ok(bytes.get(start as usize..end as usize))
5409 }
5410
5411 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5412 if index >= self.values {
5413 return Ok(None);
5414 }
5415 let (start, end) = self.span_within(index)?;
5416 Ok(Some((end - start) as usize))
5417 }
5418
5419 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5426 into.reserve(indices.len());
5427 if let Some(Some(lens)) = self.value_lens.get() {
5430 lens.extend_at(indices, into);
5431 return Ok(());
5432 }
5433 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5434 let Some(ends) = self.value_ends() else {
5435 for &index in indices {
5436 into.push(
5437 self.bytes_len_at(index as usize)?
5438 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5439 );
5440 }
5441 return Ok(());
5442 };
5443 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5444 lens.extend_at(indices, into);
5445 return Ok(());
5446 }
5447 for &index in indices {
5448 let index = index as usize;
5449 let Some(&end) = ends.get(index) else {
5451 into.push(0);
5452 continue;
5453 };
5454 let start = if index.is_multiple_of(TEXT_PAYLOAD_VALUES) { 0 } else { ends[index - 1] };
5455 if start > end {
5456 return Err(invalid("global dictionary value ends before it starts"));
5457 }
5458 into.push(i64::from(end - start));
5459 }
5460 Ok(())
5461 }
5462
5463 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5466 into.reserve(indices.len());
5467 for &index in indices {
5468 let index = index as usize;
5469 if index >= self.values {
5471 into.push(0);
5472 continue;
5473 }
5474 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5475 let len = lens
5476 .get(index % TEXT_PAYLOAD_VALUES)
5477 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5478 into.push(i64::from(*len));
5479 }
5480 Ok(())
5481 }
5482
5483 fn sweep(
5496 &self,
5497 first: usize,
5498 limit: usize,
5499 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5500 ) -> Result<usize> {
5501 let limit = limit.min(self.values);
5502 if first >= limit {
5503 return Ok(first);
5504 }
5505 let block = first / TEXT_PAYLOAD_VALUES;
5506 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5507 let mut decoded = Vec::new();
5508 let bytes = self.loaned_block(block, &mut decoded, false)?;
5509 let ends = self.ends_within(first, last)?;
5510 if ends.len() != last - first {
5511 return Err(invalid("global dictionary offsets are short"));
5512 }
5513 let mut start = u64::from(self.start_within(first)?);
5514 for (index, &end) in (first..last).zip(&ends) {
5517 let value = usize::try_from(start)
5518 .ok()
5519 .zip(usize::try_from(end).ok())
5520 .and_then(|(from, to)| bytes.get(from..to))
5521 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5522 body(index, value)?;
5523 start = end;
5524 }
5525 Ok(last)
5526 }
5527
5528 fn visit_at(
5537 &self,
5538 indices: &[u32],
5539 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5540 ) -> Result<()> {
5541 let mut order = (0..indices.len()).collect::<Vec<_>>();
5542 order.sort_unstable_by_key(|&at| indices[at]);
5543 let block_of = |at: usize| {
5544 let index = indices[at] as usize;
5545 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5546 };
5547 let mut decoded = Vec::new();
5548 let mut run = 0;
5549 while run < order.len() {
5550 let Some(block) = block_of(order[run]) else {
5551 for &at in &order[run..] {
5553 body(at, &[])?;
5554 }
5555 break;
5556 };
5557 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5558 let bytes = self.loaned_block(block, &mut decoded, true)?;
5559 for &at in &order[run..upto] {
5560 let (start, end) = self.span_within(indices[at] as usize)?;
5561 let value = bytes
5562 .get(start as usize..end as usize)
5563 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5564 body(at, value)?;
5565 }
5566 run = upto;
5567 }
5568 Ok(())
5569 }
5570
5571 fn visit(
5577 &self,
5578 indices: &[usize],
5579 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5580 ) -> Result<()> {
5581 let mut at = 0;
5582 while at < indices.len() {
5583 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5584 let upto =
5585 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5586 let wanted = &indices[at..upto];
5587 if wanted.iter().any(|&index| index >= self.values) {
5588 return Err(invalid("a visited value is past the global dictionary"));
5589 }
5590 let decoded;
5591 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5592 Some(Ok(kept)) => kept,
5593 _ => {
5594 decoded = self.decode_block(block)?;
5595 &decoded
5596 }
5597 };
5598 for (offset, &index) in wanted.iter().enumerate() {
5599 let (start, end) = self.span_within(index)?;
5600 let value = bytes
5601 .get(start as usize..end as usize)
5602 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5603 body(at + offset, value)?;
5604 }
5605 at = upto;
5606 }
5607 Ok(())
5608 }
5609
5610 fn ranks(&self) -> Option<usize> {
5611 (self.ranks > 0).then_some(self.ranks)
5612 }
5613
5614 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5622 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5623 if let Some(&answer) = memo.get(wanted) {
5624 return Ok(answer);
5625 }
5626 let answer = search_below(self, ranks, wanted)?;
5627 if memo.len() >= TEXT_SEARCH_MEMO {
5628 memo.clear();
5629 }
5630 memo.insert(wanted.to_vec(), answer);
5631 Ok(answer)
5632 }
5633
5634 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5635 let settled = self.head_at(rank)?.cmp(&head(wanted));
5639 if settled != Ordering::Equal {
5640 return Ok(settled);
5641 }
5642 let code = self.code_at_rank(rank)?;
5643 let bytes = self
5644 .bytes_at(code as usize)?
5645 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5646 Ok(bytes.cmp(wanted))
5647 }
5648
5649 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5650 let (block, within) = self.rank_parts(rank)?;
5651 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5652 let code = bitpack::tail_at(codes, self.code_bits, within)
5653 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5654 let code = u32::try_from(code)
5655 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5656 if code as usize >= self.len() {
5657 return Err(invalid("global dictionary order names a code it does not have"));
5658 }
5659 Ok(code)
5660 }
5661
5662 fn code_ranks(&self) -> Option<&[u32]> {
5663 if self.ranks == 0 || self.ranks != self.len() {
5667 return None;
5668 }
5669 self.code_ranks
5670 .get_or_init(|| {
5671 let mut ranks = vec![u32::MAX; self.ranks];
5672 let mut scratch = Vec::new();
5680 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5681 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5682 let which = first / TEXT_RANK_BLOCK;
5683 let block = match self.rank_blocks.get(which)?.get() {
5684 Some(kept) => kept.as_ref().ok()?.as_slice(),
5685 None => {
5686 self.read_rank_block(which, &mut scratch).ok()?;
5687 scratch.as_slice()
5688 }
5689 };
5690 let count = self.rank_block_len(first);
5691 let packed = self.rank_codes(block, count).ok()?;
5692 let codes = codes.get_mut(..count)?;
5693 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5694 for (within, &code) in codes.iter().enumerate() {
5695 let code = usize::try_from(code).ok()?;
5696 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5697 }
5698 }
5699 if ranks.contains(&u32::MAX) {
5700 return None;
5701 }
5702 Some(ranks)
5703 })
5704 .as_deref()
5705 }
5706
5707 fn footprint(&self) -> usize {
5708 self.offsets.capacity()
5709 + self
5710 .value_ends
5711 .get()
5712 .and_then(Option::as_ref)
5713 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5714 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5715 + self
5716 .code_ranks
5717 .get()
5718 .and_then(Option::as_ref)
5719 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5720 + self.rank_hashes.capacity() * size_of::<u64>()
5721 + self.rank_ends.capacity() * size_of::<u64>()
5722 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5723 + self
5724 .rank_blocks
5725 .iter()
5726 .filter_map(OnceLock::get)
5727 .filter_map(|result| result.as_ref().ok())
5728 .map(Vec::capacity)
5729 .sum::<usize>()
5730 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5731 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5732 + self
5733 .char_lens
5734 .iter()
5735 .filter_map(OnceLock::get)
5736 .map(|lens| lens.len() * size_of::<u32>())
5737 .sum::<usize>()
5738 + self.hashes.capacity() * size_of::<u64>()
5739 + self.starts.capacity() * size_of::<u64>()
5740 + self.lengths.capacity() * size_of::<u64>()
5741 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5742 + self
5743 .blocks
5744 .iter()
5745 .filter_map(OnceLock::get)
5746 .filter_map(|result| result.as_ref().ok())
5747 .map(Vec::capacity)
5748 .sum::<usize>()
5749 }
5750}
5751
5752fn places(table: &Table) -> Result<Vec<Place>> {
5754 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5755 for (at, stripe) in table.stripes.iter().enumerate() {
5756 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5757 for (part, &rows) in stripe.parts.iter().enumerate() {
5758 places.push(Place {
5759 stripe: index,
5760 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5761 rows,
5762 });
5763 }
5764 }
5765 Ok(places)
5766}
5767
5768fn read_index<F: Positional + ?Sized>(
5773 file: &F,
5774 stripe: &Stripe,
5775 column: usize,
5776) -> Result<Vec<PartSpan>> {
5777 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5778 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5779}
5780
5781fn read_index_span<F: Positional + ?Sized>(
5782 file: &F,
5783 index: Span,
5784 page: Span,
5785 parts: usize,
5786 column: usize,
5787) -> Result<Vec<PartSpan>> {
5788 let section = index_section(parts)?;
5789 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5790 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5791 if end > index.length as usize {
5792 return Err(invalid("index page is shorter than its columns"));
5793 }
5794 let mut bytes = vec![0; section];
5795 let offset =
5796 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5797 read_at(file, offset, &mut bytes)?;
5798 let entries = section - size_of::<u64>();
5799 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5800 if checksum(&bytes[..entries]) != stored {
5801 return Err(invalid(&format!(
5804 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5805 wanted {stored:016x} and got {:016x}",
5806 checksum(&bytes[..entries]),
5807 )));
5808 }
5809 let mut spans = Vec::with_capacity(parts);
5810 let mut start = 0_usize;
5811 for part in 0..parts {
5812 let at = part * INDEX_ENTRY;
5813 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5814 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5815 spans.push(PartSpan { start, length, hash });
5816 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5817 }
5818 if start != page.length as usize {
5819 return Err(invalid("column page length differs from its index"));
5820 }
5821 Ok(spans)
5822}
5823
5824fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5826 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5827 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5828}
5829
5830fn touch(bits: &mut Vec<u64>, part: usize, parts: usize) -> (bool, bool) {
5833 if bits.is_empty() {
5834 bits.resize(parts.div_ceil(64).max(1), 0);
5835 }
5836 let (word, bit) = (part / 64, 1_u64 << (part % 64));
5837 let Some(held) = bits.get_mut(word) else { return (false, false) };
5838 let again = *held & bit != 0;
5839 *held |= bit;
5840 let through = bits.iter().map(|word| word.count_ones() as usize).sum::<usize>() >= parts;
5841 (again, again && through)
5842}
5843
5844fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5850 if let Some(slot) = cached.index.get_mut(held.stripe)
5851 && slot.is_none()
5852 {
5853 *slot = Some(Arc::clone(&held.index));
5854 }
5855 let page = held.page.clone()?;
5856 let slot = cached.pages.get_mut(held.stripe)?;
5857 if slot.is_some() {
5858 return None;
5859 }
5860 let bytes = page.bytes.len();
5861 let used = Arc::new(AtomicBool::new(true));
5864 *slot = Some(Resident { page, used: Arc::clone(&used) });
5865 Some((bytes, used))
5866}
5867
5868#[derive(Debug, Clone)]
5877pub struct Catalog {
5878 file: Arc<File>,
5879 size: u64,
5880 entries: Arc<Vec<Entry>>,
5881 views: Arc<Vec<ViewEntry>>,
5883 opening: Opening,
5884 pool: PagePool,
5886}
5887
5888#[derive(Debug, Clone, PartialEq, Eq)]
5890pub struct CertifiedSums {
5891 pub columns: Vec<(i128, u64)>,
5892 pub rows: u64,
5893}
5894
5895#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5897pub enum IntegerExtremes {
5898 Null,
5899 Values { low: i128, high: i128 },
5900}
5901
5902pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5904
5905impl Catalog {
5906 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5915 Self::open_in(path, &PagePool::default())
5916 }
5917
5918 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5924 let path = path.as_ref();
5925 let (file, size, _, bytes, opening) = slot_bytes(path)?;
5926 let (entries, views, card) = decode_catalog(&bytes, size)?;
5927 remember_card(path, card.as_ref());
5928 Ok(Self {
5929 file: Arc::new(file),
5930 size,
5931 entries: Arc::new(entries),
5932 views: Arc::new(views),
5933 opening,
5934 pool: pool.clone(),
5935 })
5936 }
5937
5938 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5940 self.entries.iter().map(|entry| entry.name.as_str())
5941 }
5942
5943 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5950 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5951 }
5952
5953 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5959 self.views.iter()
5960 }
5961
5962 #[must_use]
5964 pub fn len(&self) -> usize {
5965 self.entries.len()
5966 }
5967
5968 #[must_use]
5971 pub fn is_empty(&self) -> bool {
5972 self.entries.is_empty()
5973 }
5974
5975 pub fn table(&self, name: &str) -> Result<Reader> {
5981 let entry = self
5982 .entries
5983 .iter()
5984 .find(|entry| entry.name == name)
5985 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5986 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5990 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5991 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5992 }
5993 let mut opening = self.opening;
5994 opening.reads += 1;
5995 opening.bytes += u64::from(entry.directory.length);
5996 Reader::build(
5997 Arc::clone(&self.file),
5998 self.size,
5999 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
6000 u64::from(entry.directory.length),
6001 opening,
6002 self.pool.clone(),
6003 )
6004 }
6005
6006 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6014 let mut counts = BTreeMap::<i64, u64>::new();
6015 let Some(()) = self.integer_fold(name, column, |value, count| {
6016 let held = counts.entry(value).or_default();
6017 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
6018 Ok(())
6019 })?
6020 else {
6021 return Ok(None);
6022 };
6023 Ok(Some(counts.into_iter().collect()))
6024 }
6025
6026 pub fn integer_fold(
6033 &self,
6034 name: &str,
6035 column: usize,
6036 mut emit: impl FnMut(i64, u64) -> Result<()>,
6037 ) -> Result<Option<()>> {
6038 let entry = self
6039 .entries
6040 .iter()
6041 .find(|entry| entry.name == name)
6042 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6043 let field =
6044 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
6045 if !signed_integer(&field.ty) {
6046 return Ok(None);
6047 }
6048 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6049 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6050 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6051 }
6052 quick_integer_fold(
6053 &self.file,
6054 Cursor::over(&self.file, offset, length),
6055 entry,
6056 self.size,
6057 column,
6058 &mut emit,
6059 )?;
6060 Ok(Some(()))
6061 }
6062
6063 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6068 let entry = self
6069 .entries
6070 .iter()
6071 .find(|entry| entry.name == name)
6072 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6073 let Some(field) = entry.fields.get(column) else {
6074 return Err(invalid("frequency column index out of range"));
6075 };
6076 if !matches!(
6077 field.ty,
6078 LogicalType::TinyInt
6079 | LogicalType::SmallInt
6080 | LogicalType::Integer
6081 | LogicalType::BigInt
6082 | LogicalType::UTinyInt
6083 | LogicalType::USmallInt
6084 | LogicalType::UInteger
6085 | LogicalType::UBigInt
6086 ) {
6087 return Ok(None);
6088 }
6089 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6090 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6091 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6092 }
6093 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
6094 return frequencies
6095 .iter()
6096 .filter(|(value, _)| value.is_some_and(|value| value != 0))
6097 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
6098 .map(Some)
6099 .ok_or_else(|| invalid("numeric frequency count overflow"));
6100 }
6101 quick_nonzero(
6102 Cursor::over(&self.file, offset, length),
6103 &entry.name,
6104 &entry.fields,
6105 entry.rows,
6106 column,
6107 )
6108 }
6109
6110 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
6113 let entry = self
6114 .entries
6115 .iter()
6116 .find(|entry| entry.name == name)
6117 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6118 let mut sums = Vec::with_capacity(columns.len());
6119 for &column in columns {
6120 let Some(field) = entry.fields.get(column) else {
6121 return Err(invalid("aggregate column index out of range"));
6122 };
6123 if !signed_integer(&field.ty) {
6124 return Ok(None);
6125 }
6126 let Some(sum) = entry.aggregates[column] else {
6127 return Ok(None);
6128 };
6129 sums.push(sum);
6130 }
6131 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6132 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6133 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6134 }
6135 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
6136 }
6137
6138 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
6140 let entry = self
6141 .entries
6142 .iter()
6143 .find(|entry| entry.name == name)
6144 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6145 let Some(count) = entry.distincts.get(column).copied() else {
6146 return Err(invalid("distinct column index out of range"));
6147 };
6148 let Some(count) = count else { return Ok(None) };
6149 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6150 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6151 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6152 }
6153 Ok(Some(count))
6154 }
6155
6156 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
6158 let entry = self
6159 .entries
6160 .iter()
6161 .find(|entry| entry.name == name)
6162 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6163 let Some(extremes) = entry.extremes.get(column).copied() else {
6164 return Err(invalid("extremes column index out of range"));
6165 };
6166 let Some(extremes) = extremes else { return Ok(None) };
6167 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6168 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6169 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6170 }
6171 Ok(Some(match extremes {
6172 None => IntegerExtremes::Null,
6173 Some((low, high)) => IntegerExtremes::Values { low, high },
6174 }))
6175 }
6176
6177 pub fn exact_numeric_frequencies(
6179 &self,
6180 name: &str,
6181 column: usize,
6182 ) -> Result<Option<NumericFrequencies>> {
6183 let entry = self
6184 .entries
6185 .iter()
6186 .find(|entry| entry.name == name)
6187 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6188 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6189 return Err(invalid("numeric frequency column index out of range"));
6190 };
6191 let Some(frequencies) = frequencies else { return Ok(None) };
6192 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6193 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6194 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6195 }
6196 Ok(Some(frequencies))
6197 }
6198
6199 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6201 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6202 }
6203}
6204
6205fn slot_offset(generation: u64) -> u64 {
6210 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6211}
6212
6213fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6218 let file = File::open(path).map_err(io)?;
6219 let size = file.metadata().map_err(io)?.len();
6220 let (slot, bytes, opening) = committed_slot(&file, size)?;
6221 Ok((file, size, slot, bytes, opening))
6222}
6223
6224fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6230 if size < HEADER {
6231 return Err(invalid("file is shorter than its header"));
6232 }
6233 let mut header = [0; HEADER as usize];
6234 read_at(file, 0, &mut header)?;
6235 let mut opening = Opening { reads: 1, bytes: HEADER };
6236 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6237 if &header[..8] != MAGIC {
6242 return Err(invalid("the header does not begin with a rudb native magic"));
6243 }
6244 if !READABLE.contains(&version) {
6245 return Err(invalid(&format!(
6246 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6247 be written again"
6248 )));
6249 }
6250 let mut selected = None;
6251 for start in [16, 16 + SLOT_BYTES] {
6252 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6253 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6254 continue;
6255 }
6256 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6257 if slot.offset < HEADER || end > size {
6258 continue;
6259 }
6260 let mut bytes = vec![0; slot.length as usize];
6261 read_at(file, slot.offset, &mut bytes)?;
6262 opening.reads += 1;
6263 opening.bytes += u64::from(slot.length);
6264 if checksum(&bytes) == slot.hash
6265 && selected
6266 .as_ref()
6267 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6268 {
6269 selected = Some((slot, bytes));
6270 }
6271 }
6272 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6273 Ok((slot, bytes, opening))
6274}
6275
6276impl Reader {
6277 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6284 let catalog = Catalog::open(path)?;
6285 let mut names = catalog.names();
6286 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6287 if names.next().is_some() {
6288 return Err(invalid(
6289 "the file holds more than one table, so it has to be opened by name",
6290 ));
6291 }
6292 catalog.table(&name)
6293 }
6294
6295 fn build(
6297 file: Arc<File>,
6298 size: u64,
6299 table: Table,
6300 directory: u64,
6301 opening: Opening,
6302 pool: PagePool,
6303 ) -> Result<Self> {
6304 let places = places(&table)?;
6305 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6306 let table_fields = table.fields.len();
6307 let stripes = table.stripes.len();
6308 let columns = (0..table.fields.len())
6309 .map(|_| {
6310 Mutex::new(Cached {
6311 pages: (0..stripes).map(|_| None).collect(),
6312 index: (0..stripes).map(|_| None).collect(),
6313 touched: vec![Vec::new(); stripes],
6314 ..Cached::default()
6315 })
6316 })
6317 .collect::<Vec<_>>();
6318 let cache = Shelf {
6319 columns,
6320 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6321 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6322 };
6323 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6324 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6325 .collect();
6326 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6327 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6328 .collect();
6329 let verified = (places.len() * table_fields).div_ceil(64);
6330 let firsts = places
6331 .iter()
6332 .scan(0, |first, place| {
6333 let at = *first;
6334 *first += place.rows as usize;
6335 Some(at)
6336 })
6337 .collect();
6338 Ok(Self {
6339 file,
6340 table: Arc::new(table),
6341 dictionaries: Arc::new(dictionaries),
6342 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6343 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6344 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6345 summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6346 opened: Arc::new(AtomicUsize::new(0)),
6347 sieves: Arc::new(sieves),
6348 part_ranges: Arc::new(part_ranges),
6349 places: Arc::new(places),
6350 cache: Arc::new(cache),
6351 pool,
6352 pages: Arc::new(AtomicUsize::new(0)),
6353 indexes: Arc::new(AtomicUsize::new(0)),
6354 verified: Arc::new((0..verified).map(|_| AtomicU64::new(0)).collect()),
6355 text_grams: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6356 firsts: Arc::new(firsts),
6357 size,
6358 directory,
6359 opening,
6360 })
6361 }
6362
6363 #[must_use]
6370 pub fn reads(&self) -> Reads {
6371 Reads {
6372 opening: self.opening,
6373 pages: self.pages.load(Atomic::Relaxed),
6374 indexes: self.indexes.load(Atomic::Relaxed),
6375 dictionaries: self.opened.load(Atomic::Relaxed),
6376 }
6377 }
6378
6379 #[must_use]
6384 pub fn layout(&self) -> Layout {
6385 let table = &self.table;
6386 let stripes = table.stripes.as_slice();
6387 let columns = table
6388 .fields
6389 .iter()
6390 .enumerate()
6391 .map(|(at, field)| ColumnLayout {
6392 name: field.name.clone(),
6393 kind: field.ty.to_string(),
6394 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6395 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6396 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6397 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6398 dictionary: dictionary_bytes(table, at),
6399 })
6400 .collect();
6401 Layout {
6402 file: self.size,
6403 rows: table.rows,
6404 stripes: stripes.len(),
6405 parts: self.places.len(),
6406 columns,
6407 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6408 directory: self.directory,
6409 header: HEADER,
6410 }
6411 }
6412
6413 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6430 let field = self
6431 .table
6432 .fields
6433 .get(column)
6434 .ok_or_else(|| invalid("stored column index out of range"))?;
6435 let mut stored = Vec::with_capacity(self.places.len());
6436 let mut row = 0;
6437 for (at, stripe) in self.table.stripes.iter().enumerate() {
6438 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6439 let index = read_index(&self.file, stripe, column)?;
6440 let mut bytes = vec![0; page.length as usize];
6441 read_at(&self.file, page.offset, &mut bytes)?;
6442 let ranges = self.stripe_part_ranges(at, column);
6443 for (part, &rows) in stripe.parts.iter().enumerate() {
6444 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6445 let held = part_bytes(&bytes, span)?;
6446 let range = ranges.and_then(|held| held.get(part));
6447 stored.push(StoredPart {
6448 stripe: at,
6449 part,
6450 row,
6451 rows: rows as usize,
6452 encoding: page_encoding(&field.ty, rows as usize, held),
6453 bytes: span.length as u64,
6454 page: page.offset,
6455 offset: span.start as u64,
6456 low: range
6457 .and_then(|range| range.low.clone())
6458 .and_then(|bound| bound.into_value(&field.ty)),
6459 high: range
6460 .and_then(|range| range.high.clone())
6461 .and_then(|bound| bound.into_value(&field.ty)),
6462 nulls: range.map(|range| range.nulls),
6463 });
6464 row += rows as usize;
6465 }
6466 }
6467 Ok(stored)
6468 }
6469
6470 #[must_use]
6472 pub fn parts(&self) -> usize {
6473 self.places.len()
6474 }
6475
6476 #[must_use]
6483 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6484 let mut runs = Vec::with_capacity(self.table.stripes.len());
6485 let mut start = 0;
6486 for stripe in &self.table.stripes {
6487 let end = start + stripe.parts.len();
6488 runs.push(start..end);
6489 start = end;
6490 }
6491 runs
6492 }
6493
6494 #[must_use]
6499 pub fn stripe_rows(&self, stripe: usize) -> usize {
6500 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6501 }
6502
6503 pub fn keep_stripes(&self, stripes: usize) {
6510 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6511 }
6512
6513 #[must_use]
6515 pub fn part_rows(&self, at: usize) -> usize {
6516 self.places.get(at).map_or(0, |place| place.rows as usize)
6517 }
6518
6519 #[must_use]
6521 pub fn table(&self) -> &Table {
6522 &self.table
6523 }
6524
6525 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6534 let field = self
6535 .table
6536 .fields
6537 .get(column)
6538 .ok_or_else(|| invalid("frequency column index out of range"))?;
6539 let Some(summary) = self.frequency_summary(column)? else {
6540 return Ok(None);
6541 };
6542 if top == 0 || summary.entries.len() < top {
6543 return Ok(None);
6544 }
6545 let boundary = summary.entries[top - 1].count;
6546 if boundary <= summary.omitted_max {
6547 return Ok(None);
6548 }
6549 self.decode_frequencies(column, &field.ty, &summary.entries)
6550 .map(|values| Some(Vec::clone(&values)))
6551 }
6552
6553 pub fn top_pair_frequencies(
6561 &self,
6562 first: usize,
6563 second: usize,
6564 _top: usize,
6565 ) -> Result<Option<PairFrequencyCounts>> {
6566 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6567 return Err(invalid("pair frequency column index out of range"));
6568 }
6569 Ok(None)
6570 }
6571
6572 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6592 let Some(prefix) = self.frequency_prefix(column)? else {
6593 return Ok(None);
6594 };
6595 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6596 }
6597
6598 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6621 Ok(self.held_prefix(column)?.map(|(entries, omitted_max)| FrequencyPrefix {
6622 entries: Vec::clone(&entries),
6623 omitted_max,
6624 }))
6625 }
6626
6627 pub(crate) fn held_prefix(&self, column: usize) -> Result<Option<(Synopsis, u64)>> {
6633 let field = self
6634 .table
6635 .fields
6636 .get(column)
6637 .ok_or_else(|| invalid("frequency column index out of range"))?;
6638 let Some(summary) = self.frequency_summary(column)? else {
6639 return Ok(None);
6640 };
6641 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6642 Ok(Some((entries, summary.omitted_max)))
6643 }
6644
6645 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6647 Ok(match self.table.frequencies.get(column) {
6648 None | Some(None) => None,
6649 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6650 Some(Some(Frequencies::Stored { span, values, entries })) => {
6651 let slot = self
6652 .frequency_summaries
6653 .get(column)
6654 .ok_or_else(|| invalid("frequency column index out of range"))?;
6655 if let Some(summary) = slot.get() {
6656 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6657 }
6658 let field = self
6659 .table
6660 .fields
6661 .get(column)
6662 .ok_or_else(|| invalid("frequency column index out of range"))?;
6663 let mut bytes = vec![0; span.length as usize];
6664 read_at(&self.file, span.offset, &mut bytes)?;
6665 let mut cur = Cursor::new(&bytes);
6666 let summary = decode_summary(&mut cur, field, self.table.rows, *values)?;
6667 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6668 if !cur.done() || summary.entries.len() != *entries {
6669 return Err(invalid("a stored synopsis differs from its directory span"));
6670 }
6671 let _ = slot.set(Arc::new(summary));
6672 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6673 }
6674 })
6675 }
6676
6677 fn decode_frequencies(
6685 &self,
6686 column: usize,
6687 ty: &LogicalType,
6688 entries: &[FrequencyEntry],
6689 ) -> Result<Synopsis> {
6690 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6691 return Ok(Arc::clone(values));
6692 }
6693 let values = Arc::new(self.decode_frequencies_once(column, ty, entries)?);
6694 if let Some(slot) = self.frequency_values.get(column) {
6695 let _ = slot.set(Arc::clone(&values));
6696 }
6697 Ok(values)
6698 }
6699
6700 fn decode_frequencies_once(
6701 &self,
6702 column: usize,
6703 ty: &LogicalType,
6704 entries: &[FrequencyEntry],
6705 ) -> Result<Vec<(Value, u64)>> {
6706 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6707 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6708 return Err(invalid("frequency text count differs from its synopsis"));
6709 }
6710 let dictionary =
6711 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6712 let mut codes = entries
6713 .iter()
6714 .filter_map(|entry| match entry.value {
6715 FrequencyValue::Code(code) => Some(code as usize),
6716 _ => None,
6717 })
6718 .collect::<Vec<_>>();
6719 codes.sort_unstable();
6720 codes.dedup();
6721 let texts = match &dictionary {
6722 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6723 _ => Vec::new(),
6724 };
6725 let mut out = Vec::with_capacity(entries.len());
6726 for (entry_at, entry) in entries.iter().enumerate() {
6727 let value = match entry.value {
6728 FrequencyValue::Null => {
6729 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6730 return Err(invalid("a null frequency entry has text"));
6731 }
6732 Value::Null
6733 }
6734 FrequencyValue::Integer(value) => match *ty {
6735 LogicalType::TinyInt => Value::TinyInt(
6736 i8::try_from(value)
6737 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6738 ),
6739 LogicalType::UTinyInt => Value::UTinyInt(
6740 u8::try_from(value)
6741 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6742 ),
6743 LogicalType::USmallInt => Value::USmallInt(
6744 u16::try_from(value)
6745 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6746 ),
6747 LogicalType::UInteger => Value::UInteger(
6748 u32::try_from(value)
6749 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6750 ),
6751 LogicalType::UBigInt => Value::UBigInt(
6752 u64::try_from(value)
6753 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6754 ),
6755 LogicalType::SmallInt => Value::SmallInt(
6756 i16::try_from(value)
6757 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6758 ),
6759 LogicalType::Integer => Value::Integer(
6760 i32::try_from(value)
6761 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6762 ),
6763 LogicalType::BigInt => Value::BigInt(
6764 i64::try_from(value)
6765 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6766 ),
6767 LogicalType::Date => Value::Date(
6768 i32::try_from(value)
6769 .map_err(|_| invalid("frequency DATE is out of range"))?,
6770 ),
6771 LogicalType::Timestamp => Value::Timestamp(
6772 i64::try_from(value)
6773 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6774 ),
6775 _ => return Err(invalid("integer frequency belongs to another type")),
6776 },
6777 FrequencyValue::Code(code) => {
6778 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6779 if *ty == LogicalType::Blob {
6780 Value::Blob(text.clone())
6781 } else {
6782 Value::Varchar(
6783 String::from_utf8(text.clone())
6784 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6785 )
6786 }
6787 } else {
6788 if dictionary.is_none() {
6789 return Err(invalid("frequency code has no dictionary or stored text"));
6790 }
6791 let at = codes
6792 .binary_search(&(code as usize))
6793 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6794 texts[at].clone()
6795 }
6796 }
6797 };
6798 out.push((value, entry.count));
6799 }
6800 Ok(out)
6801 }
6802
6803 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6813 let field = self
6814 .table
6815 .fields
6816 .get(column)
6817 .ok_or_else(|| invalid("frequency column index out of range"))?;
6818 let Some(summary) = self.frequency_summary(column)? else {
6819 return Ok(None);
6820 };
6821 if summary.ordinals.is_empty() {
6822 return Ok(None);
6823 }
6824 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6825 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6826 (
6827 entries.iter().map(|(value, _)| value.clone()).collect(),
6828 summary.ordinal_entries.clone(),
6829 )
6830 } else {
6831 (Vec::new(), Vec::new())
6832 };
6833 Ok(Some(FrequencyOccurrences {
6834 omitted_max: summary.omitted_max,
6835 ordinals: summary.ordinals.clone(),
6836 anchors,
6837 anchor_indices,
6838 }))
6839 }
6840
6841 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6867 self.table
6868 .distincts
6869 .get(column)
6870 .copied()
6871 .ok_or_else(|| invalid("distinct column index out of range"))
6872 }
6873
6874 pub fn null_count(&self, column: usize) -> Result<u64> {
6885 if column >= self.table.fields.len() {
6886 return Err(invalid("null count column index out of range"));
6887 }
6888 let mut nulls = 0_u64;
6889 for stripe in &self.table.stripes {
6890 let range = stripe
6891 .zone
6892 .column(column)
6893 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6894 nulls = nulls
6895 .checked_add(range.nulls as u64)
6896 .ok_or_else(|| invalid("null count overflow"))?;
6897 }
6898 Ok(nulls)
6899 }
6900
6901 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6916 if self.null_count(column)? > 0 || self.demoted(column) {
6917 return Ok(None);
6918 }
6919 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6920 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6921 if ranks == 0 {
6922 return Ok(None);
6923 }
6924 let low = text_at_rank(&dictionary, 0)?;
6925 let high = text_at_rank(&dictionary, ranks - 1)?;
6926 Ok(Some((low, high)))
6927 }
6928
6929 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6952 if column >= self.table.fields.len() {
6953 return Err(invalid("extremes column index out of range"));
6954 }
6955 let mut low: Option<Bound> = None;
6956 let mut high: Option<Bound> = None;
6957 for stripe in &self.table.stripes {
6958 let range = stripe
6959 .zone
6960 .column(column)
6961 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6962 if !range.exact {
6963 return Ok(None);
6964 }
6965 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6970 if stripe.rows > range.nulls {
6971 return Ok(None);
6972 }
6973 continue;
6974 };
6975 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6976 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6977 }
6978 Ok(low.zip(high))
6979 }
6980
6981 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6994 if column >= self.table.fields.len() {
6995 return Err(invalid("sum column index out of range"));
6996 }
6997 let mut total = 0_i128;
6998 let mut rows = 0_u64;
6999 for stripe in &self.table.stripes {
7000 let range = stripe
7001 .zone
7002 .column(column)
7003 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
7004 let Some(part) = range.sum else { return Ok(None) };
7005 let Some(sum) = total.checked_add(part) else { return Ok(None) };
7006 total = sum;
7007 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
7008 }
7009 Ok(Some((total, rows)))
7010 }
7011
7012 pub fn host_groups(
7014 &self,
7015 column: usize,
7016 _minimum_count: u64,
7017 ) -> Result<Option<Vec<host::HostEntry>>> {
7018 if column >= self.table.fields.len() {
7019 return Err(invalid("host group column index out of range"));
7020 }
7021 Ok(None)
7022 }
7023
7024 #[must_use]
7028 pub fn demoted(&self, column: usize) -> bool {
7029 self.table.demoted.get(column).copied().unwrap_or(false)
7030 }
7031
7032 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
7041 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
7042 if let Some(dictionary) = self.dictionaries[column].get() {
7043 return Ok(Some(Arc::clone(dictionary)));
7044 }
7045 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
7046 if let Some(dictionary) = self.dictionaries[column].get() {
7047 return Ok(Some(Arc::clone(dictionary)));
7048 }
7049 self.opened.fetch_add(1, Atomic::Relaxed);
7050 let dictionary = Arc::new(open_global_dictionary(
7051 Arc::clone(&self.file),
7052 page,
7053 &self.table.fields[column].ty,
7054 TEXT_KEEP_BUDGET,
7055 )?);
7056 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
7057 Ok(Some(dictionary))
7058 }
7059
7060 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
7067 if of.extent_bytes == 0 {
7068 return Ok(Vec::new());
7069 }
7070 let mut bytes = vec![0; of.extent_bytes as usize];
7071 read_at(&self.file, of.extent_page, &mut bytes)?;
7072 if checksum(&bytes) != of.hash {
7073 return Err(invalid("a section's extent table does not checksum"));
7074 }
7075 let extents = section::decode_extents(&bytes)?;
7076 if extents.len() != of.extents as usize {
7077 return Err(invalid("a section's extent table is not the length the entry says"));
7078 }
7079 Ok(extents)
7080 }
7081
7082 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
7092 let mut bytes = Vec::new();
7093 self.extent_into(of, &mut bytes)?;
7094 Ok(bytes)
7095 }
7096
7097 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
7099 let end = of
7100 .offset
7101 .checked_add(u64::from(of.length))
7102 .ok_or_else(|| invalid("an extent overflows the file"))?;
7103 if of.offset < HEADER || end > self.size {
7104 return Err(invalid("an extent is outside the file"));
7105 }
7106 bytes.resize(of.length as usize, 0);
7107 read_at(&self.file, of.offset, bytes)?;
7108 if checksum(bytes) != of.hash {
7109 return Err(invalid("an extent does not checksum"));
7110 }
7111 Ok(())
7112 }
7113
7114 pub fn payload_head(&self, of: &Section, len: usize) -> Result<Vec<u8>> {
7127 let extents = self.extents(of)?;
7128 let Some(first) = extents.first() else { return Ok(Vec::new()) };
7129 let end = first
7130 .offset
7131 .checked_add(u64::from(first.length))
7132 .ok_or_else(|| invalid("an extent overflows the file"))?;
7133 if first.offset < HEADER || end > self.size {
7134 return Err(invalid("an extent is outside the file"));
7135 }
7136 let mut bytes = vec![0; len.min(first.length as usize)];
7137 read_at(&self.file, first.offset, &mut bytes)?;
7138 Ok(bytes)
7139 }
7140
7141 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
7150 let extents = self.extents(of)?;
7151 let mut bytes =
7152 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
7153 for one in &extents {
7154 if one.first != bytes.len() as u64 {
7155 return Err(invalid("a section's extents do not join up"));
7156 }
7157 bytes.extend_from_slice(&self.extent(one)?);
7158 }
7159 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
7162 return Err(invalid("a section's header is longer than its payload"));
7163 }
7164 Ok(bytes)
7165 }
7166
7167 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7178 self.read_impl(part, columns, true, None)
7179 }
7180
7181 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
7191 self.read_impl(part, columns, false, None)
7192 }
7193
7194 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
7202 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7203 let field =
7204 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7205 if !matches!(
7206 field.ty,
7207 LogicalType::TinyInt
7208 | LogicalType::SmallInt
7209 | LogicalType::Integer
7210 | LogicalType::BigInt
7211 ) {
7212 return Ok(None);
7213 }
7214 let (rows, counts) = match self.with_part(place, column, |bytes| {
7215 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
7216 return Ok(None);
7217 }
7218 integer::tally(&bytes[2..]).map(Some)
7219 })? {
7220 Some(tallied) => tallied,
7221 None => return Ok(None),
7222 };
7223 if rows != place.rows as usize {
7224 return Err(invalid("encoded integer part holds the wrong number of rows"));
7225 }
7226 for &(value, _) in &counts {
7227 let fits = match field.ty {
7228 LogicalType::TinyInt => i8::try_from(value).is_ok(),
7229 LogicalType::SmallInt => i16::try_from(value).is_ok(),
7230 LogicalType::Integer => i32::try_from(value).is_ok(),
7231 LogicalType::BigInt => true,
7232 _ => false,
7233 };
7234 if !fits {
7235 return Err(invalid("encoded integer value is outside its column type"));
7236 }
7237 }
7238 Ok(Some(counts))
7239 }
7240
7241 pub fn rows_holding(
7253 &self,
7254 part: usize,
7255 column: usize,
7256 sequence: &Sequence,
7257 negated: bool,
7258 ) -> Result<Option<Vec<u32>>> {
7259 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7260 let field =
7261 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7262 if field.ty != LogicalType::Varchar {
7263 return Ok(None);
7264 }
7265 let rows = place.rows as usize;
7266 self.with_part(place, column, |bytes| {
7267 if bytes.first() != Some(&6) {
7268 return Ok(None);
7269 }
7270 let mut cur = Cursor::new(bytes);
7271 cur.u8()?;
7272 let mask = match cur.u8()? {
7273 0 => None,
7274 1 => return Ok(Some(Vec::new())),
7275 2 => {
7276 let from = cur.at;
7277 cur.take(rows.div_ceil(8))?;
7278 Some(&bytes[from..cur.at])
7279 }
7280 _ => return Err(invalid("page validity tag differs")),
7281 };
7282 let needs = sequence.needs();
7285 let first = self.firsts.get(part).copied().unwrap_or_default();
7286 let sketch = self
7287 .text_grams
7288 .get(column)
7289 .and_then(|slot| slot.get_or_init(|| grams::text_grams(self, column)).as_deref())
7290 .and_then(|words| words.get(first..first + rows));
7291 let maybe = |row: usize| sketch.is_none_or(|words| words[row] & needs == needs);
7292 let Some(held) = string::holds_in_where(&bytes[cur.at..], sequence, maybe)? else {
7293 return Ok(None);
7294 };
7295 if held.len() != rows {
7296 return Err(invalid("compressed text page holds the wrong number of rows"));
7297 }
7298 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7299 Ok(Some(
7300 (0..rows)
7301 .filter(|&row| held[row] != negated && valid(row))
7302 .map(|row| row as u32)
7303 .collect(),
7304 ))
7305 })
7306 }
7307
7308 fn is_verified(&self, bit: usize) -> bool {
7311 self.verified
7312 .get(bit / 64)
7313 .is_some_and(|word| word.load(Atomic::Relaxed) >> (bit % 64) & 1 == 1)
7314 }
7315
7316 fn set_verified(&self, bit: usize) {
7318 if let Some(word) = self.verified.get(bit / 64) {
7319 word.fetch_or(1 << (bit % 64), Atomic::Relaxed);
7320 }
7321 }
7322
7323 fn with_part<T>(
7326 &self,
7327 place: Place,
7328 column: usize,
7329 read: impl FnOnce(&[u8]) -> Result<T>,
7330 ) -> Result<T> {
7331 let stripe_index = place.stripe as usize;
7332 let stripe = self
7333 .table
7334 .stripes
7335 .get(stripe_index)
7336 .ok_or_else(|| invalid("stripe index out of range"))?;
7337 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7338 let held = self.held(stripe_index, place.part as usize, stripe, column, true)?;
7339 let span = *held
7340 .index
7341 .get(place.part as usize)
7342 .ok_or_else(|| invalid("part index out of range"))?;
7343 match &held.page {
7344 Some(page) => read(page.part(place.part as usize, span)?),
7345 None => {
7346 let offset = page
7347 .offset
7348 .checked_add(span.start as u64)
7349 .ok_or_else(|| invalid("part range overflow"))?;
7350 let mut bytes = vec![0; span.length];
7351 read_at(&self.file, offset, &mut bytes)?;
7352 verify_part(&bytes, span)?;
7353 read(&bytes)
7354 }
7355 }
7356 }
7357
7358 pub fn read_rows(
7371 &self,
7372 part: usize,
7373 columns: &[usize],
7374 positions: &[u32],
7375 whole: bool,
7376 ) -> Result<Chunk> {
7377 self.read_impl(part, columns, whole, Some(positions))
7378 }
7379
7380 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7387 if self.demoted(column) {
7390 return Ok(false);
7391 }
7392 if candidates.is_empty() {
7393 return Ok(true);
7394 }
7395 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7396 return Err(Error::internal("native code candidates are not sorted and unique"));
7397 }
7398 let stripe = self.stripe_of(part)?;
7399 let Some(page) = stripe.memberships.get(column) else {
7400 return Ok(false);
7401 };
7402 let mut bytes = vec![0; page.length as usize];
7403 read_at(&self.file, page.offset, &mut bytes)?;
7404 if checksum(&bytes) != page.hash {
7405 return Err(invalid("membership page checksum differs"));
7406 }
7407 let codes = decode_membership(&bytes)?;
7408 let mut left = 0;
7409 let mut right = 0;
7410 while left < codes.len() && right < candidates.len() {
7411 match codes[left].cmp(&candidates[right]) {
7412 Ordering::Less => left += 1,
7413 Ordering::Greater => right += 1,
7414 Ordering::Equal => return Ok(false),
7415 }
7416 }
7417 Ok(true)
7418 }
7419
7420 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7421 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7422 self.table
7423 .stripes
7424 .get(place.stripe as usize)
7425 .ok_or_else(|| invalid("stripe index out of range"))
7426 }
7427
7428 fn held(
7445 &self,
7446 at: usize,
7447 part: usize,
7448 stripe: &Stripe,
7449 column: usize,
7450 whole: bool,
7451 ) -> Result<CachedColumn> {
7452 let cache =
7453 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7454 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7455 let known = cached.index.get(at).and_then(Clone::clone);
7456 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7457 slot.used.store(true, Atomic::Relaxed);
7458 Arc::clone(&slot.page)
7459 });
7460 let (again, through) = match cached.touched.get_mut(at) {
7462 Some(bits) if whole && page.is_none() => touch(bits, part, stripe.parts.len()),
7463 _ => (false, false),
7464 };
7465 let whole = whole && again;
7466 if let Some(index) = known.clone()
7467 && (!whole || page.is_some())
7468 {
7469 return Ok(CachedColumn { stripe: at, index, page });
7470 }
7471 if cached.loading.contains(&at) {
7472 drop(cached);
7473 if let Some(index) = known {
7477 return Ok(CachedColumn { stripe: at, index, page: None });
7478 }
7479 let held = self.page_of(stripe, column, at, false, None)?;
7480 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7481 remember(&mut cached, &held);
7482 return Ok(held);
7483 }
7484 cached.loading.push(at);
7485 drop(cached);
7486
7487 let read = self.page_of(stripe, column, at, whole, known);
7488
7489 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7493 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7494 cached.loading.remove(position);
7495 }
7496 let held = read?;
7497 let taken = remember(&mut cached, &held);
7498 if taken.is_some() && !through {
7499 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7500 cached.passing.push_back(at);
7501 while cached.passing.len() > floor {
7502 let Some(old) = cached.passing.pop_front() else { break };
7503 if let Some(slot) = cached.pages.get_mut(old) {
7504 *slot = None;
7505 }
7506 }
7507 return Ok(held);
7508 }
7509 drop(cached);
7510 if let Some((bytes, used)) = taken {
7511 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7512 self.pool.admit(Held {
7513 shelf: Arc::downgrade(&self.cache),
7514 column,
7515 stripe: at,
7516 bytes,
7517 used,
7518 });
7519 }
7520 Ok(held)
7521 }
7522
7523 fn page_of(
7529 &self,
7530 stripe: &Stripe,
7531 column: usize,
7532 at: usize,
7533 whole: bool,
7534 known: Option<Arc<Vec<PartSpan>>>,
7535 ) -> Result<CachedColumn> {
7536 let index = match known {
7537 Some(index) => index,
7538 None => {
7539 self.indexes.fetch_add(1, Atomic::Relaxed);
7540 Arc::new(read_index(&self.file, stripe, column)?)
7541 }
7542 };
7543 let page = if whole {
7544 self.pages.fetch_add(1, Atomic::Relaxed);
7545 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7546 let mut bytes = vec![0; span.length as usize];
7547 read_at(&self.file, span.offset, &mut bytes)?;
7548 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7549 Some(Arc::new(HeldPage { bytes, checked }))
7550 } else {
7551 None
7552 };
7553 Ok(CachedColumn { stripe: at, index, page })
7554 }
7555
7556 fn read_impl(
7557 &self,
7558 at: usize,
7559 columns: &[usize],
7560 whole: bool,
7561 positions: Option<&[u32]>,
7562 ) -> Result<Chunk> {
7563 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7564 let index = place.stripe as usize;
7565 let stripe =
7566 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7567 let rows = place.rows as usize;
7568 let mut picked = Vec::with_capacity(columns.len());
7569 for &column in columns {
7570 let field = self
7571 .table
7572 .fields
7573 .get(column)
7574 .ok_or_else(|| invalid("column index out of range"))?;
7575 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7576 let held = self.held(index, place.part as usize, stripe, column, whole)?;
7577 let span = *held
7578 .index
7579 .get(place.part as usize)
7580 .ok_or_else(|| invalid("part index out of range"))?;
7581 let owned;
7582 let bit = at * self.table.fields.len() + column;
7583 let bytes = match &held.page {
7584 Some(held) if self.is_verified(bit) => part_bytes(&held.bytes, span),
7585 Some(held) => {
7586 held.part(place.part as usize, span).inspect(|_| self.set_verified(bit))
7587 }
7588 None => {
7589 let offset = page
7590 .offset
7591 .checked_add(span.start as u64)
7592 .ok_or_else(|| invalid("part range overflow"))?;
7593 let mut bytes = vec![0; span.length];
7594 read_at(&self.file, offset, &mut bytes)?;
7595 owned = bytes;
7596 if self.is_verified(bit) {
7597 Ok(owned.as_slice())
7598 } else {
7599 verify_part(&owned, span).map(|()| {
7600 self.set_verified(bit);
7601 owned.as_slice()
7602 })
7603 }
7604 }
7605 }
7606 .map_err(|error| {
7607 invalid(&format!(
7608 "{}, column {column} part {} of the page at {}",
7609 error.message(),
7610 place.part,
7611 page.offset,
7612 ))
7613 })?;
7614 let dictionary = self.dictionary(column)?;
7615 let mut vector = match positions {
7621 None => decode(&field.ty, rows, bytes, dictionary)?,
7622 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7623 };
7624 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7628 vector = vector.flatten()?;
7629 }
7630 picked.push(vector.into_pages());
7631 }
7632 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7633 }
7634
7635 #[must_use]
7651 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7652 let Some(place) = self.places.get(part).copied() else { return false };
7653 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7654 if stripe.zone.skips(probes) {
7655 return true;
7656 }
7657 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7658 }
7659
7660 #[must_use]
7667 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7668 let Some(place) = self.places.get(part).copied() else { return false };
7669 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7670 if stripe.zone.column(column).is_some_and(&rule) {
7671 return true;
7672 }
7673 self.stripe_part_ranges(place.stripe as usize, column)
7674 .and_then(|ranges| ranges.get(place.part as usize))
7675 .is_some_and(rule)
7676 }
7677
7678 #[must_use]
7681 pub fn part_range(&self, part: usize, column: usize) -> Option<Range> {
7682 let place = self.places.get(part).copied()?;
7683 let own = self
7684 .stripe_part_ranges(place.stripe as usize, column)
7685 .and_then(|ranges| ranges.get(place.part as usize));
7686 own.or_else(|| self.table.stripes.get(place.stripe as usize)?.zone.column(column)).cloned()
7687 }
7688
7689 #[must_use]
7691 pub fn stripe_ruled_by(
7692 &self,
7693 stripe: usize,
7694 column: usize,
7695 rule: impl Fn(&Range) -> bool,
7696 ) -> bool {
7697 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7698 }
7699
7700 fn outside(&self, place: Place, probe: &Probe) -> bool {
7706 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7707 Some(ranges) => ranges
7708 .get(place.part as usize)
7709 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7710 None => false,
7711 }
7712 }
7713
7714 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7720 let slot = self.part_ranges.get(column)?.get(stripe)?;
7721 if let Some(held) = slot.get() {
7722 return Some(held);
7723 }
7724 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7725 let mut bytes = vec![0; page.length as usize];
7726 read_at(&self.file, page.offset, &mut bytes).ok()?;
7727 if checksum(&bytes) != page.hash {
7728 return None;
7729 }
7730 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7731 let _ = slot.set(ranges);
7732 slot.get().map(|held| held.as_slice())
7733 }
7734
7735 #[must_use]
7752 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7753 let Some(place) = self.places.get(part).copied() else { return false };
7754 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7755 if stripe.zone.certain(probes) {
7756 return true;
7757 }
7758 probes
7759 .iter()
7760 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7761 }
7762
7763 fn inside(&self, place: Place, probe: &Probe) -> bool {
7769 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7770 Some(ranges) => ranges
7771 .get(place.part as usize)
7772 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7773 None => false,
7774 }
7775 }
7776
7777 #[must_use]
7788 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7789 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7790 }
7791
7792 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7798 if probe.op != Op::Equal {
7799 return false;
7800 }
7801 match self.stripe_sieves(place.stripe as usize, probe.column) {
7802 Some(sieves) => sieves
7803 .get(place.part as usize)
7804 .and_then(Option::as_ref)
7805 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7806 None => false,
7807 }
7808 }
7809
7810 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7817 let slot = self.sieves.get(column)?.get(stripe)?;
7818 if let Some(held) = slot.get() {
7819 return Some(held);
7820 }
7821 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7822 let mut bytes = vec![0; page.length as usize];
7823 read_at(&self.file, page.offset, &mut bytes).ok()?;
7824 if checksum(&bytes) != page.hash {
7825 return None;
7826 }
7827 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7828 let _ = slot.set(sieves);
7829 slot.get().map(|held| held.as_slice())
7830 }
7831}
7832
7833fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7835 let code = dictionary.code_at_rank(rank)? as usize;
7836 if dictionary.logical_type() == &LogicalType::Blob {
7837 let bytes = dictionary
7838 .try_bytes_at(code)?
7839 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7840 return Ok(Value::Blob(bytes.to_vec()));
7841 }
7842 let text = dictionary
7843 .try_text_at(code)?
7844 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7845 Ok(Value::Varchar(text.into()))
7846}
7847
7848fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7858 file.fill_at(offset, bytes)
7859}
7860
7861trait Positional {
7869 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7874}
7875
7876impl<T: Positional + ?Sized> Positional for &T {
7877 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7878 (**self).fill_at(offset, bytes)
7879 }
7880}
7881
7882impl<T: Positional + ?Sized> Positional for Arc<T> {
7883 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7884 (**self).fill_at(offset, bytes)
7885 }
7886}
7887
7888impl<T: Positional + ?Sized> Positional for Box<T> {
7889 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7890 (**self).fill_at(offset, bytes)
7891 }
7892}
7893
7894impl Positional for dyn rudb_io::File + '_ {
7895 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7896 while !bytes.is_empty() {
7897 let read = self.read_at(offset, bytes)?;
7898 if read == 0 {
7899 return Err(invalid("column page ends before its declared length"));
7900 }
7901 offset += read as u64;
7902 bytes = &mut bytes[read..];
7903 }
7904 Ok(())
7905 }
7906}
7907
7908impl Positional for File {
7909 #[cfg(unix)]
7910 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7911 use std::os::unix::fs::FileExt;
7912 while !bytes.is_empty() {
7913 let read = self.read_at(bytes, offset).map_err(io)?;
7914 if read == 0 {
7915 return Err(invalid("column page ends before its declared length"));
7916 }
7917 offset += read as u64;
7918 bytes = &mut bytes[read..];
7919 }
7920 Ok(())
7921 }
7922
7923 #[cfg(windows)]
7929 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7930 use std::os::windows::fs::FileExt;
7931 while !bytes.is_empty() {
7932 let read = self.seek_read(bytes, offset).map_err(io)?;
7933 if read == 0 {
7934 return Err(invalid("column page ends before its declared length"));
7935 }
7936 offset += read as u64;
7937 bytes = &mut bytes[read..];
7938 }
7939 Ok(())
7940 }
7941
7942 #[cfg(not(any(unix, windows)))]
7947 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7948 use std::io::{Read, Seek, SeekFrom};
7949 let mut file = self.try_clone().map_err(io)?;
7950 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7951 file.read_exact(bytes).map_err(io)
7952 }
7953}
7954
7955#[cfg(test)]
7960fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7961 use std::io::{Seek, SeekFrom, Write};
7962 let mut file = file;
7963 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7964 file.write_all(bytes).map_err(io)
7965}
7966
7967fn type_tag(ty: &LogicalType) -> Result<u8> {
7974 match ty {
7975 LogicalType::SmallInt => Ok(1),
7976 LogicalType::Integer => Ok(2),
7977 LogicalType::BigInt => Ok(3),
7978 LogicalType::Varchar => Ok(4),
7979 LogicalType::Date => Ok(5),
7980 LogicalType::Timestamp => Ok(6),
7981 LogicalType::Boolean => Ok(7),
7982 LogicalType::TinyInt => Ok(8),
7983 LogicalType::UTinyInt => Ok(9),
7984 LogicalType::USmallInt => Ok(10),
7985 LogicalType::UInteger => Ok(11),
7986 LogicalType::UBigInt => Ok(12),
7987 LogicalType::Decimal { .. } => Ok(13),
7988 LogicalType::Float => Ok(14),
7989 LogicalType::Double => Ok(15),
7990 LogicalType::HugeInt => Ok(16),
7991 LogicalType::UHugeInt => Ok(17),
7992 LogicalType::Time => Ok(18),
7993 LogicalType::TimeTz => Ok(19),
7994 LogicalType::TimestampTz => Ok(20),
7995 LogicalType::Interval => Ok(21),
7996 LogicalType::Uuid => Ok(22),
7997 LogicalType::Blob => Ok(23),
7998 LogicalType::Bit => Ok(24),
7999 LogicalType::TimestampS => Ok(25),
8000 LogicalType::TimestampMs => Ok(26),
8001 LogicalType::TimestampNs => Ok(27),
8002 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
8003 }
8004}
8005
8006fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
8012 out.push(type_tag(ty)?);
8013 if let LogicalType::Decimal { width, scale } = ty {
8014 out.push(*width);
8015 out.push(*scale);
8016 }
8017 Ok(())
8018}
8019
8020fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
8022 let tag = cur.u8()?;
8023 if tag == 13 {
8024 let width = cur.u8()?;
8025 let scale = cur.u8()?;
8026 return LogicalType::decimal(width, scale)
8027 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
8028 }
8029 tag_type(tag)
8030}
8031
8032fn tag_type(tag: u8) -> Result<LogicalType> {
8033 match tag {
8034 1 => Ok(LogicalType::SmallInt),
8035 2 => Ok(LogicalType::Integer),
8036 3 => Ok(LogicalType::BigInt),
8037 4 => Ok(LogicalType::Varchar),
8038 5 => Ok(LogicalType::Date),
8039 6 => Ok(LogicalType::Timestamp),
8040 7 => Ok(LogicalType::Boolean),
8041 8 => Ok(LogicalType::TinyInt),
8042 9 => Ok(LogicalType::UTinyInt),
8043 10 => Ok(LogicalType::USmallInt),
8044 11 => Ok(LogicalType::UInteger),
8045 12 => Ok(LogicalType::UBigInt),
8046 14 => Ok(LogicalType::Float),
8047 15 => Ok(LogicalType::Double),
8048 16 => Ok(LogicalType::HugeInt),
8049 17 => Ok(LogicalType::UHugeInt),
8050 18 => Ok(LogicalType::Time),
8051 19 => Ok(LogicalType::TimeTz),
8052 20 => Ok(LogicalType::TimestampTz),
8053 21 => Ok(LogicalType::Interval),
8054 22 => Ok(LogicalType::Uuid),
8055 23 => Ok(LogicalType::Blob),
8056 24 => Ok(LogicalType::Bit),
8057 25 => Ok(LogicalType::TimestampS),
8058 26 => Ok(LogicalType::TimestampMs),
8059 27 => Ok(LogicalType::TimestampNs),
8060 _ => Err(invalid("column type tag is unknown")),
8061 }
8062}
8063
8064fn put_u16(out: &mut Vec<u8>, value: u16) {
8065 out.extend_from_slice(&value.to_le_bytes());
8066}
8067fn put_u32(out: &mut Vec<u8>, value: u32) {
8068 out.extend_from_slice(&value.to_le_bytes());
8069}
8070fn put_u64(out: &mut Vec<u8>, value: u64) {
8071 out.extend_from_slice(&value.to_le_bytes());
8072}
8073fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
8074 while value >= 0x80 {
8075 out.push((value as u8 & 0x7f) | 0x80);
8076 value >>= 7;
8077 }
8078 out.push(value as u8);
8079}
8080
8081fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
8082 match (left, right) {
8083 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
8084 (FrequencyValue::Null, _) => Ordering::Less,
8085 (_, FrequencyValue::Null) => Ordering::Greater,
8086 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
8087 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
8088 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
8089 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
8090 }
8091}
8092
8093fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
8106 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
8107 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
8108 };
8109 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
8110 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
8111 let omitted_max = next.count;
8112 entries.truncate(FREQUENCY_ENTRIES);
8113 omitted_max
8114 } else {
8115 0
8116 };
8117 entries.sort_unstable_by(order);
8118 omitted_max
8119}
8120
8121fn code_frequency(
8122 dictionary: &GlobalDictionary,
8123 flat: &[u8],
8124 bases: &[u64],
8125) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
8126 let mut entries = dictionary
8127 .counts
8128 .iter()
8129 .enumerate()
8130 .filter(|(_, count)| **count != 0)
8131 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
8132 .collect::<Vec<_>>();
8133 if dictionary.nulls != 0 {
8134 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
8135 }
8136 let omitted_max = keep_most_frequent(&mut entries);
8137 let mut spans = Vec::with_capacity(entries.len());
8138 let mut text_bytes = 0_usize;
8139 for entry in &entries {
8140 let span = match entry.value {
8141 FrequencyValue::Code(code) => {
8142 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
8143 let bytes = flat
8144 .get(span.0..span.1)
8145 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
8146 text_bytes = text_bytes.saturating_add(bytes.len());
8147 Some(span)
8148 }
8149 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
8150 };
8151 spans.push(span);
8152 }
8153 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
8154 Vec::new()
8155 } else {
8156 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
8157 };
8158 Ok((
8159 FrequencySummary {
8160 entries,
8161 omitted_max,
8162 ordinals: Vec::new(),
8163 ordinal_entries: Vec::new(),
8164 },
8165 texts,
8166 ))
8167}
8168
8169fn encode_directory(table: &Table) -> Result<Vec<u8>> {
8170 let mut out = DIRECTORY.to_vec();
8171 let name = table.name.as_bytes();
8172 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8173 out.extend_from_slice(name);
8174 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
8175 for field in &table.fields {
8176 let name = field.name.as_bytes();
8177 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
8178 out.extend_from_slice(name);
8179 put_type(&mut out, &field.ty)?;
8180 out.push(u8::from(field.not_null));
8181 }
8182 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
8183 match dictionary {
8184 None => out.push(0),
8185 Some(page) => {
8186 out.push(dictionary_tag(&field.ty));
8187 put_u64(&mut out, page.offset);
8188 put_u32(&mut out, page.length);
8189 put_u64(&mut out, page.hash);
8190 }
8191 }
8192 }
8193 for distinct in &table.distincts {
8194 match distinct {
8195 None => out.push(0),
8196 Some(count) => {
8197 out.push(1);
8198 put_u64(&mut out, *count);
8199 }
8200 }
8201 }
8202 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
8203 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
8204 for stripe in &table.stripes {
8205 put_u32(
8206 &mut out,
8207 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
8208 );
8209 for &rows in &stripe.parts {
8210 put_u32(&mut out, rows);
8211 }
8212 put_u64(&mut out, stripe.index.offset);
8213 put_u32(&mut out, stripe.index.length);
8214 for page in &stripe.pages {
8215 put_u64(&mut out, page.offset);
8216 put_u32(&mut out, page.length);
8217 }
8218 for (column, ((field, dictionary), membership)) in
8223 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
8224 {
8225 if !coded_type(&field.ty) || dictionary.is_none() {
8226 continue;
8227 }
8228 let page = match membership {
8229 Some(page) => page,
8230 None if table.demoted.get(column).copied().unwrap_or(false) => {
8231 Page { offset: HEADER, length: 0, hash: 0 }
8232 }
8233 None => return Err(invalid("string page has no code membership index")),
8234 };
8235 put_u64(&mut out, page.offset);
8236 put_u32(&mut out, page.length);
8237 put_u64(&mut out, page.hash);
8238 }
8239 for sieve in stripe.sieves.slots() {
8240 match sieve {
8241 None => out.push(0),
8242 Some(page) => {
8243 out.push(1);
8244 put_u64(&mut out, page.offset);
8245 put_u32(&mut out, page.length);
8246 put_u64(&mut out, page.hash);
8247 }
8248 }
8249 }
8250 for held in stripe.part_ranges.slots() {
8251 match held {
8252 None => out.push(0),
8253 Some(page) => {
8254 out.push(1);
8255 put_u64(&mut out, page.offset);
8256 put_u32(&mut out, page.length);
8257 put_u64(&mut out, page.hash);
8258 }
8259 }
8260 }
8261 for range in stripe.zone.columns() {
8262 put_bound(&mut out, range.low.as_ref())?;
8263 put_bound(&mut out, range.high.as_ref())?;
8264 put_u32(
8265 &mut out,
8266 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
8267 );
8268 out.push(u8::from(range.exact));
8269 match range.sum {
8270 None => out.push(0),
8271 Some(total) => {
8272 out.push(1);
8273 out.extend_from_slice(&total.to_le_bytes());
8274 }
8275 }
8276 }
8277 }
8278 out.extend_from_slice(FREQUENCIES_SPANS);
8279 put_u16(
8280 &mut out,
8281 u16::try_from(table.frequencies.len())
8282 .map_err(|_| invalid("too many frequency columns"))?,
8283 );
8284 for summary in &table.frequencies {
8285 let summary = match summary {
8286 None => {
8287 put_u32(&mut out, 0);
8288 put_u32(&mut out, 0);
8289 continue;
8290 }
8291 Some(Frequencies::Held(summary)) => summary,
8292 Some(Frequencies::Stored { .. }) => {
8294 return Err(invalid("a synopsis left in the file cannot be written back"));
8295 }
8296 };
8297 let length_at = out.len();
8298 put_u32(&mut out, 0);
8299 put_u32(
8300 &mut out,
8301 u32::try_from(summary.entries.len())
8302 .map_err(|_| invalid("too many frequency entries"))?,
8303 );
8304 let start = out.len();
8305 out.push(1);
8306 put_u64(&mut out, summary.omitted_max);
8307 put_u32(
8308 &mut out,
8309 u32::try_from(summary.entries.len())
8310 .map_err(|_| invalid("too many frequency entries"))?,
8311 );
8312 for entry in &summary.entries {
8313 match entry.value {
8314 FrequencyValue::Null => out.push(0),
8315 FrequencyValue::Integer(value) => {
8316 out.push(1);
8317 out.extend_from_slice(&value.to_le_bytes());
8318 }
8319 FrequencyValue::Code(value) => {
8320 out.push(2);
8321 put_u32(&mut out, value);
8322 }
8323 }
8324 put_u64(&mut out, entry.count);
8325 }
8326 put_u32(
8327 &mut out,
8328 u32::try_from(summary.ordinals.len())
8329 .map_err(|_| invalid("too many frequency ordinals"))?,
8330 );
8331 let mut previous = 0_u64;
8332 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8333 let delta = if at == 0 {
8334 ordinal
8335 } else {
8336 ordinal
8337 .checked_sub(previous)
8338 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8339 };
8340 if at != 0 && delta == 0 {
8341 return Err(invalid("frequency ordinals are not unique"));
8342 }
8343 put_var_u64(&mut out, delta);
8344 previous = ordinal;
8345 }
8346 if summary.ordinal_entries.len() != summary.ordinals.len() {
8347 return Err(invalid("frequency ordinal values have a different length"));
8348 }
8349 for &entry in &summary.ordinal_entries {
8350 if entry as usize >= summary.entries.len() {
8351 return Err(invalid("frequency ordinal value is outside its entries"));
8352 }
8353 put_u16(&mut out, entry);
8354 }
8355 let length = u32::try_from(out.len() - start)
8356 .map_err(|_| invalid("a frequency synopsis is too long"))?;
8357 out[length_at..length_at + 4].copy_from_slice(&length.to_le_bytes());
8358 }
8359 if !table.pair_frequencies.is_empty() {
8360 out.extend_from_slice(PAIR_FREQUENCIES);
8361 put_u16(
8362 &mut out,
8363 u16::try_from(table.pair_frequencies.len())
8364 .map_err(|_| invalid("too many pair frequency summaries"))?,
8365 );
8366 for summary in &table.pair_frequencies {
8367 put_u16(&mut out, summary.first);
8368 put_u16(&mut out, summary.second);
8369 put_u64(&mut out, summary.omitted_max);
8370 put_u16(
8371 &mut out,
8372 u16::try_from(summary.entries.len())
8373 .map_err(|_| invalid("too many pair frequency entries"))?,
8374 );
8375 for entry in &summary.entries {
8376 put_u16(&mut out, entry.first_entry);
8377 match entry.second {
8378 None => out.push(0),
8379 Some(code) => {
8380 out.push(1);
8381 put_u32(&mut out, code);
8382 }
8383 }
8384 put_u64(&mut out, entry.count);
8385 }
8386 }
8387 }
8388 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8389 if text_columns != 0 {
8390 out.extend_from_slice(FREQUENCY_TEXTS);
8391 put_u16(
8392 &mut out,
8393 u16::try_from(text_columns)
8394 .map_err(|_| invalid("too many string frequency columns"))?,
8395 );
8396 for (column, texts) in table.frequency_texts.iter().enumerate() {
8397 if texts.is_empty() {
8398 continue;
8399 }
8400 put_u16(
8401 &mut out,
8402 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8403 );
8404 put_u16(
8405 &mut out,
8406 u16::try_from(texts.len())
8407 .map_err(|_| invalid("too many frequency text entries"))?,
8408 );
8409 for text in texts {
8410 match text {
8411 None => out.push(0),
8412 Some(text) => {
8413 out.push(1);
8414 put_u32(
8415 &mut out,
8416 u32::try_from(text.len())
8417 .map_err(|_| invalid("frequency text is too long"))?,
8418 );
8419 out.extend_from_slice(text);
8420 }
8421 }
8422 }
8423 }
8424 }
8425 if let Some(summary) = &table.host_groups {
8426 out.extend_from_slice(HOST_GROUPS);
8427 put_u16(
8428 &mut out,
8429 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8430 );
8431 put_u64(&mut out, summary.omitted_max);
8432 put_u16(
8433 &mut out,
8434 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8435 );
8436 for entry in &summary.entries {
8437 put_u32(
8438 &mut out,
8439 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8440 );
8441 out.extend_from_slice(entry.host.as_bytes());
8442 put_u64(&mut out, entry.count);
8443 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8444 put_u32(
8445 &mut out,
8446 u32::try_from(entry.minimum.len())
8447 .map_err(|_| invalid("host minimum is too long"))?,
8448 );
8449 out.extend_from_slice(entry.minimum.as_bytes());
8450 }
8451 }
8452 if let Some(clustering) = &table.clustering {
8455 out.extend_from_slice(CLUSTERING);
8456 out.push(clustering.width().tag());
8457 put_u16(
8458 &mut out,
8459 u16::try_from(clustering.columns().len())
8460 .map_err(|_| invalid("too many clustering columns"))?,
8461 );
8462 for &column in clustering.columns() {
8463 put_u16(
8464 &mut out,
8465 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8466 );
8467 }
8468 }
8469 let demoted = (0..table.fields.len())
8470 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8471 .collect::<Vec<_>>();
8472 if !demoted.is_empty() {
8473 out.extend_from_slice(DEMOTED);
8474 put_u16(
8475 &mut out,
8476 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8477 );
8478 for column in demoted {
8479 put_u16(
8480 &mut out,
8481 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8482 );
8483 }
8484 }
8485 if !table.constraints.is_empty() {
8486 out.extend_from_slice(KEYS);
8487 put_count(&mut out, table.constraints.keys.len())?;
8488 for (columns, primary) in &table.constraints.keys {
8489 out.push(u8::from(*primary));
8490 put_columns(&mut out, columns)?;
8491 }
8492 put_count(&mut out, table.constraints.foreign.len())?;
8493 for foreign in &table.constraints.foreign {
8494 put_columns(&mut out, &foreign.columns)?;
8495 put_columns(&mut out, &foreign.referenced)?;
8496 put_u32(
8497 &mut out,
8498 u32::try_from(foreign.table.len())
8499 .map_err(|_| invalid("table name is too long"))?,
8500 );
8501 out.extend_from_slice(foreign.table.as_bytes());
8502 }
8503 }
8504 out.extend_from_slice(SECTIONS);
8510 put_u64(&mut out, table.generation);
8511 put_u16(
8512 &mut out,
8513 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8514 );
8515 for held in &table.sections {
8516 held.encode(&mut out)?;
8517 }
8518 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8519 out.extend_from_slice(DICTIONARY_PAYLOADS);
8520 put_u16(
8521 &mut out,
8522 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8523 );
8524 for at in 0..table.fields.len() {
8525 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8526 }
8527 }
8528 Ok(out)
8529}
8530
8531fn signed_integer(ty: &LogicalType) -> bool {
8540 matches!(
8541 ty,
8542 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8543 )
8544}
8545
8546fn integer_or_date(ty: &LogicalType) -> bool {
8547 matches!(
8548 ty,
8549 LogicalType::TinyInt
8550 | LogicalType::SmallInt
8551 | LogicalType::Integer
8552 | LogicalType::BigInt
8553 | LogicalType::UTinyInt
8554 | LogicalType::USmallInt
8555 | LogicalType::UInteger
8556 | LogicalType::UBigInt
8557 | LogicalType::Date
8558 )
8559}
8560
8561fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8562 table
8563 .fields
8564 .iter()
8565 .enumerate()
8566 .map(|(column, field)| {
8567 if !integer_or_date(&field.ty) {
8568 return None;
8569 }
8570 let mut low: Option<i128> = None;
8571 let mut high: Option<i128> = None;
8572 for stripe in &table.stripes {
8573 let range = stripe.zone.column(column)?;
8574 if !range.exact {
8575 return None;
8576 }
8577 match (range.low.as_ref(), range.high.as_ref()) {
8578 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8579 low = Some(low.map_or(*small, |held| held.min(*small)));
8580 high = Some(high.map_or(*large, |held| held.max(*large)));
8581 }
8582 (None, None) if stripe.rows == range.nulls => {}
8583 _ => return None,
8584 }
8585 }
8586 Some(low.zip(high))
8587 })
8588 .collect()
8589}
8590
8591fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8592 reader
8593 .table
8594 .fields
8595 .iter()
8596 .enumerate()
8597 .map(|(column, field)| {
8598 if !integer_or_date(&field.ty) {
8599 return Ok(None);
8600 }
8601 match reader.exact_extremes(column)? {
8602 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8603 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8604 _ => Ok(None),
8605 }
8606 })
8607 .collect()
8608}
8609
8610fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8611 table
8612 .fields
8613 .iter()
8614 .enumerate()
8615 .map(|(column, field)| {
8616 if !integer_or_date(&field.ty) {
8617 return None;
8618 }
8619 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8620 return None;
8621 };
8622 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8623 return None;
8624 }
8625 let entries = summary
8626 .entries
8627 .iter()
8628 .map(|entry| {
8629 let value = match entry.value {
8630 FrequencyValue::Null => None,
8631 FrequencyValue::Integer(value) => Some(value),
8632 FrequencyValue::Code(_) => return None,
8633 };
8634 Some((value, entry.count))
8635 })
8636 .collect::<Option<Vec<_>>>()?;
8637 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8638 (rows == table.rows as u64).then_some(entries)
8639 })
8640 .collect()
8641}
8642
8643fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8650 if signed {
8651 FrequencyValue::Integer(i128::from(bits as i64))
8652 } else {
8653 FrequencyValue::Integer(i128::from(bits))
8654 }
8655}
8656
8657fn frequency_bits(value: &Value) -> Option<u64> {
8658 Some(match value {
8659 Value::TinyInt(value) => i64::from(*value) as u64,
8660 Value::SmallInt(value) => i64::from(*value) as u64,
8661 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8662 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8663 Value::UTinyInt(value) => u64::from(*value),
8664 Value::USmallInt(value) => u64::from(*value),
8665 Value::UInteger(value) => u64::from(*value),
8666 Value::UBigInt(value) => *value,
8667 _ => return None,
8668 })
8669}
8670
8671fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8672 Some(match value {
8673 Value::Null => None,
8674 Value::TinyInt(value) => Some(i128::from(*value)),
8675 Value::SmallInt(value) => Some(i128::from(*value)),
8676 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8677 Value::BigInt(value) => Some(i128::from(*value)),
8678 Value::UTinyInt(value) => Some(i128::from(*value)),
8679 Value::USmallInt(value) => Some(i128::from(*value)),
8680 Value::UInteger(value) => Some(i128::from(*value)),
8681 Value::UBigInt(value) => Some(i128::from(*value)),
8682 _ => return None,
8683 })
8684}
8685
8686fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8687 reader
8688 .table
8689 .fields
8690 .iter()
8691 .enumerate()
8692 .map(|(column, field)| {
8693 if !integer_or_date(&field.ty) {
8694 return Ok(None);
8695 }
8696 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8697 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8698 return Ok(None);
8699 }
8700 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8701 let Some(entries) = entries
8702 .iter()
8703 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8704 .collect::<Option<Vec<_>>>()
8705 else {
8706 return Ok(None);
8707 };
8708 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8709 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8710 })
8711 .collect()
8712}
8713
8714fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8715 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8716 let range = stripe.zone.column(column)?;
8717 let sum = sum.checked_add(range.sum?)?;
8718 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8719 Some((sum, count.checked_add(nonnull)?))
8720 })
8721}
8722
8723fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8724 table
8725 .fields
8726 .iter()
8727 .enumerate()
8728 .map(|(column, field)| {
8729 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8730 })
8731 .collect()
8732}
8733
8734fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8735 reader
8736 .table
8737 .fields
8738 .iter()
8739 .enumerate()
8740 .map(
8741 |(column, field)| {
8742 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8743 },
8744 )
8745 .collect()
8746}
8747
8748fn encode_catalog(
8749 entries: &[Entry],
8750 views: &[ViewEntry],
8751 card: Option<&KeptCard>,
8752) -> Result<Vec<u8>> {
8753 let mut out = CATALOG.to_vec();
8754 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8755 for entry in entries {
8756 let name = entry.name.as_bytes();
8757 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8758 out.extend_from_slice(name);
8759 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8760 put_u16(
8761 &mut out,
8762 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8763 );
8764 for field in &entry.fields {
8765 let name = field.name.as_bytes();
8766 put_u16(
8767 &mut out,
8768 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8769 );
8770 out.extend_from_slice(name);
8771 put_type(&mut out, &field.ty)?;
8772 out.push(u8::from(field.not_null));
8773 }
8774 put_u64(&mut out, entry.directory.offset);
8775 put_u32(&mut out, entry.directory.length);
8776 put_u64(&mut out, entry.directory.hash);
8777 }
8778 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8779 for view in views {
8780 let name = view.name.as_bytes();
8781 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8782 out.extend_from_slice(name);
8783 put_long_text(&mut out, &view.sql, "view body")?;
8784 put_long_text(&mut out, &view.statement, "view statement")?;
8785 put_u16(
8786 &mut out,
8787 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8788 );
8789 for alias in &view.aliases {
8790 let alias = alias.as_bytes();
8791 put_u16(
8792 &mut out,
8793 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8794 );
8795 out.extend_from_slice(alias);
8796 }
8797 put_u16(
8798 &mut out,
8799 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8800 );
8801 for field in &view.columns {
8802 let name = field.name.as_bytes();
8803 put_u16(
8804 &mut out,
8805 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8806 );
8807 out.extend_from_slice(name);
8808 put_type(&mut out, &field.ty)?;
8809 out.push(u8::from(field.not_null));
8810 }
8811 }
8812 out.extend_from_slice(NONZERO_COUNTS);
8813 for entry in entries {
8814 if entry.nonzero.len() != entry.fields.len() {
8815 return Err(invalid("nonzero count width differs from schema"));
8816 }
8817 for count in &entry.nonzero {
8818 match count {
8819 None => out.push(0),
8820 Some(count) => {
8821 out.push(1);
8822 put_u64(&mut out, *count);
8823 }
8824 }
8825 }
8826 }
8827 out.extend_from_slice(AGGREGATE_SUMS);
8828 for entry in entries {
8829 if entry.aggregates.len() != entry.fields.len() {
8830 return Err(invalid("aggregate sum width differs from schema"));
8831 }
8832 for summary in &entry.aggregates {
8833 match summary {
8834 None => out.push(0),
8835 Some((sum, count)) => {
8836 out.push(1);
8837 out.extend_from_slice(&sum.to_le_bytes());
8838 put_u64(&mut out, *count);
8839 }
8840 }
8841 }
8842 }
8843 out.extend_from_slice(DISTINCT_COUNTS);
8844 for entry in entries {
8845 if entry.distincts.len() != entry.fields.len() {
8846 return Err(invalid("distinct count width differs from schema"));
8847 }
8848 for count in &entry.distincts {
8849 match count {
8850 None => out.push(0),
8851 Some(count) => {
8852 if *count > entry.rows as u64 {
8853 return Err(invalid("distinct count exceeds table rows"));
8854 }
8855 out.push(1);
8856 put_u64(&mut out, *count);
8857 }
8858 }
8859 }
8860 }
8861 out.extend_from_slice(INTEGER_EXTREMES);
8862 for entry in entries {
8863 if entry.extremes.len() != entry.fields.len() {
8864 return Err(invalid("integer extremes width differs from schema"));
8865 }
8866 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8867 match extremes {
8868 None => out.push(0),
8869 Some(None) if integer_or_date(&field.ty) => out.push(1),
8870 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8871 out.push(2);
8872 out.extend_from_slice(&low.to_le_bytes());
8873 out.extend_from_slice(&high.to_le_bytes());
8874 }
8875 _ => return Err(invalid("integer extremes type or range differs")),
8876 }
8877 }
8878 }
8879 out.extend_from_slice(COMPLETE_FREQUENCIES);
8880 for entry in entries {
8881 if entry.frequencies.len() != entry.fields.len() {
8882 return Err(invalid("numeric frequency width differs from schema"));
8883 }
8884 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8885 match frequencies {
8886 None => out.push(0),
8887 Some(entries)
8888 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8889 {
8890 let mut total = 0_u64;
8891 for (at, (value, count)) in entries.iter().enumerate() {
8892 if entries[..at].iter().any(|(held, _)| held == value) {
8893 return Err(invalid("numeric frequency value repeats"));
8894 }
8895 total = total
8896 .checked_add(*count)
8897 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8898 }
8899 if total != entry.rows as u64 {
8900 return Err(invalid("numeric frequencies do not cover table rows"));
8901 }
8902 out.push(1);
8903 out.push(entries.len() as u8);
8904 for (value, count) in entries {
8905 match value {
8906 None => out.push(0),
8907 Some(value) => {
8908 out.push(1);
8909 out.extend_from_slice(&value.to_le_bytes());
8910 }
8911 }
8912 put_u64(&mut out, *count);
8913 }
8914 }
8915 _ => return Err(invalid("numeric frequency type or width differs")),
8916 }
8917 }
8918 }
8919 if let Some(card) = card {
8920 out.extend_from_slice(DEVICE_CARD);
8921 let device = card.device.as_bytes();
8922 put_u16(&mut out, u16::try_from(device.len()).map_err(|_| invalid("device id too long"))?);
8923 out.extend_from_slice(device);
8924 put_u32(&mut out, u32::try_from(card.bytes.len()).map_err(|_| invalid("card too long"))?);
8925 out.extend_from_slice(&card.bytes);
8926 }
8927 Ok(out)
8928}
8929
8930fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8932 let bytes = text.as_bytes();
8933 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8934 out.extend_from_slice(bytes);
8935 Ok(())
8936}
8937
8938fn decode_catalog(bytes: &[u8], size: u64) -> Result<Decoded> {
8941 let mut cur = Cursor::new(bytes);
8942 if cur.take(8)? != CATALOG {
8943 return Err(invalid("catalog magic differs"));
8944 }
8945 let count = cur.u32()? as usize;
8946 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8947 for _ in 0..count {
8948 let name = cur.text()?;
8949 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8950 let width = cur.u16()? as usize;
8951 let mut fields = Vec::with_capacity(width);
8952 for _ in 0..width {
8953 let name = cur.text()?;
8954 let ty = read_type(&mut cur)?;
8955 let not_null = match cur.u8()? {
8956 0 => false,
8957 1 => true,
8958 _ => return Err(invalid("nullability flag differs")),
8959 };
8960 fields.push(Field { name, ty, not_null });
8961 }
8962 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8963 let end = directory
8964 .offset
8965 .checked_add(u64::from(directory.length))
8966 .ok_or_else(|| invalid("table directory offset overflow"))?;
8967 if directory.offset < HEADER
8968 || end > size
8969 || directory.length as usize > MAX_DIRECTORY
8970 || directory.length == 0
8971 {
8972 return Err(invalid("table directory range is outside the file"));
8973 }
8974 if entries.iter().any(|held| held.name == name) {
8975 return Err(invalid("two tables in the catalog have the same name"));
8976 }
8977 let nonzero = vec![None; fields.len()];
8978 let aggregates = vec![None; fields.len()];
8979 let distincts = vec![None; fields.len()];
8980 let extremes = vec![None; fields.len()];
8981 let frequencies = vec![None; fields.len()];
8982 entries.push(Entry {
8983 name,
8984 fields,
8985 rows,
8986 directory,
8987 nonzero,
8988 aggregates,
8989 distincts,
8990 extremes,
8991 frequencies,
8992 });
8993 }
8994 let count = if cur.done() { 0 } else { cur.u32()? as usize };
8999 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
9000 for _ in 0..count {
9001 let name = cur.text()?;
9002 let sql = cur.long_text()?;
9003 let statement = cur.long_text()?;
9004 let width = cur.u16()? as usize;
9005 let mut aliases = Vec::with_capacity(width);
9006 for _ in 0..width {
9007 aliases.push(cur.text()?);
9008 }
9009 let width = cur.u16()? as usize;
9010 let mut columns = Vec::with_capacity(width);
9011 for _ in 0..width {
9012 let name = cur.text()?;
9013 let ty = read_type(&mut cur)?;
9014 let not_null = match cur.u8()? {
9015 0 => false,
9016 1 => true,
9017 _ => return Err(invalid("nullability flag differs")),
9018 };
9019 columns.push(Field { name, ty, not_null });
9020 }
9021 if views.iter().any(|held| held.name == name) {
9025 return Err(invalid("two views in the catalog have the same name"));
9026 }
9027 if entries.iter().any(|held| held.name == name) {
9028 return Err(invalid("a table and a view in the catalog have the same name"));
9029 }
9030 views.push(ViewEntry { name, sql, statement, aliases, columns });
9031 }
9032 if !cur.done() {
9033 if cur.take(8)? != NONZERO_COUNTS {
9034 return Err(invalid("catalog extension magic differs"));
9035 }
9036 for entry in &mut entries {
9037 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
9038 *count = match cur.u8()? {
9039 0 => None,
9040 1 if matches!(
9041 field.ty,
9042 LogicalType::TinyInt
9043 | LogicalType::SmallInt
9044 | LogicalType::Integer
9045 | LogicalType::BigInt
9046 | LogicalType::UTinyInt
9047 | LogicalType::USmallInt
9048 | LogicalType::UInteger
9049 | LogicalType::UBigInt
9050 ) =>
9051 {
9052 let value = cur.u64()?;
9053 if value > entry.rows as u64 {
9054 return Err(invalid("nonzero count exceeds rows"));
9055 }
9056 Some(value)
9057 }
9058 _ => return Err(invalid("nonzero count tag or column type differs")),
9059 };
9060 }
9061 }
9062 }
9063 if !cur.done() {
9064 if cur.take(8)? != AGGREGATE_SUMS {
9065 return Err(invalid("aggregate catalog extension magic differs"));
9066 }
9067 for entry in &mut entries {
9068 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
9069 *summary = match cur.u8()? {
9070 0 => None,
9071 1 if signed_integer(&field.ty) => {
9072 let sum = i128::from_le_bytes(
9073 cur.take(16)?
9074 .try_into()
9075 .map_err(|_| invalid("aggregate sum is truncated"))?,
9076 );
9077 let count = cur.u64()?;
9078 if count > entry.rows as u64 {
9079 return Err(invalid("aggregate count exceeds table rows"));
9080 }
9081 Some((sum, count))
9082 }
9083 _ => return Err(invalid("aggregate sum tag or column type differs")),
9084 };
9085 }
9086 }
9087 }
9088 if !cur.done() {
9089 if cur.take(8)? != DISTINCT_COUNTS {
9090 return Err(invalid("distinct catalog extension magic differs"));
9091 }
9092 for entry in &mut entries {
9093 for count in &mut entry.distincts {
9094 *count = match cur.u8()? {
9095 0 => None,
9096 1 => {
9097 let value = cur.u64()?;
9098 if value > entry.rows as u64 {
9099 return Err(invalid("distinct count exceeds table rows"));
9100 }
9101 Some(value)
9102 }
9103 _ => return Err(invalid("distinct count tag differs")),
9104 };
9105 }
9106 }
9107 }
9108 if !cur.done() {
9109 if cur.take(8)? != INTEGER_EXTREMES {
9110 return Err(invalid("integer extremes catalog extension magic differs"));
9111 }
9112 for entry in &mut entries {
9113 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
9114 *extremes = match cur.u8()? {
9115 0 => None,
9116 1 if integer_or_date(&field.ty) => Some(None),
9117 2 if integer_or_date(&field.ty) => {
9118 let low = i128::from_le_bytes(
9119 cur.take(16)?
9120 .try_into()
9121 .map_err(|_| invalid("minimum is truncated"))?,
9122 );
9123 let high = i128::from_le_bytes(
9124 cur.take(16)?
9125 .try_into()
9126 .map_err(|_| invalid("maximum is truncated"))?,
9127 );
9128 if low > high {
9129 return Err(invalid("integer extremes are reversed"));
9130 }
9131 Some(Some((low, high)))
9132 }
9133 _ => return Err(invalid("integer extremes tag or type differs")),
9134 };
9135 }
9136 }
9137 }
9138 if !cur.done() {
9139 if cur.take(8)? != COMPLETE_FREQUENCIES {
9140 return Err(invalid("numeric frequency catalog extension magic differs"));
9141 }
9142 for entry in &mut entries {
9143 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
9144 *frequencies = match cur.u8()? {
9145 0 => None,
9146 1 if integer_or_date(&field.ty) => {
9147 let len = cur.u8()? as usize;
9148 if len > MAX_CATALOG_FREQUENCIES {
9149 return Err(invalid("too many catalog numeric frequencies"));
9150 }
9151 let mut values = Vec::with_capacity(len);
9152 let mut total = 0_u64;
9153 for _ in 0..len {
9154 let value = match cur.u8()? {
9155 0 => None,
9156 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
9157 |_| invalid("numeric frequency value is truncated"),
9158 )?)),
9159 _ => return Err(invalid("numeric frequency value tag differs")),
9160 };
9161 if values.iter().any(|(held, _)| *held == value) {
9162 return Err(invalid("numeric frequency value repeats"));
9163 }
9164 let count = cur.u64()?;
9165 total = total
9166 .checked_add(count)
9167 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
9168 values.push((value, count));
9169 }
9170 if total != entry.rows as u64 {
9171 return Err(invalid("numeric frequencies do not cover table rows"));
9172 }
9173 Some(values)
9174 }
9175 _ => return Err(invalid("numeric frequency tag or type differs")),
9176 };
9177 }
9178 }
9179 }
9180 let mut card = None;
9181 if !cur.done() {
9182 if cur.take(8)? != DEVICE_CARD {
9183 return Err(invalid("device card catalog extension magic differs"));
9184 }
9185 let device = cur.text()?;
9186 let len = cur.u32()? as usize;
9187 if len > MAX_CARD {
9188 return Err(invalid("device card is longer than any card"));
9189 }
9190 card = Some(KeptCard { device, bytes: cur.take(len)?.to_vec() });
9191 }
9192 if !cur.done() {
9193 return Err(invalid("catalog has trailing bytes"));
9194 }
9195 Ok((entries, views, card))
9196}
9197
9198type Decoded = (Vec<Entry>, Vec<ViewEntry>, Option<KeptCard>);
9200
9201const MAX_CARD: usize = 64 << 10;
9203
9204#[derive(Debug, Clone, PartialEq, Eq)]
9211struct KeptCard {
9212 device: String,
9213 bytes: Vec<u8>,
9214}
9215
9216fn directory_of(path: &Path) -> &Path {
9218 path.parent().filter(|dir| !dir.as_os_str().is_empty()).unwrap_or(Path::new("."))
9219}
9220
9221fn card_for(path: &Path, held: Option<KeptCard>) -> Option<KeptCard> {
9228 let Ok(device) = rudb_io::device::device_key(directory_of(path)) else {
9229 return held;
9230 };
9231 match rudb_io::device::kept(&device) {
9232 Some(card) => Some(KeptCard { device, bytes: card.encode() }),
9233 None => held,
9234 }
9235}
9236
9237fn remember_card(path: &Path, card: Option<&KeptCard>) {
9239 let Some(card) = card else { return };
9240 let dir = directory_of(path);
9241 let Ok(device) = rudb_io::device::device_key(dir) else { return };
9242 if device != card.device {
9243 return;
9244 }
9245 if let Ok(decoded) = rudb_io::device::Card::decode(&card.bytes, dir) {
9246 rudb_io::device::remember(&device, decoded);
9247 }
9248}
9249
9250struct Cursor<'a> {
9258 bytes: &'a [u8],
9259 at: usize,
9260 window: Option<Window<'a>>,
9261}
9262
9263struct Window<'a> {
9265 file: &'a File,
9266 offset: u64,
9267 length: usize,
9268 start: usize,
9270 held: Vec<u8>,
9271 size: usize,
9273}
9274
9275const DIRECTORY_WINDOW: usize = 64 << 10;
9277
9278impl<'a> Cursor<'a> {
9279 fn new(bytes: &'a [u8]) -> Self {
9280 Self { bytes, at: 0, window: None }
9281 }
9282
9283 fn over(file: &'a File, offset: u64, length: usize) -> Self {
9285 let window =
9286 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
9287 Self { bytes: &[], at: 0, window: Some(window) }
9288 }
9289
9290 fn len(&self) -> usize {
9292 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
9293 }
9294
9295 fn ensure(&mut self, len: usize) -> Result<()> {
9297 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9298 if end > self.len() {
9299 return Err(invalid("directory is truncated"));
9300 }
9301 let Some(window) = &mut self.window else { return Ok(()) };
9302 if self.at < window.start || end > window.start + window.held.len() {
9303 let want = len.max(window.size).min(window.length - self.at);
9304 window.start = self.at;
9305 window.held.resize(want, 0);
9306 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
9307 }
9308 Ok(())
9309 }
9310
9311 fn held(&self, at: usize, len: usize) -> &[u8] {
9313 match &self.window {
9314 Some(window) => &window.held[at - window.start..at - window.start + len],
9315 None => &self.bytes[at..at + len],
9316 }
9317 }
9318
9319 #[inline]
9321 fn peek(&mut self, len: usize) -> Result<&[u8]> {
9322 if self.window.is_none() {
9323 let bytes = self.bytes;
9324 return Ok(&bytes[self.at..self.end(len)?]);
9325 }
9326 self.ensure(len)?;
9327 Ok(self.held(self.at, len))
9328 }
9329
9330 #[inline]
9336 fn take(&mut self, len: usize) -> Result<&[u8]> {
9337 if self.window.is_none() {
9338 let bytes = self.bytes;
9339 let (at, end) = (self.at, self.end(len)?);
9340 self.at = end;
9341 return Ok(&bytes[at..end]);
9342 }
9343 self.take_windowed(len)
9344 }
9345
9346 fn skip(&mut self, len: usize) -> Result<()> {
9348 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9349 if end > self.len() {
9350 return Err(invalid("directory is truncated"));
9351 }
9352 self.at = end;
9353 Ok(())
9354 }
9355
9356 fn skip_bound(&mut self) -> Result<()> {
9357 match self.u8()? {
9358 0 => Ok(()),
9359 1 => self.skip(16),
9360 2 => self.skip(8),
9361 3 => {
9362 let length = self.u32()? as usize;
9363 self.skip(length)
9364 }
9365 4 => self.skip(17),
9366 _ => Err(invalid("a stored bound has an unknown tag")),
9367 }
9368 }
9369
9370 #[inline]
9372 fn end(&self, len: usize) -> Result<usize> {
9373 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
9374 if end > self.bytes.len() {
9375 return Err(invalid("directory is truncated"));
9376 }
9377 Ok(end)
9378 }
9379
9380 #[inline(never)]
9382 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
9383 self.ensure(len)?;
9384 self.at += len;
9385 Ok(self.held(self.at - len, len))
9386 }
9387 #[inline]
9388 fn u8(&mut self) -> Result<u8> {
9389 Ok(self.take(1)?[0])
9390 }
9391 #[inline]
9392 fn u16(&mut self) -> Result<u16> {
9393 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
9394 }
9395 #[inline]
9396 fn u32(&mut self) -> Result<u32> {
9397 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
9398 }
9399 #[inline]
9400 fn u64(&mut self) -> Result<u64> {
9401 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
9402 }
9403 fn var_u64(&mut self) -> Result<u64> {
9404 let mut value = 0_u64;
9405 for shift in (0..=63).step_by(7) {
9406 let byte = self.u8()?;
9407 let part = u64::from(byte & 0x7f);
9408 if shift == 63 && part > 1 {
9409 return Err(invalid("frequency ordinal varint overflows"));
9410 }
9411 value |= part << shift;
9412 if byte & 0x80 == 0 {
9413 return Ok(value);
9414 }
9415 }
9416 Err(invalid("frequency ordinal varint is too long"))
9417 }
9418 fn bound(&mut self) -> Result<Option<Bound>> {
9427 let rest = self.len().saturating_sub(self.at);
9428 let mut want = 32;
9429 loop {
9430 let offered = self.peek(want.min(rest))?;
9431 let mut used = 0;
9432 match bounds::get(offered, &mut used) {
9433 Ok(bound) => {
9434 self.at += used;
9435 return Ok(bound);
9436 }
9437 Err(_) if want < rest => want *= 2,
9438 Err(error) => return Err(error),
9439 }
9440 }
9441 }
9442 fn text(&mut self) -> Result<String> {
9443 let len = self.u16()? as usize;
9444 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9445 }
9446 fn done(&self) -> bool {
9449 self.at >= self.len()
9450 }
9451 fn long_text(&mut self) -> Result<String> {
9458 let len = self.u32()? as usize;
9459 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9460 }
9461}
9462
9463fn decode_summary(
9465 cur: &mut Cursor<'_>,
9466 field: &Field,
9467 rows: usize,
9468 values: bool,
9469) -> Result<Option<FrequencySummary>> {
9470 Ok(match cur.u8()? {
9471 0 => None,
9472 1 => {
9473 let omitted_max = cur.u64()?;
9474 let count = cur.u32()? as usize;
9475 if count > FREQUENCY_ENTRIES {
9476 return Err(invalid("frequency entry count exceeds its bound"));
9477 }
9478 let mut entries = Vec::with_capacity(count);
9479 for _ in 0..count {
9481 let value = match cur.u8()? {
9482 0 => FrequencyValue::Null,
9483 1 => FrequencyValue::Integer(i128::from_le_bytes(
9484 cur.take(16)?.try_into().expect("sixteen bytes"),
9485 )),
9486 2 => FrequencyValue::Code(cur.u32()?),
9487 _ => return Err(invalid("frequency value tag differs")),
9488 };
9489 let valid = matches!(
9490 (&field.ty, value),
9491 (_, FrequencyValue::Null)
9492 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9493 | (
9494 LogicalType::TinyInt
9495 | LogicalType::SmallInt
9496 | LogicalType::Integer
9497 | LogicalType::BigInt
9498 | LogicalType::UTinyInt
9499 | LogicalType::USmallInt
9500 | LogicalType::UInteger
9501 | LogicalType::UBigInt
9502 | LogicalType::Date
9503 | LogicalType::Timestamp,
9504 FrequencyValue::Integer(_),
9505 )
9506 );
9507 if !valid {
9508 return Err(invalid("frequency value does not match its column"));
9509 }
9510 let count = cur.u64()?;
9511 if count == 0 || count > rows as u64 {
9512 return Err(invalid("frequency count is outside the table"));
9513 }
9514 entries.push(FrequencyEntry { value, count });
9515 }
9516 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9517 return Err(invalid("frequency entries are not descending"));
9518 }
9519 let ordinals = {
9520 let ordinal_count = cur.u32()? as usize;
9521 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9522 return Err(invalid("frequency ordinal count exceeds its bound"));
9523 }
9524 let mut ordinals = Vec::with_capacity(ordinal_count);
9525 let mut previous = 0_u64;
9526 for at in 0..ordinal_count {
9527 let delta = cur.var_u64()?;
9528 if at != 0 && delta == 0 {
9529 return Err(invalid("frequency ordinals are not increasing"));
9530 }
9531 let ordinal = if at == 0 {
9532 delta
9533 } else {
9534 previous
9535 .checked_add(delta)
9536 .ok_or_else(|| invalid("frequency ordinal overflows"))?
9537 };
9538 if ordinal >= rows as u64 {
9539 return Err(invalid("frequency ordinal is outside the table"));
9540 }
9541 ordinals.push(ordinal);
9542 previous = ordinal;
9543 }
9544 ordinals
9545 };
9546 let ordinal_entries = if values {
9547 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9548 for _ in 0..ordinals.len() {
9549 let entry = cur.u16()?;
9550 if entry as usize >= entries.len() {
9551 return Err(invalid("frequency ordinal value is outside its entries"));
9552 }
9553 ordinal_entries.push(entry);
9554 }
9555 ordinal_entries
9556 } else {
9557 Vec::new()
9558 };
9559 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
9560 }
9561 _ => return Err(invalid("frequency summary tag differs")),
9562 })
9563}
9564
9565fn summary_span(cur: &mut Cursor<'_>) -> Result<Option<(usize, usize)>> {
9567 let length = cur.u32()? as usize;
9568 let entries = cur.u32()? as usize;
9569 if entries > FREQUENCY_ENTRIES {
9570 return Err(invalid("frequency entry count exceeds its bound"));
9571 }
9572 if length == 0 {
9573 if entries != 0 {
9574 return Err(invalid("missing frequency synopsis has entries"));
9575 }
9576 return Ok(None);
9577 }
9578 if length > MAX_DIRECTORY || length > cur.len().saturating_sub(cur.at) {
9579 return Err(invalid("frequency synopsis span is outside the directory"));
9580 }
9581 Ok(Some((length, entries)))
9582}
9583
9584fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9587 match cur.u8()? {
9588 0 => Ok(()),
9589 1 => {
9590 cur.skip(8)?;
9591 let entries = cur.u32()? as usize;
9592 if entries > FREQUENCY_ENTRIES {
9593 return Err(invalid("frequency entry count exceeds its bound"));
9594 }
9595 for _ in 0..entries {
9596 match cur.u8()? {
9597 0 => {}
9598 1 => cur.skip(16)?,
9599 2 => cur.skip(4)?,
9600 _ => return Err(invalid("frequency value tag differs")),
9601 }
9602 cur.skip(8)?;
9603 }
9604 let ordinals = cur.u32()? as usize;
9605 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9606 return Err(invalid("frequency ordinal count exceeds its bound"));
9607 }
9608 for _ in 0..ordinals {
9609 cur.var_u64()?;
9610 }
9611 if values {
9612 cur.skip(ordinals * 2)?;
9613 }
9614 Ok(())
9615 }
9616 _ => Err(invalid("frequency summary tag differs")),
9617 }
9618}
9619
9620fn quick_nonzero(
9624 mut cur: Cursor<'_>,
9625 name: &str,
9626 fields: &[Field],
9627 rows: usize,
9628 wanted: usize,
9629) -> Result<Option<u64>> {
9630 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9631 return Err(invalid("table directory differs from the catalog"));
9632 }
9633 let width = cur.u16()? as usize;
9634 if width != fields.len() {
9635 return Err(invalid("table directory width differs from the catalog"));
9636 }
9637 for field in fields {
9638 let stored =
9639 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9640 if &stored != field {
9641 return Err(invalid("table directory schema differs from the catalog"));
9642 }
9643 }
9644 let mut dictionaries = Vec::with_capacity(width);
9645 for field in fields {
9646 let held = match cur.u8()? {
9647 0 => false,
9648 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9649 cur.skip(20)?;
9650 true
9651 }
9652 _ => return Err(invalid("dictionary page tag differs")),
9653 };
9654 dictionaries.push(held);
9655 }
9656 for _ in 0..width {
9657 match cur.u8()? {
9658 0 => {}
9659 1 => cur.skip(8)?,
9660 _ => return Err(invalid("distinct count tag differs")),
9661 }
9662 }
9663 if cur.u64()? != rows as u64 {
9664 return Err(invalid("table row count differs from the catalog"));
9665 }
9666 let stripes = cur.u32()? as usize;
9667 let mut total = 0_usize;
9668 let mut nulls = 0_u64;
9669 for _ in 0..stripes {
9670 let parts = cur.u32()? as usize;
9671 if parts == 0 || parts > STRIPE_PARTS {
9672 return Err(invalid("stripe part count is outside its bound"));
9673 }
9674 let mut stripe_rows = 0_usize;
9675 for _ in 0..parts {
9676 stripe_rows = stripe_rows
9677 .checked_add(cur.u32()? as usize)
9678 .ok_or_else(|| invalid("stripe row count overflow"))?;
9679 }
9680 total =
9681 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9682 cur.skip(12 + width * 12)?;
9683 for (field, held) in fields.iter().zip(&dictionaries) {
9684 if coded_type(&field.ty) && *held {
9685 cur.skip(20)?;
9686 }
9687 }
9688 for _ in 0..width * 2 {
9689 match cur.u8()? {
9690 0 => {}
9691 1 => cur.skip(20)?,
9692 _ => return Err(invalid("stripe page tag differs")),
9693 }
9694 }
9695 for column in 0..width {
9696 cur.skip_bound()?;
9697 cur.skip_bound()?;
9698 let count = cur.u32()? as u64;
9699 if count > stripe_rows as u64 {
9700 return Err(invalid("null count exceeds stripe rows"));
9701 }
9702 if column == wanted {
9703 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9704 }
9705 cur.skip(1)?;
9706 match cur.u8()? {
9707 0 => {}
9708 1 => cur.skip(16)?,
9709 _ => return Err(invalid("a stripe sum has an unknown tag")),
9710 }
9711 }
9712 }
9713 if total != rows {
9714 return Err(invalid("table row count differs from stripes"));
9715 }
9716 if cur.done() {
9717 return Ok(None);
9718 }
9719 let magic = cur.take(8)?;
9720 let spanned = magic == FREQUENCIES_SPANS;
9721 let values = magic == FREQUENCIES || spanned;
9722 if !values && magic != FREQUENCIES_V2 {
9723 return Err(invalid("directory extension magic differs"));
9724 }
9725 if cur.u16()? as usize != width {
9726 return Err(invalid("frequency column count differs"));
9727 }
9728 for _ in 0..wanted {
9729 if spanned {
9730 if let Some((length, _)) = summary_span(&mut cur)? {
9731 cur.skip(length)?;
9732 }
9733 } else {
9734 skip_summary(&mut cur, values, rows)?;
9735 }
9736 }
9737 let summary = if spanned {
9738 let Some((length, entries)) = summary_span(&mut cur)? else {
9739 return Ok(None);
9740 };
9741 let start = cur.at;
9742 let summary = decode_summary(&mut cur, &fields[wanted], rows, values)?
9743 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
9744 if cur.at - start != length || summary.entries.len() != entries {
9745 return Err(invalid("a stored synopsis differs from its directory span"));
9746 }
9747 summary
9748 } else {
9749 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9750 return Ok(None);
9751 };
9752 summary
9753 };
9754 let zero = summary
9755 .entries
9756 .iter()
9757 .find(|entry| entry.value == FrequencyValue::Integer(0))
9758 .map(|entry| entry.count)
9759 .or_else(|| (summary.omitted_max == 0).then_some(0));
9760 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9761}
9762
9763fn quick_integer_fold(
9766 file: &File,
9767 mut cur: Cursor<'_>,
9768 entry: &Entry,
9769 size: u64,
9770 wanted: usize,
9771 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9772) -> Result<()> {
9773 let name = &entry.name;
9774 let fields = &entry.fields;
9775 let rows = entry.rows;
9776 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9777 return Err(invalid("table directory differs from the catalog"));
9778 }
9779 let width = cur.u16()? as usize;
9780 if width != fields.len() {
9781 return Err(invalid("table directory width differs from the catalog"));
9782 }
9783 for field in fields {
9784 let stored =
9785 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9786 if &stored != field {
9787 return Err(invalid("table directory schema differs from the catalog"));
9788 }
9789 }
9790 let mut dictionaries = Vec::with_capacity(width);
9791 for field in fields {
9792 dictionaries.push(match cur.u8()? {
9793 0 => false,
9794 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9795 cur.skip(20)?;
9796 true
9797 }
9798 _ => return Err(invalid("dictionary page tag differs")),
9799 });
9800 }
9801 for _ in 0..width {
9802 match cur.u8()? {
9803 0 => {}
9804 1 => cur.skip(8)?,
9805 _ => return Err(invalid("distinct count tag differs")),
9806 }
9807 }
9808 if cur.u64()? != rows as u64 {
9809 return Err(invalid("table row count differs from the catalog"));
9810 }
9811 let stripes = cur.u32()? as usize;
9812 let mut total = 0_usize;
9813 let mut bytes = Vec::new();
9814 for _ in 0..stripes {
9815 let parts = cur.u32()? as usize;
9816 if parts == 0 || parts > STRIPE_PARTS {
9817 return Err(invalid("stripe part count is outside its bound"));
9818 }
9819 let mut part_rows = Vec::with_capacity(parts);
9820 for _ in 0..parts {
9821 let count = cur.u32()? as usize;
9822 if count == 0 {
9823 return Err(invalid("empty part"));
9824 }
9825 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9826 part_rows.push(count);
9827 }
9828 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9829 let section = index_section(parts)?;
9830 let index_length =
9831 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9832 if index.offset < HEADER
9833 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9834 || index.length as usize != index_length
9835 {
9836 return Err(invalid("index page range is outside the file"));
9837 }
9838 cur.skip(wanted * 12)?;
9839 let page = Span { offset: cur.u64()?, length: cur.u32()? };
9840 if page.offset < HEADER
9841 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9842 || page.length as usize > MAX_PAGE
9843 {
9844 return Err(invalid("column page range is outside the file"));
9845 }
9846 cur.skip((width - wanted - 1) * 12)?;
9847 for (field, held) in fields.iter().zip(&dictionaries) {
9848 if coded_type(&field.ty) && *held {
9849 cur.skip(20)?;
9850 }
9851 }
9852 for _ in 0..width * 2 {
9853 match cur.u8()? {
9854 0 => {}
9855 1 => cur.skip(20)?,
9856 _ => return Err(invalid("stripe page tag differs")),
9857 }
9858 }
9859 for _ in 0..width {
9860 cur.skip_bound()?;
9861 cur.skip_bound()?;
9862 cur.skip(5)?;
9863 match cur.u8()? {
9864 0 => {}
9865 1 => cur.skip(16)?,
9866 _ => return Err(invalid("a stripe sum has an unknown tag")),
9867 }
9868 }
9869 let spans = read_index_span(file, index, page, parts, wanted)?;
9870 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9871 bytes.resize(span.length, 0);
9872 let at = page
9873 .offset
9874 .checked_add(span.start as u64)
9875 .ok_or_else(|| invalid("part range overflow"))?;
9876 read_at(file, at, &mut bytes)?;
9877 if checksum(&bytes) != span.hash {
9878 return Err(invalid("integer part checksum differs"));
9879 }
9880 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9881 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9882 check_integer_tally_value(value, &fields[wanted].ty)?;
9883 emit(value, count)
9884 })?;
9885 if decoded_rows != expected_rows {
9886 return Err(invalid("encoded integer part holds the wrong number of rows"));
9887 }
9888 } else {
9889 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9890 if let Some(packed) = column.packed_parts() {
9891 let validity = column.validity();
9892 let all_valid = column.none_null();
9893 let base = packed.base();
9894 let mut codes = [0_u64; 64];
9895 for from in (0..expected_rows).step_by(codes.len()) {
9896 let count = (expected_rows - from).min(codes.len());
9897 packed.unpack(from, &mut codes[..count]);
9898 for (offset, &code) in codes[..count].iter().enumerate() {
9899 if all_valid || validity.is_valid(from + offset) {
9900 emit((base + i128::from(code)) as i64, 1)?;
9902 }
9903 }
9904 }
9905 continue;
9906 }
9907 let column = column.into_flat()?;
9908 let validity = column.validity();
9909 macro_rules! count_decoded {
9910 ($values:expr) => {
9911 for (row, &value) in $values.as_slice().iter().enumerate() {
9912 if validity.is_valid(row) {
9913 emit(i64::from(value), 1)?;
9914 }
9915 }
9916 };
9917 }
9918 match column.data() {
9919 Some(Data::Int8(values)) => count_decoded!(values),
9920 Some(Data::Int16(values)) => count_decoded!(values),
9921 Some(Data::Int32(values)) => count_decoded!(values),
9922 Some(Data::Int64(values)) => count_decoded!(values),
9923 _ => return Err(invalid("decoded integer part has the wrong type")),
9924 }
9925 }
9926 }
9927 }
9928 if total != rows {
9929 return Err(invalid("table row count differs from stripes"));
9930 }
9931 Ok(())
9932}
9933
9934fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9935 let fits = match ty {
9936 LogicalType::TinyInt => i8::try_from(value).is_ok(),
9937 LogicalType::SmallInt => i16::try_from(value).is_ok(),
9938 LogicalType::Integer => i32::try_from(value).is_ok(),
9939 LogicalType::BigInt => true,
9940 _ => false,
9941 };
9942 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9943}
9944
9945fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9946 read_directory(Cursor::new(bytes), size, None)
9947}
9948
9949fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9954 if cur.take(8)? != DIRECTORY {
9955 return Err(invalid("directory magic differs"));
9956 }
9957 let name = cur.text()?;
9958 let width = cur.u16()? as usize;
9959 let mut fields = Vec::with_capacity(width);
9960 for _ in 0..width {
9961 let name = cur.text()?;
9962 let ty = read_type(&mut cur)?;
9963 let not_null = match cur.u8()? {
9964 0 => false,
9965 1 => true,
9966 _ => return Err(invalid("nullability flag differs")),
9967 };
9968 fields.push(Field { name, ty, not_null });
9969 }
9970 let mut dictionaries = Vec::with_capacity(width);
9971 for field in &fields {
9972 dictionaries.push(match cur.u8()? {
9973 0 => None,
9974 tag if tag == dictionary_tag(&field.ty) => {
9975 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9976 let end = page
9977 .offset
9978 .checked_add(u64::from(page.length))
9979 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9980 if page.offset < HEADER || end > size {
9985 return Err(invalid("dictionary page range is outside the file"));
9986 }
9987 Some(page)
9988 }
9989 _ => return Err(invalid("dictionary page tag differs")),
9990 });
9991 }
9992 let mut distincts = Vec::with_capacity(width);
9993 for _ in 0..width {
9994 distincts.push(match cur.u8()? {
9995 0 => None,
9996 1 => Some(cur.u64()?),
9997 _ => return Err(invalid("distinct count tag differs")),
9998 });
9999 }
10000 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
10001 let count = cur.u32()? as usize;
10002 let mut stripes = Vec::with_capacity(count);
10003 let mut total = 0_usize;
10004 for _ in 0..count {
10005 let count = cur.u32()? as usize;
10006 if count == 0 || count > STRIPE_PARTS {
10007 return Err(invalid("stripe part count is outside its bound"));
10008 }
10009 let mut parts = Vec::with_capacity(count);
10010 let mut stripe_rows = 0_usize;
10011 for _ in 0..count {
10012 let rows = cur.u32()?;
10013 if rows == 0 {
10014 return Err(invalid("empty part"));
10015 }
10016 parts.push(rows);
10017 stripe_rows = stripe_rows
10018 .checked_add(rows as usize)
10019 .ok_or_else(|| invalid("stripe row count overflow"))?;
10020 }
10021 total =
10022 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
10023 let index = Span { offset: cur.u64()?, length: cur.u32()? };
10024 let section = index_section(count)?;
10025 let wanted = section
10026 .checked_mul(width)
10027 .and_then(|bytes| u32::try_from(bytes).ok())
10028 .ok_or_else(|| invalid("index page length overflow"))?;
10029 let end = index
10030 .offset
10031 .checked_add(u64::from(index.length))
10032 .ok_or_else(|| invalid("index page offset overflow"))?;
10033 if index.offset < HEADER || end > size || index.length != wanted {
10034 return Err(invalid("index page range is outside the file"));
10035 }
10036 let mut pages = Vec::with_capacity(width);
10037 for _ in 0..width {
10038 let offset = cur.u64()?;
10039 let length = cur.u32()?;
10040 let end = offset
10041 .checked_add(u64::from(length))
10042 .ok_or_else(|| invalid("page offset overflow"))?;
10043 if offset < HEADER || end > size || length as usize > MAX_PAGE {
10044 return Err(invalid("page range is outside the file"));
10045 }
10046 pages.push(Span { offset, length });
10047 }
10048 let mut memberships = vec![None; width];
10049 for (column, field) in fields.iter().enumerate() {
10050 if !coded_type(&field.ty) || dictionaries[column].is_none() {
10051 continue;
10052 }
10053 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10054 let end = page
10055 .offset
10056 .checked_add(u64::from(page.length))
10057 .ok_or_else(|| invalid("membership page offset overflow"))?;
10058 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10059 return Err(invalid("membership page range is outside the file"));
10060 }
10061 if page.length != 0 {
10064 memberships[column] = Some(page);
10065 }
10066 }
10067 let mut sieves = vec![None; width];
10068 for sieve in sieves.iter_mut().take(width) {
10069 match cur.u8()? {
10070 0 => continue,
10071 1 => {}
10072 _ => return Err(invalid("a sieve page has an unknown tag")),
10073 }
10074 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10075 let end = page
10076 .offset
10077 .checked_add(u64::from(page.length))
10078 .ok_or_else(|| invalid("sieve page offset overflow"))?;
10079 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10080 return Err(invalid("sieve page range is outside the file"));
10081 }
10082 *sieve = Some(page);
10083 }
10084 let mut part_ranges = vec![None; width];
10085 for held in part_ranges.iter_mut().take(width) {
10086 match cur.u8()? {
10087 0 => continue,
10088 1 => {}
10089 _ => return Err(invalid("a part range page has an unknown tag")),
10090 }
10091 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
10092 let end = page
10093 .offset
10094 .checked_add(u64::from(page.length))
10095 .ok_or_else(|| invalid("part range page offset overflow"))?;
10096 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
10097 return Err(invalid("part range page range is outside the file"));
10098 }
10099 *held = Some(page);
10100 }
10101 let mut ranges = Vec::with_capacity(width);
10102 for column in 0..width {
10103 let low = cur.bound()?;
10104 let high = cur.bound()?;
10105 let nulls = cur.u32()? as usize;
10106 if nulls > stripe_rows {
10107 return Err(invalid("null count exceeds stripe rows"));
10108 }
10109 let exact = cur.u8()? != 0;
10110 let sum = match cur.u8()? {
10111 0 => None,
10112 1 => Some(i128::from_le_bytes(
10113 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
10114 )),
10115 _ => return Err(invalid("a stripe sum has an unknown tag")),
10116 };
10117 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
10123 let low = low.map(|bound| scaled_as(bound, ty));
10124 let high = high.map(|bound| scaled_as(bound, ty));
10125 ranges.push(Range { low, high, nulls, exact, sum });
10126 }
10127 stripes.push(Stripe {
10128 rows: stripe_rows,
10129 parts,
10130 index,
10131 pages,
10132 memberships: Pages::from_slots(memberships)?,
10133 sieves: Pages::from_slots(sieves)?,
10134 part_ranges: Pages::from_slots(part_ranges)?,
10135 zone: Zone::from_ranges(ranges),
10136 });
10137 }
10138 if total != rows {
10139 return Err(invalid("table row count differs from stripes"));
10140 }
10141 let mut entry_counts = vec![0; width];
10144 let frequencies = if cur.done() {
10145 vec![None; width]
10146 } else {
10147 let frequency_magic = cur.take(8)?;
10148 let spanned = frequency_magic == FREQUENCIES_SPANS;
10149 let frequency_values = frequency_magic == FREQUENCIES || spanned;
10150 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
10151 return Err(invalid("directory extension magic differs"));
10152 }
10153 if cur.u16()? as usize != width {
10154 return Err(invalid("frequency column count differs"));
10155 }
10156 let mut frequencies = Vec::with_capacity(width);
10157 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
10158 if spanned {
10159 let Some((length, entries)) = summary_span(&mut cur)? else {
10160 frequencies.push(None);
10161 continue;
10162 };
10163 *entry_count = entries;
10164 let start = cur.at;
10165 if let Some(offset) = stored_at {
10166 cur.skip(length)?;
10167 frequencies.push(Some(Frequencies::Stored {
10168 span: Span {
10169 offset: offset
10170 .checked_add(start as u64)
10171 .ok_or_else(|| invalid("frequency synopsis offset overflow"))?,
10172 length: u32::try_from(length)
10173 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10174 },
10175 values: true,
10176 entries,
10177 }));
10178 } else {
10179 let summary = decode_summary(&mut cur, field, rows, true)?
10180 .ok_or_else(|| invalid("a stored synopsis is missing"))?;
10181 if cur.at - start != length || summary.entries.len() != entries {
10182 return Err(invalid("a stored synopsis differs from its directory span"));
10183 }
10184 frequencies.push(Some(Frequencies::Held(summary)));
10185 }
10186 continue;
10187 }
10188 let start = cur.at;
10189 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
10190 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
10191 frequencies.push(match (summary, stored_at) {
10192 (None, _) => None,
10193 (Some(summary), None) => Some(Frequencies::Held(summary)),
10194 (Some(summary), Some(offset)) => Some(Frequencies::Stored {
10195 span: Span {
10196 offset: offset + start as u64,
10197 length: u32::try_from(cur.at - start)
10198 .map_err(|_| invalid("a frequency synopsis is too long"))?,
10199 },
10200 values: frequency_values,
10201 entries: summary.entries.len(),
10202 }),
10203 });
10204 }
10205 frequencies
10206 };
10207 let mut clustering = None;
10217 let mut sections = Vec::new();
10218 let mut pair_frequencies = Vec::new();
10219 let mut seen_pair_frequencies = false;
10220 let mut frequency_texts = vec![Vec::new(); width];
10221 let mut seen_frequency_texts = false;
10222 let mut host_groups = None;
10223 let mut demoted = Vec::new();
10224 let mut seen_sections = false;
10225 let mut dictionary_payloads = Vec::new();
10226 let mut seen_payloads = false;
10227 let mut constraints = Constraints::default();
10228 let mut generation = 0;
10231 while !cur.done() {
10232 let mut tag = [0u8; 8];
10233 tag.copy_from_slice(cur.take(8)?);
10234 if &tag == PAIR_FREQUENCIES {
10235 if seen_pair_frequencies {
10236 return Err(invalid("directory names two pair frequency blocks"));
10237 }
10238 seen_pair_frequencies = true;
10239 let count = cur.u16()? as usize;
10240 if count > MAX_PAIR_FREQUENCIES {
10241 return Err(invalid("pair frequency count exceeds its bound"));
10242 }
10243 pair_frequencies = Vec::with_capacity(count);
10244 for _ in 0..count {
10245 let first = cur.u16()?;
10246 let second = cur.u16()?;
10247 let first_at = first as usize;
10248 let second_at = second as usize;
10249 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
10250 return Err(invalid("pair frequency first column has no synopsis"));
10251 }
10252 let first_entries = entry_counts[first_at];
10253 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
10254 || dictionaries.get(second_at).copied().flatten().is_none()
10255 {
10256 return Err(invalid("pair frequency second column has no stable dictionary"));
10257 }
10258 if pair_frequencies
10259 .iter()
10260 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
10261 {
10262 return Err(invalid("directory repeats a pair frequency summary"));
10263 }
10264 let omitted_max = cur.u64()?;
10265 if omitted_max > rows as u64 {
10266 return Err(invalid("pair frequency omitted count exceeds the table"));
10267 }
10268 let entries_count = cur.u16()? as usize;
10269 if entries_count > FREQUENCY_ENTRIES {
10270 return Err(invalid("pair frequency entry count exceeds its bound"));
10271 }
10272 let mut entries = Vec::with_capacity(entries_count);
10273 for _ in 0..entries_count {
10274 let first_entry = cur.u16()?;
10275 if first_entry as usize >= first_entries {
10276 return Err(invalid("pair frequency anchor is outside its synopsis"));
10277 }
10278 let second = match cur.u8()? {
10279 0 => None,
10280 1 => Some(cur.u32()?),
10281 _ => return Err(invalid("pair frequency string tag differs")),
10282 };
10283 let count = cur.u64()?;
10284 if count == 0 || count > rows as u64 {
10285 return Err(invalid("pair frequency count is outside the table"));
10286 }
10287 entries.push(PairFrequencyEntry { first_entry, second, count });
10288 }
10289 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
10290 return Err(invalid("pair frequency entries are not descending"));
10291 }
10292 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
10293 }
10294 } else if &tag == FREQUENCY_TEXTS {
10295 if seen_frequency_texts {
10296 return Err(invalid("directory names two frequency text blocks"));
10297 }
10298 seen_frequency_texts = true;
10299 let columns = cur.u16()? as usize;
10300 if columns > width {
10301 return Err(invalid("frequency text column count exceeds the schema"));
10302 }
10303 for _ in 0..columns {
10304 let column = cur.u16()? as usize;
10305 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
10306 return Err(invalid("frequency text column is repeated or out of range"));
10307 }
10308 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
10309 || dictionaries.get(column).copied().flatten().is_none()
10310 || frequencies.get(column).and_then(Option::as_ref).is_none()
10311 {
10312 return Err(invalid("frequency texts belong to a non-string synopsis"));
10313 }
10314 let count = cur.u16()? as usize;
10315 if count == 0 || count != entry_counts[column] {
10316 return Err(invalid("frequency text count differs from its synopsis"));
10317 }
10318 let mut texts = Vec::with_capacity(count);
10319 for _ in 0..count {
10320 texts.push(match cur.u8()? {
10321 0 => None,
10322 1 => {
10323 let length = cur.u32()? as usize;
10324 let bytes = cur.take(length)?.to_vec();
10325 if fields[column].ty == LogicalType::Varchar {
10326 std::str::from_utf8(&bytes)
10327 .map_err(|_| invalid("frequency text is not UTF-8"))?;
10328 }
10329 Some(bytes)
10330 }
10331 _ => return Err(invalid("frequency text tag differs")),
10332 });
10333 }
10334 frequency_texts[column] = texts;
10335 }
10336 } else if &tag == HOST_GROUPS {
10337 if host_groups.is_some() {
10338 return Err(invalid("directory names two host group blocks"));
10339 }
10340 let column = cur.u16()? as usize;
10341 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
10342 || dictionaries.get(column).copied().flatten().is_none()
10343 {
10344 return Err(invalid("host groups belong to a non-string dictionary"));
10345 }
10346 let omitted_max = cur.u64()?;
10347 if omitted_max > rows as u64 {
10348 return Err(invalid("host group bound exceeds the table"));
10349 }
10350 let count = cur.u16()? as usize;
10351 if count > host::CAPACITY {
10352 return Err(invalid("host group count exceeds its bound"));
10353 }
10354 let mut entries = Vec::with_capacity(count);
10355 let mut bytes = 0_usize;
10356 for _ in 0..count {
10357 let host_len = cur.u32()? as usize;
10358 bytes =
10359 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
10360 if bytes > host::BYTE_BUDGET {
10361 return Err(invalid("host groups exceed their byte budget"));
10362 }
10363 let host = std::str::from_utf8(cur.take(host_len)?)
10364 .map_err(|_| invalid("host is not UTF-8"))?
10365 .to_owned();
10366 let count = cur.u64()?;
10367 if count == 0 || count > rows as u64 {
10368 return Err(invalid("host group count exceeds the table"));
10369 }
10370 let bytes_sum = i128::from_le_bytes(
10371 cur.take(16)?
10372 .try_into()
10373 .map_err(|_| invalid("host length sum is truncated"))?,
10374 );
10375 if bytes_sum < 0 {
10376 return Err(invalid("host length sum is negative"));
10377 }
10378 let minimum_len = cur.u32()? as usize;
10379 bytes =
10380 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
10381 if bytes > host::BYTE_BUDGET {
10382 return Err(invalid("host groups exceed their byte budget"));
10383 }
10384 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
10385 .map_err(|_| invalid("host minimum is not UTF-8"))?
10386 .to_owned();
10387 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
10388 }
10389 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
10390 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
10391 {
10392 return Err(invalid("host groups are not in certified order"));
10393 }
10394 host_groups = Some(host::HostSummary { column, omitted_max, entries });
10395 } else if &tag == CLUSTERING {
10396 if clustering.is_some() {
10397 return Err(invalid("directory names two clustering declarations"));
10398 }
10399 let bucket = Width::from_tag(cur.u8()?)
10400 .ok_or_else(|| invalid("clustering width tag differs"))?;
10401 let count = cur.u16()? as usize;
10402 let mut columns = Vec::with_capacity(count.min(fields.len()));
10403 for _ in 0..count {
10404 columns.push(u32::from(cur.u16()?));
10405 }
10406 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
10409 invalid("stored clustering declaration does not match the table it is on")
10410 })?);
10411 } else if &tag == DEMOTED {
10412 if !demoted.is_empty() {
10413 return Err(invalid("directory names two demoted column blocks"));
10414 }
10415 let count = cur.u16()? as usize;
10416 if count == 0 || count > width {
10417 return Err(invalid("demoted column count is outside the schema"));
10418 }
10419 demoted = vec![false; width];
10420 for _ in 0..count {
10421 let column = cur.u16()? as usize;
10422 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
10423 return Err(invalid("a demoted column is repeated or has no dictionary"));
10424 }
10425 demoted[column] = true;
10426 }
10427 } else if &tag == SECTIONS {
10428 if seen_sections {
10429 return Err(invalid("directory names two section tables"));
10430 }
10431 seen_sections = true;
10432 generation = cur.u64()?;
10433 let count = cur.u16()? as usize;
10434 if count > MAX_SECTIONS {
10435 return Err(invalid("section count exceeds its bound"));
10436 }
10437 sections = Vec::with_capacity(count);
10438 for _ in 0..count {
10441 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
10442 }
10443 for held in §ions {
10444 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
10445 return Err(invalid("a section's extent table overflows the file"));
10446 };
10447 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
10451 return Err(invalid("a section's extent table is outside the file"));
10452 }
10453 if held.extents == 0 && held.extent_bytes != 0 {
10454 return Err(invalid("a section with no extents names an extent table"));
10455 }
10456 }
10457 } else if &tag == DICTIONARY_PAYLOADS {
10458 if seen_payloads {
10459 return Err(invalid("directory names two dictionary payload blocks"));
10460 }
10461 seen_payloads = true;
10462 let count = cur.u16()? as usize;
10463 if count != fields.len() {
10464 return Err(invalid("dictionary payload block does not match the table's columns"));
10465 }
10466 dictionary_payloads = Vec::with_capacity(count);
10467 for _ in 0..count {
10468 let bytes = cur.u64()?;
10469 if bytes > size {
10470 return Err(invalid("a dictionary payload is larger than the file"));
10471 }
10472 dictionary_payloads.push(bytes);
10473 }
10474 } else if &tag == KEYS {
10475 if !constraints.is_empty() {
10476 return Err(invalid("directory names two key blocks"));
10477 }
10478 let fits = |columns: &[u16]| {
10479 !columns.is_empty() && columns.iter().all(|&column| usize::from(column) < width)
10480 };
10481 let count = cur.u16()? as usize;
10482 for _ in 0..count {
10483 let primary = cur.u8()? != 0;
10484 let columns = columns_of(&mut cur)?;
10485 if !fits(&columns) {
10486 return Err(invalid("a stored key names a column the table does not have"));
10487 }
10488 constraints.keys.push((columns, primary));
10489 }
10490 let count = cur.u16()? as usize;
10491 for _ in 0..count {
10492 let columns = columns_of(&mut cur)?;
10493 let referenced = columns_of(&mut cur)?;
10494 let len = cur.u32()? as usize;
10495 let table = std::str::from_utf8(cur.take(len)?)
10496 .map_err(|_| invalid("a foreign key's table name is not UTF-8"))?
10497 .to_owned();
10498 if !fits(&columns) || referenced.len() != columns.len() || table.is_empty() {
10499 return Err(invalid("a stored foreign key does not match its table"));
10500 }
10501 constraints.foreign.push(StoredForeign { columns, table, referenced });
10502 }
10503 if constraints.is_empty() {
10504 return Err(invalid("a key block holds no key"));
10505 }
10506 } else {
10507 return Err(invalid("directory extension magic differs"));
10508 }
10509 }
10510 if !cur.done() {
10511 return Err(invalid("directory has trailing bytes"));
10512 }
10513 for stripe in &stripes {
10514 for (column, field) in fields.iter().enumerate() {
10515 if coded_type(&field.ty)
10516 && dictionaries[column].is_some()
10517 && stripe.memberships.get(column).is_none()
10518 && !demoted.get(column).copied().unwrap_or(false)
10519 {
10520 return Err(invalid("string page has no code membership index"));
10521 }
10522 }
10523 }
10524 Ok(Table {
10525 name,
10526 fields,
10527 stripes,
10528 rows,
10529 dictionaries,
10530 dictionary_payloads,
10531 demoted,
10532 distincts,
10533 frequencies,
10534 pair_frequencies,
10535 frequency_texts,
10536 host_groups,
10537 clustering,
10538 generation,
10539 sections,
10540 constraints,
10541 })
10542}
10543
10544fn put_count(out: &mut Vec<u8>, count: usize) -> Result<()> {
10546 put_u16(out, u16::try_from(count).map_err(|_| invalid("too many constraints"))?);
10547 Ok(())
10548}
10549
10550fn put_columns(out: &mut Vec<u8>, columns: &[u16]) -> Result<()> {
10552 put_count(out, columns.len())?;
10553 for &column in columns {
10554 put_u16(out, column);
10555 }
10556 Ok(())
10557}
10558
10559fn columns_of(cur: &mut Cursor<'_>) -> Result<Vec<u16>> {
10561 let count = cur.u16()? as usize;
10562 (0..count).map(|_| cur.u16()).collect()
10563}
10564
10565fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10567 bounds::put(out, bound)
10568}
10569
10570#[derive(Debug)]
10587struct Codes;
10588
10589impl chooser::Chooser for Codes {
10590 fn name(&self) -> &'static str {
10591 "codes"
10592 }
10593
10594 fn narrow_strings(
10595 &self,
10596 _values: &[&[u8]],
10597 offered: &[string::Kind],
10598 _depth: u8,
10599 ) -> Vec<string::Kind> {
10600 offered.to_vec()
10603 }
10604
10605 fn narrow_integers(
10606 &self,
10607 _values: &[i64],
10608 offered: &[integer::Kind],
10609 depth: u8,
10610 ) -> Vec<integer::Kind> {
10611 narrowed_to(Codes::keep(depth), offered)
10614 }
10615
10616 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10617 Codes::keep(depth).contains(&kind)
10618 }
10619}
10620
10621impl Codes {
10622 fn keep(depth: u8) -> &'static [integer::Kind] {
10623 if depth == 0 {
10624 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10625 } else {
10626 &[integer::Kind::Constant, integer::Kind::Packed]
10627 }
10628 }
10629}
10630
10631fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10639 let narrowed: Vec<integer::Kind> =
10640 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10641 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10642}
10643
10644#[derive(Debug)]
10656struct Fixed;
10657
10658impl chooser::Chooser for Fixed {
10659 fn name(&self) -> &'static str {
10660 "fixed"
10661 }
10662
10663 fn narrow_strings(
10664 &self,
10665 _values: &[&[u8]],
10666 offered: &[string::Kind],
10667 _depth: u8,
10668 ) -> Vec<string::Kind> {
10669 offered.to_vec()
10670 }
10671
10672 fn narrow_integers(
10673 &self,
10674 _values: &[i64],
10675 offered: &[integer::Kind],
10676 depth: u8,
10677 ) -> Vec<integer::Kind> {
10678 narrowed_to(Fixed::keep(depth), offered)
10679 }
10680
10681 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10682 Fixed::keep(depth).contains(&kind)
10683 }
10684}
10685
10686impl Fixed {
10687 fn keep(depth: u8) -> &'static [integer::Kind] {
10688 if depth == 0 {
10689 &[
10690 integer::Kind::Constant,
10691 integer::Kind::Packed,
10692 integer::Kind::Delta,
10693 integer::Kind::Rle,
10694 integer::Kind::Sparse,
10695 integer::Kind::Strided,
10696 ]
10697 } else {
10698 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10699 }
10700 }
10701}
10702
10703fn widened(data: &Data) -> Option<Vec<i64>> {
10710 match data {
10711 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10712 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10713 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10714 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10715 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10716 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10717 Data::Int64(values) => Some(values.to_vec()),
10718 _ => None,
10719 }
10720}
10721
10722fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10728 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10729 let values = integer::decode_as::<T>(bytes)
10730 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10731 if values.len() != rows {
10732 return Err(invalid("cascade page holds the wrong number of rows"));
10733 }
10734 Ok(values)
10735 }
10736 Ok(match ty {
10737 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10738 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10739 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10740 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10741 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10742 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10743 LogicalType::BigInt
10744 | LogicalType::Timestamp
10745 | LogicalType::Time
10746 | LogicalType::TimeTz
10747 | LogicalType::TimestampTz
10748 | LogicalType::TimestampS
10749 | LogicalType::TimestampMs
10750 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10751 LogicalType::Decimal { .. } => match ty.physical() {
10754 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10755 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10756 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10757 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10758 },
10759 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10760 })
10761}
10762
10763fn plain_width(ty: &LogicalType) -> Option<usize> {
10766 Some(match ty {
10767 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10768 LogicalType::SmallInt | LogicalType::USmallInt => 2,
10769 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10770 LogicalType::BigInt
10771 | LogicalType::Timestamp
10772 | LogicalType::Time
10773 | LogicalType::TimeTz
10774 | LogicalType::TimestampTz
10775 | LogicalType::TimestampS
10776 | LogicalType::TimestampMs
10777 | LogicalType::TimestampNs => 8,
10778 LogicalType::Decimal { .. } => match ty.physical() {
10779 PhysicalType::Int16 => 2,
10780 PhysicalType::Int32 => 4,
10781 PhysicalType::Int64 => 8,
10782 _ => return None,
10785 },
10786 _ => return None,
10787 })
10788}
10789
10790fn cascaded(
10796 flat: &Vector,
10797 ty: &LogicalType,
10798 packed: Option<&Packed<'_>>,
10799 settling: &mut Settling,
10800) -> Result<Option<Vec<u8>>> {
10801 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10802 let Some(values) = widened(data) else { return Ok(None) };
10803 let plain = values.len().saturating_mul(width);
10804 let best = match packed {
10805 Some(packed) => plain.min(21 + size_of_val(packed.words())),
10807 None => plain,
10808 };
10809 let out = settling.encode(&values)?;
10810 Ok((out.len() < best).then_some(out))
10811}
10812
10813const SEARCH_EVERY: usize = 16;
10820
10821#[derive(Debug, Default)]
10827struct Settling {
10828 shape: Option<Shape>,
10831 since: usize,
10833 symbols: Option<Symbols>,
10835}
10836
10837#[derive(Debug)]
10840struct Symbols {
10841 shape: chooser::Settled,
10842 len: usize,
10845 payload: usize,
10846 since: usize,
10847}
10848
10849impl Settling {
10850 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10858 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10859 {
10860 let out = string::encode_fsst(values, &symbols.shape)?;
10861 let held = match &out {
10863 None => symbols.len == 0,
10864 Some(out) => {
10865 (out.len() as u128) * (symbols.payload as u128) * 4
10866 <= (symbols.len as u128) * (payload as u128) * 5
10867 }
10868 };
10869 if held {
10870 symbols.since += 1;
10871 return Ok(out);
10872 }
10873 }
10874 let shape = string::fsst_shape(values);
10875 let out = string::encode_fsst(values, &shape)?;
10876 let len = out.as_ref().map_or(0, Vec::len);
10877 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10878 Ok(out)
10879 }
10880
10881 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10888 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10889 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10890 let out = integer::encode_with(values, &replay)?;
10891 if !replay.held() {
10892 self.settle(&out, values.len(), replay.first_offered())?;
10893 return Ok(out);
10894 }
10895 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10896 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10897 self.since += 1;
10898 return Ok(out);
10899 }
10900 }
10901 let search = chooser::Replay::new(&[], &Fixed);
10903 let out = integer::encode_with(values, &search)?;
10904 self.settle(&out, values.len(), search.first_offered())?;
10905 Ok(out)
10906 }
10907
10908 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10909 let kinds = integer::shape(out)?;
10910 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10911 self.since = 0;
10912 Ok(())
10913 }
10914}
10915
10916#[derive(Debug)]
10918struct Shape {
10919 kinds: Vec<integer::Kind>,
10920 offered: Vec<integer::Kind>,
10921 len: usize,
10922 rows: usize,
10923}
10924
10925fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10966 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10967 let mut payload = 0_usize;
10968 for row in 0..flat.len() {
10969 let text = flat.bytes_at(row).unwrap_or(b"");
10972 payload = payload.saturating_add(text.len());
10973 values.push(text);
10974 }
10975 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10977 let Some(out) = settling.text(&values, payload)? else {
10978 return Ok(None);
10979 };
10980 Ok((out.len() < plain).then_some(out))
10981}
10982
10983fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10984 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10985 let coded = integer::encode_with(&wide, &Codes)?;
10986 let plain = codes.len().saturating_mul(size_of::<u32>());
10987 Ok((coded.len() < plain).then_some(coded))
10988}
10989
10990fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10993 let flag = match flat.validity() {
10994 Validity::AllValid => 0,
10995 Validity::AllInvalid => 1,
10996 Validity::Mask(_) => 2,
10997 };
10998 out.push(flag);
10999 if flag == 2 {
11000 for group in (0..flat.len()).step_by(8) {
11001 let mut bits = 0_u8;
11002 for bit in 0..8 {
11003 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
11004 bits |= 1 << bit;
11005 }
11006 }
11007 out.push(bits);
11008 }
11009 }
11010}
11011
11012fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
11019 let coded = encoded_codes(codes)?;
11020 let mut out = Vec::with_capacity(
11021 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
11022 );
11023 out.push(if coded.is_some() { 4 } else { 3 });
11024 out.extend_from_slice(validity);
11025 match coded {
11026 Some(coded) => out.extend_from_slice(&coded),
11027 None => {
11028 for &code in codes {
11029 put_u32(&mut out, code);
11030 }
11031 }
11032 }
11033 Ok(out)
11034}
11035
11036fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
11039 let ty = vector.logical_type();
11040 let flat = vector.flatten()?;
11042 let mut out = Vec::new();
11043 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
11044 let compressed_text = if dictionary.is_none() && coded_type(ty) {
11045 text_compressed(&flat, settling)?
11046 } else {
11047 None
11048 };
11049 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
11050 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
11051 let cascade =
11055 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
11056 out.push(if cascade.is_some() {
11057 5
11058 } else if dictionary.is_some() {
11059 1
11060 } else if compressed_text.is_some() {
11061 6
11062 } else if packed.is_some() {
11063 2
11064 } else {
11065 0
11066 });
11067 push_validity(&mut out, &flat);
11068 if let Some(cascade) = cascade {
11069 out.extend_from_slice(&cascade);
11070 return Ok(out);
11071 }
11072 if let Some(dictionary) = dictionary {
11073 out.extend_from_slice(&dictionary);
11074 return Ok(out);
11075 }
11076 if let Some(compressed_text) = compressed_text {
11077 out.extend_from_slice(&compressed_text);
11078 return Ok(out);
11079 }
11080 if let Some(packed) = packed {
11081 if packed.offset() != 0 {
11082 return Err(invalid("writer received a sliced packed vector"));
11083 }
11084 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
11085 out.extend_from_slice(&packed.base().to_le_bytes());
11086 put_u32(
11087 &mut out,
11088 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
11089 );
11090 for word in packed.words() {
11091 put_u64(&mut out, *word);
11092 }
11093 return Ok(out);
11094 }
11095 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
11096 match (ty, data) {
11097 (LogicalType::TinyInt, Data::Int8(values)) => {
11098 for value in &**values {
11099 out.extend_from_slice(&value.to_le_bytes());
11100 }
11101 }
11102 (LogicalType::UTinyInt, Data::UInt8(values)) => {
11103 for value in &**values {
11104 out.extend_from_slice(&value.to_le_bytes());
11105 }
11106 }
11107 (LogicalType::SmallInt, Data::Int16(values)) => {
11108 for value in &**values {
11109 out.extend_from_slice(&value.to_le_bytes());
11110 }
11111 }
11112 (LogicalType::USmallInt, Data::UInt16(values)) => {
11113 for value in &**values {
11114 out.extend_from_slice(&value.to_le_bytes());
11115 }
11116 }
11117 (LogicalType::UInteger, Data::UInt32(values)) => {
11118 for value in &**values {
11119 out.extend_from_slice(&value.to_le_bytes());
11120 }
11121 }
11122 (LogicalType::UBigInt, Data::UInt64(values)) => {
11123 for value in &**values {
11124 out.extend_from_slice(&value.to_le_bytes());
11125 }
11126 }
11127 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
11128 for value in &**values {
11129 out.extend_from_slice(&value.to_le_bytes());
11130 }
11131 }
11132 (
11133 LogicalType::BigInt
11134 | LogicalType::Timestamp
11135 | LogicalType::Time
11136 | LogicalType::TimeTz
11137 | LogicalType::TimestampTz
11138 | LogicalType::TimestampS
11139 | LogicalType::TimestampMs
11140 | LogicalType::TimestampNs,
11141 Data::Int64(values),
11142 ) => {
11143 for value in &**values {
11144 out.extend_from_slice(&value.to_le_bytes());
11145 }
11146 }
11147 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
11150 for value in &**values {
11151 out.extend_from_slice(&value.to_le_bytes());
11152 }
11153 }
11154 (LogicalType::UHugeInt, Data::UInt128(values)) => {
11155 for value in &**values {
11156 out.extend_from_slice(&value.to_le_bytes());
11157 }
11158 }
11159 (LogicalType::Float, Data::Float32(values)) => {
11162 for value in &**values {
11163 out.extend_from_slice(&value.to_le_bytes());
11164 }
11165 }
11166 (LogicalType::Double, Data::Float64(values)) => {
11167 for value in &**values {
11168 out.extend_from_slice(&value.to_le_bytes());
11169 }
11170 }
11171 (LogicalType::Interval, Data::Interval(values)) => {
11175 for (months, days, micros) in &**values {
11176 out.extend_from_slice(&months.to_le_bytes());
11177 out.extend_from_slice(&days.to_le_bytes());
11178 out.extend_from_slice(µs.to_le_bytes());
11179 }
11180 }
11181 (LogicalType::Boolean, Data::Bool(values)) => {
11182 for value in &**values {
11183 out.push(u8::from(*value));
11184 }
11185 }
11186 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
11189 for value in &**values {
11190 out.extend_from_slice(&value.to_le_bytes());
11191 }
11192 }
11193 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
11194 for value in &**values {
11195 out.extend_from_slice(&value.to_le_bytes());
11196 }
11197 }
11198 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
11199 for value in &**values {
11200 out.extend_from_slice(&value.to_le_bytes());
11201 }
11202 }
11203 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
11204 for value in &**values {
11205 out.extend_from_slice(&value.to_le_bytes());
11206 }
11207 }
11208 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
11213 let mut bytes = Vec::new();
11214 put_u32(&mut out, 0);
11215 for row in 0..vector.len() {
11216 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
11217 bytes.extend_from_slice(value);
11218 put_u32(
11219 &mut out,
11220 u32::try_from(bytes.len())
11221 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
11222 );
11223 }
11224 out.extend_from_slice(&bytes);
11225 }
11226 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11227 }
11228 Ok(out)
11229}
11230
11231fn put_varint(out: &mut Vec<u8>, mut value: u32) {
11232 while value >= 0x80 {
11233 out.push((value as u8 & 0x7f) | 0x80);
11234 value >>= 7;
11235 }
11236 out.push(value as u8);
11237}
11238
11239fn unique_codes(codes: &[u32]) -> Vec<u32> {
11241 let mut unique = codes.to_vec();
11242 unique.sort_unstable();
11243 unique.dedup();
11244 unique
11245}
11246
11247fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
11253 let mut lists = lists;
11254 while lists.len() > 1 {
11255 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
11256 for pair in lists.chunks(2) {
11257 match pair {
11258 [left, right] => next.push(merged_pair(left, right)),
11259 [only] => next.push(only.clone()),
11260 _ => {}
11261 }
11262 }
11263 lists = next;
11264 }
11265 lists.pop().unwrap_or_default()
11266}
11267
11268fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
11269 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
11270 let mut at = 0;
11271 let mut to = 0;
11272 while at < left.len() && to < right.len() {
11273 match left[at].cmp(&right[to]) {
11274 Ordering::Less => {
11275 out.push(left[at]);
11276 at += 1;
11277 }
11278 Ordering::Greater => {
11279 out.push(right[to]);
11280 to += 1;
11281 }
11282 Ordering::Equal => {
11283 out.push(left[at]);
11284 at += 1;
11285 to += 1;
11286 }
11287 }
11288 }
11289 out.extend_from_slice(&left[at..]);
11290 out.extend_from_slice(&right[to..]);
11291 out
11292}
11293
11294fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
11299 let mut merged = Range::default();
11300 let mut first = true;
11301 for range in ranges {
11302 merged.nulls = merged.nulls.saturating_add(range.nulls);
11303 merged.sum = match (merged.sum.take(), range.sum) {
11307 (Some(held), Some(next)) if !first => held.checked_add(next),
11308 (_, next) if first => next,
11309 _ => None,
11310 };
11311 merged.exact = if first { range.exact } else { merged.exact && range.exact };
11312 if first {
11313 merged.low = range.low;
11314 merged.high = range.high;
11315 first = false;
11316 continue;
11317 }
11318 merged.low = match (merged.low.take(), range.low) {
11319 (Some(held), Some(next)) => Some(held.smaller(next)),
11320 _ => None,
11321 };
11322 merged.high = match (merged.high.take(), range.high) {
11323 (Some(held), Some(next)) => Some(held.larger(next)),
11324 _ => None,
11325 };
11326 }
11327 merged
11328}
11329
11330fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
11343 match bound {
11344 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
11345 value.truncate(PART_BOUND_BYTES);
11346 if !high {
11347 return Some(Bound::Bytes(value));
11348 }
11349 while let Some(last) = value.pop() {
11350 if last < u8::MAX {
11351 value.push(last + 1);
11352 return Some(Bound::Bytes(value));
11353 }
11354 }
11355 None
11356 }
11357 other => other,
11358 }
11359}
11360
11361fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
11369 let mut out = Vec::new();
11370 put_u32(
11371 &mut out,
11372 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11373 );
11374 for range in ranges {
11375 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
11376 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
11377 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
11378 }
11379 Ok(out)
11380}
11381
11382fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
11384 let mut cur = Cursor::new(bytes);
11385 let parts = cur.u32()? as usize;
11386 let mut out = Vec::new();
11387 for _ in 0..parts {
11388 let low = cur.bound()?;
11389 let high = cur.bound()?;
11390 let nulls = cur.u32()? as usize;
11391 out.push(Range { low, high, nulls, exact: false, sum: None });
11392 }
11393 Ok(out)
11394}
11395
11396fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
11397 let held: Vec<&Option<Sieve>> = sieves.collect();
11398 let mut out = Vec::new();
11399 put_u32(
11400 &mut out,
11401 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
11402 );
11403 for sieve in &held {
11404 let length = sieve.as_ref().map_or(0, Sieve::len);
11405 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
11406 }
11407 for sieve in held.into_iter().flatten() {
11409 out.extend_from_slice(&sieve.to_bytes());
11410 }
11411 Ok(out)
11412}
11413
11414fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
11420 let parts = u32::from_le_bytes(
11421 bytes
11422 .get(..4)
11423 .ok_or_else(|| invalid("sieve page is truncated"))?
11424 .try_into()
11425 .map_err(|_| invalid("sieve page is truncated"))?,
11426 ) as usize;
11427 let mut lengths = Vec::with_capacity(parts);
11428 for part in 0..parts {
11429 let at = 4 + part * 4;
11430 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
11431 lengths.push(u32::from_le_bytes(
11432 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
11433 ) as usize);
11434 }
11435 let mut at = 4 + parts * 4;
11436 let mut out = Vec::with_capacity(parts);
11437 for length in lengths {
11438 if length == 0 {
11439 out.push(None);
11440 continue;
11441 }
11442 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
11443 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
11444 out.push(Sieve::from_bytes(field));
11445 at = end;
11446 }
11447 if at != bytes.len() {
11448 return Err(invalid("sieve page has trailing bytes"));
11449 }
11450 Ok(out)
11451}
11452
11453fn encode_membership(unique: &[u32]) -> Vec<u8> {
11459 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
11460 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
11461 let mut previous = 0;
11462 for (at, &code) in unique.iter().enumerate() {
11463 put_varint(&mut out, if at == 0 { code } else { code - previous });
11464 previous = code;
11465 }
11466 out
11467}
11468
11469fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
11470 let mut value = 0_u32;
11471 for shift in (0..35).step_by(7) {
11472 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
11473 *at += 1;
11474 let part = u32::from(byte & 0x7f);
11475 if shift == 28 && part > 0x0f {
11476 return Err(invalid("membership varint overflow"));
11477 }
11478 value = value
11479 .checked_add(
11480 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
11481 )
11482 .ok_or_else(|| invalid("membership varint overflow"))?;
11483 if byte & 0x80 == 0 {
11484 return Ok(value);
11485 }
11486 }
11487 Err(invalid("membership varint is too long"))
11488}
11489
11490fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
11491 let mut at = 0;
11492 let count = take_varint(bytes, &mut at)? as usize;
11493 let mut codes = Vec::with_capacity(count);
11494 let mut previous = 0_u32;
11495 for index in 0..count {
11496 let delta = take_varint(bytes, &mut at)?;
11497 let code = if index == 0 {
11498 delta
11499 } else {
11500 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
11501 };
11502 if index > 0 && code <= previous {
11503 return Err(invalid("membership codes are not increasing"));
11504 }
11505 codes.push(code);
11506 previous = code;
11507 }
11508 if at != bytes.len() {
11509 return Err(invalid("membership page has trailing bytes"));
11510 }
11511 Ok(codes)
11512}
11513
11514fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
11522 let mut by_text: HashMap<&[u8], u32, Spread> =
11523 HashMap::with_capacity_and_hasher(vector.len(), Spread);
11524 let mut values = Vec::new();
11525 let mut codes = Vec::with_capacity(vector.len());
11526 let mut plain_bytes = 0_usize;
11527 for row in 0..vector.len() {
11528 let text = vector.bytes_at(row).unwrap_or(b"");
11529 plain_bytes = plain_bytes.saturating_add(text.len());
11530 let code = match by_text.get(text) {
11531 Some(&code) => code,
11532 None => {
11533 let code = u32::try_from(values.len())
11534 .map_err(|_| invalid("too many dictionary values"))?;
11535 by_text.insert(text, code);
11536 values.push(text);
11537 code
11538 }
11539 };
11540 codes.push(code);
11541 }
11542 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11543 let encoded = 8_usize
11544 .saturating_add((values.len() + 1).saturating_mul(4))
11545 .saturating_add(dictionary_bytes)
11546 .saturating_add(codes.len().saturating_mul(4));
11547 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11548 if encoded >= plain {
11549 return Ok(None);
11550 }
11551 let mut out = Vec::with_capacity(encoded);
11552 put_u32(
11553 &mut out,
11554 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11555 );
11556 put_u32(
11557 &mut out,
11558 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11559 );
11560 let mut offset = 0_u32;
11561 put_u32(&mut out, offset);
11562 for value in &values {
11563 offset = offset
11564 .checked_add(
11565 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11566 )
11567 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11568 put_u32(&mut out, offset);
11569 }
11570 for value in values {
11571 out.extend_from_slice(value);
11572 }
11573 for code in codes {
11574 put_u32(&mut out, code);
11575 }
11576 Ok(Some(out))
11577}
11578
11579struct Room<'a, T> {
11581 state: &'a Mutex<(T, usize)>,
11582 finished: &'a Condvar,
11583 bytes: usize,
11584}
11585
11586impl<T> Drop for Room<'_, T> {
11587 fn drop(&mut self) {
11588 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11589 held.1 -= self.bytes;
11590 drop(held);
11591 self.finished.notify_all();
11592 }
11593}
11594
11595enum Closing<'a> {
11597 Numeric {
11600 column: usize,
11601 counted: bool,
11602 dense: Option<(u64, usize)>,
11603 },
11604 Dictionary {
11605 index: usize,
11606 dictionary: &'a GlobalDictionary,
11607 },
11608}
11609
11610enum Closed {
11612 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11613 Dictionary(usize, ClosedDictionary),
11614}
11615
11616struct ClosedDictionary {
11618 distinct: Option<u64>,
11620 frequencies: Option<FrequencySummary>,
11621 texts: Vec<Option<Vec<u8>>>,
11622 hosts: Option<host::HostSummary>,
11623 encoded: EncodedDictionary,
11624 payload: u64,
11626}
11627
11628struct EncodedDictionary {
11629 index: Vec<u8>,
11630 ranks: Vec<u8>,
11631 grams: Vec<u8>,
11632}
11633
11634fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11675 let mut work = vec![(0, codes.len(), 0)];
11676 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11677 while let Some((from, to, depth)) = work.pop() {
11678 let part = &mut codes[from..to];
11679 keyed.clear();
11680 keyed.extend(part.iter().map(|&code| {
11681 let value = values(code);
11682 let rest = value.get(depth..).unwrap_or_default();
11683 (head(rest), rest.len().min(8) as u8, code)
11684 }));
11685 keyed.sort_unstable();
11686 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11687 *slot = entry.2;
11688 }
11689 let mut start = 0;
11690 while start < keyed.len() {
11691 let (key, taken, _) = keyed[start];
11692 let mut end = start + 1;
11693 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11694 end += 1;
11695 }
11696 if taken == 8 && end - start > 1 {
11697 work.push((from + start, from + end, depth + 8));
11698 }
11699 start = end;
11700 }
11701 }
11702}
11703
11704const PARALLEL_SORT_MIN: usize = 1 << 16;
11706
11707const BUCKETS_PER_WORKER: usize = 4;
11710
11711const SAMPLES_PER_BUCKET: usize = 32;
11713
11714fn sort_by_value_across<'a>(
11732 codes: &mut [u32],
11733 values: impl Fn(u32) -> &'a [u8] + Sync,
11734 workers: usize,
11735) {
11736 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11737 sort_by_value(codes, values);
11738 return;
11739 }
11740 let buckets = workers * BUCKETS_PER_WORKER;
11741 let wanted = buckets * SAMPLES_PER_BUCKET;
11742 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11743 sort_by_value(&mut sample, &values);
11744 let splitters =
11745 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11746 let values = &values;
11747 let splitters = &splitters;
11748 let per = codes.len().div_ceil(workers);
11749 let places = std::thread::scope(|scope| {
11751 codes
11752 .chunks(per)
11753 .map(|run| {
11754 scope.spawn(move || {
11755 run.iter()
11756 .map(|&code| {
11757 let value = values(code);
11758 splitters.partition_point(|splitter| *splitter <= value) as u32
11759 })
11760 .collect::<Vec<_>>()
11761 })
11762 })
11763 .collect::<Vec<_>>()
11764 .into_iter()
11765 .flat_map(|handle| {
11766 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11767 })
11768 .collect::<Vec<_>>()
11769 });
11770 let mut starts = vec![0_usize; buckets + 1];
11771 for &place in &places {
11772 starts[place as usize + 1] += 1;
11773 }
11774 for bucket in 0..buckets {
11775 starts[bucket + 1] += starts[bucket];
11776 }
11777 let mut laid = vec![0_u32; codes.len()];
11778 let mut next = starts.clone();
11779 for (&code, &place) in codes.iter().zip(&places) {
11780 laid[next[place as usize]] = code;
11781 next[place as usize] += 1;
11782 }
11783 drop(places);
11784 let mut runs = Vec::with_capacity(buckets);
11785 let mut rest = laid.as_mut_slice();
11786 for bucket in 0..buckets {
11787 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11788 runs.push(run);
11789 rest = after;
11790 }
11791 runs.sort_by_key(|run| run.len());
11793 let queue = Mutex::new(runs);
11794 std::thread::scope(|scope| {
11795 for _ in 0..workers {
11796 scope.spawn(|| {
11797 loop {
11798 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11799 let Some(run) = taken else { break };
11800 sort_by_value(run, values);
11801 }
11802 });
11803 }
11804 });
11805 codes.copy_from_slice(&laid);
11806}
11807
11808fn head(bytes: &[u8]) -> u64 {
11816 if let Some(word) = bytes.first_chunk::<8>() {
11817 return u64::from_be_bytes(*word);
11818 }
11819 let len = bytes.len();
11820 if len >= 4 {
11821 let front = u64::from(u32::from_be_bytes([bytes[0], bytes[1], bytes[2], bytes[3]]));
11822 let back = &bytes[len - 4..];
11823 let back = u64::from(u32::from_be_bytes([back[0], back[1], back[2], back[3]]));
11824 return (front << 32) | (back << (8 * (8 - len)));
11825 }
11826 bytes.iter().enumerate().fold(0, |word, (at, &byte)| word | (u64::from(byte) << (56 - 8 * at)))
11827}
11828
11829fn encode_global_dictionary(
11840 dictionary: &GlobalDictionary,
11841 order: &[(u64, u32)],
11842 places: &[Placed],
11843 scattered: bool,
11844) -> Result<EncodedDictionary> {
11845 let values = dictionary.values();
11846 if order.len() != values {
11847 return Err(invalid("global dictionary order does not cover its values"));
11848 }
11849 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11850 if places.len() != blocks {
11851 return Err(invalid("global dictionary payload is not the blocks it says it is"));
11852 }
11853 if dictionary.grams.len() != blocks {
11854 return Err(invalid("global dictionary signatures do not cover its blocks"));
11855 }
11856 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11857 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11858 let offset_bits = offset_width(&dictionary.ends);
11859 let payload_words = if scattered { 3 } else { 2 };
11860 let index_len = DICTIONARY_HEADER
11861 .checked_add(offset_bytes(values, offset_bits))
11862 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11863 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11864 .and_then(|len| len.checked_add(8))
11865 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11866 let mut index = Vec::with_capacity(index_len);
11867 put_u32(
11868 &mut index,
11869 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11870 );
11871 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11872 put_u32(
11873 &mut index,
11874 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11875 );
11876 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11877 | DICTIONARY_GRAMS
11878 | DICTIONARY_WIDE_GRAMS;
11879 put_u32(&mut index, offset_bits as u32 | flag);
11880 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11881 let mut end = 0_u64;
11886 for place in places {
11887 if scattered {
11888 put_u64(&mut index, place.start);
11889 put_u64(&mut index, place.length);
11890 } else {
11891 end = end
11892 .checked_add(place.length)
11893 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11894 put_u64(&mut index, end);
11895 }
11896 }
11897 for place in places {
11898 put_u64(&mut index, place.hash);
11899 }
11900 if rank_ends.len() != rank_blocks {
11903 return Err(invalid("global dictionary order is not the blocks it says it is"));
11904 }
11905 for end in &rank_ends {
11906 put_u64(&mut index, *end);
11907 }
11908 let mut at = 0_usize;
11909 for end in &rank_ends {
11910 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11911 put_u64(&mut index, checksum(&ranks[at..end]));
11912 at = end;
11913 }
11914 let gram_len = blocks
11915 .checked_mul(TEXT_GRAM_BYTES)
11916 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11917 let mut grams = Vec::with_capacity(gram_len);
11918 for block in &dictionary.grams {
11919 grams.extend_from_slice(block);
11920 }
11921 put_u64(&mut index, checksum(&grams));
11922 if index.len() != index_len {
11923 return Err(invalid("global dictionary index is not the length it was laid out for"));
11924 }
11925 Ok(EncodedDictionary { index, ranks, grams })
11926}
11927
11928const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11935
11936fn payload_shapes() -> Vec<chooser::Settled> {
11962 let integers = vec![integer::Kind::Packed];
11963 [
11964 vec![string::Kind::Front, string::Kind::Lz],
11965 vec![string::Kind::Lz, string::Kind::Fsst],
11966 vec![string::Kind::Lz, string::Kind::Plain],
11967 vec![string::Kind::Fsst],
11968 vec![string::Kind::Plain],
11969 ]
11970 .into_iter()
11971 .map(|strings| chooser::Settled::new(strings, integers.clone()))
11972 .collect()
11973}
11974
11975fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11982 let started = profile.map(|_| std::time::Instant::now());
11983 file.sync()?;
11984 if let (Some(profile), Some(started)) = (profile, started) {
11985 profile.waited(
11986 Stage::Publish,
11987 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11988 );
11989 }
11990 Ok(())
11991}
11992
11993#[derive(Debug)]
11998pub(crate) struct Unencoded {
11999 column: usize,
12000 at: usize,
12001 ends: Vec<u32>,
12002 bytes: Vec<u8>,
12003 shape: chooser::Settled,
12004}
12005
12006impl Unencoded {
12007 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
12009 let values = block_values(&self.ends, &self.bytes);
12010 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
12011 }
12012
12013 pub(crate) fn place(&self) -> (usize, usize) {
12015 (self.column, self.at)
12016 }
12017}
12018
12019pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
12023
12024fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
12026 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
12027 for value in values {
12028 for gram in value.windows(4) {
12029 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
12030 grams[bit / 8] |= 1 << (bit % 8);
12031 }
12032 }
12033 }
12034 grams
12035}
12036
12037fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
12039 let mut out = Vec::with_capacity(ends.len());
12040 let mut from = 0;
12041 for &to in ends {
12042 out.push(&bytes[from..to as usize]);
12043 from = to as usize;
12044 }
12045 out
12046}
12047
12048fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12055 for dictionary in dictionaries.iter_mut().flatten() {
12056 if !dictionary.early.is_empty() {
12057 return Err(Error::internal("a dictionary block handed out never came back"));
12058 }
12059 dictionary.seal_rest();
12060 dictionary.settle_rest()?;
12061 }
12062 encode_waiting(dictionaries)?;
12063 if dictionaries
12066 .iter()
12067 .flatten()
12068 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
12069 {
12070 return Err(Error::internal("a dictionary block handed out never came back"));
12071 }
12072 Ok(())
12073}
12074
12075fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
12078 let jobs = dictionaries
12079 .iter()
12080 .enumerate()
12081 .flat_map(|(column, held)| {
12082 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
12083 })
12084 .collect::<Vec<_>>();
12085 if jobs.is_empty() {
12086 return Ok(());
12087 }
12088 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
12089 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
12090 Ok((column, at, held.encode_waiting(at)?))
12091 };
12092 let workers = std::thread::available_parallelism()
12093 .map_or(1, usize::from)
12094 .min(MAX_FREQUENCY_WORKERS)
12095 .min(jobs.len());
12096 let made = if workers <= 1 {
12097 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
12098 } else {
12099 let next = AtomicUsize::new(0);
12100 let jobs = &jobs;
12101 let pieces = std::thread::scope(|scope| {
12102 (0..workers)
12103 .map(|_| {
12104 scope.spawn(|| {
12105 let mut mine = Vec::new();
12106 loop {
12107 let job = next.fetch_add(1, Atomic::Relaxed);
12108 let Some(&(column, at)) = jobs.get(job) else { break };
12109 mine.push(one(column, at)?);
12110 }
12111 Ok(mine)
12112 })
12113 })
12114 .collect::<Vec<_>>()
12115 .into_iter()
12116 .map(|handle| {
12117 handle
12118 .join()
12119 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
12120 })
12121 .collect::<Result<Vec<_>>>()
12122 })?;
12123 pieces.into_iter().flatten().collect()
12124 };
12125 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
12126 (0..dictionaries.len()).map(|_| Vec::new()).collect();
12127 for (column, at, bytes) in made {
12128 done[column].push((at, bytes));
12129 }
12130 for (column, mut made) in done.into_iter().enumerate() {
12131 if made.is_empty() {
12132 continue;
12133 }
12134 let Some(held) = dictionaries[column].as_mut() else { continue };
12135 made.sort_by_key(|(at, _)| *at);
12136 let waiting = std::mem::take(&mut held.waiting);
12137 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
12138 if held.encoded() != at {
12139 return Err(Error::internal("a dictionary block was encoded out of order"));
12140 }
12141 held.push_block(block);
12142 }
12143 }
12144 Ok(())
12145}
12146
12147fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
12157 let mut best: Option<(chooser::Settled, usize)> = None;
12158 for shape in payload_shapes() {
12159 let mut size = 0;
12160 for block in sample {
12161 size += string::encode_with(block, &shape)?.len();
12162 }
12163 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
12164 best = Some((shape, size));
12165 }
12166 }
12167 best.map(|(shape, _)| shape)
12168 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
12169}
12170
12171fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
12178 let mut out = Vec::with_capacity(order.len() * 4);
12179 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
12180 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
12181 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
12182 for block in order.chunks(TEXT_RANK_BLOCK) {
12183 let base = block.first().map_or(0, |&(head, _)| head);
12186 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
12187 let width = (u64::BITS - span.leading_zeros()) as usize;
12188 heads.clear();
12189 codes.clear();
12190 for &(head, code) in block {
12191 heads.push(head.wrapping_sub(base));
12192 codes.push(u64::from(code));
12193 }
12194 put_u64(&mut out, base);
12195 out.push(width as u8);
12196 bitpack::pack_tail(&heads, width, &mut out)
12197 .map_err(|_| invalid("global dictionary heads do not pack"))?;
12198 bitpack::pack_tail(&codes, code_bits, &mut out)
12199 .map_err(|_| invalid("global dictionary codes do not pack"))?;
12200 ends.push(out.len() as u64);
12201 }
12202 Ok((out, ends))
12203}
12204
12205fn open_global_dictionary(
12212 file: Arc<File>,
12213 page: Page,
12214 ty: &LogicalType,
12215 keep_budget: usize,
12216) -> Result<Vector> {
12217 if !coded_type(ty) {
12218 return Err(invalid("global dictionary belongs to a non-string column"));
12219 }
12220 let mut header = [0; DICTIONARY_HEADER];
12221 read_at(&file, page.offset, &mut header)?;
12222 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
12223 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
12224 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
12225 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12226 let scattered = width & DICTIONARY_SCATTERED != 0;
12227 let has_grams = width & DICTIONARY_GRAMS != 0;
12228 let gram_width =
12229 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
12230 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
12231 if per_block != TEXT_PAYLOAD_VALUES {
12232 return Err(invalid("global dictionary block width differs"));
12233 }
12234 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
12235 return Err(invalid("global dictionary block count differs from its value count"));
12236 }
12237 if offset_bits > u32::BITS as usize {
12238 return Err(invalid("global dictionary packs offsets past a payload"));
12239 }
12240 let offset_len = offset_bytes(count, offset_bits);
12241 let ranks = count;
12246 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
12247 let payload_words = if scattered { 3 } else { 2 };
12251 let hash_len = blocks
12252 .checked_mul(payload_words * 8)
12253 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
12254 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
12255 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
12256 let gram_len = if has_grams {
12257 blocks
12258 .checked_mul(gram_width)
12259 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
12260 } else {
12261 0
12262 };
12263 let index_len = DICTIONARY_HEADER
12264 .checked_add(offset_len)
12265 .and_then(|len| len.checked_add(hash_len))
12266 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12267 if index_len > page.length as usize {
12268 return Err(invalid("global dictionary offset index exceeds its page"));
12269 }
12270 let mut index = vec![0; index_len];
12271 index[..DICTIONARY_HEADER].copy_from_slice(&header);
12272 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
12273 if checksum(&index) != page.hash {
12274 return Err(invalid("global dictionary index checksum differs"));
12275 }
12276 let word_end = index_len - usize::from(has_grams) * 8;
12277 let gram_hash = has_grams
12278 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
12279 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
12280 .chunks_exact(8)
12281 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
12282 .collect::<Vec<_>>();
12283 let mut rest = words.split_off(blocks * payload_words);
12284 let rank_hashes = rest.split_off(rank_blocks);
12285 let rank_ends = rest;
12286 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
12289 return Err(invalid("global dictionary order blocks do not rise"));
12290 }
12291 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
12292 .map_err(|_| invalid("global dictionary rank overflow"))?;
12293 let body_len = index_len
12294 .checked_add(rank_len)
12295 .ok_or_else(|| invalid("global dictionary header overflow"))?;
12296 if body_len > page.length as usize {
12297 return Err(invalid("global dictionary order exceeds its page"));
12298 }
12299 let gram_end = body_len
12300 .checked_add(gram_len)
12301 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
12302 if gram_end > page.length as usize {
12303 return Err(invalid("global dictionary signatures exceed their page"));
12304 }
12305 let grams = gram_hash.map(|hash| NativeGrams {
12306 start: page.offset + body_len as u64,
12307 length: gram_len,
12308 width: gram_width,
12309 hash,
12310 verdicts: Mutex::new(Vec::new()),
12311 });
12312 let mut offsets = index;
12316 offsets.truncate(DICTIONARY_HEADER + offset_len);
12317 let hashes = words.split_off(blocks * (payload_words - 1));
12318 let (starts, lengths) = if scattered {
12319 let mut starts = Vec::with_capacity(blocks);
12320 let mut lengths = Vec::with_capacity(blocks);
12321 for pair in words.chunks_exact(2) {
12322 starts.push(pair[0]);
12323 lengths.push(pair[1]);
12324 }
12325 (starts, lengths)
12326 } else {
12327 let base = page.offset + gram_end as u64;
12331 let mut starts = Vec::with_capacity(blocks);
12332 let mut lengths = Vec::with_capacity(blocks);
12333 let mut at = 0_u64;
12334 for &end in &words {
12335 let len = end
12336 .checked_sub(at)
12337 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
12338 starts.push(base + at);
12339 lengths.push(len);
12340 at = end;
12341 }
12342 (starts, lengths)
12343 };
12344 let stored_len = page.length as u64 - gram_end as u64;
12350 if scattered && stored_len == 0 {
12351 let size = file.metadata().map_err(io)?.len();
12352 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
12353 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
12354 });
12355 if !inside {
12356 return Err(invalid("global dictionary block lies outside the file"));
12357 }
12358 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
12359 return Err(invalid("global dictionary blocks do not bound the payload"));
12360 }
12361 Vector::external_text(
12362 ty.clone(),
12363 Arc::new(NativeText {
12364 file,
12365 values: count,
12366 offsets,
12367 offset_bits,
12368 value_ends: OnceLock::new(),
12369 value_lens: OnceLock::new(),
12370 ends_asked: AtomicUsize::new(0),
12371 ranks,
12372 rank_at: page.offset + index_len as u64,
12373 rank_ends,
12374 rank_hashes,
12375 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
12376 code_bits: code_width(count),
12377 code_ranks: OnceLock::new(),
12378 starts,
12379 lengths,
12380 hashes,
12381 grams,
12382 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
12383 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
12384 keep_budget,
12385 payload_kept: AtomicUsize::new(0),
12386 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
12387 visit_dropped: AtomicUsize::new(0),
12388 searched: Mutex::new(HashMap::new()),
12389 }),
12390 )
12391}
12392
12393fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
12406 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
12408 let mut cur = Cursor::new(bytes);
12409 let codec = cur.u8()?;
12410 if cur.u8()? == 2 {
12411 cur.take(rows.div_ceil(8))?;
12412 }
12413 Ok((codec, cur.at))
12414 }
12415 let Ok((codec, at)) = cascade_at(rows, bytes) else {
12416 return "UNREADABLE".to_string();
12417 };
12418 let tail = &bytes[at..];
12419 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
12420 match codec {
12421 0 => match ty {
12422 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
12423 _ => "FIXED".to_string(),
12424 },
12425 1 => "DICT(PLAIN)".to_string(),
12426 2 => "FOR+BITPACK".to_string(),
12427 3 => "TABLE DICT".to_string(),
12428 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
12429 5 => described(integer::describe(tail)),
12430 6 => described(string::describe(tail)),
12431 other => format!("CODEC {other}"),
12432 }
12433}
12434
12435fn decode_selected_stable_codes(
12440 rows: usize,
12441 bytes: &[u8],
12442 positions: &[usize],
12443 out: &mut Vec<Option<u32>>,
12444) -> Result<bool> {
12445 if positions.windows(2).any(|pair| pair[0] >= pair[1])
12446 || positions.last().is_some_and(|&position| position >= rows)
12447 {
12448 return Err(invalid("selected code positions are not sorted and in range"));
12449 }
12450 let mut cur = Cursor::new(bytes);
12451 let codec = cur.u8()?;
12452 if codec != 3 && codec != 4 {
12453 return Ok(false);
12454 }
12455 let flag = cur.u8()?;
12456 let mask = match flag {
12457 0 | 1 => None,
12458 2 => {
12459 let at = cur.at;
12460 let len = rows.div_ceil(8);
12461 cur.take(len)?;
12462 Some((at, len))
12463 }
12464 _ => return Err(invalid("page validity tag differs")),
12465 };
12466 let valid = |row: usize| match flag {
12467 0 => true,
12468 1 => false,
12469 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
12470 _ => unreachable!("the validity tag was checked"),
12471 };
12472 if codec == 4 {
12473 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
12474 for (&row, code) in positions.iter().zip(wide) {
12475 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
12476 out.push(valid(row).then_some(code));
12477 }
12478 return Ok(true);
12479 }
12480 let codes_at = cur.at;
12481 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
12482 cur.take(codes_len)?;
12483 if cur.at != bytes.len() {
12484 return Err(invalid("global code page has trailing bytes"));
12485 }
12486 let codes = &bytes[codes_at..codes_at + codes_len];
12487 for &row in positions {
12488 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
12489 let code = u32::from_le_bytes(
12490 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
12491 );
12492 out.push(valid(row).then_some(code));
12493 }
12494 Ok(true)
12495}
12496
12497fn decode_at(
12503 ty: &LogicalType,
12504 rows: usize,
12505 bytes: &[u8],
12506 global: Option<Arc<Vector>>,
12507 positions: &[u32],
12508) -> Result<Vector> {
12509 if positions.last().is_some_and(|&last| last as usize >= rows) {
12510 return Err(invalid("a position is past the end of the part"));
12511 }
12512 if bytes.first() == Some(&5)
12515 && positions.len().saturating_mul(8) <= rows
12516 && bytes
12518 .get(2 + if bytes.get(1) == Some(&2) { rows.div_ceil(8) } else { 0 }..)
12519 .is_some_and(integer::pointed)
12520 {
12521 return cascade_at(ty, rows, bytes, positions);
12522 }
12523 if bytes.first() != Some(&6) {
12524 return decode(ty, rows, bytes, global)?.gather(positions);
12525 }
12526 if !coded_type(ty) {
12527 return Err(invalid("compressed text codec belongs to a non-string page"));
12528 }
12529 let mut cur = Cursor::new(bytes);
12530 cur.u8()?;
12531 let validity = match cur.u8()? {
12532 0 => Validity::AllValid,
12533 1 => Validity::AllInvalid,
12534 2 => {
12535 let mask = cur.take(rows.div_ceil(8))?;
12536 Validity::from_iter(positions.len(), |at| {
12537 let row = positions[at] as usize;
12538 mask[row / 8] >> (row % 8) & 1 == 1
12539 })
12540 }
12541 _ => return Err(invalid("page validity tag differs")),
12542 };
12543 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
12544 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12545 push_values(&mut values, ty, &ends)?;
12546 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
12547}
12548
12549fn cascade_at(ty: &LogicalType, rows: usize, bytes: &[u8], positions: &[u32]) -> Result<Vector> {
12555 fn wanted<T: integer::Lane>(values: &[i64]) -> Result<Vec<T>> {
12556 values
12557 .iter()
12558 .map(|&value| T::fit(value).ok_or_else(|| invalid("page value is not of its type")))
12559 .collect()
12560 }
12561 let mut cur = Cursor::new(bytes);
12562 cur.u8()?;
12563 let validity = match cur.u8()? {
12564 0 => Validity::AllValid,
12565 1 => Validity::AllInvalid,
12566 2 => {
12567 let mask = cur.take(rows.div_ceil(8))?;
12568 Validity::from_iter(positions.len(), |at| {
12569 let row = positions[at] as usize;
12570 mask[row / 8] >> (row % 8) & 1 == 1
12571 })
12572 }
12573 _ => return Err(invalid("page validity tag differs")),
12574 };
12575 let at: Vec<usize> = positions.iter().map(|&row| row as usize).collect();
12576 let values = integer::decode_selected(&bytes[cur.at..], &at)
12577 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
12578 if values.len() != positions.len() {
12579 return Err(invalid("cascade page holds the wrong number of rows"));
12580 }
12581 let data = match ty {
12582 LogicalType::TinyInt => Data::Int8(wanted::<i8>(&values)?.into()),
12583 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(&values)?.into()),
12584 LogicalType::SmallInt => Data::Int16(wanted::<i16>(&values)?.into()),
12585 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(&values)?.into()),
12586 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(&values)?.into()),
12587 LogicalType::UInteger => Data::UInt32(wanted::<u32>(&values)?.into()),
12588 LogicalType::BigInt
12589 | LogicalType::Timestamp
12590 | LogicalType::Time
12591 | LogicalType::TimeTz
12592 | LogicalType::TimestampTz
12593 | LogicalType::TimestampS
12594 | LogicalType::TimestampMs
12595 | LogicalType::TimestampNs => Data::Int64(values.into()),
12596 LogicalType::Decimal { .. } => match ty.physical() {
12597 PhysicalType::Int16 => Data::Int16(wanted::<i16>(&values)?.into()),
12598 PhysicalType::Int32 => Data::Int32(wanted::<i32>(&values)?.into()),
12599 PhysicalType::Int64 => Data::Int64(values.into()),
12600 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
12601 },
12602 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
12603 };
12604 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12605}
12606
12607fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
12611 if ty == &LogicalType::Varchar {
12612 return values.push_run_in_place(0, ends);
12613 }
12614 let mut start = 0;
12615 for &end in ends {
12616 let len = end
12617 .checked_sub(start)
12618 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12619 values.push_bytes_in_place(start, len)?;
12620 start = end;
12621 }
12622 Ok(())
12623}
12624
12625fn decode(
12626 ty: &LogicalType,
12627 rows: usize,
12628 bytes: &[u8],
12629 global: Option<Arc<Vector>>,
12630) -> Result<Vector> {
12631 let mut cur = Cursor::new(bytes);
12632 let codec = cur.u8()?;
12633 let flag = cur.u8()?;
12634 let validity = match flag {
12635 0 => Validity::AllValid,
12636 1 => Validity::AllInvalid,
12637 2 => {
12638 let mask = cur.take(rows.div_ceil(8))?;
12639 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12640 }
12641 _ => return Err(invalid("page validity tag differs")),
12642 };
12643 if codec == 1 {
12644 if !coded_type(ty) {
12645 return Err(invalid("dictionary codec belongs to a non-string page"));
12646 }
12647 let count = cur.u32()? as usize;
12648 let payload_len = cur.u32()? as usize;
12649 let offset_bytes = cur.take(
12650 (count + 1)
12651 .checked_mul(4)
12652 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12653 )?;
12654 let offsets = offset_bytes
12655 .chunks_exact(4)
12656 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12657 .collect::<Vec<_>>();
12658 let payload = cur.take(payload_len)?.to_vec();
12659 if offsets.first() != Some(&0)
12660 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12661 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12662 {
12663 return Err(invalid("dictionary offsets do not bound the payload"));
12664 }
12665 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12668 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12669 push_values(&mut strings, ty, &ends)?;
12670 let mut codes = Vec::with_capacity(rows);
12671 for _ in 0..rows {
12672 codes.push(cur.u32()?);
12673 }
12674 if codes.iter().any(|code| *code as usize >= count) {
12675 return Err(invalid("dictionary code is out of range"));
12676 }
12677 if cur.at != bytes.len() {
12678 return Err(invalid("dictionary page has trailing bytes"));
12679 }
12680 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12681 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12682 }
12683 if codec == 3 || codec == 4 {
12684 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12685 let codes = if codec == 4 {
12686 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12691 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12692 if codes.len() != rows {
12693 return Err(invalid("encoded code page holds the wrong number of rows"));
12694 }
12695 codes
12696 } else {
12697 let mut codes = Vec::with_capacity(rows);
12698 for _ in 0..rows {
12699 codes.push(cur.u32()?);
12700 }
12701 if cur.at != bytes.len() {
12702 return Err(invalid("global code page has trailing bytes"));
12703 }
12704 codes
12705 };
12706 let highest = codes.iter().copied().max();
12707 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12708 .with_validity(validity));
12709 }
12710 if codec == 6 {
12711 if !coded_type(ty) {
12712 return Err(invalid("compressed text codec belongs to a non-string page"));
12713 }
12714 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12718 if ends.len() != rows {
12719 return Err(invalid("compressed text page holds the wrong number of rows"));
12720 }
12721 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12724 push_values(&mut values, ty, &ends)?;
12725 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12726 }
12727 if codec == 5 {
12728 let data = cascade(ty, &bytes[cur.at..], rows)?;
12730 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12731 }
12732 if codec == 2 {
12733 let width = u32::from(cur.u8()?);
12734 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12735 let count = cur.u32()? as usize;
12736 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12737 let words: Vec<u64> = cur
12738 .take(length)?
12739 .chunks_exact(8)
12740 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12741 .collect();
12742 if cur.at != bytes.len() {
12743 return Err(invalid("packed page has trailing bytes"));
12744 }
12745 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12746 }
12747 if codec != 0 {
12748 return Err(invalid("page codec is unknown"));
12749 }
12750 let data = match ty {
12751 LogicalType::TinyInt => {
12752 let values = cur.take(rows)?;
12753 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12754 }
12755 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12756 LogicalType::SmallInt => {
12757 let values =
12758 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12759 Data::Int16(
12760 values
12761 .chunks_exact(2)
12762 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12763 .collect::<Vec<_>>()
12764 .into(),
12765 )
12766 }
12767 LogicalType::USmallInt => {
12768 let values =
12769 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12770 Data::UInt16(
12771 values
12772 .chunks_exact(2)
12773 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12774 .collect::<Vec<_>>()
12775 .into(),
12776 )
12777 }
12778 LogicalType::UInteger => {
12779 let values =
12780 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12781 Data::UInt32(
12782 values
12783 .chunks_exact(4)
12784 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12785 .collect::<Vec<_>>()
12786 .into(),
12787 )
12788 }
12789 LogicalType::UBigInt => {
12790 let values =
12791 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12792 Data::UInt64(
12793 values
12794 .chunks_exact(8)
12795 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12796 .collect::<Vec<_>>()
12797 .into(),
12798 )
12799 }
12800 LogicalType::Integer | LogicalType::Date => {
12801 let values =
12802 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12803 Data::Int32(
12804 values
12805 .chunks_exact(4)
12806 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12807 .collect::<Vec<_>>()
12808 .into(),
12809 )
12810 }
12811 LogicalType::BigInt
12812 | LogicalType::Timestamp
12813 | LogicalType::Time
12814 | LogicalType::TimeTz
12815 | LogicalType::TimestampTz
12816 | LogicalType::TimestampS
12817 | LogicalType::TimestampMs
12818 | LogicalType::TimestampNs => {
12819 let values =
12820 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12821 Data::Int64(
12822 values
12823 .chunks_exact(8)
12824 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12825 .collect::<Vec<_>>()
12826 .into(),
12827 )
12828 }
12829 LogicalType::HugeInt | LogicalType::Uuid => {
12830 let values =
12831 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12832 Data::Int128(
12833 values
12834 .chunks_exact(16)
12835 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12836 .collect::<Vec<_>>()
12837 .into(),
12838 )
12839 }
12840 LogicalType::UHugeInt => {
12841 let values =
12842 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12843 Data::UInt128(
12844 values
12845 .chunks_exact(16)
12846 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12847 .collect::<Vec<_>>()
12848 .into(),
12849 )
12850 }
12851 LogicalType::Float => {
12852 let values =
12853 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12854 Data::Float32(
12855 values
12856 .chunks_exact(4)
12857 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12858 .collect::<Vec<_>>()
12859 .into(),
12860 )
12861 }
12862 LogicalType::Double => {
12863 let values =
12864 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12865 Data::Float64(
12866 values
12867 .chunks_exact(8)
12868 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12869 .collect::<Vec<_>>()
12870 .into(),
12871 )
12872 }
12873 LogicalType::Interval => {
12874 let values =
12875 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12876 Data::Interval(
12877 values
12878 .chunks_exact(16)
12879 .map(|item| {
12880 (
12881 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12882 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12883 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12884 )
12885 })
12886 .collect::<Vec<_>>()
12887 .into(),
12888 )
12889 }
12890 LogicalType::Boolean => {
12891 let values = cur.take(rows)?;
12892 if values.iter().any(|value| *value > 1) {
12893 return Err(invalid("boolean page has another value"));
12894 }
12895 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12896 }
12897 LogicalType::Decimal { .. } => match ty.physical() {
12900 PhysicalType::Int16 => {
12901 let values =
12902 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12903 Data::Int16(
12904 values
12905 .chunks_exact(2)
12906 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12907 .collect::<Vec<_>>()
12908 .into(),
12909 )
12910 }
12911 PhysicalType::Int32 => {
12912 let values =
12913 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12914 Data::Int32(
12915 values
12916 .chunks_exact(4)
12917 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12918 .collect::<Vec<_>>()
12919 .into(),
12920 )
12921 }
12922 PhysicalType::Int64 => {
12923 let values =
12924 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12925 Data::Int64(
12926 values
12927 .chunks_exact(8)
12928 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12929 .collect::<Vec<_>>()
12930 .into(),
12931 )
12932 }
12933 _ => {
12934 let values =
12935 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12936 Data::Int128(
12937 values
12938 .chunks_exact(16)
12939 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12940 .collect::<Vec<_>>()
12941 .into(),
12942 )
12943 }
12944 },
12945 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12946 let offset_bytes = cur
12947 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12948 let offsets = offset_bytes
12949 .chunks_exact(4)
12950 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12951 .collect::<Vec<_>>();
12952 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12953 if offsets.first() != Some(&0)
12954 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12955 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12956 {
12957 return Err(invalid("string offsets do not bound the payload"));
12958 }
12959 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12967 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12968 push_values(&mut values, ty, &ends)?;
12969 Data::Varlen(values)
12970 }
12971 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12972 };
12973 if cur.at != bytes.len() {
12974 return Err(invalid("page has trailing bytes"));
12975 }
12976 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12977}
12978
12979#[cfg(test)]
12980mod tests {
12981 use std::fs::{self, OpenOptions};
12982 use std::io::{Seek, SeekFrom, Write};
12983 use std::path::PathBuf;
12984 use std::time::{SystemTime, UNIX_EPOCH};
12985
12986 use rudb_common::Stat;
12987 use rudb_common::Value;
12988 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12989 use rudb_common::stat::Provenance;
12990
12991 use super::*;
12992
12993 #[test]
12994 fn head_is_the_value_padded_to_eight_bytes() {
12995 let bytes: Vec<u8> = (1..=12).collect();
12996 for len in 0..=bytes.len() {
12997 let value = &bytes[..len];
12998 let mut word = [0; 8];
12999 let take = len.min(8);
13000 word[..take].copy_from_slice(&value[..take]);
13001 assert_eq!(head(value), u64::from_be_bytes(word), "{len} bytes");
13002 }
13003 assert!(head(b"ab") < head(b"ab\x01"));
13004 assert!(head(b"abcd") < head(b"abce"));
13005 }
13006
13007 #[test]
13008 fn spanned_frequency_header_rejects_missing_or_out_of_bounds_payloads() {
13009 for (length, entries) in [(0_u32, 1_u32), (9, 0), (1, FREQUENCY_ENTRIES as u32 + 1)] {
13010 let mut bytes = Vec::new();
13011 put_u32(&mut bytes, length);
13012 put_u32(&mut bytes, entries);
13013 bytes.push(1);
13014 assert!(summary_span(&mut Cursor::new(&bytes)).is_err());
13015 }
13016 let mut bytes = Vec::new();
13017 put_u32(&mut bytes, 1);
13018 put_u32(&mut bytes, 0);
13019 bytes.push(1);
13020 assert_eq!(summary_span(&mut Cursor::new(&bytes)).expect("one byte"), Some((1, 0)));
13021 }
13022
13023 #[test]
13024 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
13025 let bytes: Vec<u8> =
13026 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
13027 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
13028 let whole = content_name(&bytes[..length]);
13029 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
13030 let mut namer = ContentNamer::default();
13031 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
13032 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
13033 }
13034 }
13035 }
13036
13037 #[derive(Debug)]
13040 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
13041
13042 impl chooser::Chooser for TestsEverything<'_> {
13043 fn name(&self) -> &'static str {
13044 "tests everything"
13045 }
13046
13047 fn narrow_strings(
13048 &self,
13049 values: &[&[u8]],
13050 offered: &[string::Kind],
13051 depth: u8,
13052 ) -> Vec<string::Kind> {
13053 self.0.narrow_strings(values, offered, depth)
13054 }
13055
13056 fn narrow_integers(
13057 &self,
13058 values: &[i64],
13059 offered: &[integer::Kind],
13060 depth: u8,
13061 ) -> Vec<integer::Kind> {
13062 self.0.narrow_integers(values, offered, depth)
13063 }
13064 }
13065
13066 #[test]
13067 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
13068 let columns: Vec<Vec<i64>> = vec![
13069 vec![],
13070 vec![5; 1000],
13071 (0..1000).collect(),
13072 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
13073 (0..1000).map(|row| row / 50).collect(),
13074 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
13075 (0..1000).map(|row| (row * 7919) % 13).collect(),
13076 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
13077 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
13078 (0..1000).map(|row| i64::MIN + row % 3).collect(),
13079 ];
13080 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
13081 for column in &columns {
13082 for chooser in choosers {
13083 let quick = integer::encode_with(column, chooser).unwrap();
13084 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
13085 assert_eq!(
13086 quick,
13087 full,
13088 "{} on {:?}",
13089 chooser.name(),
13090 &column[..column.len().min(8)]
13091 );
13092 }
13093 }
13094 }
13095
13096 #[test]
13099 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
13100 let mut settling = Settling::default();
13101 for part in 0..STRIPE_PARTS as i64 {
13102 let values: Vec<i64> = (0..2048)
13103 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
13104 .collect();
13105 let searched = integer::encode_with(&values, &Fixed).unwrap();
13106 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
13107 }
13108 }
13109
13110 #[test]
13114 fn text_pages_share_a_table_until_the_text_changes() {
13115 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
13116 let english: Vec<Vec<u8>> = (0..1024)
13117 .map(|row: usize| {
13118 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
13119 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
13120 })
13121 .collect();
13122 let digits: Vec<Vec<u8>> =
13123 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
13124 let mut settling = Settling::default();
13125 for page in 0..8 {
13126 let values: Vec<&[u8]> =
13127 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
13128 let payload = values.iter().map(|value| value.len()).sum();
13129 let out = settling.text(&values, payload).unwrap().unwrap();
13130 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
13131 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
13132 assert!(
13133 out.len() * 4 <= alone.len() * 5,
13134 "page {page}: {} against {}",
13135 out.len(),
13136 alone.len()
13137 );
13138 let since = settling.symbols.as_ref().unwrap().since;
13139 assert_eq!(since, page % 4, "page {page}");
13140 }
13141 }
13142
13143 #[test]
13147 fn a_column_that_changes_under_the_shape_is_searched_again() {
13148 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13149 let mut noise = move || {
13150 state ^= state << 13;
13151 state ^= state >> 7;
13152 state ^= state << 17;
13153 (state % 1_000_000) as i64
13154 };
13155 let mut settling = Settling::default();
13156 for part in 0..STRIPE_PARTS as i64 {
13157 let values: Vec<i64> = match part / 16 {
13158 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
13159 1 => (0..2048).map(|_| noise()).collect(),
13160 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
13161 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
13162 };
13163 let settled = settling.encode(&values).unwrap();
13164 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
13165 let searched = integer::encode_with(&values, &Fixed).unwrap();
13166 assert!(
13167 settled.len() * 4 <= searched.len() * 5,
13168 "part {part}: {} settled against {} searched, {} against {}",
13169 settled.len(),
13170 searched.len(),
13171 integer::describe(&settled).unwrap(),
13172 integer::describe(&searched).unwrap(),
13173 );
13174 }
13175 }
13176
13177 #[test]
13178 fn checksum_matches_fixed_vectors() {
13179 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
13180 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
13181 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
13182 }
13183
13184 #[test]
13185 fn sorting_across_threads_matches_sorting_on_one() {
13186 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
13187 let mut next = move || {
13188 state ^= state << 13;
13189 state ^= state >> 7;
13190 state ^= state << 17;
13191 state
13192 };
13193 let mut values = Vec::new();
13194 for at in 0..150_000_u64 {
13195 let value = match next() % 6 {
13196 0 => Vec::new(),
13197 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
13198 2 => format!("https://example.com/path/{at}").into_bytes(),
13199 3 => b"same".to_vec(),
13200 4 => vec![0xff; (next() % 12) as usize],
13201 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
13202 };
13203 values.push(value);
13204 }
13205 let value = |code: u32| values[code as usize].as_slice();
13206 for workers in [1, 2, 3, 8, 32] {
13207 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
13208 let mut across = one.clone();
13209 sort_by_value(&mut one, value);
13210 sort_by_value_across(&mut across, value, workers);
13211 assert_eq!(one, across, "{workers} workers");
13212 }
13213 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
13214 sort_by_value_across(&mut sorted, value, 8);
13215 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
13216 }
13217
13218 fn path(label: &str) -> PathBuf {
13219 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
13220 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
13221 }
13222
13223 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
13228 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
13229 (0..dictionary.values())
13230 .map(|code| {
13231 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
13232 flat[from..to].to_vec()
13233 })
13234 .collect()
13235 }
13236
13237 fn attached(table: &Table) -> Vec<&Section> {
13244 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
13245 }
13246
13247 #[test]
13249 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
13250 const SPANS: usize = 64;
13251 const SPAN: usize = 512;
13252 let path = path("positional");
13253 let content: Vec<u8> =
13254 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
13255 fs::write(&path, &content).expect("the file is written");
13256 let file = Arc::new(File::open(&path).expect("the file opens"));
13257 std::thread::scope(|scope| {
13258 for _ in 0..8 {
13259 let file = Arc::clone(&file);
13260 scope.spawn(move || {
13261 for _ in 0..64 {
13262 for span in 0..SPANS {
13263 let mut bytes = [0_u8; SPAN];
13264 read_at(&file, (span * SPAN) as u64, &mut bytes)
13265 .expect("the span reads");
13266 assert!(
13267 bytes.iter().all(|byte| *byte == span as u8),
13268 "span {span} came back as {}",
13269 bytes[0],
13270 );
13271 }
13272 }
13273 });
13274 }
13275 });
13276 let mut past = [0_u8; SPAN];
13277 let end = (SPANS * SPAN) as u64;
13278 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
13279 assert!(error.message().contains("ends before its declared length"), "{error}");
13280 drop(file);
13281 let _ = fs::remove_file(&path);
13282 }
13283
13284 #[test]
13291 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
13292 let path = path("cursor");
13293 let mut writer = Writer::create(
13294 &path,
13295 "items",
13296 vec![
13297 Field::required("id", LogicalType::Integer),
13298 Field::new("text", LogicalType::Varchar),
13299 ],
13300 )
13301 .expect("new file");
13302 writer.append(&sample()).expect("first part");
13303 writer.append(&sample()).expect("second part");
13304 writer.finish().expect("commit");
13305 let reader = Reader::open(&path).expect("reopen from disk");
13306 assert_eq!(reader.table().rows(), 6);
13307 let ids = reader.read(0, &[0]).expect("the integer page reads back");
13308 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
13309 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
13310 let text = reader.read(1, &[1]).expect("the text page reads back");
13311 assert_eq!(text.value_at(1, 0), Value::Null);
13312 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13313 let end = reader.table().stripes().iter().flat_map(|stripe| {
13316 stripe
13317 .pages
13318 .iter()
13319 .map(|page| page.offset + u64::from(page.length))
13320 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
13321 });
13322 let last = end.fold(HEADER, u64::max);
13323 let directory = fs::metadata(&path).expect("the file is there").len();
13324 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
13325 fs::remove_file(path).expect("remove scratch file");
13326 }
13327
13328 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
13334 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
13335 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
13336 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13337 let bits = (width & !DICTIONARY_FLAGS) as usize;
13338 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
13339 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
13340 DICTIONARY_HEADER as u64
13341 + offset_bytes(count as usize, bits) as u64
13342 + blocks * payload_words * 8
13343 + rank_blocks * 16
13344 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
13345 }
13346
13347 fn sample() -> Chunk {
13348 Chunk::new(vec![
13349 Vector::from_values(
13350 LogicalType::Integer,
13351 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
13352 )
13353 .expect("integers"),
13354 Vector::from_values(
13355 LogicalType::Varchar,
13356 &[
13357 Value::Varchar("alpha".into()),
13358 Value::Null,
13359 Value::Varchar("long text after a slash".into()),
13360 ],
13361 )
13362 .expect("strings"),
13363 ])
13364 .expect("matching rows")
13365 }
13366
13367 fn sample_ids() -> Chunk {
13368 Chunk::new(vec![
13369 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
13370 .expect("integers"),
13371 ])
13372 .expect("one column")
13373 }
13374
13375 #[test]
13376 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
13377 let path = path("nulls_for_the_planner");
13380 let mut writer =
13381 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
13382 .expect("new file");
13383 let rows = Chunk::new(vec![
13384 Vector::from_values(
13385 LogicalType::Integer,
13386 &[
13387 Value::Integer(4),
13388 Value::Null,
13389 Value::Integer(9),
13390 Value::Null,
13391 Value::Integer(1),
13392 Value::Integer(2),
13393 ],
13394 )
13395 .expect("integers"),
13396 ])
13397 .expect("one column");
13398 writer.append(&rows).expect("the only part");
13399 writer.finish().expect("commit");
13400 let reader = Reader::open(&path).expect("reopen from disk");
13401 let stripes = Stripes::new(reader);
13402 let column = stripes.column("a").expect("the file has that column");
13403 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
13404 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
13407 fs::remove_file(&path).expect("clean up");
13408 }
13409
13410 #[test]
13411 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
13412 let path = path("frequencies_for_the_planner");
13415 let mut writer =
13416 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13417 .expect("new file");
13418 let rows = Chunk::new(vec![
13419 Vector::from_values(
13420 LogicalType::Integer,
13421 &[
13422 Value::Integer(4),
13423 Value::Integer(4),
13424 Value::Integer(4),
13425 Value::Integer(9),
13426 Value::Integer(9),
13427 Value::Integer(1),
13428 ],
13429 )
13430 .expect("integers"),
13431 ])
13432 .expect("one column");
13433 writer.append(&rows).expect("the only part");
13434 writer.finish().expect("commit");
13435 let reader = Reader::open(&path).expect("reopen from disk");
13436 let common = Common::new(reader);
13437 assert_eq!(common.rows(), 6);
13438 let column = common.column("id").expect("the file has that column");
13439 assert_eq!(common.column("nothing"), None);
13440 assert_eq!(
13441 common.rows_with(column, &Bound::Int(4)),
13442 Stat::exact(3, Provenance::FrequencySynopsis)
13443 );
13444 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
13446 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
13449 assert!(common.remainder(column).is_some());
13450 fs::remove_file(&path).expect("clean up");
13451 }
13452
13453 #[test]
13454 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
13455 let path = path("string_frequencies_for_the_planner");
13456 let mut writer =
13457 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13458 .expect("new file");
13459 let rows = Chunk::new(vec![
13460 Vector::from_values(
13461 LogicalType::Varchar,
13462 &[
13463 Value::Varchar(String::new()),
13464 Value::Varchar("alpha".into()),
13465 Value::Varchar(String::new()),
13466 Value::Varchar("beta".into()),
13467 Value::Varchar(String::new()),
13468 ],
13469 )
13470 .expect("strings"),
13471 ])
13472 .expect("one column");
13473 writer.append(&rows).expect("the only part");
13474 writer.finish().expect("commit");
13475
13476 let reader = Reader::open(&path).expect("reopen from disk");
13477 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
13478 let common = Common::new(reader.clone());
13479 let column = common.column("text").expect("the file has that column");
13480 assert_eq!(
13481 common.rows_with(column, &Bound::Bytes(Vec::new())),
13482 Stat::exact(3, Provenance::FrequencySynopsis)
13483 );
13484 assert_eq!(
13485 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
13486 Stat::exact(0, Provenance::FrequencySynopsis)
13487 );
13488 assert_eq!(
13489 reader.reads().dictionaries,
13490 0,
13491 "the bounded spellings answer without opening the dictionary index"
13492 );
13493 fs::remove_file(&path).expect("clean up");
13494 }
13495
13496 #[test]
13497 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
13498 let path = path("certified_host_groups");
13499 let mut writer =
13500 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
13501 .expect("new file");
13502 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
13503 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
13504 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
13505 values.push(Value::Varchar(String::new()));
13506 for part in values.chunks(512) {
13507 writer
13508 .append(
13509 &Chunk::new(vec![
13510 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
13511 ])
13512 .expect("one column"),
13513 )
13514 .expect("part written");
13515 }
13516 writer.finish().expect("commit");
13517 let reader = Reader::open(&path).expect("reopen");
13518 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
13519 fs::remove_file(&path).expect("clean up");
13520 }
13521
13522 fn bare_table(sections: Vec<Section>) -> Table {
13527 Table {
13528 name: "linked".to_owned(),
13529 fields: vec![Field::required("id", LogicalType::Integer)],
13530 stripes: Vec::new(),
13531 rows: 0,
13532 dictionaries: vec![None],
13533 dictionary_payloads: Vec::new(),
13534 demoted: Vec::new(),
13535 distincts: vec![None],
13536 frequencies: vec![None],
13537 pair_frequencies: Vec::new(),
13538 frequency_texts: Vec::new(),
13539 host_groups: None,
13540 clustering: None,
13541 constraints: Constraints::default(),
13542 generation: 1,
13543 sections,
13544 }
13545 }
13546
13547 fn a_key_map_section() -> Section {
13548 Section {
13549 kind: *section::KEY_MAP,
13550 id: 1,
13551 generation: 3,
13552 extents: 1,
13553 extent_page: HEADER,
13554 extent_bytes: section::EXTENT_BYTES as u32,
13555 hash: 0x1234_5678_9abc_def0,
13556 flags: 0,
13557 header_bytes: 24,
13558 }
13559 }
13560
13561 #[test]
13562 fn a_section_table_round_trips_through_a_directory() {
13563 let mut later = a_key_map_section();
13564 later.kind = *b"RUDBZZ9\0";
13565 later.id = 2;
13566 let table = bare_table(vec![a_key_map_section(), later]);
13567 let directory = encode_directory(&table).expect("directory");
13568 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13569 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
13570 assert!(decoded.sections()[0].known());
13574 assert!(!decoded.sections()[1].known());
13575 }
13576
13577 #[test]
13578 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
13579 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13583 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
13584 let older = &directory[..directory.len() - block];
13585 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
13586 assert!(decoded.sections().is_empty());
13587 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
13588 assert_eq!(decoded.name(), "linked");
13589 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
13590 }
13591
13592 #[test]
13593 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
13594 let path = path("format_twenty_two");
13601 let mut writer =
13602 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13603 .expect("new file");
13604 let rows = Chunk::new(vec![
13605 Vector::from_values(
13606 LogicalType::Integer,
13607 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
13608 )
13609 .expect("integers"),
13610 ])
13611 .expect("one column");
13612 writer.append(&rows).expect("the only part");
13613 writer.finish().expect("commit");
13614
13615 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13616 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13617 drop(file);
13618
13619 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
13620 assert_eq!(reader.table().rows(), 3);
13621 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
13626
13627 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13630 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
13631 drop(file);
13632 let error = Reader::open(&path).expect_err("format 21 is not readable");
13633 assert!(error.to_string().contains("format 21"), "{error}");
13634
13635 fs::remove_file(&path).expect("clean up");
13636 }
13637
13638 #[test]
13639 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
13640 let mut past = a_key_map_section();
13645 past.extent_page = 1 << 30;
13646 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
13647 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
13648 assert!(error.to_string().contains("outside the file"), "{error}");
13649
13650 let mut inside_the_header = a_key_map_section();
13651 inside_the_header.extent_page = 8;
13652 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13653 assert!(
13654 decode_directory(&directory, 1 << 20).is_err(),
13655 "a section may not overlap a header"
13656 );
13657 }
13658
13659 #[test]
13660 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13661 let not_built = Section {
13665 kind: *section::FORWARD_LINK,
13666 id: 9,
13667 generation: 3,
13668 extents: 0,
13669 extent_page: 0,
13670 extent_bytes: 0,
13671 hash: 0,
13672 flags: 0,
13673 header_bytes: 0,
13674 };
13675 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13676 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13677 assert_eq!(decoded.sections(), &[not_built]);
13678
13679 let mut incoherent = not_built;
13682 incoherent.extent_bytes = 28;
13683 incoherent.extent_page = HEADER;
13684 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13685 assert!(decode_directory(&directory, 1 << 20).is_err());
13686 }
13687
13688 #[test]
13689 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13690 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13691 let mut torn = directory.clone();
13692 let count_at = torn.len() - size_of::<u16>();
13693 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13694 assert!(decode_directory(&torn, 1 << 20).is_err());
13697 }
13698
13699 fn linked_file(label: &str, rows: i32) -> PathBuf {
13701 let path = path(label);
13702 let mut writer =
13703 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13704 .expect("new file");
13705 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13706 let chunk =
13707 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13708 .expect("one column");
13709 writer.append(&chunk).expect("the only part");
13710 writer.finish().expect("commit");
13711 path
13712 }
13713
13714 fn a_key_map_payload() -> Vec<u8> {
13715 (0..512_u32).flat_map(u32::to_le_bytes).collect()
13718 }
13719
13720 #[test]
13721 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13722 let path = linked_file("attach", 64);
13723 let payload = a_key_map_payload();
13724 let table = attach(
13725 &path,
13726 "items",
13727 &[section::Attachment {
13728 kind: *section::KEY_MAP,
13729 id: 0,
13730 flags: 2,
13731 header_bytes: 40,
13732 bytes: &payload,
13733 }],
13734 )
13735 .expect("attach a key map");
13736 assert_eq!(attached(&table).len(), 1);
13737
13738 let reader = Reader::open(&path).expect("reopen after the attach");
13739 let held = attached(reader.table());
13740 assert_eq!(held.len(), 1);
13741 assert_eq!(held[0].kind, *section::KEY_MAP);
13742 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13743 assert_eq!(held[0].header_bytes, 40);
13744 assert_eq!(held[0].generation, 1);
13748 assert!(held[0].usable(reader.table().generation()));
13749 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13750 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13751
13752 fs::remove_file(&path).expect("clean up");
13753 }
13754
13755 #[test]
13756 fn attaching_a_section_answers_every_row_exactly_as_before() {
13757 let path = linked_file("attach_changes_nothing", 300);
13762 let before = Reader::open(&path).expect("open before");
13763 let rows = before.table().rows();
13764 let first = before.read(0, &[0]).expect("read before");
13765 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
13766 let layout = before.layout().columns_total();
13767 drop(before);
13768
13769 let payload = a_key_map_payload();
13770 attach(
13771 &path,
13772 "items",
13773 &[section::Attachment {
13774 kind: *section::KEY_MAP,
13775 id: 0,
13776 flags: 0,
13777 header_bytes: 0,
13778 bytes: &payload,
13779 }],
13780 )
13781 .expect("attach");
13782
13783 let after = Reader::open(&path).expect("open after");
13784 assert_eq!(after.table().rows(), rows);
13785 let read = after.read(0, &[0]).expect("read after");
13786 for (at, value) in values.iter().enumerate() {
13787 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
13788 }
13789 assert_eq!(
13790 after.layout().columns_total(),
13791 layout,
13792 "an attach appends and does not rewrite a column page"
13793 );
13794
13795 fs::remove_file(&path).expect("clean up");
13796 }
13797
13798 #[test]
13799 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
13800 let path = linked_file("attach_twice", 32);
13804 let one = a_key_map_payload();
13805 let two = vec![7_u8; 1024];
13806 let entry = |bytes| section::Attachment {
13807 kind: *section::KEY_MAP,
13808 id: 4,
13809 flags: 1,
13810 header_bytes: 0,
13811 bytes,
13812 };
13813 attach(&path, "items", &[entry(&one)]).expect("first build");
13814 attach(&path, "items", &[entry(&two)]).expect("rebuild");
13815
13816 let reader = Reader::open(&path).expect("reopen");
13817 let held = attached(reader.table());
13818 assert_eq!(held.len(), 1, "one map per column and not one per build");
13819 assert_eq!(reader.payload(held[0]).expect("payload"), two);
13820
13821 fs::remove_file(&path).expect("clean up");
13822 }
13823
13824 #[test]
13825 fn an_attach_carries_through_a_kind_it_does_not_know() {
13826 let path = linked_file("attach_unknown", 16);
13830 let payload = vec![3_u8; 96];
13831 attach(
13832 &path,
13833 "items",
13834 &[section::Attachment {
13835 kind: *b"RUDBZZ9\0",
13836 id: 1,
13837 flags: 0,
13838 header_bytes: 0,
13839 bytes: &payload,
13840 }],
13841 )
13842 .expect("a kind this build does not know still writes");
13843 let key_map = a_key_map_payload();
13844 attach(
13845 &path,
13846 "items",
13847 &[section::Attachment {
13848 kind: *section::KEY_MAP,
13849 id: 0,
13850 flags: 0,
13851 header_bytes: 0,
13852 bytes: &key_map,
13853 }],
13854 )
13855 .expect("attach beside it");
13856
13857 let reader = Reader::open(&path).expect("reopen");
13858 let held = attached(reader.table());
13859 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13860 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13861 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13862
13863 fs::remove_file(&path).expect("clean up");
13864 }
13865
13866 #[test]
13867 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13868 let path = linked_file("attach_not_built", 8);
13869 attach(
13870 &path,
13871 "items",
13872 &[section::Attachment {
13873 kind: *section::FORWARD_LINK,
13874 id: 2,
13875 flags: 0,
13876 header_bytes: 0,
13877 bytes: &[],
13878 }],
13879 )
13880 .expect("record a link that did not fit the budget");
13881
13882 let reader = Reader::open(&path).expect("reopen");
13883 let held = attached(reader.table());
13884 assert_eq!(held.len(), 1);
13885 assert_eq!(held[0].extents, 0);
13886 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13887 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13888 assert!(reader.payload(held[0]).expect("no payload").is_empty());
13889
13890 fs::remove_file(&path).expect("clean up");
13891 }
13892
13893 #[test]
13894 fn a_payload_past_one_extent_is_split_and_joined_back() {
13895 let path = linked_file("attach_two_extents", 8);
13899 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13900 attach(
13901 &path,
13902 "items",
13903 &[section::Attachment {
13904 kind: *section::KEY_MAP,
13905 id: 0,
13906 flags: 0,
13907 header_bytes: 0,
13908 bytes: &payload,
13909 }],
13910 )
13911 .expect("attach a payload past the bound");
13912
13913 let reader = Reader::open(&path).expect("reopen");
13914 let held = attached(reader.table());
13915 let extents = reader.extents(held[0]).expect("extent table");
13916 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13917 assert_eq!(extents[0].length, section::MAX_EXTENT);
13918 assert_eq!(extents[1].length, 1);
13919 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13920 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13922 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13923
13924 fs::remove_file(&path).expect("clean up");
13925 }
13926
13927 #[test]
13928 fn a_torn_extent_is_refused_rather_than_decoded() {
13929 let path = linked_file("attach_torn", 8);
13930 let payload = a_key_map_payload();
13931 attach(
13932 &path,
13933 "items",
13934 &[section::Attachment {
13935 kind: *section::KEY_MAP,
13936 id: 0,
13937 flags: 0,
13938 header_bytes: 0,
13939 bytes: &payload,
13940 }],
13941 )
13942 .expect("attach");
13943
13944 let reader = Reader::open(&path).expect("reopen");
13945 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13946 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13947 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13948 drop(file);
13949
13950 let reader = Reader::open(&path).expect("the table still opens");
13951 let error = reader
13952 .payload(&reader.table().sections()[0])
13953 .expect_err("a corrupt payload is not handed out");
13954 assert!(error.to_string().contains("checksum"), "{error}");
13955 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13958
13959 fs::remove_file(&path).expect("clean up");
13960 }
13961
13962 #[test]
13963 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13964 let path = linked_file("attach_old_format", 8);
13967 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13968 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13969 drop(file);
13970
13971 let payload = a_key_map_payload();
13972 let error = attach(
13973 &path,
13974 "items",
13975 &[section::Attachment {
13976 kind: *section::KEY_MAP,
13977 id: 0,
13978 flags: 0,
13979 header_bytes: 0,
13980 bytes: &payload,
13981 }],
13982 )
13983 .expect_err("format 22 cannot gain a section");
13984 assert!(error.to_string().contains("format 22"), "{error}");
13985 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13986
13987 fs::remove_file(&path).expect("clean up");
13988 }
13989
13990 #[test]
13991 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13992 let path = linked_file("attach_bad_header", 8);
13993 let error = attach(
13994 &path,
13995 "items",
13996 &[section::Attachment {
13997 kind: *section::KEY_MAP,
13998 id: 0,
13999 flags: 0,
14000 header_bytes: 40,
14001 bytes: &[1, 2, 3],
14002 }],
14003 )
14004 .expect_err("a writer's bug stops at the write");
14005 assert!(error.to_string().contains("header is longer"), "{error}");
14006
14007 fs::remove_file(&path).expect("clean up");
14008 }
14009
14010 #[test]
14011 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
14012 let path = linked_file("attach_wrong_name", 8);
14013 let error = attach(&path, "orders", &[]).expect_err("no such table");
14014 assert!(error.to_string().contains("orders"), "{error}");
14015 fs::remove_file(&path).expect("clean up");
14016 }
14017
14018 #[test]
14019 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
14020 let path = path("frequency_prefix_for_the_planner");
14027 let mut writer =
14028 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14029 .expect("new file");
14030 let mut values = vec![Value::Integer(1); 10_000];
14031 for _ in 0..10 {
14032 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
14033 }
14034 for part in values.chunks(8_000) {
14037 let rows = Chunk::new(vec![
14038 Vector::from_values(LogicalType::Integer, part).expect("integers"),
14039 ])
14040 .expect("one column");
14041 writer.append(&rows).expect("a part");
14042 }
14043 writer.finish().expect("commit");
14044 let reader = Reader::open(&path).expect("reopen from disk");
14045 let prefix =
14046 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
14047 assert_eq!(prefix.entries.len(), 512);
14050 assert_eq!(prefix.omitted_max, 10);
14051 let common = Common::new(reader);
14052 assert_eq!(common.rows(), 16_000);
14053 let column = common.column("id").expect("the file has that column");
14054 assert_eq!(
14055 common.rows_with(column, &Bound::Int(1)),
14056 Stat::exact(10_000, Provenance::FrequencySynopsis)
14057 );
14058 assert_eq!(
14060 common.rows_with(column, &Bound::Int(1_100)),
14061 Stat::exact(10, Provenance::FrequencySynopsis)
14062 );
14063 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
14066 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
14069 let remainder = common.remainder(column).expect("the list is a prefix");
14073 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
14074 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
14075 fs::remove_file(&path).expect("clean up");
14076 }
14077
14078 #[test]
14080 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
14081 let path = path("empty");
14082 Writer::empty(&path, &[]).expect("a file with nothing in it");
14083 let catalog = Catalog::open(&path).expect("the empty file opens");
14084 assert_eq!(catalog.len(), 0);
14085 assert!(catalog.is_empty());
14086 assert_eq!(catalog.names().count(), 0);
14087 let mut writer =
14090 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14091 .expect("a table goes into the empty file");
14092 writer.append(&sample_ids()).expect("rows");
14093 writer.finish().expect("commit");
14094 let catalog = Catalog::open(&path).expect("the file opens again");
14095 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14096 fs::remove_file(&path).expect("clean up");
14097 }
14098
14099 #[test]
14109 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
14110 let path = path("empty-name");
14111 let field = || vec![Field::required("id", LogicalType::Integer)];
14112 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
14113 let catalog = Catalog::open(&path).expect("the file opens");
14114 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
14115
14116 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
14117 writer.append(&sample_ids()).expect("rows");
14118 writer.finish().expect("commit");
14119 let catalog = Catalog::open(&path).expect("the file opens again");
14120 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14122 let held = catalog.rows().collect::<Vec<_>>();
14123 assert_eq!(held.len(), 1);
14124 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
14125
14126 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
14128 assert!(error.to_string().contains("same name"), "{error}");
14129 fs::remove_file(&path).expect("clean up");
14130 }
14131
14132 #[test]
14133 fn a_device_card_rides_the_catalog_and_an_older_catalog_has_none() {
14134 let entry = || Entry {
14135 name: "items".to_string(),
14136 fields: vec![Field::required("id", LogicalType::Integer)],
14137 rows: 1,
14138 directory: Page { offset: HEADER, length: 8, hash: 0 },
14139 nonzero: vec![None],
14140 aggregates: vec![None],
14141 distincts: vec![None],
14142 extremes: vec![None],
14143 frequencies: vec![None],
14144 };
14145 let card = KeptCard { device: "dev:42".to_string(), bytes: vec![1, 2, 3] };
14146 let bytes = encode_catalog(&[entry()], &[], Some(&card)).expect("encodes");
14147 let (entries, views, kept) = decode_catalog(&bytes, HEADER + 8).expect("decodes");
14148 assert_eq!((entries.len(), views.len()), (1, 0));
14149 assert_eq!(kept, Some(card));
14150 let bytes = encode_catalog(&[entry()], &[], None).expect("encodes");
14151 assert_eq!(decode_catalog(&bytes, HEADER + 8).expect("decodes").2, None);
14152 }
14153
14154 fn sample_view(name: &str) -> ViewEntry {
14156 ViewEntry {
14157 name: name.to_string(),
14158 sql: "SELECT id FROM items WHERE id > 0".to_string(),
14159 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
14160 aliases: vec!["n".to_string()],
14161 columns: vec![Field::new("n", LogicalType::Integer)],
14162 }
14163 }
14164
14165 #[test]
14166 fn a_view_written_into_the_catalog_comes_back_whole() {
14167 let path = path("views");
14168 let mut writer =
14169 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14170 .expect("new file");
14171 writer.append(&sample_ids()).expect("rows");
14172 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14173 let catalog = Catalog::open(&path).expect("reopen");
14174 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
14175 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14178 fs::remove_file(&path).expect("clean up");
14179 }
14180
14181 #[test]
14183 fn appending_a_table_carries_the_views_forward() {
14184 let path = path("viewscarry");
14185 let mut writer =
14186 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14187 .expect("new file");
14188 writer.append(&sample_ids()).expect("rows");
14189 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
14190 let mut writer =
14191 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
14192 .expect("a second table");
14193 writer.append(&sample_ids()).expect("rows");
14194 writer.finish().expect("commit");
14195 let catalog = Catalog::open(&path).expect("reopen");
14196 assert_eq!(catalog.views().count(), 1);
14197 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
14198 fs::remove_file(&path).expect("clean up");
14199 }
14200
14201 #[test]
14203 fn restating_the_views_leaves_every_table_where_it_was() {
14204 let path = path("restate");
14205 let mut writer =
14206 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14207 .expect("new file");
14208 writer.append(&sample_ids()).expect("rows");
14209 writer.finish().expect("commit");
14210 let before = fs::metadata(&path).expect("the file is there").len();
14211 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
14212 let catalog = Catalog::open(&path).expect("reopen");
14213 assert_eq!(catalog.views().count(), 2);
14214 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
14215 let after = fs::metadata(&path).expect("the file is there").len();
14218 assert!(after > before, "a generation was written");
14219 assert!(after - before < before, "the table was not written again");
14220 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
14223 assert_eq!(reader.table().rows, 3);
14224 Writer::restate(&path, &[]).expect("no views at all");
14227 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
14228 fs::remove_file(&path).expect("clean up");
14229 }
14230
14231 #[test]
14233 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
14234 let bytes = encode_catalog(
14235 &[Entry {
14236 name: "items".to_string(),
14237 fields: vec![Field::required("id", LogicalType::Integer)],
14238 rows: 1,
14239 directory: Page { offset: HEADER, length: 8, hash: 0 },
14240 nonzero: vec![None],
14241 aggregates: vec![None],
14242 distincts: vec![None],
14243 extremes: vec![None],
14244 frequencies: vec![None],
14245 }],
14246 &[sample_view("items")],
14247 None,
14248 )
14249 .expect("it encodes, because encoding does not look");
14250 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
14251 assert!(error.to_string().contains("same name"), "{error}");
14252 }
14253
14254 #[test]
14257 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
14258 let rows: usize = 300;
14259 let text: Vec<String> =
14260 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
14261 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
14262 let mut page = vec![6, 2];
14263 page.extend((0..rows.div_ceil(8)).map(|byte| {
14264 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
14265 }));
14266 let compressed = string::encode_only(string::Kind::Fsst, &values)
14267 .expect("encoded")
14268 .expect("text this repetitive compresses");
14269 page.extend_from_slice(&compressed);
14270 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
14271 let positions = [0_u32, 3, 8, 13, 200, 299];
14272 let some =
14273 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
14274 assert_eq!(some.len(), positions.len());
14275 for (at, &row) in positions.iter().enumerate() {
14276 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
14277 }
14278 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
14279 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
14280 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
14281 }
14282
14283 #[test]
14286 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
14287 let path = path("rows");
14288 let mut writer = Writer::create(
14289 &path,
14290 "items",
14291 vec![
14292 Field::required("id", LogicalType::Integer),
14293 Field::new("text", LogicalType::Varchar),
14294 ],
14295 )
14296 .expect("new file");
14297 let rows = 2_000;
14298 let chunk = Chunk::new(vec![
14299 Vector::from_values(
14300 LogicalType::Integer,
14301 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
14302 )
14303 .expect("integers"),
14304 Vector::from_values(
14305 LogicalType::Varchar,
14306 &(0..rows)
14307 .map(|row| {
14308 if row % 7 == 2 {
14309 Value::Null
14310 } else {
14311 Value::Varchar(format!("a comment about order {}", row * 13))
14312 }
14313 })
14314 .collect::<Vec<_>>(),
14315 )
14316 .expect("strings"),
14317 ])
14318 .expect("matching rows");
14319 writer.append(&chunk).expect("one part");
14320 writer.finish().expect("commit");
14321 let reader = Reader::open(&path).expect("reopen from disk");
14322 let positions = [1_u32, 2, 9, 1_000, 1_999];
14323 for whole in [true, false] {
14324 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
14325 let all = reader.read(0, &[0, 1]).expect("the whole part");
14326 assert_eq!(some.len(), positions.len());
14327 for column in 0..2 {
14328 for (at, &row) in positions.iter().enumerate() {
14329 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
14330 }
14331 }
14332 }
14333 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
14334 }
14335
14336 #[test]
14337 fn committed_file_reopens_and_reads_only_requested_columns() {
14338 let path = path("reopen");
14339 let mut writer = Writer::create(
14340 &path,
14341 "items",
14342 vec![
14343 Field::required("id", LogicalType::Integer),
14344 Field::new("text", LogicalType::Varchar),
14345 ],
14346 )
14347 .expect("new file");
14348 writer.append(&sample()).expect("first part");
14349 writer.append(&sample()).expect("second part");
14350 writer.finish().expect("commit");
14351 let reader = Reader::open(&path).expect("reopen from disk");
14352 assert_eq!(reader.table().rows(), 6);
14353 assert_eq!(reader.table().stripes().len(), 1);
14356 assert_eq!(reader.parts(), 2);
14357 assert_eq!(reader.part_rows(0), 3);
14358 assert_eq!(reader.part_rows(1), 3);
14359 let text = reader.read(1, &[1]).expect("only text page");
14360 assert_eq!(text.width(), 1);
14361 assert_eq!(text.value_at(1, 0), Value::Null);
14362 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14363 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
14364 assert_eq!(sparse.width(), 1);
14365 assert_eq!(sparse.value_at(1, 0), Value::Null);
14366 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
14367 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
14368 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
14369 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
14370 let count = reader.read(0, &[]).expect("no page is needed for count");
14371 assert_eq!(count.len(), 3);
14372 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
14373 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
14374 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
14375 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
14376 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
14377 assert_eq!(integers.omitted_max, 2);
14378 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
14379 assert_eq!(strings.len(), 3);
14380 assert!(strings.contains(&(Value::Null, 2)));
14381 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
14382 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
14383 fs::remove_file(path).expect("remove scratch file");
14384 }
14385
14386 #[test]
14394 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
14395 let path = path("interleaved-runs");
14396 let mut writer =
14397 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
14398 .expect("new file");
14399 for morsel in [2_u64, 0, 3, 1] {
14400 let parts = (0..4_u64)
14401 .map(|chunk| {
14402 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
14403 let values =
14404 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
14405 let column =
14406 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
14407 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
14408 })
14409 .collect::<Vec<_>>();
14410 writer.append_stripe(parts).expect("a stripe");
14411 }
14412 writer.finish().expect("commit");
14413
14414 let reader = Reader::open(&path).expect("valid directory");
14415 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
14416 assert_eq!(reader.table().rows(), 128);
14417 for part in 0..16_usize {
14418 let read = reader.read(part, &[0]).expect("a part back");
14419 for row in 0..8_usize {
14420 let want = i64::try_from(part * 8 + row).expect("small");
14421 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
14422 }
14423 }
14424 fs::remove_file(path).expect("remove scratch file");
14425 }
14426
14427 #[test]
14430 fn runs_that_overlap_each_other_are_refused_at_commit() {
14431 let path = path("overlapping-runs");
14432 let mut writer =
14433 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
14434 .expect("new file");
14435 let one = |order: (u64, u64)| {
14436 let column =
14437 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
14438 (order, Chunk::new(vec![column]).expect("one column"))
14439 };
14440 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
14443 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
14444 let error = writer.finish().expect_err("the runs overlap");
14445 assert!(error.message().contains("source order"), "{error}");
14446 fs::remove_file(path).expect("remove scratch file");
14447 }
14448
14449 #[test]
14452 fn a_run_longer_than_a_stripe_is_refused() {
14453 let path = path("overlong-run");
14454 let mut writer =
14455 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
14456 .expect("new file");
14457 let parts = (0..=STRIPE_PARTS)
14458 .map(|at| {
14459 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
14460 .expect("a column");
14461 let chunk = Chunk::new(vec![column]).expect("one column");
14462 ((0, u64::try_from(at).expect("small")), chunk)
14463 })
14464 .collect::<Vec<_>>();
14465 let error = writer.append_stripe(parts).expect_err("one part too many");
14466 assert!(error.message().contains("more parts than it holds"), "{error}");
14467 fs::remove_file(path).expect("remove scratch file");
14468 }
14469
14470 #[test]
14476 fn parts_past_the_stripe_bound_start_a_new_stripe() {
14477 let path = path("stripe-bound");
14478 let mut writer = Writer::create(
14479 &path,
14480 "items",
14481 vec![
14482 Field::required("id", LogicalType::Integer),
14483 Field::new("text", LogicalType::Varchar),
14484 ],
14485 )
14486 .expect("new file");
14487 let parts = STRIPE_PARTS * 2 + 3;
14488 for part in 0..parts {
14489 let id = part as i32;
14490 let chunk = Chunk::new(vec![
14491 Vector::from_values(
14492 LogicalType::Integer,
14493 &[Value::Integer(id), Value::Integer(-id)],
14494 )
14495 .expect("integers"),
14496 Vector::from_values(
14497 LogicalType::Varchar,
14498 &[Value::Varchar(format!("value {part}")), Value::Null],
14499 )
14500 .expect("strings"),
14501 ])
14502 .expect("matching rows");
14503 writer.append(&chunk).expect("one part");
14504 }
14505 writer.finish().expect("commit");
14506
14507 let reader = Reader::open(&path).expect("reopen from disk");
14508 assert_eq!(reader.parts(), parts);
14509 assert_eq!(reader.table().rows(), parts * 2);
14510 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
14511 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
14512 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
14513 assert_eq!(reader.table().stripes()[2].parts(), 3);
14514 for part in (0..parts).rev() {
14517 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
14518 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
14519 for chunk in [&dense, &sparse] {
14520 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
14521 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14522 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14523 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
14524 assert_eq!(chunk.value_at(1, 1), Value::Null);
14525 }
14526 }
14527 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
14530 assert!(reader.skips(0, &above), "the first stripe stops at 63");
14531 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
14532 fs::remove_file(path).expect("remove scratch file");
14533 }
14534
14535 fn scattered(n: i64) -> i64 {
14537 n.wrapping_mul(-7_046_029_254_386_353_131)
14538 }
14539
14540 #[test]
14546 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
14547 let path = path("sieve-skip");
14548 let mut writer =
14549 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14550 .expect("new file");
14551 let parts = STRIPE_PARTS + 3;
14552 let per_part = 128;
14556 for part in 0..parts {
14557 let held: Vec<Value> = (0..per_part)
14558 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
14559 .collect();
14560 let chunk =
14561 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14562 .expect("one column");
14563 writer.append(&chunk).expect("one part");
14564 }
14565 writer.finish().expect("commit");
14566
14567 let reader = Reader::open(&path).expect("reopen from disk");
14568 let probe = |value: i64| Probe {
14569 column: 0,
14570 op: Op::Equal,
14571 value: Bound::Int(i128::from(scattered(value))),
14572 };
14573 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
14574 let tests = [probe(wanted)];
14575 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
14576 let home = wanted as usize / per_part;
14577 assert!(kept.contains(&home), "the part holding {wanted} is read");
14578 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
14582 }
14583 let absent = [probe((parts * per_part) as i64 + 1)];
14584 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
14585 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
14586 let tests = [probe(0)];
14589 assert!(
14590 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
14591 "the bounds rule out no stripe at all"
14592 );
14593 fs::remove_file(path).expect("remove scratch file");
14594 }
14595
14596 #[test]
14602 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
14603 let path = path("part-range-skip");
14604 let mut writer =
14605 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14606 .expect("new file");
14607 let parts = STRIPE_PARTS + 3;
14608 let per_part = 128;
14609 for part in 0..parts {
14610 let held: Vec<Value> = (0..per_part)
14614 .map(|row| {
14615 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14616 })
14617 .collect();
14618 let chunk =
14619 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14620 .expect("one column");
14621 writer.append(&chunk).expect("one part");
14622 }
14623 writer.finish().expect("commit");
14624
14625 let reader = Reader::open(&path).expect("reopen from disk");
14626 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14627 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
14628 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
14629 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
14631 fs::remove_file(path).expect("remove scratch file");
14632 }
14633
14634 #[test]
14638 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
14639 let path = path("part-range-certain");
14640 let mut writer =
14641 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14642 .expect("new file");
14643 let parts = STRIPE_PARTS + 3;
14644 let per_part = 128;
14645 for part in 0..parts {
14646 let held: Vec<Value> = (0..per_part)
14647 .map(|row| {
14648 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
14649 })
14650 .collect();
14651 let chunk =
14652 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14653 .expect("one column");
14654 writer.append(&chunk).expect("one part");
14655 }
14656 writer.finish().expect("commit");
14657
14658 let reader = Reader::open(&path).expect("reopen from disk");
14659 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
14660 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
14661 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
14662 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
14665 fs::remove_file(path).expect("remove scratch file");
14666 }
14667
14668 #[test]
14671 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14672 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14673 let path = path("part-range-page");
14674 let mut writer =
14675 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14676 .expect("new file");
14677 for part in 0..parts {
14678 let held: Vec<Value> = (0..128)
14679 .map(|row| {
14680 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14681 })
14682 .collect();
14683 let chunk = Chunk::new(vec![
14684 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14685 ])
14686 .expect("one column");
14687 writer.append(&chunk).expect("one part");
14688 }
14689 writer.finish().expect("commit");
14690 let reader = Reader::open(&path).expect("reopen from disk");
14691 let bytes = reader.layout().columns[0].part_ranges;
14692 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14693 fs::remove_file(path).expect("remove scratch file");
14694 }
14695 }
14696
14697 #[test]
14700 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14701 let long = vec![b'a'; PART_BOUND_BYTES * 2];
14702 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14703 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
14704 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
14705 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
14706 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
14707 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
14708 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
14709 }
14710
14711 #[test]
14714 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
14715 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
14716 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
14717 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
14718 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
14719 }
14720
14721 #[test]
14733 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
14734 let parts = 4;
14735 let per_part = 1024;
14736 let rows = parts * per_part;
14737 let written = |name: &str, keys: &[i64]| {
14738 let path = path(name);
14739 let fields = vec![Field::required("key", LogicalType::BigInt)];
14740 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
14741 for part in 0..parts {
14742 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
14743 .iter()
14744 .map(|key| Value::BigInt(*key))
14745 .collect();
14746 let chunk = Chunk::new(vec![
14747 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
14748 ])
14749 .expect("one column");
14750 writer.append(&chunk).expect("one part");
14751 }
14752 writer.finish().expect("commit");
14753 path
14754 };
14755 let climbing = |step: &dyn Fn(usize) -> i64| {
14758 let mut key = 0;
14759 (0..rows)
14760 .map(|row| {
14761 key += step(row);
14762 key
14763 })
14764 .collect::<Vec<i64>>()
14765 };
14766 let ascending = climbing(&|row| (row % 3) as i64);
14767 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
14771 let near_path = written("stored-near", &ascending);
14772 let far_path = written("stored-far", &sparse);
14773
14774 let one = Reader::open(&near_path).expect("reopen from disk");
14775 let other = Reader::open(&far_path).expect("reopen from disk");
14776 let near = one.stored(0).expect("the column is stored");
14777 let far = other.stored(0).expect("the column is stored");
14778 assert_eq!(near.len(), parts, "one row per part");
14779 assert_eq!(far.len(), parts);
14780 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
14783 assert_eq!(total(&near), one.layout().columns[0].pages);
14784 assert_eq!(total(&far), other.layout().columns[0].pages);
14785 assert!(
14786 total(&near) * 2 < total(&far),
14787 "the sparse keys cost more, {} against {}",
14788 total(&far),
14789 total(&near)
14790 );
14791 for (at, part) in near.iter().enumerate() {
14793 assert_eq!(part.part, at);
14794 assert_eq!(part.row, at * per_part);
14795 assert_eq!(part.rows, per_part);
14796 let held = &ascending[at * per_part..(at + 1) * per_part];
14797 assert_eq!(part.low, Some(Value::BigInt(held[0])));
14798 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
14799 assert_eq!(part.nulls, Some(0));
14800 }
14801 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
14804 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
14805 assert_ne!(near[0].encoding, far[0].encoding);
14806 fs::remove_file(near_path).expect("remove scratch file");
14807 fs::remove_file(far_path).expect("remove scratch file");
14808 }
14809
14810 #[test]
14820 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
14821 let path = path("sieve-pays");
14822 let fields = vec![
14823 Field::required("spread", LogicalType::BigInt),
14824 Field::required("repeated", LogicalType::BigInt),
14825 ];
14826 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
14827 let parts = 3;
14828 let per_part = 1024;
14829 for part in 0..parts {
14830 let base = (part * per_part) as i64;
14831 let spread: Vec<Value> =
14832 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
14833 let repeated: Vec<Value> =
14834 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14835 let chunk = Chunk::new(vec![
14836 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14837 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14838 ])
14839 .expect("two columns");
14840 writer.append(&chunk).expect("one part");
14841 }
14842 writer.finish().expect("commit");
14843
14844 let reader = Reader::open(&path).expect("reopen from disk");
14845 let layout = reader.layout();
14846 let spread = &layout.columns[0];
14847 let repeated = &layout.columns[1];
14848 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14849 assert_eq!(
14850 repeated.sieves, 0,
14851 "a column whose filter costs more than its parts keeps none"
14852 );
14853 for column in &layout.columns {
14856 assert!(
14857 column.sieves < column.pages,
14858 "{} spends {} on sieves over {} of data",
14859 column.name,
14860 column.sieves,
14861 column.pages
14862 );
14863 }
14864 let absent = [Probe {
14866 column: 0,
14867 op: Op::Equal,
14868 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14869 }];
14870 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14871 fs::remove_file(path).expect("remove scratch file");
14872 }
14873
14874 #[test]
14880 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14881 let path = path("sieve-damaged");
14882 let mut writer =
14883 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14884 .expect("new file");
14885 let rows = 128;
14886 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14887 let chunk =
14888 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14889 .expect("one column");
14890 writer.append(&chunk).expect("one part");
14891 writer.finish().expect("commit");
14892
14893 let page = Reader::open(&path).expect("reopen").table.stripes[0]
14894 .sieves
14895 .get(0)
14896 .expect("a sieve page");
14897 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14898 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14899 file.write_all(&[0xff]).expect("damage one byte");
14900 drop(file);
14901
14902 let reader = Reader::open(&path).expect("reopen the damaged file");
14903 let absent =
14904 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14905 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14906 assert_eq!(
14907 reader.read(0, &[0]).expect("the rows are untouched").len(),
14908 usize::try_from(rows).expect("a small count")
14909 );
14910 fs::remove_file(path).expect("remove scratch file");
14911 }
14912
14913 #[test]
14919 fn a_part_asked_for_twice_in_one_scan_keeps_its_page_only_to_the_floor() {
14920 let path = path("asked-twice");
14921 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14922 let mut writer =
14923 Writer::create(&path, "a", vec![Field::required("id", LogicalType::Integer)])
14924 .expect("new file");
14925 for part in 0..parts {
14926 let chunk = Chunk::new(vec![
14927 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14928 .expect("integers"),
14929 ])
14930 .expect("matching rows");
14931 writer.append(&chunk).expect("one part");
14932 }
14933 writer.finish().expect("commit");
14934
14935 let pool = PagePool::new(usize::MAX);
14936 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14937 let a = catalog.table("a").expect("a");
14938 let stripes = a.table().stripes().len();
14939 for part in 0..parts {
14940 for _ in 0..2 {
14941 let chunk = a.read(part, &[0]).expect("a part");
14942 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14943 }
14944 }
14945 assert_eq!(
14946 a.pages.load(Atomic::Relaxed),
14947 stripes,
14948 "a page a stripe, read on the second ask"
14949 );
14950 assert_eq!(pool.bytes(), 0, "one scan puts nothing in the pool");
14951 let column = a.cache.columns[0].lock().expect("the column");
14952 assert_eq!(column.pages.iter().flatten().count(), CACHED_STRIPES_PER_COLUMN);
14953 drop(column);
14954 drop((a, catalog));
14955 fs::remove_file(path).expect("remove scratch file");
14956 }
14957
14958 #[test]
14969 fn workers_that_want_the_same_stripe_read_it_once() {
14970 let path = path("single-flight");
14971 let mut writer =
14972 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14973 .expect("new file");
14974 for part in 0..STRIPE_PARTS {
14975 let id = part as i32;
14976 let chunk = Chunk::new(vec![
14977 Vector::from_values(
14978 LogicalType::Integer,
14979 &[Value::Integer(id), Value::Integer(-id)],
14980 )
14981 .expect("integers"),
14982 ])
14983 .expect("matching rows");
14984 writer.append(&chunk).expect("one part");
14985 }
14986 writer.finish().expect("commit");
14987
14988 let reader = Reader::open(&path).expect("reopen from disk");
14989 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14990 for part in 0..STRIPE_PARTS {
14993 reader.read(part, &[0]).expect("a part");
14994 }
14995 assert_eq!(reader.pages.load(Atomic::Relaxed), 0, "the first pass reads no page whole");
14996 let barrier = std::sync::Barrier::new(8);
14997 std::thread::scope(|scope| {
14998 for worker in 0..8 {
14999 let reader = &reader;
15000 let barrier = &barrier;
15001 scope.spawn(move || {
15002 barrier.wait();
15003 for part in (worker..STRIPE_PARTS).step_by(8) {
15004 let chunk = reader.read(part, &[0]).expect("a whole page read");
15005 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15006 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
15007 }
15008 });
15009 }
15010 });
15011 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
15012 fs::remove_file(path).expect("remove scratch file");
15013 }
15014
15015 #[test]
15028 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
15029 let opened = |label: &str, rows_per_part: i32| {
15030 let path = path(label);
15031 let mut writer =
15032 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15033 .expect("new file");
15034 for part in 0..STRIPE_PARTS * 3 {
15035 let values = (0..rows_per_part)
15039 .map(|row| {
15040 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
15041 })
15042 .collect::<Vec<_>>();
15043 let chunk = Chunk::new(vec![
15044 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
15045 ])
15046 .expect("matching rows");
15047 writer.append(&chunk).expect("one part");
15048 }
15049 writer.finish().expect("commit");
15050 let reader = Reader::open(&path).expect("reopen from disk");
15051 let size = fs::metadata(&path).expect("the file is there").len();
15052 let out = (reader.reads(), reader.table().stripes().len(), size);
15053 fs::remove_file(path).expect("remove scratch file");
15054 out
15055 };
15056
15057 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
15058 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
15059 assert_eq!(
15060 thin_stripes, fat_stripes,
15061 "the same stripe count is what makes this a fair ask"
15062 );
15063 assert!(
15064 fat_size > thin_size * 50,
15065 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
15066 );
15067
15068 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
15069 assert_eq!(thin.pages, 0, "opening read a page");
15070 assert_eq!(fat.pages, 0, "opening read a page");
15071 assert_eq!(thin.indexes, 0, "opening read an index");
15072 assert_eq!(fat.indexes, 0, "opening read an index");
15073 assert!(
15076 fat.opening.bytes < thin.opening.bytes * 2,
15077 "opening the thin file read {} bytes and the fat one read {}",
15078 thin.opening.bytes,
15079 fat.opening.bytes
15080 );
15081 }
15082
15083 #[test]
15091 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
15092 let path = path("open-twice");
15093 let mut writer =
15094 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15095 .expect("new file");
15096 for part in 0..STRIPE_PARTS * 3 {
15097 let chunk = Chunk::new(vec![
15098 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15099 .expect("integers"),
15100 ])
15101 .expect("matching rows");
15102 writer.append(&chunk).expect("one part");
15103 }
15104 writer.finish().expect("commit");
15105
15106 let first = Reader::open(&path).expect("open");
15107 for part in 0..first.parts() {
15110 first.read(part, &[0]).expect("a part");
15111 }
15112 assert!(first.reads().indexes > 0, "the scan has to have read something");
15113 let second = Reader::open(&path).expect("open again");
15114
15115 assert_eq!(first.reads().opening, second.reads().opening);
15116 assert_eq!(
15117 second.reads().pages,
15118 0,
15119 "the second open read a page off the back of the first"
15120 );
15121 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
15122 fs::remove_file(path).expect("remove scratch file");
15123 }
15124
15125 #[test]
15133 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
15134 let path = path("index-cache");
15135 let mut writer =
15136 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15137 .expect("new file");
15138 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
15139 for part in 0..parts {
15140 let id = part as i32;
15141 let chunk = Chunk::new(vec![
15142 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
15143 ])
15144 .expect("matching rows");
15145 writer.append(&chunk).expect("one part");
15146 }
15147 writer.finish().expect("commit");
15148
15149 let reader = Reader::open(&path).expect("reopen from disk");
15150 let stripes = reader.table().stripes().len();
15151 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
15152 for _ in 0..3 {
15155 for part in 0..parts {
15156 let chunk = reader.read(part, &[0]).expect("a part");
15157 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15158 }
15159 }
15160 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
15161 assert!(
15162 reader.pages.load(Atomic::Relaxed) > stripes,
15163 "the pages are the ones that get read again, which is what makes the index count mean \
15164 something"
15165 );
15166 fs::remove_file(path).expect("remove scratch file");
15167 }
15168
15169 #[test]
15176 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
15177 let path = path("page-pool");
15178 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
15179 let fields = || vec![Field::required("id", LogicalType::Integer)];
15180 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
15181 for table in ["a", "b"] {
15182 if table == "b" {
15183 writer = writer.next("b".to_string(), fields()).expect("a second table");
15184 }
15185 for part in 0..parts {
15186 let chunk = Chunk::new(vec![
15187 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15188 .expect("integers"),
15189 ])
15190 .expect("matching rows");
15191 writer.append(&chunk).expect("one part");
15192 }
15193 }
15194 writer.finish().expect("commit");
15195
15196 let pool = PagePool::new(usize::MAX);
15197 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
15198 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
15199 let stripes = a.table().stripes().len();
15200 assert!(
15201 stripes > CACHED_STRIPES_PER_COLUMN * 2,
15202 "the floor has to be smaller than a table"
15203 );
15204 let scan = |reader: &Reader| {
15205 for part in 0..parts {
15206 let chunk = reader.read(part, &[0]).expect("a part");
15207 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15208 }
15209 };
15210 scan(&a);
15213 assert_eq!(a.pages.load(Atomic::Relaxed), 0, "the first scan reads no page whole");
15214 assert_eq!(pool.bytes(), 0, "a stripe read once is not the pool's");
15215 scan(&a);
15216 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads every page");
15217 scan(&a);
15218 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the third scan reads nothing");
15219 let one = pool.bytes();
15220 assert!(one > 0, "the pool counts what the reader holds");
15221
15222 pool.budget.store(one, Atomic::Relaxed);
15224 scan(&b);
15225 scan(&b);
15226 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
15227 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
15228 let column = a.cache.columns[0].lock().expect("the column");
15229 let held = column.pages.iter().flatten().count();
15230 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
15231 drop(column);
15232
15233 drop((a, b, catalog));
15235 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
15236 scan(&c);
15237 scan(&c);
15238 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
15239 fs::remove_file(path).expect("remove scratch file");
15240 }
15241
15242 #[test]
15251 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
15252 let workers = CACHED_STRIPES_PER_COLUMN + 4;
15253 let path = path("stripe-per-worker");
15254 let mut writer =
15255 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15256 .expect("new file");
15257 for part in 0..STRIPE_PARTS * workers {
15258 let chunk = Chunk::new(vec![
15259 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
15260 .expect("integers"),
15261 ])
15262 .expect("matching rows");
15263 writer.append(&chunk).expect("one part");
15264 }
15265 writer.finish().expect("commit");
15266
15267 let read = |told: bool| {
15268 let reader = Reader::open(&path).expect("reopen from disk");
15269 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
15270 if told {
15271 reader.keep_stripes(workers);
15272 }
15273 for part in 0..reader.parts() {
15275 reader.read(part, &[0]).expect("a part");
15276 }
15277 let barrier = std::sync::Barrier::new(workers);
15278 std::thread::scope(|scope| {
15279 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
15280 let reader = &reader;
15281 let barrier = &barrier;
15282 scope.spawn(move || {
15283 for part in run {
15284 barrier.wait();
15285 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
15286 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
15287 }
15288 assert!(worker < workers);
15289 });
15290 }
15291 });
15292 reader.pages.load(Atomic::Relaxed)
15293 };
15294
15295 assert_eq!(read(true), workers, "one page read per stripe and no more");
15296 assert!(read(false) > workers, "a cache that small is read again on every part");
15297 fs::remove_file(path).expect("remove scratch file");
15298 }
15299
15300 #[test]
15305 fn a_damaged_index_page_is_an_error() {
15306 let path = path("damaged-index");
15307 let mut writer =
15308 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
15309 .expect("new file");
15310 writer.append(&sample_ids()).expect("first part");
15311 writer.append(&sample_ids()).expect("second part");
15312 writer.finish().expect("commit");
15313
15314 let reader = Reader::open(&path).expect("valid directory");
15315 let index = reader.table.stripes[0].index;
15316 let mut byte = [0; 1];
15317 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
15318 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
15319 file.seek(SeekFrom::Start(index.offset)).expect("index start");
15320 file.write_all(&[!byte[0]]).expect("damage the first part length");
15321 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
15322 assert!(error.message().contains("index page section checksum differs"), "{error}");
15323 fs::remove_file(path).expect("remove scratch file");
15324 }
15325
15326 #[test]
15333 fn every_integer_width_round_trips_through_a_page() {
15334 let path = path("integer-widths");
15335 let columns = [
15336 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
15337 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
15338 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
15339 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
15340 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
15341 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
15342 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
15343 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
15344 ];
15345 let fields = columns
15346 .iter()
15347 .enumerate()
15348 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15349 .collect::<Vec<_>>();
15350 let vectors = columns
15351 .iter()
15352 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15353 .collect::<Vec<_>>();
15354 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
15355 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15356 writer.finish().expect("commit");
15357
15358 let reader = Reader::open(&path).expect("reopen from disk");
15359 let wanted = (0..columns.len()).collect::<Vec<_>>();
15360 let read = reader.read(0, &wanted).expect("every column");
15361 assert_eq!(read.len(), 2);
15362 for (at, (ty, values)) in columns.iter().enumerate() {
15364 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15365 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15366 }
15367 fs::remove_file(path).expect("remove scratch file");
15368 }
15369
15370 #[test]
15381 fn every_other_type_the_format_knows_round_trips_through_a_page() {
15382 let path = path("other-types");
15383 let columns = [
15384 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
15385 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
15386 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
15387 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
15388 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
15389 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
15390 (
15391 LogicalType::TimestampTz,
15392 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
15393 ),
15394 (
15395 LogicalType::Interval,
15396 vec![
15397 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
15398 Value::Interval { months: 13, days: -1, micros: 1 },
15399 ],
15400 ),
15401 (
15402 LogicalType::Blob,
15403 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
15404 ),
15405 ];
15406 let fields = columns
15407 .iter()
15408 .enumerate()
15409 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
15410 .collect::<Vec<_>>();
15411 let vectors = columns
15412 .iter()
15413 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
15414 .collect::<Vec<_>>();
15415 let mut writer = Writer::create(&path, "others", fields).expect("new file");
15416 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15417 writer.finish().expect("commit");
15418
15419 let reader = Reader::open(&path).expect("reopen from disk");
15420 let wanted = (0..columns.len()).collect::<Vec<_>>();
15421 let read = reader.read(0, &wanted).expect("every column");
15422 assert_eq!(read.len(), 2);
15423 for (at, (ty, values)) in columns.iter().enumerate() {
15424 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
15425 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
15426 }
15427 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
15430 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
15431
15432 fs::remove_file(path).expect("remove scratch file");
15433 }
15434
15435 #[test]
15441 fn a_nan_survives_being_written_down() {
15442 let path = path("nan");
15443 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
15444 .expect("a NaN vector");
15445 let mut writer =
15446 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
15447 .expect("new file");
15448 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
15449 writer.finish().expect("commit");
15450 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
15451 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
15452 assert!(back.is_nan(), "a NaN came back as {back}");
15453 fs::remove_file(path).expect("remove scratch file");
15454 }
15455
15456 #[test]
15463 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
15464 let path = path("uuid-and-bit");
15465 let uuids = vec![0_i128, i128::MIN, -1];
15466 let mut bits = StringColumn::new();
15467 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
15468 bits.push_bytes(value);
15469 }
15470 let expected = bits.clone();
15471 let fields =
15472 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
15473 let vectors = vec![
15474 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
15475 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
15476 ];
15477 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
15478 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
15479 writer.finish().expect("commit");
15480
15481 let reader = Reader::open(&path).expect("reopen from disk");
15482 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
15483 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
15484 panic!("a uuid column is the 128 bit lane")
15485 };
15486 assert_eq!(back.as_slice(), uuids.as_slice());
15487 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
15488 panic!("a bit column is bytes")
15489 };
15490 for row in 0..expected.len() {
15491 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
15492 }
15493 fs::remove_file(path).expect("remove scratch file");
15494 }
15495
15496 #[test]
15499 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
15500 let mut rows: Vec<Option<u64>> = Vec::new();
15501 let mut state = 0x2545_f491_4f6c_dd1d_u64;
15502 for index in 0..400_000_u64 {
15503 state ^= state << 13;
15504 state ^= state >> 7;
15505 state ^= state << 17;
15506 let times = 1 + (state % 7) as usize;
15507 let bits = match state % 11 {
15508 0 => None,
15509 1..=3 => Some(state % 16),
15510 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
15511 };
15512 rows.extend(std::iter::repeat_n(bits, times));
15513 }
15514 let mut by_row = Candidates::default();
15515 for &bits in &rows {
15516 by_row.add(bits, 1);
15517 }
15518 let mut by_run = Candidates::default();
15519 let mut run = Run::default();
15520 let mut runs = 0_usize;
15521 for &bits in &rows {
15522 if let Some((bits, times)) = run.push(bits) {
15523 by_run.add(bits, times);
15524 runs += 1;
15525 }
15526 }
15527 if let Some((bits, times)) = run.take() {
15528 by_run.add(bits, times);
15529 }
15530 assert!(runs < rows.len() / 2, "the rows came in runs");
15531 assert!(by_row.decrements > 0, "the table filled and turned values away");
15532 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
15533 assert_eq!(by_run.nulls, by_row.nulls);
15534 assert_eq!(by_run.decrements, by_row.decrements);
15535 }
15536
15537 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
15538 let mut pairs = candidates.pairs().collect::<Vec<_>>();
15539 pairs.sort_unstable();
15540 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
15541 pairs
15542 }
15543
15544 #[derive(Default)]
15547 struct MapCandidates {
15548 counts: HashMap<u64, u32>,
15549 nulls: u32,
15550 decrements: u64,
15551 }
15552
15553 impl MapCandidates {
15554 fn add(&mut self, bits: Option<u64>, mut times: u32) {
15555 while times > 0 {
15556 let held = match bits {
15557 Some(bits) => self.counts.get_mut(&bits),
15558 None if self.nulls != 0 => Some(&mut self.nulls),
15559 None => None,
15560 };
15561 if let Some(count) = held {
15562 *count = count.saturating_add(times);
15563 return;
15564 }
15565 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
15566 match bits {
15567 Some(bits) => {
15568 self.counts.insert(bits, times);
15569 }
15570 None => self.nulls = times,
15571 }
15572 return;
15573 }
15574 self.counts.retain(|_, count| {
15575 *count -= 1;
15576 *count != 0
15577 });
15578 self.nulls = self.nulls.saturating_sub(1);
15579 self.decrements = self.decrements.saturating_add(1);
15580 times -= 1;
15581 }
15582 }
15583 }
15584
15585 #[test]
15589 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
15590 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
15591 let mut table = Candidates::default();
15592 let mut oracle = MapCandidates::default();
15593 let mut state = seed;
15594 for index in 0..300_000_u64 {
15595 state ^= state << 13;
15596 state ^= state >> 7;
15597 state ^= state << 17;
15598 let bits = match state % 13 {
15599 0 => None,
15600 1..=4 => Some(state % 40),
15601 5 => Some((index % 1000) * 1_000_000),
15602 _ => Some(state),
15603 };
15604 let times = 1 + (state >> 60) as u32 % 3;
15605 table.add(bits, times);
15606 oracle.add(bits, times);
15607 if index % 50_000 == 0 {
15608 let mut expected =
15609 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15610 expected.sort_unstable();
15611 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
15612 }
15613 }
15614 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
15615 expected.sort_unstable();
15616 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
15617 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
15618 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
15619 assert!(table.decrements > 0, "seed {seed} never filled the table");
15620 for &(bits, _) in &expected {
15621 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
15622 }
15623 }
15624 }
15625
15626 #[test]
15627 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
15628 let path = path("frequency-ordinals");
15629 let mut writer =
15630 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
15631 .expect("new file");
15632 let mut values = Vec::new();
15633 for leader in 0..10_i64 {
15634 values.extend(std::iter::repeat_n(leader, 100));
15635 }
15636 values.extend(1_000_i64..41_000);
15637 for part in values.chunks(1_024) {
15638 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
15639 .expect("big integers");
15640 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
15641 }
15642 writer.finish().expect("commit");
15643
15644 let reader = Reader::open(&path).expect("reopen from disk");
15645 let occurrences =
15646 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
15647 assert!(occurrences.omitted_max < 100);
15648 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
15649 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
15650 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
15651 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
15652 assert_eq!(
15653 &occurrences.anchor_indices[..1_000]
15654 .iter()
15655 .map(|&entry| occurrences.anchors[entry as usize].clone())
15656 .collect::<Vec<_>>(),
15657 &(0_i64..10)
15658 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
15659 .collect::<Vec<_>>()
15660 );
15661 fs::remove_file(path).expect("remove scratch file");
15662 }
15663
15664 #[test]
15665 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
15666 let path = path("frequency-bits");
15671 let mut writer = Writer::create(
15672 &path,
15673 "items",
15674 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
15675 )
15676 .expect("new file");
15677 let mut rows = Vec::new();
15678 let mut leaders = Vec::new();
15679 for leader in 0..10_u64 {
15680 let count = 300 - leader * 10;
15681 let (unsigned, signed) = if leader == 0 {
15682 (Value::Null, Value::Null)
15683 } else {
15684 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
15685 };
15686 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
15687 leaders.push(((unsigned, count), (signed, count)));
15688 }
15689 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
15690 for part in rows.chunks(1_024) {
15691 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
15692 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
15693 let chunk = Chunk::new(vec![
15694 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
15695 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
15696 ])
15697 .expect("matching columns");
15698 writer.append(&chunk).expect("rows");
15699 }
15700 writer.finish().expect("commit");
15701
15702 let reader = Reader::open(&path).expect("reopen from disk");
15703 for column in 0..2 {
15704 let prefix =
15705 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15706 let wanted = leaders
15707 .iter()
15708 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
15709 .cloned()
15710 .collect::<Vec<_>>();
15711 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
15712 assert!(prefix.omitted_max < 210, "column {column}");
15713 assert_eq!(
15714 reader.distinct_values(column).expect("valid metadata"),
15715 Some(9 + 40_000),
15716 "column {column}"
15717 );
15718 }
15719 fs::remove_file(path).expect("remove scratch file");
15720 }
15721
15722 #[test]
15723 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
15724 let path = path("frequency-tally");
15730 let types = [
15731 LogicalType::TinyInt,
15732 LogicalType::UInteger,
15733 LogicalType::Date,
15734 LogicalType::Timestamp,
15735 ];
15736 let value = |ty: &LogicalType, at: i64| match ty {
15737 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
15738 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
15739 LogicalType::Date => Value::Date(19_000 - at as i32),
15740 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
15741 };
15742 let fields = types
15743 .iter()
15744 .enumerate()
15745 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
15746 .collect::<Vec<_>>();
15747 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15748 let mut rows = Vec::new();
15749 for at in 0..250_i64 {
15750 for _ in 0..=(at % 37) {
15751 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
15752 }
15753 }
15754 for part in rows.chunks(1_000) {
15755 let columns = types
15756 .iter()
15757 .map(|ty| {
15758 let values = part
15759 .iter()
15760 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
15761 .collect::<Vec<_>>();
15762 Vector::from_values(ty.clone(), &values).expect("a column")
15763 })
15764 .collect();
15765 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
15766 }
15767 writer.finish().expect("commit");
15768
15769 let reader = Reader::open(&path).expect("reopen from disk");
15770 for (column, ty) in types.iter().enumerate() {
15771 let mut counts = HashMap::<Option<i64>, u64>::new();
15772 for row in &rows {
15773 *counts.entry(*row).or_default() += 1;
15774 }
15775 let wanted = counts
15776 .into_iter()
15777 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
15778 .collect::<Vec<_>>();
15779 let prefix =
15780 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15781 assert_eq!(prefix.entries.len(), 2, "column {column}");
15782 assert!(prefix.omitted_max > 0, "column {column}");
15783 for (value, count) in &prefix.entries {
15784 let held =
15785 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
15786 assert_eq!(held, Some(count), "column {column} value {value:?}");
15787 }
15788 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
15789 assert_eq!(
15790 reader.distinct_values(column).expect("valid metadata"),
15791 Some(wanted.len() as u64 - 1),
15792 "column {column}"
15793 );
15794 }
15795 fs::remove_file(path).expect("remove scratch file");
15796 }
15797
15798 #[test]
15799 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
15800 let edge = FREQUENCY_CANDIDATES as i64;
15805 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
15806 for with_null in [false, true] {
15807 let path = path("distinct-edge");
15808 let mut writer =
15809 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
15810 .expect("new file");
15811 let mut values = Vec::new();
15812 for round in 0..2 {
15813 for value in 0..distinct {
15814 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
15815 values.extend(std::iter::repeat_n(
15816 Value::BigInt(value * 7_919 % distinct),
15817 repeat,
15818 ));
15819 if with_null && value % 1_000 == 0 {
15820 values.push(Value::Null);
15821 }
15822 }
15823 }
15824 if with_null {
15825 values.push(Value::Null);
15826 }
15827 for part in values.chunks(1_024) {
15828 let chunk = Chunk::new(vec![
15829 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
15830 ])
15831 .expect("one column");
15832 writer.append(&chunk).expect("rows");
15833 }
15834 writer.finish().expect("commit");
15835 let reader = Reader::open(&path).expect("reopen from disk");
15836 assert_eq!(
15837 reader.distinct_values(0).expect("valid metadata"),
15838 Some(distinct as u64),
15839 "{distinct} values, null {with_null}"
15840 );
15841 fs::remove_file(path).expect("remove scratch file");
15842 }
15843 }
15844 }
15845
15846 #[test]
15847 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
15848 let path = path("quick-nonzero");
15849 let mut writer = Writer::create(
15850 &path,
15851 "items",
15852 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
15853 )
15854 .expect("create");
15855 for ids in [
15856 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
15857 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
15858 ] {
15859 let labels = vec![Value::Varchar("same".into()); ids.len()];
15860 writer
15861 .append(
15862 &Chunk::new(vec![
15863 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
15864 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
15865 ])
15866 .expect("chunk"),
15867 )
15868 .expect("append");
15869 }
15870 writer.finish().expect("finish");
15871 let catalog = Catalog::open(&path).expect("catalog");
15872 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
15873 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
15874 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
15875 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
15876 let prefix = catalog
15877 .table("items")
15878 .expect("reader")
15879 .frequency_prefix(1)
15880 .expect("valid metadata")
15881 .expect("partial frequencies");
15882 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
15883 assert_eq!(prefix.omitted_max, 1);
15884 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
15885 assert_eq!(
15886 catalog.integer_extremes("items", 1).expect("extremes"),
15887 Some(IntegerExtremes::Values { low: 0, high: 7 })
15888 );
15889 assert_eq!(
15890 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15891 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15892 );
15893 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15894 let mut legacy = catalog.clone();
15895 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15896 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15897 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15898 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15899 Writer::certify_counts(&path).expect("recertify");
15900 assert_eq!(
15901 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15902 Some(2)
15903 );
15904 assert_eq!(
15905 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15906 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15907 );
15908 assert_eq!(
15909 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15910 Some(3)
15911 );
15912 assert_eq!(
15913 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15914 Some(IntegerExtremes::Values { low: 0, high: 7 })
15915 );
15916 assert_eq!(
15917 Catalog::open(&path)
15918 .expect("reopen")
15919 .exact_numeric_frequencies("items", 1)
15920 .expect("frequencies"),
15921 None
15922 );
15923 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15924 fs::remove_file(path).expect("remove scratch file");
15925 }
15926
15927 #[test]
15928 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15929 let path = path("pair-frequencies");
15930 let mut pairs = Vec::new();
15931 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15932 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15933 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15934 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15935 let mut writer = Writer::create(
15936 &path,
15937 "items",
15938 vec![
15939 Field::required("id", LogicalType::BigInt),
15940 Field::required("phrase", LogicalType::Varchar),
15941 ],
15942 )
15943 .expect("new file");
15944 for part in pairs.chunks(1_024) {
15945 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15946 let phrases =
15947 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15948 writer
15949 .append(
15950 &Chunk::new(vec![
15951 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15952 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15953 ])
15954 .expect("matching columns"),
15955 )
15956 .expect("rows");
15957 }
15958 writer.finish().expect("commit");
15959
15960 let reader = Reader::open(&path).expect("reopen from disk");
15961 assert!(
15962 reader.table.pair_frequencies.is_empty(),
15963 "no query-specific pair result is stored"
15964 );
15965 fs::remove_file(path).expect("remove scratch file");
15966 }
15967
15968 #[test]
15969 fn legacy_group_answers_are_ignored() {
15970 let path = path("legacy-group-answers");
15971 let mut writer = Writer::create(
15972 &path,
15973 "items",
15974 vec![
15975 Field::required("id", LogicalType::BigInt),
15976 Field::required("text", LogicalType::Varchar),
15977 ],
15978 )
15979 .expect("new file");
15980 writer
15981 .append(
15982 &Chunk::new(vec![
15983 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15984 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15985 .expect("text"),
15986 ])
15987 .expect("row"),
15988 )
15989 .expect("append");
15990 writer.finish().expect("commit");
15991 let mut reader = Reader::open(&path).expect("reopen");
15992 let table = Arc::make_mut(&mut reader.table);
15993 table.pair_frequencies.push(PairFrequencySummary {
15994 first: 0,
15995 second: 1,
15996 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15997 omitted_max: 0,
15998 });
15999 table.host_groups = Some(host::HostSummary {
16000 column: 1,
16001 omitted_max: 0,
16002 entries: vec![host::HostEntry {
16003 host: "fake.test".into(),
16004 count: 999,
16005 bytes_sum: 999,
16006 minimum: "x".into(),
16007 }],
16008 });
16009 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
16010 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
16011 fs::remove_file(path).expect("remove scratch file");
16012 }
16013
16014 #[test]
16020 fn a_file_from_another_format_says_which_format_it_is() {
16021 let older = path("older-format");
16022 let mut writer =
16023 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
16024 .expect("new file");
16025 let chunk = Chunk::new(vec![
16026 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16027 .expect("integers"),
16028 ])
16029 .expect("chunk");
16030 writer.append(&chunk).expect("page written");
16031 writer.finish().expect("commit");
16032
16033 let unreadable =
16037 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
16038 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16039 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
16040 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
16041 drop(file);
16042 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
16043 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
16044 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
16045
16046 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
16047 file.seek(SeekFrom::Start(0)).expect("the magic is first");
16048 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
16049 drop(file);
16050 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
16051 assert!(complaint.contains("magic"), "{complaint}");
16052 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
16053 fs::remove_file(older).expect("remove scratch file");
16054 }
16055
16056 #[test]
16057 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
16058 let unfinished = path("unfinished");
16059 let mut writer =
16060 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
16061 .expect("new file");
16062 let chunk = Chunk::new(vec![
16063 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
16064 .expect("integers"),
16065 ])
16066 .expect("chunk");
16067 writer.append(&chunk).expect("page written");
16068 drop(writer);
16069 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
16070 fs::remove_file(unfinished).expect("remove scratch file");
16071
16072 let damaged = path("damaged");
16073 let mut writer =
16074 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
16075 .expect("new file");
16076 writer.append(&chunk).expect("page written");
16077 writer.finish().expect("commit");
16078 let reader = Reader::open(&damaged).expect("valid directory");
16079 let mut file =
16080 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
16081 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
16082 file.write_all(&[255]).expect("damage one byte");
16083 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
16084 fs::remove_file(damaged).expect("remove scratch file");
16085 }
16086
16087 #[test]
16088 fn damaged_lazy_dictionary_payload_is_an_error() {
16089 let path = path("damaged-dictionary");
16090 let mut writer = Writer::create(
16091 &path,
16092 "items",
16093 vec![
16094 Field::required("id", LogicalType::Integer),
16095 Field::new("text", LogicalType::Varchar),
16096 ],
16097 )
16098 .expect("new file");
16099 writer.append(&sample()).expect("stripe written");
16100 writer.finish().expect("commit");
16101
16102 let reader = Reader::open(&path).expect("valid directory");
16103 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
16104 let mut header = [0; DICTIONARY_HEADER];
16107 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16108 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16111 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16112 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
16113 let bits = (width & !DICTIONARY_FLAGS) as usize;
16114 let mut start = [0; 8];
16115 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
16116 read_at(&reader.file, at, &mut start).expect("the first block's start");
16117 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16118 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
16119 file.write_all(&[255]).expect("damage dictionary payload");
16120
16121 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
16122 let error =
16123 chunk.validate_external().expect_err("payload corruption must reach the caller");
16124 assert!(error.message().contains("payload checksum differs"), "{error}");
16125 fs::remove_file(path).expect("remove scratch file");
16126 }
16127
16128 #[test]
16138 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
16139 let path = path("dictionary-decide");
16140 let rows = 20_000;
16141 let unique =
16143 |row: usize| format!("{row:09} a value that appears exactly once in the table");
16144 let repeated = |row: usize| unique(row / 40);
16146 let mut writer = Writer::create(
16147 &path,
16148 "items",
16149 vec![
16150 Field::required("unique", LogicalType::Varchar),
16151 Field::required("repeated", LogicalType::Varchar),
16152 ],
16153 )
16154 .expect("new file");
16155 for part in (0..rows).step_by(1_000) {
16156 let span = part..(part + 1_000).min(rows);
16157 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
16158 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
16159 writer
16160 .append(
16161 &Chunk::new(vec![
16162 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
16163 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
16164 ])
16165 .expect("two columns"),
16166 )
16167 .expect("a part");
16168 }
16169 writer.finish().expect("commit");
16170
16171 let reader = Reader::open(&path).expect("reopen from disk");
16172 assert!(
16173 reader.table.dictionaries[0].is_none(),
16174 "a column with no repeats has nothing to say twice"
16175 );
16176 assert!(
16177 reader.table.dictionaries[1].is_some(),
16178 "a column whose values come round again keeps its dictionary"
16179 );
16180 let mut first = 0;
16181 for part in 0..reader.parts() {
16182 let chunk = reader.read(part, &[0, 1]).expect("a part");
16183 for row in 0..chunk.len() {
16184 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
16185 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
16186 }
16187 first += chunk.len();
16188 }
16189 assert_eq!(first, rows, "every row was read back");
16190 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
16191 let size = fs::metadata(&path).expect("the file is there").len() as usize;
16192 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
16193 fs::remove_file(path).expect("remove scratch file");
16194 }
16195
16196 #[test]
16209 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
16210 let path = path("dictionary-blocks");
16211 let value = |row: usize| {
16212 let row = row.saturating_sub(8_000);
16213 format!("{row:07} a value long enough to be worth a payload block")
16214 };
16215 let parts = 40;
16216 let per_part = 1000;
16217 let mut writer =
16218 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16219 .expect("new file");
16220 for part in 0..parts {
16221 let values = (0..per_part)
16222 .map(|row| Value::Varchar(value(part * per_part + row)))
16223 .collect::<Vec<_>>();
16224 let chunk = Chunk::new(vec![
16225 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16226 ])
16227 .expect("matching rows");
16228 writer.append(&chunk).expect("a part");
16229 }
16230 writer.finish().expect("commit");
16231
16232 let reader = Reader::open(&path).expect("reopen from disk");
16233 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
16234 assert!(
16235 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
16236 "the dictionary has to be several blocks for this to be testing anything"
16237 );
16238 for part in [0, parts - 1] {
16239 let chunk = reader.read(part, &[0]).expect("a part");
16240 chunk.validate_external().expect("every payload block checks out");
16241 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
16242 }
16243
16244 let mut header = [0; DICTIONARY_HEADER];
16246 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
16247 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
16248 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
16249 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
16250 let bits = (width & !DICTIONARY_FLAGS) as usize;
16251 let mut place = [0; 16];
16252 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
16253 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
16254 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
16255 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
16256 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16257 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
16258 file.write_all(&[255]).expect("damage the last payload block");
16259 let reader = Reader::open(&path).expect("the directory and the index are untouched");
16260 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
16261 let error = chunk.validate_external().expect_err("the damage must reach the caller");
16262 assert!(error.message().contains("payload checksum differs"), "{error}");
16263 fs::remove_file(path).expect("remove scratch file");
16264 }
16265
16266 #[test]
16280 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
16281 let path = path("dictionary-offsets");
16282 let value = |row: usize| {
16283 let row = row % 5_000;
16284 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
16285 };
16286 let rows = 6_000;
16287 let mut writer =
16288 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16289 .expect("new file");
16290 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
16291 for part in values.chunks(1_000) {
16292 let chunk =
16293 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
16294 .expect("matching rows");
16295 writer.append(&chunk).expect("a part");
16296 }
16297 writer.finish().expect("commit");
16298
16299 let reader = Reader::open(&path).expect("reopen from disk");
16300 assert!(
16301 rows > TEXT_PAYLOAD_VALUES * 4,
16302 "the dictionary has to be several blocks for this to be testing anything"
16303 );
16304 for part in 0..rows / 1_000 {
16305 let chunk = reader.read(part, &[0]).expect("a part");
16306 for row in 0..1_000 {
16307 let row = part * 1_000 + row;
16308 assert_eq!(
16309 chunk.value_at(row % 1_000, 0),
16310 Value::Varchar(value(row)),
16311 "value {row}"
16312 );
16313 }
16314 }
16315 for _ in 0..2 {
16318 for part in 0..rows / 1_000 {
16319 let chunk = reader.read(part, &[0]).expect("a part");
16320 let mut lens = vec![0_i64; 1_000];
16321 let column = chunk.column(0).expect("one column");
16322 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
16323 for (row, &len) in lens.iter().enumerate() {
16324 let row = part * 1_000 + row;
16325 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
16326 }
16327 }
16328 }
16329 fs::remove_file(path).expect("remove scratch file");
16330 }
16331
16332 #[test]
16334 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
16335 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
16336 ends.extend([3, 3, 10]);
16337 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
16338 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
16339 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
16340 let long = [5, 70_005, 70_006];
16342 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
16343 assert_eq!(lens, [5, 70_000, 1]);
16344 let mut read = Vec::new();
16345 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
16346 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
16347 ends.push(9);
16348 assert!(lengths_of(&ends).is_none());
16349 }
16350
16351 #[test]
16363 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
16364 let path = path("dictionary-once");
16365 let parts = 8;
16366 let per_part = 500;
16367 let value =
16368 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
16369 let mut writer =
16370 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16371 .expect("new file");
16372 for part in 0..parts {
16373 let values = (0..per_part)
16374 .map(|row| Value::Varchar(value(part * per_part + row)))
16375 .collect::<Vec<_>>();
16376 let chunk = Chunk::new(vec![
16377 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16378 ])
16379 .expect("matching rows");
16380 writer.append(&chunk).expect("a part");
16381 }
16382 writer.finish().expect("commit");
16383
16384 let reader = Reader::open(&path).expect("reopen from disk");
16385 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
16386 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
16387
16388 let workers = 16;
16389 let gate = std::sync::Barrier::new(workers);
16390 std::thread::scope(|scope| {
16391 for worker in 0..workers {
16392 let reader = reader.clone();
16393 let gate = &gate;
16394 scope.spawn(move || {
16395 gate.wait();
16396 let chunk = reader.read(worker % parts, &[0]).expect("a part");
16397 assert_eq!(
16398 chunk.value_at(0, 0),
16399 Value::Varchar(value((worker % parts) * per_part))
16400 );
16401 });
16402 }
16403 });
16404
16405 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
16406 fs::remove_file(path).expect("remove scratch file");
16407 }
16408
16409 #[test]
16414 fn a_damaged_sorted_order_is_an_error() {
16415 let path = path("damaged-order");
16416 let mut writer = Writer::create(
16417 &path,
16418 "items",
16419 vec![
16420 Field::required("id", LogicalType::Integer),
16421 Field::new("text", LogicalType::Varchar),
16422 ],
16423 )
16424 .expect("new file");
16425 writer.append(&sample()).expect("stripe written");
16426 writer.finish().expect("commit");
16427
16428 let reader = Reader::open(&path).expect("valid directory");
16429 let page = reader.table.dictionaries[1].expect("string dictionary page");
16430 let mut header = [0; DICTIONARY_HEADER];
16431 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
16432 let index_len = dictionary_index_len(&header);
16433 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16434 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
16435 file.write_all(&[255]).expect("damage the order");
16436
16437 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
16438 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
16439 assert!(error.message().contains("rank checksum differs"), "{error}");
16440 fs::remove_file(path).expect("remove scratch file");
16441 }
16442
16443 #[test]
16447 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
16448 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
16451 let path = path("dictionary-order");
16452 let mut writer =
16453 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16454 .expect("new file");
16455 writer
16456 .append(
16457 &Chunk::new(vec![
16458 Vector::from_values(
16459 LogicalType::Varchar,
16460 &spellings.map(|text| Value::Varchar(text.into())),
16461 )
16462 .expect("strings"),
16463 ])
16464 .expect("one column"),
16465 )
16466 .expect("stripe written");
16467 writer.finish().expect("commit");
16468
16469 let reader = Reader::open(&path).expect("valid directory");
16470 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16471 let count = dictionary.ranks().expect("a v10 file stores one");
16472 assert_eq!(count, spellings.len(), "every distinct value has a rank");
16473 let order = (0..count)
16474 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
16475 .collect::<Vec<_>>();
16476 let mut seen = order.clone();
16477 seen.sort_unstable();
16478 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
16479
16480 let ranked = order
16481 .iter()
16482 .map(|&code| {
16483 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16484 })
16485 .collect::<Vec<_>>();
16486 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
16487 expected.sort();
16488 assert_eq!(ranked, expected, "rank order is value order");
16489
16490 for (rank, value) in expected.iter().enumerate() {
16493 assert_eq!(
16494 dictionary.compare_rank(rank, value).expect("compare"),
16495 Ordering::Equal,
16496 "rank {rank} is its own value"
16497 );
16498 if rank > 0 {
16499 assert_eq!(
16500 dictionary.compare_rank(rank - 1, value).expect("compare"),
16501 Ordering::Less,
16502 "rank {rank} follows the one before it"
16503 );
16504 }
16505 }
16506 fs::remove_file(path).expect("remove scratch file");
16507 }
16508
16509 #[test]
16516 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
16517 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
16518 let path = path("dictionaries-at-once");
16519 let fields = (0..sizes.len())
16520 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
16521 .collect::<Vec<_>>();
16522 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16523 let rows = 10_000_usize;
16524 for start in (0..rows).step_by(1_024) {
16525 let columns = sizes
16526 .iter()
16527 .enumerate()
16528 .map(|(column, &size)| {
16529 let values = (start..(start + 1_024).min(rows))
16530 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
16531 .collect::<Vec<_>>();
16532 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
16533 })
16534 .collect::<Vec<_>>();
16535 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
16536 }
16537 writer.finish().expect("commit");
16538
16539 let reader = Reader::open(&path).expect("valid directory");
16540 for (column, &size) in sizes.iter().enumerate() {
16541 let dictionary =
16542 reader.dictionary(column).expect("read").expect("a string column has one");
16543 let count = dictionary.ranks().expect("a v10 file stores one");
16544 assert_eq!(count, size, "column {column} has its own distinct count");
16545 let ranked = (0..count)
16546 .map(|rank| {
16547 let code = dictionary.code_at_rank(rank).expect("a code");
16548 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16549 })
16550 .collect::<Vec<_>>();
16551 let expected = (0..size)
16552 .map(|value| format!("c{column}-{value:05}").into_bytes())
16553 .collect::<Vec<_>>();
16554 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
16555 }
16556 fs::remove_file(path).expect("remove scratch file");
16557 }
16558
16559 #[test]
16567 fn a_large_dictionary_ranks_in_value_order() {
16568 let path = path("dictionary-large-rank");
16569 let value = |row: u64| {
16570 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
16571 match row % 3 {
16572 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
16573 1 => format!("{mixed}"),
16574 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
16575 }
16576 };
16577 let distinct = 70_000;
16578 let parts = 4 * distinct / 1000;
16579 let per_part = 1000;
16580 let mut writer =
16581 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16582 .expect("new file");
16583 for part in 0..parts {
16584 let values = (0..per_part)
16585 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
16586 .collect::<Vec<_>>();
16587 let chunk = Chunk::new(vec![
16588 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
16589 ])
16590 .expect("matching rows");
16591 writer.append(&chunk).expect("a part");
16592 }
16593 writer.finish().expect("commit");
16594
16595 let reader = Reader::open(&path).expect("reopen from disk");
16596 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16597 let count = dictionary.ranks().expect("a ranked dictionary");
16598 assert_eq!(count, distinct as usize, "every distinct value has a rank");
16599 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
16600 let ranked = (0..count)
16601 .map(|rank| {
16602 let code = dictionary.code_at_rank(rank).expect("a code");
16603 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
16604 })
16605 .collect::<Vec<_>>();
16606 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
16607 expected.sort();
16608 assert_eq!(ranked, expected, "rank order is value order");
16609 fs::remove_file(path).expect("remove scratch file");
16610 }
16611
16612 #[test]
16625 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
16626 let path = path("windowed-directory");
16627 let fields = vec![
16628 Field::required("id", LogicalType::BigInt),
16629 Field::required("word", LogicalType::Varchar),
16630 Field::new("score", LogicalType::Double),
16631 ];
16632 let mut writer = Writer::create(&path, "items", fields).expect("new file");
16633 for part in 0..70_i64 {
16634 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
16635 let words = (0..100)
16636 .map(|row| Value::Varchar(format!("word {}", row % 13)))
16637 .collect::<Vec<_>>();
16638 let scores = (0..100)
16639 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
16640 .collect::<Vec<_>>();
16641 let chunk = Chunk::new(vec![
16642 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
16643 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
16644 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
16645 ])
16646 .expect("three columns");
16647 writer.append(&chunk).expect("a part");
16648 }
16649 writer.finish().expect("commit");
16650
16651 let catalog = Catalog::open(&path).expect("reopen");
16652 let entry = catalog.entries.first().expect("one table").directory;
16653 let (offset, length) = (entry.offset, entry.length as usize);
16654 let mut bytes = vec![0; length];
16655 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
16656 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
16657 let whole = decode_directory(&bytes, catalog.size).expect("whole");
16658 assert!(whole.stripes.len() > 1, "the table should span stripes");
16659 for size in [1, 7, 33, 4_096] {
16660 let mut cursor = Cursor::over(&catalog.file, offset, length);
16661 cursor.window.as_mut().expect("a window").size = size;
16662 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
16663 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
16664 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
16665 let mut stored = 0;
16666 for (column, (left, held)) in
16667 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
16668 {
16669 match (left, held) {
16670 (None, None) => {}
16671 (
16672 Some(super::Frequencies::Stored { span, values, entries }),
16673 Some(super::Frequencies::Held(summary)),
16674 ) => {
16675 let mut one = vec![0; span.length as usize];
16676 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
16677 let read = decode_summary(
16678 &mut Cursor::new(&one),
16679 &whole.fields[column],
16680 whole.rows,
16681 *values,
16682 )
16683 .expect("a valid synopsis")
16684 .expect("one is there");
16685 assert_eq!(*entries, read.entries.len());
16686 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
16687 stored += 1;
16688 }
16689 other => panic!("column {column} came back as {other:?}"),
16690 }
16691 }
16692 assert!(stored >= 2, "only {stored} synopses were left in the file");
16693 }
16694 let reader = catalog.table("items").expect("the table");
16695 assert!(reader.frequency_summaries[1].get().is_none());
16696 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
16697 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
16698 let clone = reader.clone();
16699 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
16700 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
16701 fs::remove_file(path).expect("remove scratch file");
16702 }
16703
16704 #[test]
16705 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
16706 let path = path("file-checksum");
16707 let bytes = (0..200_000_u32)
16708 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
16709 .collect::<Vec<_>>();
16710 fs::write(&path, &bytes).expect("scratch file");
16711 let file = File::open(&path).expect("open");
16712 for (offset, length) in [
16713 (0, 0),
16714 (3, 1),
16715 (5, 31),
16716 (0, 32),
16717 (9, 33),
16718 (1, 65_536),
16719 (7, 65_567),
16720 (0, 200_000),
16721 (11, 131_101),
16722 ] {
16723 let whole = checksum(&bytes[offset..offset + length]);
16724 assert_eq!(
16725 file_checksum(&file, offset as u64, length).expect("read"),
16726 whole,
16727 "{offset} {length}"
16728 );
16729 }
16730 fs::remove_file(path).expect("remove scratch file");
16731 }
16732
16733 #[test]
16734 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
16735 let path = path("synopsis-keeps-no-block");
16736 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
16737 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
16738 for _ in 0..3 {
16739 values.extend((0..3_000).step_by(5).map(spelled));
16740 }
16741 let mut writer =
16742 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16743 .expect("new file");
16744 for part in values.chunks(1_024) {
16745 writer
16746 .append(
16747 &Chunk::new(vec![
16748 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16749 ])
16750 .expect("one column"),
16751 )
16752 .expect("a part");
16753 }
16754 writer.finish().expect("commit");
16755
16756 let reader = Reader::open(&path).expect("reopen from disk");
16757 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16758 let resting = dictionary.footprint();
16759 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16760 assert_eq!(prefix.entries.len(), 512);
16761 for (value, count) in &prefix.entries {
16762 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
16763 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
16764 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
16765 }
16766 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
16767 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16768 assert_eq!(again.entries, prefix.entries);
16769 fs::remove_file(path).expect("remove scratch file");
16770 }
16771
16772 #[test]
16779 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
16780 let path = path("character-lengths");
16781 let spellings = (0..2_500)
16782 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
16783 .collect::<Vec<_>>();
16784 let mut writer =
16785 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16786 .expect("new file");
16787 for part in spellings.chunks(1_024) {
16788 writer
16789 .append(
16790 &Chunk::new(vec![
16791 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16792 ])
16793 .expect("one column"),
16794 )
16795 .expect("a part");
16796 }
16797 writer.finish().expect("commit");
16798
16799 let reader = Reader::open(&path).expect("reopen from disk");
16800 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16801 let resting = dictionary.footprint();
16802 let mut lens = Vec::new();
16803 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
16804 let counted = dictionary.footprint() - resting;
16805 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16806 assert!(
16807 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16808 "counting kept {counted} bytes, more than a count a value"
16809 );
16810 let expected = (0..dictionary.len())
16811 .map(|code| {
16812 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
16813 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
16814 .expect("small")
16815 })
16816 .collect::<Vec<_>>();
16817 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
16818 let mut again = Vec::new();
16819 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
16820 assert_eq!(again, lens, "the kept counts answer the second time");
16821 fs::remove_file(path).expect("remove scratch file");
16822 }
16823
16824 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
16826 let path = path(label);
16827 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
16828 let mut writer =
16829 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16830 .expect("new file");
16831 for part in values.chunks(1_024) {
16832 writer
16833 .append(
16834 &Chunk::new(vec![
16835 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16836 ])
16837 .expect("one column"),
16838 )
16839 .expect("a part");
16840 }
16841 writer.finish().expect("commit");
16842 let reader = Reader::open(&path).expect("reopen from disk");
16843 (path, reader)
16844 }
16845
16846 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
16852 let codes = (0..len)
16853 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
16854 .collect::<Vec<_>>();
16855 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
16856 (codes, valid)
16857 }
16858
16859 #[test]
16866 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
16867 let spellings = (0..2_500)
16868 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
16869 .collect::<Vec<_>>();
16870 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
16871 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16872 let (codes, valid) = scattered_rows(spellings.len());
16873 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
16874 .expect("every code is inside")
16875 .with_validity(Validity::from_run(&valid));
16876
16877 let resting = dictionary.footprint();
16878 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
16879 .expect("length reads");
16880 let counted = dictionary.footprint() - resting;
16881 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16882 assert!(
16883 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16884 "length over a vector with nulls kept {counted} bytes, more than a count a value"
16885 );
16886 let expected = (0..rows.len())
16887 .map(|row| match valid[row] {
16888 true => Value::BigInt(
16889 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16890 ),
16891 false => Value::Null,
16892 })
16893 .collect::<Vec<_>>();
16894 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16895 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16896 fs::remove_file(path).expect("remove scratch file");
16897 }
16898
16899 #[test]
16909 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16910 let spellings = (0..2_500)
16911 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16912 .collect::<Vec<_>>();
16913 let (path, reader) = stored_spellings("string-kernels", &spellings);
16914 let page = reader.table.dictionaries[0].expect("a string column has one");
16915 let starved =
16916 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16917 .expect("a dictionary opens whatever it may keep");
16918 let starved = Arc::new(starved);
16919 let (codes, valid) = scattered_rows(spellings.len());
16920 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16921 .expect("every code is inside")
16922 .with_validity(Validity::from_run(&valid));
16923 let expected = |each: &dyn Fn(&str) -> String| {
16924 (0..rows.len())
16925 .map(|row| match valid[row] {
16926 true => Value::Varchar(each(&spellings[codes[row] as usize])),
16927 false => Value::Null,
16928 })
16929 .collect::<Vec<_>>()
16930 };
16931 let answers =
16932 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16933
16934 let resting = starved.footprint();
16937 let ends = spellings.len() * size_of::<u32>();
16938 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16939 .expect("lower reads");
16940 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16941 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16942
16943 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16944 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16945 let cut =
16946 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16947 .expect("substring reads");
16948 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16949 assert_eq!(answers(&cut), expected(&cut_of), "substring");
16950 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16951
16952 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16955 .expect("upper reads");
16956 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16957 let payload = spellings.iter().map(String::len).sum::<usize>();
16958 assert!(
16959 starved.footprint() >= resting + payload,
16960 "a visit that has dropped a column's worth of blocks keeps what it reads"
16961 );
16962 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16963 .expect("upper reads kept blocks");
16964 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16965 fs::remove_file(path).expect("remove scratch file");
16966 }
16967
16968 #[test]
16978 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16979 let path = path("dictionary-sweep");
16980 let spellings = (0..2_500)
16983 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16984 .collect::<Vec<_>>();
16985 let mut writer =
16986 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16987 .expect("new file");
16988 for part in spellings.chunks(1_024) {
16991 writer
16992 .append(
16993 &Chunk::new(vec![
16994 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16995 ])
16996 .expect("one column"),
16997 )
16998 .expect("stripe written");
16999 }
17000 writer.finish().expect("commit");
17001
17002 let reader = Reader::open(&path).expect("valid directory");
17003 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17004 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17005 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
17006 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
17007 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
17008 }
17009
17010 let resting = dictionary.footprint();
17011 let sweep = || {
17012 let mut swept: Vec<Vec<u8>> = Vec::new();
17013 let mut at = 0;
17014 let mut calls = 0;
17015 while at < dictionary.len() {
17016 let stopped = dictionary
17017 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17018 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17019 swept.push(text.to_vec());
17020 Ok(())
17021 })
17022 .expect("a sweep reads");
17023 assert!(stopped > at, "a sweep moves");
17024 at = stopped;
17025 calls += 1;
17026 }
17027 assert_eq!(calls, 3, "a sweep hands over one block at a time");
17028 swept
17029 };
17030 let swept = sweep();
17031 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
17032 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
17033 let after = dictionary.footprint();
17034 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
17035
17036 let read = (0..dictionary.len())
17037 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17038 .collect::<Vec<_>>();
17039 assert_eq!(swept, read, "a sweep answers what a point read answers");
17040 let grown = dictionary.footprint() - after;
17044 assert!(
17045 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
17046 "a point read of a kept block decodes nothing, and {grown} bytes grew"
17047 );
17048 fs::remove_file(path).expect("remove scratch file");
17049 }
17050
17051 #[test]
17052 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
17053 let path = path("narrow-substring-signature");
17054 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
17055 let mut grams = Vec::new();
17056 for text in blocks {
17057 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
17058 for gram in text.windows(4) {
17059 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
17060 bits[bit / 8] |= 1 << (bit % 8);
17061 }
17062 }
17063 grams.extend(bits);
17064 }
17065 fs::write(&path, &grams).expect("scratch file");
17066 let file = File::open(&path).expect("open scratch file");
17067 let signatures = NativeGrams {
17068 start: 0,
17069 length: grams.len(),
17070 width: NARROW_GRAM_BYTES,
17071 hash: checksum(&grams),
17072 verdicts: Mutex::new(Vec::new()),
17073 };
17074 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
17075 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
17076 assert!(signatures.footprint() > 0, "a verdict is remembered");
17077 let again = signatures.verdicts(&file, b"google").expect("remembered");
17078 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
17079
17080 let damaged = NativeGrams {
17081 hash: signatures.hash ^ 1,
17082 verdicts: Mutex::new(Vec::new()),
17083 ..signatures
17084 };
17085 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
17086 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
17087 fs::remove_file(path).expect("remove scratch file");
17088 }
17089
17090 #[test]
17091 fn a_damaged_substring_signature_is_checked_only_when_used() {
17092 let path = path("damaged-substring-signature");
17093 let mut writer =
17094 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17095 .expect("new file");
17096 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
17097 writer
17098 .append(
17099 &Chunk::new(vec![
17100 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
17101 ])
17102 .expect("one column"),
17103 )
17104 .expect("stripe written");
17105 writer.finish().expect("commit");
17106
17107 let reader = Reader::open(&path).expect("valid directory");
17108 let page = reader.table.dictionaries[0].expect("string dictionary page");
17109 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
17110 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
17111 .expect("last signature byte");
17112 file.write_all(&[255]).expect("damage signature");
17113 let reader = Reader::open(&path).expect("the directory is still valid");
17114 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
17115 let error = dictionary
17116 .text_block_might_contain(0, b"goog")
17117 .expect_err("a used signature checks its own checksum");
17118 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
17119 fs::remove_file(path).expect("remove scratch file");
17120 }
17121
17122 #[test]
17133 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
17134 let path = path("dictionary-sweep-short-run");
17135 let spellings = (0..2_800)
17136 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17137 .collect::<Vec<_>>();
17138 let mut writer =
17139 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17140 .expect("new file");
17141 for part in spellings.chunks(1_024) {
17142 writer
17143 .append(
17144 &Chunk::new(vec![
17145 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17146 ])
17147 .expect("one column"),
17148 )
17149 .expect("stripe written");
17150 }
17151 writer.finish().expect("commit");
17152
17153 let reader = Reader::open(&path).expect("valid directory");
17154 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17155 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17156 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
17157 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
17158 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
17159
17160 let mut swept: Vec<Vec<u8>> = Vec::new();
17161 let mut at = 0;
17162 while at < dictionary.len() {
17163 let stopped = dictionary
17164 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
17165 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
17166 swept.push(text.to_vec());
17167 Ok(())
17168 })
17169 .expect("a sweep reads");
17170 assert!(stopped > at, "a sweep moves");
17171 at = stopped;
17172 }
17173 let read = (0..dictionary.len())
17174 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
17175 .collect::<Vec<_>>();
17176 assert_eq!(swept, read, "a sweep answers what a point read answers");
17177 fs::remove_file(path).expect("remove scratch file");
17178 }
17179
17180 #[test]
17189 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
17190 let path = path("dictionary-unpacked-ends");
17191 let spellings = (0..2_800)
17192 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
17193 .collect::<Vec<_>>();
17194 let mut writer =
17195 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17196 .expect("new file");
17197 for part in spellings.chunks(1_024) {
17198 writer
17199 .append(
17200 &Chunk::new(vec![
17201 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17202 ])
17203 .expect("one column"),
17204 )
17205 .expect("stripe written");
17206 }
17207 writer.finish().expect("commit");
17208
17209 let reader = Reader::open(&path).expect("valid directory");
17210 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
17211 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
17212 let wanted = (0..spellings.len())
17213 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
17214 .collect::<Vec<_>>();
17215
17216 let pass = |what: &str| {
17217 for (index, value) in wanted.iter().enumerate() {
17218 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
17219 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
17220 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
17221 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
17222 }
17223 };
17224 pass("the first pass");
17225 pass("the second pass");
17226
17227 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
17231 let mut whole = vec![0i64; wanted.len()];
17232 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
17233 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
17234 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
17235 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
17236 let mut through = vec![0i64; codes.len()];
17237 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
17238 for (row, &code) in codes.iter().enumerate() {
17239 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
17240 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
17241 assert_eq!(through[row], one as i64, "row {row} a row at a time");
17242 }
17243
17244 let fresh = Reader::open(&path).expect("valid directory");
17247 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
17248 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
17249 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
17250 let mut short = vec![0i64; few.len()];
17251 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
17252 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
17253 assert_eq!(short, expected, "the packed ends answer what the table answers");
17254 fs::remove_file(path).expect("remove scratch file");
17255 }
17256
17257 #[test]
17272 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
17273 let spellings = (0..3_000)
17274 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
17275 .collect::<Vec<_>>();
17276 let mut read = Vec::new();
17277 for layout in ["outside", "inside", "behind"] {
17278 let mut dictionary = GlobalDictionary::new();
17279 for text in &spellings {
17280 dictionary.code(text).expect("a code for every spelling");
17281 }
17282 dictionary.finish_blocks().expect("the last block encodes");
17283 let order = dictionary.ranked(None).expect("a sorted order");
17284 let laid = |from: u64| {
17286 let mut at = from;
17287 dictionary
17288 .blocks
17289 .iter()
17290 .map(|block| {
17291 let place =
17292 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
17293 at += block.len() as u64;
17294 place
17295 })
17296 .collect::<Vec<_>>()
17297 };
17298 let payload = dictionary.blocks.concat();
17299 let scattered = layout != "behind";
17300 let (bytes, encoded, offset, length) = if layout == "outside" {
17301 let mut bytes = vec![0; HEADER as usize];
17302 bytes.extend_from_slice(&payload);
17303 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
17304 .expect("an encoding");
17305 let offset = bytes.len() as u64;
17306 bytes.extend_from_slice(&encoded.index);
17307 bytes.extend_from_slice(&encoded.ranks);
17308 bytes.extend_from_slice(&encoded.grams);
17309 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
17310 (bytes, encoded, offset, length)
17311 } else {
17312 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
17315 .expect("an encoding");
17316 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
17317 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
17318 .expect("an encoding");
17319 let mut bytes = encoded.index.clone();
17320 bytes.extend_from_slice(&encoded.ranks);
17321 bytes.extend_from_slice(&encoded.grams);
17322 bytes.extend_from_slice(&payload);
17323 let length = bytes.len();
17324 (bytes, encoded, 0, length)
17325 };
17326 let path = path(&format!("blocks-{layout}"));
17327 fs::write(&path, &bytes).expect("the dictionary is written on its own");
17328 let file = Arc::new(File::open(&path).expect("it opens again"));
17329 let page = Page {
17330 offset,
17331 length: u32::try_from(length).expect("a test dictionary is small"),
17332 hash: checksum(&encoded.index),
17333 };
17334 let opened =
17335 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
17336 .expect("a dictionary laid out either way opens");
17337 let mut swept: Vec<Vec<u8>> = Vec::new();
17338 let mut at = 0;
17339 while at < opened.len() {
17340 at = opened
17341 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
17342 swept.push(text.to_vec());
17343 Ok(())
17344 })
17345 .expect("a sweep reads");
17346 }
17347 fs::remove_file(&path).expect("clean up");
17348 read.push(swept);
17349 }
17350 let wanted =
17351 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
17352 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
17353 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
17354 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
17355 }
17356
17357 #[test]
17365 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
17366 let path = path("dictionary-budget");
17367 let spellings = (0..2_500)
17368 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
17369 .collect::<Vec<_>>();
17370 let mut writer =
17371 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17372 .expect("new file");
17373 for part in spellings.chunks(1_024) {
17374 writer
17375 .append(
17376 &Chunk::new(vec![
17377 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
17378 ])
17379 .expect("one column"),
17380 )
17381 .expect("stripe written");
17382 }
17383 writer.finish().expect("commit");
17384
17385 let reader = Reader::open(&path).expect("valid directory");
17386 let page = reader.table.dictionaries[0].expect("a string column has one");
17387 let file = Arc::clone(&reader.file);
17388 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
17389 .expect("a dictionary opens whatever it may keep");
17390
17391 let resting = starved.footprint();
17392 let mut swept: Vec<Vec<u8>> = Vec::new();
17393 let mut at = 0;
17394 while at < starved.len() {
17395 at = starved
17396 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
17397 swept.push(text.to_vec());
17398 Ok(())
17399 })
17400 .expect("a sweep reads");
17401 }
17402 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
17403 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
17404
17405 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
17406 let read = (0..generous.len())
17407 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
17408 .collect::<Vec<_>>();
17409 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
17410 fs::remove_file(path).expect("remove scratch file");
17411 }
17412
17413 #[test]
17416 fn a_part_is_checked_once_per_open_reader() {
17417 let path = path("checked-once");
17418 let mut writer = Writer::create(
17419 &path,
17420 "items",
17421 vec![
17422 Field::required("id", LogicalType::Integer),
17423 Field::new("text", LogicalType::Varchar),
17424 ],
17425 )
17426 .expect("new file");
17427 writer.append(&sample()).expect("stripe written");
17428 writer.finish().expect("commit");
17429
17430 let reader = Reader::open(&path).expect("valid directory");
17431 let first = reader.read_rows(0, &[0], &[0, 1], false).expect("checked and read");
17432 assert!(reader.is_verified(0), "the part is remembered as checked");
17433 let page = reader.table.stripes[0].pages[0];
17434 let mut file = OpenOptions::new().write(true).open(&path).expect("open column page");
17435 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("page end");
17436 file.write_all(&[0xa5]).expect("damage page");
17437 if let Err(error) = reader.read_rows(0, &[0], &[0, 1], false) {
17438 assert!(!error.message().contains("checksum differs"), "not hashed again: {error}");
17439 }
17440 let fresh = Reader::open(&path).expect("valid directory");
17441 let error = fresh.read_rows(0, &[0], &[0, 1], false).expect_err("a new reader checks");
17442 assert!(error.message().contains("column page checksum differs"), "{error}");
17443 assert_eq!(first.len(), 2);
17444 fs::remove_file(path).expect("remove scratch file");
17445 }
17446
17447 #[test]
17448 fn damaged_membership_cannot_skip_a_string_page() {
17449 let path = path("damaged-membership");
17450 let mut writer = Writer::create(
17451 &path,
17452 "items",
17453 vec![
17454 Field::required("id", LogicalType::Integer),
17455 Field::new("text", LogicalType::Varchar),
17456 ],
17457 )
17458 .expect("new file");
17459 writer.append(&sample()).expect("stripe written");
17460 writer.finish().expect("commit");
17461
17462 let reader = Reader::open(&path).expect("valid directory");
17463 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
17464 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
17465 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
17466 file.write_all(&[255]).expect("damage membership");
17467 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
17468 assert!(error.message().contains("membership page checksum differs"), "{error}");
17469 fs::remove_file(path).expect("remove scratch file");
17470 }
17471
17472 #[test]
17473 fn membership_delta_stream_is_sorted_exact_and_bounded() {
17474 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
17475 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
17476 let encoded = encode_membership(&unique);
17477 assert_eq!(
17478 decode_membership(&encoded).expect("valid membership"),
17479 [4, 9, 72, 900, u32::MAX]
17480 );
17481 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
17484 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
17485 assert_eq!(
17486 decode_membership(&encode_membership(&merged)).expect("valid membership"),
17487 unique
17488 );
17489 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
17490 assert!(
17491 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
17492 "a value past u32 is invalid"
17493 );
17494 }
17495
17496 #[test]
17497 fn a_global_dictionary_may_be_larger_than_one_column_page() {
17498 let dictionary = Page {
17499 offset: HEADER,
17500 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
17501 hash: 0,
17502 };
17503 let table = Table {
17504 name: "items".to_owned(),
17505 fields: vec![Field::new("text", LogicalType::Varchar)],
17506 stripes: Vec::new(),
17507 rows: 0,
17508 dictionaries: vec![Some(dictionary)],
17509 dictionary_payloads: Vec::new(),
17510 demoted: Vec::new(),
17511 distincts: vec![None],
17512 frequencies: vec![None],
17513 pair_frequencies: Vec::new(),
17514 frequency_texts: Vec::new(),
17515 host_groups: None,
17516 clustering: None,
17517 constraints: Constraints::default(),
17518 generation: 1,
17519 sections: Vec::new(),
17520 };
17521 let directory = encode_directory(&table).expect("directory");
17522 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
17523
17524 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
17525 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
17526 }
17527
17528 #[test]
17529 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
17530 let path = path("constant-codes");
17531 let mut writer =
17532 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
17533 .expect("new file");
17534 let empty = vec![Value::Varchar(String::new()); 1024];
17535 for _ in 0..4 {
17536 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
17537 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
17538 }
17539 writer.finish().expect("commit");
17540
17541 let reader = Reader::open(&path).expect("valid directory");
17542 let pages = reader.layout().columns.first().expect("one column").pages;
17543 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
17547 let read = reader.read(3, &[0]).expect("the last part back");
17548 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
17549 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
17550 fs::remove_file(path).expect("remove scratch file");
17551 }
17552
17553 #[test]
17554 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
17555 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
17558 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
17559 assert!(format!("{error}").contains("not of its type"), "{error}");
17560 let low = integer::encode(&[i64::MIN]).expect("a chunk");
17561 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
17562 let zero = integer::encode(&[0]).expect("a chunk");
17563 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
17564 }
17565
17566 #[test]
17567 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
17568 let mut state: u32 = 0x9e37_79b9;
17572 let spread: Vec<u32> = (0..1024)
17573 .map(|_| {
17574 state ^= state << 13;
17575 state ^= state >> 17;
17576 state ^= state << 5;
17577 state
17578 })
17579 .collect();
17580 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
17581 let near: Vec<u32> = (0..1024).collect();
17582 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
17583 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
17584 }
17585
17586 #[test]
17592 fn two_writes_of_the_same_rows_give_the_same_bytes() {
17593 fn written(path: &PathBuf) {
17594 let fields = (0..40)
17595 .map(|column| {
17596 let ty =
17597 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
17598 Field::new(format!("c{column}"), ty)
17599 })
17600 .collect::<Vec<_>>();
17601 let mut writer = Writer::create(path, "wide", fields).expect("new file");
17602 for part in 0..70_u64 {
17603 let columns = (0..40)
17604 .map(|column| {
17605 let values = (0..64_u64)
17606 .map(|row| {
17607 let seed = part.wrapping_mul(31).wrapping_add(row);
17608 if column % 4 == 0 {
17609 Value::Varchar(format!("v{}", seed % 17))
17610 } else {
17611 Value::BigInt(i64::try_from(seed % 97).expect("small"))
17612 }
17613 })
17614 .collect::<Vec<_>>();
17615 let ty = if column % 4 == 0 {
17616 LogicalType::Varchar
17617 } else {
17618 LogicalType::BigInt
17619 };
17620 Vector::from_values(ty, &values).expect("a column")
17621 })
17622 .collect::<Vec<_>>();
17623 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
17624 }
17625 writer.finish().expect("commit");
17626 }
17627
17628 let first = path("repeatable-one");
17629 let second = path("repeatable-two");
17630 written(&first);
17631 written(&second);
17632 let left = fs::read(&first).expect("the first file");
17633 let right = fs::read(&second).expect("the second file");
17634 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
17635 assert!(left == right, "two writes of the same rows differ in their bytes");
17636
17637 let reader = Reader::open(&first).expect("valid directory");
17640 assert_eq!(reader.table().rows(), 70 * 64);
17641 let read = reader.read(0, &[0, 1]).expect("the first part back");
17642 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
17643 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
17644 fs::remove_file(first).expect("remove scratch file");
17645 fs::remove_file(second).expect("remove scratch file");
17646 }
17647
17648 fn three_tables(path: &PathBuf) {
17650 let writer = Writer::create(
17651 path,
17652 "region",
17653 vec![
17654 Field::new("r_key", LogicalType::Integer),
17655 Field::new("r_name", LogicalType::Varchar),
17656 ],
17657 )
17658 .expect("new file");
17659 let mut writer = writer;
17660 writer
17661 .append(
17662 &Chunk::new(vec![
17663 Vector::from_values(
17664 LogicalType::Integer,
17665 &[Value::Integer(0), Value::Integer(1)],
17666 )
17667 .expect("keys"),
17668 Vector::from_values(
17669 LogicalType::Varchar,
17670 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
17671 )
17672 .expect("names"),
17673 ])
17674 .expect("two columns"),
17675 )
17676 .expect("a part");
17677 let mut writer = writer
17678 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
17679 .expect("a second table");
17680 writer
17681 .append(
17682 &Chunk::new(vec![
17683 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
17684 ])
17685 .expect("one column"),
17686 )
17687 .expect("a part");
17688 let mut writer =
17689 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
17690 for part in 0..70_i64 {
17691 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
17692 writer
17693 .append(
17694 &Chunk::new(vec![
17695 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
17696 ])
17697 .expect("one column"),
17698 )
17699 .expect("a part");
17700 }
17701 writer.finish().expect("commit");
17702 }
17703
17704 #[test]
17705 fn three_tables_in_one_file_read_back_by_name() {
17706 let file = path("three-tables");
17707 three_tables(&file);
17708 let catalog = Catalog::open(&file).expect("a committed catalog");
17709 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
17710
17711 let region = catalog.table("region").expect("the first table");
17712 assert_eq!(region.table().rows(), 2);
17713 assert_eq!(
17714 region.read(0, &[1]).expect("names").value_at(1, 0),
17715 Value::Varchar("ASIA".to_owned())
17716 );
17717
17718 let wide = catalog.table("wide").expect("the third table");
17719 assert_eq!(wide.table().rows(), 70 * 64);
17720 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
17721
17722 let empty = catalog.table("empty").expect("the second table");
17725 assert_eq!(empty.table().rows(), 1);
17726 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
17727
17728 fs::remove_file(file).expect("remove scratch file");
17729 }
17730
17731 #[test]
17732 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
17733 let file = path("three-tables-missing");
17734 three_tables(&file);
17735 let catalog = Catalog::open(&file).expect("a committed catalog");
17736 let error = catalog.table("nation").expect_err("no such table");
17737 assert!(error.message().contains("nation"), "{}", error.message());
17738 fs::remove_file(file).expect("remove scratch file");
17739 }
17740
17741 #[test]
17742 fn a_file_of_three_tables_will_not_open_as_one() {
17743 let file = path("three-tables-unnamed");
17744 three_tables(&file);
17745 let error = Reader::open(&file).expect_err("more than one table");
17746 assert!(error.message().contains("more than one table"), "{}", error.message());
17747 fs::remove_file(file).expect("remove scratch file");
17748 }
17749
17750 #[test]
17752 fn decimals_of_every_storage_width_round_trip() {
17753 let file = path("decimals");
17754 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
17755 let fields = widths
17756 .iter()
17757 .enumerate()
17758 .map(|(index, (width, scale))| {
17759 Field::new(
17760 format!("d{index}"),
17761 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17762 )
17763 })
17764 .collect::<Vec<_>>();
17765 let mut writer = Writer::create(&file, "money", fields).expect("new file");
17766 let rows: [i128; 3] = [-1234, 0, 999];
17767 let columns = widths
17768 .iter()
17769 .map(|(width, scale)| {
17770 let values = rows
17771 .iter()
17772 .map(|unscaled| Value::Decimal {
17773 unscaled: *unscaled,
17774 width: *width,
17775 scale: *scale,
17776 })
17777 .collect::<Vec<_>>();
17778 Vector::from_values(
17779 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17780 &values,
17781 )
17782 .expect("a decimal column")
17783 })
17784 .collect::<Vec<_>>();
17785 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
17786 writer.finish().expect("commit");
17787
17788 let reader = Reader::open(&file).expect("a committed file");
17789 for (index, (width, scale)) in widths.iter().enumerate() {
17790 assert_eq!(
17791 reader.table().fields()[index].ty,
17792 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17793 "column {index} came back as another type"
17794 );
17795 let column = reader.read(0, &[index]).expect("the column");
17796 for (row, unscaled) in rows.iter().enumerate() {
17797 assert_eq!(
17798 column.value_at(row, 0),
17799 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
17800 "column {index} row {row}"
17801 );
17802 }
17803 }
17804 fs::remove_file(file).expect("remove scratch file");
17805 }
17806
17807 #[test]
17808 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
17809 let file = path("two-of-a-name");
17810 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
17811 .expect("new file");
17812 let error = writer
17813 .next("t", vec![Field::new("a", LogicalType::BigInt)])
17814 .expect_err("the same name twice");
17815 assert!(error.message().contains("same name"), "{}", error.message());
17816 fs::remove_file(file).expect("remove scratch file");
17817 }
17818
17819 #[test]
17820 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
17821 let file = path("integer-tally");
17822 let mut writer =
17823 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
17824 .expect("new file");
17825 let mut values = vec![Value::SmallInt(0); 1024];
17826 values[7] = Value::SmallInt(3);
17827 values[99] = Value::SmallInt(-2);
17828 values[1001] = Value::SmallInt(3);
17829 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
17830 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
17831 values[0] = Value::Null;
17832 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
17833 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
17834 writer.finish().expect("commit");
17835
17836 let reader = Reader::open(&file).expect("read file");
17837 assert_eq!(
17838 reader.integer_tally(0, 0).expect("valid part"),
17839 Some(vec![(-2, 1), (0, 1021), (3, 2)])
17840 );
17841 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
17842 let catalog = Catalog::open(&file).expect("catalog");
17843 assert_eq!(
17844 catalog.integer_tally("events", 0).expect("nullable column"),
17845 Some(vec![(-2, 2), (0, 2041), (3, 4)])
17846 );
17847 fs::remove_file(file).expect("remove scratch file");
17848 }
17849
17850 #[test]
17851 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17852 let file = path("catalog-integer-tally");
17853 let mut writer = Writer::create(
17854 &file,
17855 "events",
17856 vec![
17857 Field::new("noise", LogicalType::SmallInt),
17858 Field::new("source", LogicalType::SmallInt),
17859 ],
17860 )
17861 .expect("new file");
17862 let noise = vec![Value::SmallInt(9); 1024];
17863 let mut source = vec![Value::SmallInt(0); 1024];
17864 source[7] = Value::SmallInt(3);
17865 source[99] = Value::SmallInt(-2);
17866 let chunk = Chunk::new(vec![
17867 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17868 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17869 ])
17870 .expect("two columns");
17871 writer.append(&chunk).expect("append");
17872 writer.finish().expect("commit");
17873
17874 let catalog = Catalog::open(&file).expect("catalog");
17875 assert_eq!(
17876 catalog.integer_tally("events", 1).expect("selected column"),
17877 Some(vec![(-2, 1), (0, 1022), (3, 1)])
17878 );
17879 assert_eq!(
17880 catalog.integer_tally("events", 0).expect("other column"),
17881 Some(vec![(9, 1024)])
17882 );
17883 fs::remove_file(file).expect("remove scratch file");
17884 }
17885
17886 #[test]
17887 fn opening_the_catalog_reads_no_table_directory() {
17888 let file = path("catalog-only");
17889 three_tables(&file);
17890 let catalog = Catalog::open(&file).expect("a committed catalog");
17891 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17894 assert_eq!(catalog.names().len(), 3);
17895 fs::remove_file(file).expect("remove scratch file");
17896 }
17897
17898 #[test]
17909 fn the_checksum_answers_what_it_has_always_answered() {
17910 let bytes: Vec<u8> =
17911 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17912 for (length, expected) in [
17913 (0, 0xef46_db37_51d8_e999),
17914 (1, 0xa96c_7f0c_e858_bbb7),
17915 (3, 0x56e6_9576_32a4_87f9),
17916 (4, 0xc60d_15b1_e3ff_8f04),
17917 (5, 0x8088_1585_8624_dd4e),
17918 (7, 0xafbe_fc3d_6c6f_9a8e),
17919 (8, 0x3da5_c7aa_2696_83e0),
17920 (9, 0x465e_c429_b13c_3892),
17921 (15, 0xdee8_9d8a_065a_6233),
17922 (16, 0x1330_489a_7767_9c80),
17923 (31, 0x3391_303d_485e_846e),
17924 (32, 0x40b7_aff7_5d45_bbc8),
17925 (33, 0x4997_cae4_951c_17a5),
17926 (39, 0x5807_28fd_5c14_5739),
17927 (40, 0xf95c_f6f5_c08a_3d3b),
17928 (63, 0x2944_b4da_fc69_b206),
17929 (64, 0xbb76_f6ef_19bd_5a1b),
17930 (65, 0x814e_0c65_4a9f_d640),
17931 (127, 0x00de_aab1_31cf_f89b),
17932 (1000, 0x9e33_00c1_cde3_c58d),
17933 ] {
17934 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17935 }
17936 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17937 }
17938 #[test]
17945 fn a_declared_order_comes_back_out_of_the_file() {
17946 let path = path("clustered");
17947 let shipped = vec![
17948 Field::new("key", LogicalType::BigInt),
17949 Field::new("line", LogicalType::Integer),
17950 Field::new("shipdate", LogicalType::Date),
17951 ];
17952 let plain = vec![Field::new("a", LogicalType::Integer)];
17953 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17954
17955 let mut writer = Writer::create(&path, "lineitem", shipped)
17956 .expect("new file")
17957 .declare(stage_zero.clone())
17958 .expect("the columns are the table's");
17959 let column = |ty: LogicalType, values: &[Value]| {
17960 Vector::from_values(ty, values).expect("the values match the type")
17961 };
17962 writer
17963 .append(
17964 &Chunk::new(vec![
17965 column(
17966 LogicalType::BigInt,
17967 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17968 ),
17969 column(
17970 LogicalType::Integer,
17971 &[
17972 Value::Integer(1),
17973 Value::Integer(1),
17974 Value::Integer(1),
17975 Value::Integer(1),
17976 ],
17977 ),
17978 column(
17979 LogicalType::Date,
17980 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17981 ),
17982 ])
17983 .expect("three columns"),
17984 )
17985 .expect("four rows");
17986 let mut writer = writer.next("nation", plain).expect("a second table");
17987 writer
17988 .append(
17989 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17990 .expect("one column"),
17991 )
17992 .expect("one row");
17993 writer.finish().expect("commit");
17994
17995 let catalog = Catalog::open(&path).expect("reopen");
17996 let lineitem = catalog.table("lineitem").expect("the clustered table");
17997 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17998 let nation = catalog.table("nation").expect("the plain table");
17999 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
18000
18001 assert_eq!(lineitem.table().rows(), 4);
18004 assert_eq!(nation.table().rows(), 1);
18005 fs::remove_file(&path).ok();
18006 }
18007
18008 #[test]
18010 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
18011 let path = path("clustered-bad");
18012 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
18013 .expect("new file");
18014 let four =
18015 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
18016 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
18017 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
18018 fs::remove_file(&path).ok();
18019 }
18020
18021 #[test]
18027 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
18028 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
18029 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
18030 .collect::<Vec<_>>();
18031 let filled = || {
18032 let mut dictionary = GlobalDictionary::new();
18033 for value in &values {
18034 dictionary.code(value).expect("a code for every value");
18035 }
18036 dictionary.settle().expect("a shape");
18037 dictionary
18038 };
18039 let mut in_place = filled();
18040 in_place.finish_blocks().expect("every block encodes");
18041
18042 let mut handed = filled();
18043 let out = handed.hand_out(3);
18044 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
18045 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
18046 for job in out.iter().rev() {
18047 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
18048 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
18049 }
18050 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
18051 handed.finish_blocks().expect("the last block encodes");
18052
18053 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
18054 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
18055 }
18056
18057 #[test]
18059 fn a_block_given_back_twice_is_refused() {
18060 let mut dictionary = GlobalDictionary::new();
18061 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
18062 dictionary.code(&format!("value {at}")).expect("a code");
18063 }
18064 dictionary.settle().expect("a shape");
18065 let out = dictionary.hand_out(0);
18066 let last = out.last().expect("blocks went out");
18067 let at = last.place().1;
18068 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
18069 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
18070 }
18071
18072 #[test]
18078 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
18079 let mut values = vec![String::new(), "http://".to_owned()];
18080 for host in 0..7 {
18081 for path in 0..30 {
18082 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
18083 values.push(format!("http://example{host}.test/page/{path:04}"));
18084 }
18085 }
18086 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
18087
18088 let mut dictionary = GlobalDictionary::new();
18089 for value in &values {
18090 dictionary.code(value).expect("a code for every value");
18091 }
18092 dictionary.finish_blocks().expect("the last block encodes");
18093 let ranked = dictionary.ranked(None).expect("a sorted order");
18094 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
18095
18096 let spellings = dictionary_values(&dictionary);
18097 let seen = ranked
18098 .iter()
18099 .map(|&(_, code)| {
18100 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
18101 })
18102 .collect::<Vec<_>>();
18103 let mut wanted = values.clone();
18104 wanted.sort_unstable();
18105 assert_eq!(seen, wanted, "the order is the order the bytes give");
18106
18107 for &(carried, code) in &ranked {
18108 let value = &spellings[code as usize];
18109 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
18110 }
18111 }
18112
18113 #[test]
18118 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
18119 let entry =
18120 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
18121 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
18122 .map(|code| entry(code, u64::from(code % 7) + 1))
18123 .collect::<Vec<_>>();
18124 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
18125
18126 let mut sorted = all.clone();
18127 sorted.sort_unstable_by(|left, right| {
18128 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
18129 });
18130 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
18131 sorted.truncate(FREQUENCY_ENTRIES);
18132
18133 let mut picked = all.clone();
18134 let omitted = keep_most_frequent(&mut picked);
18135 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
18136 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
18137 assert!(
18138 picked
18139 .iter()
18140 .zip(&sorted)
18141 .all(|(one, two)| one.value == two.value && one.count == two.count),
18142 "the same entries in the same order"
18143 );
18144
18145 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
18146 let omitted = keep_most_frequent(&mut short);
18147 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
18148 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
18149 }
18150
18151 #[test]
18153 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
18154 let empty = GlobalDictionary::new();
18155 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
18156
18157 let mut dictionary = GlobalDictionary::new();
18158 for value in ["pear", "apple", "", "apples", "app"] {
18159 dictionary.code(value).expect("a code for every value");
18160 }
18161 dictionary.finish_blocks().expect("the one block encodes");
18162 let spellings = dictionary_values(&dictionary);
18163 let seen = dictionary
18164 .ranked(None)
18165 .expect("a sorted order")
18166 .iter()
18167 .map(|&(_, code)| spellings[code as usize].clone())
18168 .collect::<Vec<_>>();
18169 let wanted: Vec<Vec<u8>> =
18170 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
18171 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
18172 }
18173
18174 #[test]
18177 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
18178 let profile = LoadProfile::begin("demoted");
18179 let mut dictionary = GlobalDictionary::new();
18180 for value in 0..50_000 {
18181 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
18182 }
18183 let (_, grown) = dictionary.recharge(Some(&profile));
18184 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
18185
18186 dictionary.demote();
18187 let (before, after) = dictionary.recharge(Some(&profile));
18188 assert_eq!(before, grown);
18189 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
18192 assert_eq!(profile.held(), after, "the profile was told about the drop");
18193 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
18194
18195 dictionary.demote();
18196 assert_eq!(
18197 dictionary.recharge(Some(&profile)),
18198 (after, after),
18199 "demoting twice is a no-op"
18200 );
18201 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
18202 }
18203}