1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_io::{Filesystem, OpenMode, RealFilesystem};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60mod projection;
61mod run_projection;
62use prepare::Lent;
63pub mod section;
64pub mod stats;
65mod zones;
66
67pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
68pub use projection::build_sorted_projection;
69pub use run_projection::build_run_projection;
70pub use section::Section;
71pub use zones::{Common, Stripes, ascending, distincts};
72
73const MAGIC: &[u8; 8] = b"RUDBNV10";
74const DIRECTORY: &[u8; 8] = b"RUDBDI10";
75const CATALOG: &[u8; 8] = b"RUDBCA10";
76const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
77const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
78const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
79const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
80const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
81const MAX_CATALOG_FREQUENCIES: usize = 64;
82const FORMAT: u32 = 29;
83
84const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
115
116const HEADER: u64 = 80;
117const SLOT_BYTES: usize = 28;
118const MAX_PAGE: usize = 256 * 1024 * 1024;
119const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
120const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
121const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
122const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
130const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
132const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
138const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
153const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
173const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
181const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
189
190const MAX_SECTIONS: usize = 4096;
197const FREQUENCY_CANDIDATES: usize = 32_768;
198const FREQUENCY_ENTRIES: usize = 512;
199const FREQUENCY_BUILD_RANK: usize = 10;
200const FREQUENCY_ORDINALS: usize = 131_072;
201const MAX_PAIR_FREQUENCIES: usize = 1024;
202const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
207const MAX_FREQUENCY_WORKERS: usize = 32;
214
215fn close_workers() -> usize {
217 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
218}
219
220const CLOSE_BYTES: usize = 1 << 30;
231
232const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
235
236const MAX_ENCODE_WORKERS: usize = 32;
243
244const WRITEBACK_STRETCH: u64 = 32 << 20;
252
253const SIEVE_BUDGET: usize = 8 * 1024;
261
262const PART_BOUND_BYTES: usize = 24;
271
272fn io(error: std::io::Error) -> Error {
273 Error::io(error.to_string())
274}
275
276fn invalid(message: &str) -> Error {
277 Error::invalid_input(format!("invalid rudb native file: {message}"))
278}
279
280fn sum(counts: impl Iterator<Item = u64>) -> u64 {
282 counts.fold(0, u64::saturating_add)
283}
284
285fn span_bytes(spans: &[Span], at: usize) -> u64 {
287 spans.get(at).map_or(0, |span| u64::from(span.length))
288}
289
290fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
292 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
293}
294
295fn dictionary_bytes(table: &Table, at: usize) -> u64 {
297 page_bytes(&table.dictionaries, at)
298 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
299}
300
301fn checksum(bytes: &[u8]) -> u64 {
311 seeded_checksum(bytes, 0)
312}
313
314#[must_use]
321pub fn content_name(bytes: &[u8]) -> u128 {
322 let seed = u64::from(FORMAT);
323 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
324}
325
326#[derive(Debug, Clone)]
332pub struct ContentNamer {
333 seeds: [u64; 2],
334 lanes: [[u64; 4]; 2],
335 held: [u8; 32],
336 filled: usize,
337 length: u64,
338}
339
340impl Default for ContentNamer {
341 fn default() -> Self {
342 let seed = u64::from(FORMAT);
343 let seeds = [seed, !seed];
344 let lanes = seeds.map(|seed| {
345 [
346 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
347 seed.wrapping_add(XXH_P2),
348 seed,
349 seed.wrapping_sub(XXH_P1),
350 ]
351 });
352 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
353 }
354}
355
356impl ContentNamer {
357 pub fn update(&mut self, mut bytes: &[u8]) {
359 self.length += bytes.len() as u64;
360 if self.filled > 0 {
361 let take = (32 - self.filled).min(bytes.len());
362 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
363 self.filled += take;
364 bytes = &bytes[take..];
365 if self.filled < 32 {
366 return;
367 }
368 let block = self.held;
369 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
370 self.filled = 0;
371 }
372 let mut blocks = bytes.chunks_exact(32);
373 for block in blocks.by_ref() {
374 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
375 }
376 let rest = blocks.remainder();
377 self.held[..rest.len()].copy_from_slice(rest);
378 self.filled = rest.len();
379 }
380
381 #[must_use]
383 pub fn finish(&self) -> u128 {
384 let rest = &self.held[..self.filled];
385 let [first, second] = [0, 1].map(|at| {
386 if self.length < 32 {
387 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
388 } else {
389 finish_checksum(self.lanes[at], rest, self.length)
390 }
391 });
392 u128::from(first) << 64 | u128::from(second)
393 }
394}
395
396fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
405 let mut blocks = bytes.chunks_exact(32);
408 let rest = blocks.remainder();
409 if bytes.len() < 32 {
410 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
411 }
412 let mut lanes = [
413 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
414 seed.wrapping_add(XXH_P2),
415 seed,
416 seed.wrapping_sub(XXH_P1),
417 ];
418 for block in blocks.by_ref() {
419 checksum_block(&mut lanes, block);
420 }
421 finish_checksum(lanes, rest, bytes.len() as u64)
422}
423
424const XXH_P1: u64 = 11_400_714_785_074_694_791;
425const XXH_P2: u64 = 14_029_467_366_897_019_727;
426const XXH_P3: u64 = 1_609_587_929_392_839_161;
427const XXH_P4: u64 = 9_650_029_242_287_828_579;
428const XXH_P5: u64 = 2_870_177_450_012_600_261;
429
430fn checksum_round(state: u64, word: u64) -> u64 {
431 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
432}
433
434fn checksum_word(chunk: &[u8]) -> u64 {
435 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
436}
437
438fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
440 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
441 *lane = checksum_round(*lane, checksum_word(chunk));
442 }
443}
444
445fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
447 let merge = |state: u64, lane: u64| {
448 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
449 };
450 let [one, two, three, four] = lanes;
451 let combined = one
452 .rotate_left(1)
453 .wrapping_add(two.rotate_left(7))
454 .wrapping_add(three.rotate_left(12))
455 .wrapping_add(four.rotate_left(18));
456 let hash = merge(merge(merge(merge(combined, one), two), three), four);
457 checksum_tail(hash.wrapping_add(length), rest)
458}
459
460fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
462 let mut words = rest.chunks_exact(8);
463 for chunk in words.by_ref() {
464 hash ^= checksum_round(0, checksum_word(chunk));
465 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
466 }
467 rest = words.remainder();
468 if rest.len() >= 4 {
469 let (head, tail) = rest.split_at(4);
470 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
471 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
472 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
473 rest = tail;
474 }
475 for &byte in rest {
476 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
477 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
478 }
479 hash ^= hash >> 33;
480 hash = hash.wrapping_mul(XXH_P2);
481 hash ^= hash >> 29;
482 hash = hash.wrapping_mul(XXH_P3);
483 hash ^ (hash >> 32)
484}
485
486fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
492 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
493}
494
495fn walk_checksummed(
501 file: &File,
502 offset: u64,
503 length: usize,
504 window: usize,
505 mut each: impl FnMut(&[u8]) -> Result<()>,
506) -> Result<u64> {
507 debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
508 if length < 32 {
509 let mut bytes = vec![0; length];
510 read_at(file, offset, &mut bytes)?;
511 each(&bytes)?;
512 return Ok(checksum(&bytes));
513 }
514 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
515 let mut buffer = vec![0; window.min(length)];
516 let mut read = 0;
517 let (mut whole, mut filled) = (0, 0);
518 while read < length {
519 filled = buffer.len().min(length - read);
520 read_at(file, offset + read as u64, &mut buffer[..filled])?;
521 read += filled;
522 each(&buffer[..filled])?;
523 whole = filled / 32 * 32;
524 for block in buffer[..whole].chunks_exact(32) {
525 checksum_block(&mut lanes, block);
526 }
527 }
528 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
529}
530
531#[derive(Debug, Clone, Copy)]
532struct Slot {
533 offset: u64,
534 length: u32,
535 generation: u64,
536 hash: u64,
537}
538
539impl Slot {
540 fn bytes(self) -> [u8; SLOT_BYTES] {
541 let mut result = [0; SLOT_BYTES];
542 result[..8].copy_from_slice(&self.offset.to_le_bytes());
543 result[8..12].copy_from_slice(&self.length.to_le_bytes());
544 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
545 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
546 result
547 }
548
549 fn read(bytes: &[u8]) -> Self {
550 Self {
551 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
552 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
553 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
554 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
555 }
556 }
557}
558
559#[derive(Debug, Clone, Copy)]
560struct Page {
561 offset: u64,
562 length: u32,
563 hash: u64,
564}
565
566impl Page {
567 fn bytes(&self) -> u64 {
569 u64::from(self.length)
570 }
571}
572
573#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
574enum FrequencyValue {
575 Null,
576 Integer(i128),
577 Code(u32),
578}
579
580type FrequencyMap<V> = HashMap<u64, V, Spread>;
586
587#[derive(Debug)]
601struct Candidates {
602 slots: Vec<Candidate>,
605 held: usize,
606 nulls: u32,
607 decrements: u64,
608 survivors: Vec<Candidate>,
610}
611
612#[derive(Debug, Default, Clone, Copy)]
614struct Candidate {
615 bits: u64,
616 count: u32,
617}
618
619const FIRST_CANDIDATE_SLOTS: usize = 64;
621
622impl Default for Candidates {
623 fn default() -> Self {
624 Self {
625 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
626 held: 0,
627 nulls: 0,
628 decrements: 0,
629 survivors: Vec::new(),
630 }
631 }
632}
633
634impl Candidates {
635 fn add(&mut self, bits: Option<u64>, mut times: u32) {
642 while times > 0 {
643 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
644 match bits {
645 Some(bits) => {
646 let (at, found) = self.find(bits);
647 if found {
648 self.slots[at].count = self.slots[at].count.saturating_add(times);
649 return;
650 }
651 if room {
652 self.place(at, bits, times);
653 return;
654 }
655 }
656 None if self.nulls != 0 => {
657 self.nulls = self.nulls.saturating_add(times);
658 return;
659 }
660 None if room => {
661 self.nulls = times;
662 return;
663 }
664 None => {}
665 }
666 self.decrement();
667 times -= 1;
668 }
669 }
670
671 fn find(&self, bits: u64) -> (usize, bool) {
673 let mask = self.slots.len() - 1;
674 let mut at = home(bits, self.slots.len());
675 loop {
676 let slot = self.slots[at];
677 if slot.count == 0 {
678 return (at, false);
679 }
680 if slot.bits == bits {
681 return (at, true);
682 }
683 at = (at + 1) & mask;
684 }
685 }
686
687 fn position(&self, bits: u64) -> Option<usize> {
689 match self.find(bits) {
690 (at, true) => Some(at),
691 (_, false) => None,
692 }
693 }
694
695 fn place(&mut self, at: usize, bits: u64, count: u32) {
698 let at = if (self.held + 1) * 2 > self.slots.len() {
699 let wider = self.slots.len() * 2;
700 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
701 for slot in old.into_iter().filter(|slot| slot.count != 0) {
702 let (to, _) = self.find(slot.bits);
703 self.slots[to] = slot;
704 }
705 self.find(bits).0
706 } else {
707 at
708 };
709 self.slots[at] = Candidate { bits, count };
710 self.held += 1;
711 }
712
713 fn decrement(&mut self) {
715 let mut survivors = std::mem::take(&mut self.survivors);
716 survivors.clear();
717 survivors.extend(
718 self.slots
719 .iter()
720 .filter(|slot| slot.count > 1)
721 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
722 );
723 self.slots.fill(Candidate::default());
724 self.held = survivors.len();
725 for &slot in &survivors {
726 let (at, _) = self.find(slot.bits);
727 self.slots[at] = slot;
728 }
729 self.survivors = survivors;
730 self.nulls = self.nulls.saturating_sub(1);
731 self.decrements = self.decrements.saturating_add(1);
732 }
733
734 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
736 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
737 }
738}
739
740fn home(bits: u64, slots: usize) -> usize {
745 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
746}
747
748#[derive(Debug, Default)]
750struct Run {
751 bits: Option<u64>,
752 times: u32,
753}
754
755impl Run {
756 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
758 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
759 self.times += 1;
760 return None;
761 }
762 let ended = self.take();
763 self.bits = bits;
764 self.times = 1;
765 ended
766 }
767
768 fn take(&mut self) -> Option<(Option<u64>, u32)> {
770 let times = std::mem::take(&mut self.times);
771 (times != 0).then_some((self.bits, times))
772 }
773}
774
775#[derive(Debug, Default, Clone, Copy)]
777struct Spread;
778
779impl std::hash::BuildHasher for Spread {
780 type Hasher = SpreadHasher;
781
782 fn build_hasher(&self) -> SpreadHasher {
783 SpreadHasher(0)
784 }
785}
786
787#[derive(Debug)]
794struct SpreadHasher(u64);
795
796impl SpreadHasher {
797 fn mix(&mut self, word: u64) {
798 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
799 self.0 = (product as u64) ^ ((product >> 64) as u64);
800 }
801}
802
803impl std::hash::Hasher for SpreadHasher {
804 fn write(&mut self, bytes: &[u8]) {
805 for part in bytes.chunks(8) {
806 let mut word = [0; 8];
807 word[..part.len()].copy_from_slice(part);
808 self.mix(u64::from_le_bytes(word));
809 }
810 }
811
812 fn write_u32(&mut self, value: u32) {
813 self.mix(u64::from(value));
814 }
815
816 fn write_u64(&mut self, value: u64) {
817 self.mix(value);
818 }
819
820 fn write_i128(&mut self, value: i128) {
821 self.mix(value as u64);
822 self.mix((value >> 64) as u64);
823 }
824
825 fn write_isize(&mut self, value: isize) {
826 self.mix(value as u64);
827 }
828
829 fn finish(&self) -> u64 {
830 self.0
831 }
832}
833
834#[derive(Debug, Clone)]
835struct FrequencyEntry {
836 value: FrequencyValue,
837 count: u64,
838}
839
840#[derive(Debug, Clone)]
845struct FrequencySummary {
846 entries: Vec<FrequencyEntry>,
847 omitted_max: u64,
848 ordinals: Vec<u64>,
849 ordinal_entries: Vec<u16>,
850}
851
852#[derive(Debug, Clone)]
853struct PairFrequencyEntry {
854 first_entry: u16,
855 second: Option<u32>,
856 count: u64,
857}
858
859#[derive(Debug, Clone)]
865struct PairFrequencySummary {
866 first: u16,
867 second: u16,
868 entries: Vec<PairFrequencyEntry>,
869 omitted_max: u64,
870}
871
872#[derive(Debug, Clone)]
880enum Frequencies {
881 Held(FrequencySummary),
882 Stored {
885 span: Span,
886 values: bool,
887 },
888}
889
890#[derive(Debug, Clone)]
895pub struct FrequencyPrefix {
896 pub entries: Vec<(Value, u64)>,
898 pub omitted_max: u64,
900}
901
902#[derive(Debug, Clone, PartialEq)]
904pub struct FrequencyOccurrences {
905 pub omitted_max: u64,
907 pub ordinals: Vec<u64>,
909 pub anchors: Vec<Value>,
911 pub anchor_indices: Vec<u16>,
913}
914
915pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
917
918#[derive(Debug, Clone, Copy, Default)]
925struct Span {
926 offset: u64,
927 length: u32,
928}
929
930#[derive(Debug, Clone, Default)]
938struct Pages {
939 columns: usize,
940 held: Box<[StripePage]>,
941}
942
943#[derive(Debug, Clone, Copy)]
945struct StripePage {
946 offset: u64,
947 hash: u64,
948 length: u32,
949 column: u32,
950}
951
952impl Pages {
953 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
955 let mut held = Vec::with_capacity(slots.iter().flatten().count());
956 for (column, page) in slots.iter().enumerate() {
957 if let Some(page) = page {
958 let column =
959 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
960 held.push(StripePage {
961 offset: page.offset,
962 hash: page.hash,
963 length: page.length,
964 column,
965 });
966 }
967 }
968 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
969 }
970
971 fn get(&self, column: usize) -> Option<Page> {
973 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
974 let placed = self.held[at];
975 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
976 }
977
978 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
980 (0..self.columns).map(|column| self.get(column))
981 }
982
983 fn bytes(&self, column: usize) -> u64 {
985 self.get(column).map_or(0, |page| page.bytes())
986 }
987}
988
989#[derive(Debug, Clone)]
991pub struct Stripe {
992 rows: usize,
993 parts: Vec<u32>,
996 index: Span,
1000 pages: Vec<Span>,
1001 memberships: Pages,
1002 sieves: Pages,
1005 part_ranges: Pages,
1016 zone: Zone,
1017}
1018
1019impl Stripe {
1020 #[must_use]
1022 pub fn rows(&self) -> usize {
1023 self.rows
1024 }
1025
1026 #[must_use]
1028 pub fn parts(&self) -> usize {
1029 self.parts.len()
1030 }
1031
1032 #[must_use]
1038 pub fn zone(&self) -> &Zone {
1039 &self.zone
1040 }
1041}
1042
1043#[derive(Debug, Clone)]
1045pub struct Table {
1046 name: String,
1047 fields: Vec<Field>,
1048 stripes: Vec<Stripe>,
1049 rows: usize,
1050 dictionaries: Vec<Option<Page>>,
1051 dictionary_payloads: Vec<u64>,
1057 demoted: Vec<bool>,
1063 frequencies: Vec<Option<Frequencies>>,
1064 pair_frequencies: Vec<PairFrequencySummary>,
1065 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1070 host_groups: Option<host::HostSummary>,
1072 distincts: Vec<Option<u64>>,
1082 clustering: Option<Clustering>,
1090 generation: u64,
1104 sections: Vec<Section>,
1111}
1112
1113impl Table {
1114 #[must_use]
1116 pub fn name(&self) -> &str {
1117 &self.name
1118 }
1119
1120 #[must_use]
1122 pub fn fields(&self) -> &[Field] {
1123 &self.fields
1124 }
1125
1126 #[must_use]
1128 pub fn rows(&self) -> usize {
1129 self.rows
1130 }
1131
1132 #[must_use]
1134 pub fn stripes(&self) -> &[Stripe] {
1135 &self.stripes
1136 }
1137
1138 #[must_use]
1140 pub fn clustering(&self) -> Option<&Clustering> {
1141 self.clustering.as_ref()
1142 }
1143
1144 #[must_use]
1149 pub fn generation(&self) -> u64 {
1150 self.generation
1151 }
1152
1153 #[must_use]
1160 pub fn sections(&self) -> &[Section] {
1161 &self.sections
1162 }
1163}
1164
1165#[derive(Debug, Clone)]
1177struct Entry {
1178 name: String,
1179 fields: Vec<Field>,
1180 rows: usize,
1181 directory: Page,
1183 nonzero: Vec<Option<u64>>,
1186 aggregates: Vec<Option<(i128, u64)>>,
1188 distincts: Vec<Option<u64>>,
1190 extremes: Vec<StoredIntegerExtremes>,
1192 frequencies: Vec<StoredNumericFrequencies>,
1194}
1195
1196type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1197type StoredNumericFrequencies = Option<NumericFrequencies>;
1198
1199#[derive(Debug, Clone, PartialEq, Eq)]
1212pub struct ViewEntry {
1213 pub name: String,
1215 pub sql: String,
1217 pub statement: String,
1219 pub aliases: Vec<String>,
1221 pub columns: Vec<Field>,
1223}
1224
1225#[derive(Debug, Clone)]
1227pub struct ColumnLayout {
1228 pub name: String,
1230 pub kind: String,
1232 pub pages: u64,
1234 pub memberships: u64,
1236 pub sieves: u64,
1238 pub part_ranges: u64,
1240 pub dictionary: u64,
1242}
1243
1244impl ColumnLayout {
1245 #[must_use]
1247 pub fn total(&self) -> u64 {
1248 self.pages
1249 .saturating_add(self.memberships)
1250 .saturating_add(self.sieves)
1251 .saturating_add(self.part_ranges)
1252 .saturating_add(self.dictionary)
1253 }
1254}
1255
1256#[derive(Debug, Clone)]
1267pub struct Layout {
1268 pub file: u64,
1270 pub rows: usize,
1272 pub stripes: usize,
1274 pub parts: usize,
1276 pub columns: Vec<ColumnLayout>,
1278 pub indexes: u64,
1281 pub directory: u64,
1283 pub header: u64,
1285}
1286
1287impl Layout {
1288 #[must_use]
1290 pub fn columns_total(&self) -> u64 {
1291 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1292 }
1293
1294 #[must_use]
1300 pub fn unaccounted(&self) -> u64 {
1301 self.file
1302 .saturating_sub(self.columns_total())
1303 .saturating_sub(self.indexes)
1304 .saturating_sub(self.directory)
1305 .saturating_sub(self.header)
1306 }
1307}
1308
1309#[derive(Debug, Clone)]
1320pub struct StoredPart {
1321 pub stripe: usize,
1323 pub part: usize,
1325 pub row: usize,
1327 pub rows: usize,
1329 pub encoding: String,
1331 pub bytes: u64,
1333 pub page: u64,
1335 pub offset: u64,
1337 pub low: Option<Value>,
1339 pub high: Option<Value>,
1341 pub nulls: Option<usize>,
1343}
1344
1345const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1352
1353#[derive(Debug)]
1378struct GlobalDictionary {
1379 primary: HashMap<u64, u32, Spread>,
1383 collisions: HashMap<u64, Vec<u32>, Spread>,
1384 checks: Vec<u64>,
1386 ends: Vec<u32>,
1388 counts: Vec<u64>,
1389 nulls: u64,
1390 filling: Vec<u8>,
1392 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1398 waiting: Vec<(usize, Vec<u8>)>,
1403 sample: Vec<(usize, Vec<u8>)>,
1409 stride: usize,
1411 shape: Option<chooser::Settled>,
1413 settled: usize,
1415 blocks: Vec<Vec<u8>>,
1420 early: BTreeMap<usize, EncodedBlock>,
1426 placed: Vec<Placed>,
1428 charged: u64,
1431 demoted: bool,
1433}
1434
1435#[derive(Debug, Clone, Copy)]
1437struct Placed {
1438 start: u64,
1439 length: u64,
1440 hash: u64,
1441}
1442
1443type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1445
1446impl GlobalDictionary {
1447 fn new() -> Self {
1448 Self {
1449 primary: HashMap::default(),
1450 collisions: HashMap::default(),
1451 checks: Vec::new(),
1452 ends: Vec::new(),
1453 counts: Vec::new(),
1454 nulls: 0,
1455 filling: Vec::new(),
1456 grams: Vec::new(),
1457 waiting: Vec::new(),
1458 sample: Vec::new(),
1459 stride: 1,
1460 shape: None,
1461 settled: 0,
1462 blocks: Vec::new(),
1463 early: BTreeMap::new(),
1464 placed: Vec::new(),
1465 charged: 0,
1466 demoted: false,
1467 }
1468 }
1469
1470 fn values(&self) -> usize {
1472 self.ends.len()
1473 }
1474
1475 fn closing_bytes(&self) -> usize {
1478 let values = self.values();
1479 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1480 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1481 .sum::<usize>();
1482 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1483 }
1484
1485 fn held_bytes(&self) -> u64 {
1491 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1492 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1493 }
1494 fn spilled<T>(values: &Vec<T>) -> usize {
1495 values.capacity() * size_of::<T>()
1496 }
1497 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1498 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1499 };
1500 let bytes = table(&self.primary)
1501 + table(&self.collisions)
1502 + self.collisions.values().map(spilled).sum::<usize>()
1503 + spilled(&self.checks)
1504 + spilled(&self.ends)
1505 + spilled(&self.counts)
1506 + self.filling.capacity()
1507 + spilled(&self.grams)
1508 + raw(&self.waiting)
1509 + raw(&self.sample)
1510 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1511 + spilled(&self.placed);
1512 bytes as u64
1513 }
1514
1515 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1518 let before = self.charged;
1519 let now = self.held_bytes();
1520 if let Some(profile) = profile {
1521 if now >= before {
1522 profile.hold(now - before);
1523 } else {
1524 profile.release(before - now);
1525 }
1526 }
1527 self.charged = now;
1528 (before, now)
1529 }
1530
1531 fn demote(&mut self) {
1539 if self.demoted {
1540 return;
1541 }
1542 self.seal_rest();
1543 self.release_lookup();
1544 self.demoted = true;
1545 }
1546
1547 fn release_lookup(&mut self) {
1554 self.primary = HashMap::default();
1555 self.collisions = HashMap::default();
1556 self.checks = Vec::new();
1557 self.sample = Vec::new();
1558 self.filling = Vec::new();
1559 }
1560
1561 fn encoded(&self) -> usize {
1563 self.placed.len() + self.blocks.len()
1564 }
1565
1566 #[cfg(test)]
1567 fn code(&mut self, text: &str) -> Result<u32> {
1568 let bytes = text.as_bytes();
1569 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1570 }
1571
1572 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1578 if let Some(&code) = self.primary.get(&hash) {
1579 if self.checks.get(code as usize) == Some(&check) {
1580 return Ok(code);
1581 }
1582 if let Some(codes) = self.collisions.get(&hash) {
1583 if let Some(code) =
1584 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1585 {
1586 return Ok(code);
1587 }
1588 }
1589 let code = self.insert(text, check)?;
1590 self.collisions.entry(hash).or_default().push(code);
1591 return Ok(code);
1592 }
1593 let code = self.insert(text, check)?;
1594 self.primary.insert(hash, code);
1595 Ok(code)
1596 }
1597
1598 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1599 if self.demoted {
1600 return Err(Error::internal("a value was coded against a demoted dictionary"));
1601 }
1602 let code = u32::try_from(self.ends.len())
1603 .map_err(|_| invalid("global dictionary has too many values"))?;
1604 self.filling.extend_from_slice(text);
1605 self.ends.push(
1606 u32::try_from(self.filling.len())
1607 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1608 );
1609 self.checks.push(check);
1610 self.counts.push(0);
1611 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1612 self.seal();
1613 }
1614 Ok(code)
1615 }
1616
1617 fn seal(&mut self) {
1623 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1624 let bytes = std::mem::take(&mut self.filling);
1625 if at % self.stride == 0 {
1626 self.sample.push((at, bytes.clone()));
1627 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1628 self.stride *= 2;
1629 let stride = self.stride;
1630 self.sample.retain(|(at, _)| at % stride == 0);
1631 }
1632 }
1633 self.waiting.push((at, bytes));
1634 }
1635
1636 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1638 block_values(self.block_ends(at), bytes)
1639 }
1640
1641 fn block_ends(&self, at: usize) -> &[u32] {
1643 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1644 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1645 &self.ends[first..last]
1646 }
1647
1648 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1655 let Some(shape) = &self.shape else { return Vec::new() };
1656 let waiting = std::mem::take(&mut self.waiting);
1657 waiting
1658 .into_iter()
1659 .map(|(at, bytes)| Unencoded {
1660 column,
1661 at,
1662 ends: self.block_ends(at).to_vec(),
1663 bytes,
1664 shape: shape.clone(),
1665 })
1666 .collect()
1667 }
1668
1669 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1672 if at < self.encoded() || self.early.insert(at, block).is_some() {
1673 return Err(Error::internal("a dictionary block came back twice"));
1674 }
1675 while let Some(block) = self.early.remove(&self.encoded()) {
1676 self.push_block(block);
1677 }
1678 Ok(())
1679 }
1680
1681 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1683 self.blocks.push(bytes);
1684 self.grams.push(*grams);
1685 }
1686
1687 fn settle(&mut self) -> Result<()> {
1695 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1696 return Ok(());
1697 }
1698 self.settle_on_sample()
1699 }
1700
1701 fn settle_rest(&mut self) -> Result<()> {
1709 if self.shape.is_some() || self.sample.is_empty() {
1710 return Ok(());
1711 }
1712 self.settle_on_sample()
1713 }
1714
1715 fn settle_on_sample(&mut self) -> Result<()> {
1716 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1717 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1718 return Ok(());
1719 }
1720 let sample =
1721 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1722 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1723 self.settled = complete;
1724 Ok(())
1725 }
1726
1727 fn seal_rest(&mut self) {
1729 if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1733 self.seal();
1734 }
1735 }
1736
1737 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1740 let (block, bytes) = &self.waiting[at];
1741 let values = self.slices(*block, bytes);
1742 let encoded = match &self.shape {
1743 Some(shape) => string::encode_with(&values, shape)?,
1744 None => string::encode(&values)?,
1745 };
1746 Ok((encoded, block_grams(&values)))
1747 }
1748
1749 #[cfg(test)]
1751 fn finish_blocks(&mut self) -> Result<()> {
1752 self.seal_rest();
1753 let made = (0..self.waiting.len())
1754 .map(|at| self.encode_waiting(at))
1755 .collect::<Result<Vec<_>>>()?;
1756 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1757 if self.encoded() != at {
1758 return Err(Error::internal("a dictionary block was encoded out of order"));
1759 }
1760 self.push_block(block);
1761 }
1762 Ok(())
1763 }
1764
1765 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1783 let count = self.placed.len() + self.blocks.len();
1784 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1785 return Err(invalid("global dictionary blocks do not cover its values"));
1786 }
1787 let mut bases = Vec::with_capacity(count);
1788 let mut total = 0_usize;
1789 for block in 0..count {
1790 bases.push(total as u64);
1791 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1792 total = total
1793 .checked_add(self.ends[last] as usize)
1794 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1795 }
1796 let mut flat = vec![0_u8; total];
1797 let mut outs = Vec::with_capacity(count);
1798 let mut rest = flat.as_mut_slice();
1799 for block in 0..count {
1800 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1801 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1802 outs.push((block, out));
1803 rest = after;
1804 }
1805 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1806 let mut stored = Vec::new();
1807 for (block, out) in run {
1808 let encoded = match self.placed.get(*block) {
1809 Some(place) => {
1810 let file = file.ok_or_else(|| {
1811 Error::internal("a written dictionary block has no file")
1812 })?;
1813 let length = usize::try_from(place.length).map_err(|_| {
1814 invalid("global dictionary block does not fit in memory")
1815 })?;
1816 stored.resize(length, 0);
1817 read_at(file, place.start, &mut stored)?;
1818 if checksum(&stored) != place.hash {
1819 return Err(invalid(
1820 "a global dictionary block did not read back as written",
1821 ));
1822 }
1823 stored.as_slice()
1824 }
1825 None => &self.blocks[*block - self.placed.len()],
1826 };
1827 let decoded = string::decode_flat(encoded)?;
1828 if decoded.bytes().len() != out.len() {
1829 return Err(invalid(
1830 "a global dictionary block is not the length its ends say",
1831 ));
1832 }
1833 out.copy_from_slice(decoded.bytes());
1834 }
1835 Ok(())
1836 };
1837 let workers = close_workers().min(count / 16).max(1);
1840 if workers <= 1 {
1841 one(&mut outs)?;
1842 } else {
1843 let per = count.div_ceil(workers);
1844 std::thread::scope(|scope| {
1845 outs.chunks_mut(per)
1846 .map(|run| scope.spawn(|| one(run)))
1847 .collect::<Vec<_>>()
1848 .into_iter()
1849 .try_for_each(|handle| {
1850 handle.join().map_err(|_| {
1851 Error::internal("a global dictionary decode worker panicked")
1852 })?
1853 })
1854 })?;
1855 }
1856 drop(outs);
1857 Ok((flat, bases))
1858 }
1859
1860 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1865 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1866 let Some(&end) = ends.get(code) else { return (0, 0) };
1867 let base = base as usize;
1868 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1869 (base + from, base + end as usize)
1870 }
1871
1872 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1892 let (flat, bases) = self.decoded(file)?;
1893 let value = |code: u32| {
1894 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1895 flat.get(from..to).unwrap_or_default()
1896 };
1897 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1898 sort_by_value_across(&mut codes, value, close_workers());
1899 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1900 Ok((order, flat, bases))
1901 }
1902
1903 #[cfg(test)]
1904 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1905 self.ranked_with_values(file).map(|(order, _, _)| order)
1906 }
1907}
1908
1909#[derive(Debug)]
1917pub struct Writer {
1918 file: Box<dyn rudb_io::File>,
1921 at: u64,
1929 written_back: u64,
1931 table: Table,
1932 generation: u64,
1933 order: Vec<((u64, u64), (u64, u64))>,
1936 next_order: u64,
1937 dictionaries: Vec<Option<GlobalDictionary>>,
1938 coded: Arc<prepare::Coding>,
1941 gathers: Vec<Option<stats::Gather>>,
1947 lent: Option<Arc<Lent>>,
1950 pending: Vec<PendingChunk>,
1951 closed: Vec<Entry>,
1953 views: Vec<ViewEntry>,
1958 profile: Option<Arc<LoadProfile>>,
1964}
1965
1966#[derive(Debug)]
1974struct PendingChunk {
1975 order: (u64, u64),
1976 chunk: Chunk,
1977}
1978
1979#[derive(Debug, Clone, Copy)]
1985struct Part {
1986 order: (u64, u64),
1987 rows: usize,
1988 footprint: usize,
1989}
1990
1991impl Part {
1992 fn of(pending: &PendingChunk) -> Self {
1993 Self {
1994 order: pending.order,
1995 rows: pending.chunk.len(),
1996 footprint: pending.chunk.footprint(),
1997 }
1998 }
1999}
2000
2001#[derive(Debug, Default)]
2007struct ColumnStripe {
2008 pages: Vec<Vec<u8>>,
2009 codes: Vec<Option<Vec<u32>>>,
2010 sieves: Vec<Option<Sieve>>,
2011 ranges: Vec<Range>,
2012}
2013
2014fn coded_type(ty: &LogicalType) -> bool {
2022 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2023}
2024
2025fn dictionary_tag(ty: &LogicalType) -> u8 {
2032 if ty == &LogicalType::Blob { 2 } else { 1 }
2033}
2034
2035fn weight(ty: &LogicalType) -> usize {
2043 match ty {
2044 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2045 LogicalType::HugeInt
2046 | LogicalType::UHugeInt
2047 | LogicalType::Uuid
2048 | LogicalType::Interval => 16,
2049 LogicalType::BigInt
2050 | LogicalType::UBigInt
2051 | LogicalType::Timestamp
2052 | LogicalType::Time
2053 | LogicalType::TimeTz
2054 | LogicalType::TimestampTz
2055 | LogicalType::TimestampS
2056 | LogicalType::TimestampMs
2057 | LogicalType::TimestampNs
2058 | LogicalType::Double
2059 | LogicalType::Decimal { .. } => 8,
2060 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2061 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2062 _ => 1,
2063 }
2064}
2065
2066pub const STRIPE_PARTS: usize = 64;
2073
2074const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2082
2083const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2099
2100const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2102
2103fn index_section(parts: usize) -> Result<usize> {
2105 parts
2106 .checked_mul(INDEX_ENTRY)
2107 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2108 .ok_or_else(|| invalid("index page length overflow"))
2109}
2110
2111impl Writer {
2112 pub fn open(
2131 path: impl AsRef<Path>,
2132 name: impl Into<String>,
2133 fields: Vec<Field>,
2134 ) -> Result<Self> {
2135 Self::open_in(&RealFilesystem::new(), path, name, fields)
2136 }
2137
2138 pub fn open_in(
2145 fs: &dyn Filesystem,
2146 path: impl AsRef<Path>,
2147 name: impl Into<String>,
2148 fields: Vec<Field>,
2149 ) -> Result<Self> {
2150 for field in &fields {
2151 type_tag(&field.ty)?;
2152 }
2153 let name = name.into();
2154 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2155 let size = file.len()?;
2156 let (slot, bytes, _) = committed_slot(&*file, size)?;
2157 let (mut closed, views) = decode_catalog(&bytes, size)?;
2158 if let Some(at) = closed.iter().position(|held| held.name == name) {
2169 if closed[at].rows > 0 {
2170 return Err(invalid("two tables in one native file have the same name"));
2171 }
2172 closed.remove(at);
2173 }
2174 let generation = slot
2179 .generation
2180 .checked_add(1)
2181 .ok_or_else(|| invalid("native file generation overflow"))?;
2182 Ok(Self {
2183 file,
2184 at: size,
2187 written_back: size,
2188 dictionaries: fields
2189 .iter()
2190 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2191 .collect(),
2192 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2193 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2194 lent: None,
2195 table: Table {
2196 name,
2197 dictionaries: vec![None; fields.len()],
2198 dictionary_payloads: Vec::new(),
2199 demoted: Vec::new(),
2200 distincts: vec![None; fields.len()],
2201 fields,
2202 stripes: Vec::new(),
2203 rows: 0,
2204 frequencies: Vec::new(),
2205 pair_frequencies: Vec::new(),
2206 frequency_texts: Vec::new(),
2207 host_groups: None,
2208 clustering: None,
2209 generation,
2210 sections: Vec::new(),
2211 },
2212 generation,
2213 order: Vec::new(),
2214 next_order: 0,
2215 pending: Vec::with_capacity(STRIPE_PARTS),
2216 closed,
2217 views,
2218 profile: None,
2219 })
2220 }
2221
2222 pub fn create(
2228 path: impl AsRef<Path>,
2229 name: impl Into<String>,
2230 fields: Vec<Field>,
2231 ) -> Result<Self> {
2232 Self::create_in(&RealFilesystem::new(), path, name, fields)
2233 }
2234
2235 pub fn create_in(
2245 fs: &dyn Filesystem,
2246 path: impl AsRef<Path>,
2247 name: impl Into<String>,
2248 fields: Vec<Field>,
2249 ) -> Result<Self> {
2250 for field in &fields {
2251 type_tag(&field.ty)?;
2252 }
2253 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2254 let mut header = [0; HEADER as usize];
2255 header[..8].copy_from_slice(MAGIC);
2256 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2257 file.write_at(0, &header)?;
2258 Ok(Self {
2259 file,
2260 at: HEADER,
2261 written_back: HEADER,
2262 dictionaries: fields
2263 .iter()
2264 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2265 .collect(),
2266 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2267 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2268 lent: None,
2269 table: Table {
2270 name: name.into(),
2271 dictionaries: vec![None; fields.len()],
2272 dictionary_payloads: Vec::new(),
2273 demoted: Vec::new(),
2274 distincts: vec![None; fields.len()],
2275 fields,
2276 stripes: Vec::new(),
2277 rows: 0,
2278 frequencies: Vec::new(),
2279 pair_frequencies: Vec::new(),
2280 frequency_texts: Vec::new(),
2281 host_groups: None,
2282 clustering: None,
2283 generation: 1,
2284 sections: Vec::new(),
2285 },
2286 generation: 1,
2287 order: Vec::new(),
2288 next_order: 0,
2289 pending: Vec::with_capacity(STRIPE_PARTS),
2290 closed: Vec::new(),
2291 views: Vec::new(),
2292 profile: None,
2293 })
2294 }
2295
2296 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2318 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2319 let mut header = [0; HEADER as usize];
2320 header[..8].copy_from_slice(MAGIC);
2321 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2322 file.write_at(0, &header)?;
2323 let catalog = encode_catalog(&[], views)?;
2324 file.write_at(HEADER, &catalog)?;
2325 file.sync()?;
2329 let slot = Slot {
2330 offset: HEADER,
2331 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2332 generation: 1,
2333 hash: checksum(&catalog),
2334 };
2335 file.write_at(slot_offset(1), &slot.bytes())?;
2336 file.sync()?;
2337 Ok(())
2338 }
2339
2340 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2351 for field in &fields {
2352 type_tag(&field.ty)?;
2353 }
2354 let name = name.into();
2355 let entry = self.close()?;
2356 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2357 return Err(invalid("two tables in one native file have the same name"));
2358 }
2359 let Self { file, at, generation, mut closed, views, .. } = self;
2360 closed.push(entry);
2361 Ok(Self {
2362 file,
2363 written_back: at,
2364 at,
2365 generation,
2366 closed,
2367 views,
2368 profile: None,
2369 dictionaries: fields
2370 .iter()
2371 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2372 .collect(),
2373 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2374 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2375 lent: None,
2376 table: Table {
2377 name,
2378 dictionaries: vec![None; fields.len()],
2379 dictionary_payloads: Vec::new(),
2380 demoted: Vec::new(),
2381 distincts: vec![None; fields.len()],
2382 fields,
2383 stripes: Vec::new(),
2384 rows: 0,
2385 frequencies: Vec::new(),
2386 pair_frequencies: Vec::new(),
2387 frequency_texts: Vec::new(),
2388 host_groups: None,
2389 clustering: None,
2390 generation,
2391 sections: Vec::new(),
2392 },
2393 order: Vec::new(),
2394 next_order: 0,
2395 pending: Vec::with_capacity(STRIPE_PARTS),
2396 })
2397 }
2398
2399 #[must_use]
2409 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2410 self.views = views;
2411 self
2412 }
2413
2414 #[must_use]
2420 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2421 self.profile = Some(profile);
2422 self
2423 }
2424
2425 #[must_use]
2429 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2430 self.coded.cap(bytes);
2431 self
2432 }
2433
2434 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2449 self.table.clustering = Some(Clustering::new(
2452 clustering.columns().to_vec(),
2453 clustering.width(),
2454 &self.table.fields,
2455 )?);
2456 Ok(self)
2457 }
2458
2459 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2464 self.file.write_at(self.at, bytes)?;
2465 self.at = self
2466 .at
2467 .checked_add(bytes.len() as u64)
2468 .ok_or_else(|| invalid("native file length overflow"))?;
2469 if self.at - self.written_back >= WRITEBACK_STRETCH {
2470 self.file.start_writeback(self.written_back, self.at - self.written_back);
2471 self.written_back = self.at;
2472 }
2473 Ok(())
2474 }
2475
2476 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2482 let order = (self.next_order, 0);
2483 self.next_order = self.next_order.saturating_add(1);
2484 self.append_at(order, chunk)
2485 }
2486
2487 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2498 if chunk.is_empty() {
2499 return Ok(());
2500 }
2501 self.admit(chunk)?;
2502 if self.pending.last().is_some_and(|last| last.order > order) {
2503 self.flush_pending()?;
2504 }
2505 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2510 if self.pending.len() == STRIPE_PARTS {
2511 self.flush_pending()?;
2512 }
2513 Ok(())
2514 }
2515
2516 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2532 if parts.len() > STRIPE_PARTS {
2533 return Err(invalid("a stripe was handed more parts than it holds"));
2534 }
2535 self.flush_pending()?;
2538 for (order, chunk) in parts {
2539 if chunk.is_empty() {
2540 continue;
2541 }
2542 self.admit(&chunk)?;
2543 self.pending.push(PendingChunk { order, chunk });
2544 }
2545 self.flush_pending()
2546 }
2547
2548 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2550 if chunk.width() != self.table.fields.len() {
2551 return Err(invalid("chunk width differs from table schema"));
2552 }
2553 for (index, field) in self.table.fields.iter().enumerate() {
2554 if chunk.column(index)?.logical_type() != &field.ty {
2555 return Err(invalid("chunk type differs from table schema"));
2556 }
2557 }
2558 self.table.rows = self
2559 .table
2560 .rows
2561 .checked_add(chunk.len())
2562 .ok_or_else(|| invalid("row count overflow"))?;
2563 Ok(())
2564 }
2565
2566 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2568 let mut stripe = ColumnStripe {
2569 pages: Vec::with_capacity(columns.len()),
2570 codes: Vec::with_capacity(columns.len()),
2571 sieves: Vec::with_capacity(columns.len()),
2572 ranges: Vec::with_capacity(columns.len()),
2573 };
2574 let mut settling = Settling::default();
2575 for &column in columns {
2576 Self::encode_page(&mut stripe, &mut settling, column)?;
2577 }
2578 Ok(stripe)
2579 }
2580
2581 fn encode_page(
2584 stripe: &mut ColumnStripe,
2585 settling: &mut Settling,
2586 column: &Vector,
2587 ) -> Result<()> {
2588 let bytes = encode(column, settling)?;
2589 if bytes.len() > MAX_PAGE {
2590 return Err(invalid("column page exceeds the configured bound"));
2591 }
2592 let range = Range::of(column);
2595 let sieve =
2606 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2607 stripe.pages.push(bytes);
2608 stripe.codes.push(None);
2609 stripe.sieves.push(sieve);
2610 stripe.ranges.push(range);
2611 Ok(())
2612 }
2613
2614 fn place_blocks(&mut self) -> Result<()> {
2619 if let Some(lent) = self.lent.clone() {
2620 return self.place_lent_blocks(&lent);
2621 }
2622 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2623 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2624 for block in std::mem::take(&mut dictionary.blocks) {
2625 let start = self.at;
2626 self.put(&block)?;
2627 dictionary.placed.push(Placed {
2628 start,
2629 length: block.len() as u64,
2630 hash: checksum(&block),
2631 });
2632 }
2633 Ok(())
2634 });
2635 self.dictionaries = dictionaries;
2636 placed
2637 }
2638
2639 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2645 for column in lent.columns() {
2646 let Ok(mut held) = column.try_lock() else { continue };
2647 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2648 for block in std::mem::take(&mut dictionary.blocks) {
2649 let start = self.at;
2650 self.put(&block)?;
2651 dictionary.placed.push(Placed {
2652 start,
2653 length: block.len() as u64,
2654 hash: checksum(&block),
2655 });
2656 }
2657 }
2658 Ok(())
2659 }
2660
2661 fn reclaim(&mut self) -> Result<()> {
2665 let Some(lent) = self.lent.take() else { return Ok(()) };
2666 let (dictionaries, gathers) = lent.reclaim()?;
2667 self.dictionaries = dictionaries;
2668 self.gathers = gathers;
2669 Ok(())
2670 }
2671
2672 fn flush_pending(&mut self) -> Result<()> {
2677 if self.pending.is_empty() {
2678 return Ok(());
2679 }
2680 let held = std::mem::take(&mut self.pending);
2681 let prepared = self.preparer().prepare_held(held)?;
2682 let merged = self.merge_held(prepared)?;
2683 let paged = merged.pages()?;
2684 self.write_paged(paged)
2685 }
2686
2687 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2689 let width = self.table.fields.len();
2690 let parts = held.len();
2691 if encoded.len() != width {
2692 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2693 }
2694 let profile = self.profile.clone();
2695 if let Some(profile) = &profile {
2696 let rows = held.iter().map(|part| part.rows as u64).sum();
2697 let raw = held.iter().map(|part| part.footprint as u64).sum();
2698 let pages =
2699 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2700 profile.moved(Stage::Pages, raw, pages, rows);
2701 }
2702 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2705 let before = self.at;
2706 self.place_blocks()?;
2707 drop(timing);
2708 if let Some(profile) = &profile {
2709 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2710 }
2711 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2712 let before = self.at;
2713 let mut pages = Vec::with_capacity(width);
2714 let mut memberships = vec![None; width];
2715 let mut ranges = Vec::with_capacity(width);
2716 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2717 for stripe in &encoded {
2718 let offset = self.at;
2719 let section = index.len();
2720 let mut length = 0_usize;
2721 for bytes in &stripe.pages {
2722 self.file.write_at(self.at + length as u64, bytes)?;
2723 put_u32(
2724 &mut index,
2725 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2726 );
2727 put_u64(&mut index, checksum(bytes));
2728 length = length
2729 .checked_add(bytes.len())
2730 .ok_or_else(|| invalid("column page length overflow"))?;
2731 }
2732 let hash = checksum(&index[section..]);
2733 put_u64(&mut index, hash);
2734 if length > MAX_PAGE {
2735 return Err(invalid("column page exceeds the configured bound"));
2736 }
2737 self.at = self
2738 .at
2739 .checked_add(length as u64)
2740 .ok_or_else(|| invalid("native file length overflow"))?;
2741 pages.push(Span {
2742 offset,
2743 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2744 });
2745 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2746 }
2747 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2748 if stripe.codes.iter().all(Option::is_none) {
2749 continue;
2750 }
2751 let lists = stripe
2752 .codes
2753 .iter()
2754 .map(|codes| codes.clone().unwrap_or_default())
2755 .collect::<Vec<_>>();
2756 let bytes = encode_membership(&merged_codes(lists));
2757 let offset = self.at;
2758 self.put(&bytes)?;
2759 *membership = Some(Page {
2760 offset,
2761 length: u32::try_from(bytes.len())
2762 .map_err(|_| invalid("membership page length overflow"))?,
2763 hash: checksum(&bytes),
2764 });
2765 }
2766 let mut sieves = vec![None; width];
2767 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2768 if stripe.sieves.iter().all(Option::is_none) {
2769 continue;
2770 }
2771 let bytes = encode_sieves(stripe.sieves.iter())?;
2772 let offset = self.at;
2773 self.put(&bytes)?;
2774 *page = Some(Page {
2775 offset,
2776 length: u32::try_from(bytes.len())
2777 .map_err(|_| invalid("sieve page length overflow"))?,
2778 hash: checksum(&bytes),
2779 });
2780 }
2781 let mut part_ranges = vec![None; width];
2787 if parts > 1 {
2788 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2789 let bytes = encode_part_ranges(&stripe.ranges)?;
2790 if bytes.len() >= span.length as usize {
2791 continue;
2792 }
2793 let offset = self.at;
2794 self.put(&bytes)?;
2795 *page = Some(Page {
2796 offset,
2797 length: u32::try_from(bytes.len())
2798 .map_err(|_| invalid("part range page length overflow"))?,
2799 hash: checksum(&bytes),
2800 });
2801 }
2802 }
2803 let offset = self.at;
2804 self.put(&index)?;
2805 let index = Span {
2806 offset,
2807 length: u32::try_from(index.len())
2808 .map_err(|_| invalid("index page length overflow"))?,
2809 };
2810 let mut rows = 0_usize;
2811 let mut lengths = Vec::with_capacity(parts);
2812 let mut span = None;
2813 for part in held {
2814 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2815 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2816 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2817 }
2818 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2819 self.table.stripes.push(Stripe {
2820 rows,
2821 parts: lengths,
2822 index,
2823 pages,
2824 memberships: Pages::from_slots(memberships)?,
2825 sieves: Pages::from_slots(sieves)?,
2826 part_ranges: Pages::from_slots(part_ranges)?,
2827 zone: Zone::from_ranges(ranges),
2828 });
2829 drop(timing);
2830 if let Some(profile) = &profile {
2831 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2832 }
2833 Ok(())
2834 }
2835
2836 fn numeric_frequency(
2856 &self,
2857 column: usize,
2858 counted: bool,
2859 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2860 let signed = match self.table.fields[column].ty {
2861 LogicalType::TinyInt
2862 | LogicalType::SmallInt
2863 | LogicalType::Integer
2864 | LogicalType::BigInt
2865 | LogicalType::Date
2866 | LogicalType::Timestamp => true,
2867 LogicalType::UTinyInt
2868 | LogicalType::USmallInt
2869 | LogicalType::UInteger
2870 | LogicalType::UBigInt => false,
2871 _ => return Ok((None, None)),
2872 };
2873 let value_of = |bits: Option<u64>| match bits {
2874 None => FrequencyValue::Null,
2875 Some(bits) => integer_value(bits, signed),
2876 };
2877 let tallied = self
2882 .gathers
2883 .get(column)
2884 .and_then(Option::as_ref)
2885 .filter(|gather| gather.rows() == self.table.rows as u64)
2886 .and_then(stats::Gather::frequencies)
2887 .and_then(|(values, nulls)| {
2888 let entries = values
2889 .iter()
2890 .map(|(value, count)| {
2891 let value = value_of(Some(frequency_bits(value)?));
2892 Some(FrequencyEntry { value, count: *count })
2893 })
2894 .chain((nulls != 0).then_some(Some(FrequencyEntry {
2895 value: FrequencyValue::Null,
2896 count: nulls,
2897 })))
2898 .collect::<Option<Vec<_>>>()?;
2899 Some((entries, values.len() as u64))
2900 });
2901 let exact = match (&tallied, counted) {
2905 (None, true) => self.exact_frequency(column, signed)?,
2906 _ => None,
2907 };
2908 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
2909 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
2910 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
2911 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
2912 (None, None) => {
2913 let mut first = Candidates::default();
2917 let mut run = Run::default();
2918 self.visit_numeric(column, signed, |_, bits| {
2919 if let Some((ended, times)) = run.push(bits) {
2920 first.add(ended, times);
2921 }
2922 })?;
2923 if let Some((bits, times)) = run.take() {
2924 first.add(bits, times);
2925 }
2926 let (nulls, decrements) = (first.nulls, first.decrements);
2929 let distinct_count = (decrements == 0).then_some(first.held as u64);
2930 let (exact, null_count) = if decrements == 0 {
2931 let exact = first
2932 .pairs()
2933 .map(|(bits, count)| (bits, u64::from(count)))
2934 .collect::<FrequencyMap<_>>();
2935 (exact, (nulls != 0).then_some(u64::from(nulls)))
2936 } else {
2937 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2938 if nulls != 0 {
2939 lower.push(nulls);
2940 }
2941 lower.sort_unstable_by(|left, right| right.cmp(left));
2942 if lower.len() < FREQUENCY_BUILD_RANK
2943 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2944 {
2945 return Ok((None, distinct_count));
2946 }
2947 let mut recounts = vec![0_u64; first.slots.len()];
2950 let mut null_count = (nulls != 0).then_some(0_u64);
2951 let mut recount = |bits: Option<u64>, times: u32| {
2952 let held = match bits {
2953 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2954 None => null_count.as_mut(),
2955 };
2956 if let Some(count) = held {
2957 *count = count.saturating_add(u64::from(times));
2958 }
2959 };
2960 let mut run = Run::default();
2961 self.visit_numeric(column, signed, |_, bits| {
2962 if let Some((bits, times)) = run.push(bits) {
2963 recount(bits, times);
2964 }
2965 })?;
2966 if let Some((bits, times)) = run.take() {
2967 recount(bits, times);
2968 }
2969 let exact = first
2970 .slots
2971 .iter()
2972 .zip(&recounts)
2973 .filter(|(slot, _)| slot.count != 0)
2974 .map(|(slot, &count)| (slot.bits, count))
2975 .collect::<FrequencyMap<_>>();
2976 (exact, null_count)
2977 };
2978 let entries = exact
2979 .into_iter()
2980 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2981 .chain(
2982 null_count
2983 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
2984 )
2985 .collect::<Vec<_>>();
2986 (entries, decrements, distinct_count)
2987 }
2988 };
2989 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
2990 if omitted_max == 0 && entries.len() > 1 {
2994 let retained = entries.len().saturating_sub(1).min(2);
2995 omitted_max = entries[retained].count;
2996 entries.truncate(retained);
2997 }
2998 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2999 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3000 });
3001 let mut ordinals = Vec::new();
3002 let mut ordinal_entries = Vec::new();
3003 if let Some(kept_rows) = kept_rows {
3004 let mut kept = FrequencyMap::default();
3005 let mut null_kept = None;
3006 for (at, entry) in entries.iter().enumerate() {
3007 let at = u16::try_from(at)
3008 .map_err(|_| invalid("too many retained frequency entries"))?;
3009 match entry.value {
3010 FrequencyValue::Integer(value) => {
3011 kept.insert(value as u64, at);
3012 }
3013 FrequencyValue::Null => null_kept = Some(at),
3014 FrequencyValue::Code(_) => {}
3015 }
3016 }
3017 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3018 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3019 self.visit_numeric(column, signed, |ordinal, bits| {
3020 let held = match bits {
3021 Some(bits) => kept.get(&bits).copied(),
3022 None => null_kept,
3023 };
3024 if let Some(entry) = held {
3025 ordinals.push(ordinal);
3026 ordinal_entries.push(entry);
3027 }
3028 })?;
3029 }
3030 Ok((
3031 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3032 distinct_count,
3033 ))
3034 }
3035
3036 fn exact_frequency(
3050 &self,
3051 column: usize,
3052 signed: bool,
3053 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3054 let mut set = distinct::ExactCounts::new();
3055 let mut nulls = 0_u64;
3056 let mut run = Run::default();
3057 let mut add = |bits: Option<u64>, times: u32| match bits {
3058 Some(bits) => set.insert(bits, times),
3059 None => nulls += u64::from(times),
3060 };
3061 self.visit_numeric(column, signed, |_, bits| {
3062 if let Some((bits, times)) = run.push(bits) {
3063 add(bits, times);
3064 }
3065 })?;
3066 if let Some((bits, times)) = run.take() {
3067 add(bits, times);
3068 }
3069 let Some(distinct) = set.count() else {
3070 return Ok(None);
3071 };
3072 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3074 let mut rank = |count: u64| {
3075 if top.len() <= FREQUENCY_ENTRIES {
3076 top.push(Reverse(count));
3077 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3078 top.pop();
3079 top.push(Reverse(count));
3080 }
3081 };
3082 set.visit(|_, count| rank(count));
3083 if nulls != 0 {
3084 rank(nulls);
3085 }
3086 let top = top.into_sorted_vec();
3087 let values = distinct + u64::from(nulls != 0);
3088 if values > FREQUENCY_CANDIDATES as u64 {
3089 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3090 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3091 return Ok(Some((None, distinct)));
3092 }
3093 }
3094 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3095 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3096 set.visit(|bits, count| {
3097 if count >= least {
3098 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3099 }
3100 });
3101 if nulls != 0 && nulls >= least {
3102 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3103 }
3104 Ok(Some((Some(entries), distinct)))
3105 }
3106
3107 fn visit_numeric(
3114 &self,
3115 column: usize,
3116 signed: bool,
3117 mut visit: impl FnMut(u64, Option<u64>),
3118 ) -> Result<()> {
3119 let ty = &self.table.fields[column].ty;
3120 let mut start = 0_u64;
3121 let mut block = Vec::new();
3122 for stripe in &self.table.stripes {
3123 let spans = read_index(&self.file, stripe, column)?;
3124 let page = stripe.pages[column];
3125 let mut bytes = vec![0; page.length as usize];
3126 read_at(&self.file, page.offset, &mut bytes)?;
3127 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3128 let part = part_bytes(&bytes, *span)?;
3129 if checksum(part) != span.hash {
3130 return Err(invalid("column page checksum differs while building frequencies"));
3131 }
3132 let rows = rows as usize;
3133 let vector = decode(ty, rows, part, None)?;
3134 if signed && vector.signed_block(&mut block) && block.len() == rows {
3138 if vector.none_null() {
3139 for (row, &value) in block.iter().enumerate() {
3140 visit(start.saturating_add(row as u64), Some(value as u64));
3141 }
3142 } else {
3143 for (row, &value) in block.iter().enumerate() {
3144 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3145 visit(start.saturating_add(row as u64), bits);
3146 }
3147 }
3148 start = start.saturating_add(rows as u64);
3149 continue;
3150 }
3151 for row in 0..rows {
3153 let bits = if vector.is_null_at(row) {
3154 None
3155 } else {
3156 let widened = match vector.signed_at(row) {
3160 Some(value) => Some(value as u64),
3161 None => match vector.value_at(row) {
3162 Value::UTinyInt(value) => Some(u64::from(value)),
3163 Value::USmallInt(value) => Some(u64::from(value)),
3164 Value::UInteger(value) => Some(u64::from(value)),
3165 Value::UBigInt(value) => Some(value),
3166 _ => None,
3167 },
3168 };
3169 Some(widened.ok_or_else(|| {
3170 invalid("numeric frequency page did not contain an integer value")
3171 })?)
3172 };
3173 visit(start.saturating_add(row as u64), bits);
3174 }
3175 start = start.saturating_add(rows as u64);
3176 }
3177 }
3178 Ok(())
3179 }
3180
3181 fn numeric_columns(&self) -> Vec<usize> {
3183 self.table
3184 .fields
3185 .iter()
3186 .enumerate()
3187 .filter_map(|(column, field)| {
3188 matches!(
3189 field.ty,
3190 LogicalType::TinyInt
3191 | LogicalType::SmallInt
3192 | LogicalType::Integer
3193 | LogicalType::BigInt
3194 | LogicalType::UTinyInt
3195 | LogicalType::USmallInt
3196 | LogicalType::UInteger
3197 | LogicalType::UBigInt
3198 | LogicalType::Date
3199 | LogicalType::Timestamp
3200 )
3201 .then_some(column)
3202 })
3203 .collect()
3204 }
3205
3206 #[allow(dead_code)]
3208 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3209 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3210 return Ok(None);
3211 }
3212 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3213 return Err(invalid("frequency ordinals are not sorted and unique"));
3214 }
3215 let mut out = Vec::with_capacity(ordinals.len());
3216 let mut wanted = 0;
3217 let mut stripe_start = 0_u64;
3218 for stripe in &self.table.stripes {
3219 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3220 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3221 stripe_start = stripe_end;
3222 continue;
3223 }
3224 let spans = read_index(&self.file, stripe, column)?;
3225 let page = stripe.pages[column];
3226 let mut bytes = vec![0; page.length as usize];
3227 read_at(&self.file, page.offset, &mut bytes)?;
3228 let mut part_start = stripe_start;
3229 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3230 let part_end = part_start.saturating_add(u64::from(rows));
3231 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3232 let part = part_bytes(&bytes, *span)?;
3233 if checksum(part) != span.hash {
3234 return Err(invalid(
3235 "column page checksum differs while building pair frequencies",
3236 ));
3237 }
3238 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3239 let positions = ordinals[wanted..upto]
3240 .iter()
3241 .map(|&ordinal| {
3242 usize::try_from(ordinal.saturating_sub(part_start))
3243 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3244 })
3245 .collect::<Result<Vec<_>>>()?;
3246 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3247 return Ok(None);
3248 }
3249 wanted = upto;
3250 }
3251 part_start = part_end;
3252 }
3253 stripe_start = stripe_end;
3254 }
3255 if wanted != ordinals.len() {
3256 return Err(invalid("frequency ordinal is outside the table"));
3257 }
3258 Ok(Some(out))
3259 }
3260
3261 #[allow(dead_code)]
3263 fn pair_frequencies(
3264 &self,
3265 frequencies: &[Option<Frequencies>],
3266 ) -> Result<Vec<PairFrequencySummary>> {
3267 let anchors = frequencies
3268 .iter()
3269 .enumerate()
3270 .filter_map(|(column, summary)| {
3271 match summary {
3273 Some(Frequencies::Held(summary)) => Some(summary),
3274 _ => None,
3275 }
3276 .filter(|summary| {
3277 !summary.ordinals.is_empty()
3278 && summary.ordinal_entries.len() == summary.ordinals.len()
3279 })
3280 .cloned()
3281 .map(|summary| (column, summary))
3282 })
3283 .collect::<Vec<_>>();
3284 let strings = self
3285 .dictionaries
3286 .iter()
3287 .enumerate()
3288 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3289 .collect::<Vec<_>>();
3290 let mut summaries = Vec::new();
3291 for (first, anchors) in anchors {
3292 for &second in &strings {
3293 if summaries.len() == MAX_PAIR_FREQUENCIES {
3294 return Ok(summaries);
3295 }
3296 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3297 continue;
3298 };
3299 if codes.len() != anchors.ordinal_entries.len() {
3300 return Err(invalid("pair frequency columns have different lengths"));
3301 }
3302 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3303 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3304 *counts.entry((anchor, code)).or_default() += 1;
3305 }
3306 let mut entries = counts
3307 .into_iter()
3308 .map(|((first_entry, second), count)| PairFrequencyEntry {
3309 first_entry,
3310 second,
3311 count,
3312 })
3313 .collect::<Vec<_>>();
3314 entries.sort_unstable_by(|left, right| {
3315 right
3316 .count
3317 .cmp(&left.count)
3318 .then_with(|| left.first_entry.cmp(&right.first_entry))
3319 .then_with(|| left.second.cmp(&right.second))
3320 });
3321 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3322 entries.truncate(FREQUENCY_ENTRIES);
3323 summaries.push(PairFrequencySummary {
3324 first: u16::try_from(first)
3325 .map_err(|_| invalid("pair frequency column index overflows"))?,
3326 second: u16::try_from(second)
3327 .map_err(|_| invalid("pair frequency column index overflows"))?,
3328 entries,
3329 omitted_max: anchors.omitted_max.max(pair_omitted),
3330 });
3331 }
3332 }
3333 Ok(summaries)
3334 }
3335
3336 fn close(&mut self) -> Result<Entry> {
3347 self.reclaim()?;
3348 self.flush_pending()?;
3349 let profile = self.profile.clone();
3353 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3354 let before = self.at;
3355 let mut stripes = std::mem::take(&mut self.order)
3356 .into_iter()
3357 .zip(std::mem::take(&mut self.table.stripes))
3358 .collect::<Vec<_>>();
3359 stripes.sort_by_key(|(order, _)| order.0);
3360 let mut previous: Option<(u64, u64)> = None;
3361 for ((first, last), _) in &stripes {
3362 if previous.is_some_and(|previous| previous >= *first) {
3363 return Err(invalid("chunks did not arrive in source order"));
3364 }
3365 previous = Some(*last);
3366 }
3367 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3368 drop(timing);
3369 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3370 let placing = self.at;
3371 finish_dictionaries(&mut self.dictionaries)?;
3372 self.place_blocks()?;
3373 for dictionary in self.dictionaries.iter_mut().flatten() {
3374 dictionary.release_lookup();
3375 dictionary.recharge(profile.as_deref());
3376 }
3377 let (numeric, closed) = self.close_columns()?;
3378 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3379 numeric.into_iter().unzip();
3380 let frequencies =
3381 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3382 let pairs = Vec::new();
3384 self.table.frequencies = frequencies;
3385 self.table.distincts = distincts;
3386 self.table.pair_frequencies = pairs;
3387 if let Some(profile) = &profile {
3388 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3389 }
3390 self.table.demoted = self
3391 .dictionaries
3392 .iter()
3393 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3394 .collect();
3395 if !self.table.demoted.contains(&true) {
3396 self.table.demoted = Vec::new();
3397 }
3398 self.dictionaries = Vec::new();
3399 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3400 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3401 self.table.host_groups = None;
3402 for (index, closed) in closed.into_iter().enumerate() {
3403 let Some(closed) = closed else { continue };
3404 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3405 self.table.distincts[index] = distinct;
3406 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3407 self.table.frequency_texts[index] = texts;
3408 if hosts.is_some() {
3409 self.table.host_groups = hosts;
3410 }
3411 let offset = self.at;
3412 self.put(&encoded.index)?;
3413 self.put(&encoded.ranks)?;
3414 self.put(&encoded.grams)?;
3415 self.table.dictionary_payloads[index] = payload;
3416 let length = encoded
3417 .index
3418 .len()
3419 .checked_add(encoded.ranks.len())
3420 .and_then(|len| len.checked_add(encoded.grams.len()))
3421 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3422 self.table.dictionaries[index] = Some(Page {
3423 offset,
3424 length: u32::try_from(length)
3425 .map_err(|_| invalid("dictionary page length overflow"))?,
3426 hash: checksum(&encoded.index),
3427 });
3428 }
3429 drop(timing);
3430 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3431 let placed = self.at - placing;
3432 self.write_stats()?;
3433 let directory = encode_directory(&self.table)?;
3434 if directory.len() > MAX_DIRECTORY {
3435 return Err(invalid("directory exceeds the configured bound"));
3436 }
3437 let offset = self.at;
3438 self.put(&directory)?;
3439 drop(timing);
3440 if let Some(profile) = &profile {
3441 profile.moved(Stage::Dictionary, 0, placed, 0);
3442 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3443 }
3444 Ok(Entry {
3445 name: self.table.name.clone(),
3446 fields: self.table.fields.clone(),
3447 rows: self.table.rows,
3448 nonzero: vec![None; self.table.fields.len()],
3449 aggregates: table_aggregate_sums(&self.table),
3450 distincts: self.table.distincts.clone(),
3451 extremes: table_integer_extremes(&self.table),
3452 frequencies: table_complete_numeric_frequencies(&self.table),
3453 directory: Page {
3454 offset,
3455 length: u32::try_from(directory.len())
3456 .map_err(|_| invalid("directory length overflow"))?,
3457 hash: checksum(&directory),
3458 },
3459 })
3460 }
3461
3462 #[allow(clippy::type_complexity)]
3479 fn close_columns(
3480 &self,
3481 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3482 let numeric = self.numeric_columns().into_iter().map(|column| {
3483 let estimate =
3484 self.gathers.get(column).and_then(Option::as_ref).and_then(stats::Gather::distinct);
3485 let counted = !estimate.is_some_and(distinct::beyond);
3486 let set =
3487 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3488 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3489 (Closing::Numeric { column, counted }, NUMERIC_CLOSE_BYTES + set, cost)
3490 });
3491 let dictionaries =
3492 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3493 let dictionary = dictionary.as_ref()?;
3494 let bytes = dictionary.closing_bytes();
3495 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3496 });
3497 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3498 jobs.sort_by_key(|&(_, _, cost)| cost);
3499 let columns = self.table.fields.len();
3500 let mut frequencies = vec![(None, None); columns];
3501 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3502 let profile = self.profile.as_deref();
3503 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3504 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3505 match job {
3506 Closing::Numeric { column, counted } => {
3507 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3508 Ok(Closed::Numeric(column, self.numeric_frequency(column, counted)?))
3509 }
3510 Closing::Dictionary { index, dictionary } => {
3511 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3512 Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3513 }
3514 }
3515 };
3516 let workers = close_workers().min(jobs.len());
3517 let pieces = if workers <= 1 {
3518 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3519 } else {
3520 let state = Mutex::new((jobs, 0_usize));
3522 let finished = Condvar::new();
3523 std::thread::scope(|scope| {
3524 (0..workers)
3525 .map(|_| {
3526 scope.spawn(|| {
3527 let mut mine = Vec::new();
3528 loop {
3529 let mut held = state.lock().map_err(|_| {
3530 Error::internal("a native close worker panicked")
3531 })?;
3532 let (job, bytes) = loop {
3533 let (jobs, busy) = &mut *held;
3534 if jobs.is_empty() {
3535 return Ok(mine);
3536 }
3537 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3538 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3539 });
3540 if let Some(at) = fits {
3541 let (job, bytes, _) = jobs.remove(at);
3542 *busy += bytes;
3543 break (job, bytes);
3544 }
3545 held = finished.wait(held).map_err(|_| {
3546 Error::internal("a native close worker panicked")
3547 })?;
3548 };
3549 drop(held);
3550 let _room = Room { state: &state, finished: &finished, bytes };
3553 mine.push(run(job, bytes)?);
3554 }
3555 })
3556 })
3557 .collect::<Vec<_>>()
3558 .into_iter()
3559 .map(|handle| {
3560 handle
3561 .join()
3562 .map_err(|_| Error::internal("a native close worker panicked"))?
3563 })
3564 .collect::<Result<Vec<_>>>()
3565 })?
3566 .into_iter()
3567 .flatten()
3568 .collect()
3569 };
3570 for piece in pieces {
3571 match piece {
3572 Closed::Numeric(column, summary) => frequencies[column] = summary,
3573 Closed::Dictionary(index, one) => closed[index] = Some(one),
3574 }
3575 }
3576 Ok((frequencies, closed))
3577 }
3578
3579 fn close_dictionary(
3586 &self,
3587 _index: usize,
3588 dictionary: &GlobalDictionary,
3589 ) -> Result<ClosedDictionary> {
3590 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3591 let (distinct, frequencies, texts) = if dictionary.demoted {
3596 (None, None, Vec::new())
3597 } else {
3598 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3599 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3600 (Some(distinct), Some(frequencies), texts)
3601 };
3602 let hosts = None;
3604 drop(flat);
3605 drop(bases);
3606 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3607 let payload = dictionary
3608 .placed
3609 .iter()
3610 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3611 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3612 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3613 }
3614
3615 fn write_stats(&mut self) -> Result<()> {
3627 let gathers = std::mem::take(&mut self.gathers);
3628 let rows = self.table.rows as u64;
3629 let mut payloads = Vec::new();
3630 for (column, gather) in gathers.into_iter().enumerate() {
3631 let Some(gather) = gather else { continue };
3632 if gather.rows() != rows {
3638 continue;
3639 }
3640 let Some(stats) = gather.finish() else { continue };
3641 let mut summary = Vec::new();
3642 stats.summary.encode(&mut summary)?;
3643 let mut sketches = Vec::new();
3644 stats.sketches.encode(&mut sketches)?;
3645 payloads.push((column, summary, sketches));
3646 }
3647 if payloads.is_empty() {
3648 return Ok(());
3649 }
3650 let costs = payloads
3651 .iter()
3652 .map(|(_, summary, sketches)| summary.len() + sketches.len())
3653 .collect::<Vec<_>>();
3654 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3655 let keep = stats::within(&costs, allowance, 0);
3658 for ((column, summary, sketches), _) in
3659 payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3660 {
3661 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3662 for (kind, bytes, header_bytes) in [
3663 (*section::SUMMARY, summary, summary.len() as u32),
3666 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3667 ] {
3668 let written = write_section(
3669 &*self.file,
3670 &mut self.at,
3671 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3672 self.generation,
3673 )?;
3674 self.table.sections.push(written);
3675 }
3676 }
3677 if self.table.sections.len() > MAX_SECTIONS {
3678 return Err(invalid("the table would name more sections than the bound allows"));
3679 }
3680 Ok(())
3681 }
3682
3683 pub fn finish(mut self) -> Result<Table> {
3693 let entry = self.close()?;
3694 let profile = self.profile.take();
3695 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3696 let mut tables = std::mem::take(&mut self.closed);
3697 tables.push(entry);
3698 let catalog = encode_catalog(&tables, &self.views)?;
3699 if catalog.len() > MAX_DIRECTORY {
3700 return Err(invalid("catalog exceeds the configured bound"));
3701 }
3702 let offset = self.at;
3703 self.put(&catalog)?;
3704 if let Some(profile) = &profile {
3705 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3706 }
3707 synced(&*self.file, profile.as_deref())?;
3711 let slot = Slot {
3712 offset,
3713 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3714 generation: self.generation,
3715 hash: checksum(&catalog),
3716 };
3717 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3722 synced(&*self.file, profile.as_deref())?;
3723 Ok(self.table)
3724 }
3725
3726 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3743 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3744 let size = file.len()?;
3745 let (slot, bytes, _) = committed_slot(&*file, size)?;
3746 let (closed, _) = decode_catalog(&bytes, size)?;
3747 let generation = slot
3748 .generation
3749 .checked_add(1)
3750 .ok_or_else(|| invalid("native file generation overflow"))?;
3751 let catalog = encode_catalog(&closed, views)?;
3752 if catalog.len() > MAX_DIRECTORY {
3753 return Err(invalid("catalog exceeds the configured bound"));
3754 }
3755 file.write_at(size, &catalog)?;
3756 file.sync()?;
3757 let slot = Slot {
3758 offset: size,
3759 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3760 generation,
3761 hash: checksum(&catalog),
3762 };
3763 file.write_at(slot_offset(generation), &slot.bytes())?;
3764 file.sync()?;
3765 Ok(())
3766 }
3767
3768 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3771 let path = path.as_ref();
3772 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3773 let (mut entries, views) = decode_catalog(&bytes, size)?;
3774 let native = Catalog::open(path)?;
3775 for entry in &mut entries {
3776 let reader = native.table(&entry.name)?;
3777 entry.nonzero.fill(None);
3778 entry.aggregates = reader_aggregate_sums(&reader)?;
3779 entry.distincts = (0..entry.fields.len())
3780 .map(|column| reader.distinct_values(column))
3781 .collect::<Result<Vec<_>>>()?;
3782 entry.extremes = reader_integer_extremes(&reader)?;
3783 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3784 }
3785 let generation = slot
3786 .generation
3787 .checked_add(1)
3788 .ok_or_else(|| invalid("native file generation overflow"))?;
3789 let catalog = encode_catalog(&entries, &views)?;
3790 if catalog.len() > MAX_DIRECTORY {
3791 return Err(invalid("catalog exceeds the configured bound"));
3792 }
3793 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3794 file.write_at(size, &catalog)?;
3795 file.sync()?;
3796 let slot = Slot {
3797 offset: size,
3798 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3799 generation,
3800 hash: checksum(&catalog),
3801 };
3802 file.write_at(slot_offset(generation), &slot.bytes())?;
3803 file.sync()?;
3804 Ok(())
3805 }
3806
3807 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3809 Self::certify_summaries(path)
3810 }
3811}
3812
3813fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3819 let offset = *at;
3820 file.write_at(offset, bytes)?;
3821 *at =
3822 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3823 Ok(offset)
3824}
3825
3826fn write_section(
3832 file: &dyn rudb_io::File,
3833 at: &mut u64,
3834 one: §ion::Attachment<'_>,
3835 generation: u64,
3836) -> Result<Section> {
3837 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3841 return Err(invalid("a section's header is longer than its payload"));
3842 }
3843 let mut extents = Vec::new();
3844 let mut first = 0_u64;
3845 let extent_size =
3846 if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
3847 1 << 19
3848 } else {
3849 section::MAX_EXTENT as usize
3850 };
3851 for chunk in one.bytes.chunks(extent_size) {
3852 let offset = append(file, at, chunk)?;
3853 extents.push(section::Extent {
3854 offset,
3855 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3856 hash: checksum(chunk),
3857 first,
3858 });
3859 first += chunk.len() as u64;
3860 }
3861 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3862 section::encode_extents(&extents, &mut table)?;
3863 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3867 Ok(Section {
3868 kind: one.kind,
3869 id: one.id,
3870 generation,
3871 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3872 extent_page,
3873 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3874 hash: checksum(&table),
3875 flags: one.flags,
3876 header_bytes: one.header_bytes,
3877 })
3878}
3879
3880pub fn attach(
3904 path: impl AsRef<Path>,
3905 table: &str,
3906 attachments: &[section::Attachment<'_>],
3907) -> Result<Table> {
3908 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3909 let file = &*file;
3910 let size = file.len()?;
3911 let (slot, bytes, _) = committed_slot(file, size)?;
3912 let (mut entries, views) = decode_catalog(&bytes, size)?;
3913 let at = entries
3914 .iter()
3915 .position(|entry| entry.name == table)
3916 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3917 let mut version = [0; 4];
3918 read_at(file, 8, &mut version)?;
3919 let version = u32::from_le_bytes(version);
3920 if version != FORMAT {
3926 return Err(invalid(&format!(
3927 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3928 to be written again"
3929 )));
3930 }
3931 let mut directory = vec![0; entries[at].directory.length as usize];
3932 read_at(file, entries[at].directory.offset, &mut directory)?;
3933 if checksum(&directory) != entries[at].directory.hash {
3934 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3935 }
3936 let mut held = decode_directory(&directory, size)?;
3937 let mut cursor = size;
3938 for one in attachments {
3939 let written = write_section(file, &mut cursor, one, held.generation)?;
3940 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3941 held.sections.push(written);
3942 }
3943 if held.sections.len() > MAX_SECTIONS {
3944 return Err(invalid("the table would name more sections than the bound allows"));
3945 }
3946 let encoded = encode_directory(&held)?;
3947 if encoded.len() > MAX_DIRECTORY {
3948 return Err(invalid("directory exceeds the configured bound"));
3949 }
3950 let offset = append(file, &mut cursor, &encoded)?;
3951 entries[at].directory = Page {
3952 offset,
3953 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3954 hash: checksum(&encoded),
3955 };
3956 let catalog = encode_catalog(&entries, &views)?;
3959 if catalog.len() > MAX_DIRECTORY {
3960 return Err(invalid("catalog exceeds the configured bound"));
3961 }
3962 let offset = append(file, &mut cursor, &catalog)?;
3963 file.sync()?;
3964 let generation =
3965 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3966 let committed = Slot {
3967 offset,
3968 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3969 generation,
3970 hash: checksum(&catalog),
3971 };
3972 file.write_at(slot_offset(generation), &committed.bytes())?;
3973 file.sync()?;
3974 Ok(held)
3975}
3976
3977type Synopsis = Arc<Vec<(Value, u64)>>;
3980
3981#[derive(Debug, Clone)]
3983pub struct Reader {
3984 file: Arc<File>,
3985 table: Arc<Table>,
3986 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3987 loading: Arc<Vec<Mutex<()>>>,
3996 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3999 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4003 opened: Arc<AtomicUsize>,
4007 sieves: Arc<Vec<Vec<SieveSlot>>>,
4011 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4014 places: Arc<Vec<Place>>,
4016 cache: Arc<Shelf>,
4017 pool: PagePool,
4019 pages: Arc<AtomicUsize>,
4022 indexes: Arc<AtomicUsize>,
4025 size: u64,
4027 directory: u64,
4029 opening: Opening,
4031}
4032
4033#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4045pub struct Opening {
4046 pub reads: u32,
4049 pub bytes: u64,
4051}
4052
4053#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4055pub struct Reads {
4056 pub opening: Opening,
4058 pub pages: usize,
4060 pub indexes: usize,
4062 pub dictionaries: usize,
4065}
4066
4067#[derive(Debug, Clone, Copy)]
4069struct Place {
4070 stripe: u32,
4071 part: u32,
4072 rows: u32,
4073}
4074
4075#[derive(Debug, Clone, Copy)]
4077struct PartSpan {
4078 start: usize,
4079 length: usize,
4080 hash: u64,
4081}
4082
4083#[derive(Debug, Clone)]
4089struct CachedColumn {
4090 stripe: usize,
4091 index: Arc<Vec<PartSpan>>,
4092 page: Option<Arc<HeldPage>>,
4093}
4094
4095#[derive(Debug)]
4102struct HeldPage {
4103 bytes: Vec<u8>,
4104 checked: Vec<AtomicBool>,
4105}
4106
4107impl HeldPage {
4108 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4110 let bytes = part_bytes(&self.bytes, span)?;
4111 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4112 if !checked.load(Atomic::Relaxed) {
4113 verify_part(bytes, span)?;
4114 checked.store(true, Atomic::Relaxed);
4115 }
4116 Ok(bytes)
4117 }
4118}
4119
4120fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4122 let got = checksum(bytes);
4123 if got != span.hash {
4124 return Err(invalid(&format!(
4125 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4126 span.start, span.length, span.hash,
4127 )));
4128 }
4129 Ok(())
4130}
4131
4132#[derive(Debug, Default)]
4161struct Cached {
4162 pages: Vec<Option<Resident>>,
4163 loading: Vec<usize>,
4164 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4165 seen: Vec<bool>,
4166 passing: VecDeque<usize>,
4167}
4168
4169#[derive(Debug, Clone)]
4171struct Resident {
4172 page: Arc<HeldPage>,
4173 used: Arc<AtomicBool>,
4174}
4175
4176#[derive(Debug)]
4178struct Shelf {
4179 columns: Vec<Mutex<Cached>>,
4180 held: Vec<AtomicUsize>,
4183 kept: AtomicUsize,
4186}
4187
4188#[derive(Debug, Clone, Default)]
4207pub struct PagePool {
4208 ring: Arc<Mutex<Ring>>,
4209 budget: Arc<AtomicUsize>,
4210}
4211
4212#[derive(Debug, Default)]
4213struct Ring {
4214 held: VecDeque<Held>,
4215 bytes: usize,
4216}
4217
4218#[derive(Debug)]
4223struct Held {
4224 shelf: Weak<Shelf>,
4225 column: usize,
4226 stripe: usize,
4227 bytes: usize,
4228 used: Arc<AtomicBool>,
4229}
4230
4231impl PagePool {
4232 #[must_use]
4234 pub fn new(budget: usize) -> Self {
4235 let pool = Self::default();
4236 pool.budget.store(budget, Atomic::Relaxed);
4237 pool
4238 }
4239
4240 #[must_use]
4246 pub fn bytes(&self) -> usize {
4247 self.ring.lock().map_or(0, |ring| ring.bytes)
4248 }
4249
4250 fn admit(&self, held: Held) {
4256 let budget = self.budget.load(Atomic::Relaxed);
4257 let mut gone = Vec::new();
4258 {
4259 let Ok(mut ring) = self.ring.lock() else { return };
4260 ring.bytes += held.bytes;
4261 ring.held.push_back(held);
4262 let mut looked = 0;
4265 let limit = ring.held.len();
4266 while ring.bytes > budget && looked < limit {
4267 looked += 1;
4268 let Some(entry) = ring.held.pop_front() else { break };
4269 let Some(shelf) = entry.shelf.upgrade() else {
4270 ring.bytes -= entry.bytes;
4271 continue;
4272 };
4273 if entry.used.swap(false, Atomic::Relaxed) {
4274 ring.held.push_back(entry);
4275 continue;
4276 }
4277 let count = &shelf.held[entry.column];
4278 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4279 ring.held.push_back(entry);
4280 continue;
4281 }
4282 count.fetch_sub(1, Atomic::Relaxed);
4283 ring.bytes -= entry.bytes;
4284 gone.push((shelf, entry));
4285 }
4286 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4289 if let Some(entry) = ring.held.pop_front() {
4290 ring.bytes -= entry.bytes;
4291 }
4292 }
4293 }
4294 for (shelf, entry) in gone {
4295 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4296 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4297 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4298 *slot = None;
4299 }
4300 }
4301 }
4302 }
4303}
4304
4305const CACHED_STRIPES_PER_COLUMN: usize = 4;
4317
4318type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4320
4321type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4322
4323#[derive(Debug)]
4324struct NativeText {
4325 file: Arc<File>,
4326 values: usize,
4328 offsets: Vec<u8>,
4340 offset_bits: usize,
4343 value_ends: OnceLock<Option<Vec<u32>>>,
4356 value_lens: OnceLock<Option<Lengths>>,
4366 ends_asked: AtomicUsize,
4372 ranks: usize,
4374 rank_at: u64,
4378 rank_ends: Vec<u64>,
4382 rank_hashes: Vec<u64>,
4383 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4384 code_bits: usize,
4387 code_ranks: OnceLock<Option<Vec<u32>>>,
4394 starts: Vec<u64>,
4401 lengths: Vec<u64>,
4402 hashes: Vec<u64>,
4403 grams: Option<NativeGrams>,
4405 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4407 char_lens: Vec<OnceLock<Box<[u32]>>>,
4416 keep_budget: usize,
4419 payload_kept: AtomicUsize,
4427 swept: Vec<AtomicBool>,
4435 visit_dropped: AtomicUsize,
4450 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4467}
4468
4469#[derive(Debug)]
4470struct NativeGrams {
4471 start: u64,
4472 length: usize,
4473 width: usize,
4475 hash: u64,
4476 verdicts: Mutex<Vec<Verdict>>,
4483}
4484
4485type Verdict = (Vec<u8>, Arc<[bool]>);
4487
4488const GRAM_VERDICTS: usize = 8;
4490
4491impl NativeGrams {
4492 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4497 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4498 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4499 return Ok(Arc::clone(verdict));
4500 }
4501 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4502 let mut verdict = Vec::with_capacity(self.length / self.width);
4503 let window = GRAM_WINDOW / self.width * self.width;
4504 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4505 verdict.extend(bytes.chunks(self.width).map(|bits| {
4506 wanted
4507 .iter()
4508 .flatten()
4509 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4510 }));
4511 Ok(())
4512 })?;
4513 if hash != self.hash {
4514 return Err(invalid("global dictionary substring signatures checksum differs"));
4515 }
4516 let verdict: Arc<[bool]> = verdict.into();
4517 if held.len() >= GRAM_VERDICTS {
4518 held.remove(0);
4519 }
4520 held.push((literal.to_vec(), Arc::clone(&verdict)));
4521 Ok(verdict)
4522 }
4523
4524 fn footprint(&self) -> usize {
4525 self.verdicts.lock().map_or(0, |held| {
4526 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4527 })
4528 }
4529}
4530
4531const TEXT_SEARCH_MEMO: usize = 64;
4536
4537const TEXT_PAYLOAD_VALUES: usize = 1024;
4553
4554const TEXT_GRAM_BYTES: usize = 8192;
4565
4566const NARROW_GRAM_BYTES: usize = 2048;
4568
4569const GRAM_WINDOW: usize = 256 << 10;
4571
4572fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4575 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4576 let mut first = original ^ (original >> 16);
4577 first = first.wrapping_mul(0x7feb_352d);
4578 first ^= first >> 15;
4579 let mut second = original ^ (original >> 17);
4580 second = second.wrapping_mul(0x846c_a68b);
4581 second ^= second >> 16;
4582 let mask = width * 8 - 1;
4583 [(first as usize) & mask, (second as usize) & mask]
4584}
4585
4586const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4607
4608#[derive(Debug)]
4615enum Lengths {
4616 Narrow(Vec<u16>),
4618 Wide(Vec<u32>),
4620}
4621
4622impl Lengths {
4623 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4626 match self {
4627 Lengths::Narrow(lens) => into.extend(
4628 indices
4629 .iter()
4630 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4631 ),
4632 Lengths::Wide(lens) => into.extend(
4633 indices
4634 .iter()
4635 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4636 ),
4637 }
4638 }
4639
4640 fn footprint(&self) -> usize {
4642 match self {
4643 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4644 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4645 }
4646 }
4647}
4648
4649fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4658 match lengths_as::<u16>(ends)? {
4659 Some(narrow) => Some(Lengths::Narrow(narrow)),
4660 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4661 }
4662}
4663
4664fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4667 let mut lens = Vec::with_capacity(ends.len());
4668 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4669 let mut start = 0;
4670 for &end in block {
4671 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4672 return Some(None);
4673 };
4674 lens.push(len);
4675 start = end;
4676 }
4677 }
4678 Some(Some(lens))
4679}
4680
4681const TEXT_OFFSET_RUN: usize = 512;
4688
4689const DICTIONARY_HEADER: usize = 16;
4692
4693const DICTIONARY_SCATTERED: u32 = 1 << 31;
4707const DICTIONARY_GRAMS: u32 = 1 << 30;
4709const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4712const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4714
4715const TEXT_RANK_BLOCK: usize = 512;
4726
4727const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4741
4742impl NativeText {
4743 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4750 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4751 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4752 Ok(Some(bytes.as_slice()))
4753 }
4754
4755 fn block_chars(&self, block: usize) -> Result<&[u32]> {
4762 let slot = self
4763 .char_lens
4764 .get(block)
4765 .ok_or_else(|| invalid("a block past the global dictionary"))?;
4766 if let Some(lens) = slot.get() {
4767 return Ok(lens);
4768 }
4769 let decoded;
4770 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4771 Some(Ok(kept)) => kept,
4772 _ => {
4773 decoded = self.decode_block(block)?;
4774 &decoded
4775 }
4776 };
4777 let first = block * TEXT_PAYLOAD_VALUES;
4778 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4779 let ends = self.ends_within(first, last)?;
4780 if ends.len() != last - first {
4781 return Err(invalid("global dictionary offsets are short"));
4782 }
4783 let mut lens = Vec::with_capacity(ends.len());
4784 let mut start = u64::from(self.start_within(first)?);
4785 for &end in &ends {
4786 let value = usize::try_from(start)
4787 .ok()
4788 .zip(usize::try_from(end).ok())
4789 .and_then(|(from, to)| bytes.get(from..to))
4790 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4791 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4794 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4795 start = end;
4796 }
4797 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4798 }
4799
4800 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4805 let len = self.lengths[block];
4806 let mut stored = vec![
4807 0;
4808 usize::try_from(len).map_err(|_| invalid(
4809 "global dictionary block does not fit in memory"
4810 ))?
4811 ];
4812 read_at(&self.file, self.starts[block], &mut stored)?;
4813 if checksum(&stored) != self.hashes[block] {
4814 return Err(invalid("global dictionary payload checksum differs"));
4815 }
4816 let first = block * TEXT_PAYLOAD_VALUES;
4817 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4818 let want = self.end_within(last - 1)? as usize;
4819 let values = string::decode_flat(&stored)?;
4820 if values.len() != last - first {
4821 return Err(invalid("global dictionary block holds the wrong value count"));
4822 }
4823 let bytes = values.into_bytes();
4824 if bytes.len() != want {
4825 return Err(invalid("global dictionary block decodes to the wrong length"));
4826 }
4827 Ok(bytes)
4828 }
4829
4830 fn loaned_block<'a>(
4839 &'a self,
4840 block: usize,
4841 decoded: &'a mut Vec<u8>,
4842 scattered: bool,
4843 ) -> Result<&'a [u8]> {
4844 let kept = self.blocks.get(block).and_then(OnceLock::get);
4845 if let Some(Ok(kept)) = kept {
4846 return Ok(kept);
4847 }
4848 let again = kept.is_none()
4849 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4850 let keep = again
4851 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4852 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4853 if keep {
4854 let kept = self
4855 .payload_block(block)?
4856 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4857 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4858 return Ok(kept);
4859 }
4860 *decoded = self.decode_block(block)?;
4861 if scattered && again {
4862 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4863 }
4864 Ok(decoded)
4865 }
4866
4867 fn ends_worth_unpacking(&self) -> usize {
4884 self.values.max(TEXT_PAYLOAD_VALUES)
4885 }
4886
4887 fn value_ends(&self) -> Option<&[u32]> {
4889 if let Some(built) = self.value_ends.get() {
4890 return built.as_deref();
4891 }
4892 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4893 return None;
4894 }
4895 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4896 }
4897
4898 fn unpack_ends(&self) -> Option<Vec<u32>> {
4904 let mut ends = vec![0u32; self.values];
4905 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4906 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4907 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4908 u32::try_from(bits).unwrap_or(u32::MAX)
4909 })
4910 .ok()?;
4911 }
4912 if ends.contains(&u32::MAX) { None } else { Some(ends) }
4915 }
4916
4917 fn packed(&self) -> &[u8] {
4919 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4920 }
4921
4922 fn end_within(&self, index: usize) -> Result<u32> {
4924 if let Some(ends) = self.value_ends() {
4925 return ends
4926 .get(index)
4927 .copied()
4928 .ok_or_else(|| invalid("global dictionary offsets are short"));
4929 }
4930 let run = index / TEXT_OFFSET_RUN;
4931 let bytes = self
4932 .packed()
4933 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4934 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4935 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4936 .map_err(|_| invalid("global dictionary offsets are short"))?;
4937 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4938 }
4939
4940 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4958 let mut ends = vec![0u64; last.saturating_sub(first)];
4959 let mut scratch = Vec::new();
4960 let mut at = first;
4961 while at < last {
4962 let run = at / TEXT_OFFSET_RUN;
4963 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4964 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4965 let bytes = self
4966 .packed()
4967 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4968 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4969 let from = at % TEXT_OFFSET_RUN;
4970 let upto = stop - run * TEXT_OFFSET_RUN;
4971 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4972 return Err(invalid("global dictionary offsets are short"));
4973 }
4974 let into = &mut ends[at - first..stop - first];
4975 if from == 0 {
4976 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4977 .map_err(|_| invalid("global dictionary offsets are short"))?;
4978 } else {
4979 scratch.resize(held, 0);
4980 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4981 .map_err(|_| invalid("global dictionary offsets are short"))?;
4982 into.copy_from_slice(&scratch[from..upto]);
4983 }
4984 at = stop;
4985 }
4986 Ok(ends)
4987 }
4988
4989 fn start_within(&self, index: usize) -> Result<u32> {
4992 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4993 }
4994
4995 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5003 if let Some(ends) = self.value_ends() {
5004 let end =
5005 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5006 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5009 if start > end {
5010 return Err(invalid("global dictionary value ends before it starts"));
5011 }
5012 return Ok((start, end));
5013 }
5014 let within = index % TEXT_OFFSET_RUN;
5015 let (start, end) = if within == 0 {
5016 (self.start_within(index)?, self.end_within(index)?)
5017 } else {
5018 let run = index / TEXT_OFFSET_RUN;
5019 let bytes = self
5020 .packed()
5021 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5022 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5023 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5024 .map_err(|_| invalid("global dictionary offsets are short"))?;
5025 let ends = u32::try_from(end)
5026 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5027 let starts = u32::try_from(start)
5028 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5029 (starts, ends)
5030 };
5031 if start > end {
5032 return Err(invalid("global dictionary value ends before it starts"));
5033 }
5034 Ok((start, end))
5035 }
5036
5037 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5044 let slot = self
5045 .rank_blocks
5046 .get(rank / TEXT_RANK_BLOCK)
5047 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5048 let block = slot
5049 .get_or_init(|| {
5050 let mut bytes = Vec::new();
5051 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5052 Ok(bytes)
5053 })
5054 .as_ref()
5055 .map_err(Clone::clone)?;
5056 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5057 }
5058
5059 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5062 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5063 let end = self.rank_ends[which];
5064 bytes.clear();
5065 bytes.resize((end - start) as usize, 0);
5066 read_at(&self.file, self.rank_at + start, bytes)?;
5067 let expected = self
5068 .rank_hashes
5069 .get(which)
5070 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5071 if checksum(bytes) != *expected {
5072 return Err(invalid("global dictionary rank checksum differs"));
5073 }
5074 Ok(())
5075 }
5076
5077 fn head_at(&self, rank: usize) -> Result<u64> {
5079 let (block, within) = self.rank_parts(rank)?;
5080 let (base, width, packed) = rank_heads(block)?;
5081 let above = bitpack::tail_at(packed, width, within)
5082 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5083 Ok(base.wrapping_add(above))
5084 }
5085
5086 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5088 let (_, width, packed) = rank_heads(block)?;
5089 packed
5090 .get(bitpack::tail_len(count, width)..)
5091 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5092 }
5093
5094 fn rank_block_len(&self, rank: usize) -> usize {
5096 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5097 TEXT_RANK_BLOCK.min(self.ranks - first)
5098 }
5099}
5100
5101fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5103 let header = block
5104 .get(..RANK_BLOCK_HEADER)
5105 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5106 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5107 let width = header[8] as usize;
5108 if width > 64 {
5109 return Err(invalid("global dictionary rank block packs heads past a word"));
5110 }
5111 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5112}
5113
5114fn offset_width(ends: &[u32]) -> usize {
5121 let span = ends.iter().copied().max().unwrap_or(0);
5125 (u32::BITS - span.leading_zeros()) as usize
5126}
5127
5128fn offset_bytes(values: usize, bits: usize) -> usize {
5131 let full = values / TEXT_OFFSET_RUN;
5132 let rest = values % TEXT_OFFSET_RUN;
5133 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5134}
5135
5136fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5140 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5141 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5142 run.clear();
5143 run.extend(chunk.iter().map(|&end| u64::from(end)));
5144 bitpack::pack_tail(&run, bits, out)
5145 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5146 }
5147 Ok(())
5148}
5149
5150fn code_width(values: usize) -> usize {
5152 match u64::try_from(values).unwrap_or(u64::MAX) {
5153 0 | 1 => 0,
5154 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5155 }
5156}
5157
5158impl TextSource for NativeText {
5159 fn len(&self) -> usize {
5160 self.values
5161 }
5162
5163 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5164 let Some(grams) = &self.grams else { return Ok(true) };
5165 if literal.len() < 4 || first >= self.values {
5166 return Ok(true);
5167 }
5168 let verdict = grams.verdicts(&self.file, literal)?;
5169 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5170 }
5171
5172 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5173 if index >= self.values {
5174 return Ok(None);
5175 }
5176 let (start, end) = self.span_within(index)?;
5177 if start == end {
5178 return Ok(Some(&[]));
5179 }
5180 let block = index / TEXT_PAYLOAD_VALUES;
5183 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5184 Ok(bytes.get(start as usize..end as usize))
5185 }
5186
5187 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5188 if index >= self.values {
5189 return Ok(None);
5190 }
5191 let (start, end) = self.span_within(index)?;
5192 Ok(Some((end - start) as usize))
5193 }
5194
5195 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5202 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5203 into.reserve(indices.len());
5204 let Some(ends) = self.value_ends() else {
5205 for &index in indices {
5206 into.push(
5207 self.bytes_len_at(index as usize)?
5208 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5209 );
5210 }
5211 return Ok(());
5212 };
5213 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5214 lens.extend_at(indices, into);
5215 return Ok(());
5216 }
5217 for &index in indices {
5218 let index = index as usize;
5219 let Some(&end) = ends.get(index) else {
5221 into.push(0);
5222 continue;
5223 };
5224 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5225 if start > end {
5226 return Err(invalid("global dictionary value ends before it starts"));
5227 }
5228 into.push(i64::from(end - start));
5229 }
5230 Ok(())
5231 }
5232
5233 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5236 into.reserve(indices.len());
5237 for &index in indices {
5238 let index = index as usize;
5239 if index >= self.values {
5241 into.push(0);
5242 continue;
5243 }
5244 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5245 let len = lens
5246 .get(index % TEXT_PAYLOAD_VALUES)
5247 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5248 into.push(i64::from(*len));
5249 }
5250 Ok(())
5251 }
5252
5253 fn sweep(
5266 &self,
5267 first: usize,
5268 limit: usize,
5269 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5270 ) -> Result<usize> {
5271 let limit = limit.min(self.values);
5272 if first >= limit {
5273 return Ok(first);
5274 }
5275 let block = first / TEXT_PAYLOAD_VALUES;
5276 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5277 let mut decoded = Vec::new();
5278 let bytes = self.loaned_block(block, &mut decoded, false)?;
5279 let ends = self.ends_within(first, last)?;
5280 if ends.len() != last - first {
5281 return Err(invalid("global dictionary offsets are short"));
5282 }
5283 let mut start = u64::from(self.start_within(first)?);
5284 for (index, &end) in (first..last).zip(&ends) {
5287 let value = usize::try_from(start)
5288 .ok()
5289 .zip(usize::try_from(end).ok())
5290 .and_then(|(from, to)| bytes.get(from..to))
5291 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5292 body(index, value)?;
5293 start = end;
5294 }
5295 Ok(last)
5296 }
5297
5298 fn visit_at(
5307 &self,
5308 indices: &[u32],
5309 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5310 ) -> Result<()> {
5311 let mut order = (0..indices.len()).collect::<Vec<_>>();
5312 order.sort_unstable_by_key(|&at| indices[at]);
5313 let block_of = |at: usize| {
5314 let index = indices[at] as usize;
5315 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5316 };
5317 let mut decoded = Vec::new();
5318 let mut run = 0;
5319 while run < order.len() {
5320 let Some(block) = block_of(order[run]) else {
5321 for &at in &order[run..] {
5323 body(at, &[])?;
5324 }
5325 break;
5326 };
5327 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5328 let bytes = self.loaned_block(block, &mut decoded, true)?;
5329 for &at in &order[run..upto] {
5330 let (start, end) = self.span_within(indices[at] as usize)?;
5331 let value = bytes
5332 .get(start as usize..end as usize)
5333 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5334 body(at, value)?;
5335 }
5336 run = upto;
5337 }
5338 Ok(())
5339 }
5340
5341 fn visit(
5347 &self,
5348 indices: &[usize],
5349 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5350 ) -> Result<()> {
5351 let mut at = 0;
5352 while at < indices.len() {
5353 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5354 let upto =
5355 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5356 let wanted = &indices[at..upto];
5357 if wanted.iter().any(|&index| index >= self.values) {
5358 return Err(invalid("a visited value is past the global dictionary"));
5359 }
5360 let decoded;
5361 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5362 Some(Ok(kept)) => kept,
5363 _ => {
5364 decoded = self.decode_block(block)?;
5365 &decoded
5366 }
5367 };
5368 for (offset, &index) in wanted.iter().enumerate() {
5369 let (start, end) = self.span_within(index)?;
5370 let value = bytes
5371 .get(start as usize..end as usize)
5372 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5373 body(at + offset, value)?;
5374 }
5375 at = upto;
5376 }
5377 Ok(())
5378 }
5379
5380 fn ranks(&self) -> Option<usize> {
5381 (self.ranks > 0).then_some(self.ranks)
5382 }
5383
5384 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5392 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5393 if let Some(&answer) = memo.get(wanted) {
5394 return Ok(answer);
5395 }
5396 let answer = search_below(self, ranks, wanted)?;
5397 if memo.len() >= TEXT_SEARCH_MEMO {
5398 memo.clear();
5399 }
5400 memo.insert(wanted.to_vec(), answer);
5401 Ok(answer)
5402 }
5403
5404 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5405 let settled = self.head_at(rank)?.cmp(&head(wanted));
5409 if settled != Ordering::Equal {
5410 return Ok(settled);
5411 }
5412 let code = self.code_at_rank(rank)?;
5413 let bytes = self
5414 .bytes_at(code as usize)?
5415 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5416 Ok(bytes.cmp(wanted))
5417 }
5418
5419 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5420 let (block, within) = self.rank_parts(rank)?;
5421 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5422 let code = bitpack::tail_at(codes, self.code_bits, within)
5423 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5424 let code = u32::try_from(code)
5425 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5426 if code as usize >= self.len() {
5427 return Err(invalid("global dictionary order names a code it does not have"));
5428 }
5429 Ok(code)
5430 }
5431
5432 fn code_ranks(&self) -> Option<&[u32]> {
5433 if self.ranks == 0 || self.ranks != self.len() {
5437 return None;
5438 }
5439 self.code_ranks
5440 .get_or_init(|| {
5441 let mut ranks = vec![u32::MAX; self.ranks];
5442 let mut scratch = Vec::new();
5450 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5451 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5452 let which = first / TEXT_RANK_BLOCK;
5453 let block = match self.rank_blocks.get(which)?.get() {
5454 Some(kept) => kept.as_ref().ok()?.as_slice(),
5455 None => {
5456 self.read_rank_block(which, &mut scratch).ok()?;
5457 scratch.as_slice()
5458 }
5459 };
5460 let count = self.rank_block_len(first);
5461 let packed = self.rank_codes(block, count).ok()?;
5462 let codes = codes.get_mut(..count)?;
5463 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5464 for (within, &code) in codes.iter().enumerate() {
5465 let code = usize::try_from(code).ok()?;
5466 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5467 }
5468 }
5469 if ranks.contains(&u32::MAX) {
5470 return None;
5471 }
5472 Some(ranks)
5473 })
5474 .as_deref()
5475 }
5476
5477 fn footprint(&self) -> usize {
5478 self.offsets.capacity()
5479 + self
5480 .value_ends
5481 .get()
5482 .and_then(Option::as_ref)
5483 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5484 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5485 + self
5486 .code_ranks
5487 .get()
5488 .and_then(Option::as_ref)
5489 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5490 + self.rank_hashes.capacity() * size_of::<u64>()
5491 + self.rank_ends.capacity() * size_of::<u64>()
5492 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5493 + self
5494 .rank_blocks
5495 .iter()
5496 .filter_map(OnceLock::get)
5497 .filter_map(|result| result.as_ref().ok())
5498 .map(Vec::capacity)
5499 .sum::<usize>()
5500 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5501 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5502 + self
5503 .char_lens
5504 .iter()
5505 .filter_map(OnceLock::get)
5506 .map(|lens| lens.len() * size_of::<u32>())
5507 .sum::<usize>()
5508 + self.hashes.capacity() * size_of::<u64>()
5509 + self.starts.capacity() * size_of::<u64>()
5510 + self.lengths.capacity() * size_of::<u64>()
5511 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5512 + self
5513 .blocks
5514 .iter()
5515 .filter_map(OnceLock::get)
5516 .filter_map(|result| result.as_ref().ok())
5517 .map(Vec::capacity)
5518 .sum::<usize>()
5519 }
5520}
5521
5522fn places(table: &Table) -> Result<Vec<Place>> {
5524 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5525 for (at, stripe) in table.stripes.iter().enumerate() {
5526 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5527 for (part, &rows) in stripe.parts.iter().enumerate() {
5528 places.push(Place {
5529 stripe: index,
5530 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5531 rows,
5532 });
5533 }
5534 }
5535 Ok(places)
5536}
5537
5538fn read_index<F: Positional + ?Sized>(
5543 file: &F,
5544 stripe: &Stripe,
5545 column: usize,
5546) -> Result<Vec<PartSpan>> {
5547 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5548 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5549}
5550
5551fn read_index_span<F: Positional + ?Sized>(
5552 file: &F,
5553 index: Span,
5554 page: Span,
5555 parts: usize,
5556 column: usize,
5557) -> Result<Vec<PartSpan>> {
5558 let section = index_section(parts)?;
5559 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5560 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5561 if end > index.length as usize {
5562 return Err(invalid("index page is shorter than its columns"));
5563 }
5564 let mut bytes = vec![0; section];
5565 let offset =
5566 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5567 read_at(file, offset, &mut bytes)?;
5568 let entries = section - size_of::<u64>();
5569 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5570 if checksum(&bytes[..entries]) != stored {
5571 return Err(invalid(&format!(
5574 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5575 wanted {stored:016x} and got {:016x}",
5576 checksum(&bytes[..entries]),
5577 )));
5578 }
5579 let mut spans = Vec::with_capacity(parts);
5580 let mut start = 0_usize;
5581 for part in 0..parts {
5582 let at = part * INDEX_ENTRY;
5583 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5584 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5585 spans.push(PartSpan { start, length, hash });
5586 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5587 }
5588 if start != page.length as usize {
5589 return Err(invalid("column page length differs from its index"));
5590 }
5591 Ok(spans)
5592}
5593
5594fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5596 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5597 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5598}
5599
5600fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5606 if let Some(slot) = cached.index.get_mut(held.stripe) {
5607 if slot.is_none() {
5608 *slot = Some(Arc::clone(&held.index));
5609 }
5610 }
5611 let page = held.page.clone()?;
5612 let slot = cached.pages.get_mut(held.stripe)?;
5613 if slot.is_some() {
5614 return None;
5615 }
5616 let bytes = page.bytes.len();
5617 let used = Arc::new(AtomicBool::new(true));
5620 *slot = Some(Resident { page, used: Arc::clone(&used) });
5621 Some((bytes, used))
5622}
5623
5624#[derive(Debug, Clone)]
5633pub struct Catalog {
5634 file: Arc<File>,
5635 size: u64,
5636 entries: Arc<Vec<Entry>>,
5637 views: Arc<Vec<ViewEntry>>,
5639 opening: Opening,
5640 pool: PagePool,
5642}
5643
5644#[derive(Debug, Clone, PartialEq, Eq)]
5646pub struct CertifiedSums {
5647 pub columns: Vec<(i128, u64)>,
5648 pub rows: u64,
5649}
5650
5651#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5653pub enum IntegerExtremes {
5654 Null,
5655 Values { low: i128, high: i128 },
5656}
5657
5658pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5660
5661impl Catalog {
5662 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5671 Self::open_in(path, &PagePool::default())
5672 }
5673
5674 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5680 let (file, size, _, bytes, opening) = slot_bytes(path)?;
5681 let (entries, views) = decode_catalog(&bytes, size)?;
5682 Ok(Self {
5683 file: Arc::new(file),
5684 size,
5685 entries: Arc::new(entries),
5686 views: Arc::new(views),
5687 opening,
5688 pool: pool.clone(),
5689 })
5690 }
5691
5692 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5694 self.entries.iter().map(|entry| entry.name.as_str())
5695 }
5696
5697 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5704 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5705 }
5706
5707 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5713 self.views.iter()
5714 }
5715
5716 #[must_use]
5718 pub fn len(&self) -> usize {
5719 self.entries.len()
5720 }
5721
5722 #[must_use]
5725 pub fn is_empty(&self) -> bool {
5726 self.entries.is_empty()
5727 }
5728
5729 pub fn table(&self, name: &str) -> Result<Reader> {
5735 let entry = self
5736 .entries
5737 .iter()
5738 .find(|entry| entry.name == name)
5739 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5740 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5744 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5745 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5746 }
5747 let mut opening = self.opening;
5748 opening.reads += 1;
5749 opening.bytes += u64::from(entry.directory.length);
5750 Reader::build(
5751 Arc::clone(&self.file),
5752 self.size,
5753 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5754 u64::from(entry.directory.length),
5755 opening,
5756 self.pool.clone(),
5757 )
5758 }
5759
5760 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5768 let mut counts = BTreeMap::<i64, u64>::new();
5769 let Some(()) = self.integer_fold(name, column, |value, count| {
5770 let held = counts.entry(value).or_default();
5771 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5772 Ok(())
5773 })?
5774 else {
5775 return Ok(None);
5776 };
5777 Ok(Some(counts.into_iter().collect()))
5778 }
5779
5780 pub fn integer_fold(
5787 &self,
5788 name: &str,
5789 column: usize,
5790 mut emit: impl FnMut(i64, u64) -> Result<()>,
5791 ) -> Result<Option<()>> {
5792 let entry = self
5793 .entries
5794 .iter()
5795 .find(|entry| entry.name == name)
5796 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5797 let field =
5798 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5799 if !signed_integer(&field.ty) {
5800 return Ok(None);
5801 }
5802 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5803 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5804 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5805 }
5806 quick_integer_fold(
5807 &self.file,
5808 Cursor::over(&self.file, offset, length),
5809 entry,
5810 self.size,
5811 column,
5812 &mut emit,
5813 )?;
5814 Ok(Some(()))
5815 }
5816
5817 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5822 let entry = self
5823 .entries
5824 .iter()
5825 .find(|entry| entry.name == name)
5826 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5827 let Some(field) = entry.fields.get(column) else {
5828 return Err(invalid("frequency column index out of range"));
5829 };
5830 if !matches!(
5831 field.ty,
5832 LogicalType::TinyInt
5833 | LogicalType::SmallInt
5834 | LogicalType::Integer
5835 | LogicalType::BigInt
5836 | LogicalType::UTinyInt
5837 | LogicalType::USmallInt
5838 | LogicalType::UInteger
5839 | LogicalType::UBigInt
5840 ) {
5841 return Ok(None);
5842 }
5843 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5844 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5845 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5846 }
5847 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5848 return frequencies
5849 .iter()
5850 .filter(|(value, _)| value.is_some_and(|value| value != 0))
5851 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5852 .map(Some)
5853 .ok_or_else(|| invalid("numeric frequency count overflow"));
5854 }
5855 quick_nonzero(
5856 Cursor::over(&self.file, offset, length),
5857 &entry.name,
5858 &entry.fields,
5859 entry.rows,
5860 column,
5861 )
5862 }
5863
5864 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5867 let entry = self
5868 .entries
5869 .iter()
5870 .find(|entry| entry.name == name)
5871 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5872 let mut sums = Vec::with_capacity(columns.len());
5873 for &column in columns {
5874 let Some(field) = entry.fields.get(column) else {
5875 return Err(invalid("aggregate column index out of range"));
5876 };
5877 if !signed_integer(&field.ty) {
5878 return Ok(None);
5879 }
5880 let Some(sum) = entry.aggregates[column] else {
5881 return Ok(None);
5882 };
5883 sums.push(sum);
5884 }
5885 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5886 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5887 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5888 }
5889 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5890 }
5891
5892 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5894 let entry = self
5895 .entries
5896 .iter()
5897 .find(|entry| entry.name == name)
5898 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5899 let Some(count) = entry.distincts.get(column).copied() else {
5900 return Err(invalid("distinct column index out of range"));
5901 };
5902 let Some(count) = count else { return Ok(None) };
5903 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5904 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5905 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5906 }
5907 Ok(Some(count))
5908 }
5909
5910 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5912 let entry = self
5913 .entries
5914 .iter()
5915 .find(|entry| entry.name == name)
5916 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5917 let Some(extremes) = entry.extremes.get(column).copied() else {
5918 return Err(invalid("extremes column index out of range"));
5919 };
5920 let Some(extremes) = extremes else { return Ok(None) };
5921 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5922 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5923 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5924 }
5925 Ok(Some(match extremes {
5926 None => IntegerExtremes::Null,
5927 Some((low, high)) => IntegerExtremes::Values { low, high },
5928 }))
5929 }
5930
5931 pub fn exact_numeric_frequencies(
5933 &self,
5934 name: &str,
5935 column: usize,
5936 ) -> Result<Option<NumericFrequencies>> {
5937 let entry = self
5938 .entries
5939 .iter()
5940 .find(|entry| entry.name == name)
5941 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5942 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5943 return Err(invalid("numeric frequency column index out of range"));
5944 };
5945 let Some(frequencies) = frequencies else { return Ok(None) };
5946 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5947 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5948 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5949 }
5950 Ok(Some(frequencies))
5951 }
5952
5953 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5955 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5956 }
5957}
5958
5959fn slot_offset(generation: u64) -> u64 {
5964 16 + (generation - 1) % 2 * SLOT_BYTES as u64
5965}
5966
5967fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5972 let file = File::open(path).map_err(io)?;
5973 let size = file.metadata().map_err(io)?.len();
5974 let (slot, bytes, opening) = committed_slot(&file, size)?;
5975 Ok((file, size, slot, bytes, opening))
5976}
5977
5978fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
5984 if size < HEADER {
5985 return Err(invalid("file is shorter than its header"));
5986 }
5987 let mut header = [0; HEADER as usize];
5988 read_at(file, 0, &mut header)?;
5989 let mut opening = Opening { reads: 1, bytes: HEADER };
5990 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5991 if &header[..8] != MAGIC {
5996 return Err(invalid("the header does not begin with a rudb native magic"));
5997 }
5998 if !READABLE.contains(&version) {
5999 return Err(invalid(&format!(
6000 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6001 be written again"
6002 )));
6003 }
6004 let mut selected = None;
6005 for start in [16, 16 + SLOT_BYTES] {
6006 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6007 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6008 continue;
6009 }
6010 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6011 if slot.offset < HEADER || end > size {
6012 continue;
6013 }
6014 let mut bytes = vec![0; slot.length as usize];
6015 read_at(file, slot.offset, &mut bytes)?;
6016 opening.reads += 1;
6017 opening.bytes += u64::from(slot.length);
6018 if checksum(&bytes) == slot.hash
6019 && selected
6020 .as_ref()
6021 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6022 {
6023 selected = Some((slot, bytes));
6024 }
6025 }
6026 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6027 Ok((slot, bytes, opening))
6028}
6029
6030impl Reader {
6031 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6038 let catalog = Catalog::open(path)?;
6039 let mut names = catalog.names();
6040 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6041 if names.next().is_some() {
6042 return Err(invalid(
6043 "the file holds more than one table, so it has to be opened by name",
6044 ));
6045 }
6046 catalog.table(&name)
6047 }
6048
6049 fn build(
6051 file: Arc<File>,
6052 size: u64,
6053 table: Table,
6054 directory: u64,
6055 opening: Opening,
6056 pool: PagePool,
6057 ) -> Result<Self> {
6058 let places = places(&table)?;
6059 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6060 let table_fields = table.fields.len();
6061 let stripes = table.stripes.len();
6062 let columns = (0..table.fields.len())
6063 .map(|_| {
6064 Mutex::new(Cached {
6065 pages: (0..stripes).map(|_| None).collect(),
6066 index: (0..stripes).map(|_| None).collect(),
6067 seen: vec![false; stripes],
6068 ..Cached::default()
6069 })
6070 })
6071 .collect::<Vec<_>>();
6072 let cache = Shelf {
6073 columns,
6074 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6075 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6076 };
6077 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6078 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6079 .collect();
6080 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6081 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6082 .collect();
6083 Ok(Self {
6084 file,
6085 table: Arc::new(table),
6086 dictionaries: Arc::new(dictionaries),
6087 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6088 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6089 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6090 opened: Arc::new(AtomicUsize::new(0)),
6091 sieves: Arc::new(sieves),
6092 part_ranges: Arc::new(part_ranges),
6093 places: Arc::new(places),
6094 cache: Arc::new(cache),
6095 pool,
6096 pages: Arc::new(AtomicUsize::new(0)),
6097 indexes: Arc::new(AtomicUsize::new(0)),
6098 size,
6099 directory,
6100 opening,
6101 })
6102 }
6103
6104 #[must_use]
6111 pub fn reads(&self) -> Reads {
6112 Reads {
6113 opening: self.opening,
6114 pages: self.pages.load(Atomic::Relaxed),
6115 indexes: self.indexes.load(Atomic::Relaxed),
6116 dictionaries: self.opened.load(Atomic::Relaxed),
6117 }
6118 }
6119
6120 #[must_use]
6125 pub fn layout(&self) -> Layout {
6126 let table = &self.table;
6127 let stripes = table.stripes.as_slice();
6128 let columns = table
6129 .fields
6130 .iter()
6131 .enumerate()
6132 .map(|(at, field)| ColumnLayout {
6133 name: field.name.clone(),
6134 kind: field.ty.to_string(),
6135 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6136 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6137 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6138 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6139 dictionary: dictionary_bytes(table, at),
6140 })
6141 .collect();
6142 Layout {
6143 file: self.size,
6144 rows: table.rows,
6145 stripes: stripes.len(),
6146 parts: self.places.len(),
6147 columns,
6148 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6149 directory: self.directory,
6150 header: HEADER,
6151 }
6152 }
6153
6154 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6171 let field = self
6172 .table
6173 .fields
6174 .get(column)
6175 .ok_or_else(|| invalid("stored column index out of range"))?;
6176 let mut stored = Vec::with_capacity(self.places.len());
6177 let mut row = 0;
6178 for (at, stripe) in self.table.stripes.iter().enumerate() {
6179 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6180 let index = read_index(&self.file, stripe, column)?;
6181 let mut bytes = vec![0; page.length as usize];
6182 read_at(&self.file, page.offset, &mut bytes)?;
6183 let ranges = self.stripe_part_ranges(at, column);
6184 for (part, &rows) in stripe.parts.iter().enumerate() {
6185 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6186 let held = part_bytes(&bytes, span)?;
6187 let range = ranges.and_then(|held| held.get(part));
6188 stored.push(StoredPart {
6189 stripe: at,
6190 part,
6191 row,
6192 rows: rows as usize,
6193 encoding: page_encoding(&field.ty, rows as usize, held),
6194 bytes: span.length as u64,
6195 page: page.offset,
6196 offset: span.start as u64,
6197 low: range
6198 .and_then(|range| range.low.clone())
6199 .and_then(|bound| bound.into_value(&field.ty)),
6200 high: range
6201 .and_then(|range| range.high.clone())
6202 .and_then(|bound| bound.into_value(&field.ty)),
6203 nulls: range.map(|range| range.nulls),
6204 });
6205 row += rows as usize;
6206 }
6207 }
6208 Ok(stored)
6209 }
6210
6211 #[must_use]
6213 pub fn parts(&self) -> usize {
6214 self.places.len()
6215 }
6216
6217 #[must_use]
6224 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6225 let mut runs = Vec::with_capacity(self.table.stripes.len());
6226 let mut start = 0;
6227 for stripe in &self.table.stripes {
6228 let end = start + stripe.parts.len();
6229 runs.push(start..end);
6230 start = end;
6231 }
6232 runs
6233 }
6234
6235 #[must_use]
6240 pub fn stripe_rows(&self, stripe: usize) -> usize {
6241 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6242 }
6243
6244 pub fn keep_stripes(&self, stripes: usize) {
6251 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6252 }
6253
6254 #[must_use]
6256 pub fn part_rows(&self, at: usize) -> usize {
6257 self.places.get(at).map_or(0, |place| place.rows as usize)
6258 }
6259
6260 #[must_use]
6262 pub fn table(&self) -> &Table {
6263 &self.table
6264 }
6265
6266 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6275 let field = self
6276 .table
6277 .fields
6278 .get(column)
6279 .ok_or_else(|| invalid("frequency column index out of range"))?;
6280 let Some(summary) = self.frequency_summary(column)? else {
6281 return Ok(None);
6282 };
6283 if top == 0 || summary.entries.len() < top {
6284 return Ok(None);
6285 }
6286 let boundary = summary.entries[top - 1].count;
6287 if boundary <= summary.omitted_max {
6288 return Ok(None);
6289 }
6290 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6291 }
6292
6293 pub fn top_pair_frequencies(
6301 &self,
6302 first: usize,
6303 second: usize,
6304 _top: usize,
6305 ) -> Result<Option<PairFrequencyCounts>> {
6306 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6307 return Err(invalid("pair frequency column index out of range"));
6308 }
6309 Ok(None)
6310 }
6311
6312 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6332 let Some(prefix) = self.frequency_prefix(column)? else {
6333 return Ok(None);
6334 };
6335 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6336 }
6337
6338 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6361 let field = self
6362 .table
6363 .fields
6364 .get(column)
6365 .ok_or_else(|| invalid("frequency column index out of range"))?;
6366 let Some(summary) = self.frequency_summary(column)? else {
6367 return Ok(None);
6368 };
6369 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6370 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6371 }
6372
6373 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6375 Ok(match self.table.frequencies.get(column) {
6376 None | Some(None) => None,
6377 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6378 Some(Some(Frequencies::Stored { span, values })) => {
6379 let slot = self
6380 .frequency_summaries
6381 .get(column)
6382 .ok_or_else(|| invalid("frequency column index out of range"))?;
6383 if let Some(summary) = slot.get() {
6384 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6385 }
6386 let field = self
6387 .table
6388 .fields
6389 .get(column)
6390 .ok_or_else(|| invalid("frequency column index out of range"))?;
6391 let mut bytes = vec![0; span.length as usize];
6392 read_at(&self.file, span.offset, &mut bytes)?;
6393 let summary =
6394 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6395 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6396 let _ = slot.set(Arc::new(summary));
6397 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6398 }
6399 })
6400 }
6401
6402 fn decode_frequencies(
6410 &self,
6411 column: usize,
6412 ty: &LogicalType,
6413 entries: &[FrequencyEntry],
6414 ) -> Result<Vec<(Value, u64)>> {
6415 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6416 return Ok(values.as_ref().clone());
6417 }
6418 let values = self.decode_frequencies_once(column, ty, entries)?;
6419 if let Some(slot) = self.frequency_values.get(column) {
6420 let _ = slot.set(Arc::new(values.clone()));
6421 }
6422 Ok(values)
6423 }
6424
6425 fn decode_frequencies_once(
6426 &self,
6427 column: usize,
6428 ty: &LogicalType,
6429 entries: &[FrequencyEntry],
6430 ) -> Result<Vec<(Value, u64)>> {
6431 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6432 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6433 return Err(invalid("frequency text count differs from its synopsis"));
6434 }
6435 let dictionary =
6436 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6437 let mut codes = entries
6438 .iter()
6439 .filter_map(|entry| match entry.value {
6440 FrequencyValue::Code(code) => Some(code as usize),
6441 _ => None,
6442 })
6443 .collect::<Vec<_>>();
6444 codes.sort_unstable();
6445 codes.dedup();
6446 let texts = match &dictionary {
6447 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6448 _ => Vec::new(),
6449 };
6450 let mut out = Vec::with_capacity(entries.len());
6451 for (entry_at, entry) in entries.iter().enumerate() {
6452 let value = match entry.value {
6453 FrequencyValue::Null => {
6454 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6455 return Err(invalid("a null frequency entry has text"));
6456 }
6457 Value::Null
6458 }
6459 FrequencyValue::Integer(value) => match *ty {
6460 LogicalType::TinyInt => Value::TinyInt(
6461 i8::try_from(value)
6462 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6463 ),
6464 LogicalType::UTinyInt => Value::UTinyInt(
6465 u8::try_from(value)
6466 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6467 ),
6468 LogicalType::USmallInt => Value::USmallInt(
6469 u16::try_from(value)
6470 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6471 ),
6472 LogicalType::UInteger => Value::UInteger(
6473 u32::try_from(value)
6474 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6475 ),
6476 LogicalType::UBigInt => Value::UBigInt(
6477 u64::try_from(value)
6478 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6479 ),
6480 LogicalType::SmallInt => Value::SmallInt(
6481 i16::try_from(value)
6482 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6483 ),
6484 LogicalType::Integer => Value::Integer(
6485 i32::try_from(value)
6486 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6487 ),
6488 LogicalType::BigInt => Value::BigInt(
6489 i64::try_from(value)
6490 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6491 ),
6492 LogicalType::Date => Value::Date(
6493 i32::try_from(value)
6494 .map_err(|_| invalid("frequency DATE is out of range"))?,
6495 ),
6496 LogicalType::Timestamp => Value::Timestamp(
6497 i64::try_from(value)
6498 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6499 ),
6500 _ => return Err(invalid("integer frequency belongs to another type")),
6501 },
6502 FrequencyValue::Code(code) => {
6503 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6504 if *ty == LogicalType::Blob {
6505 Value::Blob(text.clone())
6506 } else {
6507 Value::Varchar(
6508 String::from_utf8(text.clone())
6509 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6510 )
6511 }
6512 } else {
6513 if dictionary.is_none() {
6514 return Err(invalid("frequency code has no dictionary or stored text"));
6515 }
6516 let at = codes
6517 .binary_search(&(code as usize))
6518 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6519 texts[at].clone()
6520 }
6521 }
6522 };
6523 out.push((value, entry.count));
6524 }
6525 Ok(out)
6526 }
6527
6528 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6538 let field = self
6539 .table
6540 .fields
6541 .get(column)
6542 .ok_or_else(|| invalid("frequency column index out of range"))?;
6543 let Some(summary) = self.frequency_summary(column)? else {
6544 return Ok(None);
6545 };
6546 if summary.ordinals.is_empty() {
6547 return Ok(None);
6548 }
6549 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6550 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6551 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6552 } else {
6553 (Vec::new(), Vec::new())
6554 };
6555 Ok(Some(FrequencyOccurrences {
6556 omitted_max: summary.omitted_max,
6557 ordinals: summary.ordinals.clone(),
6558 anchors,
6559 anchor_indices,
6560 }))
6561 }
6562
6563 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6589 self.table
6590 .distincts
6591 .get(column)
6592 .copied()
6593 .ok_or_else(|| invalid("distinct column index out of range"))
6594 }
6595
6596 pub fn null_count(&self, column: usize) -> Result<u64> {
6607 if column >= self.table.fields.len() {
6608 return Err(invalid("null count column index out of range"));
6609 }
6610 let mut nulls = 0_u64;
6611 for stripe in &self.table.stripes {
6612 let range = stripe
6613 .zone
6614 .column(column)
6615 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6616 nulls = nulls
6617 .checked_add(range.nulls as u64)
6618 .ok_or_else(|| invalid("null count overflow"))?;
6619 }
6620 Ok(nulls)
6621 }
6622
6623 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6638 if self.null_count(column)? > 0 || self.demoted(column) {
6639 return Ok(None);
6640 }
6641 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6642 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6643 if ranks == 0 {
6644 return Ok(None);
6645 }
6646 let low = text_at_rank(&dictionary, 0)?;
6647 let high = text_at_rank(&dictionary, ranks - 1)?;
6648 Ok(Some((low, high)))
6649 }
6650
6651 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6674 if column >= self.table.fields.len() {
6675 return Err(invalid("extremes column index out of range"));
6676 }
6677 let mut low: Option<Bound> = None;
6678 let mut high: Option<Bound> = None;
6679 for stripe in &self.table.stripes {
6680 let range = stripe
6681 .zone
6682 .column(column)
6683 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6684 if !range.exact {
6685 return Ok(None);
6686 }
6687 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6692 if stripe.rows > range.nulls {
6693 return Ok(None);
6694 }
6695 continue;
6696 };
6697 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6698 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6699 }
6700 Ok(low.zip(high))
6701 }
6702
6703 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6716 if column >= self.table.fields.len() {
6717 return Err(invalid("sum column index out of range"));
6718 }
6719 let mut total = 0_i128;
6720 let mut rows = 0_u64;
6721 for stripe in &self.table.stripes {
6722 let range = stripe
6723 .zone
6724 .column(column)
6725 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6726 let Some(part) = range.sum else { return Ok(None) };
6727 let Some(sum) = total.checked_add(part) else { return Ok(None) };
6728 total = sum;
6729 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6730 }
6731 Ok(Some((total, rows)))
6732 }
6733
6734 pub fn host_groups(
6736 &self,
6737 column: usize,
6738 _minimum_count: u64,
6739 ) -> Result<Option<Vec<host::HostEntry>>> {
6740 if column >= self.table.fields.len() {
6741 return Err(invalid("host group column index out of range"));
6742 }
6743 Ok(None)
6744 }
6745
6746 #[must_use]
6750 pub fn demoted(&self, column: usize) -> bool {
6751 self.table.demoted.get(column).copied().unwrap_or(false)
6752 }
6753
6754 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6763 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6764 if let Some(dictionary) = self.dictionaries[column].get() {
6765 return Ok(Some(Arc::clone(dictionary)));
6766 }
6767 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6768 if let Some(dictionary) = self.dictionaries[column].get() {
6769 return Ok(Some(Arc::clone(dictionary)));
6770 }
6771 self.opened.fetch_add(1, Atomic::Relaxed);
6772 let dictionary = Arc::new(open_global_dictionary(
6773 Arc::clone(&self.file),
6774 page,
6775 &self.table.fields[column].ty,
6776 TEXT_KEEP_BUDGET,
6777 )?);
6778 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6779 Ok(Some(dictionary))
6780 }
6781
6782 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6789 if of.extent_bytes == 0 {
6790 return Ok(Vec::new());
6791 }
6792 let mut bytes = vec![0; of.extent_bytes as usize];
6793 read_at(&self.file, of.extent_page, &mut bytes)?;
6794 if checksum(&bytes) != of.hash {
6795 return Err(invalid("a section's extent table does not checksum"));
6796 }
6797 let extents = section::decode_extents(&bytes)?;
6798 if extents.len() != of.extents as usize {
6799 return Err(invalid("a section's extent table is not the length the entry says"));
6800 }
6801 Ok(extents)
6802 }
6803
6804 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
6814 let mut bytes = Vec::new();
6815 self.extent_into(of, &mut bytes)?;
6816 Ok(bytes)
6817 }
6818
6819 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
6821 let end = of
6822 .offset
6823 .checked_add(u64::from(of.length))
6824 .ok_or_else(|| invalid("an extent overflows the file"))?;
6825 if of.offset < HEADER || end > self.size {
6826 return Err(invalid("an extent is outside the file"));
6827 }
6828 bytes.resize(of.length as usize, 0);
6829 read_at(&self.file, of.offset, bytes)?;
6830 if checksum(bytes) != of.hash {
6831 return Err(invalid("an extent does not checksum"));
6832 }
6833 Ok(())
6834 }
6835
6836 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6845 let extents = self.extents(of)?;
6846 let mut bytes =
6847 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6848 for one in &extents {
6849 if one.first != bytes.len() as u64 {
6850 return Err(invalid("a section's extents do not join up"));
6851 }
6852 bytes.extend_from_slice(&self.extent(one)?);
6853 }
6854 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6857 return Err(invalid("a section's header is longer than its payload"));
6858 }
6859 Ok(bytes)
6860 }
6861
6862 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6871 self.read_impl(part, columns, true, None)
6872 }
6873
6874 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6884 self.read_impl(part, columns, false, None)
6885 }
6886
6887 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6895 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6896 let field =
6897 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6898 if !matches!(
6899 field.ty,
6900 LogicalType::TinyInt
6901 | LogicalType::SmallInt
6902 | LogicalType::Integer
6903 | LogicalType::BigInt
6904 ) {
6905 return Ok(None);
6906 }
6907 let stripe_index = place.stripe as usize;
6908 let stripe = self
6909 .table
6910 .stripes
6911 .get(stripe_index)
6912 .ok_or_else(|| invalid("stripe index out of range"))?;
6913 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6914 let held = self.held(stripe_index, stripe, column, true)?;
6915 let span = *held
6916 .index
6917 .get(place.part as usize)
6918 .ok_or_else(|| invalid("part index out of range"))?;
6919 let owned;
6920 let bytes = match &held.page {
6921 Some(page) => page.part(place.part as usize, span)?,
6922 None => {
6923 let offset = page
6924 .offset
6925 .checked_add(span.start as u64)
6926 .ok_or_else(|| invalid("part range overflow"))?;
6927 let mut bytes = vec![0; span.length];
6928 read_at(&self.file, offset, &mut bytes)?;
6929 verify_part(&bytes, span)?;
6930 owned = bytes;
6931 &owned
6932 }
6933 };
6934 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6935 return Ok(None);
6936 }
6937 let (rows, counts) = integer::tally(&bytes[2..])?;
6938 if rows != place.rows as usize {
6939 return Err(invalid("encoded integer part holds the wrong number of rows"));
6940 }
6941 for &(value, _) in &counts {
6942 let fits = match field.ty {
6943 LogicalType::TinyInt => i8::try_from(value).is_ok(),
6944 LogicalType::SmallInt => i16::try_from(value).is_ok(),
6945 LogicalType::Integer => i32::try_from(value).is_ok(),
6946 LogicalType::BigInt => true,
6947 _ => false,
6948 };
6949 if !fits {
6950 return Err(invalid("encoded integer value is outside its column type"));
6951 }
6952 }
6953 Ok(Some(counts))
6954 }
6955
6956 pub fn read_rows(
6969 &self,
6970 part: usize,
6971 columns: &[usize],
6972 positions: &[u32],
6973 whole: bool,
6974 ) -> Result<Chunk> {
6975 self.read_impl(part, columns, whole, Some(positions))
6976 }
6977
6978 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6985 if self.demoted(column) {
6988 return Ok(false);
6989 }
6990 if candidates.is_empty() {
6991 return Ok(true);
6992 }
6993 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6994 return Err(Error::internal("native code candidates are not sorted and unique"));
6995 }
6996 let stripe = self.stripe_of(part)?;
6997 let Some(page) = stripe.memberships.get(column) else {
6998 return Ok(false);
6999 };
7000 let mut bytes = vec![0; page.length as usize];
7001 read_at(&self.file, page.offset, &mut bytes)?;
7002 if checksum(&bytes) != page.hash {
7003 return Err(invalid("membership page checksum differs"));
7004 }
7005 let codes = decode_membership(&bytes)?;
7006 let mut left = 0;
7007 let mut right = 0;
7008 while left < codes.len() && right < candidates.len() {
7009 match codes[left].cmp(&candidates[right]) {
7010 Ordering::Less => left += 1,
7011 Ordering::Greater => right += 1,
7012 Ordering::Equal => return Ok(false),
7013 }
7014 }
7015 Ok(true)
7016 }
7017
7018 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7019 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7020 self.table
7021 .stripes
7022 .get(place.stripe as usize)
7023 .ok_or_else(|| invalid("stripe index out of range"))
7024 }
7025
7026 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
7043 let cache =
7044 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7045 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7046 let known = cached.index.get(at).and_then(Clone::clone);
7047 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7048 slot.used.store(true, Atomic::Relaxed);
7049 Arc::clone(&slot.page)
7050 });
7051 if let Some(index) = known.clone() {
7052 if !whole || page.is_some() {
7053 return Ok(CachedColumn { stripe: at, index, page });
7054 }
7055 }
7056 if cached.loading.contains(&at) {
7057 drop(cached);
7058 if let Some(index) = known {
7062 return Ok(CachedColumn { stripe: at, index, page: None });
7063 }
7064 let held = self.page_of(stripe, column, at, false, None)?;
7065 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7066 remember(&mut cached, &held);
7067 return Ok(held);
7068 }
7069 cached.loading.push(at);
7070 drop(cached);
7071
7072 let read = self.page_of(stripe, column, at, whole, known);
7073
7074 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7078 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7079 cached.loading.remove(position);
7080 }
7081 let held = read?;
7082 let taken = remember(&mut cached, &held);
7083 let first = taken.is_some()
7084 && cached.seen.get_mut(at).is_some_and(|seen| !std::mem::replace(seen, true));
7085 if first {
7086 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7087 cached.passing.push_back(at);
7088 while cached.passing.len() > floor {
7089 let Some(old) = cached.passing.pop_front() else { break };
7090 if let Some(slot) = cached.pages.get_mut(old) {
7091 *slot = None;
7092 }
7093 }
7094 return Ok(held);
7095 }
7096 drop(cached);
7097 if let Some((bytes, used)) = taken {
7098 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7099 self.pool.admit(Held {
7100 shelf: Arc::downgrade(&self.cache),
7101 column,
7102 stripe: at,
7103 bytes,
7104 used,
7105 });
7106 }
7107 Ok(held)
7108 }
7109
7110 fn page_of(
7116 &self,
7117 stripe: &Stripe,
7118 column: usize,
7119 at: usize,
7120 whole: bool,
7121 known: Option<Arc<Vec<PartSpan>>>,
7122 ) -> Result<CachedColumn> {
7123 let index = match known {
7124 Some(index) => index,
7125 None => {
7126 self.indexes.fetch_add(1, Atomic::Relaxed);
7127 Arc::new(read_index(&self.file, stripe, column)?)
7128 }
7129 };
7130 let page = if whole {
7131 self.pages.fetch_add(1, Atomic::Relaxed);
7132 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7133 let mut bytes = vec![0; span.length as usize];
7134 read_at(&self.file, span.offset, &mut bytes)?;
7135 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7136 Some(Arc::new(HeldPage { bytes, checked }))
7137 } else {
7138 None
7139 };
7140 Ok(CachedColumn { stripe: at, index, page })
7141 }
7142
7143 fn read_impl(
7144 &self,
7145 at: usize,
7146 columns: &[usize],
7147 whole: bool,
7148 positions: Option<&[u32]>,
7149 ) -> Result<Chunk> {
7150 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7151 let index = place.stripe as usize;
7152 let stripe =
7153 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7154 let rows = place.rows as usize;
7155 let mut picked = Vec::with_capacity(columns.len());
7156 for &column in columns {
7157 let field = self
7158 .table
7159 .fields
7160 .get(column)
7161 .ok_or_else(|| invalid("column index out of range"))?;
7162 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7163 let held = self.held(index, stripe, column, whole)?;
7164 let span = *held
7165 .index
7166 .get(place.part as usize)
7167 .ok_or_else(|| invalid("part index out of range"))?;
7168 let owned;
7169 let bytes = match &held.page {
7170 Some(held) => held.part(place.part as usize, span),
7171 None => {
7172 let offset = page
7173 .offset
7174 .checked_add(span.start as u64)
7175 .ok_or_else(|| invalid("part range overflow"))?;
7176 let mut bytes = vec![0; span.length];
7177 read_at(&self.file, offset, &mut bytes)?;
7178 owned = bytes;
7179 verify_part(&owned, span).map(|()| owned.as_slice())
7180 }
7181 }
7182 .map_err(|error| {
7183 invalid(&format!(
7184 "{}, column {column} part {} of the page at {}",
7185 error.message(),
7186 place.part,
7187 page.offset,
7188 ))
7189 })?;
7190 let dictionary = self.dictionary(column)?;
7191 let mut vector = match positions {
7197 None => decode(&field.ty, rows, bytes, dictionary)?,
7198 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7199 };
7200 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7204 vector = vector.flatten()?;
7205 }
7206 picked.push(vector.into_pages());
7207 }
7208 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7209 }
7210
7211 #[must_use]
7227 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7228 let Some(place) = self.places.get(part).copied() else { return false };
7229 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7230 if stripe.zone.skips(probes) {
7231 return true;
7232 }
7233 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7234 }
7235
7236 fn outside(&self, place: Place, probe: &Probe) -> bool {
7242 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7243 Some(ranges) => ranges
7244 .get(place.part as usize)
7245 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7246 None => false,
7247 }
7248 }
7249
7250 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7256 let slot = self.part_ranges.get(column)?.get(stripe)?;
7257 if let Some(held) = slot.get() {
7258 return Some(held);
7259 }
7260 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7261 let mut bytes = vec![0; page.length as usize];
7262 read_at(&self.file, page.offset, &mut bytes).ok()?;
7263 if checksum(&bytes) != page.hash {
7264 return None;
7265 }
7266 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7267 let _ = slot.set(ranges);
7268 slot.get().map(|held| held.as_slice())
7269 }
7270
7271 #[must_use]
7288 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7289 let Some(place) = self.places.get(part).copied() else { return false };
7290 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7291 if stripe.zone.certain(probes) {
7292 return true;
7293 }
7294 probes
7295 .iter()
7296 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7297 }
7298
7299 fn inside(&self, place: Place, probe: &Probe) -> bool {
7305 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7306 Some(ranges) => ranges
7307 .get(place.part as usize)
7308 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7309 None => false,
7310 }
7311 }
7312
7313 #[must_use]
7324 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7325 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7326 }
7327
7328 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7334 if probe.op != Op::Equal {
7335 return false;
7336 }
7337 match self.stripe_sieves(place.stripe as usize, probe.column) {
7338 Some(sieves) => sieves
7339 .get(place.part as usize)
7340 .and_then(Option::as_ref)
7341 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7342 None => false,
7343 }
7344 }
7345
7346 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7353 let slot = self.sieves.get(column)?.get(stripe)?;
7354 if let Some(held) = slot.get() {
7355 return Some(held);
7356 }
7357 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7358 let mut bytes = vec![0; page.length as usize];
7359 read_at(&self.file, page.offset, &mut bytes).ok()?;
7360 if checksum(&bytes) != page.hash {
7361 return None;
7362 }
7363 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7364 let _ = slot.set(sieves);
7365 slot.get().map(|held| held.as_slice())
7366 }
7367}
7368
7369fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7371 let code = dictionary.code_at_rank(rank)? as usize;
7372 if dictionary.logical_type() == &LogicalType::Blob {
7373 let bytes = dictionary
7374 .try_bytes_at(code)?
7375 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7376 return Ok(Value::Blob(bytes.to_vec()));
7377 }
7378 let text = dictionary
7379 .try_text_at(code)?
7380 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7381 Ok(Value::Varchar(text.into()))
7382}
7383
7384fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7394 file.fill_at(offset, bytes)
7395}
7396
7397trait Positional {
7405 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7410}
7411
7412impl<T: Positional + ?Sized> Positional for &T {
7413 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7414 (**self).fill_at(offset, bytes)
7415 }
7416}
7417
7418impl<T: Positional + ?Sized> Positional for Arc<T> {
7419 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7420 (**self).fill_at(offset, bytes)
7421 }
7422}
7423
7424impl<T: Positional + ?Sized> Positional for Box<T> {
7425 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7426 (**self).fill_at(offset, bytes)
7427 }
7428}
7429
7430impl Positional for dyn rudb_io::File + '_ {
7431 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7432 while !bytes.is_empty() {
7433 let read = self.read_at(offset, bytes)?;
7434 if read == 0 {
7435 return Err(invalid("column page ends before its declared length"));
7436 }
7437 offset += read as u64;
7438 bytes = &mut bytes[read..];
7439 }
7440 Ok(())
7441 }
7442}
7443
7444impl Positional for File {
7445 #[cfg(unix)]
7446 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7447 use std::os::unix::fs::FileExt;
7448 while !bytes.is_empty() {
7449 let read = self.read_at(bytes, offset).map_err(io)?;
7450 if read == 0 {
7451 return Err(invalid("column page ends before its declared length"));
7452 }
7453 offset += read as u64;
7454 bytes = &mut bytes[read..];
7455 }
7456 Ok(())
7457 }
7458
7459 #[cfg(windows)]
7465 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7466 use std::os::windows::fs::FileExt;
7467 while !bytes.is_empty() {
7468 let read = self.seek_read(bytes, offset).map_err(io)?;
7469 if read == 0 {
7470 return Err(invalid("column page ends before its declared length"));
7471 }
7472 offset += read as u64;
7473 bytes = &mut bytes[read..];
7474 }
7475 Ok(())
7476 }
7477
7478 #[cfg(not(any(unix, windows)))]
7483 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7484 use std::io::{Read, Seek, SeekFrom};
7485 let mut file = self.try_clone().map_err(io)?;
7486 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7487 file.read_exact(bytes).map_err(io)
7488 }
7489}
7490
7491#[cfg(test)]
7496fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7497 use std::io::{Seek, SeekFrom, Write};
7498 let mut file = file;
7499 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7500 file.write_all(bytes).map_err(io)
7501}
7502
7503fn type_tag(ty: &LogicalType) -> Result<u8> {
7510 match ty {
7511 LogicalType::SmallInt => Ok(1),
7512 LogicalType::Integer => Ok(2),
7513 LogicalType::BigInt => Ok(3),
7514 LogicalType::Varchar => Ok(4),
7515 LogicalType::Date => Ok(5),
7516 LogicalType::Timestamp => Ok(6),
7517 LogicalType::Boolean => Ok(7),
7518 LogicalType::TinyInt => Ok(8),
7519 LogicalType::UTinyInt => Ok(9),
7520 LogicalType::USmallInt => Ok(10),
7521 LogicalType::UInteger => Ok(11),
7522 LogicalType::UBigInt => Ok(12),
7523 LogicalType::Decimal { .. } => Ok(13),
7524 LogicalType::Float => Ok(14),
7525 LogicalType::Double => Ok(15),
7526 LogicalType::HugeInt => Ok(16),
7527 LogicalType::UHugeInt => Ok(17),
7528 LogicalType::Time => Ok(18),
7529 LogicalType::TimeTz => Ok(19),
7530 LogicalType::TimestampTz => Ok(20),
7531 LogicalType::Interval => Ok(21),
7532 LogicalType::Uuid => Ok(22),
7533 LogicalType::Blob => Ok(23),
7534 LogicalType::Bit => Ok(24),
7535 LogicalType::TimestampS => Ok(25),
7536 LogicalType::TimestampMs => Ok(26),
7537 LogicalType::TimestampNs => Ok(27),
7538 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7539 }
7540}
7541
7542fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7548 out.push(type_tag(ty)?);
7549 if let LogicalType::Decimal { width, scale } = ty {
7550 out.push(*width);
7551 out.push(*scale);
7552 }
7553 Ok(())
7554}
7555
7556fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7558 let tag = cur.u8()?;
7559 if tag == 13 {
7560 let width = cur.u8()?;
7561 let scale = cur.u8()?;
7562 return LogicalType::decimal(width, scale)
7563 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7564 }
7565 tag_type(tag)
7566}
7567
7568fn tag_type(tag: u8) -> Result<LogicalType> {
7569 match tag {
7570 1 => Ok(LogicalType::SmallInt),
7571 2 => Ok(LogicalType::Integer),
7572 3 => Ok(LogicalType::BigInt),
7573 4 => Ok(LogicalType::Varchar),
7574 5 => Ok(LogicalType::Date),
7575 6 => Ok(LogicalType::Timestamp),
7576 7 => Ok(LogicalType::Boolean),
7577 8 => Ok(LogicalType::TinyInt),
7578 9 => Ok(LogicalType::UTinyInt),
7579 10 => Ok(LogicalType::USmallInt),
7580 11 => Ok(LogicalType::UInteger),
7581 12 => Ok(LogicalType::UBigInt),
7582 14 => Ok(LogicalType::Float),
7583 15 => Ok(LogicalType::Double),
7584 16 => Ok(LogicalType::HugeInt),
7585 17 => Ok(LogicalType::UHugeInt),
7586 18 => Ok(LogicalType::Time),
7587 19 => Ok(LogicalType::TimeTz),
7588 20 => Ok(LogicalType::TimestampTz),
7589 21 => Ok(LogicalType::Interval),
7590 22 => Ok(LogicalType::Uuid),
7591 23 => Ok(LogicalType::Blob),
7592 24 => Ok(LogicalType::Bit),
7593 25 => Ok(LogicalType::TimestampS),
7594 26 => Ok(LogicalType::TimestampMs),
7595 27 => Ok(LogicalType::TimestampNs),
7596 _ => Err(invalid("column type tag is unknown")),
7597 }
7598}
7599
7600fn put_u16(out: &mut Vec<u8>, value: u16) {
7601 out.extend_from_slice(&value.to_le_bytes());
7602}
7603fn put_u32(out: &mut Vec<u8>, value: u32) {
7604 out.extend_from_slice(&value.to_le_bytes());
7605}
7606fn put_u64(out: &mut Vec<u8>, value: u64) {
7607 out.extend_from_slice(&value.to_le_bytes());
7608}
7609fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7610 while value >= 0x80 {
7611 out.push((value as u8 & 0x7f) | 0x80);
7612 value >>= 7;
7613 }
7614 out.push(value as u8);
7615}
7616
7617fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7618 match (left, right) {
7619 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7620 (FrequencyValue::Null, _) => Ordering::Less,
7621 (_, FrequencyValue::Null) => Ordering::Greater,
7622 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7623 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7624 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7625 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7626 }
7627}
7628
7629fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7642 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7643 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7644 };
7645 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7646 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7647 let omitted_max = next.count;
7648 entries.truncate(FREQUENCY_ENTRIES);
7649 omitted_max
7650 } else {
7651 0
7652 };
7653 entries.sort_unstable_by(order);
7654 omitted_max
7655}
7656
7657fn code_frequency(
7658 dictionary: &GlobalDictionary,
7659 flat: &[u8],
7660 bases: &[u64],
7661) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7662 let mut entries = dictionary
7663 .counts
7664 .iter()
7665 .enumerate()
7666 .filter(|(_, count)| **count != 0)
7667 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7668 .collect::<Vec<_>>();
7669 if dictionary.nulls != 0 {
7670 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7671 }
7672 let omitted_max = keep_most_frequent(&mut entries);
7673 let mut spans = Vec::with_capacity(entries.len());
7674 let mut text_bytes = 0_usize;
7675 for entry in &entries {
7676 let span = match entry.value {
7677 FrequencyValue::Code(code) => {
7678 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7679 let bytes = flat
7680 .get(span.0..span.1)
7681 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7682 text_bytes = text_bytes.saturating_add(bytes.len());
7683 Some(span)
7684 }
7685 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7686 };
7687 spans.push(span);
7688 }
7689 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7690 Vec::new()
7691 } else {
7692 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7693 };
7694 Ok((
7695 FrequencySummary {
7696 entries,
7697 omitted_max,
7698 ordinals: Vec::new(),
7699 ordinal_entries: Vec::new(),
7700 },
7701 texts,
7702 ))
7703}
7704
7705fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7706 let mut out = DIRECTORY.to_vec();
7707 let name = table.name.as_bytes();
7708 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7709 out.extend_from_slice(name);
7710 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7711 for field in &table.fields {
7712 let name = field.name.as_bytes();
7713 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7714 out.extend_from_slice(name);
7715 put_type(&mut out, &field.ty)?;
7716 out.push(u8::from(field.not_null));
7717 }
7718 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7719 match dictionary {
7720 None => out.push(0),
7721 Some(page) => {
7722 out.push(dictionary_tag(&field.ty));
7723 put_u64(&mut out, page.offset);
7724 put_u32(&mut out, page.length);
7725 put_u64(&mut out, page.hash);
7726 }
7727 }
7728 }
7729 for distinct in &table.distincts {
7730 match distinct {
7731 None => out.push(0),
7732 Some(count) => {
7733 out.push(1);
7734 put_u64(&mut out, *count);
7735 }
7736 }
7737 }
7738 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7739 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7740 for stripe in &table.stripes {
7741 put_u32(
7742 &mut out,
7743 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7744 );
7745 for &rows in &stripe.parts {
7746 put_u32(&mut out, rows);
7747 }
7748 put_u64(&mut out, stripe.index.offset);
7749 put_u32(&mut out, stripe.index.length);
7750 for page in &stripe.pages {
7751 put_u64(&mut out, page.offset);
7752 put_u32(&mut out, page.length);
7753 }
7754 for (column, ((field, dictionary), membership)) in
7759 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7760 {
7761 if !coded_type(&field.ty) || dictionary.is_none() {
7762 continue;
7763 }
7764 let page = match membership {
7765 Some(page) => page,
7766 None if table.demoted.get(column).copied().unwrap_or(false) => {
7767 Page { offset: HEADER, length: 0, hash: 0 }
7768 }
7769 None => return Err(invalid("string page has no code membership index")),
7770 };
7771 put_u64(&mut out, page.offset);
7772 put_u32(&mut out, page.length);
7773 put_u64(&mut out, page.hash);
7774 }
7775 for sieve in stripe.sieves.slots() {
7776 match sieve {
7777 None => out.push(0),
7778 Some(page) => {
7779 out.push(1);
7780 put_u64(&mut out, page.offset);
7781 put_u32(&mut out, page.length);
7782 put_u64(&mut out, page.hash);
7783 }
7784 }
7785 }
7786 for held in stripe.part_ranges.slots() {
7787 match held {
7788 None => out.push(0),
7789 Some(page) => {
7790 out.push(1);
7791 put_u64(&mut out, page.offset);
7792 put_u32(&mut out, page.length);
7793 put_u64(&mut out, page.hash);
7794 }
7795 }
7796 }
7797 for range in stripe.zone.columns() {
7798 put_bound(&mut out, range.low.as_ref())?;
7799 put_bound(&mut out, range.high.as_ref())?;
7800 put_u32(
7801 &mut out,
7802 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7803 );
7804 out.push(u8::from(range.exact));
7805 match range.sum {
7806 None => out.push(0),
7807 Some(total) => {
7808 out.push(1);
7809 out.extend_from_slice(&total.to_le_bytes());
7810 }
7811 }
7812 }
7813 }
7814 out.extend_from_slice(FREQUENCIES);
7815 put_u16(
7816 &mut out,
7817 u16::try_from(table.frequencies.len())
7818 .map_err(|_| invalid("too many frequency columns"))?,
7819 );
7820 for summary in &table.frequencies {
7821 let summary = match summary {
7822 None => {
7823 out.push(0);
7824 continue;
7825 }
7826 Some(Frequencies::Held(summary)) => summary,
7827 Some(Frequencies::Stored { .. }) => {
7829 return Err(invalid("a synopsis left in the file cannot be written back"));
7830 }
7831 };
7832 out.push(1);
7833 put_u64(&mut out, summary.omitted_max);
7834 put_u32(
7835 &mut out,
7836 u32::try_from(summary.entries.len())
7837 .map_err(|_| invalid("too many frequency entries"))?,
7838 );
7839 for entry in &summary.entries {
7840 match entry.value {
7841 FrequencyValue::Null => out.push(0),
7842 FrequencyValue::Integer(value) => {
7843 out.push(1);
7844 out.extend_from_slice(&value.to_le_bytes());
7845 }
7846 FrequencyValue::Code(value) => {
7847 out.push(2);
7848 put_u32(&mut out, value);
7849 }
7850 }
7851 put_u64(&mut out, entry.count);
7852 }
7853 put_u32(
7854 &mut out,
7855 u32::try_from(summary.ordinals.len())
7856 .map_err(|_| invalid("too many frequency ordinals"))?,
7857 );
7858 let mut previous = 0_u64;
7859 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
7860 let delta = if at == 0 {
7861 ordinal
7862 } else {
7863 ordinal
7864 .checked_sub(previous)
7865 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
7866 };
7867 if at != 0 && delta == 0 {
7868 return Err(invalid("frequency ordinals are not unique"));
7869 }
7870 put_var_u64(&mut out, delta);
7871 previous = ordinal;
7872 }
7873 if summary.ordinal_entries.len() != summary.ordinals.len() {
7874 return Err(invalid("frequency ordinal values have a different length"));
7875 }
7876 for &entry in &summary.ordinal_entries {
7877 if entry as usize >= summary.entries.len() {
7878 return Err(invalid("frequency ordinal value is outside its entries"));
7879 }
7880 put_u16(&mut out, entry);
7881 }
7882 }
7883 if !table.pair_frequencies.is_empty() {
7884 out.extend_from_slice(PAIR_FREQUENCIES);
7885 put_u16(
7886 &mut out,
7887 u16::try_from(table.pair_frequencies.len())
7888 .map_err(|_| invalid("too many pair frequency summaries"))?,
7889 );
7890 for summary in &table.pair_frequencies {
7891 put_u16(&mut out, summary.first);
7892 put_u16(&mut out, summary.second);
7893 put_u64(&mut out, summary.omitted_max);
7894 put_u16(
7895 &mut out,
7896 u16::try_from(summary.entries.len())
7897 .map_err(|_| invalid("too many pair frequency entries"))?,
7898 );
7899 for entry in &summary.entries {
7900 put_u16(&mut out, entry.first_entry);
7901 match entry.second {
7902 None => out.push(0),
7903 Some(code) => {
7904 out.push(1);
7905 put_u32(&mut out, code);
7906 }
7907 }
7908 put_u64(&mut out, entry.count);
7909 }
7910 }
7911 }
7912 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7913 if text_columns != 0 {
7914 out.extend_from_slice(FREQUENCY_TEXTS);
7915 put_u16(
7916 &mut out,
7917 u16::try_from(text_columns)
7918 .map_err(|_| invalid("too many string frequency columns"))?,
7919 );
7920 for (column, texts) in table.frequency_texts.iter().enumerate() {
7921 if texts.is_empty() {
7922 continue;
7923 }
7924 put_u16(
7925 &mut out,
7926 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7927 );
7928 put_u16(
7929 &mut out,
7930 u16::try_from(texts.len())
7931 .map_err(|_| invalid("too many frequency text entries"))?,
7932 );
7933 for text in texts {
7934 match text {
7935 None => out.push(0),
7936 Some(text) => {
7937 out.push(1);
7938 put_u32(
7939 &mut out,
7940 u32::try_from(text.len())
7941 .map_err(|_| invalid("frequency text is too long"))?,
7942 );
7943 out.extend_from_slice(text);
7944 }
7945 }
7946 }
7947 }
7948 }
7949 if let Some(summary) = &table.host_groups {
7950 out.extend_from_slice(HOST_GROUPS);
7951 put_u16(
7952 &mut out,
7953 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7954 );
7955 put_u64(&mut out, summary.omitted_max);
7956 put_u16(
7957 &mut out,
7958 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7959 );
7960 for entry in &summary.entries {
7961 put_u32(
7962 &mut out,
7963 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7964 );
7965 out.extend_from_slice(entry.host.as_bytes());
7966 put_u64(&mut out, entry.count);
7967 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7968 put_u32(
7969 &mut out,
7970 u32::try_from(entry.minimum.len())
7971 .map_err(|_| invalid("host minimum is too long"))?,
7972 );
7973 out.extend_from_slice(entry.minimum.as_bytes());
7974 }
7975 }
7976 if let Some(clustering) = &table.clustering {
7979 out.extend_from_slice(CLUSTERING);
7980 out.push(clustering.width().tag());
7981 put_u16(
7982 &mut out,
7983 u16::try_from(clustering.columns().len())
7984 .map_err(|_| invalid("too many clustering columns"))?,
7985 );
7986 for &column in clustering.columns() {
7987 put_u16(
7988 &mut out,
7989 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7990 );
7991 }
7992 }
7993 let demoted = (0..table.fields.len())
7994 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
7995 .collect::<Vec<_>>();
7996 if !demoted.is_empty() {
7997 out.extend_from_slice(DEMOTED);
7998 put_u16(
7999 &mut out,
8000 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8001 );
8002 for column in demoted {
8003 put_u16(
8004 &mut out,
8005 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8006 );
8007 }
8008 }
8009 out.extend_from_slice(SECTIONS);
8015 put_u64(&mut out, table.generation);
8016 put_u16(
8017 &mut out,
8018 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8019 );
8020 for held in &table.sections {
8021 held.encode(&mut out)?;
8022 }
8023 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8024 out.extend_from_slice(DICTIONARY_PAYLOADS);
8025 put_u16(
8026 &mut out,
8027 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8028 );
8029 for at in 0..table.fields.len() {
8030 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8031 }
8032 }
8033 Ok(out)
8034}
8035
8036fn signed_integer(ty: &LogicalType) -> bool {
8045 matches!(
8046 ty,
8047 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8048 )
8049}
8050
8051fn integer_or_date(ty: &LogicalType) -> bool {
8052 matches!(
8053 ty,
8054 LogicalType::TinyInt
8055 | LogicalType::SmallInt
8056 | LogicalType::Integer
8057 | LogicalType::BigInt
8058 | LogicalType::UTinyInt
8059 | LogicalType::USmallInt
8060 | LogicalType::UInteger
8061 | LogicalType::UBigInt
8062 | LogicalType::Date
8063 )
8064}
8065
8066fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8067 table
8068 .fields
8069 .iter()
8070 .enumerate()
8071 .map(|(column, field)| {
8072 if !integer_or_date(&field.ty) {
8073 return None;
8074 }
8075 let mut low: Option<i128> = None;
8076 let mut high: Option<i128> = None;
8077 for stripe in &table.stripes {
8078 let range = stripe.zone.column(column)?;
8079 if !range.exact {
8080 return None;
8081 }
8082 match (range.low.as_ref(), range.high.as_ref()) {
8083 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8084 low = Some(low.map_or(*small, |held| held.min(*small)));
8085 high = Some(high.map_or(*large, |held| held.max(*large)));
8086 }
8087 (None, None) if stripe.rows == range.nulls => {}
8088 _ => return None,
8089 }
8090 }
8091 Some(low.zip(high))
8092 })
8093 .collect()
8094}
8095
8096fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8097 reader
8098 .table
8099 .fields
8100 .iter()
8101 .enumerate()
8102 .map(|(column, field)| {
8103 if !integer_or_date(&field.ty) {
8104 return Ok(None);
8105 }
8106 match reader.exact_extremes(column)? {
8107 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8108 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8109 _ => Ok(None),
8110 }
8111 })
8112 .collect()
8113}
8114
8115fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8116 table
8117 .fields
8118 .iter()
8119 .enumerate()
8120 .map(|(column, field)| {
8121 if !integer_or_date(&field.ty) {
8122 return None;
8123 }
8124 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8125 return None;
8126 };
8127 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8128 return None;
8129 }
8130 let entries = summary
8131 .entries
8132 .iter()
8133 .map(|entry| {
8134 let value = match entry.value {
8135 FrequencyValue::Null => None,
8136 FrequencyValue::Integer(value) => Some(value),
8137 FrequencyValue::Code(_) => return None,
8138 };
8139 Some((value, entry.count))
8140 })
8141 .collect::<Option<Vec<_>>>()?;
8142 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8143 (rows == table.rows as u64).then_some(entries)
8144 })
8145 .collect()
8146}
8147
8148fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8155 if signed {
8156 FrequencyValue::Integer(i128::from(bits as i64))
8157 } else {
8158 FrequencyValue::Integer(i128::from(bits))
8159 }
8160}
8161
8162fn frequency_bits(value: &Value) -> Option<u64> {
8163 Some(match value {
8164 Value::TinyInt(value) => i64::from(*value) as u64,
8165 Value::SmallInt(value) => i64::from(*value) as u64,
8166 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8167 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8168 Value::UTinyInt(value) => u64::from(*value),
8169 Value::USmallInt(value) => u64::from(*value),
8170 Value::UInteger(value) => u64::from(*value),
8171 Value::UBigInt(value) => *value,
8172 _ => return None,
8173 })
8174}
8175
8176fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8177 Some(match value {
8178 Value::Null => None,
8179 Value::TinyInt(value) => Some(i128::from(*value)),
8180 Value::SmallInt(value) => Some(i128::from(*value)),
8181 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8182 Value::BigInt(value) => Some(i128::from(*value)),
8183 Value::UTinyInt(value) => Some(i128::from(*value)),
8184 Value::USmallInt(value) => Some(i128::from(*value)),
8185 Value::UInteger(value) => Some(i128::from(*value)),
8186 Value::UBigInt(value) => Some(i128::from(*value)),
8187 _ => return None,
8188 })
8189}
8190
8191fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8192 reader
8193 .table
8194 .fields
8195 .iter()
8196 .enumerate()
8197 .map(|(column, field)| {
8198 if !integer_or_date(&field.ty) {
8199 return Ok(None);
8200 }
8201 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8202 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8203 return Ok(None);
8204 }
8205 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8206 let Some(entries) = entries
8207 .iter()
8208 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8209 .collect::<Option<Vec<_>>>()
8210 else {
8211 return Ok(None);
8212 };
8213 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8214 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8215 })
8216 .collect()
8217}
8218
8219fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8220 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8221 let range = stripe.zone.column(column)?;
8222 let sum = sum.checked_add(range.sum?)?;
8223 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8224 Some((sum, count.checked_add(nonnull)?))
8225 })
8226}
8227
8228fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8229 table
8230 .fields
8231 .iter()
8232 .enumerate()
8233 .map(|(column, field)| {
8234 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8235 })
8236 .collect()
8237}
8238
8239fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8240 reader
8241 .table
8242 .fields
8243 .iter()
8244 .enumerate()
8245 .map(
8246 |(column, field)| {
8247 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8248 },
8249 )
8250 .collect()
8251}
8252
8253fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8254 let mut out = CATALOG.to_vec();
8255 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8256 for entry in entries {
8257 let name = entry.name.as_bytes();
8258 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8259 out.extend_from_slice(name);
8260 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8261 put_u16(
8262 &mut out,
8263 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8264 );
8265 for field in &entry.fields {
8266 let name = field.name.as_bytes();
8267 put_u16(
8268 &mut out,
8269 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8270 );
8271 out.extend_from_slice(name);
8272 put_type(&mut out, &field.ty)?;
8273 out.push(u8::from(field.not_null));
8274 }
8275 put_u64(&mut out, entry.directory.offset);
8276 put_u32(&mut out, entry.directory.length);
8277 put_u64(&mut out, entry.directory.hash);
8278 }
8279 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8280 for view in views {
8281 let name = view.name.as_bytes();
8282 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8283 out.extend_from_slice(name);
8284 put_long_text(&mut out, &view.sql, "view body")?;
8285 put_long_text(&mut out, &view.statement, "view statement")?;
8286 put_u16(
8287 &mut out,
8288 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8289 );
8290 for alias in &view.aliases {
8291 let alias = alias.as_bytes();
8292 put_u16(
8293 &mut out,
8294 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8295 );
8296 out.extend_from_slice(alias);
8297 }
8298 put_u16(
8299 &mut out,
8300 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8301 );
8302 for field in &view.columns {
8303 let name = field.name.as_bytes();
8304 put_u16(
8305 &mut out,
8306 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8307 );
8308 out.extend_from_slice(name);
8309 put_type(&mut out, &field.ty)?;
8310 out.push(u8::from(field.not_null));
8311 }
8312 }
8313 out.extend_from_slice(NONZERO_COUNTS);
8314 for entry in entries {
8315 if entry.nonzero.len() != entry.fields.len() {
8316 return Err(invalid("nonzero count width differs from schema"));
8317 }
8318 for count in &entry.nonzero {
8319 match count {
8320 None => out.push(0),
8321 Some(count) => {
8322 out.push(1);
8323 put_u64(&mut out, *count);
8324 }
8325 }
8326 }
8327 }
8328 out.extend_from_slice(AGGREGATE_SUMS);
8329 for entry in entries {
8330 if entry.aggregates.len() != entry.fields.len() {
8331 return Err(invalid("aggregate sum width differs from schema"));
8332 }
8333 for summary in &entry.aggregates {
8334 match summary {
8335 None => out.push(0),
8336 Some((sum, count)) => {
8337 out.push(1);
8338 out.extend_from_slice(&sum.to_le_bytes());
8339 put_u64(&mut out, *count);
8340 }
8341 }
8342 }
8343 }
8344 out.extend_from_slice(DISTINCT_COUNTS);
8345 for entry in entries {
8346 if entry.distincts.len() != entry.fields.len() {
8347 return Err(invalid("distinct count width differs from schema"));
8348 }
8349 for count in &entry.distincts {
8350 match count {
8351 None => out.push(0),
8352 Some(count) => {
8353 if *count > entry.rows as u64 {
8354 return Err(invalid("distinct count exceeds table rows"));
8355 }
8356 out.push(1);
8357 put_u64(&mut out, *count);
8358 }
8359 }
8360 }
8361 }
8362 out.extend_from_slice(INTEGER_EXTREMES);
8363 for entry in entries {
8364 if entry.extremes.len() != entry.fields.len() {
8365 return Err(invalid("integer extremes width differs from schema"));
8366 }
8367 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8368 match extremes {
8369 None => out.push(0),
8370 Some(None) if integer_or_date(&field.ty) => out.push(1),
8371 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8372 out.push(2);
8373 out.extend_from_slice(&low.to_le_bytes());
8374 out.extend_from_slice(&high.to_le_bytes());
8375 }
8376 _ => return Err(invalid("integer extremes type or range differs")),
8377 }
8378 }
8379 }
8380 out.extend_from_slice(COMPLETE_FREQUENCIES);
8381 for entry in entries {
8382 if entry.frequencies.len() != entry.fields.len() {
8383 return Err(invalid("numeric frequency width differs from schema"));
8384 }
8385 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8386 match frequencies {
8387 None => out.push(0),
8388 Some(entries)
8389 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8390 {
8391 let mut total = 0_u64;
8392 for (at, (value, count)) in entries.iter().enumerate() {
8393 if entries[..at].iter().any(|(held, _)| held == value) {
8394 return Err(invalid("numeric frequency value repeats"));
8395 }
8396 total = total
8397 .checked_add(*count)
8398 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8399 }
8400 if total != entry.rows as u64 {
8401 return Err(invalid("numeric frequencies do not cover table rows"));
8402 }
8403 out.push(1);
8404 out.push(entries.len() as u8);
8405 for (value, count) in entries {
8406 match value {
8407 None => out.push(0),
8408 Some(value) => {
8409 out.push(1);
8410 out.extend_from_slice(&value.to_le_bytes());
8411 }
8412 }
8413 put_u64(&mut out, *count);
8414 }
8415 }
8416 _ => return Err(invalid("numeric frequency type or width differs")),
8417 }
8418 }
8419 }
8420 Ok(out)
8421}
8422
8423fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8425 let bytes = text.as_bytes();
8426 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8427 out.extend_from_slice(bytes);
8428 Ok(())
8429}
8430
8431fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8434 let mut cur = Cursor::new(bytes);
8435 if cur.take(8)? != CATALOG {
8436 return Err(invalid("catalog magic differs"));
8437 }
8438 let count = cur.u32()? as usize;
8439 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8440 for _ in 0..count {
8441 let name = cur.text()?;
8442 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8443 let width = cur.u16()? as usize;
8444 let mut fields = Vec::with_capacity(width);
8445 for _ in 0..width {
8446 let name = cur.text()?;
8447 let ty = read_type(&mut cur)?;
8448 let not_null = match cur.u8()? {
8449 0 => false,
8450 1 => true,
8451 _ => return Err(invalid("nullability flag differs")),
8452 };
8453 fields.push(Field { name, ty, not_null });
8454 }
8455 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8456 let end = directory
8457 .offset
8458 .checked_add(u64::from(directory.length))
8459 .ok_or_else(|| invalid("table directory offset overflow"))?;
8460 if directory.offset < HEADER
8461 || end > size
8462 || directory.length as usize > MAX_DIRECTORY
8463 || directory.length == 0
8464 {
8465 return Err(invalid("table directory range is outside the file"));
8466 }
8467 if entries.iter().any(|held| held.name == name) {
8468 return Err(invalid("two tables in the catalog have the same name"));
8469 }
8470 let nonzero = vec![None; fields.len()];
8471 let aggregates = vec![None; fields.len()];
8472 let distincts = vec![None; fields.len()];
8473 let extremes = vec![None; fields.len()];
8474 let frequencies = vec![None; fields.len()];
8475 entries.push(Entry {
8476 name,
8477 fields,
8478 rows,
8479 directory,
8480 nonzero,
8481 aggregates,
8482 distincts,
8483 extremes,
8484 frequencies,
8485 });
8486 }
8487 let count = if cur.done() { 0 } else { cur.u32()? as usize };
8492 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8493 for _ in 0..count {
8494 let name = cur.text()?;
8495 let sql = cur.long_text()?;
8496 let statement = cur.long_text()?;
8497 let width = cur.u16()? as usize;
8498 let mut aliases = Vec::with_capacity(width);
8499 for _ in 0..width {
8500 aliases.push(cur.text()?);
8501 }
8502 let width = cur.u16()? as usize;
8503 let mut columns = Vec::with_capacity(width);
8504 for _ in 0..width {
8505 let name = cur.text()?;
8506 let ty = read_type(&mut cur)?;
8507 let not_null = match cur.u8()? {
8508 0 => false,
8509 1 => true,
8510 _ => return Err(invalid("nullability flag differs")),
8511 };
8512 columns.push(Field { name, ty, not_null });
8513 }
8514 if views.iter().any(|held| held.name == name) {
8518 return Err(invalid("two views in the catalog have the same name"));
8519 }
8520 if entries.iter().any(|held| held.name == name) {
8521 return Err(invalid("a table and a view in the catalog have the same name"));
8522 }
8523 views.push(ViewEntry { name, sql, statement, aliases, columns });
8524 }
8525 if !cur.done() {
8526 if cur.take(8)? != NONZERO_COUNTS {
8527 return Err(invalid("catalog extension magic differs"));
8528 }
8529 for entry in &mut entries {
8530 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8531 *count = match cur.u8()? {
8532 0 => None,
8533 1 if matches!(
8534 field.ty,
8535 LogicalType::TinyInt
8536 | LogicalType::SmallInt
8537 | LogicalType::Integer
8538 | LogicalType::BigInt
8539 | LogicalType::UTinyInt
8540 | LogicalType::USmallInt
8541 | LogicalType::UInteger
8542 | LogicalType::UBigInt
8543 ) =>
8544 {
8545 let value = cur.u64()?;
8546 if value > entry.rows as u64 {
8547 return Err(invalid("nonzero count exceeds rows"));
8548 }
8549 Some(value)
8550 }
8551 _ => return Err(invalid("nonzero count tag or column type differs")),
8552 };
8553 }
8554 }
8555 }
8556 if !cur.done() {
8557 if cur.take(8)? != AGGREGATE_SUMS {
8558 return Err(invalid("aggregate catalog extension magic differs"));
8559 }
8560 for entry in &mut entries {
8561 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8562 *summary = match cur.u8()? {
8563 0 => None,
8564 1 if signed_integer(&field.ty) => {
8565 let sum = i128::from_le_bytes(
8566 cur.take(16)?
8567 .try_into()
8568 .map_err(|_| invalid("aggregate sum is truncated"))?,
8569 );
8570 let count = cur.u64()?;
8571 if count > entry.rows as u64 {
8572 return Err(invalid("aggregate count exceeds table rows"));
8573 }
8574 Some((sum, count))
8575 }
8576 _ => return Err(invalid("aggregate sum tag or column type differs")),
8577 };
8578 }
8579 }
8580 }
8581 if !cur.done() {
8582 if cur.take(8)? != DISTINCT_COUNTS {
8583 return Err(invalid("distinct catalog extension magic differs"));
8584 }
8585 for entry in &mut entries {
8586 for count in &mut entry.distincts {
8587 *count = match cur.u8()? {
8588 0 => None,
8589 1 => {
8590 let value = cur.u64()?;
8591 if value > entry.rows as u64 {
8592 return Err(invalid("distinct count exceeds table rows"));
8593 }
8594 Some(value)
8595 }
8596 _ => return Err(invalid("distinct count tag differs")),
8597 };
8598 }
8599 }
8600 }
8601 if !cur.done() {
8602 if cur.take(8)? != INTEGER_EXTREMES {
8603 return Err(invalid("integer extremes catalog extension magic differs"));
8604 }
8605 for entry in &mut entries {
8606 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8607 *extremes = match cur.u8()? {
8608 0 => None,
8609 1 if integer_or_date(&field.ty) => Some(None),
8610 2 if integer_or_date(&field.ty) => {
8611 let low = i128::from_le_bytes(
8612 cur.take(16)?
8613 .try_into()
8614 .map_err(|_| invalid("minimum is truncated"))?,
8615 );
8616 let high = i128::from_le_bytes(
8617 cur.take(16)?
8618 .try_into()
8619 .map_err(|_| invalid("maximum is truncated"))?,
8620 );
8621 if low > high {
8622 return Err(invalid("integer extremes are reversed"));
8623 }
8624 Some(Some((low, high)))
8625 }
8626 _ => return Err(invalid("integer extremes tag or type differs")),
8627 };
8628 }
8629 }
8630 }
8631 if !cur.done() {
8632 if cur.take(8)? != COMPLETE_FREQUENCIES {
8633 return Err(invalid("numeric frequency catalog extension magic differs"));
8634 }
8635 for entry in &mut entries {
8636 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8637 *frequencies = match cur.u8()? {
8638 0 => None,
8639 1 if integer_or_date(&field.ty) => {
8640 let len = cur.u8()? as usize;
8641 if len > MAX_CATALOG_FREQUENCIES {
8642 return Err(invalid("too many catalog numeric frequencies"));
8643 }
8644 let mut values = Vec::with_capacity(len);
8645 let mut total = 0_u64;
8646 for _ in 0..len {
8647 let value = match cur.u8()? {
8648 0 => None,
8649 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8650 |_| invalid("numeric frequency value is truncated"),
8651 )?)),
8652 _ => return Err(invalid("numeric frequency value tag differs")),
8653 };
8654 if values.iter().any(|(held, _)| *held == value) {
8655 return Err(invalid("numeric frequency value repeats"));
8656 }
8657 let count = cur.u64()?;
8658 total = total
8659 .checked_add(count)
8660 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8661 values.push((value, count));
8662 }
8663 if total != entry.rows as u64 {
8664 return Err(invalid("numeric frequencies do not cover table rows"));
8665 }
8666 Some(values)
8667 }
8668 _ => return Err(invalid("numeric frequency tag or type differs")),
8669 };
8670 }
8671 }
8672 }
8673 if !cur.done() {
8674 return Err(invalid("catalog has trailing bytes"));
8675 }
8676 Ok((entries, views))
8677}
8678
8679struct Cursor<'a> {
8687 bytes: &'a [u8],
8688 at: usize,
8689 window: Option<Window<'a>>,
8690}
8691
8692struct Window<'a> {
8694 file: &'a File,
8695 offset: u64,
8696 length: usize,
8697 start: usize,
8699 held: Vec<u8>,
8700 size: usize,
8702}
8703
8704const DIRECTORY_WINDOW: usize = 64 << 10;
8706
8707impl<'a> Cursor<'a> {
8708 fn new(bytes: &'a [u8]) -> Self {
8709 Self { bytes, at: 0, window: None }
8710 }
8711
8712 fn over(file: &'a File, offset: u64, length: usize) -> Self {
8714 let window =
8715 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8716 Self { bytes: &[], at: 0, window: Some(window) }
8717 }
8718
8719 fn len(&self) -> usize {
8721 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8722 }
8723
8724 fn ensure(&mut self, len: usize) -> Result<()> {
8726 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8727 if end > self.len() {
8728 return Err(invalid("directory is truncated"));
8729 }
8730 let Some(window) = &mut self.window else { return Ok(()) };
8731 if self.at < window.start || end > window.start + window.held.len() {
8732 let want = len.max(window.size).min(window.length - self.at);
8733 window.start = self.at;
8734 window.held.resize(want, 0);
8735 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8736 }
8737 Ok(())
8738 }
8739
8740 fn held(&self, at: usize, len: usize) -> &[u8] {
8742 match &self.window {
8743 Some(window) => &window.held[at - window.start..at - window.start + len],
8744 None => &self.bytes[at..at + len],
8745 }
8746 }
8747
8748 #[inline]
8750 fn peek(&mut self, len: usize) -> Result<&[u8]> {
8751 if self.window.is_none() {
8752 let bytes = self.bytes;
8753 return Ok(&bytes[self.at..self.end(len)?]);
8754 }
8755 self.ensure(len)?;
8756 Ok(self.held(self.at, len))
8757 }
8758
8759 #[inline]
8765 fn take(&mut self, len: usize) -> Result<&[u8]> {
8766 if self.window.is_none() {
8767 let bytes = self.bytes;
8768 let (at, end) = (self.at, self.end(len)?);
8769 self.at = end;
8770 return Ok(&bytes[at..end]);
8771 }
8772 self.take_windowed(len)
8773 }
8774
8775 fn skip(&mut self, len: usize) -> Result<()> {
8777 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8778 if end > self.len() {
8779 return Err(invalid("directory is truncated"));
8780 }
8781 self.at = end;
8782 Ok(())
8783 }
8784
8785 fn skip_bound(&mut self) -> Result<()> {
8786 match self.u8()? {
8787 0 => Ok(()),
8788 1 => self.skip(16),
8789 2 => self.skip(8),
8790 3 => {
8791 let length = self.u32()? as usize;
8792 self.skip(length)
8793 }
8794 4 => self.skip(17),
8795 _ => Err(invalid("a stored bound has an unknown tag")),
8796 }
8797 }
8798
8799 #[inline]
8801 fn end(&self, len: usize) -> Result<usize> {
8802 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8803 if end > self.bytes.len() {
8804 return Err(invalid("directory is truncated"));
8805 }
8806 Ok(end)
8807 }
8808
8809 #[inline(never)]
8811 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8812 self.ensure(len)?;
8813 self.at += len;
8814 Ok(self.held(self.at - len, len))
8815 }
8816 #[inline]
8817 fn u8(&mut self) -> Result<u8> {
8818 Ok(self.take(1)?[0])
8819 }
8820 #[inline]
8821 fn u16(&mut self) -> Result<u16> {
8822 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8823 }
8824 #[inline]
8825 fn u32(&mut self) -> Result<u32> {
8826 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8827 }
8828 #[inline]
8829 fn u64(&mut self) -> Result<u64> {
8830 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8831 }
8832 fn var_u64(&mut self) -> Result<u64> {
8833 let mut value = 0_u64;
8834 for shift in (0..=63).step_by(7) {
8835 let byte = self.u8()?;
8836 let part = u64::from(byte & 0x7f);
8837 if shift == 63 && part > 1 {
8838 return Err(invalid("frequency ordinal varint overflows"));
8839 }
8840 value |= part << shift;
8841 if byte & 0x80 == 0 {
8842 return Ok(value);
8843 }
8844 }
8845 Err(invalid("frequency ordinal varint is too long"))
8846 }
8847 fn bound(&mut self) -> Result<Option<Bound>> {
8856 let rest = self.len().saturating_sub(self.at);
8857 let mut want = 32;
8858 loop {
8859 let offered = self.peek(want.min(rest))?;
8860 let mut used = 0;
8861 match bounds::get(offered, &mut used) {
8862 Ok(bound) => {
8863 self.at += used;
8864 return Ok(bound);
8865 }
8866 Err(_) if want < rest => want *= 2,
8867 Err(error) => return Err(error),
8868 }
8869 }
8870 }
8871 fn text(&mut self) -> Result<String> {
8872 let len = self.u16()? as usize;
8873 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8874 }
8875 fn done(&self) -> bool {
8878 self.at >= self.len()
8879 }
8880 fn long_text(&mut self) -> Result<String> {
8887 let len = self.u32()? as usize;
8888 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8889 }
8890}
8891
8892fn decode_summary(
8894 cur: &mut Cursor<'_>,
8895 field: &Field,
8896 rows: usize,
8897 values: bool,
8898) -> Result<Option<FrequencySummary>> {
8899 Ok(match cur.u8()? {
8900 0 => None,
8901 1 => {
8902 let omitted_max = cur.u64()?;
8903 let count = cur.u32()? as usize;
8904 if count > FREQUENCY_ENTRIES {
8905 return Err(invalid("frequency entry count exceeds its bound"));
8906 }
8907 let mut entries = Vec::with_capacity(count);
8908 for _ in 0..count {
8910 let value = match cur.u8()? {
8911 0 => FrequencyValue::Null,
8912 1 => FrequencyValue::Integer(i128::from_le_bytes(
8913 cur.take(16)?.try_into().expect("sixteen bytes"),
8914 )),
8915 2 => FrequencyValue::Code(cur.u32()?),
8916 _ => return Err(invalid("frequency value tag differs")),
8917 };
8918 let valid = matches!(
8919 (&field.ty, value),
8920 (_, FrequencyValue::Null)
8921 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
8922 | (
8923 LogicalType::TinyInt
8924 | LogicalType::SmallInt
8925 | LogicalType::Integer
8926 | LogicalType::BigInt
8927 | LogicalType::UTinyInt
8928 | LogicalType::USmallInt
8929 | LogicalType::UInteger
8930 | LogicalType::UBigInt
8931 | LogicalType::Date
8932 | LogicalType::Timestamp,
8933 FrequencyValue::Integer(_),
8934 )
8935 );
8936 if !valid {
8937 return Err(invalid("frequency value does not match its column"));
8938 }
8939 let count = cur.u64()?;
8940 if count == 0 || count > rows as u64 {
8941 return Err(invalid("frequency count is outside the table"));
8942 }
8943 entries.push(FrequencyEntry { value, count });
8944 }
8945 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8946 return Err(invalid("frequency entries are not descending"));
8947 }
8948 let ordinals = {
8949 let ordinal_count = cur.u32()? as usize;
8950 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8951 return Err(invalid("frequency ordinal count exceeds its bound"));
8952 }
8953 let mut ordinals = Vec::with_capacity(ordinal_count);
8954 let mut previous = 0_u64;
8955 for at in 0..ordinal_count {
8956 let delta = cur.var_u64()?;
8957 if at != 0 && delta == 0 {
8958 return Err(invalid("frequency ordinals are not increasing"));
8959 }
8960 let ordinal = if at == 0 {
8961 delta
8962 } else {
8963 previous
8964 .checked_add(delta)
8965 .ok_or_else(|| invalid("frequency ordinal overflows"))?
8966 };
8967 if ordinal >= rows as u64 {
8968 return Err(invalid("frequency ordinal is outside the table"));
8969 }
8970 ordinals.push(ordinal);
8971 previous = ordinal;
8972 }
8973 ordinals
8974 };
8975 let ordinal_entries = if values {
8976 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8977 for _ in 0..ordinals.len() {
8978 let entry = cur.u16()?;
8979 if entry as usize >= entries.len() {
8980 return Err(invalid("frequency ordinal value is outside its entries"));
8981 }
8982 ordinal_entries.push(entry);
8983 }
8984 ordinal_entries
8985 } else {
8986 Vec::new()
8987 };
8988 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8989 }
8990 _ => return Err(invalid("frequency summary tag differs")),
8991 })
8992}
8993
8994fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8997 match cur.u8()? {
8998 0 => Ok(()),
8999 1 => {
9000 cur.skip(8)?;
9001 let entries = cur.u32()? as usize;
9002 if entries > FREQUENCY_ENTRIES {
9003 return Err(invalid("frequency entry count exceeds its bound"));
9004 }
9005 for _ in 0..entries {
9006 match cur.u8()? {
9007 0 => {}
9008 1 => cur.skip(16)?,
9009 2 => cur.skip(4)?,
9010 _ => return Err(invalid("frequency value tag differs")),
9011 }
9012 cur.skip(8)?;
9013 }
9014 let ordinals = cur.u32()? as usize;
9015 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9016 return Err(invalid("frequency ordinal count exceeds its bound"));
9017 }
9018 for _ in 0..ordinals {
9019 cur.var_u64()?;
9020 }
9021 if values {
9022 cur.skip(ordinals * 2)?;
9023 }
9024 Ok(())
9025 }
9026 _ => Err(invalid("frequency summary tag differs")),
9027 }
9028}
9029
9030fn quick_nonzero(
9034 mut cur: Cursor<'_>,
9035 name: &str,
9036 fields: &[Field],
9037 rows: usize,
9038 wanted: usize,
9039) -> Result<Option<u64>> {
9040 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9041 return Err(invalid("table directory differs from the catalog"));
9042 }
9043 let width = cur.u16()? as usize;
9044 if width != fields.len() {
9045 return Err(invalid("table directory width differs from the catalog"));
9046 }
9047 for field in fields {
9048 let stored =
9049 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9050 if &stored != field {
9051 return Err(invalid("table directory schema differs from the catalog"));
9052 }
9053 }
9054 let mut dictionaries = Vec::with_capacity(width);
9055 for field in fields {
9056 let held = match cur.u8()? {
9057 0 => false,
9058 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9059 cur.skip(20)?;
9060 true
9061 }
9062 _ => return Err(invalid("dictionary page tag differs")),
9063 };
9064 dictionaries.push(held);
9065 }
9066 for _ in 0..width {
9067 match cur.u8()? {
9068 0 => {}
9069 1 => cur.skip(8)?,
9070 _ => return Err(invalid("distinct count tag differs")),
9071 }
9072 }
9073 if cur.u64()? != rows as u64 {
9074 return Err(invalid("table row count differs from the catalog"));
9075 }
9076 let stripes = cur.u32()? as usize;
9077 let mut total = 0_usize;
9078 let mut nulls = 0_u64;
9079 for _ in 0..stripes {
9080 let parts = cur.u32()? as usize;
9081 if parts == 0 || parts > STRIPE_PARTS {
9082 return Err(invalid("stripe part count is outside its bound"));
9083 }
9084 let mut stripe_rows = 0_usize;
9085 for _ in 0..parts {
9086 stripe_rows = stripe_rows
9087 .checked_add(cur.u32()? as usize)
9088 .ok_or_else(|| invalid("stripe row count overflow"))?;
9089 }
9090 total =
9091 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9092 cur.skip(12 + width * 12)?;
9093 for (field, held) in fields.iter().zip(&dictionaries) {
9094 if coded_type(&field.ty) && *held {
9095 cur.skip(20)?;
9096 }
9097 }
9098 for _ in 0..width * 2 {
9099 match cur.u8()? {
9100 0 => {}
9101 1 => cur.skip(20)?,
9102 _ => return Err(invalid("stripe page tag differs")),
9103 }
9104 }
9105 for column in 0..width {
9106 cur.skip_bound()?;
9107 cur.skip_bound()?;
9108 let count = cur.u32()? as u64;
9109 if count > stripe_rows as u64 {
9110 return Err(invalid("null count exceeds stripe rows"));
9111 }
9112 if column == wanted {
9113 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9114 }
9115 cur.skip(1)?;
9116 match cur.u8()? {
9117 0 => {}
9118 1 => cur.skip(16)?,
9119 _ => return Err(invalid("a stripe sum has an unknown tag")),
9120 }
9121 }
9122 }
9123 if total != rows {
9124 return Err(invalid("table row count differs from stripes"));
9125 }
9126 if cur.done() {
9127 return Ok(None);
9128 }
9129 let magic = cur.take(8)?;
9130 let values = magic == FREQUENCIES;
9131 if !values && magic != FREQUENCIES_V2 {
9132 return Err(invalid("directory extension magic differs"));
9133 }
9134 if cur.u16()? as usize != width {
9135 return Err(invalid("frequency column count differs"));
9136 }
9137 for _ in 0..wanted {
9138 skip_summary(&mut cur, values, rows)?;
9139 }
9140 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9141 return Ok(None);
9142 };
9143 let zero = summary
9144 .entries
9145 .iter()
9146 .find(|entry| entry.value == FrequencyValue::Integer(0))
9147 .map(|entry| entry.count)
9148 .or_else(|| (summary.omitted_max == 0).then_some(0));
9149 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9150}
9151
9152fn quick_integer_fold(
9155 file: &File,
9156 mut cur: Cursor<'_>,
9157 entry: &Entry,
9158 size: u64,
9159 wanted: usize,
9160 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9161) -> Result<()> {
9162 let name = &entry.name;
9163 let fields = &entry.fields;
9164 let rows = entry.rows;
9165 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9166 return Err(invalid("table directory differs from the catalog"));
9167 }
9168 let width = cur.u16()? as usize;
9169 if width != fields.len() {
9170 return Err(invalid("table directory width differs from the catalog"));
9171 }
9172 for field in fields {
9173 let stored =
9174 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9175 if &stored != field {
9176 return Err(invalid("table directory schema differs from the catalog"));
9177 }
9178 }
9179 let mut dictionaries = Vec::with_capacity(width);
9180 for field in fields {
9181 dictionaries.push(match cur.u8()? {
9182 0 => false,
9183 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9184 cur.skip(20)?;
9185 true
9186 }
9187 _ => return Err(invalid("dictionary page tag differs")),
9188 });
9189 }
9190 for _ in 0..width {
9191 match cur.u8()? {
9192 0 => {}
9193 1 => cur.skip(8)?,
9194 _ => return Err(invalid("distinct count tag differs")),
9195 }
9196 }
9197 if cur.u64()? != rows as u64 {
9198 return Err(invalid("table row count differs from the catalog"));
9199 }
9200 let stripes = cur.u32()? as usize;
9201 let mut total = 0_usize;
9202 let mut bytes = Vec::new();
9203 for _ in 0..stripes {
9204 let parts = cur.u32()? as usize;
9205 if parts == 0 || parts > STRIPE_PARTS {
9206 return Err(invalid("stripe part count is outside its bound"));
9207 }
9208 let mut part_rows = Vec::with_capacity(parts);
9209 for _ in 0..parts {
9210 let count = cur.u32()? as usize;
9211 if count == 0 {
9212 return Err(invalid("empty part"));
9213 }
9214 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9215 part_rows.push(count);
9216 }
9217 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9218 let section = index_section(parts)?;
9219 let index_length =
9220 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9221 if index.offset < HEADER
9222 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9223 || index.length as usize != index_length
9224 {
9225 return Err(invalid("index page range is outside the file"));
9226 }
9227 cur.skip(wanted * 12)?;
9228 let page = Span { offset: cur.u64()?, length: cur.u32()? };
9229 if page.offset < HEADER
9230 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9231 || page.length as usize > MAX_PAGE
9232 {
9233 return Err(invalid("column page range is outside the file"));
9234 }
9235 cur.skip((width - wanted - 1) * 12)?;
9236 for (field, held) in fields.iter().zip(&dictionaries) {
9237 if coded_type(&field.ty) && *held {
9238 cur.skip(20)?;
9239 }
9240 }
9241 for _ in 0..width * 2 {
9242 match cur.u8()? {
9243 0 => {}
9244 1 => cur.skip(20)?,
9245 _ => return Err(invalid("stripe page tag differs")),
9246 }
9247 }
9248 for _ in 0..width {
9249 cur.skip_bound()?;
9250 cur.skip_bound()?;
9251 cur.skip(5)?;
9252 match cur.u8()? {
9253 0 => {}
9254 1 => cur.skip(16)?,
9255 _ => return Err(invalid("a stripe sum has an unknown tag")),
9256 }
9257 }
9258 let spans = read_index_span(file, index, page, parts, wanted)?;
9259 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9260 bytes.resize(span.length, 0);
9261 let at = page
9262 .offset
9263 .checked_add(span.start as u64)
9264 .ok_or_else(|| invalid("part range overflow"))?;
9265 read_at(file, at, &mut bytes)?;
9266 if checksum(&bytes) != span.hash {
9267 return Err(invalid("integer part checksum differs"));
9268 }
9269 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9270 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9271 check_integer_tally_value(value, &fields[wanted].ty)?;
9272 emit(value, count)
9273 })?;
9274 if decoded_rows != expected_rows {
9275 return Err(invalid("encoded integer part holds the wrong number of rows"));
9276 }
9277 } else {
9278 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9279 if let Some(packed) = column.packed_parts() {
9280 let validity = column.validity();
9281 let all_valid = column.none_null();
9282 let base = packed.base();
9283 let mut codes = [0_u64; 64];
9284 for from in (0..expected_rows).step_by(codes.len()) {
9285 let count = (expected_rows - from).min(codes.len());
9286 packed.unpack(from, &mut codes[..count]);
9287 for (offset, &code) in codes[..count].iter().enumerate() {
9288 if all_valid || validity.is_valid(from + offset) {
9289 emit((base + i128::from(code)) as i64, 1)?;
9291 }
9292 }
9293 }
9294 continue;
9295 }
9296 let column = column.into_flat()?;
9297 let validity = column.validity();
9298 macro_rules! count_decoded {
9299 ($values:expr) => {
9300 for (row, &value) in $values.as_slice().iter().enumerate() {
9301 if validity.is_valid(row) {
9302 emit(i64::from(value), 1)?;
9303 }
9304 }
9305 };
9306 }
9307 match column.data() {
9308 Some(Data::Int8(values)) => count_decoded!(values),
9309 Some(Data::Int16(values)) => count_decoded!(values),
9310 Some(Data::Int32(values)) => count_decoded!(values),
9311 Some(Data::Int64(values)) => count_decoded!(values),
9312 _ => return Err(invalid("decoded integer part has the wrong type")),
9313 }
9314 }
9315 }
9316 }
9317 if total != rows {
9318 return Err(invalid("table row count differs from stripes"));
9319 }
9320 Ok(())
9321}
9322
9323fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9324 let fits = match ty {
9325 LogicalType::TinyInt => i8::try_from(value).is_ok(),
9326 LogicalType::SmallInt => i16::try_from(value).is_ok(),
9327 LogicalType::Integer => i32::try_from(value).is_ok(),
9328 LogicalType::BigInt => true,
9329 _ => false,
9330 };
9331 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9332}
9333
9334fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9335 read_directory(Cursor::new(bytes), size, None)
9336}
9337
9338fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9343 if cur.take(8)? != DIRECTORY {
9344 return Err(invalid("directory magic differs"));
9345 }
9346 let name = cur.text()?;
9347 let width = cur.u16()? as usize;
9348 let mut fields = Vec::with_capacity(width);
9349 for _ in 0..width {
9350 let name = cur.text()?;
9351 let ty = read_type(&mut cur)?;
9352 let not_null = match cur.u8()? {
9353 0 => false,
9354 1 => true,
9355 _ => return Err(invalid("nullability flag differs")),
9356 };
9357 fields.push(Field { name, ty, not_null });
9358 }
9359 let mut dictionaries = Vec::with_capacity(width);
9360 for field in &fields {
9361 dictionaries.push(match cur.u8()? {
9362 0 => None,
9363 tag if tag == dictionary_tag(&field.ty) => {
9364 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9365 let end = page
9366 .offset
9367 .checked_add(u64::from(page.length))
9368 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9369 if page.offset < HEADER || end > size {
9374 return Err(invalid("dictionary page range is outside the file"));
9375 }
9376 Some(page)
9377 }
9378 _ => return Err(invalid("dictionary page tag differs")),
9379 });
9380 }
9381 let mut distincts = Vec::with_capacity(width);
9382 for _ in 0..width {
9383 distincts.push(match cur.u8()? {
9384 0 => None,
9385 1 => Some(cur.u64()?),
9386 _ => return Err(invalid("distinct count tag differs")),
9387 });
9388 }
9389 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9390 let count = cur.u32()? as usize;
9391 let mut stripes = Vec::with_capacity(count);
9392 let mut total = 0_usize;
9393 for _ in 0..count {
9394 let count = cur.u32()? as usize;
9395 if count == 0 || count > STRIPE_PARTS {
9396 return Err(invalid("stripe part count is outside its bound"));
9397 }
9398 let mut parts = Vec::with_capacity(count);
9399 let mut stripe_rows = 0_usize;
9400 for _ in 0..count {
9401 let rows = cur.u32()?;
9402 if rows == 0 {
9403 return Err(invalid("empty part"));
9404 }
9405 parts.push(rows);
9406 stripe_rows = stripe_rows
9407 .checked_add(rows as usize)
9408 .ok_or_else(|| invalid("stripe row count overflow"))?;
9409 }
9410 total =
9411 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9412 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9413 let section = index_section(count)?;
9414 let wanted = section
9415 .checked_mul(width)
9416 .and_then(|bytes| u32::try_from(bytes).ok())
9417 .ok_or_else(|| invalid("index page length overflow"))?;
9418 let end = index
9419 .offset
9420 .checked_add(u64::from(index.length))
9421 .ok_or_else(|| invalid("index page offset overflow"))?;
9422 if index.offset < HEADER || end > size || index.length != wanted {
9423 return Err(invalid("index page range is outside the file"));
9424 }
9425 let mut pages = Vec::with_capacity(width);
9426 for _ in 0..width {
9427 let offset = cur.u64()?;
9428 let length = cur.u32()?;
9429 let end = offset
9430 .checked_add(u64::from(length))
9431 .ok_or_else(|| invalid("page offset overflow"))?;
9432 if offset < HEADER || end > size || length as usize > MAX_PAGE {
9433 return Err(invalid("page range is outside the file"));
9434 }
9435 pages.push(Span { offset, length });
9436 }
9437 let mut memberships = vec![None; width];
9438 for (column, field) in fields.iter().enumerate() {
9439 if !coded_type(&field.ty) || dictionaries[column].is_none() {
9440 continue;
9441 }
9442 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9443 let end = page
9444 .offset
9445 .checked_add(u64::from(page.length))
9446 .ok_or_else(|| invalid("membership page offset overflow"))?;
9447 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9448 return Err(invalid("membership page range is outside the file"));
9449 }
9450 if page.length != 0 {
9453 memberships[column] = Some(page);
9454 }
9455 }
9456 let mut sieves = vec![None; width];
9457 for sieve in sieves.iter_mut().take(width) {
9458 match cur.u8()? {
9459 0 => continue,
9460 1 => {}
9461 _ => return Err(invalid("a sieve page has an unknown tag")),
9462 }
9463 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9464 let end = page
9465 .offset
9466 .checked_add(u64::from(page.length))
9467 .ok_or_else(|| invalid("sieve page offset overflow"))?;
9468 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9469 return Err(invalid("sieve page range is outside the file"));
9470 }
9471 *sieve = Some(page);
9472 }
9473 let mut part_ranges = vec![None; width];
9474 for held in part_ranges.iter_mut().take(width) {
9475 match cur.u8()? {
9476 0 => continue,
9477 1 => {}
9478 _ => return Err(invalid("a part range page has an unknown tag")),
9479 }
9480 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9481 let end = page
9482 .offset
9483 .checked_add(u64::from(page.length))
9484 .ok_or_else(|| invalid("part range page offset overflow"))?;
9485 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9486 return Err(invalid("part range page range is outside the file"));
9487 }
9488 *held = Some(page);
9489 }
9490 let mut ranges = Vec::with_capacity(width);
9491 for column in 0..width {
9492 let low = cur.bound()?;
9493 let high = cur.bound()?;
9494 let nulls = cur.u32()? as usize;
9495 if nulls > stripe_rows {
9496 return Err(invalid("null count exceeds stripe rows"));
9497 }
9498 let exact = cur.u8()? != 0;
9499 let sum = match cur.u8()? {
9500 0 => None,
9501 1 => Some(i128::from_le_bytes(
9502 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9503 )),
9504 _ => return Err(invalid("a stripe sum has an unknown tag")),
9505 };
9506 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9512 let low = low.map(|bound| scaled_as(bound, ty));
9513 let high = high.map(|bound| scaled_as(bound, ty));
9514 ranges.push(Range { low, high, nulls, exact, sum });
9515 }
9516 stripes.push(Stripe {
9517 rows: stripe_rows,
9518 parts,
9519 index,
9520 pages,
9521 memberships: Pages::from_slots(memberships)?,
9522 sieves: Pages::from_slots(sieves)?,
9523 part_ranges: Pages::from_slots(part_ranges)?,
9524 zone: Zone::from_ranges(ranges),
9525 });
9526 }
9527 if total != rows {
9528 return Err(invalid("table row count differs from stripes"));
9529 }
9530 let mut entry_counts = vec![0; width];
9533 let frequencies = if cur.done() {
9534 vec![None; width]
9535 } else {
9536 let frequency_magic = cur.take(8)?;
9537 let frequency_values = frequency_magic == FREQUENCIES;
9538 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9539 return Err(invalid("directory extension magic differs"));
9540 }
9541 if cur.u16()? as usize != width {
9542 return Err(invalid("frequency column count differs"));
9543 }
9544 let mut frequencies = Vec::with_capacity(width);
9545 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9546 let start = cur.at;
9547 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9548 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9549 frequencies.push(match (summary, stored_at) {
9550 (None, _) => None,
9551 (Some(summary), None) => Some(Frequencies::Held(summary)),
9552 (Some(_), Some(offset)) => Some(Frequencies::Stored {
9553 span: Span {
9554 offset: offset + start as u64,
9555 length: u32::try_from(cur.at - start)
9556 .map_err(|_| invalid("a frequency synopsis is too long"))?,
9557 },
9558 values: frequency_values,
9559 }),
9560 });
9561 }
9562 frequencies
9563 };
9564 let mut clustering = None;
9574 let mut sections = Vec::new();
9575 let mut pair_frequencies = Vec::new();
9576 let mut seen_pair_frequencies = false;
9577 let mut frequency_texts = vec![Vec::new(); width];
9578 let mut seen_frequency_texts = false;
9579 let mut host_groups = None;
9580 let mut demoted = Vec::new();
9581 let mut seen_sections = false;
9582 let mut dictionary_payloads = Vec::new();
9583 let mut seen_payloads = false;
9584 let mut generation = 0;
9587 while !cur.done() {
9588 let mut tag = [0u8; 8];
9589 tag.copy_from_slice(cur.take(8)?);
9590 if &tag == PAIR_FREQUENCIES {
9591 if seen_pair_frequencies {
9592 return Err(invalid("directory names two pair frequency blocks"));
9593 }
9594 seen_pair_frequencies = true;
9595 let count = cur.u16()? as usize;
9596 if count > MAX_PAIR_FREQUENCIES {
9597 return Err(invalid("pair frequency count exceeds its bound"));
9598 }
9599 pair_frequencies = Vec::with_capacity(count);
9600 for _ in 0..count {
9601 let first = cur.u16()?;
9602 let second = cur.u16()?;
9603 let first_at = first as usize;
9604 let second_at = second as usize;
9605 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9606 return Err(invalid("pair frequency first column has no synopsis"));
9607 }
9608 let first_entries = entry_counts[first_at];
9609 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9610 || dictionaries.get(second_at).copied().flatten().is_none()
9611 {
9612 return Err(invalid("pair frequency second column has no stable dictionary"));
9613 }
9614 if pair_frequencies
9615 .iter()
9616 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9617 {
9618 return Err(invalid("directory repeats a pair frequency summary"));
9619 }
9620 let omitted_max = cur.u64()?;
9621 if omitted_max > rows as u64 {
9622 return Err(invalid("pair frequency omitted count exceeds the table"));
9623 }
9624 let entries_count = cur.u16()? as usize;
9625 if entries_count > FREQUENCY_ENTRIES {
9626 return Err(invalid("pair frequency entry count exceeds its bound"));
9627 }
9628 let mut entries = Vec::with_capacity(entries_count);
9629 for _ in 0..entries_count {
9630 let first_entry = cur.u16()?;
9631 if first_entry as usize >= first_entries {
9632 return Err(invalid("pair frequency anchor is outside its synopsis"));
9633 }
9634 let second = match cur.u8()? {
9635 0 => None,
9636 1 => Some(cur.u32()?),
9637 _ => return Err(invalid("pair frequency string tag differs")),
9638 };
9639 let count = cur.u64()?;
9640 if count == 0 || count > rows as u64 {
9641 return Err(invalid("pair frequency count is outside the table"));
9642 }
9643 entries.push(PairFrequencyEntry { first_entry, second, count });
9644 }
9645 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9646 return Err(invalid("pair frequency entries are not descending"));
9647 }
9648 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9649 }
9650 } else if &tag == FREQUENCY_TEXTS {
9651 if seen_frequency_texts {
9652 return Err(invalid("directory names two frequency text blocks"));
9653 }
9654 seen_frequency_texts = true;
9655 let columns = cur.u16()? as usize;
9656 if columns > width {
9657 return Err(invalid("frequency text column count exceeds the schema"));
9658 }
9659 for _ in 0..columns {
9660 let column = cur.u16()? as usize;
9661 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9662 return Err(invalid("frequency text column is repeated or out of range"));
9663 }
9664 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9665 || dictionaries.get(column).copied().flatten().is_none()
9666 || frequencies.get(column).and_then(Option::as_ref).is_none()
9667 {
9668 return Err(invalid("frequency texts belong to a non-string synopsis"));
9669 }
9670 let count = cur.u16()? as usize;
9671 if count == 0 || count != entry_counts[column] {
9672 return Err(invalid("frequency text count differs from its synopsis"));
9673 }
9674 let mut texts = Vec::with_capacity(count);
9675 for _ in 0..count {
9676 texts.push(match cur.u8()? {
9677 0 => None,
9678 1 => {
9679 let length = cur.u32()? as usize;
9680 let bytes = cur.take(length)?.to_vec();
9681 if fields[column].ty == LogicalType::Varchar {
9682 std::str::from_utf8(&bytes)
9683 .map_err(|_| invalid("frequency text is not UTF-8"))?;
9684 }
9685 Some(bytes)
9686 }
9687 _ => return Err(invalid("frequency text tag differs")),
9688 });
9689 }
9690 frequency_texts[column] = texts;
9691 }
9692 } else if &tag == HOST_GROUPS {
9693 if host_groups.is_some() {
9694 return Err(invalid("directory names two host group blocks"));
9695 }
9696 let column = cur.u16()? as usize;
9697 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9698 || dictionaries.get(column).copied().flatten().is_none()
9699 {
9700 return Err(invalid("host groups belong to a non-string dictionary"));
9701 }
9702 let omitted_max = cur.u64()?;
9703 if omitted_max > rows as u64 {
9704 return Err(invalid("host group bound exceeds the table"));
9705 }
9706 let count = cur.u16()? as usize;
9707 if count > host::CAPACITY {
9708 return Err(invalid("host group count exceeds its bound"));
9709 }
9710 let mut entries = Vec::with_capacity(count);
9711 let mut bytes = 0_usize;
9712 for _ in 0..count {
9713 let host_len = cur.u32()? as usize;
9714 bytes =
9715 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9716 if bytes > host::BYTE_BUDGET {
9717 return Err(invalid("host groups exceed their byte budget"));
9718 }
9719 let host = std::str::from_utf8(cur.take(host_len)?)
9720 .map_err(|_| invalid("host is not UTF-8"))?
9721 .to_owned();
9722 let count = cur.u64()?;
9723 if count == 0 || count > rows as u64 {
9724 return Err(invalid("host group count exceeds the table"));
9725 }
9726 let bytes_sum = i128::from_le_bytes(
9727 cur.take(16)?
9728 .try_into()
9729 .map_err(|_| invalid("host length sum is truncated"))?,
9730 );
9731 if bytes_sum < 0 {
9732 return Err(invalid("host length sum is negative"));
9733 }
9734 let minimum_len = cur.u32()? as usize;
9735 bytes =
9736 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9737 if bytes > host::BYTE_BUDGET {
9738 return Err(invalid("host groups exceed their byte budget"));
9739 }
9740 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9741 .map_err(|_| invalid("host minimum is not UTF-8"))?
9742 .to_owned();
9743 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9744 }
9745 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9746 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9747 {
9748 return Err(invalid("host groups are not in certified order"));
9749 }
9750 host_groups = Some(host::HostSummary { column, omitted_max, entries });
9751 } else if &tag == CLUSTERING {
9752 if clustering.is_some() {
9753 return Err(invalid("directory names two clustering declarations"));
9754 }
9755 let bucket = Width::from_tag(cur.u8()?)
9756 .ok_or_else(|| invalid("clustering width tag differs"))?;
9757 let count = cur.u16()? as usize;
9758 let mut columns = Vec::with_capacity(count.min(fields.len()));
9759 for _ in 0..count {
9760 columns.push(u32::from(cur.u16()?));
9761 }
9762 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9765 invalid("stored clustering declaration does not match the table it is on")
9766 })?);
9767 } else if &tag == DEMOTED {
9768 if !demoted.is_empty() {
9769 return Err(invalid("directory names two demoted column blocks"));
9770 }
9771 let count = cur.u16()? as usize;
9772 if count == 0 || count > width {
9773 return Err(invalid("demoted column count is outside the schema"));
9774 }
9775 demoted = vec![false; width];
9776 for _ in 0..count {
9777 let column = cur.u16()? as usize;
9778 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9779 return Err(invalid("a demoted column is repeated or has no dictionary"));
9780 }
9781 demoted[column] = true;
9782 }
9783 } else if &tag == SECTIONS {
9784 if seen_sections {
9785 return Err(invalid("directory names two section tables"));
9786 }
9787 seen_sections = true;
9788 generation = cur.u64()?;
9789 let count = cur.u16()? as usize;
9790 if count > MAX_SECTIONS {
9791 return Err(invalid("section count exceeds its bound"));
9792 }
9793 sections = Vec::with_capacity(count);
9794 for _ in 0..count {
9797 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9798 }
9799 for held in §ions {
9800 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9801 return Err(invalid("a section's extent table overflows the file"));
9802 };
9803 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9807 return Err(invalid("a section's extent table is outside the file"));
9808 }
9809 if held.extents == 0 && held.extent_bytes != 0 {
9810 return Err(invalid("a section with no extents names an extent table"));
9811 }
9812 }
9813 } else if &tag == DICTIONARY_PAYLOADS {
9814 if seen_payloads {
9815 return Err(invalid("directory names two dictionary payload blocks"));
9816 }
9817 seen_payloads = true;
9818 let count = cur.u16()? as usize;
9819 if count != fields.len() {
9820 return Err(invalid("dictionary payload block does not match the table's columns"));
9821 }
9822 dictionary_payloads = Vec::with_capacity(count);
9823 for _ in 0..count {
9824 let bytes = cur.u64()?;
9825 if bytes > size {
9826 return Err(invalid("a dictionary payload is larger than the file"));
9827 }
9828 dictionary_payloads.push(bytes);
9829 }
9830 } else {
9831 return Err(invalid("directory extension magic differs"));
9832 }
9833 }
9834 if !cur.done() {
9835 return Err(invalid("directory has trailing bytes"));
9836 }
9837 for stripe in &stripes {
9838 for (column, field) in fields.iter().enumerate() {
9839 if coded_type(&field.ty)
9840 && dictionaries[column].is_some()
9841 && stripe.memberships.get(column).is_none()
9842 && !demoted.get(column).copied().unwrap_or(false)
9843 {
9844 return Err(invalid("string page has no code membership index"));
9845 }
9846 }
9847 }
9848 Ok(Table {
9849 name,
9850 fields,
9851 stripes,
9852 rows,
9853 dictionaries,
9854 dictionary_payloads,
9855 demoted,
9856 distincts,
9857 frequencies,
9858 pair_frequencies,
9859 frequency_texts,
9860 host_groups,
9861 clustering,
9862 generation,
9863 sections,
9864 })
9865}
9866
9867fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
9869 bounds::put(out, bound)
9870}
9871
9872#[derive(Debug)]
9889struct Codes;
9890
9891impl chooser::Chooser for Codes {
9892 fn name(&self) -> &'static str {
9893 "codes"
9894 }
9895
9896 fn narrow_strings(
9897 &self,
9898 _values: &[&[u8]],
9899 offered: &[string::Kind],
9900 _depth: u8,
9901 ) -> Vec<string::Kind> {
9902 offered.to_vec()
9905 }
9906
9907 fn narrow_integers(
9908 &self,
9909 _values: &[i64],
9910 offered: &[integer::Kind],
9911 depth: u8,
9912 ) -> Vec<integer::Kind> {
9913 narrowed_to(Codes::keep(depth), offered)
9916 }
9917
9918 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9919 Codes::keep(depth).contains(&kind)
9920 }
9921}
9922
9923impl Codes {
9924 fn keep(depth: u8) -> &'static [integer::Kind] {
9925 if depth == 0 {
9926 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
9927 } else {
9928 &[integer::Kind::Constant, integer::Kind::Packed]
9929 }
9930 }
9931}
9932
9933fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
9941 let narrowed: Vec<integer::Kind> =
9942 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
9943 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
9944}
9945
9946#[derive(Debug)]
9958struct Fixed;
9959
9960impl chooser::Chooser for Fixed {
9961 fn name(&self) -> &'static str {
9962 "fixed"
9963 }
9964
9965 fn narrow_strings(
9966 &self,
9967 _values: &[&[u8]],
9968 offered: &[string::Kind],
9969 _depth: u8,
9970 ) -> Vec<string::Kind> {
9971 offered.to_vec()
9972 }
9973
9974 fn narrow_integers(
9975 &self,
9976 _values: &[i64],
9977 offered: &[integer::Kind],
9978 depth: u8,
9979 ) -> Vec<integer::Kind> {
9980 narrowed_to(Fixed::keep(depth), offered)
9981 }
9982
9983 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9984 Fixed::keep(depth).contains(&kind)
9985 }
9986}
9987
9988impl Fixed {
9989 fn keep(depth: u8) -> &'static [integer::Kind] {
9990 if depth == 0 {
9991 &[
9992 integer::Kind::Constant,
9993 integer::Kind::Packed,
9994 integer::Kind::Delta,
9995 integer::Kind::Rle,
9996 integer::Kind::Sparse,
9997 integer::Kind::Strided,
9998 ]
9999 } else {
10000 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10001 }
10002 }
10003}
10004
10005fn widened(data: &Data) -> Option<Vec<i64>> {
10012 match data {
10013 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10014 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10015 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10016 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10017 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10018 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10019 Data::Int64(values) => Some(values.to_vec()),
10020 _ => None,
10021 }
10022}
10023
10024fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10030 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10031 let values = integer::decode_as::<T>(bytes)
10032 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10033 if values.len() != rows {
10034 return Err(invalid("cascade page holds the wrong number of rows"));
10035 }
10036 Ok(values)
10037 }
10038 Ok(match ty {
10039 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10040 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10041 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10042 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10043 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10044 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10045 LogicalType::BigInt
10046 | LogicalType::Timestamp
10047 | LogicalType::Time
10048 | LogicalType::TimeTz
10049 | LogicalType::TimestampTz
10050 | LogicalType::TimestampS
10051 | LogicalType::TimestampMs
10052 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10053 LogicalType::Decimal { .. } => match ty.physical() {
10056 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10057 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10058 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10059 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10060 },
10061 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10062 })
10063}
10064
10065fn plain_width(ty: &LogicalType) -> Option<usize> {
10068 Some(match ty {
10069 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10070 LogicalType::SmallInt | LogicalType::USmallInt => 2,
10071 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10072 LogicalType::BigInt
10073 | LogicalType::Timestamp
10074 | LogicalType::Time
10075 | LogicalType::TimeTz
10076 | LogicalType::TimestampTz
10077 | LogicalType::TimestampS
10078 | LogicalType::TimestampMs
10079 | LogicalType::TimestampNs => 8,
10080 LogicalType::Decimal { .. } => match ty.physical() {
10081 PhysicalType::Int16 => 2,
10082 PhysicalType::Int32 => 4,
10083 PhysicalType::Int64 => 8,
10084 _ => return None,
10087 },
10088 _ => return None,
10089 })
10090}
10091
10092fn cascaded(
10098 flat: &Vector,
10099 ty: &LogicalType,
10100 packed: Option<&Packed<'_>>,
10101 settling: &mut Settling,
10102) -> Result<Option<Vec<u8>>> {
10103 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10104 let Some(values) = widened(data) else { return Ok(None) };
10105 let plain = values.len().saturating_mul(width);
10106 let best = match packed {
10107 Some(packed) => plain.min(21 + size_of_val(packed.words())),
10109 None => plain,
10110 };
10111 let out = settling.encode(&values)?;
10112 Ok((out.len() < best).then_some(out))
10113}
10114
10115const SEARCH_EVERY: usize = 16;
10122
10123#[derive(Debug, Default)]
10129struct Settling {
10130 shape: Option<Shape>,
10133 since: usize,
10135 symbols: Option<Symbols>,
10137}
10138
10139#[derive(Debug)]
10142struct Symbols {
10143 shape: chooser::Settled,
10144 len: usize,
10147 payload: usize,
10148 since: usize,
10149}
10150
10151impl Settling {
10152 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10160 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10161 {
10162 let out = string::encode_fsst(values, &symbols.shape)?;
10163 let held = match &out {
10165 None => symbols.len == 0,
10166 Some(out) => {
10167 (out.len() as u128) * (symbols.payload as u128) * 4
10168 <= (symbols.len as u128) * (payload as u128) * 5
10169 }
10170 };
10171 if held {
10172 symbols.since += 1;
10173 return Ok(out);
10174 }
10175 }
10176 let shape = string::fsst_shape(values);
10177 let out = string::encode_fsst(values, &shape)?;
10178 let len = out.as_ref().map_or(0, Vec::len);
10179 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10180 Ok(out)
10181 }
10182
10183 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10190 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10191 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10192 let out = integer::encode_with(values, &replay)?;
10193 if !replay.held() {
10194 self.settle(&out, values.len(), replay.first_offered())?;
10195 return Ok(out);
10196 }
10197 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10198 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10199 self.since += 1;
10200 return Ok(out);
10201 }
10202 }
10203 let search = chooser::Replay::new(&[], &Fixed);
10205 let out = integer::encode_with(values, &search)?;
10206 self.settle(&out, values.len(), search.first_offered())?;
10207 Ok(out)
10208 }
10209
10210 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10211 let kinds = integer::shape(out)?;
10212 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10213 self.since = 0;
10214 Ok(())
10215 }
10216}
10217
10218#[derive(Debug)]
10220struct Shape {
10221 kinds: Vec<integer::Kind>,
10222 offered: Vec<integer::Kind>,
10223 len: usize,
10224 rows: usize,
10225}
10226
10227fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10268 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10269 let mut payload = 0_usize;
10270 for row in 0..flat.len() {
10271 let text = flat.bytes_at(row).unwrap_or(b"");
10274 payload = payload.saturating_add(text.len());
10275 values.push(text);
10276 }
10277 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10279 let Some(out) = settling.text(&values, payload)? else {
10280 return Ok(None);
10281 };
10282 Ok((out.len() < plain).then_some(out))
10283}
10284
10285fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10286 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10287 let coded = integer::encode_with(&wide, &Codes)?;
10288 let plain = codes.len().saturating_mul(size_of::<u32>());
10289 Ok((coded.len() < plain).then_some(coded))
10290}
10291
10292fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10295 let flag = match flat.validity() {
10296 Validity::AllValid => 0,
10297 Validity::AllInvalid => 1,
10298 Validity::Mask(_) => 2,
10299 };
10300 out.push(flag);
10301 if flag == 2 {
10302 for group in (0..flat.len()).step_by(8) {
10303 let mut bits = 0_u8;
10304 for bit in 0..8 {
10305 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10306 bits |= 1 << bit;
10307 }
10308 }
10309 out.push(bits);
10310 }
10311 }
10312}
10313
10314fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10321 let coded = encoded_codes(codes)?;
10322 let mut out = Vec::with_capacity(
10323 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10324 );
10325 out.push(if coded.is_some() { 4 } else { 3 });
10326 out.extend_from_slice(validity);
10327 match coded {
10328 Some(coded) => out.extend_from_slice(&coded),
10329 None => {
10330 for &code in codes {
10331 put_u32(&mut out, code);
10332 }
10333 }
10334 }
10335 Ok(out)
10336}
10337
10338fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10341 let ty = vector.logical_type();
10342 let flat = vector.flatten()?;
10344 let mut out = Vec::new();
10345 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10346 let compressed_text = if dictionary.is_none() && coded_type(ty) {
10347 text_compressed(&flat, settling)?
10348 } else {
10349 None
10350 };
10351 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10352 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10353 let cascade =
10357 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10358 out.push(if cascade.is_some() {
10359 5
10360 } else if dictionary.is_some() {
10361 1
10362 } else if compressed_text.is_some() {
10363 6
10364 } else if packed.is_some() {
10365 2
10366 } else {
10367 0
10368 });
10369 push_validity(&mut out, &flat);
10370 if let Some(cascade) = cascade {
10371 out.extend_from_slice(&cascade);
10372 return Ok(out);
10373 }
10374 if let Some(dictionary) = dictionary {
10375 out.extend_from_slice(&dictionary);
10376 return Ok(out);
10377 }
10378 if let Some(compressed_text) = compressed_text {
10379 out.extend_from_slice(&compressed_text);
10380 return Ok(out);
10381 }
10382 if let Some(packed) = packed {
10383 if packed.offset() != 0 {
10384 return Err(invalid("writer received a sliced packed vector"));
10385 }
10386 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10387 out.extend_from_slice(&packed.base().to_le_bytes());
10388 put_u32(
10389 &mut out,
10390 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10391 );
10392 for word in packed.words() {
10393 put_u64(&mut out, *word);
10394 }
10395 return Ok(out);
10396 }
10397 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10398 match (ty, data) {
10399 (LogicalType::TinyInt, Data::Int8(values)) => {
10400 for value in &**values {
10401 out.extend_from_slice(&value.to_le_bytes());
10402 }
10403 }
10404 (LogicalType::UTinyInt, Data::UInt8(values)) => {
10405 for value in &**values {
10406 out.extend_from_slice(&value.to_le_bytes());
10407 }
10408 }
10409 (LogicalType::SmallInt, Data::Int16(values)) => {
10410 for value in &**values {
10411 out.extend_from_slice(&value.to_le_bytes());
10412 }
10413 }
10414 (LogicalType::USmallInt, Data::UInt16(values)) => {
10415 for value in &**values {
10416 out.extend_from_slice(&value.to_le_bytes());
10417 }
10418 }
10419 (LogicalType::UInteger, Data::UInt32(values)) => {
10420 for value in &**values {
10421 out.extend_from_slice(&value.to_le_bytes());
10422 }
10423 }
10424 (LogicalType::UBigInt, Data::UInt64(values)) => {
10425 for value in &**values {
10426 out.extend_from_slice(&value.to_le_bytes());
10427 }
10428 }
10429 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10430 for value in &**values {
10431 out.extend_from_slice(&value.to_le_bytes());
10432 }
10433 }
10434 (
10435 LogicalType::BigInt
10436 | LogicalType::Timestamp
10437 | LogicalType::Time
10438 | LogicalType::TimeTz
10439 | LogicalType::TimestampTz
10440 | LogicalType::TimestampS
10441 | LogicalType::TimestampMs
10442 | LogicalType::TimestampNs,
10443 Data::Int64(values),
10444 ) => {
10445 for value in &**values {
10446 out.extend_from_slice(&value.to_le_bytes());
10447 }
10448 }
10449 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10452 for value in &**values {
10453 out.extend_from_slice(&value.to_le_bytes());
10454 }
10455 }
10456 (LogicalType::UHugeInt, Data::UInt128(values)) => {
10457 for value in &**values {
10458 out.extend_from_slice(&value.to_le_bytes());
10459 }
10460 }
10461 (LogicalType::Float, Data::Float32(values)) => {
10464 for value in &**values {
10465 out.extend_from_slice(&value.to_le_bytes());
10466 }
10467 }
10468 (LogicalType::Double, Data::Float64(values)) => {
10469 for value in &**values {
10470 out.extend_from_slice(&value.to_le_bytes());
10471 }
10472 }
10473 (LogicalType::Interval, Data::Interval(values)) => {
10477 for (months, days, micros) in &**values {
10478 out.extend_from_slice(&months.to_le_bytes());
10479 out.extend_from_slice(&days.to_le_bytes());
10480 out.extend_from_slice(µs.to_le_bytes());
10481 }
10482 }
10483 (LogicalType::Boolean, Data::Bool(values)) => {
10484 for value in &**values {
10485 out.push(u8::from(*value));
10486 }
10487 }
10488 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10491 for value in &**values {
10492 out.extend_from_slice(&value.to_le_bytes());
10493 }
10494 }
10495 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10496 for value in &**values {
10497 out.extend_from_slice(&value.to_le_bytes());
10498 }
10499 }
10500 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10501 for value in &**values {
10502 out.extend_from_slice(&value.to_le_bytes());
10503 }
10504 }
10505 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10506 for value in &**values {
10507 out.extend_from_slice(&value.to_le_bytes());
10508 }
10509 }
10510 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10515 let mut bytes = Vec::new();
10516 put_u32(&mut out, 0);
10517 for row in 0..vector.len() {
10518 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10519 bytes.extend_from_slice(value);
10520 put_u32(
10521 &mut out,
10522 u32::try_from(bytes.len())
10523 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10524 );
10525 }
10526 out.extend_from_slice(&bytes);
10527 }
10528 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10529 }
10530 Ok(out)
10531}
10532
10533fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10534 while value >= 0x80 {
10535 out.push((value as u8 & 0x7f) | 0x80);
10536 value >>= 7;
10537 }
10538 out.push(value as u8);
10539}
10540
10541fn unique_codes(codes: &[u32]) -> Vec<u32> {
10543 let mut unique = codes.to_vec();
10544 unique.sort_unstable();
10545 unique.dedup();
10546 unique
10547}
10548
10549fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10555 let mut lists = lists;
10556 while lists.len() > 1 {
10557 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10558 for pair in lists.chunks(2) {
10559 match pair {
10560 [left, right] => next.push(merged_pair(left, right)),
10561 [only] => next.push(only.clone()),
10562 _ => {}
10563 }
10564 }
10565 lists = next;
10566 }
10567 lists.pop().unwrap_or_default()
10568}
10569
10570fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10571 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10572 let mut at = 0;
10573 let mut to = 0;
10574 while at < left.len() && to < right.len() {
10575 match left[at].cmp(&right[to]) {
10576 Ordering::Less => {
10577 out.push(left[at]);
10578 at += 1;
10579 }
10580 Ordering::Greater => {
10581 out.push(right[to]);
10582 to += 1;
10583 }
10584 Ordering::Equal => {
10585 out.push(left[at]);
10586 at += 1;
10587 to += 1;
10588 }
10589 }
10590 }
10591 out.extend_from_slice(&left[at..]);
10592 out.extend_from_slice(&right[to..]);
10593 out
10594}
10595
10596fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10601 let mut merged = Range::default();
10602 let mut first = true;
10603 for range in ranges {
10604 merged.nulls = merged.nulls.saturating_add(range.nulls);
10605 merged.sum = match (merged.sum.take(), range.sum) {
10609 (Some(held), Some(next)) if !first => held.checked_add(next),
10610 (_, next) if first => next,
10611 _ => None,
10612 };
10613 merged.exact = if first { range.exact } else { merged.exact && range.exact };
10614 if first {
10615 merged.low = range.low;
10616 merged.high = range.high;
10617 first = false;
10618 continue;
10619 }
10620 merged.low = match (merged.low.take(), range.low) {
10621 (Some(held), Some(next)) => Some(held.smaller(next)),
10622 _ => None,
10623 };
10624 merged.high = match (merged.high.take(), range.high) {
10625 (Some(held), Some(next)) => Some(held.larger(next)),
10626 _ => None,
10627 };
10628 }
10629 merged
10630}
10631
10632fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10645 match bound {
10646 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10647 value.truncate(PART_BOUND_BYTES);
10648 if !high {
10649 return Some(Bound::Bytes(value));
10650 }
10651 while let Some(last) = value.pop() {
10652 if last < u8::MAX {
10653 value.push(last + 1);
10654 return Some(Bound::Bytes(value));
10655 }
10656 }
10657 None
10658 }
10659 other => other,
10660 }
10661}
10662
10663fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10671 let mut out = Vec::new();
10672 put_u32(
10673 &mut out,
10674 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10675 );
10676 for range in ranges {
10677 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10678 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10679 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10680 }
10681 Ok(out)
10682}
10683
10684fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10686 let mut cur = Cursor::new(bytes);
10687 let parts = cur.u32()? as usize;
10688 let mut out = Vec::new();
10689 for _ in 0..parts {
10690 let low = cur.bound()?;
10691 let high = cur.bound()?;
10692 let nulls = cur.u32()? as usize;
10693 out.push(Range { low, high, nulls, exact: false, sum: None });
10694 }
10695 Ok(out)
10696}
10697
10698fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10699 let held: Vec<&Option<Sieve>> = sieves.collect();
10700 let mut out = Vec::new();
10701 put_u32(
10702 &mut out,
10703 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10704 );
10705 for sieve in &held {
10706 let length = sieve.as_ref().map_or(0, Sieve::len);
10707 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10708 }
10709 for sieve in held.into_iter().flatten() {
10711 out.extend_from_slice(&sieve.to_bytes());
10712 }
10713 Ok(out)
10714}
10715
10716fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10722 let parts = u32::from_le_bytes(
10723 bytes
10724 .get(..4)
10725 .ok_or_else(|| invalid("sieve page is truncated"))?
10726 .try_into()
10727 .map_err(|_| invalid("sieve page is truncated"))?,
10728 ) as usize;
10729 let mut lengths = Vec::with_capacity(parts);
10730 for part in 0..parts {
10731 let at = 4 + part * 4;
10732 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10733 lengths.push(u32::from_le_bytes(
10734 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10735 ) as usize);
10736 }
10737 let mut at = 4 + parts * 4;
10738 let mut out = Vec::with_capacity(parts);
10739 for length in lengths {
10740 if length == 0 {
10741 out.push(None);
10742 continue;
10743 }
10744 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10745 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10746 out.push(Sieve::from_bytes(field));
10747 at = end;
10748 }
10749 if at != bytes.len() {
10750 return Err(invalid("sieve page has trailing bytes"));
10751 }
10752 Ok(out)
10753}
10754
10755fn encode_membership(unique: &[u32]) -> Vec<u8> {
10761 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10762 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10763 let mut previous = 0;
10764 for (at, &code) in unique.iter().enumerate() {
10765 put_varint(&mut out, if at == 0 { code } else { code - previous });
10766 previous = code;
10767 }
10768 out
10769}
10770
10771fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10772 let mut value = 0_u32;
10773 for shift in (0..35).step_by(7) {
10774 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10775 *at += 1;
10776 let part = u32::from(byte & 0x7f);
10777 if shift == 28 && part > 0x0f {
10778 return Err(invalid("membership varint overflow"));
10779 }
10780 value = value
10781 .checked_add(
10782 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10783 )
10784 .ok_or_else(|| invalid("membership varint overflow"))?;
10785 if byte & 0x80 == 0 {
10786 return Ok(value);
10787 }
10788 }
10789 Err(invalid("membership varint is too long"))
10790}
10791
10792fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10793 let mut at = 0;
10794 let count = take_varint(bytes, &mut at)? as usize;
10795 let mut codes = Vec::with_capacity(count);
10796 let mut previous = 0_u32;
10797 for index in 0..count {
10798 let delta = take_varint(bytes, &mut at)?;
10799 let code = if index == 0 {
10800 delta
10801 } else {
10802 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10803 };
10804 if index > 0 && code <= previous {
10805 return Err(invalid("membership codes are not increasing"));
10806 }
10807 codes.push(code);
10808 previous = code;
10809 }
10810 if at != bytes.len() {
10811 return Err(invalid("membership page has trailing bytes"));
10812 }
10813 Ok(codes)
10814}
10815
10816fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10824 let mut by_text: HashMap<&[u8], u32, Spread> =
10825 HashMap::with_capacity_and_hasher(vector.len(), Spread);
10826 let mut values = Vec::new();
10827 let mut codes = Vec::with_capacity(vector.len());
10828 let mut plain_bytes = 0_usize;
10829 for row in 0..vector.len() {
10830 let text = vector.bytes_at(row).unwrap_or(b"");
10831 plain_bytes = plain_bytes.saturating_add(text.len());
10832 let code = match by_text.get(text) {
10833 Some(&code) => code,
10834 None => {
10835 let code = u32::try_from(values.len())
10836 .map_err(|_| invalid("too many dictionary values"))?;
10837 by_text.insert(text, code);
10838 values.push(text);
10839 code
10840 }
10841 };
10842 codes.push(code);
10843 }
10844 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
10845 let encoded = 8_usize
10846 .saturating_add((values.len() + 1).saturating_mul(4))
10847 .saturating_add(dictionary_bytes)
10848 .saturating_add(codes.len().saturating_mul(4));
10849 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
10850 if encoded >= plain {
10851 return Ok(None);
10852 }
10853 let mut out = Vec::with_capacity(encoded);
10854 put_u32(
10855 &mut out,
10856 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
10857 );
10858 put_u32(
10859 &mut out,
10860 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
10861 );
10862 let mut offset = 0_u32;
10863 put_u32(&mut out, offset);
10864 for value in &values {
10865 offset = offset
10866 .checked_add(
10867 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
10868 )
10869 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
10870 put_u32(&mut out, offset);
10871 }
10872 for value in values {
10873 out.extend_from_slice(value);
10874 }
10875 for code in codes {
10876 put_u32(&mut out, code);
10877 }
10878 Ok(Some(out))
10879}
10880
10881struct Room<'a, T> {
10883 state: &'a Mutex<(T, usize)>,
10884 finished: &'a Condvar,
10885 bytes: usize,
10886}
10887
10888impl<T> Drop for Room<'_, T> {
10889 fn drop(&mut self) {
10890 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
10891 held.1 -= self.bytes;
10892 drop(held);
10893 self.finished.notify_all();
10894 }
10895}
10896
10897enum Closing<'a> {
10899 Numeric {
10901 column: usize,
10902 counted: bool,
10903 },
10904 Dictionary {
10905 index: usize,
10906 dictionary: &'a GlobalDictionary,
10907 },
10908}
10909
10910enum Closed {
10912 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
10913 Dictionary(usize, ClosedDictionary),
10914}
10915
10916struct ClosedDictionary {
10918 distinct: Option<u64>,
10920 frequencies: Option<FrequencySummary>,
10921 texts: Vec<Option<Vec<u8>>>,
10922 hosts: Option<host::HostSummary>,
10923 encoded: EncodedDictionary,
10924 payload: u64,
10926}
10927
10928struct EncodedDictionary {
10929 index: Vec<u8>,
10930 ranks: Vec<u8>,
10931 grams: Vec<u8>,
10932}
10933
10934fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
10975 let mut work = vec![(0, codes.len(), 0)];
10976 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
10977 while let Some((from, to, depth)) = work.pop() {
10978 let part = &mut codes[from..to];
10979 keyed.clear();
10980 keyed.extend(part.iter().map(|&code| {
10981 let value = values(code);
10982 let rest = value.get(depth..).unwrap_or_default();
10983 (head(rest), rest.len().min(8) as u8, code)
10984 }));
10985 keyed.sort_unstable();
10986 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
10987 *slot = entry.2;
10988 }
10989 let mut start = 0;
10990 while start < keyed.len() {
10991 let (key, taken, _) = keyed[start];
10992 let mut end = start + 1;
10993 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
10994 end += 1;
10995 }
10996 if taken == 8 && end - start > 1 {
10997 work.push((from + start, from + end, depth + 8));
10998 }
10999 start = end;
11000 }
11001 }
11002}
11003
11004const PARALLEL_SORT_MIN: usize = 1 << 16;
11006
11007const BUCKETS_PER_WORKER: usize = 4;
11010
11011const SAMPLES_PER_BUCKET: usize = 32;
11013
11014fn sort_by_value_across<'a>(
11032 codes: &mut [u32],
11033 values: impl Fn(u32) -> &'a [u8] + Sync,
11034 workers: usize,
11035) {
11036 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11037 sort_by_value(codes, values);
11038 return;
11039 }
11040 let buckets = workers * BUCKETS_PER_WORKER;
11041 let wanted = buckets * SAMPLES_PER_BUCKET;
11042 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11043 sort_by_value(&mut sample, &values);
11044 let splitters =
11045 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11046 let values = &values;
11047 let splitters = &splitters;
11048 let per = codes.len().div_ceil(workers);
11049 let places = std::thread::scope(|scope| {
11051 codes
11052 .chunks(per)
11053 .map(|run| {
11054 scope.spawn(move || {
11055 run.iter()
11056 .map(|&code| {
11057 let value = values(code);
11058 splitters.partition_point(|splitter| *splitter <= value) as u32
11059 })
11060 .collect::<Vec<_>>()
11061 })
11062 })
11063 .collect::<Vec<_>>()
11064 .into_iter()
11065 .flat_map(|handle| {
11066 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11067 })
11068 .collect::<Vec<_>>()
11069 });
11070 let mut starts = vec![0_usize; buckets + 1];
11071 for &place in &places {
11072 starts[place as usize + 1] += 1;
11073 }
11074 for bucket in 0..buckets {
11075 starts[bucket + 1] += starts[bucket];
11076 }
11077 let mut laid = vec![0_u32; codes.len()];
11078 let mut next = starts.clone();
11079 for (&code, &place) in codes.iter().zip(&places) {
11080 laid[next[place as usize]] = code;
11081 next[place as usize] += 1;
11082 }
11083 drop(places);
11084 let mut runs = Vec::with_capacity(buckets);
11085 let mut rest = laid.as_mut_slice();
11086 for bucket in 0..buckets {
11087 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11088 runs.push(run);
11089 rest = after;
11090 }
11091 runs.sort_by_key(|run| run.len());
11093 let queue = Mutex::new(runs);
11094 std::thread::scope(|scope| {
11095 for _ in 0..workers {
11096 scope.spawn(|| {
11097 loop {
11098 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11099 let Some(run) = taken else { break };
11100 sort_by_value(run, values);
11101 }
11102 });
11103 }
11104 });
11105 codes.copy_from_slice(&laid);
11106}
11107
11108fn head(bytes: &[u8]) -> u64 {
11110 let mut word = [0; 8];
11111 let take = bytes.len().min(8);
11112 word[..take].copy_from_slice(&bytes[..take]);
11113 u64::from_be_bytes(word)
11114}
11115
11116fn encode_global_dictionary(
11127 dictionary: &GlobalDictionary,
11128 order: &[(u64, u32)],
11129 places: &[Placed],
11130 scattered: bool,
11131) -> Result<EncodedDictionary> {
11132 let values = dictionary.values();
11133 if order.len() != values {
11134 return Err(invalid("global dictionary order does not cover its values"));
11135 }
11136 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11137 if places.len() != blocks {
11138 return Err(invalid("global dictionary payload is not the blocks it says it is"));
11139 }
11140 if dictionary.grams.len() != blocks {
11141 return Err(invalid("global dictionary signatures do not cover its blocks"));
11142 }
11143 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11144 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11145 let offset_bits = offset_width(&dictionary.ends);
11146 let payload_words = if scattered { 3 } else { 2 };
11147 let index_len = DICTIONARY_HEADER
11148 .checked_add(offset_bytes(values, offset_bits))
11149 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11150 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11151 .and_then(|len| len.checked_add(8))
11152 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11153 let mut index = Vec::with_capacity(index_len);
11154 put_u32(
11155 &mut index,
11156 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11157 );
11158 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11159 put_u32(
11160 &mut index,
11161 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11162 );
11163 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11164 | DICTIONARY_GRAMS
11165 | DICTIONARY_WIDE_GRAMS;
11166 put_u32(&mut index, offset_bits as u32 | flag);
11167 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11168 let mut end = 0_u64;
11173 for place in places {
11174 if scattered {
11175 put_u64(&mut index, place.start);
11176 put_u64(&mut index, place.length);
11177 } else {
11178 end = end
11179 .checked_add(place.length)
11180 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11181 put_u64(&mut index, end);
11182 }
11183 }
11184 for place in places {
11185 put_u64(&mut index, place.hash);
11186 }
11187 if rank_ends.len() != rank_blocks {
11190 return Err(invalid("global dictionary order is not the blocks it says it is"));
11191 }
11192 for end in &rank_ends {
11193 put_u64(&mut index, *end);
11194 }
11195 let mut at = 0_usize;
11196 for end in &rank_ends {
11197 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11198 put_u64(&mut index, checksum(&ranks[at..end]));
11199 at = end;
11200 }
11201 let gram_len = blocks
11202 .checked_mul(TEXT_GRAM_BYTES)
11203 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11204 let mut grams = Vec::with_capacity(gram_len);
11205 for block in &dictionary.grams {
11206 grams.extend_from_slice(block);
11207 }
11208 put_u64(&mut index, checksum(&grams));
11209 if index.len() != index_len {
11210 return Err(invalid("global dictionary index is not the length it was laid out for"));
11211 }
11212 Ok(EncodedDictionary { index, ranks, grams })
11213}
11214
11215const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11222
11223fn payload_shapes() -> Vec<chooser::Settled> {
11249 let integers = vec![integer::Kind::Packed];
11250 [
11251 vec![string::Kind::Front, string::Kind::Lz],
11252 vec![string::Kind::Lz, string::Kind::Fsst],
11253 vec![string::Kind::Lz, string::Kind::Plain],
11254 vec![string::Kind::Fsst],
11255 vec![string::Kind::Plain],
11256 ]
11257 .into_iter()
11258 .map(|strings| chooser::Settled::new(strings, integers.clone()))
11259 .collect()
11260}
11261
11262fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11269 let started = profile.map(|_| std::time::Instant::now());
11270 file.sync()?;
11271 if let (Some(profile), Some(started)) = (profile, started) {
11272 profile.waited(
11273 Stage::Publish,
11274 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11275 );
11276 }
11277 Ok(())
11278}
11279
11280#[derive(Debug)]
11285pub(crate) struct Unencoded {
11286 column: usize,
11287 at: usize,
11288 ends: Vec<u32>,
11289 bytes: Vec<u8>,
11290 shape: chooser::Settled,
11291}
11292
11293impl Unencoded {
11294 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11296 let values = block_values(&self.ends, &self.bytes);
11297 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11298 }
11299
11300 pub(crate) fn place(&self) -> (usize, usize) {
11302 (self.column, self.at)
11303 }
11304}
11305
11306pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11310
11311fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11313 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11314 for value in values {
11315 for gram in value.windows(4) {
11316 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11317 grams[bit / 8] |= 1 << (bit % 8);
11318 }
11319 }
11320 }
11321 grams
11322}
11323
11324fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11326 let mut out = Vec::with_capacity(ends.len());
11327 let mut from = 0;
11328 for &to in ends {
11329 out.push(&bytes[from..to as usize]);
11330 from = to as usize;
11331 }
11332 out
11333}
11334
11335fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11342 for dictionary in dictionaries.iter_mut().flatten() {
11343 if !dictionary.early.is_empty() {
11344 return Err(Error::internal("a dictionary block handed out never came back"));
11345 }
11346 dictionary.seal_rest();
11347 dictionary.settle_rest()?;
11348 }
11349 encode_waiting(dictionaries)?;
11350 if dictionaries
11353 .iter()
11354 .flatten()
11355 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11356 {
11357 return Err(Error::internal("a dictionary block handed out never came back"));
11358 }
11359 Ok(())
11360}
11361
11362fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11365 let jobs = dictionaries
11366 .iter()
11367 .enumerate()
11368 .flat_map(|(column, held)| {
11369 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11370 })
11371 .collect::<Vec<_>>();
11372 if jobs.is_empty() {
11373 return Ok(());
11374 }
11375 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11376 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11377 Ok((column, at, held.encode_waiting(at)?))
11378 };
11379 let workers = std::thread::available_parallelism()
11380 .map_or(1, usize::from)
11381 .min(MAX_FREQUENCY_WORKERS)
11382 .min(jobs.len());
11383 let made = if workers <= 1 {
11384 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11385 } else {
11386 let next = AtomicUsize::new(0);
11387 let jobs = &jobs;
11388 let pieces = std::thread::scope(|scope| {
11389 (0..workers)
11390 .map(|_| {
11391 scope.spawn(|| {
11392 let mut mine = Vec::new();
11393 loop {
11394 let job = next.fetch_add(1, Atomic::Relaxed);
11395 let Some(&(column, at)) = jobs.get(job) else { break };
11396 mine.push(one(column, at)?);
11397 }
11398 Ok(mine)
11399 })
11400 })
11401 .collect::<Vec<_>>()
11402 .into_iter()
11403 .map(|handle| {
11404 handle
11405 .join()
11406 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11407 })
11408 .collect::<Result<Vec<_>>>()
11409 })?;
11410 pieces.into_iter().flatten().collect()
11411 };
11412 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11413 (0..dictionaries.len()).map(|_| Vec::new()).collect();
11414 for (column, at, bytes) in made {
11415 done[column].push((at, bytes));
11416 }
11417 for (column, mut made) in done.into_iter().enumerate() {
11418 if made.is_empty() {
11419 continue;
11420 }
11421 let Some(held) = dictionaries[column].as_mut() else { continue };
11422 made.sort_by_key(|(at, _)| *at);
11423 let waiting = std::mem::take(&mut held.waiting);
11424 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11425 if held.encoded() != at {
11426 return Err(Error::internal("a dictionary block was encoded out of order"));
11427 }
11428 held.push_block(block);
11429 }
11430 }
11431 Ok(())
11432}
11433
11434fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11444 let mut best: Option<(chooser::Settled, usize)> = None;
11445 for shape in payload_shapes() {
11446 let mut size = 0;
11447 for block in sample {
11448 size += string::encode_with(block, &shape)?.len();
11449 }
11450 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11451 best = Some((shape, size));
11452 }
11453 }
11454 best.map(|(shape, _)| shape)
11455 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11456}
11457
11458fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11465 let mut out = Vec::with_capacity(order.len() * 4);
11466 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11467 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11468 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11469 for block in order.chunks(TEXT_RANK_BLOCK) {
11470 let base = block.first().map_or(0, |&(head, _)| head);
11473 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11474 let width = (u64::BITS - span.leading_zeros()) as usize;
11475 heads.clear();
11476 codes.clear();
11477 for &(head, code) in block {
11478 heads.push(head.wrapping_sub(base));
11479 codes.push(u64::from(code));
11480 }
11481 put_u64(&mut out, base);
11482 out.push(width as u8);
11483 bitpack::pack_tail(&heads, width, &mut out)
11484 .map_err(|_| invalid("global dictionary heads do not pack"))?;
11485 bitpack::pack_tail(&codes, code_bits, &mut out)
11486 .map_err(|_| invalid("global dictionary codes do not pack"))?;
11487 ends.push(out.len() as u64);
11488 }
11489 Ok((out, ends))
11490}
11491
11492fn open_global_dictionary(
11499 file: Arc<File>,
11500 page: Page,
11501 ty: &LogicalType,
11502 keep_budget: usize,
11503) -> Result<Vector> {
11504 if !coded_type(ty) {
11505 return Err(invalid("global dictionary belongs to a non-string column"));
11506 }
11507 let mut header = [0; DICTIONARY_HEADER];
11508 read_at(&file, page.offset, &mut header)?;
11509 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11510 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11511 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11512 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11513 let scattered = width & DICTIONARY_SCATTERED != 0;
11514 let has_grams = width & DICTIONARY_GRAMS != 0;
11515 let gram_width =
11516 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11517 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11518 if per_block != TEXT_PAYLOAD_VALUES {
11519 return Err(invalid("global dictionary block width differs"));
11520 }
11521 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11522 return Err(invalid("global dictionary block count differs from its value count"));
11523 }
11524 if offset_bits > u32::BITS as usize {
11525 return Err(invalid("global dictionary packs offsets past a payload"));
11526 }
11527 let offset_len = offset_bytes(count, offset_bits);
11528 let ranks = count;
11533 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11534 let payload_words = if scattered { 3 } else { 2 };
11538 let hash_len = blocks
11539 .checked_mul(payload_words * 8)
11540 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11541 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11542 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11543 let gram_len = if has_grams {
11544 blocks
11545 .checked_mul(gram_width)
11546 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11547 } else {
11548 0
11549 };
11550 let index_len = DICTIONARY_HEADER
11551 .checked_add(offset_len)
11552 .and_then(|len| len.checked_add(hash_len))
11553 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11554 if index_len > page.length as usize {
11555 return Err(invalid("global dictionary offset index exceeds its page"));
11556 }
11557 let mut index = vec![0; index_len];
11558 index[..DICTIONARY_HEADER].copy_from_slice(&header);
11559 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11560 if checksum(&index) != page.hash {
11561 return Err(invalid("global dictionary index checksum differs"));
11562 }
11563 let word_end = index_len - usize::from(has_grams) * 8;
11564 let gram_hash = has_grams
11565 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11566 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11567 .chunks_exact(8)
11568 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11569 .collect::<Vec<_>>();
11570 let mut rest = words.split_off(blocks * payload_words);
11571 let rank_hashes = rest.split_off(rank_blocks);
11572 let rank_ends = rest;
11573 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11576 return Err(invalid("global dictionary order blocks do not rise"));
11577 }
11578 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11579 .map_err(|_| invalid("global dictionary rank overflow"))?;
11580 let body_len = index_len
11581 .checked_add(rank_len)
11582 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11583 if body_len > page.length as usize {
11584 return Err(invalid("global dictionary order exceeds its page"));
11585 }
11586 let gram_end = body_len
11587 .checked_add(gram_len)
11588 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11589 if gram_end > page.length as usize {
11590 return Err(invalid("global dictionary signatures exceed their page"));
11591 }
11592 let grams = gram_hash.map(|hash| NativeGrams {
11593 start: page.offset + body_len as u64,
11594 length: gram_len,
11595 width: gram_width,
11596 hash,
11597 verdicts: Mutex::new(Vec::new()),
11598 });
11599 let mut offsets = index;
11603 offsets.truncate(DICTIONARY_HEADER + offset_len);
11604 let hashes = words.split_off(blocks * (payload_words - 1));
11605 let (starts, lengths) = if scattered {
11606 let mut starts = Vec::with_capacity(blocks);
11607 let mut lengths = Vec::with_capacity(blocks);
11608 for pair in words.chunks_exact(2) {
11609 starts.push(pair[0]);
11610 lengths.push(pair[1]);
11611 }
11612 (starts, lengths)
11613 } else {
11614 let base = page.offset + gram_end as u64;
11618 let mut starts = Vec::with_capacity(blocks);
11619 let mut lengths = Vec::with_capacity(blocks);
11620 let mut at = 0_u64;
11621 for &end in &words {
11622 let len = end
11623 .checked_sub(at)
11624 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11625 starts.push(base + at);
11626 lengths.push(len);
11627 at = end;
11628 }
11629 (starts, lengths)
11630 };
11631 let stored_len = page.length as u64 - gram_end as u64;
11637 if scattered && stored_len == 0 {
11638 let size = file.metadata().map_err(io)?.len();
11639 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11640 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11641 });
11642 if !inside {
11643 return Err(invalid("global dictionary block lies outside the file"));
11644 }
11645 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11646 return Err(invalid("global dictionary blocks do not bound the payload"));
11647 }
11648 Vector::external_text(
11649 ty.clone(),
11650 Arc::new(NativeText {
11651 file,
11652 values: count,
11653 offsets,
11654 offset_bits,
11655 value_ends: OnceLock::new(),
11656 value_lens: OnceLock::new(),
11657 ends_asked: AtomicUsize::new(0),
11658 ranks,
11659 rank_at: page.offset + index_len as u64,
11660 rank_ends,
11661 rank_hashes,
11662 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11663 code_bits: code_width(count),
11664 code_ranks: OnceLock::new(),
11665 starts,
11666 lengths,
11667 hashes,
11668 grams,
11669 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11670 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11671 keep_budget,
11672 payload_kept: AtomicUsize::new(0),
11673 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11674 visit_dropped: AtomicUsize::new(0),
11675 searched: Mutex::new(HashMap::new()),
11676 }),
11677 )
11678}
11679
11680fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11693 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11695 let mut cur = Cursor::new(bytes);
11696 let codec = cur.u8()?;
11697 if cur.u8()? == 2 {
11698 cur.take(rows.div_ceil(8))?;
11699 }
11700 Ok((codec, cur.at))
11701 }
11702 let Ok((codec, at)) = cascade_at(rows, bytes) else {
11703 return "UNREADABLE".to_string();
11704 };
11705 let tail = &bytes[at..];
11706 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11707 match codec {
11708 0 => match ty {
11709 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11710 _ => "FIXED".to_string(),
11711 },
11712 1 => "DICT(PLAIN)".to_string(),
11713 2 => "FOR+BITPACK".to_string(),
11714 3 => "TABLE DICT".to_string(),
11715 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11716 5 => described(integer::describe(tail)),
11717 6 => described(string::describe(tail)),
11718 other => format!("CODEC {other}"),
11719 }
11720}
11721
11722fn decode_selected_stable_codes(
11727 rows: usize,
11728 bytes: &[u8],
11729 positions: &[usize],
11730 out: &mut Vec<Option<u32>>,
11731) -> Result<bool> {
11732 if positions.windows(2).any(|pair| pair[0] >= pair[1])
11733 || positions.last().is_some_and(|&position| position >= rows)
11734 {
11735 return Err(invalid("selected code positions are not sorted and in range"));
11736 }
11737 let mut cur = Cursor::new(bytes);
11738 let codec = cur.u8()?;
11739 if codec != 3 && codec != 4 {
11740 return Ok(false);
11741 }
11742 let flag = cur.u8()?;
11743 let mask = match flag {
11744 0 | 1 => None,
11745 2 => {
11746 let at = cur.at;
11747 let len = rows.div_ceil(8);
11748 cur.take(len)?;
11749 Some((at, len))
11750 }
11751 _ => return Err(invalid("page validity tag differs")),
11752 };
11753 let valid = |row: usize| match flag {
11754 0 => true,
11755 1 => false,
11756 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11757 _ => unreachable!("the validity tag was checked"),
11758 };
11759 if codec == 4 {
11760 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11761 for (&row, code) in positions.iter().zip(wide) {
11762 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11763 out.push(valid(row).then_some(code));
11764 }
11765 return Ok(true);
11766 }
11767 let codes_at = cur.at;
11768 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11769 cur.take(codes_len)?;
11770 if cur.at != bytes.len() {
11771 return Err(invalid("global code page has trailing bytes"));
11772 }
11773 let codes = &bytes[codes_at..codes_at + codes_len];
11774 for &row in positions {
11775 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11776 let code = u32::from_le_bytes(
11777 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11778 );
11779 out.push(valid(row).then_some(code));
11780 }
11781 Ok(true)
11782}
11783
11784fn decode_at(
11790 ty: &LogicalType,
11791 rows: usize,
11792 bytes: &[u8],
11793 global: Option<Arc<Vector>>,
11794 positions: &[u32],
11795) -> Result<Vector> {
11796 if positions.last().is_some_and(|&last| last as usize >= rows) {
11797 return Err(invalid("a position is past the end of the part"));
11798 }
11799 if bytes.first() != Some(&6) {
11800 return decode(ty, rows, bytes, global)?.gather(positions);
11801 }
11802 if !coded_type(ty) {
11803 return Err(invalid("compressed text codec belongs to a non-string page"));
11804 }
11805 let mut cur = Cursor::new(bytes);
11806 cur.u8()?;
11807 let validity = match cur.u8()? {
11808 0 => Validity::AllValid,
11809 1 => Validity::AllInvalid,
11810 2 => {
11811 let mask = cur.take(rows.div_ceil(8))?;
11812 Validity::from_iter(positions.len(), |at| {
11813 let row = positions[at] as usize;
11814 mask[row / 8] >> (row % 8) & 1 == 1
11815 })
11816 }
11817 _ => return Err(invalid("page validity tag differs")),
11818 };
11819 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11820 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11821 push_values(&mut values, ty, &ends)?;
11822 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11823}
11824
11825fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
11829 if ty == &LogicalType::Varchar {
11830 return values.push_run_in_place(0, ends);
11831 }
11832 let mut start = 0;
11833 for &end in ends {
11834 let len = end
11835 .checked_sub(start)
11836 .ok_or_else(|| invalid("a string value ends before it starts"))?;
11837 values.push_bytes_in_place(start, len)?;
11838 start = end;
11839 }
11840 Ok(())
11841}
11842
11843fn decode(
11844 ty: &LogicalType,
11845 rows: usize,
11846 bytes: &[u8],
11847 global: Option<Arc<Vector>>,
11848) -> Result<Vector> {
11849 let mut cur = Cursor::new(bytes);
11850 let codec = cur.u8()?;
11851 let flag = cur.u8()?;
11852 let validity = match flag {
11853 0 => Validity::AllValid,
11854 1 => Validity::AllInvalid,
11855 2 => {
11856 let mask = cur.take(rows.div_ceil(8))?;
11857 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
11858 }
11859 _ => return Err(invalid("page validity tag differs")),
11860 };
11861 if codec == 1 {
11862 if !coded_type(ty) {
11863 return Err(invalid("dictionary codec belongs to a non-string page"));
11864 }
11865 let count = cur.u32()? as usize;
11866 let payload_len = cur.u32()? as usize;
11867 let offset_bytes = cur.take(
11868 (count + 1)
11869 .checked_mul(4)
11870 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
11871 )?;
11872 let offsets = offset_bytes
11873 .chunks_exact(4)
11874 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11875 .collect::<Vec<_>>();
11876 let payload = cur.take(payload_len)?.to_vec();
11877 if offsets.first() != Some(&0)
11878 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11879 || offsets.windows(2).any(|pair| pair[0] > pair[1])
11880 {
11881 return Err(invalid("dictionary offsets do not bound the payload"));
11882 }
11883 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
11886 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
11887 push_values(&mut strings, ty, &ends)?;
11888 let mut codes = Vec::with_capacity(rows);
11889 for _ in 0..rows {
11890 codes.push(cur.u32()?);
11891 }
11892 if codes.iter().any(|code| *code as usize >= count) {
11893 return Err(invalid("dictionary code is out of range"));
11894 }
11895 if cur.at != bytes.len() {
11896 return Err(invalid("dictionary page has trailing bytes"));
11897 }
11898 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
11899 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
11900 }
11901 if codec == 3 || codec == 4 {
11902 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
11903 let codes = if codec == 4 {
11904 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
11909 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
11910 if codes.len() != rows {
11911 return Err(invalid("encoded code page holds the wrong number of rows"));
11912 }
11913 codes
11914 } else {
11915 let mut codes = Vec::with_capacity(rows);
11916 for _ in 0..rows {
11917 codes.push(cur.u32()?);
11918 }
11919 if cur.at != bytes.len() {
11920 return Err(invalid("global code page has trailing bytes"));
11921 }
11922 codes
11923 };
11924 let highest = codes.iter().copied().max();
11925 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
11926 .with_validity(validity));
11927 }
11928 if codec == 6 {
11929 if !coded_type(ty) {
11930 return Err(invalid("compressed text codec belongs to a non-string page"));
11931 }
11932 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
11936 if ends.len() != rows {
11937 return Err(invalid("compressed text page holds the wrong number of rows"));
11938 }
11939 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11942 push_values(&mut values, ty, &ends)?;
11943 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
11944 }
11945 if codec == 5 {
11946 let data = cascade(ty, &bytes[cur.at..], rows)?;
11948 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
11949 }
11950 if codec == 2 {
11951 let width = u32::from(cur.u8()?);
11952 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
11953 let count = cur.u32()? as usize;
11954 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
11955 let words: Vec<u64> = cur
11956 .take(length)?
11957 .chunks_exact(8)
11958 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
11959 .collect();
11960 if cur.at != bytes.len() {
11961 return Err(invalid("packed page has trailing bytes"));
11962 }
11963 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
11964 }
11965 if codec != 0 {
11966 return Err(invalid("page codec is unknown"));
11967 }
11968 let data = match ty {
11969 LogicalType::TinyInt => {
11970 let values = cur.take(rows)?;
11971 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
11972 }
11973 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
11974 LogicalType::SmallInt => {
11975 let values =
11976 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11977 Data::Int16(
11978 values
11979 .chunks_exact(2)
11980 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
11981 .collect::<Vec<_>>()
11982 .into(),
11983 )
11984 }
11985 LogicalType::USmallInt => {
11986 let values =
11987 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11988 Data::UInt16(
11989 values
11990 .chunks_exact(2)
11991 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
11992 .collect::<Vec<_>>()
11993 .into(),
11994 )
11995 }
11996 LogicalType::UInteger => {
11997 let values =
11998 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11999 Data::UInt32(
12000 values
12001 .chunks_exact(4)
12002 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12003 .collect::<Vec<_>>()
12004 .into(),
12005 )
12006 }
12007 LogicalType::UBigInt => {
12008 let values =
12009 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12010 Data::UInt64(
12011 values
12012 .chunks_exact(8)
12013 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12014 .collect::<Vec<_>>()
12015 .into(),
12016 )
12017 }
12018 LogicalType::Integer | LogicalType::Date => {
12019 let values =
12020 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12021 Data::Int32(
12022 values
12023 .chunks_exact(4)
12024 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12025 .collect::<Vec<_>>()
12026 .into(),
12027 )
12028 }
12029 LogicalType::BigInt
12030 | LogicalType::Timestamp
12031 | LogicalType::Time
12032 | LogicalType::TimeTz
12033 | LogicalType::TimestampTz
12034 | LogicalType::TimestampS
12035 | LogicalType::TimestampMs
12036 | LogicalType::TimestampNs => {
12037 let values =
12038 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12039 Data::Int64(
12040 values
12041 .chunks_exact(8)
12042 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12043 .collect::<Vec<_>>()
12044 .into(),
12045 )
12046 }
12047 LogicalType::HugeInt | LogicalType::Uuid => {
12048 let values =
12049 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12050 Data::Int128(
12051 values
12052 .chunks_exact(16)
12053 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12054 .collect::<Vec<_>>()
12055 .into(),
12056 )
12057 }
12058 LogicalType::UHugeInt => {
12059 let values =
12060 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12061 Data::UInt128(
12062 values
12063 .chunks_exact(16)
12064 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12065 .collect::<Vec<_>>()
12066 .into(),
12067 )
12068 }
12069 LogicalType::Float => {
12070 let values =
12071 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12072 Data::Float32(
12073 values
12074 .chunks_exact(4)
12075 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12076 .collect::<Vec<_>>()
12077 .into(),
12078 )
12079 }
12080 LogicalType::Double => {
12081 let values =
12082 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12083 Data::Float64(
12084 values
12085 .chunks_exact(8)
12086 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12087 .collect::<Vec<_>>()
12088 .into(),
12089 )
12090 }
12091 LogicalType::Interval => {
12092 let values =
12093 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12094 Data::Interval(
12095 values
12096 .chunks_exact(16)
12097 .map(|item| {
12098 (
12099 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12100 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12101 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12102 )
12103 })
12104 .collect::<Vec<_>>()
12105 .into(),
12106 )
12107 }
12108 LogicalType::Boolean => {
12109 let values = cur.take(rows)?;
12110 if values.iter().any(|value| *value > 1) {
12111 return Err(invalid("boolean page has another value"));
12112 }
12113 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12114 }
12115 LogicalType::Decimal { .. } => match ty.physical() {
12118 PhysicalType::Int16 => {
12119 let values =
12120 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12121 Data::Int16(
12122 values
12123 .chunks_exact(2)
12124 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12125 .collect::<Vec<_>>()
12126 .into(),
12127 )
12128 }
12129 PhysicalType::Int32 => {
12130 let values =
12131 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12132 Data::Int32(
12133 values
12134 .chunks_exact(4)
12135 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12136 .collect::<Vec<_>>()
12137 .into(),
12138 )
12139 }
12140 PhysicalType::Int64 => {
12141 let values =
12142 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12143 Data::Int64(
12144 values
12145 .chunks_exact(8)
12146 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12147 .collect::<Vec<_>>()
12148 .into(),
12149 )
12150 }
12151 _ => {
12152 let values =
12153 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12154 Data::Int128(
12155 values
12156 .chunks_exact(16)
12157 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12158 .collect::<Vec<_>>()
12159 .into(),
12160 )
12161 }
12162 },
12163 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12164 let offset_bytes = cur
12165 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12166 let offsets = offset_bytes
12167 .chunks_exact(4)
12168 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12169 .collect::<Vec<_>>();
12170 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12171 if offsets.first() != Some(&0)
12172 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12173 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12174 {
12175 return Err(invalid("string offsets do not bound the payload"));
12176 }
12177 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12185 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12186 push_values(&mut values, ty, &ends)?;
12187 Data::Varlen(values)
12188 }
12189 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12190 };
12191 if cur.at != bytes.len() {
12192 return Err(invalid("page has trailing bytes"));
12193 }
12194 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12195}
12196
12197#[cfg(test)]
12198mod tests {
12199 use std::fs::{self, OpenOptions};
12200 use std::io::{Seek, SeekFrom, Write};
12201 use std::path::PathBuf;
12202 use std::time::{SystemTime, UNIX_EPOCH};
12203
12204 use rudb_common::Stat;
12205 use rudb_common::Value;
12206 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12207 use rudb_common::stat::Provenance;
12208
12209 use super::*;
12210
12211 #[test]
12212 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12213 let bytes: Vec<u8> =
12214 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12215 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12216 let whole = content_name(&bytes[..length]);
12217 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12218 let mut namer = ContentNamer::default();
12219 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12220 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12221 }
12222 }
12223 }
12224
12225 #[derive(Debug)]
12228 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12229
12230 impl chooser::Chooser for TestsEverything<'_> {
12231 fn name(&self) -> &'static str {
12232 "tests everything"
12233 }
12234
12235 fn narrow_strings(
12236 &self,
12237 values: &[&[u8]],
12238 offered: &[string::Kind],
12239 depth: u8,
12240 ) -> Vec<string::Kind> {
12241 self.0.narrow_strings(values, offered, depth)
12242 }
12243
12244 fn narrow_integers(
12245 &self,
12246 values: &[i64],
12247 offered: &[integer::Kind],
12248 depth: u8,
12249 ) -> Vec<integer::Kind> {
12250 self.0.narrow_integers(values, offered, depth)
12251 }
12252 }
12253
12254 #[test]
12255 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12256 let columns: Vec<Vec<i64>> = vec![
12257 vec![],
12258 vec![5; 1000],
12259 (0..1000).collect(),
12260 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12261 (0..1000).map(|row| row / 50).collect(),
12262 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12263 (0..1000).map(|row| (row * 7919) % 13).collect(),
12264 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12265 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12266 (0..1000).map(|row| i64::MIN + row % 3).collect(),
12267 ];
12268 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12269 for column in &columns {
12270 for chooser in choosers {
12271 let quick = integer::encode_with(column, chooser).unwrap();
12272 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12273 assert_eq!(
12274 quick,
12275 full,
12276 "{} on {:?}",
12277 chooser.name(),
12278 &column[..column.len().min(8)]
12279 );
12280 }
12281 }
12282 }
12283
12284 #[test]
12287 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12288 let mut settling = Settling::default();
12289 for part in 0..STRIPE_PARTS as i64 {
12290 let values: Vec<i64> = (0..2048)
12291 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12292 .collect();
12293 let searched = integer::encode_with(&values, &Fixed).unwrap();
12294 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12295 }
12296 }
12297
12298 #[test]
12302 fn text_pages_share_a_table_until_the_text_changes() {
12303 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
12304 let english: Vec<Vec<u8>> = (0..1024)
12305 .map(|row: usize| {
12306 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
12307 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
12308 })
12309 .collect();
12310 let digits: Vec<Vec<u8>> =
12311 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
12312 let mut settling = Settling::default();
12313 for page in 0..8 {
12314 let values: Vec<&[u8]> =
12315 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
12316 let payload = values.iter().map(|value| value.len()).sum();
12317 let out = settling.text(&values, payload).unwrap().unwrap();
12318 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
12319 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
12320 assert!(
12321 out.len() * 4 <= alone.len() * 5,
12322 "page {page}: {} against {}",
12323 out.len(),
12324 alone.len()
12325 );
12326 let since = settling.symbols.as_ref().unwrap().since;
12327 assert_eq!(since, page % 4, "page {page}");
12328 }
12329 }
12330
12331 #[test]
12335 fn a_column_that_changes_under_the_shape_is_searched_again() {
12336 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12337 let mut noise = move || {
12338 state ^= state << 13;
12339 state ^= state >> 7;
12340 state ^= state << 17;
12341 (state % 1_000_000) as i64
12342 };
12343 let mut settling = Settling::default();
12344 for part in 0..STRIPE_PARTS as i64 {
12345 let values: Vec<i64> = match part / 16 {
12346 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12347 1 => (0..2048).map(|_| noise()).collect(),
12348 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12349 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12350 };
12351 let settled = settling.encode(&values).unwrap();
12352 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12353 let searched = integer::encode_with(&values, &Fixed).unwrap();
12354 assert!(
12355 settled.len() * 4 <= searched.len() * 5,
12356 "part {part}: {} settled against {} searched, {} against {}",
12357 settled.len(),
12358 searched.len(),
12359 integer::describe(&settled).unwrap(),
12360 integer::describe(&searched).unwrap(),
12361 );
12362 }
12363 }
12364
12365 #[test]
12366 fn checksum_matches_fixed_vectors() {
12367 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12368 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12369 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12370 }
12371
12372 #[test]
12373 fn sorting_across_threads_matches_sorting_on_one() {
12374 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12375 let mut next = move || {
12376 state ^= state << 13;
12377 state ^= state >> 7;
12378 state ^= state << 17;
12379 state
12380 };
12381 let mut values = Vec::new();
12382 for at in 0..150_000_u64 {
12383 let value = match next() % 6 {
12384 0 => Vec::new(),
12385 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12386 2 => format!("https://example.com/path/{at}").into_bytes(),
12387 3 => b"same".to_vec(),
12388 4 => vec![0xff; (next() % 12) as usize],
12389 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12390 };
12391 values.push(value);
12392 }
12393 let value = |code: u32| values[code as usize].as_slice();
12394 for workers in [1, 2, 3, 8, 32] {
12395 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12396 let mut across = one.clone();
12397 sort_by_value(&mut one, value);
12398 sort_by_value_across(&mut across, value, workers);
12399 assert_eq!(one, across, "{workers} workers");
12400 }
12401 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12402 sort_by_value_across(&mut sorted, value, 8);
12403 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12404 }
12405
12406 fn path(label: &str) -> PathBuf {
12407 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12408 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12409 }
12410
12411 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12416 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12417 (0..dictionary.values())
12418 .map(|code| {
12419 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12420 flat[from..to].to_vec()
12421 })
12422 .collect()
12423 }
12424
12425 fn attached(table: &Table) -> Vec<&Section> {
12432 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12433 }
12434
12435 #[test]
12437 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12438 const SPANS: usize = 64;
12439 const SPAN: usize = 512;
12440 let path = path("positional");
12441 let content: Vec<u8> =
12442 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12443 fs::write(&path, &content).expect("the file is written");
12444 let file = Arc::new(File::open(&path).expect("the file opens"));
12445 std::thread::scope(|scope| {
12446 for _ in 0..8 {
12447 let file = Arc::clone(&file);
12448 scope.spawn(move || {
12449 for _ in 0..64 {
12450 for span in 0..SPANS {
12451 let mut bytes = [0_u8; SPAN];
12452 read_at(&file, (span * SPAN) as u64, &mut bytes)
12453 .expect("the span reads");
12454 assert!(
12455 bytes.iter().all(|byte| *byte == span as u8),
12456 "span {span} came back as {}",
12457 bytes[0],
12458 );
12459 }
12460 }
12461 });
12462 }
12463 });
12464 let mut past = [0_u8; SPAN];
12465 let end = (SPANS * SPAN) as u64;
12466 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12467 assert!(error.message().contains("ends before its declared length"), "{error}");
12468 drop(file);
12469 let _ = fs::remove_file(&path);
12470 }
12471
12472 #[test]
12479 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12480 let path = path("cursor");
12481 let mut writer = Writer::create(
12482 &path,
12483 "items",
12484 vec![
12485 Field::required("id", LogicalType::Integer),
12486 Field::new("text", LogicalType::Varchar),
12487 ],
12488 )
12489 .expect("new file");
12490 writer.append(&sample()).expect("first part");
12491 writer.append(&sample()).expect("second part");
12492 writer.finish().expect("commit");
12493 let reader = Reader::open(&path).expect("reopen from disk");
12494 assert_eq!(reader.table().rows(), 6);
12495 let ids = reader.read(0, &[0]).expect("the integer page reads back");
12496 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12497 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12498 let text = reader.read(1, &[1]).expect("the text page reads back");
12499 assert_eq!(text.value_at(1, 0), Value::Null);
12500 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12501 let end = reader.table().stripes().iter().flat_map(|stripe| {
12504 stripe
12505 .pages
12506 .iter()
12507 .map(|page| page.offset + u64::from(page.length))
12508 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12509 });
12510 let last = end.fold(HEADER, u64::max);
12511 let directory = fs::metadata(&path).expect("the file is there").len();
12512 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12513 fs::remove_file(path).expect("remove scratch file");
12514 }
12515
12516 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12522 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12523 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12524 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12525 let bits = (width & !DICTIONARY_FLAGS) as usize;
12526 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12527 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12528 DICTIONARY_HEADER as u64
12529 + offset_bytes(count as usize, bits) as u64
12530 + blocks * payload_words * 8
12531 + rank_blocks * 16
12532 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12533 }
12534
12535 fn sample() -> Chunk {
12536 Chunk::new(vec![
12537 Vector::from_values(
12538 LogicalType::Integer,
12539 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12540 )
12541 .expect("integers"),
12542 Vector::from_values(
12543 LogicalType::Varchar,
12544 &[
12545 Value::Varchar("alpha".into()),
12546 Value::Null,
12547 Value::Varchar("long text after a slash".into()),
12548 ],
12549 )
12550 .expect("strings"),
12551 ])
12552 .expect("matching rows")
12553 }
12554
12555 fn sample_ids() -> Chunk {
12556 Chunk::new(vec![
12557 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12558 .expect("integers"),
12559 ])
12560 .expect("one column")
12561 }
12562
12563 #[test]
12564 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12565 let path = path("nulls_for_the_planner");
12568 let mut writer =
12569 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12570 .expect("new file");
12571 let rows = Chunk::new(vec![
12572 Vector::from_values(
12573 LogicalType::Integer,
12574 &[
12575 Value::Integer(4),
12576 Value::Null,
12577 Value::Integer(9),
12578 Value::Null,
12579 Value::Integer(1),
12580 Value::Integer(2),
12581 ],
12582 )
12583 .expect("integers"),
12584 ])
12585 .expect("one column");
12586 writer.append(&rows).expect("the only part");
12587 writer.finish().expect("commit");
12588 let reader = Reader::open(&path).expect("reopen from disk");
12589 let stripes = Stripes::new(reader);
12590 let column = stripes.column("a").expect("the file has that column");
12591 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12592 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12595 fs::remove_file(&path).expect("clean up");
12596 }
12597
12598 #[test]
12599 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
12600 let path = path("frequencies_for_the_planner");
12603 let mut writer =
12604 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12605 .expect("new file");
12606 let rows = Chunk::new(vec![
12607 Vector::from_values(
12608 LogicalType::Integer,
12609 &[
12610 Value::Integer(4),
12611 Value::Integer(4),
12612 Value::Integer(4),
12613 Value::Integer(9),
12614 Value::Integer(9),
12615 Value::Integer(1),
12616 ],
12617 )
12618 .expect("integers"),
12619 ])
12620 .expect("one column");
12621 writer.append(&rows).expect("the only part");
12622 writer.finish().expect("commit");
12623 let reader = Reader::open(&path).expect("reopen from disk");
12624 let common = Common::new(reader);
12625 assert_eq!(common.rows(), 6);
12626 let column = common.column("id").expect("the file has that column");
12627 assert_eq!(common.column("nothing"), None);
12628 assert_eq!(
12629 common.rows_with(column, &Bound::Int(4)),
12630 Stat::exact(3, Provenance::FrequencySynopsis)
12631 );
12632 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
12634 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12637 assert!(common.remainder(column).is_some());
12638 fs::remove_file(&path).expect("clean up");
12639 }
12640
12641 #[test]
12642 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12643 let path = path("string_frequencies_for_the_planner");
12644 let mut writer =
12645 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12646 .expect("new file");
12647 let rows = Chunk::new(vec![
12648 Vector::from_values(
12649 LogicalType::Varchar,
12650 &[
12651 Value::Varchar(String::new()),
12652 Value::Varchar("alpha".into()),
12653 Value::Varchar(String::new()),
12654 Value::Varchar("beta".into()),
12655 Value::Varchar(String::new()),
12656 ],
12657 )
12658 .expect("strings"),
12659 ])
12660 .expect("one column");
12661 writer.append(&rows).expect("the only part");
12662 writer.finish().expect("commit");
12663
12664 let reader = Reader::open(&path).expect("reopen from disk");
12665 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12666 let common = Common::new(reader.clone());
12667 let column = common.column("text").expect("the file has that column");
12668 assert_eq!(
12669 common.rows_with(column, &Bound::Bytes(Vec::new())),
12670 Stat::exact(3, Provenance::FrequencySynopsis)
12671 );
12672 assert_eq!(
12673 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12674 Stat::exact(0, Provenance::FrequencySynopsis)
12675 );
12676 assert_eq!(
12677 reader.reads().dictionaries,
12678 0,
12679 "the bounded spellings answer without opening the dictionary index"
12680 );
12681 fs::remove_file(&path).expect("clean up");
12682 }
12683
12684 #[test]
12685 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12686 let path = path("certified_host_groups");
12687 let mut writer =
12688 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12689 .expect("new file");
12690 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12691 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12692 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12693 values.push(Value::Varchar(String::new()));
12694 for part in values.chunks(512) {
12695 writer
12696 .append(
12697 &Chunk::new(vec![
12698 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12699 ])
12700 .expect("one column"),
12701 )
12702 .expect("part written");
12703 }
12704 writer.finish().expect("commit");
12705 let reader = Reader::open(&path).expect("reopen");
12706 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12707 fs::remove_file(&path).expect("clean up");
12708 }
12709
12710 fn bare_table(sections: Vec<Section>) -> Table {
12715 Table {
12716 name: "linked".to_owned(),
12717 fields: vec![Field::required("id", LogicalType::Integer)],
12718 stripes: Vec::new(),
12719 rows: 0,
12720 dictionaries: vec![None],
12721 dictionary_payloads: Vec::new(),
12722 demoted: Vec::new(),
12723 distincts: vec![None],
12724 frequencies: vec![None],
12725 pair_frequencies: Vec::new(),
12726 frequency_texts: Vec::new(),
12727 host_groups: None,
12728 clustering: None,
12729 generation: 1,
12730 sections,
12731 }
12732 }
12733
12734 fn a_key_map_section() -> Section {
12735 Section {
12736 kind: *section::KEY_MAP,
12737 id: 1,
12738 generation: 3,
12739 extents: 1,
12740 extent_page: HEADER,
12741 extent_bytes: section::EXTENT_BYTES as u32,
12742 hash: 0x1234_5678_9abc_def0,
12743 flags: 0,
12744 header_bytes: 24,
12745 }
12746 }
12747
12748 #[test]
12749 fn a_section_table_round_trips_through_a_directory() {
12750 let mut later = a_key_map_section();
12751 later.kind = *b"RUDBZZ9\0";
12752 later.id = 2;
12753 let table = bare_table(vec![a_key_map_section(), later]);
12754 let directory = encode_directory(&table).expect("directory");
12755 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12756 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12757 assert!(decoded.sections()[0].known());
12761 assert!(!decoded.sections()[1].known());
12762 }
12763
12764 #[test]
12765 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12766 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12770 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
12771 let older = &directory[..directory.len() - block];
12772 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
12773 assert!(decoded.sections().is_empty());
12774 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
12775 assert_eq!(decoded.name(), "linked");
12776 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
12777 }
12778
12779 #[test]
12780 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
12781 let path = path("format_twenty_two");
12788 let mut writer =
12789 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12790 .expect("new file");
12791 let rows = Chunk::new(vec![
12792 Vector::from_values(
12793 LogicalType::Integer,
12794 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
12795 )
12796 .expect("integers"),
12797 ])
12798 .expect("one column");
12799 writer.append(&rows).expect("the only part");
12800 writer.finish().expect("commit");
12801
12802 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12803 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12804 drop(file);
12805
12806 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
12807 assert_eq!(reader.table().rows(), 3);
12808 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
12813
12814 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12817 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
12818 drop(file);
12819 let error = Reader::open(&path).expect_err("format 21 is not readable");
12820 assert!(error.to_string().contains("format 21"), "{error}");
12821
12822 fs::remove_file(&path).expect("clean up");
12823 }
12824
12825 #[test]
12826 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
12827 let mut past = a_key_map_section();
12832 past.extent_page = 1 << 30;
12833 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
12834 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
12835 assert!(error.to_string().contains("outside the file"), "{error}");
12836
12837 let mut inside_the_header = a_key_map_section();
12838 inside_the_header.extent_page = 8;
12839 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
12840 assert!(
12841 decode_directory(&directory, 1 << 20).is_err(),
12842 "a section may not overlap a header"
12843 );
12844 }
12845
12846 #[test]
12847 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
12848 let not_built = Section {
12852 kind: *section::FORWARD_LINK,
12853 id: 9,
12854 generation: 3,
12855 extents: 0,
12856 extent_page: 0,
12857 extent_bytes: 0,
12858 hash: 0,
12859 flags: 0,
12860 header_bytes: 0,
12861 };
12862 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
12863 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12864 assert_eq!(decoded.sections(), &[not_built]);
12865
12866 let mut incoherent = not_built;
12869 incoherent.extent_bytes = 28;
12870 incoherent.extent_page = HEADER;
12871 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
12872 assert!(decode_directory(&directory, 1 << 20).is_err());
12873 }
12874
12875 #[test]
12876 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
12877 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12878 let mut torn = directory.clone();
12879 let count_at = torn.len() - size_of::<u16>();
12880 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
12881 assert!(decode_directory(&torn, 1 << 20).is_err());
12884 }
12885
12886 fn linked_file(label: &str, rows: i32) -> PathBuf {
12888 let path = path(label);
12889 let mut writer =
12890 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12891 .expect("new file");
12892 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
12893 let chunk =
12894 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
12895 .expect("one column");
12896 writer.append(&chunk).expect("the only part");
12897 writer.finish().expect("commit");
12898 path
12899 }
12900
12901 fn a_key_map_payload() -> Vec<u8> {
12902 (0..512_u32).flat_map(u32::to_le_bytes).collect()
12905 }
12906
12907 #[test]
12908 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
12909 let path = linked_file("attach", 64);
12910 let payload = a_key_map_payload();
12911 let table = attach(
12912 &path,
12913 "items",
12914 &[section::Attachment {
12915 kind: *section::KEY_MAP,
12916 id: 0,
12917 flags: 2,
12918 header_bytes: 40,
12919 bytes: &payload,
12920 }],
12921 )
12922 .expect("attach a key map");
12923 assert_eq!(attached(&table).len(), 1);
12924
12925 let reader = Reader::open(&path).expect("reopen after the attach");
12926 let held = attached(reader.table());
12927 assert_eq!(held.len(), 1);
12928 assert_eq!(held[0].kind, *section::KEY_MAP);
12929 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
12930 assert_eq!(held[0].header_bytes, 40);
12931 assert_eq!(held[0].generation, 1);
12935 assert!(held[0].usable(reader.table().generation()));
12936 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
12937 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
12938
12939 fs::remove_file(&path).expect("clean up");
12940 }
12941
12942 #[test]
12943 fn attaching_a_section_answers_every_row_exactly_as_before() {
12944 let path = linked_file("attach_changes_nothing", 300);
12949 let before = Reader::open(&path).expect("open before");
12950 let rows = before.table().rows();
12951 let first = before.read(0, &[0]).expect("read before");
12952 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
12953 let layout = before.layout().columns_total();
12954 drop(before);
12955
12956 let payload = a_key_map_payload();
12957 attach(
12958 &path,
12959 "items",
12960 &[section::Attachment {
12961 kind: *section::KEY_MAP,
12962 id: 0,
12963 flags: 0,
12964 header_bytes: 0,
12965 bytes: &payload,
12966 }],
12967 )
12968 .expect("attach");
12969
12970 let after = Reader::open(&path).expect("open after");
12971 assert_eq!(after.table().rows(), rows);
12972 let read = after.read(0, &[0]).expect("read after");
12973 for (at, value) in values.iter().enumerate() {
12974 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
12975 }
12976 assert_eq!(
12977 after.layout().columns_total(),
12978 layout,
12979 "an attach appends and does not rewrite a column page"
12980 );
12981
12982 fs::remove_file(&path).expect("clean up");
12983 }
12984
12985 #[test]
12986 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
12987 let path = linked_file("attach_twice", 32);
12991 let one = a_key_map_payload();
12992 let two = vec![7_u8; 1024];
12993 let entry = |bytes| section::Attachment {
12994 kind: *section::KEY_MAP,
12995 id: 4,
12996 flags: 1,
12997 header_bytes: 0,
12998 bytes,
12999 };
13000 attach(&path, "items", &[entry(&one)]).expect("first build");
13001 attach(&path, "items", &[entry(&two)]).expect("rebuild");
13002
13003 let reader = Reader::open(&path).expect("reopen");
13004 let held = attached(reader.table());
13005 assert_eq!(held.len(), 1, "one map per column and not one per build");
13006 assert_eq!(reader.payload(held[0]).expect("payload"), two);
13007
13008 fs::remove_file(&path).expect("clean up");
13009 }
13010
13011 #[test]
13012 fn an_attach_carries_through_a_kind_it_does_not_know() {
13013 let path = linked_file("attach_unknown", 16);
13017 let payload = vec![3_u8; 96];
13018 attach(
13019 &path,
13020 "items",
13021 &[section::Attachment {
13022 kind: *b"RUDBZZ9\0",
13023 id: 1,
13024 flags: 0,
13025 header_bytes: 0,
13026 bytes: &payload,
13027 }],
13028 )
13029 .expect("a kind this build does not know still writes");
13030 let key_map = a_key_map_payload();
13031 attach(
13032 &path,
13033 "items",
13034 &[section::Attachment {
13035 kind: *section::KEY_MAP,
13036 id: 0,
13037 flags: 0,
13038 header_bytes: 0,
13039 bytes: &key_map,
13040 }],
13041 )
13042 .expect("attach beside it");
13043
13044 let reader = Reader::open(&path).expect("reopen");
13045 let held = attached(reader.table());
13046 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13047 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13048 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13049
13050 fs::remove_file(&path).expect("clean up");
13051 }
13052
13053 #[test]
13054 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13055 let path = linked_file("attach_not_built", 8);
13056 attach(
13057 &path,
13058 "items",
13059 &[section::Attachment {
13060 kind: *section::FORWARD_LINK,
13061 id: 2,
13062 flags: 0,
13063 header_bytes: 0,
13064 bytes: &[],
13065 }],
13066 )
13067 .expect("record a link that did not fit the budget");
13068
13069 let reader = Reader::open(&path).expect("reopen");
13070 let held = attached(reader.table());
13071 assert_eq!(held.len(), 1);
13072 assert_eq!(held[0].extents, 0);
13073 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13074 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13075 assert!(reader.payload(held[0]).expect("no payload").is_empty());
13076
13077 fs::remove_file(&path).expect("clean up");
13078 }
13079
13080 #[test]
13081 fn a_payload_past_one_extent_is_split_and_joined_back() {
13082 let path = linked_file("attach_two_extents", 8);
13086 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13087 attach(
13088 &path,
13089 "items",
13090 &[section::Attachment {
13091 kind: *section::KEY_MAP,
13092 id: 0,
13093 flags: 0,
13094 header_bytes: 0,
13095 bytes: &payload,
13096 }],
13097 )
13098 .expect("attach a payload past the bound");
13099
13100 let reader = Reader::open(&path).expect("reopen");
13101 let held = attached(reader.table());
13102 let extents = reader.extents(held[0]).expect("extent table");
13103 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13104 assert_eq!(extents[0].length, section::MAX_EXTENT);
13105 assert_eq!(extents[1].length, 1);
13106 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13107 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13109 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13110
13111 fs::remove_file(&path).expect("clean up");
13112 }
13113
13114 #[test]
13115 fn a_torn_extent_is_refused_rather_than_decoded() {
13116 let path = linked_file("attach_torn", 8);
13117 let payload = a_key_map_payload();
13118 attach(
13119 &path,
13120 "items",
13121 &[section::Attachment {
13122 kind: *section::KEY_MAP,
13123 id: 0,
13124 flags: 0,
13125 header_bytes: 0,
13126 bytes: &payload,
13127 }],
13128 )
13129 .expect("attach");
13130
13131 let reader = Reader::open(&path).expect("reopen");
13132 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13133 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13134 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13135 drop(file);
13136
13137 let reader = Reader::open(&path).expect("the table still opens");
13138 let error = reader
13139 .payload(&reader.table().sections()[0])
13140 .expect_err("a corrupt payload is not handed out");
13141 assert!(error.to_string().contains("checksum"), "{error}");
13142 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13145
13146 fs::remove_file(&path).expect("clean up");
13147 }
13148
13149 #[test]
13150 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13151 let path = linked_file("attach_old_format", 8);
13154 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13155 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13156 drop(file);
13157
13158 let payload = a_key_map_payload();
13159 let error = attach(
13160 &path,
13161 "items",
13162 &[section::Attachment {
13163 kind: *section::KEY_MAP,
13164 id: 0,
13165 flags: 0,
13166 header_bytes: 0,
13167 bytes: &payload,
13168 }],
13169 )
13170 .expect_err("format 22 cannot gain a section");
13171 assert!(error.to_string().contains("format 22"), "{error}");
13172 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13173
13174 fs::remove_file(&path).expect("clean up");
13175 }
13176
13177 #[test]
13178 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13179 let path = linked_file("attach_bad_header", 8);
13180 let error = attach(
13181 &path,
13182 "items",
13183 &[section::Attachment {
13184 kind: *section::KEY_MAP,
13185 id: 0,
13186 flags: 0,
13187 header_bytes: 40,
13188 bytes: &[1, 2, 3],
13189 }],
13190 )
13191 .expect_err("a writer's bug stops at the write");
13192 assert!(error.to_string().contains("header is longer"), "{error}");
13193
13194 fs::remove_file(&path).expect("clean up");
13195 }
13196
13197 #[test]
13198 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13199 let path = linked_file("attach_wrong_name", 8);
13200 let error = attach(&path, "orders", &[]).expect_err("no such table");
13201 assert!(error.to_string().contains("orders"), "{error}");
13202 fs::remove_file(&path).expect("clean up");
13203 }
13204
13205 #[test]
13206 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13207 let path = path("frequency_prefix_for_the_planner");
13214 let mut writer =
13215 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13216 .expect("new file");
13217 let mut values = vec![Value::Integer(1); 10_000];
13218 for _ in 0..10 {
13219 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13220 }
13221 for part in values.chunks(8_000) {
13224 let rows = Chunk::new(vec![
13225 Vector::from_values(LogicalType::Integer, part).expect("integers"),
13226 ])
13227 .expect("one column");
13228 writer.append(&rows).expect("a part");
13229 }
13230 writer.finish().expect("commit");
13231 let reader = Reader::open(&path).expect("reopen from disk");
13232 let prefix =
13233 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13234 assert_eq!(prefix.entries.len(), 512);
13237 assert_eq!(prefix.omitted_max, 10);
13238 let common = Common::new(reader);
13239 assert_eq!(common.rows(), 16_000);
13240 let column = common.column("id").expect("the file has that column");
13241 assert_eq!(
13242 common.rows_with(column, &Bound::Int(1)),
13243 Stat::exact(10_000, Provenance::FrequencySynopsis)
13244 );
13245 assert_eq!(
13247 common.rows_with(column, &Bound::Int(1_100)),
13248 Stat::exact(10, Provenance::FrequencySynopsis)
13249 );
13250 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13253 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13256 let remainder = common.remainder(column).expect("the list is a prefix");
13260 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13261 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13262 fs::remove_file(&path).expect("clean up");
13263 }
13264
13265 #[test]
13267 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13268 let path = path("empty");
13269 Writer::empty(&path, &[]).expect("a file with nothing in it");
13270 let catalog = Catalog::open(&path).expect("the empty file opens");
13271 assert_eq!(catalog.len(), 0);
13272 assert!(catalog.is_empty());
13273 assert_eq!(catalog.names().count(), 0);
13274 let mut writer =
13277 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13278 .expect("a table goes into the empty file");
13279 writer.append(&sample_ids()).expect("rows");
13280 writer.finish().expect("commit");
13281 let catalog = Catalog::open(&path).expect("the file opens again");
13282 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13283 fs::remove_file(&path).expect("clean up");
13284 }
13285
13286 #[test]
13296 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13297 let path = path("empty-name");
13298 let field = || vec![Field::required("id", LogicalType::Integer)];
13299 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13300 let catalog = Catalog::open(&path).expect("the file opens");
13301 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13302
13303 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13304 writer.append(&sample_ids()).expect("rows");
13305 writer.finish().expect("commit");
13306 let catalog = Catalog::open(&path).expect("the file opens again");
13307 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13309 let held = catalog.rows().collect::<Vec<_>>();
13310 assert_eq!(held.len(), 1);
13311 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13312
13313 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13315 assert!(error.to_string().contains("same name"), "{error}");
13316 fs::remove_file(&path).expect("clean up");
13317 }
13318
13319 fn sample_view(name: &str) -> ViewEntry {
13321 ViewEntry {
13322 name: name.to_string(),
13323 sql: "SELECT id FROM items WHERE id > 0".to_string(),
13324 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13325 aliases: vec!["n".to_string()],
13326 columns: vec![Field::new("n", LogicalType::Integer)],
13327 }
13328 }
13329
13330 #[test]
13331 fn a_view_written_into_the_catalog_comes_back_whole() {
13332 let path = path("views");
13333 let mut writer =
13334 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13335 .expect("new file");
13336 writer.append(&sample_ids()).expect("rows");
13337 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13338 let catalog = Catalog::open(&path).expect("reopen");
13339 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13340 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13343 fs::remove_file(&path).expect("clean up");
13344 }
13345
13346 #[test]
13348 fn appending_a_table_carries_the_views_forward() {
13349 let path = path("viewscarry");
13350 let mut writer =
13351 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13352 .expect("new file");
13353 writer.append(&sample_ids()).expect("rows");
13354 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13355 let mut writer =
13356 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13357 .expect("a second table");
13358 writer.append(&sample_ids()).expect("rows");
13359 writer.finish().expect("commit");
13360 let catalog = Catalog::open(&path).expect("reopen");
13361 assert_eq!(catalog.views().count(), 1);
13362 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13363 fs::remove_file(&path).expect("clean up");
13364 }
13365
13366 #[test]
13368 fn restating_the_views_leaves_every_table_where_it_was() {
13369 let path = path("restate");
13370 let mut writer =
13371 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13372 .expect("new file");
13373 writer.append(&sample_ids()).expect("rows");
13374 writer.finish().expect("commit");
13375 let before = fs::metadata(&path).expect("the file is there").len();
13376 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13377 let catalog = Catalog::open(&path).expect("reopen");
13378 assert_eq!(catalog.views().count(), 2);
13379 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13380 let after = fs::metadata(&path).expect("the file is there").len();
13383 assert!(after > before, "a generation was written");
13384 assert!(after - before < before, "the table was not written again");
13385 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13388 assert_eq!(reader.table().rows, 3);
13389 Writer::restate(&path, &[]).expect("no views at all");
13392 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13393 fs::remove_file(&path).expect("clean up");
13394 }
13395
13396 #[test]
13398 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13399 let bytes = encode_catalog(
13400 &[Entry {
13401 name: "items".to_string(),
13402 fields: vec![Field::required("id", LogicalType::Integer)],
13403 rows: 1,
13404 directory: Page { offset: HEADER, length: 8, hash: 0 },
13405 nonzero: vec![None],
13406 aggregates: vec![None],
13407 distincts: vec![None],
13408 extremes: vec![None],
13409 frequencies: vec![None],
13410 }],
13411 &[sample_view("items")],
13412 )
13413 .expect("it encodes, because encoding does not look");
13414 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13415 assert!(error.to_string().contains("same name"), "{error}");
13416 }
13417
13418 #[test]
13421 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13422 let rows: usize = 300;
13423 let text: Vec<String> =
13424 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13425 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13426 let mut page = vec![6, 2];
13427 page.extend((0..rows.div_ceil(8)).map(|byte| {
13428 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13429 }));
13430 let compressed = string::encode_only(string::Kind::Fsst, &values)
13431 .expect("encoded")
13432 .expect("text this repetitive compresses");
13433 page.extend_from_slice(&compressed);
13434 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13435 let positions = [0_u32, 3, 8, 13, 200, 299];
13436 let some =
13437 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13438 assert_eq!(some.len(), positions.len());
13439 for (at, &row) in positions.iter().enumerate() {
13440 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13441 }
13442 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13443 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13444 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13445 }
13446
13447 #[test]
13450 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13451 let path = path("rows");
13452 let mut writer = Writer::create(
13453 &path,
13454 "items",
13455 vec![
13456 Field::required("id", LogicalType::Integer),
13457 Field::new("text", LogicalType::Varchar),
13458 ],
13459 )
13460 .expect("new file");
13461 let rows = 2_000;
13462 let chunk = Chunk::new(vec![
13463 Vector::from_values(
13464 LogicalType::Integer,
13465 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13466 )
13467 .expect("integers"),
13468 Vector::from_values(
13469 LogicalType::Varchar,
13470 &(0..rows)
13471 .map(|row| {
13472 if row % 7 == 2 {
13473 Value::Null
13474 } else {
13475 Value::Varchar(format!("a comment about order {}", row * 13))
13476 }
13477 })
13478 .collect::<Vec<_>>(),
13479 )
13480 .expect("strings"),
13481 ])
13482 .expect("matching rows");
13483 writer.append(&chunk).expect("one part");
13484 writer.finish().expect("commit");
13485 let reader = Reader::open(&path).expect("reopen from disk");
13486 let positions = [1_u32, 2, 9, 1_000, 1_999];
13487 for whole in [true, false] {
13488 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13489 let all = reader.read(0, &[0, 1]).expect("the whole part");
13490 assert_eq!(some.len(), positions.len());
13491 for column in 0..2 {
13492 for (at, &row) in positions.iter().enumerate() {
13493 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13494 }
13495 }
13496 }
13497 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13498 }
13499
13500 #[test]
13501 fn committed_file_reopens_and_reads_only_requested_columns() {
13502 let path = path("reopen");
13503 let mut writer = Writer::create(
13504 &path,
13505 "items",
13506 vec![
13507 Field::required("id", LogicalType::Integer),
13508 Field::new("text", LogicalType::Varchar),
13509 ],
13510 )
13511 .expect("new file");
13512 writer.append(&sample()).expect("first part");
13513 writer.append(&sample()).expect("second part");
13514 writer.finish().expect("commit");
13515 let reader = Reader::open(&path).expect("reopen from disk");
13516 assert_eq!(reader.table().rows(), 6);
13517 assert_eq!(reader.table().stripes().len(), 1);
13520 assert_eq!(reader.parts(), 2);
13521 assert_eq!(reader.part_rows(0), 3);
13522 assert_eq!(reader.part_rows(1), 3);
13523 let text = reader.read(1, &[1]).expect("only text page");
13524 assert_eq!(text.width(), 1);
13525 assert_eq!(text.value_at(1, 0), Value::Null);
13526 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13527 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13528 assert_eq!(sparse.width(), 1);
13529 assert_eq!(sparse.value_at(1, 0), Value::Null);
13530 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13531 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13532 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13533 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13534 let count = reader.read(0, &[]).expect("no page is needed for count");
13535 assert_eq!(count.len(), 3);
13536 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13537 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13538 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
13539 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
13540 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
13541 assert_eq!(integers.omitted_max, 2);
13542 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13543 assert_eq!(strings.len(), 3);
13544 assert!(strings.contains(&(Value::Null, 2)));
13545 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13546 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13547 fs::remove_file(path).expect("remove scratch file");
13548 }
13549
13550 #[test]
13558 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13559 let path = path("interleaved-runs");
13560 let mut writer =
13561 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13562 .expect("new file");
13563 for morsel in [2_u64, 0, 3, 1] {
13564 let parts = (0..4_u64)
13565 .map(|chunk| {
13566 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13567 let values =
13568 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13569 let column =
13570 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13571 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13572 })
13573 .collect::<Vec<_>>();
13574 writer.append_stripe(parts).expect("a stripe");
13575 }
13576 writer.finish().expect("commit");
13577
13578 let reader = Reader::open(&path).expect("valid directory");
13579 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13580 assert_eq!(reader.table().rows(), 128);
13581 for part in 0..16_usize {
13582 let read = reader.read(part, &[0]).expect("a part back");
13583 for row in 0..8_usize {
13584 let want = i64::try_from(part * 8 + row).expect("small");
13585 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13586 }
13587 }
13588 fs::remove_file(path).expect("remove scratch file");
13589 }
13590
13591 #[test]
13594 fn runs_that_overlap_each_other_are_refused_at_commit() {
13595 let path = path("overlapping-runs");
13596 let mut writer =
13597 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13598 .expect("new file");
13599 let one = |order: (u64, u64)| {
13600 let column =
13601 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13602 (order, Chunk::new(vec![column]).expect("one column"))
13603 };
13604 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13607 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13608 let error = writer.finish().expect_err("the runs overlap");
13609 assert!(error.message().contains("source order"), "{error}");
13610 fs::remove_file(path).expect("remove scratch file");
13611 }
13612
13613 #[test]
13616 fn a_run_longer_than_a_stripe_is_refused() {
13617 let path = path("overlong-run");
13618 let mut writer =
13619 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13620 .expect("new file");
13621 let parts = (0..=STRIPE_PARTS)
13622 .map(|at| {
13623 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13624 .expect("a column");
13625 let chunk = Chunk::new(vec![column]).expect("one column");
13626 ((0, u64::try_from(at).expect("small")), chunk)
13627 })
13628 .collect::<Vec<_>>();
13629 let error = writer.append_stripe(parts).expect_err("one part too many");
13630 assert!(error.message().contains("more parts than it holds"), "{error}");
13631 fs::remove_file(path).expect("remove scratch file");
13632 }
13633
13634 #[test]
13640 fn parts_past_the_stripe_bound_start_a_new_stripe() {
13641 let path = path("stripe-bound");
13642 let mut writer = Writer::create(
13643 &path,
13644 "items",
13645 vec![
13646 Field::required("id", LogicalType::Integer),
13647 Field::new("text", LogicalType::Varchar),
13648 ],
13649 )
13650 .expect("new file");
13651 let parts = STRIPE_PARTS * 2 + 3;
13652 for part in 0..parts {
13653 let id = part as i32;
13654 let chunk = Chunk::new(vec![
13655 Vector::from_values(
13656 LogicalType::Integer,
13657 &[Value::Integer(id), Value::Integer(-id)],
13658 )
13659 .expect("integers"),
13660 Vector::from_values(
13661 LogicalType::Varchar,
13662 &[Value::Varchar(format!("value {part}")), Value::Null],
13663 )
13664 .expect("strings"),
13665 ])
13666 .expect("matching rows");
13667 writer.append(&chunk).expect("one part");
13668 }
13669 writer.finish().expect("commit");
13670
13671 let reader = Reader::open(&path).expect("reopen from disk");
13672 assert_eq!(reader.parts(), parts);
13673 assert_eq!(reader.table().rows(), parts * 2);
13674 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13675 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13676 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13677 assert_eq!(reader.table().stripes()[2].parts(), 3);
13678 for part in (0..parts).rev() {
13681 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13682 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13683 for chunk in [&dense, &sparse] {
13684 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13685 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13686 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13687 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13688 assert_eq!(chunk.value_at(1, 1), Value::Null);
13689 }
13690 }
13691 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13694 assert!(reader.skips(0, &above), "the first stripe stops at 63");
13695 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13696 fs::remove_file(path).expect("remove scratch file");
13697 }
13698
13699 fn scattered(n: i64) -> i64 {
13701 n.wrapping_mul(-7_046_029_254_386_353_131)
13702 }
13703
13704 #[test]
13710 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13711 let path = path("sieve-skip");
13712 let mut writer =
13713 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13714 .expect("new file");
13715 let parts = STRIPE_PARTS + 3;
13716 let per_part = 128;
13720 for part in 0..parts {
13721 let held: Vec<Value> = (0..per_part)
13722 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13723 .collect();
13724 let chunk =
13725 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13726 .expect("one column");
13727 writer.append(&chunk).expect("one part");
13728 }
13729 writer.finish().expect("commit");
13730
13731 let reader = Reader::open(&path).expect("reopen from disk");
13732 let probe = |value: i64| Probe {
13733 column: 0,
13734 op: Op::Equal,
13735 value: Bound::Int(i128::from(scattered(value))),
13736 };
13737 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13738 let tests = [probe(wanted)];
13739 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13740 let home = wanted as usize / per_part;
13741 assert!(kept.contains(&home), "the part holding {wanted} is read");
13742 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13746 }
13747 let absent = [probe((parts * per_part) as i64 + 1)];
13748 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13749 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13750 let tests = [probe(0)];
13753 assert!(
13754 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13755 "the bounds rule out no stripe at all"
13756 );
13757 fs::remove_file(path).expect("remove scratch file");
13758 }
13759
13760 #[test]
13766 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
13767 let path = path("part-range-skip");
13768 let mut writer =
13769 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13770 .expect("new file");
13771 let parts = STRIPE_PARTS + 3;
13772 let per_part = 128;
13773 for part in 0..parts {
13774 let held: Vec<Value> = (0..per_part)
13778 .map(|row| {
13779 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13780 })
13781 .collect();
13782 let chunk =
13783 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13784 .expect("one column");
13785 writer.append(&chunk).expect("one part");
13786 }
13787 writer.finish().expect("commit");
13788
13789 let reader = Reader::open(&path).expect("reopen from disk");
13790 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13791 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
13792 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
13793 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
13795 fs::remove_file(path).expect("remove scratch file");
13796 }
13797
13798 #[test]
13802 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
13803 let path = path("part-range-certain");
13804 let mut writer =
13805 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13806 .expect("new file");
13807 let parts = STRIPE_PARTS + 3;
13808 let per_part = 128;
13809 for part in 0..parts {
13810 let held: Vec<Value> = (0..per_part)
13811 .map(|row| {
13812 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13813 })
13814 .collect();
13815 let chunk =
13816 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13817 .expect("one column");
13818 writer.append(&chunk).expect("one part");
13819 }
13820 writer.finish().expect("commit");
13821
13822 let reader = Reader::open(&path).expect("reopen from disk");
13823 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13824 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
13825 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
13826 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
13829 fs::remove_file(path).expect("remove scratch file");
13830 }
13831
13832 #[test]
13835 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
13836 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
13837 let path = path("part-range-page");
13838 let mut writer =
13839 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13840 .expect("new file");
13841 for part in 0..parts {
13842 let held: Vec<Value> = (0..128)
13843 .map(|row| {
13844 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
13845 })
13846 .collect();
13847 let chunk = Chunk::new(vec![
13848 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
13849 ])
13850 .expect("one column");
13851 writer.append(&chunk).expect("one part");
13852 }
13853 writer.finish().expect("commit");
13854 let reader = Reader::open(&path).expect("reopen from disk");
13855 let bytes = reader.layout().columns[0].part_ranges;
13856 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
13857 fs::remove_file(path).expect("remove scratch file");
13858 }
13859 }
13860
13861 #[test]
13864 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
13865 let long = vec![b'a'; PART_BOUND_BYTES * 2];
13866 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
13867 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
13868 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
13869 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
13870 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
13871 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
13872 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
13873 }
13874
13875 #[test]
13878 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
13879 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
13880 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
13881 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
13882 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
13883 }
13884
13885 #[test]
13897 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
13898 let parts = 4;
13899 let per_part = 1024;
13900 let rows = parts * per_part;
13901 let written = |name: &str, keys: &[i64]| {
13902 let path = path(name);
13903 let fields = vec![Field::required("key", LogicalType::BigInt)];
13904 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
13905 for part in 0..parts {
13906 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
13907 .iter()
13908 .map(|key| Value::BigInt(*key))
13909 .collect();
13910 let chunk = Chunk::new(vec![
13911 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
13912 ])
13913 .expect("one column");
13914 writer.append(&chunk).expect("one part");
13915 }
13916 writer.finish().expect("commit");
13917 path
13918 };
13919 let climbing = |step: &dyn Fn(usize) -> i64| {
13922 let mut key = 0;
13923 (0..rows)
13924 .map(|row| {
13925 key += step(row);
13926 key
13927 })
13928 .collect::<Vec<i64>>()
13929 };
13930 let ascending = climbing(&|row| (row % 3) as i64);
13931 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
13935 let near_path = written("stored-near", &ascending);
13936 let far_path = written("stored-far", &sparse);
13937
13938 let one = Reader::open(&near_path).expect("reopen from disk");
13939 let other = Reader::open(&far_path).expect("reopen from disk");
13940 let near = one.stored(0).expect("the column is stored");
13941 let far = other.stored(0).expect("the column is stored");
13942 assert_eq!(near.len(), parts, "one row per part");
13943 assert_eq!(far.len(), parts);
13944 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
13947 assert_eq!(total(&near), one.layout().columns[0].pages);
13948 assert_eq!(total(&far), other.layout().columns[0].pages);
13949 assert!(
13950 total(&near) * 2 < total(&far),
13951 "the sparse keys cost more, {} against {}",
13952 total(&far),
13953 total(&near)
13954 );
13955 for (at, part) in near.iter().enumerate() {
13957 assert_eq!(part.part, at);
13958 assert_eq!(part.row, at * per_part);
13959 assert_eq!(part.rows, per_part);
13960 let held = &ascending[at * per_part..(at + 1) * per_part];
13961 assert_eq!(part.low, Some(Value::BigInt(held[0])));
13962 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
13963 assert_eq!(part.nulls, Some(0));
13964 }
13965 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
13968 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
13969 assert_ne!(near[0].encoding, far[0].encoding);
13970 fs::remove_file(near_path).expect("remove scratch file");
13971 fs::remove_file(far_path).expect("remove scratch file");
13972 }
13973
13974 #[test]
13984 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
13985 let path = path("sieve-pays");
13986 let fields = vec![
13987 Field::required("spread", LogicalType::BigInt),
13988 Field::required("repeated", LogicalType::BigInt),
13989 ];
13990 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
13991 let parts = 3;
13992 let per_part = 1024;
13993 for part in 0..parts {
13994 let base = (part * per_part) as i64;
13995 let spread: Vec<Value> =
13996 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
13997 let repeated: Vec<Value> =
13998 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
13999 let chunk = Chunk::new(vec![
14000 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14001 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14002 ])
14003 .expect("two columns");
14004 writer.append(&chunk).expect("one part");
14005 }
14006 writer.finish().expect("commit");
14007
14008 let reader = Reader::open(&path).expect("reopen from disk");
14009 let layout = reader.layout();
14010 let spread = &layout.columns[0];
14011 let repeated = &layout.columns[1];
14012 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14013 assert_eq!(
14014 repeated.sieves, 0,
14015 "a column whose filter costs more than its parts keeps none"
14016 );
14017 for column in &layout.columns {
14020 assert!(
14021 column.sieves < column.pages,
14022 "{} spends {} on sieves over {} of data",
14023 column.name,
14024 column.sieves,
14025 column.pages
14026 );
14027 }
14028 let absent = [Probe {
14030 column: 0,
14031 op: Op::Equal,
14032 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14033 }];
14034 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14035 fs::remove_file(path).expect("remove scratch file");
14036 }
14037
14038 #[test]
14044 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14045 let path = path("sieve-damaged");
14046 let mut writer =
14047 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14048 .expect("new file");
14049 let rows = 128;
14050 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14051 let chunk =
14052 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14053 .expect("one column");
14054 writer.append(&chunk).expect("one part");
14055 writer.finish().expect("commit");
14056
14057 let page = Reader::open(&path).expect("reopen").table.stripes[0]
14058 .sieves
14059 .get(0)
14060 .expect("a sieve page");
14061 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14062 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14063 file.write_all(&[0xff]).expect("damage one byte");
14064 drop(file);
14065
14066 let reader = Reader::open(&path).expect("reopen the damaged file");
14067 let absent =
14068 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14069 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14070 assert_eq!(
14071 reader.read(0, &[0]).expect("the rows are untouched").len(),
14072 usize::try_from(rows).expect("a small count")
14073 );
14074 fs::remove_file(path).expect("remove scratch file");
14075 }
14076
14077 #[test]
14088 fn workers_that_want_the_same_stripe_read_it_once() {
14089 let path = path("single-flight");
14090 let mut writer =
14091 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14092 .expect("new file");
14093 for part in 0..STRIPE_PARTS {
14094 let id = part as i32;
14095 let chunk = Chunk::new(vec![
14096 Vector::from_values(
14097 LogicalType::Integer,
14098 &[Value::Integer(id), Value::Integer(-id)],
14099 )
14100 .expect("integers"),
14101 ])
14102 .expect("matching rows");
14103 writer.append(&chunk).expect("one part");
14104 }
14105 writer.finish().expect("commit");
14106
14107 let reader = Reader::open(&path).expect("reopen from disk");
14108 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14109 let barrier = std::sync::Barrier::new(8);
14110 std::thread::scope(|scope| {
14111 for worker in 0..8 {
14112 let reader = &reader;
14113 let barrier = &barrier;
14114 scope.spawn(move || {
14115 barrier.wait();
14116 for part in (worker..STRIPE_PARTS).step_by(8) {
14117 let chunk = reader.read(part, &[0]).expect("a whole page read");
14118 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14119 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14120 }
14121 });
14122 }
14123 });
14124 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14125 fs::remove_file(path).expect("remove scratch file");
14126 }
14127
14128 #[test]
14141 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14142 let opened = |label: &str, rows_per_part: i32| {
14143 let path = path(label);
14144 let mut writer =
14145 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14146 .expect("new file");
14147 for part in 0..STRIPE_PARTS * 3 {
14148 let values = (0..rows_per_part)
14152 .map(|row| {
14153 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14154 })
14155 .collect::<Vec<_>>();
14156 let chunk = Chunk::new(vec![
14157 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14158 ])
14159 .expect("matching rows");
14160 writer.append(&chunk).expect("one part");
14161 }
14162 writer.finish().expect("commit");
14163 let reader = Reader::open(&path).expect("reopen from disk");
14164 let size = fs::metadata(&path).expect("the file is there").len();
14165 let out = (reader.reads(), reader.table().stripes().len(), size);
14166 fs::remove_file(path).expect("remove scratch file");
14167 out
14168 };
14169
14170 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14171 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14172 assert_eq!(
14173 thin_stripes, fat_stripes,
14174 "the same stripe count is what makes this a fair ask"
14175 );
14176 assert!(
14177 fat_size > thin_size * 50,
14178 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14179 );
14180
14181 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14182 assert_eq!(thin.pages, 0, "opening read a page");
14183 assert_eq!(fat.pages, 0, "opening read a page");
14184 assert_eq!(thin.indexes, 0, "opening read an index");
14185 assert_eq!(fat.indexes, 0, "opening read an index");
14186 assert!(
14189 fat.opening.bytes < thin.opening.bytes * 2,
14190 "opening the thin file read {} bytes and the fat one read {}",
14191 thin.opening.bytes,
14192 fat.opening.bytes
14193 );
14194 }
14195
14196 #[test]
14204 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14205 let path = path("open-twice");
14206 let mut writer =
14207 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14208 .expect("new file");
14209 for part in 0..STRIPE_PARTS * 3 {
14210 let chunk = Chunk::new(vec![
14211 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14212 .expect("integers"),
14213 ])
14214 .expect("matching rows");
14215 writer.append(&chunk).expect("one part");
14216 }
14217 writer.finish().expect("commit");
14218
14219 let first = Reader::open(&path).expect("open");
14220 for part in 0..first.parts() {
14223 first.read(part, &[0]).expect("a part");
14224 }
14225 assert!(first.reads().pages > 0, "the scan has to have read something");
14226 let second = Reader::open(&path).expect("open again");
14227
14228 assert_eq!(first.reads().opening, second.reads().opening);
14229 assert_eq!(
14230 second.reads().pages,
14231 0,
14232 "the second open read a page off the back of the first"
14233 );
14234 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14235 fs::remove_file(path).expect("remove scratch file");
14236 }
14237
14238 #[test]
14246 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14247 let path = path("index-cache");
14248 let mut writer =
14249 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14250 .expect("new file");
14251 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14252 for part in 0..parts {
14253 let id = part as i32;
14254 let chunk = Chunk::new(vec![
14255 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14256 ])
14257 .expect("matching rows");
14258 writer.append(&chunk).expect("one part");
14259 }
14260 writer.finish().expect("commit");
14261
14262 let reader = Reader::open(&path).expect("reopen from disk");
14263 let stripes = reader.table().stripes().len();
14264 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14265 for _ in 0..2 {
14267 for part in 0..parts {
14268 let chunk = reader.read(part, &[0]).expect("a part");
14269 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14270 }
14271 }
14272 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14273 assert!(
14274 reader.pages.load(Atomic::Relaxed) > stripes,
14275 "the pages are the ones that get read again, which is what makes the index count mean \
14276 something"
14277 );
14278 fs::remove_file(path).expect("remove scratch file");
14279 }
14280
14281 #[test]
14288 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14289 let path = path("page-pool");
14290 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
14291 let fields = || vec![Field::required("id", LogicalType::Integer)];
14292 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14293 for table in ["a", "b"] {
14294 if table == "b" {
14295 writer = writer.next("b".to_string(), fields()).expect("a second table");
14296 }
14297 for part in 0..parts {
14298 let chunk = Chunk::new(vec![
14299 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14300 .expect("integers"),
14301 ])
14302 .expect("matching rows");
14303 writer.append(&chunk).expect("one part");
14304 }
14305 }
14306 writer.finish().expect("commit");
14307
14308 let pool = PagePool::new(usize::MAX);
14309 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14310 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14311 let stripes = a.table().stripes().len();
14312 assert!(
14313 stripes > CACHED_STRIPES_PER_COLUMN * 2,
14314 "the floor has to be smaller than a table"
14315 );
14316 let scan = |reader: &Reader| {
14317 for part in 0..parts {
14318 let chunk = reader.read(part, &[0]).expect("a part");
14319 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14320 }
14321 };
14322 scan(&a);
14325 assert_eq!(pool.bytes(), 0, "a page read once is not the pool's");
14326 scan(&a);
14327 let twice = stripes * 2 - CACHED_STRIPES_PER_COLUMN;
14328 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the second scan reads the rest again");
14329 scan(&a);
14330 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the third scan reads nothing");
14331 let one = pool.bytes();
14332 assert!(one > 0, "the pool counts what the reader holds");
14333
14334 pool.budget.store(one, Atomic::Relaxed);
14336 scan(&b);
14337 scan(&b);
14338 assert_eq!(b.pages.load(Atomic::Relaxed), twice, "a page is never let go while in use");
14339 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14340 let column = a.cache.columns[0].lock().expect("the column");
14341 let held = column.pages.iter().flatten().count();
14342 assert_eq!(
14343 held,
14344 CACHED_STRIPES_PER_COLUMN + column.passing.len(),
14345 "the count and the slots agree"
14346 );
14347 drop(column);
14348
14349 drop((a, b, catalog));
14351 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14352 scan(&c);
14353 scan(&c);
14354 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14355 fs::remove_file(path).expect("remove scratch file");
14356 }
14357
14358 #[test]
14367 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14368 let workers = CACHED_STRIPES_PER_COLUMN + 4;
14369 let path = path("stripe-per-worker");
14370 let mut writer =
14371 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14372 .expect("new file");
14373 for part in 0..STRIPE_PARTS * workers {
14374 let chunk = Chunk::new(vec![
14375 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14376 .expect("integers"),
14377 ])
14378 .expect("matching rows");
14379 writer.append(&chunk).expect("one part");
14380 }
14381 writer.finish().expect("commit");
14382
14383 let read = |told: bool| {
14384 let reader = Reader::open(&path).expect("reopen from disk");
14385 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14386 if told {
14387 reader.keep_stripes(workers);
14388 }
14389 let barrier = std::sync::Barrier::new(workers);
14390 std::thread::scope(|scope| {
14391 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14392 let reader = &reader;
14393 let barrier = &barrier;
14394 scope.spawn(move || {
14395 for part in run {
14396 barrier.wait();
14397 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14398 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14399 }
14400 assert!(worker < workers);
14401 });
14402 }
14403 });
14404 reader.pages.load(Atomic::Relaxed)
14405 };
14406
14407 assert_eq!(read(true), workers, "one page read per stripe and no more");
14408 assert!(read(false) > workers, "a cache that small is read again on every part");
14409 fs::remove_file(path).expect("remove scratch file");
14410 }
14411
14412 #[test]
14417 fn a_damaged_index_page_is_an_error() {
14418 let path = path("damaged-index");
14419 let mut writer =
14420 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14421 .expect("new file");
14422 writer.append(&sample_ids()).expect("first part");
14423 writer.append(&sample_ids()).expect("second part");
14424 writer.finish().expect("commit");
14425
14426 let reader = Reader::open(&path).expect("valid directory");
14427 let index = reader.table.stripes[0].index;
14428 let mut byte = [0; 1];
14429 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14430 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14431 file.seek(SeekFrom::Start(index.offset)).expect("index start");
14432 file.write_all(&[!byte[0]]).expect("damage the first part length");
14433 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14434 assert!(error.message().contains("index page section checksum differs"), "{error}");
14435 fs::remove_file(path).expect("remove scratch file");
14436 }
14437
14438 #[test]
14445 fn every_integer_width_round_trips_through_a_page() {
14446 let path = path("integer-widths");
14447 let columns = [
14448 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14449 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14450 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14451 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14452 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14453 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14454 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14455 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14456 ];
14457 let fields = columns
14458 .iter()
14459 .enumerate()
14460 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14461 .collect::<Vec<_>>();
14462 let vectors = columns
14463 .iter()
14464 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14465 .collect::<Vec<_>>();
14466 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14467 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14468 writer.finish().expect("commit");
14469
14470 let reader = Reader::open(&path).expect("reopen from disk");
14471 let wanted = (0..columns.len()).collect::<Vec<_>>();
14472 let read = reader.read(0, &wanted).expect("every column");
14473 assert_eq!(read.len(), 2);
14474 for (at, (ty, values)) in columns.iter().enumerate() {
14476 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14477 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14478 }
14479 fs::remove_file(path).expect("remove scratch file");
14480 }
14481
14482 #[test]
14493 fn every_other_type_the_format_knows_round_trips_through_a_page() {
14494 let path = path("other-types");
14495 let columns = [
14496 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14497 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14498 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14499 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14500 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14501 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14502 (
14503 LogicalType::TimestampTz,
14504 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14505 ),
14506 (
14507 LogicalType::Interval,
14508 vec![
14509 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14510 Value::Interval { months: 13, days: -1, micros: 1 },
14511 ],
14512 ),
14513 (
14514 LogicalType::Blob,
14515 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14516 ),
14517 ];
14518 let fields = columns
14519 .iter()
14520 .enumerate()
14521 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14522 .collect::<Vec<_>>();
14523 let vectors = columns
14524 .iter()
14525 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14526 .collect::<Vec<_>>();
14527 let mut writer = Writer::create(&path, "others", fields).expect("new file");
14528 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14529 writer.finish().expect("commit");
14530
14531 let reader = Reader::open(&path).expect("reopen from disk");
14532 let wanted = (0..columns.len()).collect::<Vec<_>>();
14533 let read = reader.read(0, &wanted).expect("every column");
14534 assert_eq!(read.len(), 2);
14535 for (at, (ty, values)) in columns.iter().enumerate() {
14536 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14537 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14538 }
14539 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14542 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14543
14544 fs::remove_file(path).expect("remove scratch file");
14545 }
14546
14547 #[test]
14553 fn a_nan_survives_being_written_down() {
14554 let path = path("nan");
14555 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14556 .expect("a NaN vector");
14557 let mut writer =
14558 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14559 .expect("new file");
14560 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14561 writer.finish().expect("commit");
14562 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14563 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14564 assert!(back.is_nan(), "a NaN came back as {back}");
14565 fs::remove_file(path).expect("remove scratch file");
14566 }
14567
14568 #[test]
14575 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14576 let path = path("uuid-and-bit");
14577 let uuids = vec![0_i128, i128::MIN, -1];
14578 let mut bits = StringColumn::new();
14579 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14580 bits.push_bytes(value);
14581 }
14582 let expected = bits.clone();
14583 let fields =
14584 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14585 let vectors = vec![
14586 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14587 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14588 ];
14589 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14590 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14591 writer.finish().expect("commit");
14592
14593 let reader = Reader::open(&path).expect("reopen from disk");
14594 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14595 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14596 panic!("a uuid column is the 128 bit lane")
14597 };
14598 assert_eq!(back.as_slice(), uuids.as_slice());
14599 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14600 panic!("a bit column is bytes")
14601 };
14602 for row in 0..expected.len() {
14603 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14604 }
14605 fs::remove_file(path).expect("remove scratch file");
14606 }
14607
14608 #[test]
14611 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14612 let mut rows: Vec<Option<u64>> = Vec::new();
14613 let mut state = 0x2545_f491_4f6c_dd1d_u64;
14614 for index in 0..400_000_u64 {
14615 state ^= state << 13;
14616 state ^= state >> 7;
14617 state ^= state << 17;
14618 let times = 1 + (state % 7) as usize;
14619 let bits = match state % 11 {
14620 0 => None,
14621 1..=3 => Some(state % 16),
14622 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14623 };
14624 rows.extend(std::iter::repeat_n(bits, times));
14625 }
14626 let mut by_row = Candidates::default();
14627 for &bits in &rows {
14628 by_row.add(bits, 1);
14629 }
14630 let mut by_run = Candidates::default();
14631 let mut run = Run::default();
14632 let mut runs = 0_usize;
14633 for &bits in &rows {
14634 if let Some((bits, times)) = run.push(bits) {
14635 by_run.add(bits, times);
14636 runs += 1;
14637 }
14638 }
14639 if let Some((bits, times)) = run.take() {
14640 by_run.add(bits, times);
14641 }
14642 assert!(runs < rows.len() / 2, "the rows came in runs");
14643 assert!(by_row.decrements > 0, "the table filled and turned values away");
14644 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14645 assert_eq!(by_run.nulls, by_row.nulls);
14646 assert_eq!(by_run.decrements, by_row.decrements);
14647 }
14648
14649 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14650 let mut pairs = candidates.pairs().collect::<Vec<_>>();
14651 pairs.sort_unstable();
14652 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14653 pairs
14654 }
14655
14656 #[derive(Default)]
14659 struct MapCandidates {
14660 counts: HashMap<u64, u32>,
14661 nulls: u32,
14662 decrements: u64,
14663 }
14664
14665 impl MapCandidates {
14666 fn add(&mut self, bits: Option<u64>, mut times: u32) {
14667 while times > 0 {
14668 let held = match bits {
14669 Some(bits) => self.counts.get_mut(&bits),
14670 None if self.nulls != 0 => Some(&mut self.nulls),
14671 None => None,
14672 };
14673 if let Some(count) = held {
14674 *count = count.saturating_add(times);
14675 return;
14676 }
14677 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14678 match bits {
14679 Some(bits) => {
14680 self.counts.insert(bits, times);
14681 }
14682 None => self.nulls = times,
14683 }
14684 return;
14685 }
14686 self.counts.retain(|_, count| {
14687 *count -= 1;
14688 *count != 0
14689 });
14690 self.nulls = self.nulls.saturating_sub(1);
14691 self.decrements = self.decrements.saturating_add(1);
14692 times -= 1;
14693 }
14694 }
14695 }
14696
14697 #[test]
14701 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14702 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14703 let mut table = Candidates::default();
14704 let mut oracle = MapCandidates::default();
14705 let mut state = seed;
14706 for index in 0..300_000_u64 {
14707 state ^= state << 13;
14708 state ^= state >> 7;
14709 state ^= state << 17;
14710 let bits = match state % 13 {
14711 0 => None,
14712 1..=4 => Some(state % 40),
14713 5 => Some((index % 1000) * 1_000_000),
14714 _ => Some(state),
14715 };
14716 let times = 1 + (state >> 60) as u32 % 3;
14717 table.add(bits, times);
14718 oracle.add(bits, times);
14719 if index % 50_000 == 0 {
14720 let mut expected =
14721 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14722 expected.sort_unstable();
14723 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14724 }
14725 }
14726 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14727 expected.sort_unstable();
14728 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14729 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14730 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14731 assert!(table.decrements > 0, "seed {seed} never filled the table");
14732 for &(bits, _) in &expected {
14733 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14734 }
14735 }
14736 }
14737
14738 #[test]
14739 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14740 let path = path("frequency-ordinals");
14741 let mut writer =
14742 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14743 .expect("new file");
14744 let mut values = Vec::new();
14745 for leader in 0..10_i64 {
14746 values.extend(std::iter::repeat_n(leader, 100));
14747 }
14748 values.extend(1_000_i64..41_000);
14749 for part in values.chunks(1_024) {
14750 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14751 .expect("big integers");
14752 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14753 }
14754 writer.finish().expect("commit");
14755
14756 let reader = Reader::open(&path).expect("reopen from disk");
14757 let occurrences =
14758 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14759 assert!(occurrences.omitted_max < 100);
14760 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14761 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14762 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14763 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14764 assert_eq!(
14765 &occurrences.anchor_indices[..1_000]
14766 .iter()
14767 .map(|&entry| occurrences.anchors[entry as usize].clone())
14768 .collect::<Vec<_>>(),
14769 &(0_i64..10)
14770 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
14771 .collect::<Vec<_>>()
14772 );
14773 fs::remove_file(path).expect("remove scratch file");
14774 }
14775
14776 #[test]
14777 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
14778 let path = path("frequency-bits");
14783 let mut writer = Writer::create(
14784 &path,
14785 "items",
14786 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
14787 )
14788 .expect("new file");
14789 let mut rows = Vec::new();
14790 let mut leaders = Vec::new();
14791 for leader in 0..10_u64 {
14792 let count = 300 - leader * 10;
14793 let (unsigned, signed) = if leader == 0 {
14794 (Value::Null, Value::Null)
14795 } else {
14796 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
14797 };
14798 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
14799 leaders.push(((unsigned, count), (signed, count)));
14800 }
14801 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
14802 for part in rows.chunks(1_024) {
14803 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
14804 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
14805 let chunk = Chunk::new(vec![
14806 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
14807 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
14808 ])
14809 .expect("matching columns");
14810 writer.append(&chunk).expect("rows");
14811 }
14812 writer.finish().expect("commit");
14813
14814 let reader = Reader::open(&path).expect("reopen from disk");
14815 for column in 0..2 {
14816 let prefix =
14817 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14818 let wanted = leaders
14819 .iter()
14820 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
14821 .cloned()
14822 .collect::<Vec<_>>();
14823 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
14824 assert!(prefix.omitted_max < 210, "column {column}");
14825 assert_eq!(
14826 reader.distinct_values(column).expect("valid metadata"),
14827 Some(9 + 40_000),
14828 "column {column}"
14829 );
14830 }
14831 fs::remove_file(path).expect("remove scratch file");
14832 }
14833
14834 #[test]
14835 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
14836 let path = path("frequency-tally");
14842 let types = [
14843 LogicalType::TinyInt,
14844 LogicalType::UInteger,
14845 LogicalType::Date,
14846 LogicalType::Timestamp,
14847 ];
14848 let value = |ty: &LogicalType, at: i64| match ty {
14849 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
14850 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
14851 LogicalType::Date => Value::Date(19_000 - at as i32),
14852 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
14853 };
14854 let fields = types
14855 .iter()
14856 .enumerate()
14857 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
14858 .collect::<Vec<_>>();
14859 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14860 let mut rows = Vec::new();
14861 for at in 0..250_i64 {
14862 for _ in 0..=(at % 37) {
14863 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
14864 }
14865 }
14866 for part in rows.chunks(1_000) {
14867 let columns = types
14868 .iter()
14869 .map(|ty| {
14870 let values = part
14871 .iter()
14872 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
14873 .collect::<Vec<_>>();
14874 Vector::from_values(ty.clone(), &values).expect("a column")
14875 })
14876 .collect();
14877 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
14878 }
14879 writer.finish().expect("commit");
14880
14881 let reader = Reader::open(&path).expect("reopen from disk");
14882 for (column, ty) in types.iter().enumerate() {
14883 let mut counts = HashMap::<Option<i64>, u64>::new();
14884 for row in &rows {
14885 *counts.entry(*row).or_default() += 1;
14886 }
14887 let wanted = counts
14888 .into_iter()
14889 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
14890 .collect::<Vec<_>>();
14891 let prefix =
14892 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14893 assert_eq!(prefix.entries.len(), 2, "column {column}");
14894 assert!(prefix.omitted_max > 0, "column {column}");
14895 for (value, count) in &prefix.entries {
14896 let held =
14897 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
14898 assert_eq!(held, Some(count), "column {column} value {value:?}");
14899 }
14900 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
14901 assert_eq!(
14902 reader.distinct_values(column).expect("valid metadata"),
14903 Some(wanted.len() as u64 - 1),
14904 "column {column}"
14905 );
14906 }
14907 fs::remove_file(path).expect("remove scratch file");
14908 }
14909
14910 #[test]
14911 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
14912 let edge = FREQUENCY_CANDIDATES as i64;
14917 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
14918 for with_null in [false, true] {
14919 let path = path("distinct-edge");
14920 let mut writer =
14921 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
14922 .expect("new file");
14923 let mut values = Vec::new();
14924 for round in 0..2 {
14925 for value in 0..distinct {
14926 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
14927 values.extend(std::iter::repeat_n(
14928 Value::BigInt(value * 7_919 % distinct),
14929 repeat,
14930 ));
14931 if with_null && value % 1_000 == 0 {
14932 values.push(Value::Null);
14933 }
14934 }
14935 }
14936 if with_null {
14937 values.push(Value::Null);
14938 }
14939 for part in values.chunks(1_024) {
14940 let chunk = Chunk::new(vec![
14941 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
14942 ])
14943 .expect("one column");
14944 writer.append(&chunk).expect("rows");
14945 }
14946 writer.finish().expect("commit");
14947 let reader = Reader::open(&path).expect("reopen from disk");
14948 assert_eq!(
14949 reader.distinct_values(0).expect("valid metadata"),
14950 Some(distinct as u64),
14951 "{distinct} values, null {with_null}"
14952 );
14953 fs::remove_file(path).expect("remove scratch file");
14954 }
14955 }
14956 }
14957
14958 #[test]
14959 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
14960 let path = path("quick-nonzero");
14961 let mut writer = Writer::create(
14962 &path,
14963 "items",
14964 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
14965 )
14966 .expect("create");
14967 for ids in [
14968 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
14969 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
14970 ] {
14971 let labels = vec![Value::Varchar("same".into()); ids.len()];
14972 writer
14973 .append(
14974 &Chunk::new(vec![
14975 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
14976 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
14977 ])
14978 .expect("chunk"),
14979 )
14980 .expect("append");
14981 }
14982 writer.finish().expect("finish");
14983 let catalog = Catalog::open(&path).expect("catalog");
14984 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
14985 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
14986 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
14987 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
14988 let prefix = catalog
14989 .table("items")
14990 .expect("reader")
14991 .frequency_prefix(1)
14992 .expect("valid metadata")
14993 .expect("partial frequencies");
14994 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
14995 assert_eq!(prefix.omitted_max, 1);
14996 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
14997 assert_eq!(
14998 catalog.integer_extremes("items", 1).expect("extremes"),
14999 Some(IntegerExtremes::Values { low: 0, high: 7 })
15000 );
15001 assert_eq!(
15002 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15003 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15004 );
15005 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15006 let mut legacy = catalog.clone();
15007 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15008 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15009 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15010 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15011 Writer::certify_counts(&path).expect("recertify");
15012 assert_eq!(
15013 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15014 Some(2)
15015 );
15016 assert_eq!(
15017 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15018 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15019 );
15020 assert_eq!(
15021 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15022 Some(3)
15023 );
15024 assert_eq!(
15025 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15026 Some(IntegerExtremes::Values { low: 0, high: 7 })
15027 );
15028 assert_eq!(
15029 Catalog::open(&path)
15030 .expect("reopen")
15031 .exact_numeric_frequencies("items", 1)
15032 .expect("frequencies"),
15033 None
15034 );
15035 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15036 fs::remove_file(path).expect("remove scratch file");
15037 }
15038
15039 #[test]
15040 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15041 let path = path("pair-frequencies");
15042 let mut pairs = Vec::new();
15043 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15044 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15045 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15046 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15047 let mut writer = Writer::create(
15048 &path,
15049 "items",
15050 vec![
15051 Field::required("id", LogicalType::BigInt),
15052 Field::required("phrase", LogicalType::Varchar),
15053 ],
15054 )
15055 .expect("new file");
15056 for part in pairs.chunks(1_024) {
15057 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15058 let phrases =
15059 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15060 writer
15061 .append(
15062 &Chunk::new(vec![
15063 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15064 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15065 ])
15066 .expect("matching columns"),
15067 )
15068 .expect("rows");
15069 }
15070 writer.finish().expect("commit");
15071
15072 let reader = Reader::open(&path).expect("reopen from disk");
15073 assert!(
15074 reader.table.pair_frequencies.is_empty(),
15075 "no query-specific pair result is stored"
15076 );
15077 fs::remove_file(path).expect("remove scratch file");
15078 }
15079
15080 #[test]
15081 fn legacy_group_answers_are_ignored() {
15082 let path = path("legacy-group-answers");
15083 let mut writer = Writer::create(
15084 &path,
15085 "items",
15086 vec![
15087 Field::required("id", LogicalType::BigInt),
15088 Field::required("text", LogicalType::Varchar),
15089 ],
15090 )
15091 .expect("new file");
15092 writer
15093 .append(
15094 &Chunk::new(vec![
15095 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15096 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15097 .expect("text"),
15098 ])
15099 .expect("row"),
15100 )
15101 .expect("append");
15102 writer.finish().expect("commit");
15103 let mut reader = Reader::open(&path).expect("reopen");
15104 let table = Arc::make_mut(&mut reader.table);
15105 table.pair_frequencies.push(PairFrequencySummary {
15106 first: 0,
15107 second: 1,
15108 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15109 omitted_max: 0,
15110 });
15111 table.host_groups = Some(host::HostSummary {
15112 column: 1,
15113 omitted_max: 0,
15114 entries: vec![host::HostEntry {
15115 host: "fake.test".into(),
15116 count: 999,
15117 bytes_sum: 999,
15118 minimum: "x".into(),
15119 }],
15120 });
15121 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
15122 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
15123 fs::remove_file(path).expect("remove scratch file");
15124 }
15125
15126 #[test]
15132 fn a_file_from_another_format_says_which_format_it_is() {
15133 let older = path("older-format");
15134 let mut writer =
15135 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15136 .expect("new file");
15137 let chunk = Chunk::new(vec![
15138 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15139 .expect("integers"),
15140 ])
15141 .expect("chunk");
15142 writer.append(&chunk).expect("page written");
15143 writer.finish().expect("commit");
15144
15145 let unreadable =
15149 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15150 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15151 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15152 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15153 drop(file);
15154 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15155 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15156 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15157
15158 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15159 file.seek(SeekFrom::Start(0)).expect("the magic is first");
15160 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15161 drop(file);
15162 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15163 assert!(complaint.contains("magic"), "{complaint}");
15164 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15165 fs::remove_file(older).expect("remove scratch file");
15166 }
15167
15168 #[test]
15169 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15170 let unfinished = path("unfinished");
15171 let mut writer =
15172 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15173 .expect("new file");
15174 let chunk = Chunk::new(vec![
15175 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15176 .expect("integers"),
15177 ])
15178 .expect("chunk");
15179 writer.append(&chunk).expect("page written");
15180 drop(writer);
15181 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15182 fs::remove_file(unfinished).expect("remove scratch file");
15183
15184 let damaged = path("damaged");
15185 let mut writer =
15186 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15187 .expect("new file");
15188 writer.append(&chunk).expect("page written");
15189 writer.finish().expect("commit");
15190 let reader = Reader::open(&damaged).expect("valid directory");
15191 let mut file =
15192 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15193 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15194 file.write_all(&[255]).expect("damage one byte");
15195 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15196 fs::remove_file(damaged).expect("remove scratch file");
15197 }
15198
15199 #[test]
15200 fn damaged_lazy_dictionary_payload_is_an_error() {
15201 let path = path("damaged-dictionary");
15202 let mut writer = Writer::create(
15203 &path,
15204 "items",
15205 vec![
15206 Field::required("id", LogicalType::Integer),
15207 Field::new("text", LogicalType::Varchar),
15208 ],
15209 )
15210 .expect("new file");
15211 writer.append(&sample()).expect("stripe written");
15212 writer.finish().expect("commit");
15213
15214 let reader = Reader::open(&path).expect("valid directory");
15215 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15216 let mut header = [0; DICTIONARY_HEADER];
15219 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15220 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15223 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15224 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15225 let bits = (width & !DICTIONARY_FLAGS) as usize;
15226 let mut start = [0; 8];
15227 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15228 read_at(&reader.file, at, &mut start).expect("the first block's start");
15229 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15230 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15231 file.write_all(&[255]).expect("damage dictionary payload");
15232
15233 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15234 let error =
15235 chunk.validate_external().expect_err("payload corruption must reach the caller");
15236 assert!(error.message().contains("payload checksum differs"), "{error}");
15237 fs::remove_file(path).expect("remove scratch file");
15238 }
15239
15240 #[test]
15250 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15251 let path = path("dictionary-decide");
15252 let rows = 20_000;
15253 let unique =
15255 |row: usize| format!("{row:09} a value that appears exactly once in the table");
15256 let repeated = |row: usize| unique(row / 40);
15258 let mut writer = Writer::create(
15259 &path,
15260 "items",
15261 vec![
15262 Field::required("unique", LogicalType::Varchar),
15263 Field::required("repeated", LogicalType::Varchar),
15264 ],
15265 )
15266 .expect("new file");
15267 for part in (0..rows).step_by(1_000) {
15268 let span = part..(part + 1_000).min(rows);
15269 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15270 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15271 writer
15272 .append(
15273 &Chunk::new(vec![
15274 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15275 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15276 ])
15277 .expect("two columns"),
15278 )
15279 .expect("a part");
15280 }
15281 writer.finish().expect("commit");
15282
15283 let reader = Reader::open(&path).expect("reopen from disk");
15284 assert!(
15285 reader.table.dictionaries[0].is_none(),
15286 "a column with no repeats has nothing to say twice"
15287 );
15288 assert!(
15289 reader.table.dictionaries[1].is_some(),
15290 "a column whose values come round again keeps its dictionary"
15291 );
15292 let mut first = 0;
15293 for part in 0..reader.parts() {
15294 let chunk = reader.read(part, &[0, 1]).expect("a part");
15295 for row in 0..chunk.len() {
15296 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15297 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15298 }
15299 first += chunk.len();
15300 }
15301 assert_eq!(first, rows, "every row was read back");
15302 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15303 let size = fs::metadata(&path).expect("the file is there").len() as usize;
15304 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15305 fs::remove_file(path).expect("remove scratch file");
15306 }
15307
15308 #[test]
15321 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15322 let path = path("dictionary-blocks");
15323 let value = |row: usize| {
15324 let row = row.saturating_sub(8_000);
15325 format!("{row:07} a value long enough to be worth a payload block")
15326 };
15327 let parts = 40;
15328 let per_part = 1000;
15329 let mut writer =
15330 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15331 .expect("new file");
15332 for part in 0..parts {
15333 let values = (0..per_part)
15334 .map(|row| Value::Varchar(value(part * per_part + row)))
15335 .collect::<Vec<_>>();
15336 let chunk = Chunk::new(vec![
15337 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15338 ])
15339 .expect("matching rows");
15340 writer.append(&chunk).expect("a part");
15341 }
15342 writer.finish().expect("commit");
15343
15344 let reader = Reader::open(&path).expect("reopen from disk");
15345 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15346 assert!(
15347 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15348 "the dictionary has to be several blocks for this to be testing anything"
15349 );
15350 for part in [0, parts - 1] {
15351 let chunk = reader.read(part, &[0]).expect("a part");
15352 chunk.validate_external().expect("every payload block checks out");
15353 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15354 }
15355
15356 let mut header = [0; DICTIONARY_HEADER];
15358 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15359 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15360 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15361 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15362 let bits = (width & !DICTIONARY_FLAGS) as usize;
15363 let mut place = [0; 16];
15364 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15365 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15366 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15367 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15368 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15369 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15370 file.write_all(&[255]).expect("damage the last payload block");
15371 let reader = Reader::open(&path).expect("the directory and the index are untouched");
15372 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15373 let error = chunk.validate_external().expect_err("the damage must reach the caller");
15374 assert!(error.message().contains("payload checksum differs"), "{error}");
15375 fs::remove_file(path).expect("remove scratch file");
15376 }
15377
15378 #[test]
15392 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15393 let path = path("dictionary-offsets");
15394 let value = |row: usize| {
15395 let row = row % 5_000;
15396 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15397 };
15398 let rows = 6_000;
15399 let mut writer =
15400 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15401 .expect("new file");
15402 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15403 for part in values.chunks(1_000) {
15404 let chunk =
15405 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15406 .expect("matching rows");
15407 writer.append(&chunk).expect("a part");
15408 }
15409 writer.finish().expect("commit");
15410
15411 let reader = Reader::open(&path).expect("reopen from disk");
15412 assert!(
15413 rows > TEXT_PAYLOAD_VALUES * 4,
15414 "the dictionary has to be several blocks for this to be testing anything"
15415 );
15416 for part in 0..rows / 1_000 {
15417 let chunk = reader.read(part, &[0]).expect("a part");
15418 for row in 0..1_000 {
15419 let row = part * 1_000 + row;
15420 assert_eq!(
15421 chunk.value_at(row % 1_000, 0),
15422 Value::Varchar(value(row)),
15423 "value {row}"
15424 );
15425 }
15426 }
15427 for _ in 0..2 {
15430 for part in 0..rows / 1_000 {
15431 let chunk = reader.read(part, &[0]).expect("a part");
15432 let mut lens = vec![0_i64; 1_000];
15433 let column = chunk.column(0).expect("one column");
15434 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15435 for (row, &len) in lens.iter().enumerate() {
15436 let row = part * 1_000 + row;
15437 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15438 }
15439 }
15440 }
15441 fs::remove_file(path).expect("remove scratch file");
15442 }
15443
15444 #[test]
15446 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15447 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15448 ends.extend([3, 3, 10]);
15449 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15450 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15451 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15452 let long = [5, 70_005, 70_006];
15454 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15455 assert_eq!(lens, [5, 70_000, 1]);
15456 let mut read = Vec::new();
15457 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15458 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15459 ends.push(9);
15460 assert!(lengths_of(&ends).is_none());
15461 }
15462
15463 #[test]
15475 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15476 let path = path("dictionary-once");
15477 let parts = 8;
15478 let per_part = 500;
15479 let value =
15480 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15481 let mut writer =
15482 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15483 .expect("new file");
15484 for part in 0..parts {
15485 let values = (0..per_part)
15486 .map(|row| Value::Varchar(value(part * per_part + row)))
15487 .collect::<Vec<_>>();
15488 let chunk = Chunk::new(vec![
15489 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15490 ])
15491 .expect("matching rows");
15492 writer.append(&chunk).expect("a part");
15493 }
15494 writer.finish().expect("commit");
15495
15496 let reader = Reader::open(&path).expect("reopen from disk");
15497 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15498 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15499
15500 let workers = 16;
15501 let gate = std::sync::Barrier::new(workers);
15502 std::thread::scope(|scope| {
15503 for worker in 0..workers {
15504 let reader = reader.clone();
15505 let gate = &gate;
15506 scope.spawn(move || {
15507 gate.wait();
15508 let chunk = reader.read(worker % parts, &[0]).expect("a part");
15509 assert_eq!(
15510 chunk.value_at(0, 0),
15511 Value::Varchar(value((worker % parts) * per_part))
15512 );
15513 });
15514 }
15515 });
15516
15517 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15518 fs::remove_file(path).expect("remove scratch file");
15519 }
15520
15521 #[test]
15526 fn a_damaged_sorted_order_is_an_error() {
15527 let path = path("damaged-order");
15528 let mut writer = Writer::create(
15529 &path,
15530 "items",
15531 vec![
15532 Field::required("id", LogicalType::Integer),
15533 Field::new("text", LogicalType::Varchar),
15534 ],
15535 )
15536 .expect("new file");
15537 writer.append(&sample()).expect("stripe written");
15538 writer.finish().expect("commit");
15539
15540 let reader = Reader::open(&path).expect("valid directory");
15541 let page = reader.table.dictionaries[1].expect("string dictionary page");
15542 let mut header = [0; DICTIONARY_HEADER];
15543 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15544 let index_len = dictionary_index_len(&header);
15545 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15546 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15547 file.write_all(&[255]).expect("damage the order");
15548
15549 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15550 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15551 assert!(error.message().contains("rank checksum differs"), "{error}");
15552 fs::remove_file(path).expect("remove scratch file");
15553 }
15554
15555 #[test]
15559 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15560 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15563 let path = path("dictionary-order");
15564 let mut writer =
15565 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15566 .expect("new file");
15567 writer
15568 .append(
15569 &Chunk::new(vec![
15570 Vector::from_values(
15571 LogicalType::Varchar,
15572 &spellings.map(|text| Value::Varchar(text.into())),
15573 )
15574 .expect("strings"),
15575 ])
15576 .expect("one column"),
15577 )
15578 .expect("stripe written");
15579 writer.finish().expect("commit");
15580
15581 let reader = Reader::open(&path).expect("valid directory");
15582 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15583 let count = dictionary.ranks().expect("a v10 file stores one");
15584 assert_eq!(count, spellings.len(), "every distinct value has a rank");
15585 let order = (0..count)
15586 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15587 .collect::<Vec<_>>();
15588 let mut seen = order.clone();
15589 seen.sort_unstable();
15590 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15591
15592 let ranked = order
15593 .iter()
15594 .map(|&code| {
15595 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15596 })
15597 .collect::<Vec<_>>();
15598 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15599 expected.sort();
15600 assert_eq!(ranked, expected, "rank order is value order");
15601
15602 for (rank, value) in expected.iter().enumerate() {
15605 assert_eq!(
15606 dictionary.compare_rank(rank, value).expect("compare"),
15607 Ordering::Equal,
15608 "rank {rank} is its own value"
15609 );
15610 if rank > 0 {
15611 assert_eq!(
15612 dictionary.compare_rank(rank - 1, value).expect("compare"),
15613 Ordering::Less,
15614 "rank {rank} follows the one before it"
15615 );
15616 }
15617 }
15618 fs::remove_file(path).expect("remove scratch file");
15619 }
15620
15621 #[test]
15628 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15629 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15630 let path = path("dictionaries-at-once");
15631 let fields = (0..sizes.len())
15632 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15633 .collect::<Vec<_>>();
15634 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15635 let rows = 10_000_usize;
15636 for start in (0..rows).step_by(1_024) {
15637 let columns = sizes
15638 .iter()
15639 .enumerate()
15640 .map(|(column, &size)| {
15641 let values = (start..(start + 1_024).min(rows))
15642 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15643 .collect::<Vec<_>>();
15644 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15645 })
15646 .collect::<Vec<_>>();
15647 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15648 }
15649 writer.finish().expect("commit");
15650
15651 let reader = Reader::open(&path).expect("valid directory");
15652 for (column, &size) in sizes.iter().enumerate() {
15653 let dictionary =
15654 reader.dictionary(column).expect("read").expect("a string column has one");
15655 let count = dictionary.ranks().expect("a v10 file stores one");
15656 assert_eq!(count, size, "column {column} has its own distinct count");
15657 let ranked = (0..count)
15658 .map(|rank| {
15659 let code = dictionary.code_at_rank(rank).expect("a code");
15660 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15661 })
15662 .collect::<Vec<_>>();
15663 let expected = (0..size)
15664 .map(|value| format!("c{column}-{value:05}").into_bytes())
15665 .collect::<Vec<_>>();
15666 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15667 }
15668 fs::remove_file(path).expect("remove scratch file");
15669 }
15670
15671 #[test]
15679 fn a_large_dictionary_ranks_in_value_order() {
15680 let path = path("dictionary-large-rank");
15681 let value = |row: u64| {
15682 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15683 match row % 3 {
15684 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15685 1 => format!("{mixed}"),
15686 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15687 }
15688 };
15689 let distinct = 70_000;
15690 let parts = 4 * distinct / 1000;
15691 let per_part = 1000;
15692 let mut writer =
15693 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15694 .expect("new file");
15695 for part in 0..parts {
15696 let values = (0..per_part)
15697 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15698 .collect::<Vec<_>>();
15699 let chunk = Chunk::new(vec![
15700 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15701 ])
15702 .expect("matching rows");
15703 writer.append(&chunk).expect("a part");
15704 }
15705 writer.finish().expect("commit");
15706
15707 let reader = Reader::open(&path).expect("reopen from disk");
15708 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15709 let count = dictionary.ranks().expect("a ranked dictionary");
15710 assert_eq!(count, distinct as usize, "every distinct value has a rank");
15711 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15712 let ranked = (0..count)
15713 .map(|rank| {
15714 let code = dictionary.code_at_rank(rank).expect("a code");
15715 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15716 })
15717 .collect::<Vec<_>>();
15718 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15719 expected.sort();
15720 assert_eq!(ranked, expected, "rank order is value order");
15721 fs::remove_file(path).expect("remove scratch file");
15722 }
15723
15724 #[test]
15737 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15738 let path = path("windowed-directory");
15739 let fields = vec![
15740 Field::required("id", LogicalType::BigInt),
15741 Field::required("word", LogicalType::Varchar),
15742 Field::new("score", LogicalType::Double),
15743 ];
15744 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15745 for part in 0..70_i64 {
15746 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15747 let words = (0..100)
15748 .map(|row| Value::Varchar(format!("word {}", row % 13)))
15749 .collect::<Vec<_>>();
15750 let scores = (0..100)
15751 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15752 .collect::<Vec<_>>();
15753 let chunk = Chunk::new(vec![
15754 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15755 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15756 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15757 ])
15758 .expect("three columns");
15759 writer.append(&chunk).expect("a part");
15760 }
15761 writer.finish().expect("commit");
15762
15763 let catalog = Catalog::open(&path).expect("reopen");
15764 let entry = catalog.entries.first().expect("one table").directory;
15765 let (offset, length) = (entry.offset, entry.length as usize);
15766 let mut bytes = vec![0; length];
15767 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
15768 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
15769 let whole = decode_directory(&bytes, catalog.size).expect("whole");
15770 assert!(whole.stripes.len() > 1, "the table should span stripes");
15771 for size in [1, 7, 33, 4_096] {
15772 let mut cursor = Cursor::over(&catalog.file, offset, length);
15773 cursor.window.as_mut().expect("a window").size = size;
15774 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
15775 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
15776 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
15777 let mut stored = 0;
15778 for (column, (left, held)) in
15779 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
15780 {
15781 match (left, held) {
15782 (None, None) => {}
15783 (
15784 Some(super::Frequencies::Stored { span, values }),
15785 Some(super::Frequencies::Held(summary)),
15786 ) => {
15787 let mut one = vec![0; span.length as usize];
15788 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
15789 let read = decode_summary(
15790 &mut Cursor::new(&one),
15791 &whole.fields[column],
15792 whole.rows,
15793 *values,
15794 )
15795 .expect("a valid synopsis")
15796 .expect("one is there");
15797 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
15798 stored += 1;
15799 }
15800 other => panic!("column {column} came back as {other:?}"),
15801 }
15802 }
15803 assert!(stored >= 2, "only {stored} synopses were left in the file");
15804 }
15805 let reader = catalog.table("items").expect("the table");
15806 assert!(reader.frequency_summaries[1].get().is_none());
15807 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
15808 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
15809 let clone = reader.clone();
15810 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
15811 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
15812 fs::remove_file(path).expect("remove scratch file");
15813 }
15814
15815 #[test]
15816 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
15817 let path = path("file-checksum");
15818 let bytes = (0..200_000_u32)
15819 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
15820 .collect::<Vec<_>>();
15821 fs::write(&path, &bytes).expect("scratch file");
15822 let file = File::open(&path).expect("open");
15823 for (offset, length) in [
15824 (0, 0),
15825 (3, 1),
15826 (5, 31),
15827 (0, 32),
15828 (9, 33),
15829 (1, 65_536),
15830 (7, 65_567),
15831 (0, 200_000),
15832 (11, 131_101),
15833 ] {
15834 let whole = checksum(&bytes[offset..offset + length]);
15835 assert_eq!(
15836 file_checksum(&file, offset as u64, length).expect("read"),
15837 whole,
15838 "{offset} {length}"
15839 );
15840 }
15841 fs::remove_file(path).expect("remove scratch file");
15842 }
15843
15844 #[test]
15845 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
15846 let path = path("synopsis-keeps-no-block");
15847 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
15848 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
15849 for _ in 0..3 {
15850 values.extend((0..3_000).step_by(5).map(spelled));
15851 }
15852 let mut writer =
15853 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15854 .expect("new file");
15855 for part in values.chunks(1_024) {
15856 writer
15857 .append(
15858 &Chunk::new(vec![
15859 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15860 ])
15861 .expect("one column"),
15862 )
15863 .expect("a part");
15864 }
15865 writer.finish().expect("commit");
15866
15867 let reader = Reader::open(&path).expect("reopen from disk");
15868 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15869 let resting = dictionary.footprint();
15870 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15871 assert_eq!(prefix.entries.len(), 512);
15872 for (value, count) in &prefix.entries {
15873 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
15874 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
15875 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
15876 }
15877 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
15878 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15879 assert_eq!(again.entries, prefix.entries);
15880 fs::remove_file(path).expect("remove scratch file");
15881 }
15882
15883 #[test]
15890 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
15891 let path = path("character-lengths");
15892 let spellings = (0..2_500)
15893 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
15894 .collect::<Vec<_>>();
15895 let mut writer =
15896 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15897 .expect("new file");
15898 for part in spellings.chunks(1_024) {
15899 writer
15900 .append(
15901 &Chunk::new(vec![
15902 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15903 ])
15904 .expect("one column"),
15905 )
15906 .expect("a part");
15907 }
15908 writer.finish().expect("commit");
15909
15910 let reader = Reader::open(&path).expect("reopen from disk");
15911 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15912 let resting = dictionary.footprint();
15913 let mut lens = Vec::new();
15914 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
15915 let counted = dictionary.footprint() - resting;
15916 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15917 assert!(
15918 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15919 "counting kept {counted} bytes, more than a count a value"
15920 );
15921 let expected = (0..dictionary.len())
15922 .map(|code| {
15923 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
15924 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
15925 .expect("small")
15926 })
15927 .collect::<Vec<_>>();
15928 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
15929 let mut again = Vec::new();
15930 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
15931 assert_eq!(again, lens, "the kept counts answer the second time");
15932 fs::remove_file(path).expect("remove scratch file");
15933 }
15934
15935 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
15937 let path = path(label);
15938 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
15939 let mut writer =
15940 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15941 .expect("new file");
15942 for part in values.chunks(1_024) {
15943 writer
15944 .append(
15945 &Chunk::new(vec![
15946 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15947 ])
15948 .expect("one column"),
15949 )
15950 .expect("a part");
15951 }
15952 writer.finish().expect("commit");
15953 let reader = Reader::open(&path).expect("reopen from disk");
15954 (path, reader)
15955 }
15956
15957 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
15963 let codes = (0..len)
15964 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
15965 .collect::<Vec<_>>();
15966 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
15967 (codes, valid)
15968 }
15969
15970 #[test]
15977 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
15978 let spellings = (0..2_500)
15979 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
15980 .collect::<Vec<_>>();
15981 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
15982 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15983 let (codes, valid) = scattered_rows(spellings.len());
15984 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
15985 .expect("every code is inside")
15986 .with_validity(Validity::from_run(&valid));
15987
15988 let resting = dictionary.footprint();
15989 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
15990 .expect("length reads");
15991 let counted = dictionary.footprint() - resting;
15992 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15993 assert!(
15994 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15995 "length over a vector with nulls kept {counted} bytes, more than a count a value"
15996 );
15997 let expected = (0..rows.len())
15998 .map(|row| match valid[row] {
15999 true => Value::BigInt(
16000 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16001 ),
16002 false => Value::Null,
16003 })
16004 .collect::<Vec<_>>();
16005 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16006 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16007 fs::remove_file(path).expect("remove scratch file");
16008 }
16009
16010 #[test]
16020 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16021 let spellings = (0..2_500)
16022 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16023 .collect::<Vec<_>>();
16024 let (path, reader) = stored_spellings("string-kernels", &spellings);
16025 let page = reader.table.dictionaries[0].expect("a string column has one");
16026 let starved =
16027 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16028 .expect("a dictionary opens whatever it may keep");
16029 let starved = Arc::new(starved);
16030 let (codes, valid) = scattered_rows(spellings.len());
16031 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16032 .expect("every code is inside")
16033 .with_validity(Validity::from_run(&valid));
16034 let expected = |each: &dyn Fn(&str) -> String| {
16035 (0..rows.len())
16036 .map(|row| match valid[row] {
16037 true => Value::Varchar(each(&spellings[codes[row] as usize])),
16038 false => Value::Null,
16039 })
16040 .collect::<Vec<_>>()
16041 };
16042 let answers =
16043 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16044
16045 let resting = starved.footprint();
16048 let ends = spellings.len() * size_of::<u32>();
16049 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16050 .expect("lower reads");
16051 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16052 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16053
16054 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16055 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16056 let cut =
16057 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16058 .expect("substring reads");
16059 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16060 assert_eq!(answers(&cut), expected(&cut_of), "substring");
16061 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16062
16063 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16066 .expect("upper reads");
16067 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16068 let payload = spellings.iter().map(String::len).sum::<usize>();
16069 assert!(
16070 starved.footprint() >= resting + payload,
16071 "a visit that has dropped a column's worth of blocks keeps what it reads"
16072 );
16073 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16074 .expect("upper reads kept blocks");
16075 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16076 fs::remove_file(path).expect("remove scratch file");
16077 }
16078
16079 #[test]
16089 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16090 let path = path("dictionary-sweep");
16091 let spellings = (0..2_500)
16094 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16095 .collect::<Vec<_>>();
16096 let mut writer =
16097 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16098 .expect("new file");
16099 for part in spellings.chunks(1_024) {
16102 writer
16103 .append(
16104 &Chunk::new(vec![
16105 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16106 ])
16107 .expect("one column"),
16108 )
16109 .expect("stripe written");
16110 }
16111 writer.finish().expect("commit");
16112
16113 let reader = Reader::open(&path).expect("valid directory");
16114 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16115 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16116 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
16117 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
16118 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
16119 }
16120
16121 let resting = dictionary.footprint();
16122 let sweep = || {
16123 let mut swept: Vec<Vec<u8>> = Vec::new();
16124 let mut at = 0;
16125 let mut calls = 0;
16126 while at < dictionary.len() {
16127 let stopped = dictionary
16128 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16129 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16130 swept.push(text.to_vec());
16131 Ok(())
16132 })
16133 .expect("a sweep reads");
16134 assert!(stopped > at, "a sweep moves");
16135 at = stopped;
16136 calls += 1;
16137 }
16138 assert_eq!(calls, 3, "a sweep hands over one block at a time");
16139 swept
16140 };
16141 let swept = sweep();
16142 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16143 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16144 let after = dictionary.footprint();
16145 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16146
16147 let read = (0..dictionary.len())
16148 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16149 .collect::<Vec<_>>();
16150 assert_eq!(swept, read, "a sweep answers what a point read answers");
16151 let grown = dictionary.footprint() - after;
16155 assert!(
16156 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16157 "a point read of a kept block decodes nothing, and {grown} bytes grew"
16158 );
16159 fs::remove_file(path).expect("remove scratch file");
16160 }
16161
16162 #[test]
16163 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16164 let path = path("narrow-substring-signature");
16165 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16166 let mut grams = Vec::new();
16167 for text in blocks {
16168 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16169 for gram in text.windows(4) {
16170 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16171 bits[bit / 8] |= 1 << (bit % 8);
16172 }
16173 }
16174 grams.extend(bits);
16175 }
16176 fs::write(&path, &grams).expect("scratch file");
16177 let file = File::open(&path).expect("open scratch file");
16178 let signatures = NativeGrams {
16179 start: 0,
16180 length: grams.len(),
16181 width: NARROW_GRAM_BYTES,
16182 hash: checksum(&grams),
16183 verdicts: Mutex::new(Vec::new()),
16184 };
16185 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16186 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16187 assert!(signatures.footprint() > 0, "a verdict is remembered");
16188 let again = signatures.verdicts(&file, b"google").expect("remembered");
16189 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16190
16191 let damaged = NativeGrams {
16192 hash: signatures.hash ^ 1,
16193 verdicts: Mutex::new(Vec::new()),
16194 ..signatures
16195 };
16196 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16197 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16198 fs::remove_file(path).expect("remove scratch file");
16199 }
16200
16201 #[test]
16202 fn a_damaged_substring_signature_is_checked_only_when_used() {
16203 let path = path("damaged-substring-signature");
16204 let mut writer =
16205 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16206 .expect("new file");
16207 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16208 writer
16209 .append(
16210 &Chunk::new(vec![
16211 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16212 ])
16213 .expect("one column"),
16214 )
16215 .expect("stripe written");
16216 writer.finish().expect("commit");
16217
16218 let reader = Reader::open(&path).expect("valid directory");
16219 let page = reader.table.dictionaries[0].expect("string dictionary page");
16220 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16221 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16222 .expect("last signature byte");
16223 file.write_all(&[255]).expect("damage signature");
16224 let reader = Reader::open(&path).expect("the directory is still valid");
16225 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16226 let error = dictionary
16227 .text_block_might_contain(0, b"goog")
16228 .expect_err("a used signature checks its own checksum");
16229 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16230 fs::remove_file(path).expect("remove scratch file");
16231 }
16232
16233 #[test]
16244 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16245 let path = path("dictionary-sweep-short-run");
16246 let spellings = (0..2_800)
16247 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16248 .collect::<Vec<_>>();
16249 let mut writer =
16250 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16251 .expect("new file");
16252 for part in spellings.chunks(1_024) {
16253 writer
16254 .append(
16255 &Chunk::new(vec![
16256 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16257 ])
16258 .expect("one column"),
16259 )
16260 .expect("stripe written");
16261 }
16262 writer.finish().expect("commit");
16263
16264 let reader = Reader::open(&path).expect("valid directory");
16265 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16266 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16267 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16268 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16269 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16270
16271 let mut swept: Vec<Vec<u8>> = Vec::new();
16272 let mut at = 0;
16273 while at < dictionary.len() {
16274 let stopped = dictionary
16275 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16276 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16277 swept.push(text.to_vec());
16278 Ok(())
16279 })
16280 .expect("a sweep reads");
16281 assert!(stopped > at, "a sweep moves");
16282 at = stopped;
16283 }
16284 let read = (0..dictionary.len())
16285 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16286 .collect::<Vec<_>>();
16287 assert_eq!(swept, read, "a sweep answers what a point read answers");
16288 fs::remove_file(path).expect("remove scratch file");
16289 }
16290
16291 #[test]
16300 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16301 let path = path("dictionary-unpacked-ends");
16302 let spellings = (0..2_800)
16303 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16304 .collect::<Vec<_>>();
16305 let mut writer =
16306 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16307 .expect("new file");
16308 for part in spellings.chunks(1_024) {
16309 writer
16310 .append(
16311 &Chunk::new(vec![
16312 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16313 ])
16314 .expect("one column"),
16315 )
16316 .expect("stripe written");
16317 }
16318 writer.finish().expect("commit");
16319
16320 let reader = Reader::open(&path).expect("valid directory");
16321 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16322 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16323 let wanted = (0..spellings.len())
16324 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16325 .collect::<Vec<_>>();
16326
16327 let pass = |what: &str| {
16328 for (index, value) in wanted.iter().enumerate() {
16329 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16330 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16331 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16332 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16333 }
16334 };
16335 pass("the first pass");
16336 pass("the second pass");
16337
16338 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16342 let mut whole = vec![0i64; wanted.len()];
16343 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16344 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16345 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16346 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16347 let mut through = vec![0i64; codes.len()];
16348 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16349 for (row, &code) in codes.iter().enumerate() {
16350 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16351 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16352 assert_eq!(through[row], one as i64, "row {row} a row at a time");
16353 }
16354
16355 let fresh = Reader::open(&path).expect("valid directory");
16358 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16359 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16360 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16361 let mut short = vec![0i64; few.len()];
16362 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16363 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16364 assert_eq!(short, expected, "the packed ends answer what the table answers");
16365 fs::remove_file(path).expect("remove scratch file");
16366 }
16367
16368 #[test]
16383 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16384 let spellings = (0..3_000)
16385 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16386 .collect::<Vec<_>>();
16387 let mut read = Vec::new();
16388 for layout in ["outside", "inside", "behind"] {
16389 let mut dictionary = GlobalDictionary::new();
16390 for text in &spellings {
16391 dictionary.code(text).expect("a code for every spelling");
16392 }
16393 dictionary.finish_blocks().expect("the last block encodes");
16394 let order = dictionary.ranked(None).expect("a sorted order");
16395 let laid = |from: u64| {
16397 let mut at = from;
16398 dictionary
16399 .blocks
16400 .iter()
16401 .map(|block| {
16402 let place =
16403 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16404 at += block.len() as u64;
16405 place
16406 })
16407 .collect::<Vec<_>>()
16408 };
16409 let payload = dictionary.blocks.concat();
16410 let scattered = layout != "behind";
16411 let (bytes, encoded, offset, length) = if layout == "outside" {
16412 let mut bytes = vec![0; HEADER as usize];
16413 bytes.extend_from_slice(&payload);
16414 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16415 .expect("an encoding");
16416 let offset = bytes.len() as u64;
16417 bytes.extend_from_slice(&encoded.index);
16418 bytes.extend_from_slice(&encoded.ranks);
16419 bytes.extend_from_slice(&encoded.grams);
16420 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16421 (bytes, encoded, offset, length)
16422 } else {
16423 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16426 .expect("an encoding");
16427 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16428 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16429 .expect("an encoding");
16430 let mut bytes = encoded.index.clone();
16431 bytes.extend_from_slice(&encoded.ranks);
16432 bytes.extend_from_slice(&encoded.grams);
16433 bytes.extend_from_slice(&payload);
16434 let length = bytes.len();
16435 (bytes, encoded, 0, length)
16436 };
16437 let path = path(&format!("blocks-{layout}"));
16438 fs::write(&path, &bytes).expect("the dictionary is written on its own");
16439 let file = Arc::new(File::open(&path).expect("it opens again"));
16440 let page = Page {
16441 offset,
16442 length: u32::try_from(length).expect("a test dictionary is small"),
16443 hash: checksum(&encoded.index),
16444 };
16445 let opened =
16446 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16447 .expect("a dictionary laid out either way opens");
16448 let mut swept: Vec<Vec<u8>> = Vec::new();
16449 let mut at = 0;
16450 while at < opened.len() {
16451 at = opened
16452 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16453 swept.push(text.to_vec());
16454 Ok(())
16455 })
16456 .expect("a sweep reads");
16457 }
16458 fs::remove_file(&path).expect("clean up");
16459 read.push(swept);
16460 }
16461 let wanted =
16462 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16463 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16464 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16465 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16466 }
16467
16468 #[test]
16476 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16477 let path = path("dictionary-budget");
16478 let spellings = (0..2_500)
16479 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16480 .collect::<Vec<_>>();
16481 let mut writer =
16482 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16483 .expect("new file");
16484 for part in spellings.chunks(1_024) {
16485 writer
16486 .append(
16487 &Chunk::new(vec![
16488 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16489 ])
16490 .expect("one column"),
16491 )
16492 .expect("stripe written");
16493 }
16494 writer.finish().expect("commit");
16495
16496 let reader = Reader::open(&path).expect("valid directory");
16497 let page = reader.table.dictionaries[0].expect("a string column has one");
16498 let file = Arc::clone(&reader.file);
16499 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16500 .expect("a dictionary opens whatever it may keep");
16501
16502 let resting = starved.footprint();
16503 let mut swept: Vec<Vec<u8>> = Vec::new();
16504 let mut at = 0;
16505 while at < starved.len() {
16506 at = starved
16507 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16508 swept.push(text.to_vec());
16509 Ok(())
16510 })
16511 .expect("a sweep reads");
16512 }
16513 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16514 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16515
16516 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16517 let read = (0..generous.len())
16518 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16519 .collect::<Vec<_>>();
16520 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16521 fs::remove_file(path).expect("remove scratch file");
16522 }
16523
16524 #[test]
16525 fn damaged_membership_cannot_skip_a_string_page() {
16526 let path = path("damaged-membership");
16527 let mut writer = Writer::create(
16528 &path,
16529 "items",
16530 vec![
16531 Field::required("id", LogicalType::Integer),
16532 Field::new("text", LogicalType::Varchar),
16533 ],
16534 )
16535 .expect("new file");
16536 writer.append(&sample()).expect("stripe written");
16537 writer.finish().expect("commit");
16538
16539 let reader = Reader::open(&path).expect("valid directory");
16540 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16541 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16542 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16543 file.write_all(&[255]).expect("damage membership");
16544 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16545 assert!(error.message().contains("membership page checksum differs"), "{error}");
16546 fs::remove_file(path).expect("remove scratch file");
16547 }
16548
16549 #[test]
16550 fn membership_delta_stream_is_sorted_exact_and_bounded() {
16551 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16552 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16553 let encoded = encode_membership(&unique);
16554 assert_eq!(
16555 decode_membership(&encoded).expect("valid membership"),
16556 [4, 9, 72, 900, u32::MAX]
16557 );
16558 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16561 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16562 assert_eq!(
16563 decode_membership(&encode_membership(&merged)).expect("valid membership"),
16564 unique
16565 );
16566 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16567 assert!(
16568 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16569 "a value past u32 is invalid"
16570 );
16571 }
16572
16573 #[test]
16574 fn a_global_dictionary_may_be_larger_than_one_column_page() {
16575 let dictionary = Page {
16576 offset: HEADER,
16577 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16578 hash: 0,
16579 };
16580 let table = Table {
16581 name: "items".to_owned(),
16582 fields: vec![Field::new("text", LogicalType::Varchar)],
16583 stripes: Vec::new(),
16584 rows: 0,
16585 dictionaries: vec![Some(dictionary)],
16586 dictionary_payloads: Vec::new(),
16587 demoted: Vec::new(),
16588 distincts: vec![None],
16589 frequencies: vec![None],
16590 pair_frequencies: Vec::new(),
16591 frequency_texts: Vec::new(),
16592 host_groups: None,
16593 clustering: None,
16594 generation: 1,
16595 sections: Vec::new(),
16596 };
16597 let directory = encode_directory(&table).expect("directory");
16598 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16599
16600 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16601 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16602 }
16603
16604 #[test]
16605 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16606 let path = path("constant-codes");
16607 let mut writer =
16608 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16609 .expect("new file");
16610 let empty = vec![Value::Varchar(String::new()); 1024];
16611 for _ in 0..4 {
16612 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16613 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16614 }
16615 writer.finish().expect("commit");
16616
16617 let reader = Reader::open(&path).expect("valid directory");
16618 let pages = reader.layout().columns.first().expect("one column").pages;
16619 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16623 let read = reader.read(3, &[0]).expect("the last part back");
16624 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16625 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16626 fs::remove_file(path).expect("remove scratch file");
16627 }
16628
16629 #[test]
16630 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16631 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
16634 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
16635 assert!(format!("{error}").contains("not of its type"), "{error}");
16636 let low = integer::encode(&[i64::MIN]).expect("a chunk");
16637 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
16638 let zero = integer::encode(&[0]).expect("a chunk");
16639 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
16640 }
16641
16642 #[test]
16643 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16644 let mut state: u32 = 0x9e37_79b9;
16648 let spread: Vec<u32> = (0..1024)
16649 .map(|_| {
16650 state ^= state << 13;
16651 state ^= state >> 17;
16652 state ^= state << 5;
16653 state
16654 })
16655 .collect();
16656 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16657 let near: Vec<u32> = (0..1024).collect();
16658 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16659 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16660 }
16661
16662 #[test]
16668 fn two_writes_of_the_same_rows_give_the_same_bytes() {
16669 fn written(path: &PathBuf) {
16670 let fields = (0..40)
16671 .map(|column| {
16672 let ty =
16673 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16674 Field::new(format!("c{column}"), ty)
16675 })
16676 .collect::<Vec<_>>();
16677 let mut writer = Writer::create(path, "wide", fields).expect("new file");
16678 for part in 0..70_u64 {
16679 let columns = (0..40)
16680 .map(|column| {
16681 let values = (0..64_u64)
16682 .map(|row| {
16683 let seed = part.wrapping_mul(31).wrapping_add(row);
16684 if column % 4 == 0 {
16685 Value::Varchar(format!("v{}", seed % 17))
16686 } else {
16687 Value::BigInt(i64::try_from(seed % 97).expect("small"))
16688 }
16689 })
16690 .collect::<Vec<_>>();
16691 let ty = if column % 4 == 0 {
16692 LogicalType::Varchar
16693 } else {
16694 LogicalType::BigInt
16695 };
16696 Vector::from_values(ty, &values).expect("a column")
16697 })
16698 .collect::<Vec<_>>();
16699 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16700 }
16701 writer.finish().expect("commit");
16702 }
16703
16704 let first = path("repeatable-one");
16705 let second = path("repeatable-two");
16706 written(&first);
16707 written(&second);
16708 let left = fs::read(&first).expect("the first file");
16709 let right = fs::read(&second).expect("the second file");
16710 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16711 assert!(left == right, "two writes of the same rows differ in their bytes");
16712
16713 let reader = Reader::open(&first).expect("valid directory");
16716 assert_eq!(reader.table().rows(), 70 * 64);
16717 let read = reader.read(0, &[0, 1]).expect("the first part back");
16718 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16719 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16720 fs::remove_file(first).expect("remove scratch file");
16721 fs::remove_file(second).expect("remove scratch file");
16722 }
16723
16724 fn three_tables(path: &PathBuf) {
16726 let writer = Writer::create(
16727 path,
16728 "region",
16729 vec![
16730 Field::new("r_key", LogicalType::Integer),
16731 Field::new("r_name", LogicalType::Varchar),
16732 ],
16733 )
16734 .expect("new file");
16735 let mut writer = writer;
16736 writer
16737 .append(
16738 &Chunk::new(vec![
16739 Vector::from_values(
16740 LogicalType::Integer,
16741 &[Value::Integer(0), Value::Integer(1)],
16742 )
16743 .expect("keys"),
16744 Vector::from_values(
16745 LogicalType::Varchar,
16746 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16747 )
16748 .expect("names"),
16749 ])
16750 .expect("two columns"),
16751 )
16752 .expect("a part");
16753 let mut writer = writer
16754 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16755 .expect("a second table");
16756 writer
16757 .append(
16758 &Chunk::new(vec![
16759 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16760 ])
16761 .expect("one column"),
16762 )
16763 .expect("a part");
16764 let mut writer =
16765 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16766 for part in 0..70_i64 {
16767 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
16768 writer
16769 .append(
16770 &Chunk::new(vec![
16771 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
16772 ])
16773 .expect("one column"),
16774 )
16775 .expect("a part");
16776 }
16777 writer.finish().expect("commit");
16778 }
16779
16780 #[test]
16781 fn three_tables_in_one_file_read_back_by_name() {
16782 let file = path("three-tables");
16783 three_tables(&file);
16784 let catalog = Catalog::open(&file).expect("a committed catalog");
16785 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
16786
16787 let region = catalog.table("region").expect("the first table");
16788 assert_eq!(region.table().rows(), 2);
16789 assert_eq!(
16790 region.read(0, &[1]).expect("names").value_at(1, 0),
16791 Value::Varchar("ASIA".to_owned())
16792 );
16793
16794 let wide = catalog.table("wide").expect("the third table");
16795 assert_eq!(wide.table().rows(), 70 * 64);
16796 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
16797
16798 let empty = catalog.table("empty").expect("the second table");
16801 assert_eq!(empty.table().rows(), 1);
16802 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
16803
16804 fs::remove_file(file).expect("remove scratch file");
16805 }
16806
16807 #[test]
16808 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
16809 let file = path("three-tables-missing");
16810 three_tables(&file);
16811 let catalog = Catalog::open(&file).expect("a committed catalog");
16812 let error = catalog.table("nation").expect_err("no such table");
16813 assert!(error.message().contains("nation"), "{}", error.message());
16814 fs::remove_file(file).expect("remove scratch file");
16815 }
16816
16817 #[test]
16818 fn a_file_of_three_tables_will_not_open_as_one() {
16819 let file = path("three-tables-unnamed");
16820 three_tables(&file);
16821 let error = Reader::open(&file).expect_err("more than one table");
16822 assert!(error.message().contains("more than one table"), "{}", error.message());
16823 fs::remove_file(file).expect("remove scratch file");
16824 }
16825
16826 #[test]
16828 fn decimals_of_every_storage_width_round_trip() {
16829 let file = path("decimals");
16830 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
16831 let fields = widths
16832 .iter()
16833 .enumerate()
16834 .map(|(index, (width, scale))| {
16835 Field::new(
16836 format!("d{index}"),
16837 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16838 )
16839 })
16840 .collect::<Vec<_>>();
16841 let mut writer = Writer::create(&file, "money", fields).expect("new file");
16842 let rows: [i128; 3] = [-1234, 0, 999];
16843 let columns = widths
16844 .iter()
16845 .map(|(width, scale)| {
16846 let values = rows
16847 .iter()
16848 .map(|unscaled| Value::Decimal {
16849 unscaled: *unscaled,
16850 width: *width,
16851 scale: *scale,
16852 })
16853 .collect::<Vec<_>>();
16854 Vector::from_values(
16855 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16856 &values,
16857 )
16858 .expect("a decimal column")
16859 })
16860 .collect::<Vec<_>>();
16861 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
16862 writer.finish().expect("commit");
16863
16864 let reader = Reader::open(&file).expect("a committed file");
16865 for (index, (width, scale)) in widths.iter().enumerate() {
16866 assert_eq!(
16867 reader.table().fields()[index].ty,
16868 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16869 "column {index} came back as another type"
16870 );
16871 let column = reader.read(0, &[index]).expect("the column");
16872 for (row, unscaled) in rows.iter().enumerate() {
16873 assert_eq!(
16874 column.value_at(row, 0),
16875 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
16876 "column {index} row {row}"
16877 );
16878 }
16879 }
16880 fs::remove_file(file).expect("remove scratch file");
16881 }
16882
16883 #[test]
16884 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
16885 let file = path("two-of-a-name");
16886 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
16887 .expect("new file");
16888 let error = writer
16889 .next("t", vec![Field::new("a", LogicalType::BigInt)])
16890 .expect_err("the same name twice");
16891 assert!(error.message().contains("same name"), "{}", error.message());
16892 fs::remove_file(file).expect("remove scratch file");
16893 }
16894
16895 #[test]
16896 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
16897 let file = path("integer-tally");
16898 let mut writer =
16899 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
16900 .expect("new file");
16901 let mut values = vec![Value::SmallInt(0); 1024];
16902 values[7] = Value::SmallInt(3);
16903 values[99] = Value::SmallInt(-2);
16904 values[1001] = Value::SmallInt(3);
16905 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
16906 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
16907 values[0] = Value::Null;
16908 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
16909 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
16910 writer.finish().expect("commit");
16911
16912 let reader = Reader::open(&file).expect("read file");
16913 assert_eq!(
16914 reader.integer_tally(0, 0).expect("valid part"),
16915 Some(vec![(-2, 1), (0, 1021), (3, 2)])
16916 );
16917 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
16918 let catalog = Catalog::open(&file).expect("catalog");
16919 assert_eq!(
16920 catalog.integer_tally("events", 0).expect("nullable column"),
16921 Some(vec![(-2, 2), (0, 2041), (3, 4)])
16922 );
16923 fs::remove_file(file).expect("remove scratch file");
16924 }
16925
16926 #[test]
16927 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
16928 let file = path("catalog-integer-tally");
16929 let mut writer = Writer::create(
16930 &file,
16931 "events",
16932 vec![
16933 Field::new("noise", LogicalType::SmallInt),
16934 Field::new("source", LogicalType::SmallInt),
16935 ],
16936 )
16937 .expect("new file");
16938 let noise = vec![Value::SmallInt(9); 1024];
16939 let mut source = vec![Value::SmallInt(0); 1024];
16940 source[7] = Value::SmallInt(3);
16941 source[99] = Value::SmallInt(-2);
16942 let chunk = Chunk::new(vec![
16943 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
16944 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
16945 ])
16946 .expect("two columns");
16947 writer.append(&chunk).expect("append");
16948 writer.finish().expect("commit");
16949
16950 let catalog = Catalog::open(&file).expect("catalog");
16951 assert_eq!(
16952 catalog.integer_tally("events", 1).expect("selected column"),
16953 Some(vec![(-2, 1), (0, 1022), (3, 1)])
16954 );
16955 assert_eq!(
16956 catalog.integer_tally("events", 0).expect("other column"),
16957 Some(vec![(9, 1024)])
16958 );
16959 fs::remove_file(file).expect("remove scratch file");
16960 }
16961
16962 #[test]
16963 fn opening_the_catalog_reads_no_table_directory() {
16964 let file = path("catalog-only");
16965 three_tables(&file);
16966 let catalog = Catalog::open(&file).expect("a committed catalog");
16967 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
16970 assert_eq!(catalog.names().len(), 3);
16971 fs::remove_file(file).expect("remove scratch file");
16972 }
16973
16974 #[test]
16985 fn the_checksum_answers_what_it_has_always_answered() {
16986 let bytes: Vec<u8> =
16987 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
16988 for (length, expected) in [
16989 (0, 0xef46_db37_51d8_e999),
16990 (1, 0xa96c_7f0c_e858_bbb7),
16991 (3, 0x56e6_9576_32a4_87f9),
16992 (4, 0xc60d_15b1_e3ff_8f04),
16993 (5, 0x8088_1585_8624_dd4e),
16994 (7, 0xafbe_fc3d_6c6f_9a8e),
16995 (8, 0x3da5_c7aa_2696_83e0),
16996 (9, 0x465e_c429_b13c_3892),
16997 (15, 0xdee8_9d8a_065a_6233),
16998 (16, 0x1330_489a_7767_9c80),
16999 (31, 0x3391_303d_485e_846e),
17000 (32, 0x40b7_aff7_5d45_bbc8),
17001 (33, 0x4997_cae4_951c_17a5),
17002 (39, 0x5807_28fd_5c14_5739),
17003 (40, 0xf95c_f6f5_c08a_3d3b),
17004 (63, 0x2944_b4da_fc69_b206),
17005 (64, 0xbb76_f6ef_19bd_5a1b),
17006 (65, 0x814e_0c65_4a9f_d640),
17007 (127, 0x00de_aab1_31cf_f89b),
17008 (1000, 0x9e33_00c1_cde3_c58d),
17009 ] {
17010 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17011 }
17012 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17013 }
17014 #[test]
17021 fn a_declared_order_comes_back_out_of_the_file() {
17022 let path = path("clustered");
17023 let shipped = vec![
17024 Field::new("key", LogicalType::BigInt),
17025 Field::new("line", LogicalType::Integer),
17026 Field::new("shipdate", LogicalType::Date),
17027 ];
17028 let plain = vec![Field::new("a", LogicalType::Integer)];
17029 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17030
17031 let mut writer = Writer::create(&path, "lineitem", shipped)
17032 .expect("new file")
17033 .declare(stage_zero.clone())
17034 .expect("the columns are the table's");
17035 let column = |ty: LogicalType, values: &[Value]| {
17036 Vector::from_values(ty, values).expect("the values match the type")
17037 };
17038 writer
17039 .append(
17040 &Chunk::new(vec![
17041 column(
17042 LogicalType::BigInt,
17043 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17044 ),
17045 column(
17046 LogicalType::Integer,
17047 &[
17048 Value::Integer(1),
17049 Value::Integer(1),
17050 Value::Integer(1),
17051 Value::Integer(1),
17052 ],
17053 ),
17054 column(
17055 LogicalType::Date,
17056 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17057 ),
17058 ])
17059 .expect("three columns"),
17060 )
17061 .expect("four rows");
17062 let mut writer = writer.next("nation", plain).expect("a second table");
17063 writer
17064 .append(
17065 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17066 .expect("one column"),
17067 )
17068 .expect("one row");
17069 writer.finish().expect("commit");
17070
17071 let catalog = Catalog::open(&path).expect("reopen");
17072 let lineitem = catalog.table("lineitem").expect("the clustered table");
17073 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17074 let nation = catalog.table("nation").expect("the plain table");
17075 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17076
17077 assert_eq!(lineitem.table().rows(), 4);
17080 assert_eq!(nation.table().rows(), 1);
17081 fs::remove_file(&path).ok();
17082 }
17083
17084 #[test]
17086 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17087 let path = path("clustered-bad");
17088 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17089 .expect("new file");
17090 let four =
17091 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17092 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17093 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17094 fs::remove_file(&path).ok();
17095 }
17096
17097 #[test]
17103 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17104 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17105 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17106 .collect::<Vec<_>>();
17107 let filled = || {
17108 let mut dictionary = GlobalDictionary::new();
17109 for value in &values {
17110 dictionary.code(value).expect("a code for every value");
17111 }
17112 dictionary.settle().expect("a shape");
17113 dictionary
17114 };
17115 let mut in_place = filled();
17116 in_place.finish_blocks().expect("every block encodes");
17117
17118 let mut handed = filled();
17119 let out = handed.hand_out(3);
17120 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17121 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17122 for job in out.iter().rev() {
17123 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17124 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17125 }
17126 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17127 handed.finish_blocks().expect("the last block encodes");
17128
17129 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17130 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17131 }
17132
17133 #[test]
17135 fn a_block_given_back_twice_is_refused() {
17136 let mut dictionary = GlobalDictionary::new();
17137 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17138 dictionary.code(&format!("value {at}")).expect("a code");
17139 }
17140 dictionary.settle().expect("a shape");
17141 let out = dictionary.hand_out(0);
17142 let last = out.last().expect("blocks went out");
17143 let at = last.place().1;
17144 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17145 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17146 }
17147
17148 #[test]
17154 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17155 let mut values = vec![String::new(), "http://".to_owned()];
17156 for host in 0..7 {
17157 for path in 0..30 {
17158 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17159 values.push(format!("http://example{host}.test/page/{path:04}"));
17160 }
17161 }
17162 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17163
17164 let mut dictionary = GlobalDictionary::new();
17165 for value in &values {
17166 dictionary.code(value).expect("a code for every value");
17167 }
17168 dictionary.finish_blocks().expect("the last block encodes");
17169 let ranked = dictionary.ranked(None).expect("a sorted order");
17170 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17171
17172 let spellings = dictionary_values(&dictionary);
17173 let seen = ranked
17174 .iter()
17175 .map(|&(_, code)| {
17176 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17177 })
17178 .collect::<Vec<_>>();
17179 let mut wanted = values.clone();
17180 wanted.sort_unstable();
17181 assert_eq!(seen, wanted, "the order is the order the bytes give");
17182
17183 for &(carried, code) in &ranked {
17184 let value = &spellings[code as usize];
17185 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17186 }
17187 }
17188
17189 #[test]
17194 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17195 let entry =
17196 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17197 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17198 .map(|code| entry(code, u64::from(code % 7) + 1))
17199 .collect::<Vec<_>>();
17200 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17201
17202 let mut sorted = all.clone();
17203 sorted.sort_unstable_by(|left, right| {
17204 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17205 });
17206 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17207 sorted.truncate(FREQUENCY_ENTRIES);
17208
17209 let mut picked = all.clone();
17210 let omitted = keep_most_frequent(&mut picked);
17211 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17212 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17213 assert!(
17214 picked
17215 .iter()
17216 .zip(&sorted)
17217 .all(|(one, two)| one.value == two.value && one.count == two.count),
17218 "the same entries in the same order"
17219 );
17220
17221 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17222 let omitted = keep_most_frequent(&mut short);
17223 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17224 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17225 }
17226
17227 #[test]
17229 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17230 let empty = GlobalDictionary::new();
17231 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17232
17233 let mut dictionary = GlobalDictionary::new();
17234 for value in ["pear", "apple", "", "apples", "app"] {
17235 dictionary.code(value).expect("a code for every value");
17236 }
17237 dictionary.finish_blocks().expect("the one block encodes");
17238 let spellings = dictionary_values(&dictionary);
17239 let seen = dictionary
17240 .ranked(None)
17241 .expect("a sorted order")
17242 .iter()
17243 .map(|&(_, code)| spellings[code as usize].clone())
17244 .collect::<Vec<_>>();
17245 let wanted: Vec<Vec<u8>> =
17246 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17247 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17248 }
17249
17250 #[test]
17253 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17254 let profile = LoadProfile::begin("demoted");
17255 let mut dictionary = GlobalDictionary::new();
17256 for value in 0..50_000 {
17257 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17258 }
17259 let (_, grown) = dictionary.recharge(Some(&profile));
17260 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17261
17262 dictionary.demote();
17263 let (before, after) = dictionary.recharge(Some(&profile));
17264 assert_eq!(before, grown);
17265 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17268 assert_eq!(profile.held(), after, "the profile was told about the drop");
17269 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17270
17271 dictionary.demote();
17272 assert_eq!(
17273 dictionary.recharge(Some(&profile)),
17274 (after, after),
17275 "demoting twice is a no-op"
17276 );
17277 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17278 }
17279}