1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::{Ordering, Reverse};
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::sequence::Sequence;
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_io::{Filesystem, OpenMode, RealFilesystem};
50use rudb_metrics::{LoadProfile, Stage};
51use rudb_storage::sieve::Sieve;
52use rudb_storage::{Probe, Range, Zone};
53use rudb_vector::string::StringColumn;
54use rudb_vector::validity::Validity;
55use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
56
57mod distinct;
58pub mod graph;
59pub mod host;
60mod prepare;
61mod projection;
62mod run_projection;
63use prepare::Lent;
64pub mod section;
65pub mod stats;
66mod zones;
67
68pub use prepare::{Building, DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
69pub use projection::build_sorted_projection;
70pub use run_projection::build_run_projection;
71pub use section::Section;
72pub use zones::{Common, Stripes, ascending, distincts, widths};
73
74const MAGIC: &[u8; 8] = b"RUDBNV10";
75const DIRECTORY: &[u8; 8] = b"RUDBDI10";
76const CATALOG: &[u8; 8] = b"RUDBCA10";
77const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
78const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
79const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
80const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
81const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
82const MAX_CATALOG_FREQUENCIES: usize = 64;
83const FORMAT: u32 = 29;
84
85const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
116
117const HEADER: u64 = 80;
118const SLOT_BYTES: usize = 28;
119const MAX_PAGE: usize = 256 * 1024 * 1024;
120const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
121const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
122const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
123const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
131const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
133const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
139const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
154const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
174const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
182const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
190
191const MAX_SECTIONS: usize = 4096;
198const FREQUENCY_CANDIDATES: usize = 32_768;
199const FREQUENCY_ENTRIES: usize = 512;
200const FREQUENCY_BUILD_RANK: usize = 10;
201const FREQUENCY_ORDINALS: usize = 131_072;
202const MAX_PAIR_FREQUENCIES: usize = 1024;
203const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
208const MAX_FREQUENCY_WORKERS: usize = 32;
215
216fn close_workers() -> usize {
218 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
219}
220
221const CLOSE_BYTES: usize = 1 << 30;
232
233const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
236
237const MAX_ENCODE_WORKERS: usize = 32;
244
245const WRITEBACK_STRETCH: u64 = 32 << 20;
253
254const SIEVE_BUDGET: usize = 8 * 1024;
262
263const PART_BOUND_BYTES: usize = 24;
272
273fn io(error: std::io::Error) -> Error {
274 Error::io(error.to_string())
275}
276
277fn invalid(message: &str) -> Error {
278 Error::invalid_input(format!("invalid rudb native file: {message}"))
279}
280
281fn sum(counts: impl Iterator<Item = u64>) -> u64 {
283 counts.fold(0, u64::saturating_add)
284}
285
286fn span_bytes(spans: &[Span], at: usize) -> u64 {
288 spans.get(at).map_or(0, |span| u64::from(span.length))
289}
290
291fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
293 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
294}
295
296fn dictionary_bytes(table: &Table, at: usize) -> u64 {
298 page_bytes(&table.dictionaries, at)
299 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
300}
301
302fn checksum(bytes: &[u8]) -> u64 {
312 seeded_checksum(bytes, 0)
313}
314
315#[must_use]
322pub fn content_name(bytes: &[u8]) -> u128 {
323 let seed = u64::from(FORMAT);
324 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
325}
326
327#[derive(Debug, Clone)]
333pub struct ContentNamer {
334 seeds: [u64; 2],
335 lanes: [[u64; 4]; 2],
336 held: [u8; 32],
337 filled: usize,
338 length: u64,
339}
340
341impl Default for ContentNamer {
342 fn default() -> Self {
343 let seed = u64::from(FORMAT);
344 let seeds = [seed, !seed];
345 let lanes = seeds.map(|seed| {
346 [
347 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
348 seed.wrapping_add(XXH_P2),
349 seed,
350 seed.wrapping_sub(XXH_P1),
351 ]
352 });
353 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
354 }
355}
356
357impl ContentNamer {
358 pub fn update(&mut self, mut bytes: &[u8]) {
360 self.length += bytes.len() as u64;
361 if self.filled > 0 {
362 let take = (32 - self.filled).min(bytes.len());
363 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
364 self.filled += take;
365 bytes = &bytes[take..];
366 if self.filled < 32 {
367 return;
368 }
369 let block = self.held;
370 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
371 self.filled = 0;
372 }
373 let mut blocks = bytes.chunks_exact(32);
374 for block in blocks.by_ref() {
375 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
376 }
377 let rest = blocks.remainder();
378 self.held[..rest.len()].copy_from_slice(rest);
379 self.filled = rest.len();
380 }
381
382 #[must_use]
384 pub fn finish(&self) -> u128 {
385 let rest = &self.held[..self.filled];
386 let [first, second] = [0, 1].map(|at| {
387 if self.length < 32 {
388 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
389 } else {
390 finish_checksum(self.lanes[at], rest, self.length)
391 }
392 });
393 u128::from(first) << 64 | u128::from(second)
394 }
395}
396
397fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
406 let mut blocks = bytes.chunks_exact(32);
409 let rest = blocks.remainder();
410 if bytes.len() < 32 {
411 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
412 }
413 let mut lanes = [
414 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
415 seed.wrapping_add(XXH_P2),
416 seed,
417 seed.wrapping_sub(XXH_P1),
418 ];
419 for block in blocks.by_ref() {
420 checksum_block(&mut lanes, block);
421 }
422 finish_checksum(lanes, rest, bytes.len() as u64)
423}
424
425const XXH_P1: u64 = 11_400_714_785_074_694_791;
426const XXH_P2: u64 = 14_029_467_366_897_019_727;
427const XXH_P3: u64 = 1_609_587_929_392_839_161;
428const XXH_P4: u64 = 9_650_029_242_287_828_579;
429const XXH_P5: u64 = 2_870_177_450_012_600_261;
430
431fn checksum_round(state: u64, word: u64) -> u64 {
432 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
433}
434
435fn checksum_word(chunk: &[u8]) -> u64 {
436 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
437}
438
439fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
441 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
442 *lane = checksum_round(*lane, checksum_word(chunk));
443 }
444}
445
446fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
448 let merge = |state: u64, lane: u64| {
449 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
450 };
451 let [one, two, three, four] = lanes;
452 let combined = one
453 .rotate_left(1)
454 .wrapping_add(two.rotate_left(7))
455 .wrapping_add(three.rotate_left(12))
456 .wrapping_add(four.rotate_left(18));
457 let hash = merge(merge(merge(merge(combined, one), two), three), four);
458 checksum_tail(hash.wrapping_add(length), rest)
459}
460
461fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
463 let mut words = rest.chunks_exact(8);
464 for chunk in words.by_ref() {
465 hash ^= checksum_round(0, checksum_word(chunk));
466 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
467 }
468 rest = words.remainder();
469 if rest.len() >= 4 {
470 let (head, tail) = rest.split_at(4);
471 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
472 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
473 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
474 rest = tail;
475 }
476 for &byte in rest {
477 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
478 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
479 }
480 hash ^= hash >> 33;
481 hash = hash.wrapping_mul(XXH_P2);
482 hash ^= hash >> 29;
483 hash = hash.wrapping_mul(XXH_P3);
484 hash ^ (hash >> 32)
485}
486
487fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
493 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
494}
495
496fn walk_checksummed(
502 file: &File,
503 offset: u64,
504 length: usize,
505 window: usize,
506 mut each: impl FnMut(&[u8]) -> Result<()>,
507) -> Result<u64> {
508 debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
509 if length < 32 {
510 let mut bytes = vec![0; length];
511 read_at(file, offset, &mut bytes)?;
512 each(&bytes)?;
513 return Ok(checksum(&bytes));
514 }
515 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
516 let mut buffer = vec![0; window.min(length)];
517 let mut read = 0;
518 let (mut whole, mut filled) = (0, 0);
519 while read < length {
520 filled = buffer.len().min(length - read);
521 read_at(file, offset + read as u64, &mut buffer[..filled])?;
522 read += filled;
523 each(&buffer[..filled])?;
524 whole = filled / 32 * 32;
525 for block in buffer[..whole].chunks_exact(32) {
526 checksum_block(&mut lanes, block);
527 }
528 }
529 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
530}
531
532#[derive(Debug, Clone, Copy)]
533struct Slot {
534 offset: u64,
535 length: u32,
536 generation: u64,
537 hash: u64,
538}
539
540impl Slot {
541 fn bytes(self) -> [u8; SLOT_BYTES] {
542 let mut result = [0; SLOT_BYTES];
543 result[..8].copy_from_slice(&self.offset.to_le_bytes());
544 result[8..12].copy_from_slice(&self.length.to_le_bytes());
545 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
546 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
547 result
548 }
549
550 fn read(bytes: &[u8]) -> Self {
551 Self {
552 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
553 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
554 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
555 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
556 }
557 }
558}
559
560#[derive(Debug, Clone, Copy)]
561struct Page {
562 offset: u64,
563 length: u32,
564 hash: u64,
565}
566
567impl Page {
568 fn bytes(&self) -> u64 {
570 u64::from(self.length)
571 }
572}
573
574#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
575enum FrequencyValue {
576 Null,
577 Integer(i128),
578 Code(u32),
579}
580
581type FrequencyMap<V> = HashMap<u64, V, Spread>;
587
588#[derive(Debug)]
602struct Candidates {
603 slots: Vec<Candidate>,
606 held: usize,
607 nulls: u32,
608 decrements: u64,
609 survivors: Vec<Candidate>,
611}
612
613#[derive(Debug, Default, Clone, Copy)]
615struct Candidate {
616 bits: u64,
617 count: u32,
618}
619
620const FIRST_CANDIDATE_SLOTS: usize = 64;
622
623impl Default for Candidates {
624 fn default() -> Self {
625 Self {
626 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
627 held: 0,
628 nulls: 0,
629 decrements: 0,
630 survivors: Vec::new(),
631 }
632 }
633}
634
635impl Candidates {
636 fn add(&mut self, bits: Option<u64>, mut times: u32) {
643 while times > 0 {
644 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
645 match bits {
646 Some(bits) => {
647 let (at, found) = self.find(bits);
648 if found {
649 self.slots[at].count = self.slots[at].count.saturating_add(times);
650 return;
651 }
652 if room {
653 self.place(at, bits, times);
654 return;
655 }
656 }
657 None if self.nulls != 0 => {
658 self.nulls = self.nulls.saturating_add(times);
659 return;
660 }
661 None if room => {
662 self.nulls = times;
663 return;
664 }
665 None => {}
666 }
667 self.decrement();
668 times -= 1;
669 }
670 }
671
672 fn find(&self, bits: u64) -> (usize, bool) {
674 let mask = self.slots.len() - 1;
675 let mut at = home(bits, self.slots.len());
676 loop {
677 let slot = self.slots[at];
678 if slot.count == 0 {
679 return (at, false);
680 }
681 if slot.bits == bits {
682 return (at, true);
683 }
684 at = (at + 1) & mask;
685 }
686 }
687
688 fn position(&self, bits: u64) -> Option<usize> {
690 match self.find(bits) {
691 (at, true) => Some(at),
692 (_, false) => None,
693 }
694 }
695
696 fn place(&mut self, at: usize, bits: u64, count: u32) {
699 let at = if (self.held + 1) * 2 > self.slots.len() {
700 let wider = self.slots.len() * 2;
701 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
702 for slot in old.into_iter().filter(|slot| slot.count != 0) {
703 let (to, _) = self.find(slot.bits);
704 self.slots[to] = slot;
705 }
706 self.find(bits).0
707 } else {
708 at
709 };
710 self.slots[at] = Candidate { bits, count };
711 self.held += 1;
712 }
713
714 fn decrement(&mut self) {
716 let mut survivors = std::mem::take(&mut self.survivors);
717 survivors.clear();
718 survivors.extend(
719 self.slots
720 .iter()
721 .filter(|slot| slot.count > 1)
722 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
723 );
724 self.slots.fill(Candidate::default());
725 self.held = survivors.len();
726 for &slot in &survivors {
727 let (at, _) = self.find(slot.bits);
728 self.slots[at] = slot;
729 }
730 self.survivors = survivors;
731 self.nulls = self.nulls.saturating_sub(1);
732 self.decrements = self.decrements.saturating_add(1);
733 }
734
735 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
737 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
738 }
739}
740
741fn home(bits: u64, slots: usize) -> usize {
746 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
747}
748
749#[derive(Debug, Default)]
751struct Run {
752 bits: Option<u64>,
753 times: u32,
754}
755
756impl Run {
757 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
759 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
760 self.times += 1;
761 return None;
762 }
763 let ended = self.take();
764 self.bits = bits;
765 self.times = 1;
766 ended
767 }
768
769 fn take(&mut self) -> Option<(Option<u64>, u32)> {
771 let times = std::mem::take(&mut self.times);
772 (times != 0).then_some((self.bits, times))
773 }
774}
775
776#[derive(Debug, Default, Clone, Copy)]
778struct Spread;
779
780impl std::hash::BuildHasher for Spread {
781 type Hasher = SpreadHasher;
782
783 fn build_hasher(&self) -> SpreadHasher {
784 SpreadHasher(0)
785 }
786}
787
788#[derive(Debug)]
795struct SpreadHasher(u64);
796
797impl SpreadHasher {
798 fn mix(&mut self, word: u64) {
799 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
800 self.0 = (product as u64) ^ ((product >> 64) as u64);
801 }
802}
803
804impl std::hash::Hasher for SpreadHasher {
805 fn write(&mut self, bytes: &[u8]) {
806 for part in bytes.chunks(8) {
807 let mut word = [0; 8];
808 word[..part.len()].copy_from_slice(part);
809 self.mix(u64::from_le_bytes(word));
810 }
811 }
812
813 fn write_u32(&mut self, value: u32) {
814 self.mix(u64::from(value));
815 }
816
817 fn write_u64(&mut self, value: u64) {
818 self.mix(value);
819 }
820
821 fn write_i128(&mut self, value: i128) {
822 self.mix(value as u64);
823 self.mix((value >> 64) as u64);
824 }
825
826 fn write_isize(&mut self, value: isize) {
827 self.mix(value as u64);
828 }
829
830 fn finish(&self) -> u64 {
831 self.0
832 }
833}
834
835#[derive(Debug, Clone)]
836struct FrequencyEntry {
837 value: FrequencyValue,
838 count: u64,
839}
840
841#[derive(Debug, Clone)]
846struct FrequencySummary {
847 entries: Vec<FrequencyEntry>,
848 omitted_max: u64,
849 ordinals: Vec<u64>,
850 ordinal_entries: Vec<u16>,
851}
852
853#[derive(Debug, Clone)]
854struct PairFrequencyEntry {
855 first_entry: u16,
856 second: Option<u32>,
857 count: u64,
858}
859
860#[derive(Debug, Clone)]
866struct PairFrequencySummary {
867 first: u16,
868 second: u16,
869 entries: Vec<PairFrequencyEntry>,
870 omitted_max: u64,
871}
872
873#[derive(Debug, Clone)]
881enum Frequencies {
882 Held(FrequencySummary),
883 Stored {
886 span: Span,
887 values: bool,
888 },
889}
890
891#[derive(Debug, Clone)]
896pub struct FrequencyPrefix {
897 pub entries: Vec<(Value, u64)>,
899 pub omitted_max: u64,
901}
902
903#[derive(Debug, Clone, PartialEq)]
905pub struct FrequencyOccurrences {
906 pub omitted_max: u64,
908 pub ordinals: Vec<u64>,
910 pub anchors: Vec<Value>,
912 pub anchor_indices: Vec<u16>,
914}
915
916pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
918
919#[derive(Debug, Clone, Copy, Default)]
926struct Span {
927 offset: u64,
928 length: u32,
929}
930
931#[derive(Debug, Clone, Default)]
939struct Pages {
940 columns: usize,
941 held: Box<[StripePage]>,
942}
943
944#[derive(Debug, Clone, Copy)]
946struct StripePage {
947 offset: u64,
948 hash: u64,
949 length: u32,
950 column: u32,
951}
952
953impl Pages {
954 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
956 let mut held = Vec::with_capacity(slots.iter().flatten().count());
957 for (column, page) in slots.iter().enumerate() {
958 if let Some(page) = page {
959 let column =
960 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
961 held.push(StripePage {
962 offset: page.offset,
963 hash: page.hash,
964 length: page.length,
965 column,
966 });
967 }
968 }
969 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
970 }
971
972 fn get(&self, column: usize) -> Option<Page> {
974 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
975 let placed = self.held[at];
976 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
977 }
978
979 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
981 (0..self.columns).map(|column| self.get(column))
982 }
983
984 fn bytes(&self, column: usize) -> u64 {
986 self.get(column).map_or(0, |page| page.bytes())
987 }
988}
989
990#[derive(Debug, Clone)]
992pub struct Stripe {
993 rows: usize,
994 parts: Vec<u32>,
997 index: Span,
1001 pages: Vec<Span>,
1002 memberships: Pages,
1003 sieves: Pages,
1006 part_ranges: Pages,
1017 zone: Zone,
1018}
1019
1020impl Stripe {
1021 #[must_use]
1023 pub fn rows(&self) -> usize {
1024 self.rows
1025 }
1026
1027 #[must_use]
1029 pub fn parts(&self) -> usize {
1030 self.parts.len()
1031 }
1032
1033 #[must_use]
1039 pub fn zone(&self) -> &Zone {
1040 &self.zone
1041 }
1042}
1043
1044#[derive(Debug, Clone)]
1046pub struct Table {
1047 name: String,
1048 fields: Vec<Field>,
1049 stripes: Vec<Stripe>,
1050 rows: usize,
1051 dictionaries: Vec<Option<Page>>,
1052 dictionary_payloads: Vec<u64>,
1058 demoted: Vec<bool>,
1064 frequencies: Vec<Option<Frequencies>>,
1065 pair_frequencies: Vec<PairFrequencySummary>,
1066 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1071 host_groups: Option<host::HostSummary>,
1073 distincts: Vec<Option<u64>>,
1083 clustering: Option<Clustering>,
1091 generation: u64,
1105 sections: Vec<Section>,
1112}
1113
1114impl Table {
1115 #[must_use]
1117 pub fn name(&self) -> &str {
1118 &self.name
1119 }
1120
1121 #[must_use]
1123 pub fn fields(&self) -> &[Field] {
1124 &self.fields
1125 }
1126
1127 #[must_use]
1129 pub fn rows(&self) -> usize {
1130 self.rows
1131 }
1132
1133 #[must_use]
1135 pub fn stripes(&self) -> &[Stripe] {
1136 &self.stripes
1137 }
1138
1139 #[must_use]
1141 pub fn clustering(&self) -> Option<&Clustering> {
1142 self.clustering.as_ref()
1143 }
1144
1145 #[must_use]
1150 pub fn generation(&self) -> u64 {
1151 self.generation
1152 }
1153
1154 #[must_use]
1161 pub fn sections(&self) -> &[Section] {
1162 &self.sections
1163 }
1164}
1165
1166#[derive(Debug, Clone)]
1178struct Entry {
1179 name: String,
1180 fields: Vec<Field>,
1181 rows: usize,
1182 directory: Page,
1184 nonzero: Vec<Option<u64>>,
1187 aggregates: Vec<Option<(i128, u64)>>,
1189 distincts: Vec<Option<u64>>,
1191 extremes: Vec<StoredIntegerExtremes>,
1193 frequencies: Vec<StoredNumericFrequencies>,
1195}
1196
1197type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1198type StoredNumericFrequencies = Option<NumericFrequencies>;
1199
1200#[derive(Debug, Clone, PartialEq, Eq)]
1213pub struct ViewEntry {
1214 pub name: String,
1216 pub sql: String,
1218 pub statement: String,
1220 pub aliases: Vec<String>,
1222 pub columns: Vec<Field>,
1224}
1225
1226#[derive(Debug, Clone)]
1228pub struct ColumnLayout {
1229 pub name: String,
1231 pub kind: String,
1233 pub pages: u64,
1235 pub memberships: u64,
1237 pub sieves: u64,
1239 pub part_ranges: u64,
1241 pub dictionary: u64,
1243}
1244
1245impl ColumnLayout {
1246 #[must_use]
1248 pub fn total(&self) -> u64 {
1249 self.pages
1250 .saturating_add(self.memberships)
1251 .saturating_add(self.sieves)
1252 .saturating_add(self.part_ranges)
1253 .saturating_add(self.dictionary)
1254 }
1255}
1256
1257#[derive(Debug, Clone)]
1268pub struct Layout {
1269 pub file: u64,
1271 pub rows: usize,
1273 pub stripes: usize,
1275 pub parts: usize,
1277 pub columns: Vec<ColumnLayout>,
1279 pub indexes: u64,
1282 pub directory: u64,
1284 pub header: u64,
1286}
1287
1288impl Layout {
1289 #[must_use]
1291 pub fn columns_total(&self) -> u64 {
1292 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1293 }
1294
1295 #[must_use]
1301 pub fn unaccounted(&self) -> u64 {
1302 self.file
1303 .saturating_sub(self.columns_total())
1304 .saturating_sub(self.indexes)
1305 .saturating_sub(self.directory)
1306 .saturating_sub(self.header)
1307 }
1308}
1309
1310#[derive(Debug, Clone)]
1321pub struct StoredPart {
1322 pub stripe: usize,
1324 pub part: usize,
1326 pub row: usize,
1328 pub rows: usize,
1330 pub encoding: String,
1332 pub bytes: u64,
1334 pub page: u64,
1336 pub offset: u64,
1338 pub low: Option<Value>,
1340 pub high: Option<Value>,
1342 pub nulls: Option<usize>,
1344}
1345
1346const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1353
1354#[derive(Debug)]
1379struct GlobalDictionary {
1380 primary: HashMap<u64, u32, Spread>,
1384 collisions: HashMap<u64, Vec<u32>, Spread>,
1385 checks: Vec<u64>,
1387 ends: Vec<u32>,
1389 counts: Vec<u64>,
1390 nulls: u64,
1391 filling: Vec<u8>,
1393 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1399 waiting: Vec<(usize, Vec<u8>)>,
1404 sample: Vec<(usize, Vec<u8>)>,
1410 stride: usize,
1412 shape: Option<chooser::Settled>,
1414 settled: usize,
1416 blocks: Vec<Vec<u8>>,
1421 early: BTreeMap<usize, EncodedBlock>,
1427 placed: Vec<Placed>,
1429 charged: u64,
1432 demoted: bool,
1434}
1435
1436#[derive(Debug, Clone, Copy)]
1438struct Placed {
1439 start: u64,
1440 length: u64,
1441 hash: u64,
1442}
1443
1444type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1446
1447impl GlobalDictionary {
1448 fn new() -> Self {
1449 Self {
1450 primary: HashMap::default(),
1451 collisions: HashMap::default(),
1452 checks: Vec::new(),
1453 ends: Vec::new(),
1454 counts: Vec::new(),
1455 nulls: 0,
1456 filling: Vec::new(),
1457 grams: Vec::new(),
1458 waiting: Vec::new(),
1459 sample: Vec::new(),
1460 stride: 1,
1461 shape: None,
1462 settled: 0,
1463 blocks: Vec::new(),
1464 early: BTreeMap::new(),
1465 placed: Vec::new(),
1466 charged: 0,
1467 demoted: false,
1468 }
1469 }
1470
1471 fn values(&self) -> usize {
1473 self.ends.len()
1474 }
1475
1476 fn closing_bytes(&self) -> usize {
1479 let values = self.values();
1480 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1481 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1482 .sum::<usize>();
1483 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1484 }
1485
1486 fn held_bytes(&self) -> u64 {
1492 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1493 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1494 }
1495 fn spilled<T>(values: &Vec<T>) -> usize {
1496 values.capacity() * size_of::<T>()
1497 }
1498 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1499 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1500 };
1501 let bytes = table(&self.primary)
1502 + table(&self.collisions)
1503 + self.collisions.values().map(spilled).sum::<usize>()
1504 + spilled(&self.checks)
1505 + spilled(&self.ends)
1506 + spilled(&self.counts)
1507 + self.filling.capacity()
1508 + spilled(&self.grams)
1509 + raw(&self.waiting)
1510 + raw(&self.sample)
1511 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1512 + spilled(&self.placed);
1513 bytes as u64
1514 }
1515
1516 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1519 let before = self.charged;
1520 let now = self.held_bytes();
1521 if let Some(profile) = profile {
1522 if now >= before {
1523 profile.hold(now - before);
1524 } else {
1525 profile.release(before - now);
1526 }
1527 }
1528 self.charged = now;
1529 (before, now)
1530 }
1531
1532 fn demote(&mut self) {
1540 if self.demoted {
1541 return;
1542 }
1543 self.seal_rest();
1544 self.release_lookup();
1545 self.demoted = true;
1546 }
1547
1548 fn release_lookup(&mut self) {
1555 self.primary = HashMap::default();
1556 self.collisions = HashMap::default();
1557 self.checks = Vec::new();
1558 self.sample = Vec::new();
1559 self.filling = Vec::new();
1560 }
1561
1562 fn encoded(&self) -> usize {
1564 self.placed.len() + self.blocks.len()
1565 }
1566
1567 #[cfg(test)]
1568 fn code(&mut self, text: &str) -> Result<u32> {
1569 let bytes = text.as_bytes();
1570 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1571 }
1572
1573 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1579 if let Some(&code) = self.primary.get(&hash) {
1580 if self.checks.get(code as usize) == Some(&check) {
1581 return Ok(code);
1582 }
1583 if let Some(codes) = self.collisions.get(&hash) {
1584 if let Some(code) =
1585 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1586 {
1587 return Ok(code);
1588 }
1589 }
1590 let code = self.insert(text, check)?;
1591 self.collisions.entry(hash).or_default().push(code);
1592 return Ok(code);
1593 }
1594 let code = self.insert(text, check)?;
1595 self.primary.insert(hash, code);
1596 Ok(code)
1597 }
1598
1599 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1600 if self.demoted {
1601 return Err(Error::internal("a value was coded against a demoted dictionary"));
1602 }
1603 let code = u32::try_from(self.ends.len())
1604 .map_err(|_| invalid("global dictionary has too many values"))?;
1605 self.filling.extend_from_slice(text);
1606 self.ends.push(
1607 u32::try_from(self.filling.len())
1608 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1609 );
1610 self.checks.push(check);
1611 self.counts.push(0);
1612 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1613 self.seal();
1614 }
1615 Ok(code)
1616 }
1617
1618 fn seal(&mut self) {
1624 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1625 let bytes = std::mem::take(&mut self.filling);
1626 if at % self.stride == 0 {
1627 self.sample.push((at, bytes.clone()));
1628 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1629 self.stride *= 2;
1630 let stride = self.stride;
1631 self.sample.retain(|(at, _)| at % stride == 0);
1632 }
1633 }
1634 self.waiting.push((at, bytes));
1635 }
1636
1637 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1639 block_values(self.block_ends(at), bytes)
1640 }
1641
1642 fn block_ends(&self, at: usize) -> &[u32] {
1644 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1645 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1646 &self.ends[first..last]
1647 }
1648
1649 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1656 let Some(shape) = &self.shape else { return Vec::new() };
1657 let waiting = std::mem::take(&mut self.waiting);
1658 waiting
1659 .into_iter()
1660 .map(|(at, bytes)| Unencoded {
1661 column,
1662 at,
1663 ends: self.block_ends(at).to_vec(),
1664 bytes,
1665 shape: shape.clone(),
1666 })
1667 .collect()
1668 }
1669
1670 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1673 if at < self.encoded() || self.early.insert(at, block).is_some() {
1674 return Err(Error::internal("a dictionary block came back twice"));
1675 }
1676 while let Some(block) = self.early.remove(&self.encoded()) {
1677 self.push_block(block);
1678 }
1679 Ok(())
1680 }
1681
1682 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1684 self.blocks.push(bytes);
1685 self.grams.push(*grams);
1686 }
1687
1688 fn settle(&mut self) -> Result<()> {
1696 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1697 return Ok(());
1698 }
1699 self.settle_on_sample()
1700 }
1701
1702 fn settle_rest(&mut self) -> Result<()> {
1710 if self.shape.is_some() || self.sample.is_empty() {
1711 return Ok(());
1712 }
1713 self.settle_on_sample()
1714 }
1715
1716 fn settle_on_sample(&mut self) -> Result<()> {
1717 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1718 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1719 return Ok(());
1720 }
1721 let sample =
1722 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1723 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1724 self.settled = complete;
1725 Ok(())
1726 }
1727
1728 fn seal_rest(&mut self) {
1730 if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1734 self.seal();
1735 }
1736 }
1737
1738 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1741 let (block, bytes) = &self.waiting[at];
1742 let values = self.slices(*block, bytes);
1743 let encoded = match &self.shape {
1744 Some(shape) => string::encode_with(&values, shape)?,
1745 None => string::encode(&values)?,
1746 };
1747 Ok((encoded, block_grams(&values)))
1748 }
1749
1750 #[cfg(test)]
1752 fn finish_blocks(&mut self) -> Result<()> {
1753 self.seal_rest();
1754 let made = (0..self.waiting.len())
1755 .map(|at| self.encode_waiting(at))
1756 .collect::<Result<Vec<_>>>()?;
1757 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1758 if self.encoded() != at {
1759 return Err(Error::internal("a dictionary block was encoded out of order"));
1760 }
1761 self.push_block(block);
1762 }
1763 Ok(())
1764 }
1765
1766 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1784 let count = self.placed.len() + self.blocks.len();
1785 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1786 return Err(invalid("global dictionary blocks do not cover its values"));
1787 }
1788 let mut bases = Vec::with_capacity(count);
1789 let mut total = 0_usize;
1790 for block in 0..count {
1791 bases.push(total as u64);
1792 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1793 total = total
1794 .checked_add(self.ends[last] as usize)
1795 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1796 }
1797 let mut flat = vec![0_u8; total];
1798 let mut outs = Vec::with_capacity(count);
1799 let mut rest = flat.as_mut_slice();
1800 for block in 0..count {
1801 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1802 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1803 outs.push((block, out));
1804 rest = after;
1805 }
1806 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1807 let mut stored = Vec::new();
1808 for (block, out) in run {
1809 let encoded = match self.placed.get(*block) {
1810 Some(place) => {
1811 let file = file.ok_or_else(|| {
1812 Error::internal("a written dictionary block has no file")
1813 })?;
1814 let length = usize::try_from(place.length).map_err(|_| {
1815 invalid("global dictionary block does not fit in memory")
1816 })?;
1817 stored.resize(length, 0);
1818 read_at(file, place.start, &mut stored)?;
1819 if checksum(&stored) != place.hash {
1820 return Err(invalid(
1821 "a global dictionary block did not read back as written",
1822 ));
1823 }
1824 stored.as_slice()
1825 }
1826 None => &self.blocks[*block - self.placed.len()],
1827 };
1828 let decoded = string::decode_flat(encoded)?;
1829 if decoded.bytes().len() != out.len() {
1830 return Err(invalid(
1831 "a global dictionary block is not the length its ends say",
1832 ));
1833 }
1834 out.copy_from_slice(decoded.bytes());
1835 }
1836 Ok(())
1837 };
1838 let workers = close_workers().min(count / 16).max(1);
1841 if workers <= 1 {
1842 one(&mut outs)?;
1843 } else {
1844 let per = count.div_ceil(workers);
1845 std::thread::scope(|scope| {
1846 outs.chunks_mut(per)
1847 .map(|run| scope.spawn(|| one(run)))
1848 .collect::<Vec<_>>()
1849 .into_iter()
1850 .try_for_each(|handle| {
1851 handle.join().map_err(|_| {
1852 Error::internal("a global dictionary decode worker panicked")
1853 })?
1854 })
1855 })?;
1856 }
1857 drop(outs);
1858 Ok((flat, bases))
1859 }
1860
1861 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1866 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1867 let Some(&end) = ends.get(code) else { return (0, 0) };
1868 let base = base as usize;
1869 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1870 (base + from, base + end as usize)
1871 }
1872
1873 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1893 let (flat, bases) = self.decoded(file)?;
1894 let value = |code: u32| {
1895 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1896 flat.get(from..to).unwrap_or_default()
1897 };
1898 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1899 sort_by_value_across(&mut codes, value, close_workers());
1900 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1901 Ok((order, flat, bases))
1902 }
1903
1904 #[cfg(test)]
1905 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1906 self.ranked_with_values(file).map(|(order, _, _)| order)
1907 }
1908}
1909
1910#[derive(Debug)]
1918pub struct Writer {
1919 file: Box<dyn rudb_io::File>,
1922 at: u64,
1930 written_back: u64,
1932 table: Table,
1933 generation: u64,
1934 order: Vec<((u64, u64), (u64, u64))>,
1937 next_order: u64,
1938 dictionaries: Vec<Option<GlobalDictionary>>,
1939 coded: Arc<prepare::Coding>,
1942 gathers: Vec<Option<stats::Gather>>,
1948 lent: Option<Arc<Lent>>,
1951 pending: Vec<PendingChunk>,
1952 closed: Vec<Entry>,
1954 views: Vec<ViewEntry>,
1959 profile: Option<Arc<LoadProfile>>,
1965}
1966
1967#[derive(Debug)]
1975struct PendingChunk {
1976 order: (u64, u64),
1977 chunk: Chunk,
1978}
1979
1980#[derive(Debug, Clone, Copy)]
1986struct Part {
1987 order: (u64, u64),
1988 rows: usize,
1989 footprint: usize,
1990}
1991
1992impl Part {
1993 fn of(pending: &PendingChunk) -> Self {
1994 Self {
1995 order: pending.order,
1996 rows: pending.chunk.len(),
1997 footprint: pending.chunk.footprint(),
1998 }
1999 }
2000}
2001
2002#[derive(Debug, Default)]
2008struct ColumnStripe {
2009 pages: Vec<Vec<u8>>,
2010 codes: Vec<Option<Vec<u32>>>,
2011 sieves: Vec<Option<Sieve>>,
2012 ranges: Vec<Range>,
2013}
2014
2015fn coded_type(ty: &LogicalType) -> bool {
2023 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2024}
2025
2026fn dictionary_tag(ty: &LogicalType) -> u8 {
2033 if ty == &LogicalType::Blob { 2 } else { 1 }
2034}
2035
2036fn weight(ty: &LogicalType) -> usize {
2044 match ty {
2045 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2046 LogicalType::HugeInt
2047 | LogicalType::UHugeInt
2048 | LogicalType::Uuid
2049 | LogicalType::Interval => 16,
2050 LogicalType::BigInt
2051 | LogicalType::UBigInt
2052 | LogicalType::Timestamp
2053 | LogicalType::Time
2054 | LogicalType::TimeTz
2055 | LogicalType::TimestampTz
2056 | LogicalType::TimestampS
2057 | LogicalType::TimestampMs
2058 | LogicalType::TimestampNs
2059 | LogicalType::Double
2060 | LogicalType::Decimal { .. } => 8,
2061 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2062 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2063 _ => 1,
2064 }
2065}
2066
2067pub const STRIPE_PARTS: usize = 64;
2074
2075const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2083
2084const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2100
2101const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2103
2104fn index_section(parts: usize) -> Result<usize> {
2106 parts
2107 .checked_mul(INDEX_ENTRY)
2108 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2109 .ok_or_else(|| invalid("index page length overflow"))
2110}
2111
2112impl Writer {
2113 pub fn open(
2132 path: impl AsRef<Path>,
2133 name: impl Into<String>,
2134 fields: Vec<Field>,
2135 ) -> Result<Self> {
2136 Self::open_in(&RealFilesystem::new(), path, name, fields)
2137 }
2138
2139 pub fn open_in(
2146 fs: &dyn Filesystem,
2147 path: impl AsRef<Path>,
2148 name: impl Into<String>,
2149 fields: Vec<Field>,
2150 ) -> Result<Self> {
2151 for field in &fields {
2152 type_tag(&field.ty)?;
2153 }
2154 let name = name.into();
2155 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2156 let size = file.len()?;
2157 let (slot, bytes, _) = committed_slot(&*file, size)?;
2158 let (mut closed, views) = decode_catalog(&bytes, size)?;
2159 if let Some(at) = closed.iter().position(|held| held.name == name) {
2170 if closed[at].rows > 0 {
2171 return Err(invalid("two tables in one native file have the same name"));
2172 }
2173 closed.remove(at);
2174 }
2175 let generation = slot
2180 .generation
2181 .checked_add(1)
2182 .ok_or_else(|| invalid("native file generation overflow"))?;
2183 Ok(Self {
2184 file,
2185 at: size,
2188 written_back: size,
2189 dictionaries: fields
2190 .iter()
2191 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2192 .collect(),
2193 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2194 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2195 lent: None,
2196 table: Table {
2197 name,
2198 dictionaries: vec![None; fields.len()],
2199 dictionary_payloads: Vec::new(),
2200 demoted: Vec::new(),
2201 distincts: vec![None; fields.len()],
2202 fields,
2203 stripes: Vec::new(),
2204 rows: 0,
2205 frequencies: Vec::new(),
2206 pair_frequencies: Vec::new(),
2207 frequency_texts: Vec::new(),
2208 host_groups: None,
2209 clustering: None,
2210 generation,
2211 sections: Vec::new(),
2212 },
2213 generation,
2214 order: Vec::new(),
2215 next_order: 0,
2216 pending: Vec::with_capacity(STRIPE_PARTS),
2217 closed,
2218 views,
2219 profile: None,
2220 })
2221 }
2222
2223 pub fn create(
2229 path: impl AsRef<Path>,
2230 name: impl Into<String>,
2231 fields: Vec<Field>,
2232 ) -> Result<Self> {
2233 Self::create_in(&RealFilesystem::new(), path, name, fields)
2234 }
2235
2236 pub fn create_in(
2246 fs: &dyn Filesystem,
2247 path: impl AsRef<Path>,
2248 name: impl Into<String>,
2249 fields: Vec<Field>,
2250 ) -> Result<Self> {
2251 for field in &fields {
2252 type_tag(&field.ty)?;
2253 }
2254 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2255 let mut header = [0; HEADER as usize];
2256 header[..8].copy_from_slice(MAGIC);
2257 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2258 file.write_at(0, &header)?;
2259 Ok(Self {
2260 file,
2261 at: HEADER,
2262 written_back: HEADER,
2263 dictionaries: fields
2264 .iter()
2265 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2266 .collect(),
2267 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2268 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2269 lent: None,
2270 table: Table {
2271 name: name.into(),
2272 dictionaries: vec![None; fields.len()],
2273 dictionary_payloads: Vec::new(),
2274 demoted: Vec::new(),
2275 distincts: vec![None; fields.len()],
2276 fields,
2277 stripes: Vec::new(),
2278 rows: 0,
2279 frequencies: Vec::new(),
2280 pair_frequencies: Vec::new(),
2281 frequency_texts: Vec::new(),
2282 host_groups: None,
2283 clustering: None,
2284 generation: 1,
2285 sections: Vec::new(),
2286 },
2287 generation: 1,
2288 order: Vec::new(),
2289 next_order: 0,
2290 pending: Vec::with_capacity(STRIPE_PARTS),
2291 closed: Vec::new(),
2292 views: Vec::new(),
2293 profile: None,
2294 })
2295 }
2296
2297 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2319 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2320 let mut header = [0; HEADER as usize];
2321 header[..8].copy_from_slice(MAGIC);
2322 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2323 file.write_at(0, &header)?;
2324 let catalog = encode_catalog(&[], views)?;
2325 file.write_at(HEADER, &catalog)?;
2326 file.sync()?;
2330 let slot = Slot {
2331 offset: HEADER,
2332 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2333 generation: 1,
2334 hash: checksum(&catalog),
2335 };
2336 file.write_at(slot_offset(1), &slot.bytes())?;
2337 file.sync()?;
2338 Ok(())
2339 }
2340
2341 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2352 for field in &fields {
2353 type_tag(&field.ty)?;
2354 }
2355 let name = name.into();
2356 let entry = self.close()?;
2357 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2358 return Err(invalid("two tables in one native file have the same name"));
2359 }
2360 let Self { file, at, generation, mut closed, views, .. } = self;
2361 closed.push(entry);
2362 Ok(Self {
2363 file,
2364 written_back: at,
2365 at,
2366 generation,
2367 closed,
2368 views,
2369 profile: None,
2370 dictionaries: fields
2371 .iter()
2372 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2373 .collect(),
2374 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2375 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2376 lent: None,
2377 table: Table {
2378 name,
2379 dictionaries: vec![None; fields.len()],
2380 dictionary_payloads: Vec::new(),
2381 demoted: Vec::new(),
2382 distincts: vec![None; fields.len()],
2383 fields,
2384 stripes: Vec::new(),
2385 rows: 0,
2386 frequencies: Vec::new(),
2387 pair_frequencies: Vec::new(),
2388 frequency_texts: Vec::new(),
2389 host_groups: None,
2390 clustering: None,
2391 generation,
2392 sections: Vec::new(),
2393 },
2394 order: Vec::new(),
2395 next_order: 0,
2396 pending: Vec::with_capacity(STRIPE_PARTS),
2397 })
2398 }
2399
2400 #[must_use]
2410 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2411 self.views = views;
2412 self
2413 }
2414
2415 #[must_use]
2421 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2422 self.profile = Some(profile);
2423 self
2424 }
2425
2426 #[must_use]
2430 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2431 self.coded.cap(bytes);
2432 self
2433 }
2434
2435 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2450 self.table.clustering = Some(Clustering::new(
2453 clustering.columns().to_vec(),
2454 clustering.width(),
2455 &self.table.fields,
2456 )?);
2457 Ok(self)
2458 }
2459
2460 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2465 self.file.write_at(self.at, bytes)?;
2466 self.at = self
2467 .at
2468 .checked_add(bytes.len() as u64)
2469 .ok_or_else(|| invalid("native file length overflow"))?;
2470 if self.at - self.written_back >= WRITEBACK_STRETCH {
2471 self.file.start_writeback(self.written_back, self.at - self.written_back);
2472 self.written_back = self.at;
2473 }
2474 Ok(())
2475 }
2476
2477 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2483 let order = (self.next_order, 0);
2484 self.next_order = self.next_order.saturating_add(1);
2485 self.append_at(order, chunk)
2486 }
2487
2488 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2499 if chunk.is_empty() {
2500 return Ok(());
2501 }
2502 self.admit(chunk)?;
2503 if self.pending.last().is_some_and(|last| last.order > order) {
2504 self.flush_pending()?;
2505 }
2506 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2511 if self.pending.len() == STRIPE_PARTS {
2512 self.flush_pending()?;
2513 }
2514 Ok(())
2515 }
2516
2517 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2533 if parts.len() > STRIPE_PARTS {
2534 return Err(invalid("a stripe was handed more parts than it holds"));
2535 }
2536 self.flush_pending()?;
2539 for (order, chunk) in parts {
2540 if chunk.is_empty() {
2541 continue;
2542 }
2543 self.admit(&chunk)?;
2544 self.pending.push(PendingChunk { order, chunk });
2545 }
2546 self.flush_pending()
2547 }
2548
2549 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2551 if chunk.width() != self.table.fields.len() {
2552 return Err(invalid("chunk width differs from table schema"));
2553 }
2554 for (index, field) in self.table.fields.iter().enumerate() {
2555 if chunk.column(index)?.logical_type() != &field.ty {
2556 return Err(invalid("chunk type differs from table schema"));
2557 }
2558 }
2559 self.table.rows = self
2560 .table
2561 .rows
2562 .checked_add(chunk.len())
2563 .ok_or_else(|| invalid("row count overflow"))?;
2564 Ok(())
2565 }
2566
2567 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2569 let mut stripe = ColumnStripe {
2570 pages: Vec::with_capacity(columns.len()),
2571 codes: Vec::with_capacity(columns.len()),
2572 sieves: Vec::with_capacity(columns.len()),
2573 ranges: Vec::with_capacity(columns.len()),
2574 };
2575 let mut settling = Settling::default();
2576 for &column in columns {
2577 Self::encode_page(&mut stripe, &mut settling, column)?;
2578 }
2579 Ok(stripe)
2580 }
2581
2582 fn encode_page(
2585 stripe: &mut ColumnStripe,
2586 settling: &mut Settling,
2587 column: &Vector,
2588 ) -> Result<()> {
2589 let bytes = encode(column, settling)?;
2590 if bytes.len() > MAX_PAGE {
2591 return Err(invalid("column page exceeds the configured bound"));
2592 }
2593 let range = Range::of(column);
2596 let sieve =
2607 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2608 stripe.pages.push(bytes);
2609 stripe.codes.push(None);
2610 stripe.sieves.push(sieve);
2611 stripe.ranges.push(range);
2612 Ok(())
2613 }
2614
2615 fn place_blocks(&mut self) -> Result<()> {
2620 if let Some(lent) = self.lent.clone() {
2621 return self.place_lent_blocks(&lent);
2622 }
2623 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2624 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2625 for block in std::mem::take(&mut dictionary.blocks) {
2626 let start = self.at;
2627 self.put(&block)?;
2628 dictionary.placed.push(Placed {
2629 start,
2630 length: block.len() as u64,
2631 hash: checksum(&block),
2632 });
2633 }
2634 Ok(())
2635 });
2636 self.dictionaries = dictionaries;
2637 placed
2638 }
2639
2640 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2646 for column in lent.columns() {
2647 let Ok(mut held) = column.try_lock() else { continue };
2648 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2649 for block in std::mem::take(&mut dictionary.blocks) {
2650 let start = self.at;
2651 self.put(&block)?;
2652 dictionary.placed.push(Placed {
2653 start,
2654 length: block.len() as u64,
2655 hash: checksum(&block),
2656 });
2657 }
2658 }
2659 Ok(())
2660 }
2661
2662 fn reclaim(&mut self) -> Result<()> {
2666 let Some(lent) = self.lent.take() else { return Ok(()) };
2667 let (dictionaries, gathers) = lent.reclaim()?;
2668 self.dictionaries = dictionaries;
2669 self.gathers = gathers;
2670 Ok(())
2671 }
2672
2673 fn flush_pending(&mut self) -> Result<()> {
2678 if self.pending.is_empty() {
2679 return Ok(());
2680 }
2681 let held = std::mem::take(&mut self.pending);
2682 let prepared = self.preparer().prepare_held(held)?;
2683 let merged = self.merge_held(prepared)?;
2684 let paged = merged.pages()?;
2685 self.write_paged(paged)
2686 }
2687
2688 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2690 let width = self.table.fields.len();
2691 let parts = held.len();
2692 if encoded.len() != width {
2693 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2694 }
2695 let profile = self.profile.clone();
2696 if let Some(profile) = &profile {
2697 let rows = held.iter().map(|part| part.rows as u64).sum();
2698 let raw = held.iter().map(|part| part.footprint as u64).sum();
2699 let pages =
2700 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2701 profile.moved(Stage::Pages, raw, pages, rows);
2702 }
2703 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2706 let before = self.at;
2707 self.place_blocks()?;
2708 drop(timing);
2709 if let Some(profile) = &profile {
2710 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2711 }
2712 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2713 let before = self.at;
2714 let mut pages = Vec::with_capacity(width);
2715 let mut memberships = vec![None; width];
2716 let mut ranges = Vec::with_capacity(width);
2717 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2718 for stripe in &encoded {
2719 let offset = self.at;
2720 let section = index.len();
2721 let mut length = 0_usize;
2722 for bytes in &stripe.pages {
2723 self.file.write_at(self.at + length as u64, bytes)?;
2724 put_u32(
2725 &mut index,
2726 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2727 );
2728 put_u64(&mut index, checksum(bytes));
2729 length = length
2730 .checked_add(bytes.len())
2731 .ok_or_else(|| invalid("column page length overflow"))?;
2732 }
2733 let hash = checksum(&index[section..]);
2734 put_u64(&mut index, hash);
2735 if length > MAX_PAGE {
2736 return Err(invalid("column page exceeds the configured bound"));
2737 }
2738 self.at = self
2739 .at
2740 .checked_add(length as u64)
2741 .ok_or_else(|| invalid("native file length overflow"))?;
2742 pages.push(Span {
2743 offset,
2744 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2745 });
2746 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2747 }
2748 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2749 if stripe.codes.iter().all(Option::is_none) {
2750 continue;
2751 }
2752 let lists = stripe
2753 .codes
2754 .iter()
2755 .map(|codes| codes.clone().unwrap_or_default())
2756 .collect::<Vec<_>>();
2757 let bytes = encode_membership(&merged_codes(lists));
2758 let offset = self.at;
2759 self.put(&bytes)?;
2760 *membership = Some(Page {
2761 offset,
2762 length: u32::try_from(bytes.len())
2763 .map_err(|_| invalid("membership page length overflow"))?,
2764 hash: checksum(&bytes),
2765 });
2766 }
2767 let mut sieves = vec![None; width];
2768 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2769 if stripe.sieves.iter().all(Option::is_none) {
2770 continue;
2771 }
2772 let bytes = encode_sieves(stripe.sieves.iter())?;
2773 let offset = self.at;
2774 self.put(&bytes)?;
2775 *page = Some(Page {
2776 offset,
2777 length: u32::try_from(bytes.len())
2778 .map_err(|_| invalid("sieve page length overflow"))?,
2779 hash: checksum(&bytes),
2780 });
2781 }
2782 let mut part_ranges = vec![None; width];
2788 if parts > 1 {
2789 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2790 let bytes = encode_part_ranges(&stripe.ranges)?;
2791 if bytes.len() >= span.length as usize {
2792 continue;
2793 }
2794 let offset = self.at;
2795 self.put(&bytes)?;
2796 *page = Some(Page {
2797 offset,
2798 length: u32::try_from(bytes.len())
2799 .map_err(|_| invalid("part range page length overflow"))?,
2800 hash: checksum(&bytes),
2801 });
2802 }
2803 }
2804 let offset = self.at;
2805 self.put(&index)?;
2806 let index = Span {
2807 offset,
2808 length: u32::try_from(index.len())
2809 .map_err(|_| invalid("index page length overflow"))?,
2810 };
2811 let mut rows = 0_usize;
2812 let mut lengths = Vec::with_capacity(parts);
2813 let mut span = None;
2814 for part in held {
2815 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2816 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2817 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2818 }
2819 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2820 self.table.stripes.push(Stripe {
2821 rows,
2822 parts: lengths,
2823 index,
2824 pages,
2825 memberships: Pages::from_slots(memberships)?,
2826 sieves: Pages::from_slots(sieves)?,
2827 part_ranges: Pages::from_slots(part_ranges)?,
2828 zone: Zone::from_ranges(ranges),
2829 });
2830 drop(timing);
2831 if let Some(profile) = &profile {
2832 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2833 }
2834 Ok(())
2835 }
2836
2837 fn numeric_frequency(
2857 &self,
2858 column: usize,
2859 counted: bool,
2860 dense: Option<(u64, usize)>,
2861 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2862 let signed = match self.table.fields[column].ty {
2863 LogicalType::TinyInt
2864 | LogicalType::SmallInt
2865 | LogicalType::Integer
2866 | LogicalType::BigInt
2867 | LogicalType::Date
2868 | LogicalType::Timestamp => true,
2869 LogicalType::UTinyInt
2870 | LogicalType::USmallInt
2871 | LogicalType::UInteger
2872 | LogicalType::UBigInt => false,
2873 _ => return Ok((None, None)),
2874 };
2875 let value_of = |bits: Option<u64>| match bits {
2876 None => FrequencyValue::Null,
2877 Some(bits) => integer_value(bits, signed),
2878 };
2879 let tallied = self
2884 .gathers
2885 .get(column)
2886 .and_then(Option::as_ref)
2887 .filter(|gather| gather.rows() == self.table.rows as u64)
2888 .and_then(stats::Gather::frequencies)
2889 .and_then(|(values, nulls)| {
2890 let entries = values
2891 .iter()
2892 .map(|(value, count)| {
2893 let value = value_of(Some(frequency_bits(value)?));
2894 Some(FrequencyEntry { value, count: *count })
2895 })
2896 .chain((nulls != 0).then_some(Some(FrequencyEntry {
2897 value: FrequencyValue::Null,
2898 count: nulls,
2899 })))
2900 .collect::<Option<Vec<_>>>()?;
2901 Some((entries, values.len() as u64))
2902 });
2903 let exact = match (&tallied, counted) {
2907 (None, true) => self.exact_frequency(column, signed, dense)?,
2908 _ => None,
2909 };
2910 let (mut entries, decrements, distinct_count) = match (tallied, exact) {
2911 (Some((entries, distinct)), _) => (entries, 0, Some(distinct)),
2912 (None, Some((Some(entries), distinct))) => (entries, 0, Some(distinct)),
2913 (None, Some((None, distinct))) => return Ok((None, Some(distinct))),
2914 (None, None) => {
2915 let mut first = Candidates::default();
2919 let mut run = Run::default();
2920 self.visit_numeric(column, signed, |_, bits| {
2921 if let Some((ended, times)) = run.push(bits) {
2922 first.add(ended, times);
2923 }
2924 })?;
2925 if let Some((bits, times)) = run.take() {
2926 first.add(bits, times);
2927 }
2928 let (nulls, decrements) = (first.nulls, first.decrements);
2931 let distinct_count = (decrements == 0).then_some(first.held as u64);
2932 let (exact, null_count) = if decrements == 0 {
2933 let exact = first
2934 .pairs()
2935 .map(|(bits, count)| (bits, u64::from(count)))
2936 .collect::<FrequencyMap<_>>();
2937 (exact, (nulls != 0).then_some(u64::from(nulls)))
2938 } else {
2939 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2940 if nulls != 0 {
2941 lower.push(nulls);
2942 }
2943 lower.sort_unstable_by(|left, right| right.cmp(left));
2944 if lower.len() < FREQUENCY_BUILD_RANK
2945 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2946 {
2947 return Ok((None, distinct_count));
2948 }
2949 let mut recounts = vec![0_u64; first.slots.len()];
2952 let mut null_count = (nulls != 0).then_some(0_u64);
2953 let mut recount = |bits: Option<u64>, times: u32| {
2954 let held = match bits {
2955 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2956 None => null_count.as_mut(),
2957 };
2958 if let Some(count) = held {
2959 *count = count.saturating_add(u64::from(times));
2960 }
2961 };
2962 let mut run = Run::default();
2963 self.visit_numeric(column, signed, |_, bits| {
2964 if let Some((bits, times)) = run.push(bits) {
2965 recount(bits, times);
2966 }
2967 })?;
2968 if let Some((bits, times)) = run.take() {
2969 recount(bits, times);
2970 }
2971 let exact = first
2972 .slots
2973 .iter()
2974 .zip(&recounts)
2975 .filter(|(slot, _)| slot.count != 0)
2976 .map(|(slot, &count)| (slot.bits, count))
2977 .collect::<FrequencyMap<_>>();
2978 (exact, null_count)
2979 };
2980 let entries = exact
2981 .into_iter()
2982 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2983 .chain(
2984 null_count
2985 .map(|count| FrequencyEntry { value: FrequencyValue::Null, count }),
2986 )
2987 .collect::<Vec<_>>();
2988 (entries, decrements, distinct_count)
2989 }
2990 };
2991 let mut omitted_max = keep_most_frequent(&mut entries).max(decrements);
2992 if omitted_max == 0 && entries.len() > 1 {
2996 let retained = entries.len().saturating_sub(1).min(2);
2997 omitted_max = entries[retained].count;
2998 entries.truncate(retained);
2999 }
3000 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3001 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3002 });
3003 let mut ordinals = Vec::new();
3004 let mut ordinal_entries = Vec::new();
3005 if let Some(kept_rows) = kept_rows {
3006 let mut kept = FrequencyMap::default();
3007 let mut null_kept = None;
3008 for (at, entry) in entries.iter().enumerate() {
3009 let at = u16::try_from(at)
3010 .map_err(|_| invalid("too many retained frequency entries"))?;
3011 match entry.value {
3012 FrequencyValue::Integer(value) => {
3013 kept.insert(value as u64, at);
3014 }
3015 FrequencyValue::Null => null_kept = Some(at),
3016 FrequencyValue::Code(_) => {}
3017 }
3018 }
3019 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3020 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3021 self.visit_numeric(column, signed, |ordinal, bits| {
3022 let held = match bits {
3023 Some(bits) => kept.get(&bits).copied(),
3024 None => null_kept,
3025 };
3026 if let Some(entry) = held {
3027 ordinals.push(ordinal);
3028 ordinal_entries.push(entry);
3029 }
3030 })?;
3031 }
3032 Ok((
3033 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3034 distinct_count,
3035 ))
3036 }
3037
3038 fn dense_range(&self, gather: &stats::Gather, set: usize) -> Option<(u64, usize)> {
3045 let rows = self.table.rows;
3046 if gather.rows() != rows as u64 || u32::try_from(rows).is_err() {
3047 return None;
3048 }
3049 let (low, high) = gather.span()?;
3050 let len = usize::try_from(high.checked_sub(low)?.checked_add(1)?).ok()?;
3051 #[allow(clippy::cast_sign_loss, clippy::cast_possible_truncation)]
3052 let bits = low as u64;
3053 (len.checked_mul(size_of::<u32>())? <= set.max(1 << 20)).then_some((bits, len))
3054 }
3055
3056 fn exact_frequency(
3070 &self,
3071 column: usize,
3072 signed: bool,
3073 dense: Option<(u64, usize)>,
3074 ) -> Result<Option<(Option<Vec<FrequencyEntry>>, u64)>> {
3075 if let Some((low, len)) = dense {
3078 let mut counts = distinct::DenseCounts::new(low, len);
3079 let nulls =
3080 self.count_numeric(column, signed, |bits, times| counts.insert(bits, times))?;
3081 if let Some(distinct) = counts.count() {
3082 let Some(distinct) = distinct else { return Ok(None) };
3083 return Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3084 counts.visit(visit);
3085 })));
3086 }
3087 }
3088 let mut set = distinct::ExactCounts::new();
3089 let nulls = self.count_numeric(column, signed, |bits, times| set.insert(bits, times))?;
3090 let Some(distinct) = set.count() else {
3091 return Ok(None);
3092 };
3093 Ok(Some(self.frequent_entries(signed, distinct, nulls, |visit| {
3094 set.visit(visit);
3095 })))
3096 }
3097
3098 fn count_numeric(
3101 &self,
3102 column: usize,
3103 signed: bool,
3104 mut add: impl FnMut(u64, u32),
3105 ) -> Result<u64> {
3106 let mut nulls = 0_u64;
3107 let mut run = Run::default();
3108 let mut take = |bits: Option<u64>, times: u32| match bits {
3109 Some(bits) => add(bits, times),
3110 None => nulls += u64::from(times),
3111 };
3112 self.visit_numeric(column, signed, |_, bits| {
3113 if let Some((bits, times)) = run.push(bits) {
3114 take(bits, times);
3115 }
3116 })?;
3117 if let Some((bits, times)) = run.take() {
3118 take(bits, times);
3119 }
3120 Ok(nulls)
3121 }
3122
3123 fn frequent_entries(
3126 &self,
3127 signed: bool,
3128 distinct: u64,
3129 nulls: u64,
3130 mut visit: impl FnMut(&mut dyn FnMut(u64, u64)),
3131 ) -> (Option<Vec<FrequencyEntry>>, u64) {
3132 let mut top = std::collections::BinaryHeap::with_capacity(FREQUENCY_ENTRIES + 2);
3134 let mut rank = |count: u64| {
3135 if top.len() <= FREQUENCY_ENTRIES {
3136 top.push(Reverse(count));
3137 } else if top.peek().is_some_and(|&Reverse(least)| count > least) {
3138 top.pop();
3139 top.push(Reverse(count));
3140 }
3141 };
3142 visit(&mut |_, count| rank(count));
3143 if nulls != 0 {
3144 rank(nulls);
3145 }
3146 let top = top.into_sorted_vec();
3147 let values = distinct + u64::from(nulls != 0);
3148 if values > FREQUENCY_CANDIDATES as u64 {
3149 let bound = self.table.rows as u64 / (FREQUENCY_CANDIDATES as u64 + 1);
3150 if top.get(FREQUENCY_BUILD_RANK - 1).is_none_or(|&Reverse(count)| count <= bound) {
3151 return (None, distinct);
3152 }
3153 }
3154 let least = top.get(FREQUENCY_ENTRIES).map_or(0, |&Reverse(count)| count);
3155 let mut entries = Vec::with_capacity(FREQUENCY_ENTRIES + 1);
3156 visit(&mut |bits, count| {
3157 if count >= least {
3158 entries.push(FrequencyEntry { value: integer_value(bits, signed), count });
3159 }
3160 });
3161 if nulls != 0 && nulls >= least {
3162 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: nulls });
3163 }
3164 (Some(entries), distinct)
3165 }
3166
3167 fn visit_numeric(
3174 &self,
3175 column: usize,
3176 signed: bool,
3177 mut visit: impl FnMut(u64, Option<u64>),
3178 ) -> Result<()> {
3179 let ty = &self.table.fields[column].ty;
3180 let mut start = 0_u64;
3181 let mut block = Vec::new();
3182 for stripe in &self.table.stripes {
3183 let spans = read_index(&self.file, stripe, column)?;
3184 let page = stripe.pages[column];
3185 let mut bytes = vec![0; page.length as usize];
3186 read_at(&self.file, page.offset, &mut bytes)?;
3187 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3188 let part = part_bytes(&bytes, *span)?;
3189 if checksum(part) != span.hash {
3190 return Err(invalid("column page checksum differs while building frequencies"));
3191 }
3192 let rows = rows as usize;
3193 let vector = decode(ty, rows, part, None)?;
3194 if signed && vector.signed_block(&mut block) && block.len() == rows {
3198 if vector.none_null() {
3199 for (row, &value) in block.iter().enumerate() {
3200 visit(start.saturating_add(row as u64), Some(value as u64));
3201 }
3202 } else {
3203 for (row, &value) in block.iter().enumerate() {
3204 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3205 visit(start.saturating_add(row as u64), bits);
3206 }
3207 }
3208 start = start.saturating_add(rows as u64);
3209 continue;
3210 }
3211 for row in 0..rows {
3213 let bits = if vector.is_null_at(row) {
3214 None
3215 } else {
3216 let widened = match vector.signed_at(row) {
3220 Some(value) => Some(value as u64),
3221 None => match vector.value_at(row) {
3222 Value::UTinyInt(value) => Some(u64::from(value)),
3223 Value::USmallInt(value) => Some(u64::from(value)),
3224 Value::UInteger(value) => Some(u64::from(value)),
3225 Value::UBigInt(value) => Some(value),
3226 _ => None,
3227 },
3228 };
3229 Some(widened.ok_or_else(|| {
3230 invalid("numeric frequency page did not contain an integer value")
3231 })?)
3232 };
3233 visit(start.saturating_add(row as u64), bits);
3234 }
3235 start = start.saturating_add(rows as u64);
3236 }
3237 }
3238 Ok(())
3239 }
3240
3241 fn numeric_columns(&self) -> Vec<usize> {
3243 self.table
3244 .fields
3245 .iter()
3246 .enumerate()
3247 .filter_map(|(column, field)| {
3248 matches!(
3249 field.ty,
3250 LogicalType::TinyInt
3251 | LogicalType::SmallInt
3252 | LogicalType::Integer
3253 | LogicalType::BigInt
3254 | LogicalType::UTinyInt
3255 | LogicalType::USmallInt
3256 | LogicalType::UInteger
3257 | LogicalType::UBigInt
3258 | LogicalType::Date
3259 | LogicalType::Timestamp
3260 )
3261 .then_some(column)
3262 })
3263 .collect()
3264 }
3265
3266 #[allow(dead_code)]
3268 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3269 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3270 return Ok(None);
3271 }
3272 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3273 return Err(invalid("frequency ordinals are not sorted and unique"));
3274 }
3275 let mut out = Vec::with_capacity(ordinals.len());
3276 let mut wanted = 0;
3277 let mut stripe_start = 0_u64;
3278 for stripe in &self.table.stripes {
3279 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3280 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3281 stripe_start = stripe_end;
3282 continue;
3283 }
3284 let spans = read_index(&self.file, stripe, column)?;
3285 let page = stripe.pages[column];
3286 let mut bytes = vec![0; page.length as usize];
3287 read_at(&self.file, page.offset, &mut bytes)?;
3288 let mut part_start = stripe_start;
3289 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3290 let part_end = part_start.saturating_add(u64::from(rows));
3291 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3292 let part = part_bytes(&bytes, *span)?;
3293 if checksum(part) != span.hash {
3294 return Err(invalid(
3295 "column page checksum differs while building pair frequencies",
3296 ));
3297 }
3298 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3299 let positions = ordinals[wanted..upto]
3300 .iter()
3301 .map(|&ordinal| {
3302 usize::try_from(ordinal.saturating_sub(part_start))
3303 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3304 })
3305 .collect::<Result<Vec<_>>>()?;
3306 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3307 return Ok(None);
3308 }
3309 wanted = upto;
3310 }
3311 part_start = part_end;
3312 }
3313 stripe_start = stripe_end;
3314 }
3315 if wanted != ordinals.len() {
3316 return Err(invalid("frequency ordinal is outside the table"));
3317 }
3318 Ok(Some(out))
3319 }
3320
3321 #[allow(dead_code)]
3323 fn pair_frequencies(
3324 &self,
3325 frequencies: &[Option<Frequencies>],
3326 ) -> Result<Vec<PairFrequencySummary>> {
3327 let anchors = frequencies
3328 .iter()
3329 .enumerate()
3330 .filter_map(|(column, summary)| {
3331 match summary {
3333 Some(Frequencies::Held(summary)) => Some(summary),
3334 _ => None,
3335 }
3336 .filter(|summary| {
3337 !summary.ordinals.is_empty()
3338 && summary.ordinal_entries.len() == summary.ordinals.len()
3339 })
3340 .cloned()
3341 .map(|summary| (column, summary))
3342 })
3343 .collect::<Vec<_>>();
3344 let strings = self
3345 .dictionaries
3346 .iter()
3347 .enumerate()
3348 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3349 .collect::<Vec<_>>();
3350 let mut summaries = Vec::new();
3351 for (first, anchors) in anchors {
3352 for &second in &strings {
3353 if summaries.len() == MAX_PAIR_FREQUENCIES {
3354 return Ok(summaries);
3355 }
3356 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3357 continue;
3358 };
3359 if codes.len() != anchors.ordinal_entries.len() {
3360 return Err(invalid("pair frequency columns have different lengths"));
3361 }
3362 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3363 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3364 *counts.entry((anchor, code)).or_default() += 1;
3365 }
3366 let mut entries = counts
3367 .into_iter()
3368 .map(|((first_entry, second), count)| PairFrequencyEntry {
3369 first_entry,
3370 second,
3371 count,
3372 })
3373 .collect::<Vec<_>>();
3374 entries.sort_unstable_by(|left, right| {
3375 right
3376 .count
3377 .cmp(&left.count)
3378 .then_with(|| left.first_entry.cmp(&right.first_entry))
3379 .then_with(|| left.second.cmp(&right.second))
3380 });
3381 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3382 entries.truncate(FREQUENCY_ENTRIES);
3383 summaries.push(PairFrequencySummary {
3384 first: u16::try_from(first)
3385 .map_err(|_| invalid("pair frequency column index overflows"))?,
3386 second: u16::try_from(second)
3387 .map_err(|_| invalid("pair frequency column index overflows"))?,
3388 entries,
3389 omitted_max: anchors.omitted_max.max(pair_omitted),
3390 });
3391 }
3392 }
3393 Ok(summaries)
3394 }
3395
3396 fn close(&mut self) -> Result<Entry> {
3407 self.reclaim()?;
3408 self.flush_pending()?;
3409 let profile = self.profile.clone();
3413 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3414 let before = self.at;
3415 let mut stripes = std::mem::take(&mut self.order)
3416 .into_iter()
3417 .zip(std::mem::take(&mut self.table.stripes))
3418 .collect::<Vec<_>>();
3419 stripes.sort_by_key(|(order, _)| order.0);
3420 let mut previous: Option<(u64, u64)> = None;
3421 for ((first, last), _) in &stripes {
3422 if previous.is_some_and(|previous| previous >= *first) {
3423 return Err(invalid("chunks did not arrive in source order"));
3424 }
3425 previous = Some(*last);
3426 }
3427 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3428 drop(timing);
3429 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3430 let placing = self.at;
3431 finish_dictionaries(&mut self.dictionaries)?;
3432 self.place_blocks()?;
3433 for dictionary in self.dictionaries.iter_mut().flatten() {
3434 dictionary.release_lookup();
3435 dictionary.recharge(profile.as_deref());
3436 }
3437 let (numeric, closed) = self.close_columns()?;
3438 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3439 numeric.into_iter().unzip();
3440 let frequencies =
3441 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3442 let pairs = Vec::new();
3444 self.table.frequencies = frequencies;
3445 self.table.distincts = distincts;
3446 self.table.pair_frequencies = pairs;
3447 if let Some(profile) = &profile {
3448 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3449 }
3450 self.table.demoted = self
3451 .dictionaries
3452 .iter()
3453 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3454 .collect();
3455 if !self.table.demoted.contains(&true) {
3456 self.table.demoted = Vec::new();
3457 }
3458 self.dictionaries = Vec::new();
3459 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3460 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3461 self.table.host_groups = None;
3462 for (index, closed) in closed.into_iter().enumerate() {
3463 let Some(closed) = closed else { continue };
3464 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3465 self.table.distincts[index] = distinct;
3466 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3467 self.table.frequency_texts[index] = texts;
3468 if hosts.is_some() {
3469 self.table.host_groups = hosts;
3470 }
3471 let offset = self.at;
3472 self.put(&encoded.index)?;
3473 self.put(&encoded.ranks)?;
3474 self.put(&encoded.grams)?;
3475 self.table.dictionary_payloads[index] = payload;
3476 let length = encoded
3477 .index
3478 .len()
3479 .checked_add(encoded.ranks.len())
3480 .and_then(|len| len.checked_add(encoded.grams.len()))
3481 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3482 self.table.dictionaries[index] = Some(Page {
3483 offset,
3484 length: u32::try_from(length)
3485 .map_err(|_| invalid("dictionary page length overflow"))?,
3486 hash: checksum(&encoded.index),
3487 });
3488 }
3489 drop(timing);
3490 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3491 let placed = self.at - placing;
3492 self.write_stats()?;
3493 let directory = encode_directory(&self.table)?;
3494 if directory.len() > MAX_DIRECTORY {
3495 return Err(invalid("directory exceeds the configured bound"));
3496 }
3497 let offset = self.at;
3498 self.put(&directory)?;
3499 drop(timing);
3500 if let Some(profile) = &profile {
3501 profile.moved(Stage::Dictionary, 0, placed, 0);
3502 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3503 }
3504 Ok(Entry {
3505 name: self.table.name.clone(),
3506 fields: self.table.fields.clone(),
3507 rows: self.table.rows,
3508 nonzero: vec![None; self.table.fields.len()],
3509 aggregates: table_aggregate_sums(&self.table),
3510 distincts: self.table.distincts.clone(),
3511 extremes: table_integer_extremes(&self.table),
3512 frequencies: table_complete_numeric_frequencies(&self.table),
3513 directory: Page {
3514 offset,
3515 length: u32::try_from(directory.len())
3516 .map_err(|_| invalid("directory length overflow"))?,
3517 hash: checksum(&directory),
3518 },
3519 })
3520 }
3521
3522 #[allow(clippy::type_complexity)]
3539 fn close_columns(
3540 &self,
3541 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3542 let numeric = self.numeric_columns().into_iter().map(|column| {
3543 let gather = self.gathers.get(column).and_then(Option::as_ref);
3544 let estimate = gather.and_then(stats::Gather::distinct);
3545 let counted = !estimate.is_some_and(distinct::beyond);
3546 let set =
3547 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3548 let dense = gather.filter(|_| counted).and_then(|gather| self.dense_range(gather, set));
3549 let set = dense.map_or(set, |(_, len)| len * size_of::<u32>());
3550 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3551 (Closing::Numeric { column, counted, dense }, NUMERIC_CLOSE_BYTES + set, cost)
3552 });
3553 let dictionaries =
3554 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3555 let dictionary = dictionary.as_ref()?;
3556 let bytes = dictionary.closing_bytes();
3557 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3558 });
3559 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3560 jobs.sort_by_key(|&(_, _, cost)| cost);
3561 let columns = self.table.fields.len();
3562 let mut frequencies = vec![(None, None); columns];
3563 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3564 let profile = self.profile.as_deref();
3565 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3566 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3567 match job {
3568 Closing::Numeric { column, counted, dense } => {
3569 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3570 Ok(Closed::Numeric(column, self.numeric_frequency(column, counted, dense)?))
3571 }
3572 Closing::Dictionary { index, dictionary } => {
3573 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3574 Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3575 }
3576 }
3577 };
3578 let workers = close_workers().min(jobs.len());
3579 let pieces = if workers <= 1 {
3580 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3581 } else {
3582 let state = Mutex::new((jobs, 0_usize));
3584 let finished = Condvar::new();
3585 std::thread::scope(|scope| {
3586 (0..workers)
3587 .map(|_| {
3588 scope.spawn(|| {
3589 let mut mine = Vec::new();
3590 loop {
3591 let mut held = state.lock().map_err(|_| {
3592 Error::internal("a native close worker panicked")
3593 })?;
3594 let (job, bytes) = loop {
3595 let (jobs, busy) = &mut *held;
3596 if jobs.is_empty() {
3597 return Ok(mine);
3598 }
3599 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3600 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3601 });
3602 if let Some(at) = fits {
3603 let (job, bytes, _) = jobs.remove(at);
3604 *busy += bytes;
3605 break (job, bytes);
3606 }
3607 held = finished.wait(held).map_err(|_| {
3608 Error::internal("a native close worker panicked")
3609 })?;
3610 };
3611 drop(held);
3612 let _room = Room { state: &state, finished: &finished, bytes };
3615 mine.push(run(job, bytes)?);
3616 }
3617 })
3618 })
3619 .collect::<Vec<_>>()
3620 .into_iter()
3621 .map(|handle| {
3622 handle
3623 .join()
3624 .map_err(|_| Error::internal("a native close worker panicked"))?
3625 })
3626 .collect::<Result<Vec<_>>>()
3627 })?
3628 .into_iter()
3629 .flatten()
3630 .collect()
3631 };
3632 for piece in pieces {
3633 match piece {
3634 Closed::Numeric(column, summary) => frequencies[column] = summary,
3635 Closed::Dictionary(index, one) => closed[index] = Some(one),
3636 }
3637 }
3638 Ok((frequencies, closed))
3639 }
3640
3641 fn close_dictionary(
3648 &self,
3649 _index: usize,
3650 dictionary: &GlobalDictionary,
3651 ) -> Result<ClosedDictionary> {
3652 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3653 let (distinct, frequencies, texts) = if dictionary.demoted {
3658 (None, None, Vec::new())
3659 } else {
3660 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3661 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3662 (Some(distinct), Some(frequencies), texts)
3663 };
3664 let hosts = None;
3666 drop(flat);
3667 drop(bases);
3668 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3669 let payload = dictionary
3670 .placed
3671 .iter()
3672 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3673 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3674 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3675 }
3676
3677 fn write_stats(&mut self) -> Result<()> {
3689 let gathers = std::mem::take(&mut self.gathers);
3690 let rows = self.table.rows as u64;
3691 let mut payloads = Vec::new();
3692 for (column, gather) in gathers.into_iter().enumerate() {
3693 let Some(gather) = gather else { continue };
3694 if gather.rows() != rows {
3700 continue;
3701 }
3702 let Some(stats) = gather.finish() else { continue };
3703 let mut summary = Vec::new();
3704 stats.summary.encode(&mut summary)?;
3705 let mut sketches = Vec::new();
3706 stats.sketches.encode(&mut sketches)?;
3707 payloads.push((column, summary, sketches));
3708 }
3709 if payloads.is_empty() {
3710 return Ok(());
3711 }
3712 let summaries = payloads.iter().map(|(_, summary, _)| summary.len()).collect::<Vec<_>>();
3713 let sketches = payloads.iter().map(|(_, _, sketches)| sketches.len()).collect::<Vec<_>>();
3714 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3715 let keep = stats::kept(&summaries, &sketches, allowance, 0);
3718 for ((column, summary, sketches), &(built, sketched)) in payloads.iter().zip(&keep) {
3719 if !built {
3720 continue;
3721 }
3722 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3723 let sections = [
3724 (*section::SUMMARY, summary, summary.len() as u32),
3727 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3728 ];
3729 let wanted = 1 + usize::from(sketched);
3730 for (kind, bytes, header_bytes) in sections.into_iter().take(wanted) {
3731 let written = write_section(
3732 &*self.file,
3733 &mut self.at,
3734 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3735 self.generation,
3736 )?;
3737 self.table.sections.push(written);
3738 }
3739 }
3740 if self.table.sections.len() > MAX_SECTIONS {
3741 return Err(invalid("the table would name more sections than the bound allows"));
3742 }
3743 Ok(())
3744 }
3745
3746 pub fn finish(mut self) -> Result<Table> {
3756 let entry = self.close()?;
3757 let profile = self.profile.take();
3758 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3759 let mut tables = std::mem::take(&mut self.closed);
3760 tables.push(entry);
3761 let catalog = encode_catalog(&tables, &self.views)?;
3762 if catalog.len() > MAX_DIRECTORY {
3763 return Err(invalid("catalog exceeds the configured bound"));
3764 }
3765 let offset = self.at;
3766 self.put(&catalog)?;
3767 if let Some(profile) = &profile {
3768 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3769 }
3770 synced(&*self.file, profile.as_deref())?;
3774 let slot = Slot {
3775 offset,
3776 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3777 generation: self.generation,
3778 hash: checksum(&catalog),
3779 };
3780 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3785 synced(&*self.file, profile.as_deref())?;
3786 Ok(self.table)
3787 }
3788
3789 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3806 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3807 let size = file.len()?;
3808 let (slot, bytes, _) = committed_slot(&*file, size)?;
3809 let (closed, _) = decode_catalog(&bytes, size)?;
3810 let generation = slot
3811 .generation
3812 .checked_add(1)
3813 .ok_or_else(|| invalid("native file generation overflow"))?;
3814 let catalog = encode_catalog(&closed, views)?;
3815 if catalog.len() > MAX_DIRECTORY {
3816 return Err(invalid("catalog exceeds the configured bound"));
3817 }
3818 file.write_at(size, &catalog)?;
3819 file.sync()?;
3820 let slot = Slot {
3821 offset: size,
3822 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3823 generation,
3824 hash: checksum(&catalog),
3825 };
3826 file.write_at(slot_offset(generation), &slot.bytes())?;
3827 file.sync()?;
3828 Ok(())
3829 }
3830
3831 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3834 let path = path.as_ref();
3835 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3836 let (mut entries, views) = decode_catalog(&bytes, size)?;
3837 let native = Catalog::open(path)?;
3838 for entry in &mut entries {
3839 let reader = native.table(&entry.name)?;
3840 entry.nonzero.fill(None);
3841 entry.aggregates = reader_aggregate_sums(&reader)?;
3842 entry.distincts = (0..entry.fields.len())
3843 .map(|column| reader.distinct_values(column))
3844 .collect::<Result<Vec<_>>>()?;
3845 entry.extremes = reader_integer_extremes(&reader)?;
3846 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3847 }
3848 let generation = slot
3849 .generation
3850 .checked_add(1)
3851 .ok_or_else(|| invalid("native file generation overflow"))?;
3852 let catalog = encode_catalog(&entries, &views)?;
3853 if catalog.len() > MAX_DIRECTORY {
3854 return Err(invalid("catalog exceeds the configured bound"));
3855 }
3856 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3857 file.write_at(size, &catalog)?;
3858 file.sync()?;
3859 let slot = Slot {
3860 offset: size,
3861 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3862 generation,
3863 hash: checksum(&catalog),
3864 };
3865 file.write_at(slot_offset(generation), &slot.bytes())?;
3866 file.sync()?;
3867 Ok(())
3868 }
3869
3870 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3872 Self::certify_summaries(path)
3873 }
3874}
3875
3876fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3882 let offset = *at;
3883 file.write_at(offset, bytes)?;
3884 *at =
3885 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3886 Ok(offset)
3887}
3888
3889fn write_section(
3895 file: &dyn rudb_io::File,
3896 at: &mut u64,
3897 one: §ion::Attachment<'_>,
3898 generation: u64,
3899) -> Result<Section> {
3900 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3904 return Err(invalid("a section's header is longer than its payload"));
3905 }
3906 let mut extents = Vec::new();
3907 let mut first = 0_u64;
3908 let extent_size =
3909 if one.kind == *section::SORTED_PROJECTION || one.kind == *section::RUN_PROJECTION {
3910 1 << 19
3911 } else {
3912 section::MAX_EXTENT as usize
3913 };
3914 for chunk in one.bytes.chunks(extent_size) {
3915 let offset = append(file, at, chunk)?;
3916 extents.push(section::Extent {
3917 offset,
3918 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3919 hash: checksum(chunk),
3920 first,
3921 });
3922 first += chunk.len() as u64;
3923 }
3924 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3925 section::encode_extents(&extents, &mut table)?;
3926 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3930 Ok(Section {
3931 kind: one.kind,
3932 id: one.id,
3933 generation,
3934 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3935 extent_page,
3936 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3937 hash: checksum(&table),
3938 flags: one.flags,
3939 header_bytes: one.header_bytes,
3940 })
3941}
3942
3943pub fn attach(
3967 path: impl AsRef<Path>,
3968 table: &str,
3969 attachments: &[section::Attachment<'_>],
3970) -> Result<Table> {
3971 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3972 let file = &*file;
3973 let size = file.len()?;
3974 let (slot, bytes, _) = committed_slot(file, size)?;
3975 let (mut entries, views) = decode_catalog(&bytes, size)?;
3976 let at = entries
3977 .iter()
3978 .position(|entry| entry.name == table)
3979 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3980 let mut version = [0; 4];
3981 read_at(file, 8, &mut version)?;
3982 let version = u32::from_le_bytes(version);
3983 if version != FORMAT {
3989 return Err(invalid(&format!(
3990 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3991 to be written again"
3992 )));
3993 }
3994 let mut directory = vec![0; entries[at].directory.length as usize];
3995 read_at(file, entries[at].directory.offset, &mut directory)?;
3996 if checksum(&directory) != entries[at].directory.hash {
3997 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3998 }
3999 let mut held = decode_directory(&directory, size)?;
4000 let mut cursor = size;
4001 for one in attachments {
4002 let written = write_section(file, &mut cursor, one, held.generation)?;
4003 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
4004 held.sections.push(written);
4005 }
4006 if held.sections.len() > MAX_SECTIONS {
4007 return Err(invalid("the table would name more sections than the bound allows"));
4008 }
4009 let encoded = encode_directory(&held)?;
4010 if encoded.len() > MAX_DIRECTORY {
4011 return Err(invalid("directory exceeds the configured bound"));
4012 }
4013 let offset = append(file, &mut cursor, &encoded)?;
4014 entries[at].directory = Page {
4015 offset,
4016 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
4017 hash: checksum(&encoded),
4018 };
4019 let catalog = encode_catalog(&entries, &views)?;
4022 if catalog.len() > MAX_DIRECTORY {
4023 return Err(invalid("catalog exceeds the configured bound"));
4024 }
4025 let offset = append(file, &mut cursor, &catalog)?;
4026 file.sync()?;
4027 let generation =
4028 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
4029 let committed = Slot {
4030 offset,
4031 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
4032 generation,
4033 hash: checksum(&catalog),
4034 };
4035 file.write_at(slot_offset(generation), &committed.bytes())?;
4036 file.sync()?;
4037 Ok(held)
4038}
4039
4040type Synopsis = Arc<Vec<(Value, u64)>>;
4043
4044#[derive(Debug, Clone)]
4046pub struct Reader {
4047 file: Arc<File>,
4048 table: Arc<Table>,
4049 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
4050 loading: Arc<Vec<Mutex<()>>>,
4059 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
4062 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
4066 opened: Arc<AtomicUsize>,
4070 sieves: Arc<Vec<Vec<SieveSlot>>>,
4074 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
4077 places: Arc<Vec<Place>>,
4079 cache: Arc<Shelf>,
4080 pool: PagePool,
4082 pages: Arc<AtomicUsize>,
4085 indexes: Arc<AtomicUsize>,
4088 size: u64,
4090 directory: u64,
4092 opening: Opening,
4094}
4095
4096#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4108pub struct Opening {
4109 pub reads: u32,
4112 pub bytes: u64,
4114}
4115
4116#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
4118pub struct Reads {
4119 pub opening: Opening,
4121 pub pages: usize,
4123 pub indexes: usize,
4125 pub dictionaries: usize,
4128}
4129
4130#[derive(Debug, Clone, Copy)]
4132struct Place {
4133 stripe: u32,
4134 part: u32,
4135 rows: u32,
4136}
4137
4138#[derive(Debug, Clone, Copy)]
4140struct PartSpan {
4141 start: usize,
4142 length: usize,
4143 hash: u64,
4144}
4145
4146#[derive(Debug, Clone)]
4152struct CachedColumn {
4153 stripe: usize,
4154 index: Arc<Vec<PartSpan>>,
4155 page: Option<Arc<HeldPage>>,
4156}
4157
4158#[derive(Debug)]
4165struct HeldPage {
4166 bytes: Vec<u8>,
4167 checked: Vec<AtomicBool>,
4168}
4169
4170impl HeldPage {
4171 fn part(&self, part: usize, span: PartSpan) -> Result<&[u8]> {
4173 let bytes = part_bytes(&self.bytes, span)?;
4174 let checked = self.checked.get(part).ok_or_else(|| invalid("part index out of range"))?;
4175 if !checked.load(Atomic::Relaxed) {
4176 verify_part(bytes, span)?;
4177 checked.store(true, Atomic::Relaxed);
4178 }
4179 Ok(bytes)
4180 }
4181}
4182
4183fn verify_part(bytes: &[u8], span: PartSpan) -> Result<()> {
4185 let got = checksum(bytes);
4186 if got != span.hash {
4187 return Err(invalid(&format!(
4188 "column page checksum differs, part at {}+{} bytes, wanted {:016x} and got {got:016x}",
4189 span.start, span.length, span.hash,
4190 )));
4191 }
4192 Ok(())
4193}
4194
4195#[derive(Debug, Default)]
4224struct Cached {
4225 pages: Vec<Option<Resident>>,
4226 loading: Vec<usize>,
4227 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4228 seen: Vec<bool>,
4229 passing: VecDeque<usize>,
4230}
4231
4232#[derive(Debug, Clone)]
4234struct Resident {
4235 page: Arc<HeldPage>,
4236 used: Arc<AtomicBool>,
4237}
4238
4239#[derive(Debug)]
4241struct Shelf {
4242 columns: Vec<Mutex<Cached>>,
4243 held: Vec<AtomicUsize>,
4246 kept: AtomicUsize,
4249}
4250
4251#[derive(Debug, Clone, Default)]
4270pub struct PagePool {
4271 ring: Arc<Mutex<Ring>>,
4272 budget: Arc<AtomicUsize>,
4273}
4274
4275#[derive(Debug, Default)]
4276struct Ring {
4277 held: VecDeque<Held>,
4278 bytes: usize,
4279}
4280
4281#[derive(Debug)]
4286struct Held {
4287 shelf: Weak<Shelf>,
4288 column: usize,
4289 stripe: usize,
4290 bytes: usize,
4291 used: Arc<AtomicBool>,
4292}
4293
4294impl PagePool {
4295 #[must_use]
4297 pub fn new(budget: usize) -> Self {
4298 let pool = Self::default();
4299 pool.budget.store(budget, Atomic::Relaxed);
4300 pool
4301 }
4302
4303 #[must_use]
4309 pub fn bytes(&self) -> usize {
4310 self.ring.lock().map_or(0, |ring| ring.bytes)
4311 }
4312
4313 fn admit(&self, held: Held) {
4319 let budget = self.budget.load(Atomic::Relaxed);
4320 let mut gone = Vec::new();
4321 {
4322 let Ok(mut ring) = self.ring.lock() else { return };
4323 ring.bytes += held.bytes;
4324 ring.held.push_back(held);
4325 let mut looked = 0;
4328 let limit = ring.held.len();
4329 while ring.bytes > budget && looked < limit {
4330 looked += 1;
4331 let Some(entry) = ring.held.pop_front() else { break };
4332 let Some(shelf) = entry.shelf.upgrade() else {
4333 ring.bytes -= entry.bytes;
4334 continue;
4335 };
4336 if entry.used.swap(false, Atomic::Relaxed) {
4337 ring.held.push_back(entry);
4338 continue;
4339 }
4340 let count = &shelf.held[entry.column];
4341 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4342 ring.held.push_back(entry);
4343 continue;
4344 }
4345 count.fetch_sub(1, Atomic::Relaxed);
4346 ring.bytes -= entry.bytes;
4347 gone.push((shelf, entry));
4348 }
4349 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4352 if let Some(entry) = ring.held.pop_front() {
4353 ring.bytes -= entry.bytes;
4354 }
4355 }
4356 }
4357 for (shelf, entry) in gone {
4358 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4359 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4360 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4361 *slot = None;
4362 }
4363 }
4364 }
4365 }
4366}
4367
4368const CACHED_STRIPES_PER_COLUMN: usize = 4;
4380
4381type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4383
4384type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4385
4386#[derive(Debug)]
4387struct NativeText {
4388 file: Arc<File>,
4389 values: usize,
4391 offsets: Vec<u8>,
4403 offset_bits: usize,
4406 value_ends: OnceLock<Option<Vec<u32>>>,
4419 value_lens: OnceLock<Option<Lengths>>,
4429 ends_asked: AtomicUsize,
4435 ranks: usize,
4437 rank_at: u64,
4441 rank_ends: Vec<u64>,
4445 rank_hashes: Vec<u64>,
4446 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4447 code_bits: usize,
4450 code_ranks: OnceLock<Option<Vec<u32>>>,
4457 starts: Vec<u64>,
4464 lengths: Vec<u64>,
4465 hashes: Vec<u64>,
4466 grams: Option<NativeGrams>,
4468 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4470 char_lens: Vec<OnceLock<Box<[u32]>>>,
4479 keep_budget: usize,
4482 payload_kept: AtomicUsize,
4490 swept: Vec<AtomicBool>,
4498 visit_dropped: AtomicUsize,
4513 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4530}
4531
4532#[derive(Debug)]
4533struct NativeGrams {
4534 start: u64,
4535 length: usize,
4536 width: usize,
4538 hash: u64,
4539 verdicts: Mutex<Vec<Verdict>>,
4546}
4547
4548type Verdict = (Vec<u8>, Arc<[bool]>);
4550
4551const GRAM_VERDICTS: usize = 8;
4553
4554impl NativeGrams {
4555 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4560 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4561 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4562 return Ok(Arc::clone(verdict));
4563 }
4564 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4565 let mut verdict = Vec::with_capacity(self.length / self.width);
4566 let window = GRAM_WINDOW / self.width * self.width;
4567 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4568 verdict.extend(bytes.chunks(self.width).map(|bits| {
4569 wanted
4570 .iter()
4571 .flatten()
4572 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4573 }));
4574 Ok(())
4575 })?;
4576 if hash != self.hash {
4577 return Err(invalid("global dictionary substring signatures checksum differs"));
4578 }
4579 let verdict: Arc<[bool]> = verdict.into();
4580 if held.len() >= GRAM_VERDICTS {
4581 held.remove(0);
4582 }
4583 held.push((literal.to_vec(), Arc::clone(&verdict)));
4584 Ok(verdict)
4585 }
4586
4587 fn footprint(&self) -> usize {
4588 self.verdicts.lock().map_or(0, |held| {
4589 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4590 })
4591 }
4592}
4593
4594const TEXT_SEARCH_MEMO: usize = 64;
4599
4600const TEXT_PAYLOAD_VALUES: usize = 1024;
4616
4617const TEXT_GRAM_BYTES: usize = 8192;
4628
4629const NARROW_GRAM_BYTES: usize = 2048;
4631
4632const GRAM_WINDOW: usize = 256 << 10;
4634
4635fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4638 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4639 let mut first = original ^ (original >> 16);
4640 first = first.wrapping_mul(0x7feb_352d);
4641 first ^= first >> 15;
4642 let mut second = original ^ (original >> 17);
4643 second = second.wrapping_mul(0x846c_a68b);
4644 second ^= second >> 16;
4645 let mask = width * 8 - 1;
4646 [(first as usize) & mask, (second as usize) & mask]
4647}
4648
4649const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4670
4671#[derive(Debug)]
4678enum Lengths {
4679 Narrow(Vec<u16>),
4681 Wide(Vec<u32>),
4683}
4684
4685impl Lengths {
4686 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4689 match self {
4690 Lengths::Narrow(lens) => into.extend(
4691 indices
4692 .iter()
4693 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4694 ),
4695 Lengths::Wide(lens) => into.extend(
4696 indices
4697 .iter()
4698 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4699 ),
4700 }
4701 }
4702
4703 fn footprint(&self) -> usize {
4705 match self {
4706 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4707 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4708 }
4709 }
4710}
4711
4712fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4721 match lengths_as::<u16>(ends)? {
4722 Some(narrow) => Some(Lengths::Narrow(narrow)),
4723 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4724 }
4725}
4726
4727fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4730 let mut lens = Vec::with_capacity(ends.len());
4731 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4732 let mut start = 0;
4733 for &end in block {
4734 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4735 return Some(None);
4736 };
4737 lens.push(len);
4738 start = end;
4739 }
4740 }
4741 Some(Some(lens))
4742}
4743
4744const TEXT_OFFSET_RUN: usize = 512;
4751
4752const DICTIONARY_HEADER: usize = 16;
4755
4756const DICTIONARY_SCATTERED: u32 = 1 << 31;
4770const DICTIONARY_GRAMS: u32 = 1 << 30;
4772const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4775const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4777
4778const TEXT_RANK_BLOCK: usize = 512;
4789
4790const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4804
4805impl NativeText {
4806 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4813 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4814 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4815 Ok(Some(bytes.as_slice()))
4816 }
4817
4818 fn block_chars(&self, block: usize) -> Result<&[u32]> {
4825 let slot = self
4826 .char_lens
4827 .get(block)
4828 .ok_or_else(|| invalid("a block past the global dictionary"))?;
4829 if let Some(lens) = slot.get() {
4830 return Ok(lens);
4831 }
4832 let decoded;
4833 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4834 Some(Ok(kept)) => kept,
4835 _ => {
4836 decoded = self.decode_block(block)?;
4837 &decoded
4838 }
4839 };
4840 let first = block * TEXT_PAYLOAD_VALUES;
4841 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4842 let ends = self.ends_within(first, last)?;
4843 if ends.len() != last - first {
4844 return Err(invalid("global dictionary offsets are short"));
4845 }
4846 let mut lens = Vec::with_capacity(ends.len());
4847 let mut start = u64::from(self.start_within(first)?);
4848 for &end in &ends {
4849 let value = usize::try_from(start)
4850 .ok()
4851 .zip(usize::try_from(end).ok())
4852 .and_then(|(from, to)| bytes.get(from..to))
4853 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4854 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4857 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4858 start = end;
4859 }
4860 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4861 }
4862
4863 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4868 let len = self.lengths[block];
4869 let mut stored = vec![
4870 0;
4871 usize::try_from(len).map_err(|_| invalid(
4872 "global dictionary block does not fit in memory"
4873 ))?
4874 ];
4875 read_at(&self.file, self.starts[block], &mut stored)?;
4876 if checksum(&stored) != self.hashes[block] {
4877 return Err(invalid("global dictionary payload checksum differs"));
4878 }
4879 let first = block * TEXT_PAYLOAD_VALUES;
4880 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4881 let want = self.end_within(last - 1)? as usize;
4882 let values = string::decode_flat(&stored)?;
4883 if values.len() != last - first {
4884 return Err(invalid("global dictionary block holds the wrong value count"));
4885 }
4886 let bytes = values.into_bytes();
4887 if bytes.len() != want {
4888 return Err(invalid("global dictionary block decodes to the wrong length"));
4889 }
4890 Ok(bytes)
4891 }
4892
4893 fn loaned_block<'a>(
4902 &'a self,
4903 block: usize,
4904 decoded: &'a mut Vec<u8>,
4905 scattered: bool,
4906 ) -> Result<&'a [u8]> {
4907 let kept = self.blocks.get(block).and_then(OnceLock::get);
4908 if let Some(Ok(kept)) = kept {
4909 return Ok(kept);
4910 }
4911 let again = kept.is_none()
4912 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4913 let keep = again
4914 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4915 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4916 if keep {
4917 let kept = self
4918 .payload_block(block)?
4919 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4920 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4921 return Ok(kept);
4922 }
4923 *decoded = self.decode_block(block)?;
4924 if scattered && again {
4925 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4926 }
4927 Ok(decoded)
4928 }
4929
4930 fn ends_worth_unpacking(&self) -> usize {
4947 self.values.max(TEXT_PAYLOAD_VALUES)
4948 }
4949
4950 fn value_ends(&self) -> Option<&[u32]> {
4952 if let Some(built) = self.value_ends.get() {
4953 return built.as_deref();
4954 }
4955 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4956 return None;
4957 }
4958 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4959 }
4960
4961 fn unpack_ends(&self) -> Option<Vec<u32>> {
4967 let mut ends = vec![0u32; self.values];
4968 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4969 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4970 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4971 u32::try_from(bits).unwrap_or(u32::MAX)
4972 })
4973 .ok()?;
4974 }
4975 if ends.contains(&u32::MAX) { None } else { Some(ends) }
4978 }
4979
4980 fn packed(&self) -> &[u8] {
4982 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4983 }
4984
4985 fn end_within(&self, index: usize) -> Result<u32> {
4987 if let Some(ends) = self.value_ends() {
4988 return ends
4989 .get(index)
4990 .copied()
4991 .ok_or_else(|| invalid("global dictionary offsets are short"));
4992 }
4993 let run = index / TEXT_OFFSET_RUN;
4994 let bytes = self
4995 .packed()
4996 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4997 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4998 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4999 .map_err(|_| invalid("global dictionary offsets are short"))?;
5000 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
5001 }
5002
5003 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
5021 let mut ends = vec![0u64; last.saturating_sub(first)];
5022 let mut scratch = Vec::new();
5023 let mut at = first;
5024 while at < last {
5025 let run = at / TEXT_OFFSET_RUN;
5026 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
5027 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
5028 let bytes = self
5029 .packed()
5030 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5031 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5032 let from = at % TEXT_OFFSET_RUN;
5033 let upto = stop - run * TEXT_OFFSET_RUN;
5034 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
5035 return Err(invalid("global dictionary offsets are short"));
5036 }
5037 let into = &mut ends[at - first..stop - first];
5038 if from == 0 {
5039 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
5040 .map_err(|_| invalid("global dictionary offsets are short"))?;
5041 } else {
5042 scratch.resize(held, 0);
5043 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
5044 .map_err(|_| invalid("global dictionary offsets are short"))?;
5045 into.copy_from_slice(&scratch[from..upto]);
5046 }
5047 at = stop;
5048 }
5049 Ok(ends)
5050 }
5051
5052 fn start_within(&self, index: usize) -> Result<u32> {
5055 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
5056 }
5057
5058 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
5066 if let Some(ends) = self.value_ends() {
5067 let end =
5068 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
5069 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5072 if start > end {
5073 return Err(invalid("global dictionary value ends before it starts"));
5074 }
5075 return Ok((start, end));
5076 }
5077 let within = index % TEXT_OFFSET_RUN;
5078 let (start, end) = if within == 0 {
5079 (self.start_within(index)?, self.end_within(index)?)
5080 } else {
5081 let run = index / TEXT_OFFSET_RUN;
5082 let bytes = self
5083 .packed()
5084 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
5085 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
5086 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
5087 .map_err(|_| invalid("global dictionary offsets are short"))?;
5088 let ends = u32::try_from(end)
5089 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5090 let starts = u32::try_from(start)
5091 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
5092 (starts, ends)
5093 };
5094 if start > end {
5095 return Err(invalid("global dictionary value ends before it starts"));
5096 }
5097 Ok((start, end))
5098 }
5099
5100 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
5107 let slot = self
5108 .rank_blocks
5109 .get(rank / TEXT_RANK_BLOCK)
5110 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
5111 let block = slot
5112 .get_or_init(|| {
5113 let mut bytes = Vec::new();
5114 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
5115 Ok(bytes)
5116 })
5117 .as_ref()
5118 .map_err(Clone::clone)?;
5119 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
5120 }
5121
5122 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
5125 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
5126 let end = self.rank_ends[which];
5127 bytes.clear();
5128 bytes.resize((end - start) as usize, 0);
5129 read_at(&self.file, self.rank_at + start, bytes)?;
5130 let expected = self
5131 .rank_hashes
5132 .get(which)
5133 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
5134 if checksum(bytes) != *expected {
5135 return Err(invalid("global dictionary rank checksum differs"));
5136 }
5137 Ok(())
5138 }
5139
5140 fn head_at(&self, rank: usize) -> Result<u64> {
5142 let (block, within) = self.rank_parts(rank)?;
5143 let (base, width, packed) = rank_heads(block)?;
5144 let above = bitpack::tail_at(packed, width, within)
5145 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
5146 Ok(base.wrapping_add(above))
5147 }
5148
5149 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
5151 let (_, width, packed) = rank_heads(block)?;
5152 packed
5153 .get(bitpack::tail_len(count, width)..)
5154 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
5155 }
5156
5157 fn rank_block_len(&self, rank: usize) -> usize {
5159 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
5160 TEXT_RANK_BLOCK.min(self.ranks - first)
5161 }
5162}
5163
5164fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
5166 let header = block
5167 .get(..RANK_BLOCK_HEADER)
5168 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
5169 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
5170 let width = header[8] as usize;
5171 if width > 64 {
5172 return Err(invalid("global dictionary rank block packs heads past a word"));
5173 }
5174 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5175}
5176
5177fn offset_width(ends: &[u32]) -> usize {
5184 let span = ends.iter().copied().max().unwrap_or(0);
5188 (u32::BITS - span.leading_zeros()) as usize
5189}
5190
5191fn offset_bytes(values: usize, bits: usize) -> usize {
5194 let full = values / TEXT_OFFSET_RUN;
5195 let rest = values % TEXT_OFFSET_RUN;
5196 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5197}
5198
5199fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5203 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5204 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5205 run.clear();
5206 run.extend(chunk.iter().map(|&end| u64::from(end)));
5207 bitpack::pack_tail(&run, bits, out)
5208 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5209 }
5210 Ok(())
5211}
5212
5213fn code_width(values: usize) -> usize {
5215 match u64::try_from(values).unwrap_or(u64::MAX) {
5216 0 | 1 => 0,
5217 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5218 }
5219}
5220
5221impl TextSource for NativeText {
5222 fn len(&self) -> usize {
5223 self.values
5224 }
5225
5226 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5227 let Some(grams) = &self.grams else { return Ok(true) };
5228 if literal.len() < 4 || first >= self.values {
5229 return Ok(true);
5230 }
5231 let verdict = grams.verdicts(&self.file, literal)?;
5232 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5233 }
5234
5235 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5236 if index >= self.values {
5237 return Ok(None);
5238 }
5239 let (start, end) = self.span_within(index)?;
5240 if start == end {
5241 return Ok(Some(&[]));
5242 }
5243 let block = index / TEXT_PAYLOAD_VALUES;
5246 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5247 Ok(bytes.get(start as usize..end as usize))
5248 }
5249
5250 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5251 if index >= self.values {
5252 return Ok(None);
5253 }
5254 let (start, end) = self.span_within(index)?;
5255 Ok(Some((end - start) as usize))
5256 }
5257
5258 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5265 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5266 into.reserve(indices.len());
5267 let Some(ends) = self.value_ends() else {
5268 for &index in indices {
5269 into.push(
5270 self.bytes_len_at(index as usize)?
5271 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5272 );
5273 }
5274 return Ok(());
5275 };
5276 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5277 lens.extend_at(indices, into);
5278 return Ok(());
5279 }
5280 for &index in indices {
5281 let index = index as usize;
5282 let Some(&end) = ends.get(index) else {
5284 into.push(0);
5285 continue;
5286 };
5287 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5288 if start > end {
5289 return Err(invalid("global dictionary value ends before it starts"));
5290 }
5291 into.push(i64::from(end - start));
5292 }
5293 Ok(())
5294 }
5295
5296 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5299 into.reserve(indices.len());
5300 for &index in indices {
5301 let index = index as usize;
5302 if index >= self.values {
5304 into.push(0);
5305 continue;
5306 }
5307 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5308 let len = lens
5309 .get(index % TEXT_PAYLOAD_VALUES)
5310 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5311 into.push(i64::from(*len));
5312 }
5313 Ok(())
5314 }
5315
5316 fn sweep(
5329 &self,
5330 first: usize,
5331 limit: usize,
5332 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5333 ) -> Result<usize> {
5334 let limit = limit.min(self.values);
5335 if first >= limit {
5336 return Ok(first);
5337 }
5338 let block = first / TEXT_PAYLOAD_VALUES;
5339 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5340 let mut decoded = Vec::new();
5341 let bytes = self.loaned_block(block, &mut decoded, false)?;
5342 let ends = self.ends_within(first, last)?;
5343 if ends.len() != last - first {
5344 return Err(invalid("global dictionary offsets are short"));
5345 }
5346 let mut start = u64::from(self.start_within(first)?);
5347 for (index, &end) in (first..last).zip(&ends) {
5350 let value = usize::try_from(start)
5351 .ok()
5352 .zip(usize::try_from(end).ok())
5353 .and_then(|(from, to)| bytes.get(from..to))
5354 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5355 body(index, value)?;
5356 start = end;
5357 }
5358 Ok(last)
5359 }
5360
5361 fn visit_at(
5370 &self,
5371 indices: &[u32],
5372 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5373 ) -> Result<()> {
5374 let mut order = (0..indices.len()).collect::<Vec<_>>();
5375 order.sort_unstable_by_key(|&at| indices[at]);
5376 let block_of = |at: usize| {
5377 let index = indices[at] as usize;
5378 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5379 };
5380 let mut decoded = Vec::new();
5381 let mut run = 0;
5382 while run < order.len() {
5383 let Some(block) = block_of(order[run]) else {
5384 for &at in &order[run..] {
5386 body(at, &[])?;
5387 }
5388 break;
5389 };
5390 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5391 let bytes = self.loaned_block(block, &mut decoded, true)?;
5392 for &at in &order[run..upto] {
5393 let (start, end) = self.span_within(indices[at] as usize)?;
5394 let value = bytes
5395 .get(start as usize..end as usize)
5396 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5397 body(at, value)?;
5398 }
5399 run = upto;
5400 }
5401 Ok(())
5402 }
5403
5404 fn visit(
5410 &self,
5411 indices: &[usize],
5412 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5413 ) -> Result<()> {
5414 let mut at = 0;
5415 while at < indices.len() {
5416 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5417 let upto =
5418 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5419 let wanted = &indices[at..upto];
5420 if wanted.iter().any(|&index| index >= self.values) {
5421 return Err(invalid("a visited value is past the global dictionary"));
5422 }
5423 let decoded;
5424 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5425 Some(Ok(kept)) => kept,
5426 _ => {
5427 decoded = self.decode_block(block)?;
5428 &decoded
5429 }
5430 };
5431 for (offset, &index) in wanted.iter().enumerate() {
5432 let (start, end) = self.span_within(index)?;
5433 let value = bytes
5434 .get(start as usize..end as usize)
5435 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5436 body(at + offset, value)?;
5437 }
5438 at = upto;
5439 }
5440 Ok(())
5441 }
5442
5443 fn ranks(&self) -> Option<usize> {
5444 (self.ranks > 0).then_some(self.ranks)
5445 }
5446
5447 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5455 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5456 if let Some(&answer) = memo.get(wanted) {
5457 return Ok(answer);
5458 }
5459 let answer = search_below(self, ranks, wanted)?;
5460 if memo.len() >= TEXT_SEARCH_MEMO {
5461 memo.clear();
5462 }
5463 memo.insert(wanted.to_vec(), answer);
5464 Ok(answer)
5465 }
5466
5467 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5468 let settled = self.head_at(rank)?.cmp(&head(wanted));
5472 if settled != Ordering::Equal {
5473 return Ok(settled);
5474 }
5475 let code = self.code_at_rank(rank)?;
5476 let bytes = self
5477 .bytes_at(code as usize)?
5478 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5479 Ok(bytes.cmp(wanted))
5480 }
5481
5482 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5483 let (block, within) = self.rank_parts(rank)?;
5484 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5485 let code = bitpack::tail_at(codes, self.code_bits, within)
5486 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5487 let code = u32::try_from(code)
5488 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5489 if code as usize >= self.len() {
5490 return Err(invalid("global dictionary order names a code it does not have"));
5491 }
5492 Ok(code)
5493 }
5494
5495 fn code_ranks(&self) -> Option<&[u32]> {
5496 if self.ranks == 0 || self.ranks != self.len() {
5500 return None;
5501 }
5502 self.code_ranks
5503 .get_or_init(|| {
5504 let mut ranks = vec![u32::MAX; self.ranks];
5505 let mut scratch = Vec::new();
5513 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5514 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5515 let which = first / TEXT_RANK_BLOCK;
5516 let block = match self.rank_blocks.get(which)?.get() {
5517 Some(kept) => kept.as_ref().ok()?.as_slice(),
5518 None => {
5519 self.read_rank_block(which, &mut scratch).ok()?;
5520 scratch.as_slice()
5521 }
5522 };
5523 let count = self.rank_block_len(first);
5524 let packed = self.rank_codes(block, count).ok()?;
5525 let codes = codes.get_mut(..count)?;
5526 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5527 for (within, &code) in codes.iter().enumerate() {
5528 let code = usize::try_from(code).ok()?;
5529 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5530 }
5531 }
5532 if ranks.contains(&u32::MAX) {
5533 return None;
5534 }
5535 Some(ranks)
5536 })
5537 .as_deref()
5538 }
5539
5540 fn footprint(&self) -> usize {
5541 self.offsets.capacity()
5542 + self
5543 .value_ends
5544 .get()
5545 .and_then(Option::as_ref)
5546 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5547 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5548 + self
5549 .code_ranks
5550 .get()
5551 .and_then(Option::as_ref)
5552 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5553 + self.rank_hashes.capacity() * size_of::<u64>()
5554 + self.rank_ends.capacity() * size_of::<u64>()
5555 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5556 + self
5557 .rank_blocks
5558 .iter()
5559 .filter_map(OnceLock::get)
5560 .filter_map(|result| result.as_ref().ok())
5561 .map(Vec::capacity)
5562 .sum::<usize>()
5563 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5564 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5565 + self
5566 .char_lens
5567 .iter()
5568 .filter_map(OnceLock::get)
5569 .map(|lens| lens.len() * size_of::<u32>())
5570 .sum::<usize>()
5571 + self.hashes.capacity() * size_of::<u64>()
5572 + self.starts.capacity() * size_of::<u64>()
5573 + self.lengths.capacity() * size_of::<u64>()
5574 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5575 + self
5576 .blocks
5577 .iter()
5578 .filter_map(OnceLock::get)
5579 .filter_map(|result| result.as_ref().ok())
5580 .map(Vec::capacity)
5581 .sum::<usize>()
5582 }
5583}
5584
5585fn places(table: &Table) -> Result<Vec<Place>> {
5587 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5588 for (at, stripe) in table.stripes.iter().enumerate() {
5589 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5590 for (part, &rows) in stripe.parts.iter().enumerate() {
5591 places.push(Place {
5592 stripe: index,
5593 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5594 rows,
5595 });
5596 }
5597 }
5598 Ok(places)
5599}
5600
5601fn read_index<F: Positional + ?Sized>(
5606 file: &F,
5607 stripe: &Stripe,
5608 column: usize,
5609) -> Result<Vec<PartSpan>> {
5610 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5611 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5612}
5613
5614fn read_index_span<F: Positional + ?Sized>(
5615 file: &F,
5616 index: Span,
5617 page: Span,
5618 parts: usize,
5619 column: usize,
5620) -> Result<Vec<PartSpan>> {
5621 let section = index_section(parts)?;
5622 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5623 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5624 if end > index.length as usize {
5625 return Err(invalid("index page is shorter than its columns"));
5626 }
5627 let mut bytes = vec![0; section];
5628 let offset =
5629 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5630 read_at(file, offset, &mut bytes)?;
5631 let entries = section - size_of::<u64>();
5632 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5633 if checksum(&bytes[..entries]) != stored {
5634 return Err(invalid(&format!(
5637 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5638 wanted {stored:016x} and got {:016x}",
5639 checksum(&bytes[..entries]),
5640 )));
5641 }
5642 let mut spans = Vec::with_capacity(parts);
5643 let mut start = 0_usize;
5644 for part in 0..parts {
5645 let at = part * INDEX_ENTRY;
5646 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5647 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5648 spans.push(PartSpan { start, length, hash });
5649 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5650 }
5651 if start != page.length as usize {
5652 return Err(invalid("column page length differs from its index"));
5653 }
5654 Ok(spans)
5655}
5656
5657fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5659 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5660 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5661}
5662
5663fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5669 if let Some(slot) = cached.index.get_mut(held.stripe) {
5670 if slot.is_none() {
5671 *slot = Some(Arc::clone(&held.index));
5672 }
5673 }
5674 let page = held.page.clone()?;
5675 let slot = cached.pages.get_mut(held.stripe)?;
5676 if slot.is_some() {
5677 return None;
5678 }
5679 let bytes = page.bytes.len();
5680 let used = Arc::new(AtomicBool::new(true));
5683 *slot = Some(Resident { page, used: Arc::clone(&used) });
5684 Some((bytes, used))
5685}
5686
5687#[derive(Debug, Clone)]
5696pub struct Catalog {
5697 file: Arc<File>,
5698 size: u64,
5699 entries: Arc<Vec<Entry>>,
5700 views: Arc<Vec<ViewEntry>>,
5702 opening: Opening,
5703 pool: PagePool,
5705}
5706
5707#[derive(Debug, Clone, PartialEq, Eq)]
5709pub struct CertifiedSums {
5710 pub columns: Vec<(i128, u64)>,
5711 pub rows: u64,
5712}
5713
5714#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5716pub enum IntegerExtremes {
5717 Null,
5718 Values { low: i128, high: i128 },
5719}
5720
5721pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5723
5724impl Catalog {
5725 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5734 Self::open_in(path, &PagePool::default())
5735 }
5736
5737 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5743 let (file, size, _, bytes, opening) = slot_bytes(path)?;
5744 let (entries, views) = decode_catalog(&bytes, size)?;
5745 Ok(Self {
5746 file: Arc::new(file),
5747 size,
5748 entries: Arc::new(entries),
5749 views: Arc::new(views),
5750 opening,
5751 pool: pool.clone(),
5752 })
5753 }
5754
5755 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5757 self.entries.iter().map(|entry| entry.name.as_str())
5758 }
5759
5760 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5767 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5768 }
5769
5770 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5776 self.views.iter()
5777 }
5778
5779 #[must_use]
5781 pub fn len(&self) -> usize {
5782 self.entries.len()
5783 }
5784
5785 #[must_use]
5788 pub fn is_empty(&self) -> bool {
5789 self.entries.is_empty()
5790 }
5791
5792 pub fn table(&self, name: &str) -> Result<Reader> {
5798 let entry = self
5799 .entries
5800 .iter()
5801 .find(|entry| entry.name == name)
5802 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5803 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5807 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5808 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5809 }
5810 let mut opening = self.opening;
5811 opening.reads += 1;
5812 opening.bytes += u64::from(entry.directory.length);
5813 Reader::build(
5814 Arc::clone(&self.file),
5815 self.size,
5816 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5817 u64::from(entry.directory.length),
5818 opening,
5819 self.pool.clone(),
5820 )
5821 }
5822
5823 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5831 let mut counts = BTreeMap::<i64, u64>::new();
5832 let Some(()) = self.integer_fold(name, column, |value, count| {
5833 let held = counts.entry(value).or_default();
5834 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5835 Ok(())
5836 })?
5837 else {
5838 return Ok(None);
5839 };
5840 Ok(Some(counts.into_iter().collect()))
5841 }
5842
5843 pub fn integer_fold(
5850 &self,
5851 name: &str,
5852 column: usize,
5853 mut emit: impl FnMut(i64, u64) -> Result<()>,
5854 ) -> Result<Option<()>> {
5855 let entry = self
5856 .entries
5857 .iter()
5858 .find(|entry| entry.name == name)
5859 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5860 let field =
5861 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5862 if !signed_integer(&field.ty) {
5863 return Ok(None);
5864 }
5865 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5866 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5867 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5868 }
5869 quick_integer_fold(
5870 &self.file,
5871 Cursor::over(&self.file, offset, length),
5872 entry,
5873 self.size,
5874 column,
5875 &mut emit,
5876 )?;
5877 Ok(Some(()))
5878 }
5879
5880 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5885 let entry = self
5886 .entries
5887 .iter()
5888 .find(|entry| entry.name == name)
5889 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5890 let Some(field) = entry.fields.get(column) else {
5891 return Err(invalid("frequency column index out of range"));
5892 };
5893 if !matches!(
5894 field.ty,
5895 LogicalType::TinyInt
5896 | LogicalType::SmallInt
5897 | LogicalType::Integer
5898 | LogicalType::BigInt
5899 | LogicalType::UTinyInt
5900 | LogicalType::USmallInt
5901 | LogicalType::UInteger
5902 | LogicalType::UBigInt
5903 ) {
5904 return Ok(None);
5905 }
5906 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5907 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5908 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5909 }
5910 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5911 return frequencies
5912 .iter()
5913 .filter(|(value, _)| value.is_some_and(|value| value != 0))
5914 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5915 .map(Some)
5916 .ok_or_else(|| invalid("numeric frequency count overflow"));
5917 }
5918 quick_nonzero(
5919 Cursor::over(&self.file, offset, length),
5920 &entry.name,
5921 &entry.fields,
5922 entry.rows,
5923 column,
5924 )
5925 }
5926
5927 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5930 let entry = self
5931 .entries
5932 .iter()
5933 .find(|entry| entry.name == name)
5934 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5935 let mut sums = Vec::with_capacity(columns.len());
5936 for &column in columns {
5937 let Some(field) = entry.fields.get(column) else {
5938 return Err(invalid("aggregate column index out of range"));
5939 };
5940 if !signed_integer(&field.ty) {
5941 return Ok(None);
5942 }
5943 let Some(sum) = entry.aggregates[column] else {
5944 return Ok(None);
5945 };
5946 sums.push(sum);
5947 }
5948 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5949 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5950 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5951 }
5952 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5953 }
5954
5955 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5957 let entry = self
5958 .entries
5959 .iter()
5960 .find(|entry| entry.name == name)
5961 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5962 let Some(count) = entry.distincts.get(column).copied() else {
5963 return Err(invalid("distinct column index out of range"));
5964 };
5965 let Some(count) = count else { return Ok(None) };
5966 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5967 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5968 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5969 }
5970 Ok(Some(count))
5971 }
5972
5973 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5975 let entry = self
5976 .entries
5977 .iter()
5978 .find(|entry| entry.name == name)
5979 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5980 let Some(extremes) = entry.extremes.get(column).copied() else {
5981 return Err(invalid("extremes column index out of range"));
5982 };
5983 let Some(extremes) = extremes else { return Ok(None) };
5984 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5985 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5986 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5987 }
5988 Ok(Some(match extremes {
5989 None => IntegerExtremes::Null,
5990 Some((low, high)) => IntegerExtremes::Values { low, high },
5991 }))
5992 }
5993
5994 pub fn exact_numeric_frequencies(
5996 &self,
5997 name: &str,
5998 column: usize,
5999 ) -> Result<Option<NumericFrequencies>> {
6000 let entry = self
6001 .entries
6002 .iter()
6003 .find(|entry| entry.name == name)
6004 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
6005 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
6006 return Err(invalid("numeric frequency column index out of range"));
6007 };
6008 let Some(frequencies) = frequencies else { return Ok(None) };
6009 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
6010 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
6011 return Err(invalid(&format!("the directory of table {name} does not checksum")));
6012 }
6013 Ok(Some(frequencies))
6014 }
6015
6016 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
6018 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
6019 }
6020}
6021
6022fn slot_offset(generation: u64) -> u64 {
6027 16 + (generation - 1) % 2 * SLOT_BYTES as u64
6028}
6029
6030fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
6035 let file = File::open(path).map_err(io)?;
6036 let size = file.metadata().map_err(io)?.len();
6037 let (slot, bytes, opening) = committed_slot(&file, size)?;
6038 Ok((file, size, slot, bytes, opening))
6039}
6040
6041fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
6047 if size < HEADER {
6048 return Err(invalid("file is shorter than its header"));
6049 }
6050 let mut header = [0; HEADER as usize];
6051 read_at(file, 0, &mut header)?;
6052 let mut opening = Opening { reads: 1, bytes: HEADER };
6053 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
6054 if &header[..8] != MAGIC {
6059 return Err(invalid("the header does not begin with a rudb native magic"));
6060 }
6061 if !READABLE.contains(&version) {
6062 return Err(invalid(&format!(
6063 "the file is format {version} and this build reads format {FORMAT}, so it has to \
6064 be written again"
6065 )));
6066 }
6067 let mut selected = None;
6068 for start in [16, 16 + SLOT_BYTES] {
6069 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
6070 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
6071 continue;
6072 }
6073 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
6074 if slot.offset < HEADER || end > size {
6075 continue;
6076 }
6077 let mut bytes = vec![0; slot.length as usize];
6078 read_at(file, slot.offset, &mut bytes)?;
6079 opening.reads += 1;
6080 opening.bytes += u64::from(slot.length);
6081 if checksum(&bytes) == slot.hash
6082 && selected
6083 .as_ref()
6084 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
6085 {
6086 selected = Some((slot, bytes));
6087 }
6088 }
6089 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
6090 Ok((slot, bytes, opening))
6091}
6092
6093impl Reader {
6094 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
6101 let catalog = Catalog::open(path)?;
6102 let mut names = catalog.names();
6103 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
6104 if names.next().is_some() {
6105 return Err(invalid(
6106 "the file holds more than one table, so it has to be opened by name",
6107 ));
6108 }
6109 catalog.table(&name)
6110 }
6111
6112 fn build(
6114 file: Arc<File>,
6115 size: u64,
6116 table: Table,
6117 directory: u64,
6118 opening: Opening,
6119 pool: PagePool,
6120 ) -> Result<Self> {
6121 let places = places(&table)?;
6122 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
6123 let table_fields = table.fields.len();
6124 let stripes = table.stripes.len();
6125 let columns = (0..table.fields.len())
6126 .map(|_| {
6127 Mutex::new(Cached {
6128 pages: (0..stripes).map(|_| None).collect(),
6129 index: (0..stripes).map(|_| None).collect(),
6130 seen: vec![false; stripes],
6131 ..Cached::default()
6132 })
6133 })
6134 .collect::<Vec<_>>();
6135 let cache = Shelf {
6136 columns,
6137 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
6138 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
6139 };
6140 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
6141 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6142 .collect();
6143 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
6144 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
6145 .collect();
6146 Ok(Self {
6147 file,
6148 table: Arc::new(table),
6149 dictionaries: Arc::new(dictionaries),
6150 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
6151 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6152 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
6153 opened: Arc::new(AtomicUsize::new(0)),
6154 sieves: Arc::new(sieves),
6155 part_ranges: Arc::new(part_ranges),
6156 places: Arc::new(places),
6157 cache: Arc::new(cache),
6158 pool,
6159 pages: Arc::new(AtomicUsize::new(0)),
6160 indexes: Arc::new(AtomicUsize::new(0)),
6161 size,
6162 directory,
6163 opening,
6164 })
6165 }
6166
6167 #[must_use]
6174 pub fn reads(&self) -> Reads {
6175 Reads {
6176 opening: self.opening,
6177 pages: self.pages.load(Atomic::Relaxed),
6178 indexes: self.indexes.load(Atomic::Relaxed),
6179 dictionaries: self.opened.load(Atomic::Relaxed),
6180 }
6181 }
6182
6183 #[must_use]
6188 pub fn layout(&self) -> Layout {
6189 let table = &self.table;
6190 let stripes = table.stripes.as_slice();
6191 let columns = table
6192 .fields
6193 .iter()
6194 .enumerate()
6195 .map(|(at, field)| ColumnLayout {
6196 name: field.name.clone(),
6197 kind: field.ty.to_string(),
6198 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6199 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6200 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6201 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6202 dictionary: dictionary_bytes(table, at),
6203 })
6204 .collect();
6205 Layout {
6206 file: self.size,
6207 rows: table.rows,
6208 stripes: stripes.len(),
6209 parts: self.places.len(),
6210 columns,
6211 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6212 directory: self.directory,
6213 header: HEADER,
6214 }
6215 }
6216
6217 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6234 let field = self
6235 .table
6236 .fields
6237 .get(column)
6238 .ok_or_else(|| invalid("stored column index out of range"))?;
6239 let mut stored = Vec::with_capacity(self.places.len());
6240 let mut row = 0;
6241 for (at, stripe) in self.table.stripes.iter().enumerate() {
6242 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6243 let index = read_index(&self.file, stripe, column)?;
6244 let mut bytes = vec![0; page.length as usize];
6245 read_at(&self.file, page.offset, &mut bytes)?;
6246 let ranges = self.stripe_part_ranges(at, column);
6247 for (part, &rows) in stripe.parts.iter().enumerate() {
6248 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6249 let held = part_bytes(&bytes, span)?;
6250 let range = ranges.and_then(|held| held.get(part));
6251 stored.push(StoredPart {
6252 stripe: at,
6253 part,
6254 row,
6255 rows: rows as usize,
6256 encoding: page_encoding(&field.ty, rows as usize, held),
6257 bytes: span.length as u64,
6258 page: page.offset,
6259 offset: span.start as u64,
6260 low: range
6261 .and_then(|range| range.low.clone())
6262 .and_then(|bound| bound.into_value(&field.ty)),
6263 high: range
6264 .and_then(|range| range.high.clone())
6265 .and_then(|bound| bound.into_value(&field.ty)),
6266 nulls: range.map(|range| range.nulls),
6267 });
6268 row += rows as usize;
6269 }
6270 }
6271 Ok(stored)
6272 }
6273
6274 #[must_use]
6276 pub fn parts(&self) -> usize {
6277 self.places.len()
6278 }
6279
6280 #[must_use]
6287 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6288 let mut runs = Vec::with_capacity(self.table.stripes.len());
6289 let mut start = 0;
6290 for stripe in &self.table.stripes {
6291 let end = start + stripe.parts.len();
6292 runs.push(start..end);
6293 start = end;
6294 }
6295 runs
6296 }
6297
6298 #[must_use]
6303 pub fn stripe_rows(&self, stripe: usize) -> usize {
6304 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6305 }
6306
6307 pub fn keep_stripes(&self, stripes: usize) {
6314 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6315 }
6316
6317 #[must_use]
6319 pub fn part_rows(&self, at: usize) -> usize {
6320 self.places.get(at).map_or(0, |place| place.rows as usize)
6321 }
6322
6323 #[must_use]
6325 pub fn table(&self) -> &Table {
6326 &self.table
6327 }
6328
6329 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6338 let field = self
6339 .table
6340 .fields
6341 .get(column)
6342 .ok_or_else(|| invalid("frequency column index out of range"))?;
6343 let Some(summary) = self.frequency_summary(column)? else {
6344 return Ok(None);
6345 };
6346 if top == 0 || summary.entries.len() < top {
6347 return Ok(None);
6348 }
6349 let boundary = summary.entries[top - 1].count;
6350 if boundary <= summary.omitted_max {
6351 return Ok(None);
6352 }
6353 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6354 }
6355
6356 pub fn top_pair_frequencies(
6364 &self,
6365 first: usize,
6366 second: usize,
6367 _top: usize,
6368 ) -> Result<Option<PairFrequencyCounts>> {
6369 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6370 return Err(invalid("pair frequency column index out of range"));
6371 }
6372 Ok(None)
6373 }
6374
6375 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6395 let Some(prefix) = self.frequency_prefix(column)? else {
6396 return Ok(None);
6397 };
6398 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6399 }
6400
6401 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6424 let field = self
6425 .table
6426 .fields
6427 .get(column)
6428 .ok_or_else(|| invalid("frequency column index out of range"))?;
6429 let Some(summary) = self.frequency_summary(column)? else {
6430 return Ok(None);
6431 };
6432 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6433 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6434 }
6435
6436 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6438 Ok(match self.table.frequencies.get(column) {
6439 None | Some(None) => None,
6440 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6441 Some(Some(Frequencies::Stored { span, values })) => {
6442 let slot = self
6443 .frequency_summaries
6444 .get(column)
6445 .ok_or_else(|| invalid("frequency column index out of range"))?;
6446 if let Some(summary) = slot.get() {
6447 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6448 }
6449 let field = self
6450 .table
6451 .fields
6452 .get(column)
6453 .ok_or_else(|| invalid("frequency column index out of range"))?;
6454 let mut bytes = vec![0; span.length as usize];
6455 read_at(&self.file, span.offset, &mut bytes)?;
6456 let summary =
6457 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6458 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6459 let _ = slot.set(Arc::new(summary));
6460 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6461 }
6462 })
6463 }
6464
6465 fn decode_frequencies(
6473 &self,
6474 column: usize,
6475 ty: &LogicalType,
6476 entries: &[FrequencyEntry],
6477 ) -> Result<Vec<(Value, u64)>> {
6478 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6479 return Ok(values.as_ref().clone());
6480 }
6481 let values = self.decode_frequencies_once(column, ty, entries)?;
6482 if let Some(slot) = self.frequency_values.get(column) {
6483 let _ = slot.set(Arc::new(values.clone()));
6484 }
6485 Ok(values)
6486 }
6487
6488 fn decode_frequencies_once(
6489 &self,
6490 column: usize,
6491 ty: &LogicalType,
6492 entries: &[FrequencyEntry],
6493 ) -> Result<Vec<(Value, u64)>> {
6494 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6495 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6496 return Err(invalid("frequency text count differs from its synopsis"));
6497 }
6498 let dictionary =
6499 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6500 let mut codes = entries
6501 .iter()
6502 .filter_map(|entry| match entry.value {
6503 FrequencyValue::Code(code) => Some(code as usize),
6504 _ => None,
6505 })
6506 .collect::<Vec<_>>();
6507 codes.sort_unstable();
6508 codes.dedup();
6509 let texts = match &dictionary {
6510 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6511 _ => Vec::new(),
6512 };
6513 let mut out = Vec::with_capacity(entries.len());
6514 for (entry_at, entry) in entries.iter().enumerate() {
6515 let value = match entry.value {
6516 FrequencyValue::Null => {
6517 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6518 return Err(invalid("a null frequency entry has text"));
6519 }
6520 Value::Null
6521 }
6522 FrequencyValue::Integer(value) => match *ty {
6523 LogicalType::TinyInt => Value::TinyInt(
6524 i8::try_from(value)
6525 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6526 ),
6527 LogicalType::UTinyInt => Value::UTinyInt(
6528 u8::try_from(value)
6529 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6530 ),
6531 LogicalType::USmallInt => Value::USmallInt(
6532 u16::try_from(value)
6533 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6534 ),
6535 LogicalType::UInteger => Value::UInteger(
6536 u32::try_from(value)
6537 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6538 ),
6539 LogicalType::UBigInt => Value::UBigInt(
6540 u64::try_from(value)
6541 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6542 ),
6543 LogicalType::SmallInt => Value::SmallInt(
6544 i16::try_from(value)
6545 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6546 ),
6547 LogicalType::Integer => Value::Integer(
6548 i32::try_from(value)
6549 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6550 ),
6551 LogicalType::BigInt => Value::BigInt(
6552 i64::try_from(value)
6553 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6554 ),
6555 LogicalType::Date => Value::Date(
6556 i32::try_from(value)
6557 .map_err(|_| invalid("frequency DATE is out of range"))?,
6558 ),
6559 LogicalType::Timestamp => Value::Timestamp(
6560 i64::try_from(value)
6561 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6562 ),
6563 _ => return Err(invalid("integer frequency belongs to another type")),
6564 },
6565 FrequencyValue::Code(code) => {
6566 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6567 if *ty == LogicalType::Blob {
6568 Value::Blob(text.clone())
6569 } else {
6570 Value::Varchar(
6571 String::from_utf8(text.clone())
6572 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6573 )
6574 }
6575 } else {
6576 if dictionary.is_none() {
6577 return Err(invalid("frequency code has no dictionary or stored text"));
6578 }
6579 let at = codes
6580 .binary_search(&(code as usize))
6581 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6582 texts[at].clone()
6583 }
6584 }
6585 };
6586 out.push((value, entry.count));
6587 }
6588 Ok(out)
6589 }
6590
6591 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6601 let field = self
6602 .table
6603 .fields
6604 .get(column)
6605 .ok_or_else(|| invalid("frequency column index out of range"))?;
6606 let Some(summary) = self.frequency_summary(column)? else {
6607 return Ok(None);
6608 };
6609 if summary.ordinals.is_empty() {
6610 return Ok(None);
6611 }
6612 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6613 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6614 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6615 } else {
6616 (Vec::new(), Vec::new())
6617 };
6618 Ok(Some(FrequencyOccurrences {
6619 omitted_max: summary.omitted_max,
6620 ordinals: summary.ordinals.clone(),
6621 anchors,
6622 anchor_indices,
6623 }))
6624 }
6625
6626 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6652 self.table
6653 .distincts
6654 .get(column)
6655 .copied()
6656 .ok_or_else(|| invalid("distinct column index out of range"))
6657 }
6658
6659 pub fn null_count(&self, column: usize) -> Result<u64> {
6670 if column >= self.table.fields.len() {
6671 return Err(invalid("null count column index out of range"));
6672 }
6673 let mut nulls = 0_u64;
6674 for stripe in &self.table.stripes {
6675 let range = stripe
6676 .zone
6677 .column(column)
6678 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6679 nulls = nulls
6680 .checked_add(range.nulls as u64)
6681 .ok_or_else(|| invalid("null count overflow"))?;
6682 }
6683 Ok(nulls)
6684 }
6685
6686 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6701 if self.null_count(column)? > 0 || self.demoted(column) {
6702 return Ok(None);
6703 }
6704 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6705 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6706 if ranks == 0 {
6707 return Ok(None);
6708 }
6709 let low = text_at_rank(&dictionary, 0)?;
6710 let high = text_at_rank(&dictionary, ranks - 1)?;
6711 Ok(Some((low, high)))
6712 }
6713
6714 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6737 if column >= self.table.fields.len() {
6738 return Err(invalid("extremes column index out of range"));
6739 }
6740 let mut low: Option<Bound> = None;
6741 let mut high: Option<Bound> = None;
6742 for stripe in &self.table.stripes {
6743 let range = stripe
6744 .zone
6745 .column(column)
6746 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6747 if !range.exact {
6748 return Ok(None);
6749 }
6750 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6755 if stripe.rows > range.nulls {
6756 return Ok(None);
6757 }
6758 continue;
6759 };
6760 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6761 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6762 }
6763 Ok(low.zip(high))
6764 }
6765
6766 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6779 if column >= self.table.fields.len() {
6780 return Err(invalid("sum column index out of range"));
6781 }
6782 let mut total = 0_i128;
6783 let mut rows = 0_u64;
6784 for stripe in &self.table.stripes {
6785 let range = stripe
6786 .zone
6787 .column(column)
6788 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6789 let Some(part) = range.sum else { return Ok(None) };
6790 let Some(sum) = total.checked_add(part) else { return Ok(None) };
6791 total = sum;
6792 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6793 }
6794 Ok(Some((total, rows)))
6795 }
6796
6797 pub fn host_groups(
6799 &self,
6800 column: usize,
6801 _minimum_count: u64,
6802 ) -> Result<Option<Vec<host::HostEntry>>> {
6803 if column >= self.table.fields.len() {
6804 return Err(invalid("host group column index out of range"));
6805 }
6806 Ok(None)
6807 }
6808
6809 #[must_use]
6813 pub fn demoted(&self, column: usize) -> bool {
6814 self.table.demoted.get(column).copied().unwrap_or(false)
6815 }
6816
6817 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6826 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6827 if let Some(dictionary) = self.dictionaries[column].get() {
6828 return Ok(Some(Arc::clone(dictionary)));
6829 }
6830 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6831 if let Some(dictionary) = self.dictionaries[column].get() {
6832 return Ok(Some(Arc::clone(dictionary)));
6833 }
6834 self.opened.fetch_add(1, Atomic::Relaxed);
6835 let dictionary = Arc::new(open_global_dictionary(
6836 Arc::clone(&self.file),
6837 page,
6838 &self.table.fields[column].ty,
6839 TEXT_KEEP_BUDGET,
6840 )?);
6841 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6842 Ok(Some(dictionary))
6843 }
6844
6845 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6852 if of.extent_bytes == 0 {
6853 return Ok(Vec::new());
6854 }
6855 let mut bytes = vec![0; of.extent_bytes as usize];
6856 read_at(&self.file, of.extent_page, &mut bytes)?;
6857 if checksum(&bytes) != of.hash {
6858 return Err(invalid("a section's extent table does not checksum"));
6859 }
6860 let extents = section::decode_extents(&bytes)?;
6861 if extents.len() != of.extents as usize {
6862 return Err(invalid("a section's extent table is not the length the entry says"));
6863 }
6864 Ok(extents)
6865 }
6866
6867 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
6877 let mut bytes = Vec::new();
6878 self.extent_into(of, &mut bytes)?;
6879 Ok(bytes)
6880 }
6881
6882 fn extent_into(&self, of: §ion::Extent, bytes: &mut Vec<u8>) -> Result<()> {
6884 let end = of
6885 .offset
6886 .checked_add(u64::from(of.length))
6887 .ok_or_else(|| invalid("an extent overflows the file"))?;
6888 if of.offset < HEADER || end > self.size {
6889 return Err(invalid("an extent is outside the file"));
6890 }
6891 bytes.resize(of.length as usize, 0);
6892 read_at(&self.file, of.offset, bytes)?;
6893 if checksum(bytes) != of.hash {
6894 return Err(invalid("an extent does not checksum"));
6895 }
6896 Ok(())
6897 }
6898
6899 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6908 let extents = self.extents(of)?;
6909 let mut bytes =
6910 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6911 for one in &extents {
6912 if one.first != bytes.len() as u64 {
6913 return Err(invalid("a section's extents do not join up"));
6914 }
6915 bytes.extend_from_slice(&self.extent(one)?);
6916 }
6917 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6920 return Err(invalid("a section's header is longer than its payload"));
6921 }
6922 Ok(bytes)
6923 }
6924
6925 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6934 self.read_impl(part, columns, true, None)
6935 }
6936
6937 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6947 self.read_impl(part, columns, false, None)
6948 }
6949
6950 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6958 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6959 let field =
6960 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6961 if !matches!(
6962 field.ty,
6963 LogicalType::TinyInt
6964 | LogicalType::SmallInt
6965 | LogicalType::Integer
6966 | LogicalType::BigInt
6967 ) {
6968 return Ok(None);
6969 }
6970 let (rows, counts) = match self.with_part(place, column, |bytes| {
6971 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6972 return Ok(None);
6973 }
6974 integer::tally(&bytes[2..]).map(Some)
6975 })? {
6976 Some(tallied) => tallied,
6977 None => return Ok(None),
6978 };
6979 if rows != place.rows as usize {
6980 return Err(invalid("encoded integer part holds the wrong number of rows"));
6981 }
6982 for &(value, _) in &counts {
6983 let fits = match field.ty {
6984 LogicalType::TinyInt => i8::try_from(value).is_ok(),
6985 LogicalType::SmallInt => i16::try_from(value).is_ok(),
6986 LogicalType::Integer => i32::try_from(value).is_ok(),
6987 LogicalType::BigInt => true,
6988 _ => false,
6989 };
6990 if !fits {
6991 return Err(invalid("encoded integer value is outside its column type"));
6992 }
6993 }
6994 Ok(Some(counts))
6995 }
6996
6997 pub fn rows_holding(
7009 &self,
7010 part: usize,
7011 column: usize,
7012 sequence: &Sequence,
7013 negated: bool,
7014 ) -> Result<Option<Vec<u32>>> {
7015 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7016 let field =
7017 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
7018 if field.ty != LogicalType::Varchar {
7019 return Ok(None);
7020 }
7021 let rows = place.rows as usize;
7022 self.with_part(place, column, |bytes| {
7023 if bytes.first() != Some(&6) {
7024 return Ok(None);
7025 }
7026 let mut cur = Cursor::new(bytes);
7027 cur.u8()?;
7028 let mask = match cur.u8()? {
7029 0 => None,
7030 1 => return Ok(Some(Vec::new())),
7031 2 => {
7032 let from = cur.at;
7033 cur.take(rows.div_ceil(8))?;
7034 Some(&bytes[from..cur.at])
7035 }
7036 _ => return Err(invalid("page validity tag differs")),
7037 };
7038 let Some(held) = string::holds_in(&bytes[cur.at..], sequence)? else {
7039 return Ok(None);
7040 };
7041 if held.len() != rows {
7042 return Err(invalid("compressed text page holds the wrong number of rows"));
7043 }
7044 let valid = |row: usize| mask.is_none_or(|mask| mask[row / 8] >> (row % 8) & 1 == 1);
7045 Ok(Some(
7046 (0..rows)
7047 .filter(|&row| held[row] != negated && valid(row))
7048 .map(|row| row as u32)
7049 .collect(),
7050 ))
7051 })
7052 }
7053
7054 fn with_part<T>(
7057 &self,
7058 place: Place,
7059 column: usize,
7060 read: impl FnOnce(&[u8]) -> Result<T>,
7061 ) -> Result<T> {
7062 let stripe_index = place.stripe as usize;
7063 let stripe = self
7064 .table
7065 .stripes
7066 .get(stripe_index)
7067 .ok_or_else(|| invalid("stripe index out of range"))?;
7068 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7069 let held = self.held(stripe_index, stripe, column, true)?;
7070 let span = *held
7071 .index
7072 .get(place.part as usize)
7073 .ok_or_else(|| invalid("part index out of range"))?;
7074 match &held.page {
7075 Some(page) => read(page.part(place.part as usize, span)?),
7076 None => {
7077 let offset = page
7078 .offset
7079 .checked_add(span.start as u64)
7080 .ok_or_else(|| invalid("part range overflow"))?;
7081 let mut bytes = vec![0; span.length];
7082 read_at(&self.file, offset, &mut bytes)?;
7083 verify_part(&bytes, span)?;
7084 read(&bytes)
7085 }
7086 }
7087 }
7088
7089 pub fn read_rows(
7102 &self,
7103 part: usize,
7104 columns: &[usize],
7105 positions: &[u32],
7106 whole: bool,
7107 ) -> Result<Chunk> {
7108 self.read_impl(part, columns, whole, Some(positions))
7109 }
7110
7111 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
7118 if self.demoted(column) {
7121 return Ok(false);
7122 }
7123 if candidates.is_empty() {
7124 return Ok(true);
7125 }
7126 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
7127 return Err(Error::internal("native code candidates are not sorted and unique"));
7128 }
7129 let stripe = self.stripe_of(part)?;
7130 let Some(page) = stripe.memberships.get(column) else {
7131 return Ok(false);
7132 };
7133 let mut bytes = vec![0; page.length as usize];
7134 read_at(&self.file, page.offset, &mut bytes)?;
7135 if checksum(&bytes) != page.hash {
7136 return Err(invalid("membership page checksum differs"));
7137 }
7138 let codes = decode_membership(&bytes)?;
7139 let mut left = 0;
7140 let mut right = 0;
7141 while left < codes.len() && right < candidates.len() {
7142 match codes[left].cmp(&candidates[right]) {
7143 Ordering::Less => left += 1,
7144 Ordering::Greater => right += 1,
7145 Ordering::Equal => return Ok(false),
7146 }
7147 }
7148 Ok(true)
7149 }
7150
7151 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
7152 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
7153 self.table
7154 .stripes
7155 .get(place.stripe as usize)
7156 .ok_or_else(|| invalid("stripe index out of range"))
7157 }
7158
7159 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
7176 let cache =
7177 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
7178 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7179 let known = cached.index.get(at).and_then(Clone::clone);
7180 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
7181 slot.used.store(true, Atomic::Relaxed);
7182 Arc::clone(&slot.page)
7183 });
7184 if let Some(index) = known.clone() {
7185 if !whole || page.is_some() {
7186 return Ok(CachedColumn { stripe: at, index, page });
7187 }
7188 }
7189 if cached.loading.contains(&at) {
7190 drop(cached);
7191 if let Some(index) = known {
7195 return Ok(CachedColumn { stripe: at, index, page: None });
7196 }
7197 let held = self.page_of(stripe, column, at, false, None)?;
7198 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7199 remember(&mut cached, &held);
7200 return Ok(held);
7201 }
7202 cached.loading.push(at);
7203 drop(cached);
7204
7205 let read = self.page_of(stripe, column, at, whole, known);
7206
7207 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7211 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7212 cached.loading.remove(position);
7213 }
7214 let held = read?;
7215 let taken = remember(&mut cached, &held);
7216 let first = taken.is_some()
7217 && cached.seen.get_mut(at).is_some_and(|seen| !std::mem::replace(seen, true));
7218 if first {
7219 let floor = self.cache.kept.load(Atomic::Relaxed).max(1);
7220 cached.passing.push_back(at);
7221 while cached.passing.len() > floor {
7222 let Some(old) = cached.passing.pop_front() else { break };
7223 if let Some(slot) = cached.pages.get_mut(old) {
7224 *slot = None;
7225 }
7226 }
7227 return Ok(held);
7228 }
7229 drop(cached);
7230 if let Some((bytes, used)) = taken {
7231 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7232 self.pool.admit(Held {
7233 shelf: Arc::downgrade(&self.cache),
7234 column,
7235 stripe: at,
7236 bytes,
7237 used,
7238 });
7239 }
7240 Ok(held)
7241 }
7242
7243 fn page_of(
7249 &self,
7250 stripe: &Stripe,
7251 column: usize,
7252 at: usize,
7253 whole: bool,
7254 known: Option<Arc<Vec<PartSpan>>>,
7255 ) -> Result<CachedColumn> {
7256 let index = match known {
7257 Some(index) => index,
7258 None => {
7259 self.indexes.fetch_add(1, Atomic::Relaxed);
7260 Arc::new(read_index(&self.file, stripe, column)?)
7261 }
7262 };
7263 let page = if whole {
7264 self.pages.fetch_add(1, Atomic::Relaxed);
7265 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7266 let mut bytes = vec![0; span.length as usize];
7267 read_at(&self.file, span.offset, &mut bytes)?;
7268 let checked = index.iter().map(|_| AtomicBool::new(false)).collect();
7269 Some(Arc::new(HeldPage { bytes, checked }))
7270 } else {
7271 None
7272 };
7273 Ok(CachedColumn { stripe: at, index, page })
7274 }
7275
7276 fn read_impl(
7277 &self,
7278 at: usize,
7279 columns: &[usize],
7280 whole: bool,
7281 positions: Option<&[u32]>,
7282 ) -> Result<Chunk> {
7283 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7284 let index = place.stripe as usize;
7285 let stripe =
7286 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7287 let rows = place.rows as usize;
7288 let mut picked = Vec::with_capacity(columns.len());
7289 for &column in columns {
7290 let field = self
7291 .table
7292 .fields
7293 .get(column)
7294 .ok_or_else(|| invalid("column index out of range"))?;
7295 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7296 let held = self.held(index, stripe, column, whole)?;
7297 let span = *held
7298 .index
7299 .get(place.part as usize)
7300 .ok_or_else(|| invalid("part index out of range"))?;
7301 let owned;
7302 let bytes = match &held.page {
7303 Some(held) => held.part(place.part as usize, span),
7304 None => {
7305 let offset = page
7306 .offset
7307 .checked_add(span.start as u64)
7308 .ok_or_else(|| invalid("part range overflow"))?;
7309 let mut bytes = vec![0; span.length];
7310 read_at(&self.file, offset, &mut bytes)?;
7311 owned = bytes;
7312 verify_part(&owned, span).map(|()| owned.as_slice())
7313 }
7314 }
7315 .map_err(|error| {
7316 invalid(&format!(
7317 "{}, column {column} part {} of the page at {}",
7318 error.message(),
7319 place.part,
7320 page.offset,
7321 ))
7322 })?;
7323 let dictionary = self.dictionary(column)?;
7324 let mut vector = match positions {
7330 None => decode(&field.ty, rows, bytes, dictionary)?,
7331 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7332 };
7333 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7337 vector = vector.flatten()?;
7338 }
7339 picked.push(vector.into_pages());
7340 }
7341 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7342 }
7343
7344 #[must_use]
7360 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7361 let Some(place) = self.places.get(part).copied() else { return false };
7362 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7363 if stripe.zone.skips(probes) {
7364 return true;
7365 }
7366 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7367 }
7368
7369 #[must_use]
7376 pub fn ruled_by(&self, part: usize, column: usize, rule: impl Fn(&Range) -> bool) -> bool {
7377 let Some(place) = self.places.get(part).copied() else { return false };
7378 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7379 if stripe.zone.column(column).is_some_and(&rule) {
7380 return true;
7381 }
7382 self.stripe_part_ranges(place.stripe as usize, column)
7383 .and_then(|ranges| ranges.get(place.part as usize))
7384 .is_some_and(rule)
7385 }
7386
7387 #[must_use]
7389 pub fn stripe_ruled_by(
7390 &self,
7391 stripe: usize,
7392 column: usize,
7393 rule: impl Fn(&Range) -> bool,
7394 ) -> bool {
7395 self.table.stripes.get(stripe).and_then(|held| held.zone.column(column)).is_some_and(rule)
7396 }
7397
7398 fn outside(&self, place: Place, probe: &Probe) -> bool {
7404 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7405 Some(ranges) => ranges
7406 .get(place.part as usize)
7407 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7408 None => false,
7409 }
7410 }
7411
7412 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7418 let slot = self.part_ranges.get(column)?.get(stripe)?;
7419 if let Some(held) = slot.get() {
7420 return Some(held);
7421 }
7422 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7423 let mut bytes = vec![0; page.length as usize];
7424 read_at(&self.file, page.offset, &mut bytes).ok()?;
7425 if checksum(&bytes) != page.hash {
7426 return None;
7427 }
7428 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7429 let _ = slot.set(ranges);
7430 slot.get().map(|held| held.as_slice())
7431 }
7432
7433 #[must_use]
7450 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7451 let Some(place) = self.places.get(part).copied() else { return false };
7452 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7453 if stripe.zone.certain(probes) {
7454 return true;
7455 }
7456 probes
7457 .iter()
7458 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7459 }
7460
7461 fn inside(&self, place: Place, probe: &Probe) -> bool {
7467 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7468 Some(ranges) => ranges
7469 .get(place.part as usize)
7470 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7471 None => false,
7472 }
7473 }
7474
7475 #[must_use]
7486 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7487 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7488 }
7489
7490 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7496 if probe.op != Op::Equal {
7497 return false;
7498 }
7499 match self.stripe_sieves(place.stripe as usize, probe.column) {
7500 Some(sieves) => sieves
7501 .get(place.part as usize)
7502 .and_then(Option::as_ref)
7503 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7504 None => false,
7505 }
7506 }
7507
7508 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7515 let slot = self.sieves.get(column)?.get(stripe)?;
7516 if let Some(held) = slot.get() {
7517 return Some(held);
7518 }
7519 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7520 let mut bytes = vec![0; page.length as usize];
7521 read_at(&self.file, page.offset, &mut bytes).ok()?;
7522 if checksum(&bytes) != page.hash {
7523 return None;
7524 }
7525 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7526 let _ = slot.set(sieves);
7527 slot.get().map(|held| held.as_slice())
7528 }
7529}
7530
7531fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7533 let code = dictionary.code_at_rank(rank)? as usize;
7534 if dictionary.logical_type() == &LogicalType::Blob {
7535 let bytes = dictionary
7536 .try_bytes_at(code)?
7537 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7538 return Ok(Value::Blob(bytes.to_vec()));
7539 }
7540 let text = dictionary
7541 .try_text_at(code)?
7542 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7543 Ok(Value::Varchar(text.into()))
7544}
7545
7546fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7556 file.fill_at(offset, bytes)
7557}
7558
7559trait Positional {
7567 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7572}
7573
7574impl<T: Positional + ?Sized> Positional for &T {
7575 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7576 (**self).fill_at(offset, bytes)
7577 }
7578}
7579
7580impl<T: Positional + ?Sized> Positional for Arc<T> {
7581 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7582 (**self).fill_at(offset, bytes)
7583 }
7584}
7585
7586impl<T: Positional + ?Sized> Positional for Box<T> {
7587 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7588 (**self).fill_at(offset, bytes)
7589 }
7590}
7591
7592impl Positional for dyn rudb_io::File + '_ {
7593 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7594 while !bytes.is_empty() {
7595 let read = self.read_at(offset, bytes)?;
7596 if read == 0 {
7597 return Err(invalid("column page ends before its declared length"));
7598 }
7599 offset += read as u64;
7600 bytes = &mut bytes[read..];
7601 }
7602 Ok(())
7603 }
7604}
7605
7606impl Positional for File {
7607 #[cfg(unix)]
7608 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7609 use std::os::unix::fs::FileExt;
7610 while !bytes.is_empty() {
7611 let read = self.read_at(bytes, offset).map_err(io)?;
7612 if read == 0 {
7613 return Err(invalid("column page ends before its declared length"));
7614 }
7615 offset += read as u64;
7616 bytes = &mut bytes[read..];
7617 }
7618 Ok(())
7619 }
7620
7621 #[cfg(windows)]
7627 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7628 use std::os::windows::fs::FileExt;
7629 while !bytes.is_empty() {
7630 let read = self.seek_read(bytes, offset).map_err(io)?;
7631 if read == 0 {
7632 return Err(invalid("column page ends before its declared length"));
7633 }
7634 offset += read as u64;
7635 bytes = &mut bytes[read..];
7636 }
7637 Ok(())
7638 }
7639
7640 #[cfg(not(any(unix, windows)))]
7645 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7646 use std::io::{Read, Seek, SeekFrom};
7647 let mut file = self.try_clone().map_err(io)?;
7648 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7649 file.read_exact(bytes).map_err(io)
7650 }
7651}
7652
7653#[cfg(test)]
7658fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7659 use std::io::{Seek, SeekFrom, Write};
7660 let mut file = file;
7661 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7662 file.write_all(bytes).map_err(io)
7663}
7664
7665fn type_tag(ty: &LogicalType) -> Result<u8> {
7672 match ty {
7673 LogicalType::SmallInt => Ok(1),
7674 LogicalType::Integer => Ok(2),
7675 LogicalType::BigInt => Ok(3),
7676 LogicalType::Varchar => Ok(4),
7677 LogicalType::Date => Ok(5),
7678 LogicalType::Timestamp => Ok(6),
7679 LogicalType::Boolean => Ok(7),
7680 LogicalType::TinyInt => Ok(8),
7681 LogicalType::UTinyInt => Ok(9),
7682 LogicalType::USmallInt => Ok(10),
7683 LogicalType::UInteger => Ok(11),
7684 LogicalType::UBigInt => Ok(12),
7685 LogicalType::Decimal { .. } => Ok(13),
7686 LogicalType::Float => Ok(14),
7687 LogicalType::Double => Ok(15),
7688 LogicalType::HugeInt => Ok(16),
7689 LogicalType::UHugeInt => Ok(17),
7690 LogicalType::Time => Ok(18),
7691 LogicalType::TimeTz => Ok(19),
7692 LogicalType::TimestampTz => Ok(20),
7693 LogicalType::Interval => Ok(21),
7694 LogicalType::Uuid => Ok(22),
7695 LogicalType::Blob => Ok(23),
7696 LogicalType::Bit => Ok(24),
7697 LogicalType::TimestampS => Ok(25),
7698 LogicalType::TimestampMs => Ok(26),
7699 LogicalType::TimestampNs => Ok(27),
7700 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7701 }
7702}
7703
7704fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7710 out.push(type_tag(ty)?);
7711 if let LogicalType::Decimal { width, scale } = ty {
7712 out.push(*width);
7713 out.push(*scale);
7714 }
7715 Ok(())
7716}
7717
7718fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7720 let tag = cur.u8()?;
7721 if tag == 13 {
7722 let width = cur.u8()?;
7723 let scale = cur.u8()?;
7724 return LogicalType::decimal(width, scale)
7725 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7726 }
7727 tag_type(tag)
7728}
7729
7730fn tag_type(tag: u8) -> Result<LogicalType> {
7731 match tag {
7732 1 => Ok(LogicalType::SmallInt),
7733 2 => Ok(LogicalType::Integer),
7734 3 => Ok(LogicalType::BigInt),
7735 4 => Ok(LogicalType::Varchar),
7736 5 => Ok(LogicalType::Date),
7737 6 => Ok(LogicalType::Timestamp),
7738 7 => Ok(LogicalType::Boolean),
7739 8 => Ok(LogicalType::TinyInt),
7740 9 => Ok(LogicalType::UTinyInt),
7741 10 => Ok(LogicalType::USmallInt),
7742 11 => Ok(LogicalType::UInteger),
7743 12 => Ok(LogicalType::UBigInt),
7744 14 => Ok(LogicalType::Float),
7745 15 => Ok(LogicalType::Double),
7746 16 => Ok(LogicalType::HugeInt),
7747 17 => Ok(LogicalType::UHugeInt),
7748 18 => Ok(LogicalType::Time),
7749 19 => Ok(LogicalType::TimeTz),
7750 20 => Ok(LogicalType::TimestampTz),
7751 21 => Ok(LogicalType::Interval),
7752 22 => Ok(LogicalType::Uuid),
7753 23 => Ok(LogicalType::Blob),
7754 24 => Ok(LogicalType::Bit),
7755 25 => Ok(LogicalType::TimestampS),
7756 26 => Ok(LogicalType::TimestampMs),
7757 27 => Ok(LogicalType::TimestampNs),
7758 _ => Err(invalid("column type tag is unknown")),
7759 }
7760}
7761
7762fn put_u16(out: &mut Vec<u8>, value: u16) {
7763 out.extend_from_slice(&value.to_le_bytes());
7764}
7765fn put_u32(out: &mut Vec<u8>, value: u32) {
7766 out.extend_from_slice(&value.to_le_bytes());
7767}
7768fn put_u64(out: &mut Vec<u8>, value: u64) {
7769 out.extend_from_slice(&value.to_le_bytes());
7770}
7771fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7772 while value >= 0x80 {
7773 out.push((value as u8 & 0x7f) | 0x80);
7774 value >>= 7;
7775 }
7776 out.push(value as u8);
7777}
7778
7779fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7780 match (left, right) {
7781 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7782 (FrequencyValue::Null, _) => Ordering::Less,
7783 (_, FrequencyValue::Null) => Ordering::Greater,
7784 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7785 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7786 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7787 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7788 }
7789}
7790
7791fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7804 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7805 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7806 };
7807 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7808 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7809 let omitted_max = next.count;
7810 entries.truncate(FREQUENCY_ENTRIES);
7811 omitted_max
7812 } else {
7813 0
7814 };
7815 entries.sort_unstable_by(order);
7816 omitted_max
7817}
7818
7819fn code_frequency(
7820 dictionary: &GlobalDictionary,
7821 flat: &[u8],
7822 bases: &[u64],
7823) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7824 let mut entries = dictionary
7825 .counts
7826 .iter()
7827 .enumerate()
7828 .filter(|(_, count)| **count != 0)
7829 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7830 .collect::<Vec<_>>();
7831 if dictionary.nulls != 0 {
7832 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7833 }
7834 let omitted_max = keep_most_frequent(&mut entries);
7835 let mut spans = Vec::with_capacity(entries.len());
7836 let mut text_bytes = 0_usize;
7837 for entry in &entries {
7838 let span = match entry.value {
7839 FrequencyValue::Code(code) => {
7840 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7841 let bytes = flat
7842 .get(span.0..span.1)
7843 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7844 text_bytes = text_bytes.saturating_add(bytes.len());
7845 Some(span)
7846 }
7847 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7848 };
7849 spans.push(span);
7850 }
7851 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7852 Vec::new()
7853 } else {
7854 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7855 };
7856 Ok((
7857 FrequencySummary {
7858 entries,
7859 omitted_max,
7860 ordinals: Vec::new(),
7861 ordinal_entries: Vec::new(),
7862 },
7863 texts,
7864 ))
7865}
7866
7867fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7868 let mut out = DIRECTORY.to_vec();
7869 let name = table.name.as_bytes();
7870 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7871 out.extend_from_slice(name);
7872 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7873 for field in &table.fields {
7874 let name = field.name.as_bytes();
7875 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7876 out.extend_from_slice(name);
7877 put_type(&mut out, &field.ty)?;
7878 out.push(u8::from(field.not_null));
7879 }
7880 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7881 match dictionary {
7882 None => out.push(0),
7883 Some(page) => {
7884 out.push(dictionary_tag(&field.ty));
7885 put_u64(&mut out, page.offset);
7886 put_u32(&mut out, page.length);
7887 put_u64(&mut out, page.hash);
7888 }
7889 }
7890 }
7891 for distinct in &table.distincts {
7892 match distinct {
7893 None => out.push(0),
7894 Some(count) => {
7895 out.push(1);
7896 put_u64(&mut out, *count);
7897 }
7898 }
7899 }
7900 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7901 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7902 for stripe in &table.stripes {
7903 put_u32(
7904 &mut out,
7905 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7906 );
7907 for &rows in &stripe.parts {
7908 put_u32(&mut out, rows);
7909 }
7910 put_u64(&mut out, stripe.index.offset);
7911 put_u32(&mut out, stripe.index.length);
7912 for page in &stripe.pages {
7913 put_u64(&mut out, page.offset);
7914 put_u32(&mut out, page.length);
7915 }
7916 for (column, ((field, dictionary), membership)) in
7921 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7922 {
7923 if !coded_type(&field.ty) || dictionary.is_none() {
7924 continue;
7925 }
7926 let page = match membership {
7927 Some(page) => page,
7928 None if table.demoted.get(column).copied().unwrap_or(false) => {
7929 Page { offset: HEADER, length: 0, hash: 0 }
7930 }
7931 None => return Err(invalid("string page has no code membership index")),
7932 };
7933 put_u64(&mut out, page.offset);
7934 put_u32(&mut out, page.length);
7935 put_u64(&mut out, page.hash);
7936 }
7937 for sieve in stripe.sieves.slots() {
7938 match sieve {
7939 None => out.push(0),
7940 Some(page) => {
7941 out.push(1);
7942 put_u64(&mut out, page.offset);
7943 put_u32(&mut out, page.length);
7944 put_u64(&mut out, page.hash);
7945 }
7946 }
7947 }
7948 for held in stripe.part_ranges.slots() {
7949 match held {
7950 None => out.push(0),
7951 Some(page) => {
7952 out.push(1);
7953 put_u64(&mut out, page.offset);
7954 put_u32(&mut out, page.length);
7955 put_u64(&mut out, page.hash);
7956 }
7957 }
7958 }
7959 for range in stripe.zone.columns() {
7960 put_bound(&mut out, range.low.as_ref())?;
7961 put_bound(&mut out, range.high.as_ref())?;
7962 put_u32(
7963 &mut out,
7964 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7965 );
7966 out.push(u8::from(range.exact));
7967 match range.sum {
7968 None => out.push(0),
7969 Some(total) => {
7970 out.push(1);
7971 out.extend_from_slice(&total.to_le_bytes());
7972 }
7973 }
7974 }
7975 }
7976 out.extend_from_slice(FREQUENCIES);
7977 put_u16(
7978 &mut out,
7979 u16::try_from(table.frequencies.len())
7980 .map_err(|_| invalid("too many frequency columns"))?,
7981 );
7982 for summary in &table.frequencies {
7983 let summary = match summary {
7984 None => {
7985 out.push(0);
7986 continue;
7987 }
7988 Some(Frequencies::Held(summary)) => summary,
7989 Some(Frequencies::Stored { .. }) => {
7991 return Err(invalid("a synopsis left in the file cannot be written back"));
7992 }
7993 };
7994 out.push(1);
7995 put_u64(&mut out, summary.omitted_max);
7996 put_u32(
7997 &mut out,
7998 u32::try_from(summary.entries.len())
7999 .map_err(|_| invalid("too many frequency entries"))?,
8000 );
8001 for entry in &summary.entries {
8002 match entry.value {
8003 FrequencyValue::Null => out.push(0),
8004 FrequencyValue::Integer(value) => {
8005 out.push(1);
8006 out.extend_from_slice(&value.to_le_bytes());
8007 }
8008 FrequencyValue::Code(value) => {
8009 out.push(2);
8010 put_u32(&mut out, value);
8011 }
8012 }
8013 put_u64(&mut out, entry.count);
8014 }
8015 put_u32(
8016 &mut out,
8017 u32::try_from(summary.ordinals.len())
8018 .map_err(|_| invalid("too many frequency ordinals"))?,
8019 );
8020 let mut previous = 0_u64;
8021 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
8022 let delta = if at == 0 {
8023 ordinal
8024 } else {
8025 ordinal
8026 .checked_sub(previous)
8027 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
8028 };
8029 if at != 0 && delta == 0 {
8030 return Err(invalid("frequency ordinals are not unique"));
8031 }
8032 put_var_u64(&mut out, delta);
8033 previous = ordinal;
8034 }
8035 if summary.ordinal_entries.len() != summary.ordinals.len() {
8036 return Err(invalid("frequency ordinal values have a different length"));
8037 }
8038 for &entry in &summary.ordinal_entries {
8039 if entry as usize >= summary.entries.len() {
8040 return Err(invalid("frequency ordinal value is outside its entries"));
8041 }
8042 put_u16(&mut out, entry);
8043 }
8044 }
8045 if !table.pair_frequencies.is_empty() {
8046 out.extend_from_slice(PAIR_FREQUENCIES);
8047 put_u16(
8048 &mut out,
8049 u16::try_from(table.pair_frequencies.len())
8050 .map_err(|_| invalid("too many pair frequency summaries"))?,
8051 );
8052 for summary in &table.pair_frequencies {
8053 put_u16(&mut out, summary.first);
8054 put_u16(&mut out, summary.second);
8055 put_u64(&mut out, summary.omitted_max);
8056 put_u16(
8057 &mut out,
8058 u16::try_from(summary.entries.len())
8059 .map_err(|_| invalid("too many pair frequency entries"))?,
8060 );
8061 for entry in &summary.entries {
8062 put_u16(&mut out, entry.first_entry);
8063 match entry.second {
8064 None => out.push(0),
8065 Some(code) => {
8066 out.push(1);
8067 put_u32(&mut out, code);
8068 }
8069 }
8070 put_u64(&mut out, entry.count);
8071 }
8072 }
8073 }
8074 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
8075 if text_columns != 0 {
8076 out.extend_from_slice(FREQUENCY_TEXTS);
8077 put_u16(
8078 &mut out,
8079 u16::try_from(text_columns)
8080 .map_err(|_| invalid("too many string frequency columns"))?,
8081 );
8082 for (column, texts) in table.frequency_texts.iter().enumerate() {
8083 if texts.is_empty() {
8084 continue;
8085 }
8086 put_u16(
8087 &mut out,
8088 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
8089 );
8090 put_u16(
8091 &mut out,
8092 u16::try_from(texts.len())
8093 .map_err(|_| invalid("too many frequency text entries"))?,
8094 );
8095 for text in texts {
8096 match text {
8097 None => out.push(0),
8098 Some(text) => {
8099 out.push(1);
8100 put_u32(
8101 &mut out,
8102 u32::try_from(text.len())
8103 .map_err(|_| invalid("frequency text is too long"))?,
8104 );
8105 out.extend_from_slice(text);
8106 }
8107 }
8108 }
8109 }
8110 }
8111 if let Some(summary) = &table.host_groups {
8112 out.extend_from_slice(HOST_GROUPS);
8113 put_u16(
8114 &mut out,
8115 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
8116 );
8117 put_u64(&mut out, summary.omitted_max);
8118 put_u16(
8119 &mut out,
8120 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
8121 );
8122 for entry in &summary.entries {
8123 put_u32(
8124 &mut out,
8125 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
8126 );
8127 out.extend_from_slice(entry.host.as_bytes());
8128 put_u64(&mut out, entry.count);
8129 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
8130 put_u32(
8131 &mut out,
8132 u32::try_from(entry.minimum.len())
8133 .map_err(|_| invalid("host minimum is too long"))?,
8134 );
8135 out.extend_from_slice(entry.minimum.as_bytes());
8136 }
8137 }
8138 if let Some(clustering) = &table.clustering {
8141 out.extend_from_slice(CLUSTERING);
8142 out.push(clustering.width().tag());
8143 put_u16(
8144 &mut out,
8145 u16::try_from(clustering.columns().len())
8146 .map_err(|_| invalid("too many clustering columns"))?,
8147 );
8148 for &column in clustering.columns() {
8149 put_u16(
8150 &mut out,
8151 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
8152 );
8153 }
8154 }
8155 let demoted = (0..table.fields.len())
8156 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
8157 .collect::<Vec<_>>();
8158 if !demoted.is_empty() {
8159 out.extend_from_slice(DEMOTED);
8160 put_u16(
8161 &mut out,
8162 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
8163 );
8164 for column in demoted {
8165 put_u16(
8166 &mut out,
8167 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
8168 );
8169 }
8170 }
8171 out.extend_from_slice(SECTIONS);
8177 put_u64(&mut out, table.generation);
8178 put_u16(
8179 &mut out,
8180 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
8181 );
8182 for held in &table.sections {
8183 held.encode(&mut out)?;
8184 }
8185 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
8186 out.extend_from_slice(DICTIONARY_PAYLOADS);
8187 put_u16(
8188 &mut out,
8189 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
8190 );
8191 for at in 0..table.fields.len() {
8192 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
8193 }
8194 }
8195 Ok(out)
8196}
8197
8198fn signed_integer(ty: &LogicalType) -> bool {
8207 matches!(
8208 ty,
8209 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
8210 )
8211}
8212
8213fn integer_or_date(ty: &LogicalType) -> bool {
8214 matches!(
8215 ty,
8216 LogicalType::TinyInt
8217 | LogicalType::SmallInt
8218 | LogicalType::Integer
8219 | LogicalType::BigInt
8220 | LogicalType::UTinyInt
8221 | LogicalType::USmallInt
8222 | LogicalType::UInteger
8223 | LogicalType::UBigInt
8224 | LogicalType::Date
8225 )
8226}
8227
8228fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
8229 table
8230 .fields
8231 .iter()
8232 .enumerate()
8233 .map(|(column, field)| {
8234 if !integer_or_date(&field.ty) {
8235 return None;
8236 }
8237 let mut low: Option<i128> = None;
8238 let mut high: Option<i128> = None;
8239 for stripe in &table.stripes {
8240 let range = stripe.zone.column(column)?;
8241 if !range.exact {
8242 return None;
8243 }
8244 match (range.low.as_ref(), range.high.as_ref()) {
8245 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8246 low = Some(low.map_or(*small, |held| held.min(*small)));
8247 high = Some(high.map_or(*large, |held| held.max(*large)));
8248 }
8249 (None, None) if stripe.rows == range.nulls => {}
8250 _ => return None,
8251 }
8252 }
8253 Some(low.zip(high))
8254 })
8255 .collect()
8256}
8257
8258fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8259 reader
8260 .table
8261 .fields
8262 .iter()
8263 .enumerate()
8264 .map(|(column, field)| {
8265 if !integer_or_date(&field.ty) {
8266 return Ok(None);
8267 }
8268 match reader.exact_extremes(column)? {
8269 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8270 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8271 _ => Ok(None),
8272 }
8273 })
8274 .collect()
8275}
8276
8277fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8278 table
8279 .fields
8280 .iter()
8281 .enumerate()
8282 .map(|(column, field)| {
8283 if !integer_or_date(&field.ty) {
8284 return None;
8285 }
8286 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8287 return None;
8288 };
8289 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8290 return None;
8291 }
8292 let entries = summary
8293 .entries
8294 .iter()
8295 .map(|entry| {
8296 let value = match entry.value {
8297 FrequencyValue::Null => None,
8298 FrequencyValue::Integer(value) => Some(value),
8299 FrequencyValue::Code(_) => return None,
8300 };
8301 Some((value, entry.count))
8302 })
8303 .collect::<Option<Vec<_>>>()?;
8304 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8305 (rows == table.rows as u64).then_some(entries)
8306 })
8307 .collect()
8308}
8309
8310fn integer_value(bits: u64, signed: bool) -> FrequencyValue {
8317 if signed {
8318 FrequencyValue::Integer(i128::from(bits as i64))
8319 } else {
8320 FrequencyValue::Integer(i128::from(bits))
8321 }
8322}
8323
8324fn frequency_bits(value: &Value) -> Option<u64> {
8325 Some(match value {
8326 Value::TinyInt(value) => i64::from(*value) as u64,
8327 Value::SmallInt(value) => i64::from(*value) as u64,
8328 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8329 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8330 Value::UTinyInt(value) => u64::from(*value),
8331 Value::USmallInt(value) => u64::from(*value),
8332 Value::UInteger(value) => u64::from(*value),
8333 Value::UBigInt(value) => *value,
8334 _ => return None,
8335 })
8336}
8337
8338fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8339 Some(match value {
8340 Value::Null => None,
8341 Value::TinyInt(value) => Some(i128::from(*value)),
8342 Value::SmallInt(value) => Some(i128::from(*value)),
8343 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8344 Value::BigInt(value) => Some(i128::from(*value)),
8345 Value::UTinyInt(value) => Some(i128::from(*value)),
8346 Value::USmallInt(value) => Some(i128::from(*value)),
8347 Value::UInteger(value) => Some(i128::from(*value)),
8348 Value::UBigInt(value) => Some(i128::from(*value)),
8349 _ => return None,
8350 })
8351}
8352
8353fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8354 reader
8355 .table
8356 .fields
8357 .iter()
8358 .enumerate()
8359 .map(|(column, field)| {
8360 if !integer_or_date(&field.ty) {
8361 return Ok(None);
8362 }
8363 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8364 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8365 return Ok(None);
8366 }
8367 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8368 let Some(entries) = entries
8369 .iter()
8370 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8371 .collect::<Option<Vec<_>>>()
8372 else {
8373 return Ok(None);
8374 };
8375 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8376 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8377 })
8378 .collect()
8379}
8380
8381fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8382 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8383 let range = stripe.zone.column(column)?;
8384 let sum = sum.checked_add(range.sum?)?;
8385 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8386 Some((sum, count.checked_add(nonnull)?))
8387 })
8388}
8389
8390fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8391 table
8392 .fields
8393 .iter()
8394 .enumerate()
8395 .map(|(column, field)| {
8396 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8397 })
8398 .collect()
8399}
8400
8401fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8402 reader
8403 .table
8404 .fields
8405 .iter()
8406 .enumerate()
8407 .map(
8408 |(column, field)| {
8409 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8410 },
8411 )
8412 .collect()
8413}
8414
8415fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8416 let mut out = CATALOG.to_vec();
8417 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8418 for entry in entries {
8419 let name = entry.name.as_bytes();
8420 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8421 out.extend_from_slice(name);
8422 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8423 put_u16(
8424 &mut out,
8425 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8426 );
8427 for field in &entry.fields {
8428 let name = field.name.as_bytes();
8429 put_u16(
8430 &mut out,
8431 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8432 );
8433 out.extend_from_slice(name);
8434 put_type(&mut out, &field.ty)?;
8435 out.push(u8::from(field.not_null));
8436 }
8437 put_u64(&mut out, entry.directory.offset);
8438 put_u32(&mut out, entry.directory.length);
8439 put_u64(&mut out, entry.directory.hash);
8440 }
8441 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8442 for view in views {
8443 let name = view.name.as_bytes();
8444 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8445 out.extend_from_slice(name);
8446 put_long_text(&mut out, &view.sql, "view body")?;
8447 put_long_text(&mut out, &view.statement, "view statement")?;
8448 put_u16(
8449 &mut out,
8450 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8451 );
8452 for alias in &view.aliases {
8453 let alias = alias.as_bytes();
8454 put_u16(
8455 &mut out,
8456 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8457 );
8458 out.extend_from_slice(alias);
8459 }
8460 put_u16(
8461 &mut out,
8462 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8463 );
8464 for field in &view.columns {
8465 let name = field.name.as_bytes();
8466 put_u16(
8467 &mut out,
8468 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8469 );
8470 out.extend_from_slice(name);
8471 put_type(&mut out, &field.ty)?;
8472 out.push(u8::from(field.not_null));
8473 }
8474 }
8475 out.extend_from_slice(NONZERO_COUNTS);
8476 for entry in entries {
8477 if entry.nonzero.len() != entry.fields.len() {
8478 return Err(invalid("nonzero count width differs from schema"));
8479 }
8480 for count in &entry.nonzero {
8481 match count {
8482 None => out.push(0),
8483 Some(count) => {
8484 out.push(1);
8485 put_u64(&mut out, *count);
8486 }
8487 }
8488 }
8489 }
8490 out.extend_from_slice(AGGREGATE_SUMS);
8491 for entry in entries {
8492 if entry.aggregates.len() != entry.fields.len() {
8493 return Err(invalid("aggregate sum width differs from schema"));
8494 }
8495 for summary in &entry.aggregates {
8496 match summary {
8497 None => out.push(0),
8498 Some((sum, count)) => {
8499 out.push(1);
8500 out.extend_from_slice(&sum.to_le_bytes());
8501 put_u64(&mut out, *count);
8502 }
8503 }
8504 }
8505 }
8506 out.extend_from_slice(DISTINCT_COUNTS);
8507 for entry in entries {
8508 if entry.distincts.len() != entry.fields.len() {
8509 return Err(invalid("distinct count width differs from schema"));
8510 }
8511 for count in &entry.distincts {
8512 match count {
8513 None => out.push(0),
8514 Some(count) => {
8515 if *count > entry.rows as u64 {
8516 return Err(invalid("distinct count exceeds table rows"));
8517 }
8518 out.push(1);
8519 put_u64(&mut out, *count);
8520 }
8521 }
8522 }
8523 }
8524 out.extend_from_slice(INTEGER_EXTREMES);
8525 for entry in entries {
8526 if entry.extremes.len() != entry.fields.len() {
8527 return Err(invalid("integer extremes width differs from schema"));
8528 }
8529 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8530 match extremes {
8531 None => out.push(0),
8532 Some(None) if integer_or_date(&field.ty) => out.push(1),
8533 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8534 out.push(2);
8535 out.extend_from_slice(&low.to_le_bytes());
8536 out.extend_from_slice(&high.to_le_bytes());
8537 }
8538 _ => return Err(invalid("integer extremes type or range differs")),
8539 }
8540 }
8541 }
8542 out.extend_from_slice(COMPLETE_FREQUENCIES);
8543 for entry in entries {
8544 if entry.frequencies.len() != entry.fields.len() {
8545 return Err(invalid("numeric frequency width differs from schema"));
8546 }
8547 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8548 match frequencies {
8549 None => out.push(0),
8550 Some(entries)
8551 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8552 {
8553 let mut total = 0_u64;
8554 for (at, (value, count)) in entries.iter().enumerate() {
8555 if entries[..at].iter().any(|(held, _)| held == value) {
8556 return Err(invalid("numeric frequency value repeats"));
8557 }
8558 total = total
8559 .checked_add(*count)
8560 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8561 }
8562 if total != entry.rows as u64 {
8563 return Err(invalid("numeric frequencies do not cover table rows"));
8564 }
8565 out.push(1);
8566 out.push(entries.len() as u8);
8567 for (value, count) in entries {
8568 match value {
8569 None => out.push(0),
8570 Some(value) => {
8571 out.push(1);
8572 out.extend_from_slice(&value.to_le_bytes());
8573 }
8574 }
8575 put_u64(&mut out, *count);
8576 }
8577 }
8578 _ => return Err(invalid("numeric frequency type or width differs")),
8579 }
8580 }
8581 }
8582 Ok(out)
8583}
8584
8585fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8587 let bytes = text.as_bytes();
8588 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8589 out.extend_from_slice(bytes);
8590 Ok(())
8591}
8592
8593fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8596 let mut cur = Cursor::new(bytes);
8597 if cur.take(8)? != CATALOG {
8598 return Err(invalid("catalog magic differs"));
8599 }
8600 let count = cur.u32()? as usize;
8601 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8602 for _ in 0..count {
8603 let name = cur.text()?;
8604 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8605 let width = cur.u16()? as usize;
8606 let mut fields = Vec::with_capacity(width);
8607 for _ in 0..width {
8608 let name = cur.text()?;
8609 let ty = read_type(&mut cur)?;
8610 let not_null = match cur.u8()? {
8611 0 => false,
8612 1 => true,
8613 _ => return Err(invalid("nullability flag differs")),
8614 };
8615 fields.push(Field { name, ty, not_null });
8616 }
8617 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8618 let end = directory
8619 .offset
8620 .checked_add(u64::from(directory.length))
8621 .ok_or_else(|| invalid("table directory offset overflow"))?;
8622 if directory.offset < HEADER
8623 || end > size
8624 || directory.length as usize > MAX_DIRECTORY
8625 || directory.length == 0
8626 {
8627 return Err(invalid("table directory range is outside the file"));
8628 }
8629 if entries.iter().any(|held| held.name == name) {
8630 return Err(invalid("two tables in the catalog have the same name"));
8631 }
8632 let nonzero = vec![None; fields.len()];
8633 let aggregates = vec![None; fields.len()];
8634 let distincts = vec![None; fields.len()];
8635 let extremes = vec![None; fields.len()];
8636 let frequencies = vec![None; fields.len()];
8637 entries.push(Entry {
8638 name,
8639 fields,
8640 rows,
8641 directory,
8642 nonzero,
8643 aggregates,
8644 distincts,
8645 extremes,
8646 frequencies,
8647 });
8648 }
8649 let count = if cur.done() { 0 } else { cur.u32()? as usize };
8654 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8655 for _ in 0..count {
8656 let name = cur.text()?;
8657 let sql = cur.long_text()?;
8658 let statement = cur.long_text()?;
8659 let width = cur.u16()? as usize;
8660 let mut aliases = Vec::with_capacity(width);
8661 for _ in 0..width {
8662 aliases.push(cur.text()?);
8663 }
8664 let width = cur.u16()? as usize;
8665 let mut columns = Vec::with_capacity(width);
8666 for _ in 0..width {
8667 let name = cur.text()?;
8668 let ty = read_type(&mut cur)?;
8669 let not_null = match cur.u8()? {
8670 0 => false,
8671 1 => true,
8672 _ => return Err(invalid("nullability flag differs")),
8673 };
8674 columns.push(Field { name, ty, not_null });
8675 }
8676 if views.iter().any(|held| held.name == name) {
8680 return Err(invalid("two views in the catalog have the same name"));
8681 }
8682 if entries.iter().any(|held| held.name == name) {
8683 return Err(invalid("a table and a view in the catalog have the same name"));
8684 }
8685 views.push(ViewEntry { name, sql, statement, aliases, columns });
8686 }
8687 if !cur.done() {
8688 if cur.take(8)? != NONZERO_COUNTS {
8689 return Err(invalid("catalog extension magic differs"));
8690 }
8691 for entry in &mut entries {
8692 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8693 *count = match cur.u8()? {
8694 0 => None,
8695 1 if matches!(
8696 field.ty,
8697 LogicalType::TinyInt
8698 | LogicalType::SmallInt
8699 | LogicalType::Integer
8700 | LogicalType::BigInt
8701 | LogicalType::UTinyInt
8702 | LogicalType::USmallInt
8703 | LogicalType::UInteger
8704 | LogicalType::UBigInt
8705 ) =>
8706 {
8707 let value = cur.u64()?;
8708 if value > entry.rows as u64 {
8709 return Err(invalid("nonzero count exceeds rows"));
8710 }
8711 Some(value)
8712 }
8713 _ => return Err(invalid("nonzero count tag or column type differs")),
8714 };
8715 }
8716 }
8717 }
8718 if !cur.done() {
8719 if cur.take(8)? != AGGREGATE_SUMS {
8720 return Err(invalid("aggregate catalog extension magic differs"));
8721 }
8722 for entry in &mut entries {
8723 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8724 *summary = match cur.u8()? {
8725 0 => None,
8726 1 if signed_integer(&field.ty) => {
8727 let sum = i128::from_le_bytes(
8728 cur.take(16)?
8729 .try_into()
8730 .map_err(|_| invalid("aggregate sum is truncated"))?,
8731 );
8732 let count = cur.u64()?;
8733 if count > entry.rows as u64 {
8734 return Err(invalid("aggregate count exceeds table rows"));
8735 }
8736 Some((sum, count))
8737 }
8738 _ => return Err(invalid("aggregate sum tag or column type differs")),
8739 };
8740 }
8741 }
8742 }
8743 if !cur.done() {
8744 if cur.take(8)? != DISTINCT_COUNTS {
8745 return Err(invalid("distinct catalog extension magic differs"));
8746 }
8747 for entry in &mut entries {
8748 for count in &mut entry.distincts {
8749 *count = match cur.u8()? {
8750 0 => None,
8751 1 => {
8752 let value = cur.u64()?;
8753 if value > entry.rows as u64 {
8754 return Err(invalid("distinct count exceeds table rows"));
8755 }
8756 Some(value)
8757 }
8758 _ => return Err(invalid("distinct count tag differs")),
8759 };
8760 }
8761 }
8762 }
8763 if !cur.done() {
8764 if cur.take(8)? != INTEGER_EXTREMES {
8765 return Err(invalid("integer extremes catalog extension magic differs"));
8766 }
8767 for entry in &mut entries {
8768 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8769 *extremes = match cur.u8()? {
8770 0 => None,
8771 1 if integer_or_date(&field.ty) => Some(None),
8772 2 if integer_or_date(&field.ty) => {
8773 let low = i128::from_le_bytes(
8774 cur.take(16)?
8775 .try_into()
8776 .map_err(|_| invalid("minimum is truncated"))?,
8777 );
8778 let high = i128::from_le_bytes(
8779 cur.take(16)?
8780 .try_into()
8781 .map_err(|_| invalid("maximum is truncated"))?,
8782 );
8783 if low > high {
8784 return Err(invalid("integer extremes are reversed"));
8785 }
8786 Some(Some((low, high)))
8787 }
8788 _ => return Err(invalid("integer extremes tag or type differs")),
8789 };
8790 }
8791 }
8792 }
8793 if !cur.done() {
8794 if cur.take(8)? != COMPLETE_FREQUENCIES {
8795 return Err(invalid("numeric frequency catalog extension magic differs"));
8796 }
8797 for entry in &mut entries {
8798 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8799 *frequencies = match cur.u8()? {
8800 0 => None,
8801 1 if integer_or_date(&field.ty) => {
8802 let len = cur.u8()? as usize;
8803 if len > MAX_CATALOG_FREQUENCIES {
8804 return Err(invalid("too many catalog numeric frequencies"));
8805 }
8806 let mut values = Vec::with_capacity(len);
8807 let mut total = 0_u64;
8808 for _ in 0..len {
8809 let value = match cur.u8()? {
8810 0 => None,
8811 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8812 |_| invalid("numeric frequency value is truncated"),
8813 )?)),
8814 _ => return Err(invalid("numeric frequency value tag differs")),
8815 };
8816 if values.iter().any(|(held, _)| *held == value) {
8817 return Err(invalid("numeric frequency value repeats"));
8818 }
8819 let count = cur.u64()?;
8820 total = total
8821 .checked_add(count)
8822 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8823 values.push((value, count));
8824 }
8825 if total != entry.rows as u64 {
8826 return Err(invalid("numeric frequencies do not cover table rows"));
8827 }
8828 Some(values)
8829 }
8830 _ => return Err(invalid("numeric frequency tag or type differs")),
8831 };
8832 }
8833 }
8834 }
8835 if !cur.done() {
8836 return Err(invalid("catalog has trailing bytes"));
8837 }
8838 Ok((entries, views))
8839}
8840
8841struct Cursor<'a> {
8849 bytes: &'a [u8],
8850 at: usize,
8851 window: Option<Window<'a>>,
8852}
8853
8854struct Window<'a> {
8856 file: &'a File,
8857 offset: u64,
8858 length: usize,
8859 start: usize,
8861 held: Vec<u8>,
8862 size: usize,
8864}
8865
8866const DIRECTORY_WINDOW: usize = 64 << 10;
8868
8869impl<'a> Cursor<'a> {
8870 fn new(bytes: &'a [u8]) -> Self {
8871 Self { bytes, at: 0, window: None }
8872 }
8873
8874 fn over(file: &'a File, offset: u64, length: usize) -> Self {
8876 let window =
8877 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8878 Self { bytes: &[], at: 0, window: Some(window) }
8879 }
8880
8881 fn len(&self) -> usize {
8883 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8884 }
8885
8886 fn ensure(&mut self, len: usize) -> Result<()> {
8888 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8889 if end > self.len() {
8890 return Err(invalid("directory is truncated"));
8891 }
8892 let Some(window) = &mut self.window else { return Ok(()) };
8893 if self.at < window.start || end > window.start + window.held.len() {
8894 let want = len.max(window.size).min(window.length - self.at);
8895 window.start = self.at;
8896 window.held.resize(want, 0);
8897 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8898 }
8899 Ok(())
8900 }
8901
8902 fn held(&self, at: usize, len: usize) -> &[u8] {
8904 match &self.window {
8905 Some(window) => &window.held[at - window.start..at - window.start + len],
8906 None => &self.bytes[at..at + len],
8907 }
8908 }
8909
8910 #[inline]
8912 fn peek(&mut self, len: usize) -> Result<&[u8]> {
8913 if self.window.is_none() {
8914 let bytes = self.bytes;
8915 return Ok(&bytes[self.at..self.end(len)?]);
8916 }
8917 self.ensure(len)?;
8918 Ok(self.held(self.at, len))
8919 }
8920
8921 #[inline]
8927 fn take(&mut self, len: usize) -> Result<&[u8]> {
8928 if self.window.is_none() {
8929 let bytes = self.bytes;
8930 let (at, end) = (self.at, self.end(len)?);
8931 self.at = end;
8932 return Ok(&bytes[at..end]);
8933 }
8934 self.take_windowed(len)
8935 }
8936
8937 fn skip(&mut self, len: usize) -> Result<()> {
8939 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8940 if end > self.len() {
8941 return Err(invalid("directory is truncated"));
8942 }
8943 self.at = end;
8944 Ok(())
8945 }
8946
8947 fn skip_bound(&mut self) -> Result<()> {
8948 match self.u8()? {
8949 0 => Ok(()),
8950 1 => self.skip(16),
8951 2 => self.skip(8),
8952 3 => {
8953 let length = self.u32()? as usize;
8954 self.skip(length)
8955 }
8956 4 => self.skip(17),
8957 _ => Err(invalid("a stored bound has an unknown tag")),
8958 }
8959 }
8960
8961 #[inline]
8963 fn end(&self, len: usize) -> Result<usize> {
8964 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8965 if end > self.bytes.len() {
8966 return Err(invalid("directory is truncated"));
8967 }
8968 Ok(end)
8969 }
8970
8971 #[inline(never)]
8973 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8974 self.ensure(len)?;
8975 self.at += len;
8976 Ok(self.held(self.at - len, len))
8977 }
8978 #[inline]
8979 fn u8(&mut self) -> Result<u8> {
8980 Ok(self.take(1)?[0])
8981 }
8982 #[inline]
8983 fn u16(&mut self) -> Result<u16> {
8984 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8985 }
8986 #[inline]
8987 fn u32(&mut self) -> Result<u32> {
8988 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8989 }
8990 #[inline]
8991 fn u64(&mut self) -> Result<u64> {
8992 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8993 }
8994 fn var_u64(&mut self) -> Result<u64> {
8995 let mut value = 0_u64;
8996 for shift in (0..=63).step_by(7) {
8997 let byte = self.u8()?;
8998 let part = u64::from(byte & 0x7f);
8999 if shift == 63 && part > 1 {
9000 return Err(invalid("frequency ordinal varint overflows"));
9001 }
9002 value |= part << shift;
9003 if byte & 0x80 == 0 {
9004 return Ok(value);
9005 }
9006 }
9007 Err(invalid("frequency ordinal varint is too long"))
9008 }
9009 fn bound(&mut self) -> Result<Option<Bound>> {
9018 let rest = self.len().saturating_sub(self.at);
9019 let mut want = 32;
9020 loop {
9021 let offered = self.peek(want.min(rest))?;
9022 let mut used = 0;
9023 match bounds::get(offered, &mut used) {
9024 Ok(bound) => {
9025 self.at += used;
9026 return Ok(bound);
9027 }
9028 Err(_) if want < rest => want *= 2,
9029 Err(error) => return Err(error),
9030 }
9031 }
9032 }
9033 fn text(&mut self) -> Result<String> {
9034 let len = self.u16()? as usize;
9035 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
9036 }
9037 fn done(&self) -> bool {
9040 self.at >= self.len()
9041 }
9042 fn long_text(&mut self) -> Result<String> {
9049 let len = self.u32()? as usize;
9050 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
9051 }
9052}
9053
9054fn decode_summary(
9056 cur: &mut Cursor<'_>,
9057 field: &Field,
9058 rows: usize,
9059 values: bool,
9060) -> Result<Option<FrequencySummary>> {
9061 Ok(match cur.u8()? {
9062 0 => None,
9063 1 => {
9064 let omitted_max = cur.u64()?;
9065 let count = cur.u32()? as usize;
9066 if count > FREQUENCY_ENTRIES {
9067 return Err(invalid("frequency entry count exceeds its bound"));
9068 }
9069 let mut entries = Vec::with_capacity(count);
9070 for _ in 0..count {
9072 let value = match cur.u8()? {
9073 0 => FrequencyValue::Null,
9074 1 => FrequencyValue::Integer(i128::from_le_bytes(
9075 cur.take(16)?.try_into().expect("sixteen bytes"),
9076 )),
9077 2 => FrequencyValue::Code(cur.u32()?),
9078 _ => return Err(invalid("frequency value tag differs")),
9079 };
9080 let valid = matches!(
9081 (&field.ty, value),
9082 (_, FrequencyValue::Null)
9083 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
9084 | (
9085 LogicalType::TinyInt
9086 | LogicalType::SmallInt
9087 | LogicalType::Integer
9088 | LogicalType::BigInt
9089 | LogicalType::UTinyInt
9090 | LogicalType::USmallInt
9091 | LogicalType::UInteger
9092 | LogicalType::UBigInt
9093 | LogicalType::Date
9094 | LogicalType::Timestamp,
9095 FrequencyValue::Integer(_),
9096 )
9097 );
9098 if !valid {
9099 return Err(invalid("frequency value does not match its column"));
9100 }
9101 let count = cur.u64()?;
9102 if count == 0 || count > rows as u64 {
9103 return Err(invalid("frequency count is outside the table"));
9104 }
9105 entries.push(FrequencyEntry { value, count });
9106 }
9107 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9108 return Err(invalid("frequency entries are not descending"));
9109 }
9110 let ordinals = {
9111 let ordinal_count = cur.u32()? as usize;
9112 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
9113 return Err(invalid("frequency ordinal count exceeds its bound"));
9114 }
9115 let mut ordinals = Vec::with_capacity(ordinal_count);
9116 let mut previous = 0_u64;
9117 for at in 0..ordinal_count {
9118 let delta = cur.var_u64()?;
9119 if at != 0 && delta == 0 {
9120 return Err(invalid("frequency ordinals are not increasing"));
9121 }
9122 let ordinal = if at == 0 {
9123 delta
9124 } else {
9125 previous
9126 .checked_add(delta)
9127 .ok_or_else(|| invalid("frequency ordinal overflows"))?
9128 };
9129 if ordinal >= rows as u64 {
9130 return Err(invalid("frequency ordinal is outside the table"));
9131 }
9132 ordinals.push(ordinal);
9133 previous = ordinal;
9134 }
9135 ordinals
9136 };
9137 let ordinal_entries = if values {
9138 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
9139 for _ in 0..ordinals.len() {
9140 let entry = cur.u16()?;
9141 if entry as usize >= entries.len() {
9142 return Err(invalid("frequency ordinal value is outside its entries"));
9143 }
9144 ordinal_entries.push(entry);
9145 }
9146 ordinal_entries
9147 } else {
9148 Vec::new()
9149 };
9150 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
9151 }
9152 _ => return Err(invalid("frequency summary tag differs")),
9153 })
9154}
9155
9156fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
9159 match cur.u8()? {
9160 0 => Ok(()),
9161 1 => {
9162 cur.skip(8)?;
9163 let entries = cur.u32()? as usize;
9164 if entries > FREQUENCY_ENTRIES {
9165 return Err(invalid("frequency entry count exceeds its bound"));
9166 }
9167 for _ in 0..entries {
9168 match cur.u8()? {
9169 0 => {}
9170 1 => cur.skip(16)?,
9171 2 => cur.skip(4)?,
9172 _ => return Err(invalid("frequency value tag differs")),
9173 }
9174 cur.skip(8)?;
9175 }
9176 let ordinals = cur.u32()? as usize;
9177 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
9178 return Err(invalid("frequency ordinal count exceeds its bound"));
9179 }
9180 for _ in 0..ordinals {
9181 cur.var_u64()?;
9182 }
9183 if values {
9184 cur.skip(ordinals * 2)?;
9185 }
9186 Ok(())
9187 }
9188 _ => Err(invalid("frequency summary tag differs")),
9189 }
9190}
9191
9192fn quick_nonzero(
9196 mut cur: Cursor<'_>,
9197 name: &str,
9198 fields: &[Field],
9199 rows: usize,
9200 wanted: usize,
9201) -> Result<Option<u64>> {
9202 if cur.take(8)? != DIRECTORY || cur.text()? != name {
9203 return Err(invalid("table directory differs from the catalog"));
9204 }
9205 let width = cur.u16()? as usize;
9206 if width != fields.len() {
9207 return Err(invalid("table directory width differs from the catalog"));
9208 }
9209 for field in fields {
9210 let stored =
9211 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9212 if &stored != field {
9213 return Err(invalid("table directory schema differs from the catalog"));
9214 }
9215 }
9216 let mut dictionaries = Vec::with_capacity(width);
9217 for field in fields {
9218 let held = match cur.u8()? {
9219 0 => false,
9220 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9221 cur.skip(20)?;
9222 true
9223 }
9224 _ => return Err(invalid("dictionary page tag differs")),
9225 };
9226 dictionaries.push(held);
9227 }
9228 for _ in 0..width {
9229 match cur.u8()? {
9230 0 => {}
9231 1 => cur.skip(8)?,
9232 _ => return Err(invalid("distinct count tag differs")),
9233 }
9234 }
9235 if cur.u64()? != rows as u64 {
9236 return Err(invalid("table row count differs from the catalog"));
9237 }
9238 let stripes = cur.u32()? as usize;
9239 let mut total = 0_usize;
9240 let mut nulls = 0_u64;
9241 for _ in 0..stripes {
9242 let parts = cur.u32()? as usize;
9243 if parts == 0 || parts > STRIPE_PARTS {
9244 return Err(invalid("stripe part count is outside its bound"));
9245 }
9246 let mut stripe_rows = 0_usize;
9247 for _ in 0..parts {
9248 stripe_rows = stripe_rows
9249 .checked_add(cur.u32()? as usize)
9250 .ok_or_else(|| invalid("stripe row count overflow"))?;
9251 }
9252 total =
9253 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9254 cur.skip(12 + width * 12)?;
9255 for (field, held) in fields.iter().zip(&dictionaries) {
9256 if coded_type(&field.ty) && *held {
9257 cur.skip(20)?;
9258 }
9259 }
9260 for _ in 0..width * 2 {
9261 match cur.u8()? {
9262 0 => {}
9263 1 => cur.skip(20)?,
9264 _ => return Err(invalid("stripe page tag differs")),
9265 }
9266 }
9267 for column in 0..width {
9268 cur.skip_bound()?;
9269 cur.skip_bound()?;
9270 let count = cur.u32()? as u64;
9271 if count > stripe_rows as u64 {
9272 return Err(invalid("null count exceeds stripe rows"));
9273 }
9274 if column == wanted {
9275 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9276 }
9277 cur.skip(1)?;
9278 match cur.u8()? {
9279 0 => {}
9280 1 => cur.skip(16)?,
9281 _ => return Err(invalid("a stripe sum has an unknown tag")),
9282 }
9283 }
9284 }
9285 if total != rows {
9286 return Err(invalid("table row count differs from stripes"));
9287 }
9288 if cur.done() {
9289 return Ok(None);
9290 }
9291 let magic = cur.take(8)?;
9292 let values = magic == FREQUENCIES;
9293 if !values && magic != FREQUENCIES_V2 {
9294 return Err(invalid("directory extension magic differs"));
9295 }
9296 if cur.u16()? as usize != width {
9297 return Err(invalid("frequency column count differs"));
9298 }
9299 for _ in 0..wanted {
9300 skip_summary(&mut cur, values, rows)?;
9301 }
9302 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9303 return Ok(None);
9304 };
9305 let zero = summary
9306 .entries
9307 .iter()
9308 .find(|entry| entry.value == FrequencyValue::Integer(0))
9309 .map(|entry| entry.count)
9310 .or_else(|| (summary.omitted_max == 0).then_some(0));
9311 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9312}
9313
9314fn quick_integer_fold(
9317 file: &File,
9318 mut cur: Cursor<'_>,
9319 entry: &Entry,
9320 size: u64,
9321 wanted: usize,
9322 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9323) -> Result<()> {
9324 let name = &entry.name;
9325 let fields = &entry.fields;
9326 let rows = entry.rows;
9327 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9328 return Err(invalid("table directory differs from the catalog"));
9329 }
9330 let width = cur.u16()? as usize;
9331 if width != fields.len() {
9332 return Err(invalid("table directory width differs from the catalog"));
9333 }
9334 for field in fields {
9335 let stored =
9336 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9337 if &stored != field {
9338 return Err(invalid("table directory schema differs from the catalog"));
9339 }
9340 }
9341 let mut dictionaries = Vec::with_capacity(width);
9342 for field in fields {
9343 dictionaries.push(match cur.u8()? {
9344 0 => false,
9345 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9346 cur.skip(20)?;
9347 true
9348 }
9349 _ => return Err(invalid("dictionary page tag differs")),
9350 });
9351 }
9352 for _ in 0..width {
9353 match cur.u8()? {
9354 0 => {}
9355 1 => cur.skip(8)?,
9356 _ => return Err(invalid("distinct count tag differs")),
9357 }
9358 }
9359 if cur.u64()? != rows as u64 {
9360 return Err(invalid("table row count differs from the catalog"));
9361 }
9362 let stripes = cur.u32()? as usize;
9363 let mut total = 0_usize;
9364 let mut bytes = Vec::new();
9365 for _ in 0..stripes {
9366 let parts = cur.u32()? as usize;
9367 if parts == 0 || parts > STRIPE_PARTS {
9368 return Err(invalid("stripe part count is outside its bound"));
9369 }
9370 let mut part_rows = Vec::with_capacity(parts);
9371 for _ in 0..parts {
9372 let count = cur.u32()? as usize;
9373 if count == 0 {
9374 return Err(invalid("empty part"));
9375 }
9376 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9377 part_rows.push(count);
9378 }
9379 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9380 let section = index_section(parts)?;
9381 let index_length =
9382 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9383 if index.offset < HEADER
9384 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9385 || index.length as usize != index_length
9386 {
9387 return Err(invalid("index page range is outside the file"));
9388 }
9389 cur.skip(wanted * 12)?;
9390 let page = Span { offset: cur.u64()?, length: cur.u32()? };
9391 if page.offset < HEADER
9392 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9393 || page.length as usize > MAX_PAGE
9394 {
9395 return Err(invalid("column page range is outside the file"));
9396 }
9397 cur.skip((width - wanted - 1) * 12)?;
9398 for (field, held) in fields.iter().zip(&dictionaries) {
9399 if coded_type(&field.ty) && *held {
9400 cur.skip(20)?;
9401 }
9402 }
9403 for _ in 0..width * 2 {
9404 match cur.u8()? {
9405 0 => {}
9406 1 => cur.skip(20)?,
9407 _ => return Err(invalid("stripe page tag differs")),
9408 }
9409 }
9410 for _ in 0..width {
9411 cur.skip_bound()?;
9412 cur.skip_bound()?;
9413 cur.skip(5)?;
9414 match cur.u8()? {
9415 0 => {}
9416 1 => cur.skip(16)?,
9417 _ => return Err(invalid("a stripe sum has an unknown tag")),
9418 }
9419 }
9420 let spans = read_index_span(file, index, page, parts, wanted)?;
9421 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9422 bytes.resize(span.length, 0);
9423 let at = page
9424 .offset
9425 .checked_add(span.start as u64)
9426 .ok_or_else(|| invalid("part range overflow"))?;
9427 read_at(file, at, &mut bytes)?;
9428 if checksum(&bytes) != span.hash {
9429 return Err(invalid("integer part checksum differs"));
9430 }
9431 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9432 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9433 check_integer_tally_value(value, &fields[wanted].ty)?;
9434 emit(value, count)
9435 })?;
9436 if decoded_rows != expected_rows {
9437 return Err(invalid("encoded integer part holds the wrong number of rows"));
9438 }
9439 } else {
9440 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9441 if let Some(packed) = column.packed_parts() {
9442 let validity = column.validity();
9443 let all_valid = column.none_null();
9444 let base = packed.base();
9445 let mut codes = [0_u64; 64];
9446 for from in (0..expected_rows).step_by(codes.len()) {
9447 let count = (expected_rows - from).min(codes.len());
9448 packed.unpack(from, &mut codes[..count]);
9449 for (offset, &code) in codes[..count].iter().enumerate() {
9450 if all_valid || validity.is_valid(from + offset) {
9451 emit((base + i128::from(code)) as i64, 1)?;
9453 }
9454 }
9455 }
9456 continue;
9457 }
9458 let column = column.into_flat()?;
9459 let validity = column.validity();
9460 macro_rules! count_decoded {
9461 ($values:expr) => {
9462 for (row, &value) in $values.as_slice().iter().enumerate() {
9463 if validity.is_valid(row) {
9464 emit(i64::from(value), 1)?;
9465 }
9466 }
9467 };
9468 }
9469 match column.data() {
9470 Some(Data::Int8(values)) => count_decoded!(values),
9471 Some(Data::Int16(values)) => count_decoded!(values),
9472 Some(Data::Int32(values)) => count_decoded!(values),
9473 Some(Data::Int64(values)) => count_decoded!(values),
9474 _ => return Err(invalid("decoded integer part has the wrong type")),
9475 }
9476 }
9477 }
9478 }
9479 if total != rows {
9480 return Err(invalid("table row count differs from stripes"));
9481 }
9482 Ok(())
9483}
9484
9485fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9486 let fits = match ty {
9487 LogicalType::TinyInt => i8::try_from(value).is_ok(),
9488 LogicalType::SmallInt => i16::try_from(value).is_ok(),
9489 LogicalType::Integer => i32::try_from(value).is_ok(),
9490 LogicalType::BigInt => true,
9491 _ => false,
9492 };
9493 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9494}
9495
9496fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9497 read_directory(Cursor::new(bytes), size, None)
9498}
9499
9500fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9505 if cur.take(8)? != DIRECTORY {
9506 return Err(invalid("directory magic differs"));
9507 }
9508 let name = cur.text()?;
9509 let width = cur.u16()? as usize;
9510 let mut fields = Vec::with_capacity(width);
9511 for _ in 0..width {
9512 let name = cur.text()?;
9513 let ty = read_type(&mut cur)?;
9514 let not_null = match cur.u8()? {
9515 0 => false,
9516 1 => true,
9517 _ => return Err(invalid("nullability flag differs")),
9518 };
9519 fields.push(Field { name, ty, not_null });
9520 }
9521 let mut dictionaries = Vec::with_capacity(width);
9522 for field in &fields {
9523 dictionaries.push(match cur.u8()? {
9524 0 => None,
9525 tag if tag == dictionary_tag(&field.ty) => {
9526 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9527 let end = page
9528 .offset
9529 .checked_add(u64::from(page.length))
9530 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9531 if page.offset < HEADER || end > size {
9536 return Err(invalid("dictionary page range is outside the file"));
9537 }
9538 Some(page)
9539 }
9540 _ => return Err(invalid("dictionary page tag differs")),
9541 });
9542 }
9543 let mut distincts = Vec::with_capacity(width);
9544 for _ in 0..width {
9545 distincts.push(match cur.u8()? {
9546 0 => None,
9547 1 => Some(cur.u64()?),
9548 _ => return Err(invalid("distinct count tag differs")),
9549 });
9550 }
9551 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9552 let count = cur.u32()? as usize;
9553 let mut stripes = Vec::with_capacity(count);
9554 let mut total = 0_usize;
9555 for _ in 0..count {
9556 let count = cur.u32()? as usize;
9557 if count == 0 || count > STRIPE_PARTS {
9558 return Err(invalid("stripe part count is outside its bound"));
9559 }
9560 let mut parts = Vec::with_capacity(count);
9561 let mut stripe_rows = 0_usize;
9562 for _ in 0..count {
9563 let rows = cur.u32()?;
9564 if rows == 0 {
9565 return Err(invalid("empty part"));
9566 }
9567 parts.push(rows);
9568 stripe_rows = stripe_rows
9569 .checked_add(rows as usize)
9570 .ok_or_else(|| invalid("stripe row count overflow"))?;
9571 }
9572 total =
9573 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9574 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9575 let section = index_section(count)?;
9576 let wanted = section
9577 .checked_mul(width)
9578 .and_then(|bytes| u32::try_from(bytes).ok())
9579 .ok_or_else(|| invalid("index page length overflow"))?;
9580 let end = index
9581 .offset
9582 .checked_add(u64::from(index.length))
9583 .ok_or_else(|| invalid("index page offset overflow"))?;
9584 if index.offset < HEADER || end > size || index.length != wanted {
9585 return Err(invalid("index page range is outside the file"));
9586 }
9587 let mut pages = Vec::with_capacity(width);
9588 for _ in 0..width {
9589 let offset = cur.u64()?;
9590 let length = cur.u32()?;
9591 let end = offset
9592 .checked_add(u64::from(length))
9593 .ok_or_else(|| invalid("page offset overflow"))?;
9594 if offset < HEADER || end > size || length as usize > MAX_PAGE {
9595 return Err(invalid("page range is outside the file"));
9596 }
9597 pages.push(Span { offset, length });
9598 }
9599 let mut memberships = vec![None; width];
9600 for (column, field) in fields.iter().enumerate() {
9601 if !coded_type(&field.ty) || dictionaries[column].is_none() {
9602 continue;
9603 }
9604 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9605 let end = page
9606 .offset
9607 .checked_add(u64::from(page.length))
9608 .ok_or_else(|| invalid("membership page offset overflow"))?;
9609 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9610 return Err(invalid("membership page range is outside the file"));
9611 }
9612 if page.length != 0 {
9615 memberships[column] = Some(page);
9616 }
9617 }
9618 let mut sieves = vec![None; width];
9619 for sieve in sieves.iter_mut().take(width) {
9620 match cur.u8()? {
9621 0 => continue,
9622 1 => {}
9623 _ => return Err(invalid("a sieve page has an unknown tag")),
9624 }
9625 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9626 let end = page
9627 .offset
9628 .checked_add(u64::from(page.length))
9629 .ok_or_else(|| invalid("sieve page offset overflow"))?;
9630 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9631 return Err(invalid("sieve page range is outside the file"));
9632 }
9633 *sieve = Some(page);
9634 }
9635 let mut part_ranges = vec![None; width];
9636 for held in part_ranges.iter_mut().take(width) {
9637 match cur.u8()? {
9638 0 => continue,
9639 1 => {}
9640 _ => return Err(invalid("a part range page has an unknown tag")),
9641 }
9642 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9643 let end = page
9644 .offset
9645 .checked_add(u64::from(page.length))
9646 .ok_or_else(|| invalid("part range page offset overflow"))?;
9647 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9648 return Err(invalid("part range page range is outside the file"));
9649 }
9650 *held = Some(page);
9651 }
9652 let mut ranges = Vec::with_capacity(width);
9653 for column in 0..width {
9654 let low = cur.bound()?;
9655 let high = cur.bound()?;
9656 let nulls = cur.u32()? as usize;
9657 if nulls > stripe_rows {
9658 return Err(invalid("null count exceeds stripe rows"));
9659 }
9660 let exact = cur.u8()? != 0;
9661 let sum = match cur.u8()? {
9662 0 => None,
9663 1 => Some(i128::from_le_bytes(
9664 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9665 )),
9666 _ => return Err(invalid("a stripe sum has an unknown tag")),
9667 };
9668 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9674 let low = low.map(|bound| scaled_as(bound, ty));
9675 let high = high.map(|bound| scaled_as(bound, ty));
9676 ranges.push(Range { low, high, nulls, exact, sum });
9677 }
9678 stripes.push(Stripe {
9679 rows: stripe_rows,
9680 parts,
9681 index,
9682 pages,
9683 memberships: Pages::from_slots(memberships)?,
9684 sieves: Pages::from_slots(sieves)?,
9685 part_ranges: Pages::from_slots(part_ranges)?,
9686 zone: Zone::from_ranges(ranges),
9687 });
9688 }
9689 if total != rows {
9690 return Err(invalid("table row count differs from stripes"));
9691 }
9692 let mut entry_counts = vec![0; width];
9695 let frequencies = if cur.done() {
9696 vec![None; width]
9697 } else {
9698 let frequency_magic = cur.take(8)?;
9699 let frequency_values = frequency_magic == FREQUENCIES;
9700 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9701 return Err(invalid("directory extension magic differs"));
9702 }
9703 if cur.u16()? as usize != width {
9704 return Err(invalid("frequency column count differs"));
9705 }
9706 let mut frequencies = Vec::with_capacity(width);
9707 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9708 let start = cur.at;
9709 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9710 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9711 frequencies.push(match (summary, stored_at) {
9712 (None, _) => None,
9713 (Some(summary), None) => Some(Frequencies::Held(summary)),
9714 (Some(_), Some(offset)) => Some(Frequencies::Stored {
9715 span: Span {
9716 offset: offset + start as u64,
9717 length: u32::try_from(cur.at - start)
9718 .map_err(|_| invalid("a frequency synopsis is too long"))?,
9719 },
9720 values: frequency_values,
9721 }),
9722 });
9723 }
9724 frequencies
9725 };
9726 let mut clustering = None;
9736 let mut sections = Vec::new();
9737 let mut pair_frequencies = Vec::new();
9738 let mut seen_pair_frequencies = false;
9739 let mut frequency_texts = vec![Vec::new(); width];
9740 let mut seen_frequency_texts = false;
9741 let mut host_groups = None;
9742 let mut demoted = Vec::new();
9743 let mut seen_sections = false;
9744 let mut dictionary_payloads = Vec::new();
9745 let mut seen_payloads = false;
9746 let mut generation = 0;
9749 while !cur.done() {
9750 let mut tag = [0u8; 8];
9751 tag.copy_from_slice(cur.take(8)?);
9752 if &tag == PAIR_FREQUENCIES {
9753 if seen_pair_frequencies {
9754 return Err(invalid("directory names two pair frequency blocks"));
9755 }
9756 seen_pair_frequencies = true;
9757 let count = cur.u16()? as usize;
9758 if count > MAX_PAIR_FREQUENCIES {
9759 return Err(invalid("pair frequency count exceeds its bound"));
9760 }
9761 pair_frequencies = Vec::with_capacity(count);
9762 for _ in 0..count {
9763 let first = cur.u16()?;
9764 let second = cur.u16()?;
9765 let first_at = first as usize;
9766 let second_at = second as usize;
9767 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9768 return Err(invalid("pair frequency first column has no synopsis"));
9769 }
9770 let first_entries = entry_counts[first_at];
9771 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9772 || dictionaries.get(second_at).copied().flatten().is_none()
9773 {
9774 return Err(invalid("pair frequency second column has no stable dictionary"));
9775 }
9776 if pair_frequencies
9777 .iter()
9778 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9779 {
9780 return Err(invalid("directory repeats a pair frequency summary"));
9781 }
9782 let omitted_max = cur.u64()?;
9783 if omitted_max > rows as u64 {
9784 return Err(invalid("pair frequency omitted count exceeds the table"));
9785 }
9786 let entries_count = cur.u16()? as usize;
9787 if entries_count > FREQUENCY_ENTRIES {
9788 return Err(invalid("pair frequency entry count exceeds its bound"));
9789 }
9790 let mut entries = Vec::with_capacity(entries_count);
9791 for _ in 0..entries_count {
9792 let first_entry = cur.u16()?;
9793 if first_entry as usize >= first_entries {
9794 return Err(invalid("pair frequency anchor is outside its synopsis"));
9795 }
9796 let second = match cur.u8()? {
9797 0 => None,
9798 1 => Some(cur.u32()?),
9799 _ => return Err(invalid("pair frequency string tag differs")),
9800 };
9801 let count = cur.u64()?;
9802 if count == 0 || count > rows as u64 {
9803 return Err(invalid("pair frequency count is outside the table"));
9804 }
9805 entries.push(PairFrequencyEntry { first_entry, second, count });
9806 }
9807 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9808 return Err(invalid("pair frequency entries are not descending"));
9809 }
9810 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9811 }
9812 } else if &tag == FREQUENCY_TEXTS {
9813 if seen_frequency_texts {
9814 return Err(invalid("directory names two frequency text blocks"));
9815 }
9816 seen_frequency_texts = true;
9817 let columns = cur.u16()? as usize;
9818 if columns > width {
9819 return Err(invalid("frequency text column count exceeds the schema"));
9820 }
9821 for _ in 0..columns {
9822 let column = cur.u16()? as usize;
9823 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9824 return Err(invalid("frequency text column is repeated or out of range"));
9825 }
9826 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9827 || dictionaries.get(column).copied().flatten().is_none()
9828 || frequencies.get(column).and_then(Option::as_ref).is_none()
9829 {
9830 return Err(invalid("frequency texts belong to a non-string synopsis"));
9831 }
9832 let count = cur.u16()? as usize;
9833 if count == 0 || count != entry_counts[column] {
9834 return Err(invalid("frequency text count differs from its synopsis"));
9835 }
9836 let mut texts = Vec::with_capacity(count);
9837 for _ in 0..count {
9838 texts.push(match cur.u8()? {
9839 0 => None,
9840 1 => {
9841 let length = cur.u32()? as usize;
9842 let bytes = cur.take(length)?.to_vec();
9843 if fields[column].ty == LogicalType::Varchar {
9844 std::str::from_utf8(&bytes)
9845 .map_err(|_| invalid("frequency text is not UTF-8"))?;
9846 }
9847 Some(bytes)
9848 }
9849 _ => return Err(invalid("frequency text tag differs")),
9850 });
9851 }
9852 frequency_texts[column] = texts;
9853 }
9854 } else if &tag == HOST_GROUPS {
9855 if host_groups.is_some() {
9856 return Err(invalid("directory names two host group blocks"));
9857 }
9858 let column = cur.u16()? as usize;
9859 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9860 || dictionaries.get(column).copied().flatten().is_none()
9861 {
9862 return Err(invalid("host groups belong to a non-string dictionary"));
9863 }
9864 let omitted_max = cur.u64()?;
9865 if omitted_max > rows as u64 {
9866 return Err(invalid("host group bound exceeds the table"));
9867 }
9868 let count = cur.u16()? as usize;
9869 if count > host::CAPACITY {
9870 return Err(invalid("host group count exceeds its bound"));
9871 }
9872 let mut entries = Vec::with_capacity(count);
9873 let mut bytes = 0_usize;
9874 for _ in 0..count {
9875 let host_len = cur.u32()? as usize;
9876 bytes =
9877 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9878 if bytes > host::BYTE_BUDGET {
9879 return Err(invalid("host groups exceed their byte budget"));
9880 }
9881 let host = std::str::from_utf8(cur.take(host_len)?)
9882 .map_err(|_| invalid("host is not UTF-8"))?
9883 .to_owned();
9884 let count = cur.u64()?;
9885 if count == 0 || count > rows as u64 {
9886 return Err(invalid("host group count exceeds the table"));
9887 }
9888 let bytes_sum = i128::from_le_bytes(
9889 cur.take(16)?
9890 .try_into()
9891 .map_err(|_| invalid("host length sum is truncated"))?,
9892 );
9893 if bytes_sum < 0 {
9894 return Err(invalid("host length sum is negative"));
9895 }
9896 let minimum_len = cur.u32()? as usize;
9897 bytes =
9898 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9899 if bytes > host::BYTE_BUDGET {
9900 return Err(invalid("host groups exceed their byte budget"));
9901 }
9902 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9903 .map_err(|_| invalid("host minimum is not UTF-8"))?
9904 .to_owned();
9905 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9906 }
9907 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9908 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9909 {
9910 return Err(invalid("host groups are not in certified order"));
9911 }
9912 host_groups = Some(host::HostSummary { column, omitted_max, entries });
9913 } else if &tag == CLUSTERING {
9914 if clustering.is_some() {
9915 return Err(invalid("directory names two clustering declarations"));
9916 }
9917 let bucket = Width::from_tag(cur.u8()?)
9918 .ok_or_else(|| invalid("clustering width tag differs"))?;
9919 let count = cur.u16()? as usize;
9920 let mut columns = Vec::with_capacity(count.min(fields.len()));
9921 for _ in 0..count {
9922 columns.push(u32::from(cur.u16()?));
9923 }
9924 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9927 invalid("stored clustering declaration does not match the table it is on")
9928 })?);
9929 } else if &tag == DEMOTED {
9930 if !demoted.is_empty() {
9931 return Err(invalid("directory names two demoted column blocks"));
9932 }
9933 let count = cur.u16()? as usize;
9934 if count == 0 || count > width {
9935 return Err(invalid("demoted column count is outside the schema"));
9936 }
9937 demoted = vec![false; width];
9938 for _ in 0..count {
9939 let column = cur.u16()? as usize;
9940 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9941 return Err(invalid("a demoted column is repeated or has no dictionary"));
9942 }
9943 demoted[column] = true;
9944 }
9945 } else if &tag == SECTIONS {
9946 if seen_sections {
9947 return Err(invalid("directory names two section tables"));
9948 }
9949 seen_sections = true;
9950 generation = cur.u64()?;
9951 let count = cur.u16()? as usize;
9952 if count > MAX_SECTIONS {
9953 return Err(invalid("section count exceeds its bound"));
9954 }
9955 sections = Vec::with_capacity(count);
9956 for _ in 0..count {
9959 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9960 }
9961 for held in §ions {
9962 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9963 return Err(invalid("a section's extent table overflows the file"));
9964 };
9965 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9969 return Err(invalid("a section's extent table is outside the file"));
9970 }
9971 if held.extents == 0 && held.extent_bytes != 0 {
9972 return Err(invalid("a section with no extents names an extent table"));
9973 }
9974 }
9975 } else if &tag == DICTIONARY_PAYLOADS {
9976 if seen_payloads {
9977 return Err(invalid("directory names two dictionary payload blocks"));
9978 }
9979 seen_payloads = true;
9980 let count = cur.u16()? as usize;
9981 if count != fields.len() {
9982 return Err(invalid("dictionary payload block does not match the table's columns"));
9983 }
9984 dictionary_payloads = Vec::with_capacity(count);
9985 for _ in 0..count {
9986 let bytes = cur.u64()?;
9987 if bytes > size {
9988 return Err(invalid("a dictionary payload is larger than the file"));
9989 }
9990 dictionary_payloads.push(bytes);
9991 }
9992 } else {
9993 return Err(invalid("directory extension magic differs"));
9994 }
9995 }
9996 if !cur.done() {
9997 return Err(invalid("directory has trailing bytes"));
9998 }
9999 for stripe in &stripes {
10000 for (column, field) in fields.iter().enumerate() {
10001 if coded_type(&field.ty)
10002 && dictionaries[column].is_some()
10003 && stripe.memberships.get(column).is_none()
10004 && !demoted.get(column).copied().unwrap_or(false)
10005 {
10006 return Err(invalid("string page has no code membership index"));
10007 }
10008 }
10009 }
10010 Ok(Table {
10011 name,
10012 fields,
10013 stripes,
10014 rows,
10015 dictionaries,
10016 dictionary_payloads,
10017 demoted,
10018 distincts,
10019 frequencies,
10020 pair_frequencies,
10021 frequency_texts,
10022 host_groups,
10023 clustering,
10024 generation,
10025 sections,
10026 })
10027}
10028
10029fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
10031 bounds::put(out, bound)
10032}
10033
10034#[derive(Debug)]
10051struct Codes;
10052
10053impl chooser::Chooser for Codes {
10054 fn name(&self) -> &'static str {
10055 "codes"
10056 }
10057
10058 fn narrow_strings(
10059 &self,
10060 _values: &[&[u8]],
10061 offered: &[string::Kind],
10062 _depth: u8,
10063 ) -> Vec<string::Kind> {
10064 offered.to_vec()
10067 }
10068
10069 fn narrow_integers(
10070 &self,
10071 _values: &[i64],
10072 offered: &[integer::Kind],
10073 depth: u8,
10074 ) -> Vec<integer::Kind> {
10075 narrowed_to(Codes::keep(depth), offered)
10078 }
10079
10080 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10081 Codes::keep(depth).contains(&kind)
10082 }
10083}
10084
10085impl Codes {
10086 fn keep(depth: u8) -> &'static [integer::Kind] {
10087 if depth == 0 {
10088 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
10089 } else {
10090 &[integer::Kind::Constant, integer::Kind::Packed]
10091 }
10092 }
10093}
10094
10095fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
10103 let narrowed: Vec<integer::Kind> =
10104 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
10105 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
10106}
10107
10108#[derive(Debug)]
10120struct Fixed;
10121
10122impl chooser::Chooser for Fixed {
10123 fn name(&self) -> &'static str {
10124 "fixed"
10125 }
10126
10127 fn narrow_strings(
10128 &self,
10129 _values: &[&[u8]],
10130 offered: &[string::Kind],
10131 _depth: u8,
10132 ) -> Vec<string::Kind> {
10133 offered.to_vec()
10134 }
10135
10136 fn narrow_integers(
10137 &self,
10138 _values: &[i64],
10139 offered: &[integer::Kind],
10140 depth: u8,
10141 ) -> Vec<integer::Kind> {
10142 narrowed_to(Fixed::keep(depth), offered)
10143 }
10144
10145 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
10146 Fixed::keep(depth).contains(&kind)
10147 }
10148}
10149
10150impl Fixed {
10151 fn keep(depth: u8) -> &'static [integer::Kind] {
10152 if depth == 0 {
10153 &[
10154 integer::Kind::Constant,
10155 integer::Kind::Packed,
10156 integer::Kind::Delta,
10157 integer::Kind::Rle,
10158 integer::Kind::Sparse,
10159 integer::Kind::Strided,
10160 ]
10161 } else {
10162 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
10163 }
10164 }
10165}
10166
10167fn widened(data: &Data) -> Option<Vec<i64>> {
10174 match data {
10175 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10176 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10177 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10178 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10179 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10180 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
10181 Data::Int64(values) => Some(values.to_vec()),
10182 _ => None,
10183 }
10184}
10185
10186fn cascade(ty: &LogicalType, bytes: &[u8], rows: usize) -> Result<Data> {
10192 fn wanted<T: integer::Lane>(bytes: &[u8], rows: usize) -> Result<Vec<T>> {
10193 let values = integer::decode_as::<T>(bytes)
10194 .map_err(|error| invalid(&format!("page value is not of its type: {error}")))?;
10195 if values.len() != rows {
10196 return Err(invalid("cascade page holds the wrong number of rows"));
10197 }
10198 Ok(values)
10199 }
10200 Ok(match ty {
10201 LogicalType::TinyInt => Data::Int8(wanted::<i8>(bytes, rows)?.into()),
10202 LogicalType::UTinyInt => Data::UInt8(wanted::<u8>(bytes, rows)?.into()),
10203 LogicalType::SmallInt => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10204 LogicalType::USmallInt => Data::UInt16(wanted::<u16>(bytes, rows)?.into()),
10205 LogicalType::Integer | LogicalType::Date => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10206 LogicalType::UInteger => Data::UInt32(wanted::<u32>(bytes, rows)?.into()),
10207 LogicalType::BigInt
10208 | LogicalType::Timestamp
10209 | LogicalType::Time
10210 | LogicalType::TimeTz
10211 | LogicalType::TimestampTz
10212 | LogicalType::TimestampS
10213 | LogicalType::TimestampMs
10214 | LogicalType::TimestampNs => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10215 LogicalType::Decimal { .. } => match ty.physical() {
10218 PhysicalType::Int16 => Data::Int16(wanted::<i16>(bytes, rows)?.into()),
10219 PhysicalType::Int32 => Data::Int32(wanted::<i32>(bytes, rows)?.into()),
10220 PhysicalType::Int64 => Data::Int64(wanted::<i64>(bytes, rows)?.into()),
10221 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10222 },
10223 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10224 })
10225}
10226
10227fn plain_width(ty: &LogicalType) -> Option<usize> {
10230 Some(match ty {
10231 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10232 LogicalType::SmallInt | LogicalType::USmallInt => 2,
10233 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10234 LogicalType::BigInt
10235 | LogicalType::Timestamp
10236 | LogicalType::Time
10237 | LogicalType::TimeTz
10238 | LogicalType::TimestampTz
10239 | LogicalType::TimestampS
10240 | LogicalType::TimestampMs
10241 | LogicalType::TimestampNs => 8,
10242 LogicalType::Decimal { .. } => match ty.physical() {
10243 PhysicalType::Int16 => 2,
10244 PhysicalType::Int32 => 4,
10245 PhysicalType::Int64 => 8,
10246 _ => return None,
10249 },
10250 _ => return None,
10251 })
10252}
10253
10254fn cascaded(
10260 flat: &Vector,
10261 ty: &LogicalType,
10262 packed: Option<&Packed<'_>>,
10263 settling: &mut Settling,
10264) -> Result<Option<Vec<u8>>> {
10265 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10266 let Some(values) = widened(data) else { return Ok(None) };
10267 let plain = values.len().saturating_mul(width);
10268 let best = match packed {
10269 Some(packed) => plain.min(21 + size_of_val(packed.words())),
10271 None => plain,
10272 };
10273 let out = settling.encode(&values)?;
10274 Ok((out.len() < best).then_some(out))
10275}
10276
10277const SEARCH_EVERY: usize = 16;
10284
10285#[derive(Debug, Default)]
10291struct Settling {
10292 shape: Option<Shape>,
10295 since: usize,
10297 symbols: Option<Symbols>,
10299}
10300
10301#[derive(Debug)]
10304struct Symbols {
10305 shape: chooser::Settled,
10306 len: usize,
10309 payload: usize,
10310 since: usize,
10311}
10312
10313impl Settling {
10314 fn text(&mut self, values: &[&[u8]], payload: usize) -> Result<Option<Vec<u8>>> {
10322 if let Some(symbols) = self.symbols.as_mut().filter(|symbols| symbols.since < SEARCH_EVERY)
10323 {
10324 let out = string::encode_fsst(values, &symbols.shape)?;
10325 let held = match &out {
10327 None => symbols.len == 0,
10328 Some(out) => {
10329 (out.len() as u128) * (symbols.payload as u128) * 4
10330 <= (symbols.len as u128) * (payload as u128) * 5
10331 }
10332 };
10333 if held {
10334 symbols.since += 1;
10335 return Ok(out);
10336 }
10337 }
10338 let shape = string::fsst_shape(values);
10339 let out = string::encode_fsst(values, &shape)?;
10340 let len = out.as_ref().map_or(0, Vec::len);
10341 self.symbols = Some(Symbols { shape, len, payload: payload.max(1), since: 0 });
10342 Ok(out)
10343 }
10344
10345 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10352 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10353 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10354 let out = integer::encode_with(values, &replay)?;
10355 if !replay.held() {
10356 self.settle(&out, values.len(), replay.first_offered())?;
10357 return Ok(out);
10358 }
10359 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10360 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10361 self.since += 1;
10362 return Ok(out);
10363 }
10364 }
10365 let search = chooser::Replay::new(&[], &Fixed);
10367 let out = integer::encode_with(values, &search)?;
10368 self.settle(&out, values.len(), search.first_offered())?;
10369 Ok(out)
10370 }
10371
10372 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10373 let kinds = integer::shape(out)?;
10374 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10375 self.since = 0;
10376 Ok(())
10377 }
10378}
10379
10380#[derive(Debug)]
10382struct Shape {
10383 kinds: Vec<integer::Kind>,
10384 offered: Vec<integer::Kind>,
10385 len: usize,
10386 rows: usize,
10387}
10388
10389fn text_compressed(flat: &Vector, settling: &mut Settling) -> Result<Option<Vec<u8>>> {
10430 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10431 let mut payload = 0_usize;
10432 for row in 0..flat.len() {
10433 let text = flat.bytes_at(row).unwrap_or(b"");
10436 payload = payload.saturating_add(text.len());
10437 values.push(text);
10438 }
10439 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10441 let Some(out) = settling.text(&values, payload)? else {
10442 return Ok(None);
10443 };
10444 Ok((out.len() < plain).then_some(out))
10445}
10446
10447fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10448 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10449 let coded = integer::encode_with(&wide, &Codes)?;
10450 let plain = codes.len().saturating_mul(size_of::<u32>());
10451 Ok((coded.len() < plain).then_some(coded))
10452}
10453
10454fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10457 let flag = match flat.validity() {
10458 Validity::AllValid => 0,
10459 Validity::AllInvalid => 1,
10460 Validity::Mask(_) => 2,
10461 };
10462 out.push(flag);
10463 if flag == 2 {
10464 for group in (0..flat.len()).step_by(8) {
10465 let mut bits = 0_u8;
10466 for bit in 0..8 {
10467 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10468 bits |= 1 << bit;
10469 }
10470 }
10471 out.push(bits);
10472 }
10473 }
10474}
10475
10476fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10483 let coded = encoded_codes(codes)?;
10484 let mut out = Vec::with_capacity(
10485 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10486 );
10487 out.push(if coded.is_some() { 4 } else { 3 });
10488 out.extend_from_slice(validity);
10489 match coded {
10490 Some(coded) => out.extend_from_slice(&coded),
10491 None => {
10492 for &code in codes {
10493 put_u32(&mut out, code);
10494 }
10495 }
10496 }
10497 Ok(out)
10498}
10499
10500fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10503 let ty = vector.logical_type();
10504 let flat = vector.flatten()?;
10506 let mut out = Vec::new();
10507 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10508 let compressed_text = if dictionary.is_none() && coded_type(ty) {
10509 text_compressed(&flat, settling)?
10510 } else {
10511 None
10512 };
10513 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10514 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10515 let cascade =
10519 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10520 out.push(if cascade.is_some() {
10521 5
10522 } else if dictionary.is_some() {
10523 1
10524 } else if compressed_text.is_some() {
10525 6
10526 } else if packed.is_some() {
10527 2
10528 } else {
10529 0
10530 });
10531 push_validity(&mut out, &flat);
10532 if let Some(cascade) = cascade {
10533 out.extend_from_slice(&cascade);
10534 return Ok(out);
10535 }
10536 if let Some(dictionary) = dictionary {
10537 out.extend_from_slice(&dictionary);
10538 return Ok(out);
10539 }
10540 if let Some(compressed_text) = compressed_text {
10541 out.extend_from_slice(&compressed_text);
10542 return Ok(out);
10543 }
10544 if let Some(packed) = packed {
10545 if packed.offset() != 0 {
10546 return Err(invalid("writer received a sliced packed vector"));
10547 }
10548 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10549 out.extend_from_slice(&packed.base().to_le_bytes());
10550 put_u32(
10551 &mut out,
10552 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10553 );
10554 for word in packed.words() {
10555 put_u64(&mut out, *word);
10556 }
10557 return Ok(out);
10558 }
10559 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10560 match (ty, data) {
10561 (LogicalType::TinyInt, Data::Int8(values)) => {
10562 for value in &**values {
10563 out.extend_from_slice(&value.to_le_bytes());
10564 }
10565 }
10566 (LogicalType::UTinyInt, Data::UInt8(values)) => {
10567 for value in &**values {
10568 out.extend_from_slice(&value.to_le_bytes());
10569 }
10570 }
10571 (LogicalType::SmallInt, Data::Int16(values)) => {
10572 for value in &**values {
10573 out.extend_from_slice(&value.to_le_bytes());
10574 }
10575 }
10576 (LogicalType::USmallInt, Data::UInt16(values)) => {
10577 for value in &**values {
10578 out.extend_from_slice(&value.to_le_bytes());
10579 }
10580 }
10581 (LogicalType::UInteger, Data::UInt32(values)) => {
10582 for value in &**values {
10583 out.extend_from_slice(&value.to_le_bytes());
10584 }
10585 }
10586 (LogicalType::UBigInt, Data::UInt64(values)) => {
10587 for value in &**values {
10588 out.extend_from_slice(&value.to_le_bytes());
10589 }
10590 }
10591 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10592 for value in &**values {
10593 out.extend_from_slice(&value.to_le_bytes());
10594 }
10595 }
10596 (
10597 LogicalType::BigInt
10598 | LogicalType::Timestamp
10599 | LogicalType::Time
10600 | LogicalType::TimeTz
10601 | LogicalType::TimestampTz
10602 | LogicalType::TimestampS
10603 | LogicalType::TimestampMs
10604 | LogicalType::TimestampNs,
10605 Data::Int64(values),
10606 ) => {
10607 for value in &**values {
10608 out.extend_from_slice(&value.to_le_bytes());
10609 }
10610 }
10611 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10614 for value in &**values {
10615 out.extend_from_slice(&value.to_le_bytes());
10616 }
10617 }
10618 (LogicalType::UHugeInt, Data::UInt128(values)) => {
10619 for value in &**values {
10620 out.extend_from_slice(&value.to_le_bytes());
10621 }
10622 }
10623 (LogicalType::Float, Data::Float32(values)) => {
10626 for value in &**values {
10627 out.extend_from_slice(&value.to_le_bytes());
10628 }
10629 }
10630 (LogicalType::Double, Data::Float64(values)) => {
10631 for value in &**values {
10632 out.extend_from_slice(&value.to_le_bytes());
10633 }
10634 }
10635 (LogicalType::Interval, Data::Interval(values)) => {
10639 for (months, days, micros) in &**values {
10640 out.extend_from_slice(&months.to_le_bytes());
10641 out.extend_from_slice(&days.to_le_bytes());
10642 out.extend_from_slice(µs.to_le_bytes());
10643 }
10644 }
10645 (LogicalType::Boolean, Data::Bool(values)) => {
10646 for value in &**values {
10647 out.push(u8::from(*value));
10648 }
10649 }
10650 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10653 for value in &**values {
10654 out.extend_from_slice(&value.to_le_bytes());
10655 }
10656 }
10657 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10658 for value in &**values {
10659 out.extend_from_slice(&value.to_le_bytes());
10660 }
10661 }
10662 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10663 for value in &**values {
10664 out.extend_from_slice(&value.to_le_bytes());
10665 }
10666 }
10667 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10668 for value in &**values {
10669 out.extend_from_slice(&value.to_le_bytes());
10670 }
10671 }
10672 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10677 let mut bytes = Vec::new();
10678 put_u32(&mut out, 0);
10679 for row in 0..vector.len() {
10680 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10681 bytes.extend_from_slice(value);
10682 put_u32(
10683 &mut out,
10684 u32::try_from(bytes.len())
10685 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10686 );
10687 }
10688 out.extend_from_slice(&bytes);
10689 }
10690 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10691 }
10692 Ok(out)
10693}
10694
10695fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10696 while value >= 0x80 {
10697 out.push((value as u8 & 0x7f) | 0x80);
10698 value >>= 7;
10699 }
10700 out.push(value as u8);
10701}
10702
10703fn unique_codes(codes: &[u32]) -> Vec<u32> {
10705 let mut unique = codes.to_vec();
10706 unique.sort_unstable();
10707 unique.dedup();
10708 unique
10709}
10710
10711fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10717 let mut lists = lists;
10718 while lists.len() > 1 {
10719 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10720 for pair in lists.chunks(2) {
10721 match pair {
10722 [left, right] => next.push(merged_pair(left, right)),
10723 [only] => next.push(only.clone()),
10724 _ => {}
10725 }
10726 }
10727 lists = next;
10728 }
10729 lists.pop().unwrap_or_default()
10730}
10731
10732fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10733 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10734 let mut at = 0;
10735 let mut to = 0;
10736 while at < left.len() && to < right.len() {
10737 match left[at].cmp(&right[to]) {
10738 Ordering::Less => {
10739 out.push(left[at]);
10740 at += 1;
10741 }
10742 Ordering::Greater => {
10743 out.push(right[to]);
10744 to += 1;
10745 }
10746 Ordering::Equal => {
10747 out.push(left[at]);
10748 at += 1;
10749 to += 1;
10750 }
10751 }
10752 }
10753 out.extend_from_slice(&left[at..]);
10754 out.extend_from_slice(&right[to..]);
10755 out
10756}
10757
10758fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10763 let mut merged = Range::default();
10764 let mut first = true;
10765 for range in ranges {
10766 merged.nulls = merged.nulls.saturating_add(range.nulls);
10767 merged.sum = match (merged.sum.take(), range.sum) {
10771 (Some(held), Some(next)) if !first => held.checked_add(next),
10772 (_, next) if first => next,
10773 _ => None,
10774 };
10775 merged.exact = if first { range.exact } else { merged.exact && range.exact };
10776 if first {
10777 merged.low = range.low;
10778 merged.high = range.high;
10779 first = false;
10780 continue;
10781 }
10782 merged.low = match (merged.low.take(), range.low) {
10783 (Some(held), Some(next)) => Some(held.smaller(next)),
10784 _ => None,
10785 };
10786 merged.high = match (merged.high.take(), range.high) {
10787 (Some(held), Some(next)) => Some(held.larger(next)),
10788 _ => None,
10789 };
10790 }
10791 merged
10792}
10793
10794fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10807 match bound {
10808 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10809 value.truncate(PART_BOUND_BYTES);
10810 if !high {
10811 return Some(Bound::Bytes(value));
10812 }
10813 while let Some(last) = value.pop() {
10814 if last < u8::MAX {
10815 value.push(last + 1);
10816 return Some(Bound::Bytes(value));
10817 }
10818 }
10819 None
10820 }
10821 other => other,
10822 }
10823}
10824
10825fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10833 let mut out = Vec::new();
10834 put_u32(
10835 &mut out,
10836 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10837 );
10838 for range in ranges {
10839 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10840 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10841 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10842 }
10843 Ok(out)
10844}
10845
10846fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10848 let mut cur = Cursor::new(bytes);
10849 let parts = cur.u32()? as usize;
10850 let mut out = Vec::new();
10851 for _ in 0..parts {
10852 let low = cur.bound()?;
10853 let high = cur.bound()?;
10854 let nulls = cur.u32()? as usize;
10855 out.push(Range { low, high, nulls, exact: false, sum: None });
10856 }
10857 Ok(out)
10858}
10859
10860fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10861 let held: Vec<&Option<Sieve>> = sieves.collect();
10862 let mut out = Vec::new();
10863 put_u32(
10864 &mut out,
10865 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10866 );
10867 for sieve in &held {
10868 let length = sieve.as_ref().map_or(0, Sieve::len);
10869 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10870 }
10871 for sieve in held.into_iter().flatten() {
10873 out.extend_from_slice(&sieve.to_bytes());
10874 }
10875 Ok(out)
10876}
10877
10878fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10884 let parts = u32::from_le_bytes(
10885 bytes
10886 .get(..4)
10887 .ok_or_else(|| invalid("sieve page is truncated"))?
10888 .try_into()
10889 .map_err(|_| invalid("sieve page is truncated"))?,
10890 ) as usize;
10891 let mut lengths = Vec::with_capacity(parts);
10892 for part in 0..parts {
10893 let at = 4 + part * 4;
10894 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10895 lengths.push(u32::from_le_bytes(
10896 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10897 ) as usize);
10898 }
10899 let mut at = 4 + parts * 4;
10900 let mut out = Vec::with_capacity(parts);
10901 for length in lengths {
10902 if length == 0 {
10903 out.push(None);
10904 continue;
10905 }
10906 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10907 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10908 out.push(Sieve::from_bytes(field));
10909 at = end;
10910 }
10911 if at != bytes.len() {
10912 return Err(invalid("sieve page has trailing bytes"));
10913 }
10914 Ok(out)
10915}
10916
10917fn encode_membership(unique: &[u32]) -> Vec<u8> {
10923 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10924 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10925 let mut previous = 0;
10926 for (at, &code) in unique.iter().enumerate() {
10927 put_varint(&mut out, if at == 0 { code } else { code - previous });
10928 previous = code;
10929 }
10930 out
10931}
10932
10933fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10934 let mut value = 0_u32;
10935 for shift in (0..35).step_by(7) {
10936 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10937 *at += 1;
10938 let part = u32::from(byte & 0x7f);
10939 if shift == 28 && part > 0x0f {
10940 return Err(invalid("membership varint overflow"));
10941 }
10942 value = value
10943 .checked_add(
10944 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10945 )
10946 .ok_or_else(|| invalid("membership varint overflow"))?;
10947 if byte & 0x80 == 0 {
10948 return Ok(value);
10949 }
10950 }
10951 Err(invalid("membership varint is too long"))
10952}
10953
10954fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10955 let mut at = 0;
10956 let count = take_varint(bytes, &mut at)? as usize;
10957 let mut codes = Vec::with_capacity(count);
10958 let mut previous = 0_u32;
10959 for index in 0..count {
10960 let delta = take_varint(bytes, &mut at)?;
10961 let code = if index == 0 {
10962 delta
10963 } else {
10964 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10965 };
10966 if index > 0 && code <= previous {
10967 return Err(invalid("membership codes are not increasing"));
10968 }
10969 codes.push(code);
10970 previous = code;
10971 }
10972 if at != bytes.len() {
10973 return Err(invalid("membership page has trailing bytes"));
10974 }
10975 Ok(codes)
10976}
10977
10978fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10986 let mut by_text: HashMap<&[u8], u32, Spread> =
10987 HashMap::with_capacity_and_hasher(vector.len(), Spread);
10988 let mut values = Vec::new();
10989 let mut codes = Vec::with_capacity(vector.len());
10990 let mut plain_bytes = 0_usize;
10991 for row in 0..vector.len() {
10992 let text = vector.bytes_at(row).unwrap_or(b"");
10993 plain_bytes = plain_bytes.saturating_add(text.len());
10994 let code = match by_text.get(text) {
10995 Some(&code) => code,
10996 None => {
10997 let code = u32::try_from(values.len())
10998 .map_err(|_| invalid("too many dictionary values"))?;
10999 by_text.insert(text, code);
11000 values.push(text);
11001 code
11002 }
11003 };
11004 codes.push(code);
11005 }
11006 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
11007 let encoded = 8_usize
11008 .saturating_add((values.len() + 1).saturating_mul(4))
11009 .saturating_add(dictionary_bytes)
11010 .saturating_add(codes.len().saturating_mul(4));
11011 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
11012 if encoded >= plain {
11013 return Ok(None);
11014 }
11015 let mut out = Vec::with_capacity(encoded);
11016 put_u32(
11017 &mut out,
11018 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
11019 );
11020 put_u32(
11021 &mut out,
11022 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
11023 );
11024 let mut offset = 0_u32;
11025 put_u32(&mut out, offset);
11026 for value in &values {
11027 offset = offset
11028 .checked_add(
11029 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
11030 )
11031 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
11032 put_u32(&mut out, offset);
11033 }
11034 for value in values {
11035 out.extend_from_slice(value);
11036 }
11037 for code in codes {
11038 put_u32(&mut out, code);
11039 }
11040 Ok(Some(out))
11041}
11042
11043struct Room<'a, T> {
11045 state: &'a Mutex<(T, usize)>,
11046 finished: &'a Condvar,
11047 bytes: usize,
11048}
11049
11050impl<T> Drop for Room<'_, T> {
11051 fn drop(&mut self) {
11052 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
11053 held.1 -= self.bytes;
11054 drop(held);
11055 self.finished.notify_all();
11056 }
11057}
11058
11059enum Closing<'a> {
11061 Numeric {
11064 column: usize,
11065 counted: bool,
11066 dense: Option<(u64, usize)>,
11067 },
11068 Dictionary {
11069 index: usize,
11070 dictionary: &'a GlobalDictionary,
11071 },
11072}
11073
11074enum Closed {
11076 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
11077 Dictionary(usize, ClosedDictionary),
11078}
11079
11080struct ClosedDictionary {
11082 distinct: Option<u64>,
11084 frequencies: Option<FrequencySummary>,
11085 texts: Vec<Option<Vec<u8>>>,
11086 hosts: Option<host::HostSummary>,
11087 encoded: EncodedDictionary,
11088 payload: u64,
11090}
11091
11092struct EncodedDictionary {
11093 index: Vec<u8>,
11094 ranks: Vec<u8>,
11095 grams: Vec<u8>,
11096}
11097
11098fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
11139 let mut work = vec![(0, codes.len(), 0)];
11140 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
11141 while let Some((from, to, depth)) = work.pop() {
11142 let part = &mut codes[from..to];
11143 keyed.clear();
11144 keyed.extend(part.iter().map(|&code| {
11145 let value = values(code);
11146 let rest = value.get(depth..).unwrap_or_default();
11147 (head(rest), rest.len().min(8) as u8, code)
11148 }));
11149 keyed.sort_unstable();
11150 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
11151 *slot = entry.2;
11152 }
11153 let mut start = 0;
11154 while start < keyed.len() {
11155 let (key, taken, _) = keyed[start];
11156 let mut end = start + 1;
11157 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
11158 end += 1;
11159 }
11160 if taken == 8 && end - start > 1 {
11161 work.push((from + start, from + end, depth + 8));
11162 }
11163 start = end;
11164 }
11165 }
11166}
11167
11168const PARALLEL_SORT_MIN: usize = 1 << 16;
11170
11171const BUCKETS_PER_WORKER: usize = 4;
11174
11175const SAMPLES_PER_BUCKET: usize = 32;
11177
11178fn sort_by_value_across<'a>(
11196 codes: &mut [u32],
11197 values: impl Fn(u32) -> &'a [u8] + Sync,
11198 workers: usize,
11199) {
11200 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
11201 sort_by_value(codes, values);
11202 return;
11203 }
11204 let buckets = workers * BUCKETS_PER_WORKER;
11205 let wanted = buckets * SAMPLES_PER_BUCKET;
11206 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
11207 sort_by_value(&mut sample, &values);
11208 let splitters =
11209 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
11210 let values = &values;
11211 let splitters = &splitters;
11212 let per = codes.len().div_ceil(workers);
11213 let places = std::thread::scope(|scope| {
11215 codes
11216 .chunks(per)
11217 .map(|run| {
11218 scope.spawn(move || {
11219 run.iter()
11220 .map(|&code| {
11221 let value = values(code);
11222 splitters.partition_point(|splitter| *splitter <= value) as u32
11223 })
11224 .collect::<Vec<_>>()
11225 })
11226 })
11227 .collect::<Vec<_>>()
11228 .into_iter()
11229 .flat_map(|handle| {
11230 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11231 })
11232 .collect::<Vec<_>>()
11233 });
11234 let mut starts = vec![0_usize; buckets + 1];
11235 for &place in &places {
11236 starts[place as usize + 1] += 1;
11237 }
11238 for bucket in 0..buckets {
11239 starts[bucket + 1] += starts[bucket];
11240 }
11241 let mut laid = vec![0_u32; codes.len()];
11242 let mut next = starts.clone();
11243 for (&code, &place) in codes.iter().zip(&places) {
11244 laid[next[place as usize]] = code;
11245 next[place as usize] += 1;
11246 }
11247 drop(places);
11248 let mut runs = Vec::with_capacity(buckets);
11249 let mut rest = laid.as_mut_slice();
11250 for bucket in 0..buckets {
11251 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11252 runs.push(run);
11253 rest = after;
11254 }
11255 runs.sort_by_key(|run| run.len());
11257 let queue = Mutex::new(runs);
11258 std::thread::scope(|scope| {
11259 for _ in 0..workers {
11260 scope.spawn(|| {
11261 loop {
11262 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11263 let Some(run) = taken else { break };
11264 sort_by_value(run, values);
11265 }
11266 });
11267 }
11268 });
11269 codes.copy_from_slice(&laid);
11270}
11271
11272fn head(bytes: &[u8]) -> u64 {
11274 let mut word = [0; 8];
11275 let take = bytes.len().min(8);
11276 word[..take].copy_from_slice(&bytes[..take]);
11277 u64::from_be_bytes(word)
11278}
11279
11280fn encode_global_dictionary(
11291 dictionary: &GlobalDictionary,
11292 order: &[(u64, u32)],
11293 places: &[Placed],
11294 scattered: bool,
11295) -> Result<EncodedDictionary> {
11296 let values = dictionary.values();
11297 if order.len() != values {
11298 return Err(invalid("global dictionary order does not cover its values"));
11299 }
11300 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11301 if places.len() != blocks {
11302 return Err(invalid("global dictionary payload is not the blocks it says it is"));
11303 }
11304 if dictionary.grams.len() != blocks {
11305 return Err(invalid("global dictionary signatures do not cover its blocks"));
11306 }
11307 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11308 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11309 let offset_bits = offset_width(&dictionary.ends);
11310 let payload_words = if scattered { 3 } else { 2 };
11311 let index_len = DICTIONARY_HEADER
11312 .checked_add(offset_bytes(values, offset_bits))
11313 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11314 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11315 .and_then(|len| len.checked_add(8))
11316 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11317 let mut index = Vec::with_capacity(index_len);
11318 put_u32(
11319 &mut index,
11320 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11321 );
11322 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11323 put_u32(
11324 &mut index,
11325 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11326 );
11327 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11328 | DICTIONARY_GRAMS
11329 | DICTIONARY_WIDE_GRAMS;
11330 put_u32(&mut index, offset_bits as u32 | flag);
11331 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11332 let mut end = 0_u64;
11337 for place in places {
11338 if scattered {
11339 put_u64(&mut index, place.start);
11340 put_u64(&mut index, place.length);
11341 } else {
11342 end = end
11343 .checked_add(place.length)
11344 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11345 put_u64(&mut index, end);
11346 }
11347 }
11348 for place in places {
11349 put_u64(&mut index, place.hash);
11350 }
11351 if rank_ends.len() != rank_blocks {
11354 return Err(invalid("global dictionary order is not the blocks it says it is"));
11355 }
11356 for end in &rank_ends {
11357 put_u64(&mut index, *end);
11358 }
11359 let mut at = 0_usize;
11360 for end in &rank_ends {
11361 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11362 put_u64(&mut index, checksum(&ranks[at..end]));
11363 at = end;
11364 }
11365 let gram_len = blocks
11366 .checked_mul(TEXT_GRAM_BYTES)
11367 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11368 let mut grams = Vec::with_capacity(gram_len);
11369 for block in &dictionary.grams {
11370 grams.extend_from_slice(block);
11371 }
11372 put_u64(&mut index, checksum(&grams));
11373 if index.len() != index_len {
11374 return Err(invalid("global dictionary index is not the length it was laid out for"));
11375 }
11376 Ok(EncodedDictionary { index, ranks, grams })
11377}
11378
11379const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11386
11387fn payload_shapes() -> Vec<chooser::Settled> {
11413 let integers = vec![integer::Kind::Packed];
11414 [
11415 vec![string::Kind::Front, string::Kind::Lz],
11416 vec![string::Kind::Lz, string::Kind::Fsst],
11417 vec![string::Kind::Lz, string::Kind::Plain],
11418 vec![string::Kind::Fsst],
11419 vec![string::Kind::Plain],
11420 ]
11421 .into_iter()
11422 .map(|strings| chooser::Settled::new(strings, integers.clone()))
11423 .collect()
11424}
11425
11426fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11433 let started = profile.map(|_| std::time::Instant::now());
11434 file.sync()?;
11435 if let (Some(profile), Some(started)) = (profile, started) {
11436 profile.waited(
11437 Stage::Publish,
11438 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11439 );
11440 }
11441 Ok(())
11442}
11443
11444#[derive(Debug)]
11449pub(crate) struct Unencoded {
11450 column: usize,
11451 at: usize,
11452 ends: Vec<u32>,
11453 bytes: Vec<u8>,
11454 shape: chooser::Settled,
11455}
11456
11457impl Unencoded {
11458 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11460 let values = block_values(&self.ends, &self.bytes);
11461 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11462 }
11463
11464 pub(crate) fn place(&self) -> (usize, usize) {
11466 (self.column, self.at)
11467 }
11468}
11469
11470pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11474
11475fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11477 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11478 for value in values {
11479 for gram in value.windows(4) {
11480 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11481 grams[bit / 8] |= 1 << (bit % 8);
11482 }
11483 }
11484 }
11485 grams
11486}
11487
11488fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11490 let mut out = Vec::with_capacity(ends.len());
11491 let mut from = 0;
11492 for &to in ends {
11493 out.push(&bytes[from..to as usize]);
11494 from = to as usize;
11495 }
11496 out
11497}
11498
11499fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11506 for dictionary in dictionaries.iter_mut().flatten() {
11507 if !dictionary.early.is_empty() {
11508 return Err(Error::internal("a dictionary block handed out never came back"));
11509 }
11510 dictionary.seal_rest();
11511 dictionary.settle_rest()?;
11512 }
11513 encode_waiting(dictionaries)?;
11514 if dictionaries
11517 .iter()
11518 .flatten()
11519 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11520 {
11521 return Err(Error::internal("a dictionary block handed out never came back"));
11522 }
11523 Ok(())
11524}
11525
11526fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11529 let jobs = dictionaries
11530 .iter()
11531 .enumerate()
11532 .flat_map(|(column, held)| {
11533 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11534 })
11535 .collect::<Vec<_>>();
11536 if jobs.is_empty() {
11537 return Ok(());
11538 }
11539 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11540 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11541 Ok((column, at, held.encode_waiting(at)?))
11542 };
11543 let workers = std::thread::available_parallelism()
11544 .map_or(1, usize::from)
11545 .min(MAX_FREQUENCY_WORKERS)
11546 .min(jobs.len());
11547 let made = if workers <= 1 {
11548 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11549 } else {
11550 let next = AtomicUsize::new(0);
11551 let jobs = &jobs;
11552 let pieces = std::thread::scope(|scope| {
11553 (0..workers)
11554 .map(|_| {
11555 scope.spawn(|| {
11556 let mut mine = Vec::new();
11557 loop {
11558 let job = next.fetch_add(1, Atomic::Relaxed);
11559 let Some(&(column, at)) = jobs.get(job) else { break };
11560 mine.push(one(column, at)?);
11561 }
11562 Ok(mine)
11563 })
11564 })
11565 .collect::<Vec<_>>()
11566 .into_iter()
11567 .map(|handle| {
11568 handle
11569 .join()
11570 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11571 })
11572 .collect::<Result<Vec<_>>>()
11573 })?;
11574 pieces.into_iter().flatten().collect()
11575 };
11576 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11577 (0..dictionaries.len()).map(|_| Vec::new()).collect();
11578 for (column, at, bytes) in made {
11579 done[column].push((at, bytes));
11580 }
11581 for (column, mut made) in done.into_iter().enumerate() {
11582 if made.is_empty() {
11583 continue;
11584 }
11585 let Some(held) = dictionaries[column].as_mut() else { continue };
11586 made.sort_by_key(|(at, _)| *at);
11587 let waiting = std::mem::take(&mut held.waiting);
11588 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11589 if held.encoded() != at {
11590 return Err(Error::internal("a dictionary block was encoded out of order"));
11591 }
11592 held.push_block(block);
11593 }
11594 }
11595 Ok(())
11596}
11597
11598fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11608 let mut best: Option<(chooser::Settled, usize)> = None;
11609 for shape in payload_shapes() {
11610 let mut size = 0;
11611 for block in sample {
11612 size += string::encode_with(block, &shape)?.len();
11613 }
11614 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11615 best = Some((shape, size));
11616 }
11617 }
11618 best.map(|(shape, _)| shape)
11619 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11620}
11621
11622fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11629 let mut out = Vec::with_capacity(order.len() * 4);
11630 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11631 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11632 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11633 for block in order.chunks(TEXT_RANK_BLOCK) {
11634 let base = block.first().map_or(0, |&(head, _)| head);
11637 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11638 let width = (u64::BITS - span.leading_zeros()) as usize;
11639 heads.clear();
11640 codes.clear();
11641 for &(head, code) in block {
11642 heads.push(head.wrapping_sub(base));
11643 codes.push(u64::from(code));
11644 }
11645 put_u64(&mut out, base);
11646 out.push(width as u8);
11647 bitpack::pack_tail(&heads, width, &mut out)
11648 .map_err(|_| invalid("global dictionary heads do not pack"))?;
11649 bitpack::pack_tail(&codes, code_bits, &mut out)
11650 .map_err(|_| invalid("global dictionary codes do not pack"))?;
11651 ends.push(out.len() as u64);
11652 }
11653 Ok((out, ends))
11654}
11655
11656fn open_global_dictionary(
11663 file: Arc<File>,
11664 page: Page,
11665 ty: &LogicalType,
11666 keep_budget: usize,
11667) -> Result<Vector> {
11668 if !coded_type(ty) {
11669 return Err(invalid("global dictionary belongs to a non-string column"));
11670 }
11671 let mut header = [0; DICTIONARY_HEADER];
11672 read_at(&file, page.offset, &mut header)?;
11673 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11674 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11675 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11676 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11677 let scattered = width & DICTIONARY_SCATTERED != 0;
11678 let has_grams = width & DICTIONARY_GRAMS != 0;
11679 let gram_width =
11680 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11681 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11682 if per_block != TEXT_PAYLOAD_VALUES {
11683 return Err(invalid("global dictionary block width differs"));
11684 }
11685 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11686 return Err(invalid("global dictionary block count differs from its value count"));
11687 }
11688 if offset_bits > u32::BITS as usize {
11689 return Err(invalid("global dictionary packs offsets past a payload"));
11690 }
11691 let offset_len = offset_bytes(count, offset_bits);
11692 let ranks = count;
11697 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11698 let payload_words = if scattered { 3 } else { 2 };
11702 let hash_len = blocks
11703 .checked_mul(payload_words * 8)
11704 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11705 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11706 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11707 let gram_len = if has_grams {
11708 blocks
11709 .checked_mul(gram_width)
11710 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11711 } else {
11712 0
11713 };
11714 let index_len = DICTIONARY_HEADER
11715 .checked_add(offset_len)
11716 .and_then(|len| len.checked_add(hash_len))
11717 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11718 if index_len > page.length as usize {
11719 return Err(invalid("global dictionary offset index exceeds its page"));
11720 }
11721 let mut index = vec![0; index_len];
11722 index[..DICTIONARY_HEADER].copy_from_slice(&header);
11723 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11724 if checksum(&index) != page.hash {
11725 return Err(invalid("global dictionary index checksum differs"));
11726 }
11727 let word_end = index_len - usize::from(has_grams) * 8;
11728 let gram_hash = has_grams
11729 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11730 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11731 .chunks_exact(8)
11732 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11733 .collect::<Vec<_>>();
11734 let mut rest = words.split_off(blocks * payload_words);
11735 let rank_hashes = rest.split_off(rank_blocks);
11736 let rank_ends = rest;
11737 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11740 return Err(invalid("global dictionary order blocks do not rise"));
11741 }
11742 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11743 .map_err(|_| invalid("global dictionary rank overflow"))?;
11744 let body_len = index_len
11745 .checked_add(rank_len)
11746 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11747 if body_len > page.length as usize {
11748 return Err(invalid("global dictionary order exceeds its page"));
11749 }
11750 let gram_end = body_len
11751 .checked_add(gram_len)
11752 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11753 if gram_end > page.length as usize {
11754 return Err(invalid("global dictionary signatures exceed their page"));
11755 }
11756 let grams = gram_hash.map(|hash| NativeGrams {
11757 start: page.offset + body_len as u64,
11758 length: gram_len,
11759 width: gram_width,
11760 hash,
11761 verdicts: Mutex::new(Vec::new()),
11762 });
11763 let mut offsets = index;
11767 offsets.truncate(DICTIONARY_HEADER + offset_len);
11768 let hashes = words.split_off(blocks * (payload_words - 1));
11769 let (starts, lengths) = if scattered {
11770 let mut starts = Vec::with_capacity(blocks);
11771 let mut lengths = Vec::with_capacity(blocks);
11772 for pair in words.chunks_exact(2) {
11773 starts.push(pair[0]);
11774 lengths.push(pair[1]);
11775 }
11776 (starts, lengths)
11777 } else {
11778 let base = page.offset + gram_end as u64;
11782 let mut starts = Vec::with_capacity(blocks);
11783 let mut lengths = Vec::with_capacity(blocks);
11784 let mut at = 0_u64;
11785 for &end in &words {
11786 let len = end
11787 .checked_sub(at)
11788 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11789 starts.push(base + at);
11790 lengths.push(len);
11791 at = end;
11792 }
11793 (starts, lengths)
11794 };
11795 let stored_len = page.length as u64 - gram_end as u64;
11801 if scattered && stored_len == 0 {
11802 let size = file.metadata().map_err(io)?.len();
11803 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11804 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11805 });
11806 if !inside {
11807 return Err(invalid("global dictionary block lies outside the file"));
11808 }
11809 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11810 return Err(invalid("global dictionary blocks do not bound the payload"));
11811 }
11812 Vector::external_text(
11813 ty.clone(),
11814 Arc::new(NativeText {
11815 file,
11816 values: count,
11817 offsets,
11818 offset_bits,
11819 value_ends: OnceLock::new(),
11820 value_lens: OnceLock::new(),
11821 ends_asked: AtomicUsize::new(0),
11822 ranks,
11823 rank_at: page.offset + index_len as u64,
11824 rank_ends,
11825 rank_hashes,
11826 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11827 code_bits: code_width(count),
11828 code_ranks: OnceLock::new(),
11829 starts,
11830 lengths,
11831 hashes,
11832 grams,
11833 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11834 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11835 keep_budget,
11836 payload_kept: AtomicUsize::new(0),
11837 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11838 visit_dropped: AtomicUsize::new(0),
11839 searched: Mutex::new(HashMap::new()),
11840 }),
11841 )
11842}
11843
11844fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11857 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11859 let mut cur = Cursor::new(bytes);
11860 let codec = cur.u8()?;
11861 if cur.u8()? == 2 {
11862 cur.take(rows.div_ceil(8))?;
11863 }
11864 Ok((codec, cur.at))
11865 }
11866 let Ok((codec, at)) = cascade_at(rows, bytes) else {
11867 return "UNREADABLE".to_string();
11868 };
11869 let tail = &bytes[at..];
11870 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11871 match codec {
11872 0 => match ty {
11873 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11874 _ => "FIXED".to_string(),
11875 },
11876 1 => "DICT(PLAIN)".to_string(),
11877 2 => "FOR+BITPACK".to_string(),
11878 3 => "TABLE DICT".to_string(),
11879 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11880 5 => described(integer::describe(tail)),
11881 6 => described(string::describe(tail)),
11882 other => format!("CODEC {other}"),
11883 }
11884}
11885
11886fn decode_selected_stable_codes(
11891 rows: usize,
11892 bytes: &[u8],
11893 positions: &[usize],
11894 out: &mut Vec<Option<u32>>,
11895) -> Result<bool> {
11896 if positions.windows(2).any(|pair| pair[0] >= pair[1])
11897 || positions.last().is_some_and(|&position| position >= rows)
11898 {
11899 return Err(invalid("selected code positions are not sorted and in range"));
11900 }
11901 let mut cur = Cursor::new(bytes);
11902 let codec = cur.u8()?;
11903 if codec != 3 && codec != 4 {
11904 return Ok(false);
11905 }
11906 let flag = cur.u8()?;
11907 let mask = match flag {
11908 0 | 1 => None,
11909 2 => {
11910 let at = cur.at;
11911 let len = rows.div_ceil(8);
11912 cur.take(len)?;
11913 Some((at, len))
11914 }
11915 _ => return Err(invalid("page validity tag differs")),
11916 };
11917 let valid = |row: usize| match flag {
11918 0 => true,
11919 1 => false,
11920 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11921 _ => unreachable!("the validity tag was checked"),
11922 };
11923 if codec == 4 {
11924 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11925 for (&row, code) in positions.iter().zip(wide) {
11926 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11927 out.push(valid(row).then_some(code));
11928 }
11929 return Ok(true);
11930 }
11931 let codes_at = cur.at;
11932 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11933 cur.take(codes_len)?;
11934 if cur.at != bytes.len() {
11935 return Err(invalid("global code page has trailing bytes"));
11936 }
11937 let codes = &bytes[codes_at..codes_at + codes_len];
11938 for &row in positions {
11939 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11940 let code = u32::from_le_bytes(
11941 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11942 );
11943 out.push(valid(row).then_some(code));
11944 }
11945 Ok(true)
11946}
11947
11948fn decode_at(
11954 ty: &LogicalType,
11955 rows: usize,
11956 bytes: &[u8],
11957 global: Option<Arc<Vector>>,
11958 positions: &[u32],
11959) -> Result<Vector> {
11960 if positions.last().is_some_and(|&last| last as usize >= rows) {
11961 return Err(invalid("a position is past the end of the part"));
11962 }
11963 if bytes.first() != Some(&6) {
11964 return decode(ty, rows, bytes, global)?.gather(positions);
11965 }
11966 if !coded_type(ty) {
11967 return Err(invalid("compressed text codec belongs to a non-string page"));
11968 }
11969 let mut cur = Cursor::new(bytes);
11970 cur.u8()?;
11971 let validity = match cur.u8()? {
11972 0 => Validity::AllValid,
11973 1 => Validity::AllInvalid,
11974 2 => {
11975 let mask = cur.take(rows.div_ceil(8))?;
11976 Validity::from_iter(positions.len(), |at| {
11977 let row = positions[at] as usize;
11978 mask[row / 8] >> (row % 8) & 1 == 1
11979 })
11980 }
11981 _ => return Err(invalid("page validity tag differs")),
11982 };
11983 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11984 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11985 push_values(&mut values, ty, &ends)?;
11986 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11987}
11988
11989fn push_values(values: &mut StringColumn, ty: &LogicalType, ends: &[usize]) -> Result<()> {
11993 if ty == &LogicalType::Varchar {
11994 return values.push_run_in_place(0, ends);
11995 }
11996 let mut start = 0;
11997 for &end in ends {
11998 let len = end
11999 .checked_sub(start)
12000 .ok_or_else(|| invalid("a string value ends before it starts"))?;
12001 values.push_bytes_in_place(start, len)?;
12002 start = end;
12003 }
12004 Ok(())
12005}
12006
12007fn decode(
12008 ty: &LogicalType,
12009 rows: usize,
12010 bytes: &[u8],
12011 global: Option<Arc<Vector>>,
12012) -> Result<Vector> {
12013 let mut cur = Cursor::new(bytes);
12014 let codec = cur.u8()?;
12015 let flag = cur.u8()?;
12016 let validity = match flag {
12017 0 => Validity::AllValid,
12018 1 => Validity::AllInvalid,
12019 2 => {
12020 let mask = cur.take(rows.div_ceil(8))?;
12021 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
12022 }
12023 _ => return Err(invalid("page validity tag differs")),
12024 };
12025 if codec == 1 {
12026 if !coded_type(ty) {
12027 return Err(invalid("dictionary codec belongs to a non-string page"));
12028 }
12029 let count = cur.u32()? as usize;
12030 let payload_len = cur.u32()? as usize;
12031 let offset_bytes = cur.take(
12032 (count + 1)
12033 .checked_mul(4)
12034 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
12035 )?;
12036 let offsets = offset_bytes
12037 .chunks_exact(4)
12038 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12039 .collect::<Vec<_>>();
12040 let payload = cur.take(payload_len)?.to_vec();
12041 if offsets.first() != Some(&0)
12042 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12043 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12044 {
12045 return Err(invalid("dictionary offsets do not bound the payload"));
12046 }
12047 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
12050 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12051 push_values(&mut strings, ty, &ends)?;
12052 let mut codes = Vec::with_capacity(rows);
12053 for _ in 0..rows {
12054 codes.push(cur.u32()?);
12055 }
12056 if codes.iter().any(|code| *code as usize >= count) {
12057 return Err(invalid("dictionary code is out of range"));
12058 }
12059 if cur.at != bytes.len() {
12060 return Err(invalid("dictionary page has trailing bytes"));
12061 }
12062 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
12063 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
12064 }
12065 if codec == 3 || codec == 4 {
12066 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
12067 let codes = if codec == 4 {
12068 let codes = integer::decode_as::<u32>(&bytes[cur.at..])
12073 .map_err(|error| invalid(&format!("code is not a code: {error}")))?;
12074 if codes.len() != rows {
12075 return Err(invalid("encoded code page holds the wrong number of rows"));
12076 }
12077 codes
12078 } else {
12079 let mut codes = Vec::with_capacity(rows);
12080 for _ in 0..rows {
12081 codes.push(cur.u32()?);
12082 }
12083 if cur.at != bytes.len() {
12084 return Err(invalid("global code page has trailing bytes"));
12085 }
12086 codes
12087 };
12088 let highest = codes.iter().copied().max();
12089 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
12090 .with_validity(validity));
12091 }
12092 if codec == 6 {
12093 if !coded_type(ty) {
12094 return Err(invalid("compressed text codec belongs to a non-string page"));
12095 }
12096 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
12100 if ends.len() != rows {
12101 return Err(invalid("compressed text page holds the wrong number of rows"));
12102 }
12103 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12106 push_values(&mut values, ty, &ends)?;
12107 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
12108 }
12109 if codec == 5 {
12110 let data = cascade(ty, &bytes[cur.at..], rows)?;
12112 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
12113 }
12114 if codec == 2 {
12115 let width = u32::from(cur.u8()?);
12116 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
12117 let count = cur.u32()? as usize;
12118 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
12119 let words: Vec<u64> = cur
12120 .take(length)?
12121 .chunks_exact(8)
12122 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
12123 .collect();
12124 if cur.at != bytes.len() {
12125 return Err(invalid("packed page has trailing bytes"));
12126 }
12127 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
12128 }
12129 if codec != 0 {
12130 return Err(invalid("page codec is unknown"));
12131 }
12132 let data = match ty {
12133 LogicalType::TinyInt => {
12134 let values = cur.take(rows)?;
12135 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
12136 }
12137 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
12138 LogicalType::SmallInt => {
12139 let values =
12140 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12141 Data::Int16(
12142 values
12143 .chunks_exact(2)
12144 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12145 .collect::<Vec<_>>()
12146 .into(),
12147 )
12148 }
12149 LogicalType::USmallInt => {
12150 let values =
12151 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12152 Data::UInt16(
12153 values
12154 .chunks_exact(2)
12155 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
12156 .collect::<Vec<_>>()
12157 .into(),
12158 )
12159 }
12160 LogicalType::UInteger => {
12161 let values =
12162 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12163 Data::UInt32(
12164 values
12165 .chunks_exact(4)
12166 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
12167 .collect::<Vec<_>>()
12168 .into(),
12169 )
12170 }
12171 LogicalType::UBigInt => {
12172 let values =
12173 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12174 Data::UInt64(
12175 values
12176 .chunks_exact(8)
12177 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
12178 .collect::<Vec<_>>()
12179 .into(),
12180 )
12181 }
12182 LogicalType::Integer | LogicalType::Date => {
12183 let values =
12184 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12185 Data::Int32(
12186 values
12187 .chunks_exact(4)
12188 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12189 .collect::<Vec<_>>()
12190 .into(),
12191 )
12192 }
12193 LogicalType::BigInt
12194 | LogicalType::Timestamp
12195 | LogicalType::Time
12196 | LogicalType::TimeTz
12197 | LogicalType::TimestampTz
12198 | LogicalType::TimestampS
12199 | LogicalType::TimestampMs
12200 | LogicalType::TimestampNs => {
12201 let values =
12202 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12203 Data::Int64(
12204 values
12205 .chunks_exact(8)
12206 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12207 .collect::<Vec<_>>()
12208 .into(),
12209 )
12210 }
12211 LogicalType::HugeInt | LogicalType::Uuid => {
12212 let values =
12213 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12214 Data::Int128(
12215 values
12216 .chunks_exact(16)
12217 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12218 .collect::<Vec<_>>()
12219 .into(),
12220 )
12221 }
12222 LogicalType::UHugeInt => {
12223 let values =
12224 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12225 Data::UInt128(
12226 values
12227 .chunks_exact(16)
12228 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12229 .collect::<Vec<_>>()
12230 .into(),
12231 )
12232 }
12233 LogicalType::Float => {
12234 let values =
12235 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12236 Data::Float32(
12237 values
12238 .chunks_exact(4)
12239 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12240 .collect::<Vec<_>>()
12241 .into(),
12242 )
12243 }
12244 LogicalType::Double => {
12245 let values =
12246 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12247 Data::Float64(
12248 values
12249 .chunks_exact(8)
12250 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12251 .collect::<Vec<_>>()
12252 .into(),
12253 )
12254 }
12255 LogicalType::Interval => {
12256 let values =
12257 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12258 Data::Interval(
12259 values
12260 .chunks_exact(16)
12261 .map(|item| {
12262 (
12263 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12264 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12265 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12266 )
12267 })
12268 .collect::<Vec<_>>()
12269 .into(),
12270 )
12271 }
12272 LogicalType::Boolean => {
12273 let values = cur.take(rows)?;
12274 if values.iter().any(|value| *value > 1) {
12275 return Err(invalid("boolean page has another value"));
12276 }
12277 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12278 }
12279 LogicalType::Decimal { .. } => match ty.physical() {
12282 PhysicalType::Int16 => {
12283 let values =
12284 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12285 Data::Int16(
12286 values
12287 .chunks_exact(2)
12288 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12289 .collect::<Vec<_>>()
12290 .into(),
12291 )
12292 }
12293 PhysicalType::Int32 => {
12294 let values =
12295 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12296 Data::Int32(
12297 values
12298 .chunks_exact(4)
12299 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12300 .collect::<Vec<_>>()
12301 .into(),
12302 )
12303 }
12304 PhysicalType::Int64 => {
12305 let values =
12306 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12307 Data::Int64(
12308 values
12309 .chunks_exact(8)
12310 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12311 .collect::<Vec<_>>()
12312 .into(),
12313 )
12314 }
12315 _ => {
12316 let values =
12317 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12318 Data::Int128(
12319 values
12320 .chunks_exact(16)
12321 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12322 .collect::<Vec<_>>()
12323 .into(),
12324 )
12325 }
12326 },
12327 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12328 let offset_bytes = cur
12329 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12330 let offsets = offset_bytes
12331 .chunks_exact(4)
12332 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12333 .collect::<Vec<_>>();
12334 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12335 if offsets.first() != Some(&0)
12336 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12337 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12338 {
12339 return Err(invalid("string offsets do not bound the payload"));
12340 }
12341 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12349 let ends: Vec<usize> = offsets[1..].iter().map(|&end| end as usize).collect();
12350 push_values(&mut values, ty, &ends)?;
12351 Data::Varlen(values)
12352 }
12353 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12354 };
12355 if cur.at != bytes.len() {
12356 return Err(invalid("page has trailing bytes"));
12357 }
12358 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12359}
12360
12361#[cfg(test)]
12362mod tests {
12363 use std::fs::{self, OpenOptions};
12364 use std::io::{Seek, SeekFrom, Write};
12365 use std::path::PathBuf;
12366 use std::time::{SystemTime, UNIX_EPOCH};
12367
12368 use rudb_common::Stat;
12369 use rudb_common::Value;
12370 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12371 use rudb_common::stat::Provenance;
12372
12373 use super::*;
12374
12375 #[test]
12376 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12377 let bytes: Vec<u8> =
12378 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12379 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12380 let whole = content_name(&bytes[..length]);
12381 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12382 let mut namer = ContentNamer::default();
12383 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12384 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12385 }
12386 }
12387 }
12388
12389 #[derive(Debug)]
12392 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12393
12394 impl chooser::Chooser for TestsEverything<'_> {
12395 fn name(&self) -> &'static str {
12396 "tests everything"
12397 }
12398
12399 fn narrow_strings(
12400 &self,
12401 values: &[&[u8]],
12402 offered: &[string::Kind],
12403 depth: u8,
12404 ) -> Vec<string::Kind> {
12405 self.0.narrow_strings(values, offered, depth)
12406 }
12407
12408 fn narrow_integers(
12409 &self,
12410 values: &[i64],
12411 offered: &[integer::Kind],
12412 depth: u8,
12413 ) -> Vec<integer::Kind> {
12414 self.0.narrow_integers(values, offered, depth)
12415 }
12416 }
12417
12418 #[test]
12419 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12420 let columns: Vec<Vec<i64>> = vec![
12421 vec![],
12422 vec![5; 1000],
12423 (0..1000).collect(),
12424 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12425 (0..1000).map(|row| row / 50).collect(),
12426 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12427 (0..1000).map(|row| (row * 7919) % 13).collect(),
12428 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12429 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12430 (0..1000).map(|row| i64::MIN + row % 3).collect(),
12431 ];
12432 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12433 for column in &columns {
12434 for chooser in choosers {
12435 let quick = integer::encode_with(column, chooser).unwrap();
12436 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12437 assert_eq!(
12438 quick,
12439 full,
12440 "{} on {:?}",
12441 chooser.name(),
12442 &column[..column.len().min(8)]
12443 );
12444 }
12445 }
12446 }
12447
12448 #[test]
12451 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12452 let mut settling = Settling::default();
12453 for part in 0..STRIPE_PARTS as i64 {
12454 let values: Vec<i64> = (0..2048)
12455 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12456 .collect();
12457 let searched = integer::encode_with(&values, &Fixed).unwrap();
12458 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12459 }
12460 }
12461
12462 #[test]
12466 fn text_pages_share_a_table_until_the_text_changes() {
12467 let words = ["carefully", "final", "deposits", "sleep", "furiously", "quickly", "among"];
12468 let english: Vec<Vec<u8>> = (0..1024)
12469 .map(|row: usize| {
12470 let pick = |at: usize| words[(row * 7 + at * 3) % words.len()];
12471 format!("{} {} {} the {}", pick(0), pick(1), pick(2), pick(3)).into_bytes()
12472 })
12473 .collect();
12474 let digits: Vec<Vec<u8>> =
12475 (0..1024_u64).map(|row| format!("{:020}", row * 7_919_993).into_bytes()).collect();
12476 let mut settling = Settling::default();
12477 for page in 0..8 {
12478 let values: Vec<&[u8]> =
12479 if page < 4 { &english } else { &digits }.iter().map(Vec::as_slice).collect();
12480 let payload = values.iter().map(|value| value.len()).sum();
12481 let out = settling.text(&values, payload).unwrap().unwrap();
12482 assert_eq!(string::decode(&out).unwrap(), values, "page {page}");
12483 let alone = string::encode_only(string::Kind::Fsst, &values).unwrap().unwrap();
12484 assert!(
12485 out.len() * 4 <= alone.len() * 5,
12486 "page {page}: {} against {}",
12487 out.len(),
12488 alone.len()
12489 );
12490 let since = settling.symbols.as_ref().unwrap().since;
12491 assert_eq!(since, page % 4, "page {page}");
12492 }
12493 }
12494
12495 #[test]
12499 fn a_column_that_changes_under_the_shape_is_searched_again() {
12500 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12501 let mut noise = move || {
12502 state ^= state << 13;
12503 state ^= state >> 7;
12504 state ^= state << 17;
12505 (state % 1_000_000) as i64
12506 };
12507 let mut settling = Settling::default();
12508 for part in 0..STRIPE_PARTS as i64 {
12509 let values: Vec<i64> = match part / 16 {
12510 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12511 1 => (0..2048).map(|_| noise()).collect(),
12512 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12513 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12514 };
12515 let settled = settling.encode(&values).unwrap();
12516 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12517 let searched = integer::encode_with(&values, &Fixed).unwrap();
12518 assert!(
12519 settled.len() * 4 <= searched.len() * 5,
12520 "part {part}: {} settled against {} searched, {} against {}",
12521 settled.len(),
12522 searched.len(),
12523 integer::describe(&settled).unwrap(),
12524 integer::describe(&searched).unwrap(),
12525 );
12526 }
12527 }
12528
12529 #[test]
12530 fn checksum_matches_fixed_vectors() {
12531 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12532 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12533 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12534 }
12535
12536 #[test]
12537 fn sorting_across_threads_matches_sorting_on_one() {
12538 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12539 let mut next = move || {
12540 state ^= state << 13;
12541 state ^= state >> 7;
12542 state ^= state << 17;
12543 state
12544 };
12545 let mut values = Vec::new();
12546 for at in 0..150_000_u64 {
12547 let value = match next() % 6 {
12548 0 => Vec::new(),
12549 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12550 2 => format!("https://example.com/path/{at}").into_bytes(),
12551 3 => b"same".to_vec(),
12552 4 => vec![0xff; (next() % 12) as usize],
12553 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12554 };
12555 values.push(value);
12556 }
12557 let value = |code: u32| values[code as usize].as_slice();
12558 for workers in [1, 2, 3, 8, 32] {
12559 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12560 let mut across = one.clone();
12561 sort_by_value(&mut one, value);
12562 sort_by_value_across(&mut across, value, workers);
12563 assert_eq!(one, across, "{workers} workers");
12564 }
12565 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12566 sort_by_value_across(&mut sorted, value, 8);
12567 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12568 }
12569
12570 fn path(label: &str) -> PathBuf {
12571 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12572 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12573 }
12574
12575 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12580 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12581 (0..dictionary.values())
12582 .map(|code| {
12583 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12584 flat[from..to].to_vec()
12585 })
12586 .collect()
12587 }
12588
12589 fn attached(table: &Table) -> Vec<&Section> {
12596 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12597 }
12598
12599 #[test]
12601 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12602 const SPANS: usize = 64;
12603 const SPAN: usize = 512;
12604 let path = path("positional");
12605 let content: Vec<u8> =
12606 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12607 fs::write(&path, &content).expect("the file is written");
12608 let file = Arc::new(File::open(&path).expect("the file opens"));
12609 std::thread::scope(|scope| {
12610 for _ in 0..8 {
12611 let file = Arc::clone(&file);
12612 scope.spawn(move || {
12613 for _ in 0..64 {
12614 for span in 0..SPANS {
12615 let mut bytes = [0_u8; SPAN];
12616 read_at(&file, (span * SPAN) as u64, &mut bytes)
12617 .expect("the span reads");
12618 assert!(
12619 bytes.iter().all(|byte| *byte == span as u8),
12620 "span {span} came back as {}",
12621 bytes[0],
12622 );
12623 }
12624 }
12625 });
12626 }
12627 });
12628 let mut past = [0_u8; SPAN];
12629 let end = (SPANS * SPAN) as u64;
12630 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12631 assert!(error.message().contains("ends before its declared length"), "{error}");
12632 drop(file);
12633 let _ = fs::remove_file(&path);
12634 }
12635
12636 #[test]
12643 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12644 let path = path("cursor");
12645 let mut writer = Writer::create(
12646 &path,
12647 "items",
12648 vec![
12649 Field::required("id", LogicalType::Integer),
12650 Field::new("text", LogicalType::Varchar),
12651 ],
12652 )
12653 .expect("new file");
12654 writer.append(&sample()).expect("first part");
12655 writer.append(&sample()).expect("second part");
12656 writer.finish().expect("commit");
12657 let reader = Reader::open(&path).expect("reopen from disk");
12658 assert_eq!(reader.table().rows(), 6);
12659 let ids = reader.read(0, &[0]).expect("the integer page reads back");
12660 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12661 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12662 let text = reader.read(1, &[1]).expect("the text page reads back");
12663 assert_eq!(text.value_at(1, 0), Value::Null);
12664 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12665 let end = reader.table().stripes().iter().flat_map(|stripe| {
12668 stripe
12669 .pages
12670 .iter()
12671 .map(|page| page.offset + u64::from(page.length))
12672 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12673 });
12674 let last = end.fold(HEADER, u64::max);
12675 let directory = fs::metadata(&path).expect("the file is there").len();
12676 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12677 fs::remove_file(path).expect("remove scratch file");
12678 }
12679
12680 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12686 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12687 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12688 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12689 let bits = (width & !DICTIONARY_FLAGS) as usize;
12690 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12691 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12692 DICTIONARY_HEADER as u64
12693 + offset_bytes(count as usize, bits) as u64
12694 + blocks * payload_words * 8
12695 + rank_blocks * 16
12696 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12697 }
12698
12699 fn sample() -> Chunk {
12700 Chunk::new(vec![
12701 Vector::from_values(
12702 LogicalType::Integer,
12703 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12704 )
12705 .expect("integers"),
12706 Vector::from_values(
12707 LogicalType::Varchar,
12708 &[
12709 Value::Varchar("alpha".into()),
12710 Value::Null,
12711 Value::Varchar("long text after a slash".into()),
12712 ],
12713 )
12714 .expect("strings"),
12715 ])
12716 .expect("matching rows")
12717 }
12718
12719 fn sample_ids() -> Chunk {
12720 Chunk::new(vec![
12721 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12722 .expect("integers"),
12723 ])
12724 .expect("one column")
12725 }
12726
12727 #[test]
12728 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12729 let path = path("nulls_for_the_planner");
12732 let mut writer =
12733 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12734 .expect("new file");
12735 let rows = Chunk::new(vec![
12736 Vector::from_values(
12737 LogicalType::Integer,
12738 &[
12739 Value::Integer(4),
12740 Value::Null,
12741 Value::Integer(9),
12742 Value::Null,
12743 Value::Integer(1),
12744 Value::Integer(2),
12745 ],
12746 )
12747 .expect("integers"),
12748 ])
12749 .expect("one column");
12750 writer.append(&rows).expect("the only part");
12751 writer.finish().expect("commit");
12752 let reader = Reader::open(&path).expect("reopen from disk");
12753 let stripes = Stripes::new(reader);
12754 let column = stripes.column("a").expect("the file has that column");
12755 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12756 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12759 fs::remove_file(&path).expect("clean up");
12760 }
12761
12762 #[test]
12763 fn the_planner_gets_a_leading_count_without_a_complete_numeric_synopsis() {
12764 let path = path("frequencies_for_the_planner");
12767 let mut writer =
12768 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12769 .expect("new file");
12770 let rows = Chunk::new(vec![
12771 Vector::from_values(
12772 LogicalType::Integer,
12773 &[
12774 Value::Integer(4),
12775 Value::Integer(4),
12776 Value::Integer(4),
12777 Value::Integer(9),
12778 Value::Integer(9),
12779 Value::Integer(1),
12780 ],
12781 )
12782 .expect("integers"),
12783 ])
12784 .expect("one column");
12785 writer.append(&rows).expect("the only part");
12786 writer.finish().expect("commit");
12787 let reader = Reader::open(&path).expect("reopen from disk");
12788 let common = Common::new(reader);
12789 assert_eq!(common.rows(), 6);
12790 let column = common.column("id").expect("the file has that column");
12791 assert_eq!(common.column("nothing"), None);
12792 assert_eq!(
12793 common.rows_with(column, &Bound::Int(4)),
12794 Stat::exact(3, Provenance::FrequencySynopsis)
12795 );
12796 assert_eq!(common.rows_with(column, &Bound::Int(7)), Stat::Unknown);
12798 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12801 assert!(common.remainder(column).is_some());
12802 fs::remove_file(&path).expect("clean up");
12803 }
12804
12805 #[test]
12806 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12807 let path = path("string_frequencies_for_the_planner");
12808 let mut writer =
12809 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12810 .expect("new file");
12811 let rows = Chunk::new(vec![
12812 Vector::from_values(
12813 LogicalType::Varchar,
12814 &[
12815 Value::Varchar(String::new()),
12816 Value::Varchar("alpha".into()),
12817 Value::Varchar(String::new()),
12818 Value::Varchar("beta".into()),
12819 Value::Varchar(String::new()),
12820 ],
12821 )
12822 .expect("strings"),
12823 ])
12824 .expect("one column");
12825 writer.append(&rows).expect("the only part");
12826 writer.finish().expect("commit");
12827
12828 let reader = Reader::open(&path).expect("reopen from disk");
12829 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12830 let common = Common::new(reader.clone());
12831 let column = common.column("text").expect("the file has that column");
12832 assert_eq!(
12833 common.rows_with(column, &Bound::Bytes(Vec::new())),
12834 Stat::exact(3, Provenance::FrequencySynopsis)
12835 );
12836 assert_eq!(
12837 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12838 Stat::exact(0, Provenance::FrequencySynopsis)
12839 );
12840 assert_eq!(
12841 reader.reads().dictionaries,
12842 0,
12843 "the bounded spellings answer without opening the dictionary index"
12844 );
12845 fs::remove_file(&path).expect("clean up");
12846 }
12847
12848 #[test]
12849 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12850 let path = path("certified_host_groups");
12851 let mut writer =
12852 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12853 .expect("new file");
12854 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12855 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12856 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12857 values.push(Value::Varchar(String::new()));
12858 for part in values.chunks(512) {
12859 writer
12860 .append(
12861 &Chunk::new(vec![
12862 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12863 ])
12864 .expect("one column"),
12865 )
12866 .expect("part written");
12867 }
12868 writer.finish().expect("commit");
12869 let reader = Reader::open(&path).expect("reopen");
12870 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12871 fs::remove_file(&path).expect("clean up");
12872 }
12873
12874 fn bare_table(sections: Vec<Section>) -> Table {
12879 Table {
12880 name: "linked".to_owned(),
12881 fields: vec![Field::required("id", LogicalType::Integer)],
12882 stripes: Vec::new(),
12883 rows: 0,
12884 dictionaries: vec![None],
12885 dictionary_payloads: Vec::new(),
12886 demoted: Vec::new(),
12887 distincts: vec![None],
12888 frequencies: vec![None],
12889 pair_frequencies: Vec::new(),
12890 frequency_texts: Vec::new(),
12891 host_groups: None,
12892 clustering: None,
12893 generation: 1,
12894 sections,
12895 }
12896 }
12897
12898 fn a_key_map_section() -> Section {
12899 Section {
12900 kind: *section::KEY_MAP,
12901 id: 1,
12902 generation: 3,
12903 extents: 1,
12904 extent_page: HEADER,
12905 extent_bytes: section::EXTENT_BYTES as u32,
12906 hash: 0x1234_5678_9abc_def0,
12907 flags: 0,
12908 header_bytes: 24,
12909 }
12910 }
12911
12912 #[test]
12913 fn a_section_table_round_trips_through_a_directory() {
12914 let mut later = a_key_map_section();
12915 later.kind = *b"RUDBZZ9\0";
12916 later.id = 2;
12917 let table = bare_table(vec![a_key_map_section(), later]);
12918 let directory = encode_directory(&table).expect("directory");
12919 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12920 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12921 assert!(decoded.sections()[0].known());
12925 assert!(!decoded.sections()[1].known());
12926 }
12927
12928 #[test]
12929 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12930 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12934 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
12935 let older = &directory[..directory.len() - block];
12936 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
12937 assert!(decoded.sections().is_empty());
12938 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
12939 assert_eq!(decoded.name(), "linked");
12940 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
12941 }
12942
12943 #[test]
12944 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
12945 let path = path("format_twenty_two");
12952 let mut writer =
12953 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12954 .expect("new file");
12955 let rows = Chunk::new(vec![
12956 Vector::from_values(
12957 LogicalType::Integer,
12958 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
12959 )
12960 .expect("integers"),
12961 ])
12962 .expect("one column");
12963 writer.append(&rows).expect("the only part");
12964 writer.finish().expect("commit");
12965
12966 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12967 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12968 drop(file);
12969
12970 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
12971 assert_eq!(reader.table().rows(), 3);
12972 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
12977
12978 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12981 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
12982 drop(file);
12983 let error = Reader::open(&path).expect_err("format 21 is not readable");
12984 assert!(error.to_string().contains("format 21"), "{error}");
12985
12986 fs::remove_file(&path).expect("clean up");
12987 }
12988
12989 #[test]
12990 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
12991 let mut past = a_key_map_section();
12996 past.extent_page = 1 << 30;
12997 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
12998 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
12999 assert!(error.to_string().contains("outside the file"), "{error}");
13000
13001 let mut inside_the_header = a_key_map_section();
13002 inside_the_header.extent_page = 8;
13003 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
13004 assert!(
13005 decode_directory(&directory, 1 << 20).is_err(),
13006 "a section may not overlap a header"
13007 );
13008 }
13009
13010 #[test]
13011 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
13012 let not_built = Section {
13016 kind: *section::FORWARD_LINK,
13017 id: 9,
13018 generation: 3,
13019 extents: 0,
13020 extent_page: 0,
13021 extent_bytes: 0,
13022 hash: 0,
13023 flags: 0,
13024 header_bytes: 0,
13025 };
13026 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
13027 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
13028 assert_eq!(decoded.sections(), &[not_built]);
13029
13030 let mut incoherent = not_built;
13033 incoherent.extent_bytes = 28;
13034 incoherent.extent_page = HEADER;
13035 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
13036 assert!(decode_directory(&directory, 1 << 20).is_err());
13037 }
13038
13039 #[test]
13040 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
13041 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
13042 let mut torn = directory.clone();
13043 let count_at = torn.len() - size_of::<u16>();
13044 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
13045 assert!(decode_directory(&torn, 1 << 20).is_err());
13048 }
13049
13050 fn linked_file(label: &str, rows: i32) -> PathBuf {
13052 let path = path(label);
13053 let mut writer =
13054 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13055 .expect("new file");
13056 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
13057 let chunk =
13058 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
13059 .expect("one column");
13060 writer.append(&chunk).expect("the only part");
13061 writer.finish().expect("commit");
13062 path
13063 }
13064
13065 fn a_key_map_payload() -> Vec<u8> {
13066 (0..512_u32).flat_map(u32::to_le_bytes).collect()
13069 }
13070
13071 #[test]
13072 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
13073 let path = linked_file("attach", 64);
13074 let payload = a_key_map_payload();
13075 let table = attach(
13076 &path,
13077 "items",
13078 &[section::Attachment {
13079 kind: *section::KEY_MAP,
13080 id: 0,
13081 flags: 2,
13082 header_bytes: 40,
13083 bytes: &payload,
13084 }],
13085 )
13086 .expect("attach a key map");
13087 assert_eq!(attached(&table).len(), 1);
13088
13089 let reader = Reader::open(&path).expect("reopen after the attach");
13090 let held = attached(reader.table());
13091 assert_eq!(held.len(), 1);
13092 assert_eq!(held[0].kind, *section::KEY_MAP);
13093 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
13094 assert_eq!(held[0].header_bytes, 40);
13095 assert_eq!(held[0].generation, 1);
13099 assert!(held[0].usable(reader.table().generation()));
13100 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
13101 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
13102
13103 fs::remove_file(&path).expect("clean up");
13104 }
13105
13106 #[test]
13107 fn attaching_a_section_answers_every_row_exactly_as_before() {
13108 let path = linked_file("attach_changes_nothing", 300);
13113 let before = Reader::open(&path).expect("open before");
13114 let rows = before.table().rows();
13115 let first = before.read(0, &[0]).expect("read before");
13116 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
13117 let layout = before.layout().columns_total();
13118 drop(before);
13119
13120 let payload = a_key_map_payload();
13121 attach(
13122 &path,
13123 "items",
13124 &[section::Attachment {
13125 kind: *section::KEY_MAP,
13126 id: 0,
13127 flags: 0,
13128 header_bytes: 0,
13129 bytes: &payload,
13130 }],
13131 )
13132 .expect("attach");
13133
13134 let after = Reader::open(&path).expect("open after");
13135 assert_eq!(after.table().rows(), rows);
13136 let read = after.read(0, &[0]).expect("read after");
13137 for (at, value) in values.iter().enumerate() {
13138 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
13139 }
13140 assert_eq!(
13141 after.layout().columns_total(),
13142 layout,
13143 "an attach appends and does not rewrite a column page"
13144 );
13145
13146 fs::remove_file(&path).expect("clean up");
13147 }
13148
13149 #[test]
13150 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
13151 let path = linked_file("attach_twice", 32);
13155 let one = a_key_map_payload();
13156 let two = vec![7_u8; 1024];
13157 let entry = |bytes| section::Attachment {
13158 kind: *section::KEY_MAP,
13159 id: 4,
13160 flags: 1,
13161 header_bytes: 0,
13162 bytes,
13163 };
13164 attach(&path, "items", &[entry(&one)]).expect("first build");
13165 attach(&path, "items", &[entry(&two)]).expect("rebuild");
13166
13167 let reader = Reader::open(&path).expect("reopen");
13168 let held = attached(reader.table());
13169 assert_eq!(held.len(), 1, "one map per column and not one per build");
13170 assert_eq!(reader.payload(held[0]).expect("payload"), two);
13171
13172 fs::remove_file(&path).expect("clean up");
13173 }
13174
13175 #[test]
13176 fn an_attach_carries_through_a_kind_it_does_not_know() {
13177 let path = linked_file("attach_unknown", 16);
13181 let payload = vec![3_u8; 96];
13182 attach(
13183 &path,
13184 "items",
13185 &[section::Attachment {
13186 kind: *b"RUDBZZ9\0",
13187 id: 1,
13188 flags: 0,
13189 header_bytes: 0,
13190 bytes: &payload,
13191 }],
13192 )
13193 .expect("a kind this build does not know still writes");
13194 let key_map = a_key_map_payload();
13195 attach(
13196 &path,
13197 "items",
13198 &[section::Attachment {
13199 kind: *section::KEY_MAP,
13200 id: 0,
13201 flags: 0,
13202 header_bytes: 0,
13203 bytes: &key_map,
13204 }],
13205 )
13206 .expect("attach beside it");
13207
13208 let reader = Reader::open(&path).expect("reopen");
13209 let held = attached(reader.table());
13210 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
13211 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
13212 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
13213
13214 fs::remove_file(&path).expect("clean up");
13215 }
13216
13217 #[test]
13218 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
13219 let path = linked_file("attach_not_built", 8);
13220 attach(
13221 &path,
13222 "items",
13223 &[section::Attachment {
13224 kind: *section::FORWARD_LINK,
13225 id: 2,
13226 flags: 0,
13227 header_bytes: 0,
13228 bytes: &[],
13229 }],
13230 )
13231 .expect("record a link that did not fit the budget");
13232
13233 let reader = Reader::open(&path).expect("reopen");
13234 let held = attached(reader.table());
13235 assert_eq!(held.len(), 1);
13236 assert_eq!(held[0].extents, 0);
13237 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13238 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13239 assert!(reader.payload(held[0]).expect("no payload").is_empty());
13240
13241 fs::remove_file(&path).expect("clean up");
13242 }
13243
13244 #[test]
13245 fn a_payload_past_one_extent_is_split_and_joined_back() {
13246 let path = linked_file("attach_two_extents", 8);
13250 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13251 attach(
13252 &path,
13253 "items",
13254 &[section::Attachment {
13255 kind: *section::KEY_MAP,
13256 id: 0,
13257 flags: 0,
13258 header_bytes: 0,
13259 bytes: &payload,
13260 }],
13261 )
13262 .expect("attach a payload past the bound");
13263
13264 let reader = Reader::open(&path).expect("reopen");
13265 let held = attached(reader.table());
13266 let extents = reader.extents(held[0]).expect("extent table");
13267 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13268 assert_eq!(extents[0].length, section::MAX_EXTENT);
13269 assert_eq!(extents[1].length, 1);
13270 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13271 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13273 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13274
13275 fs::remove_file(&path).expect("clean up");
13276 }
13277
13278 #[test]
13279 fn a_torn_extent_is_refused_rather_than_decoded() {
13280 let path = linked_file("attach_torn", 8);
13281 let payload = a_key_map_payload();
13282 attach(
13283 &path,
13284 "items",
13285 &[section::Attachment {
13286 kind: *section::KEY_MAP,
13287 id: 0,
13288 flags: 0,
13289 header_bytes: 0,
13290 bytes: &payload,
13291 }],
13292 )
13293 .expect("attach");
13294
13295 let reader = Reader::open(&path).expect("reopen");
13296 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13297 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13298 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13299 drop(file);
13300
13301 let reader = Reader::open(&path).expect("the table still opens");
13302 let error = reader
13303 .payload(&reader.table().sections()[0])
13304 .expect_err("a corrupt payload is not handed out");
13305 assert!(error.to_string().contains("checksum"), "{error}");
13306 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13309
13310 fs::remove_file(&path).expect("clean up");
13311 }
13312
13313 #[test]
13314 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13315 let path = linked_file("attach_old_format", 8);
13318 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13319 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13320 drop(file);
13321
13322 let payload = a_key_map_payload();
13323 let error = attach(
13324 &path,
13325 "items",
13326 &[section::Attachment {
13327 kind: *section::KEY_MAP,
13328 id: 0,
13329 flags: 0,
13330 header_bytes: 0,
13331 bytes: &payload,
13332 }],
13333 )
13334 .expect_err("format 22 cannot gain a section");
13335 assert!(error.to_string().contains("format 22"), "{error}");
13336 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13337
13338 fs::remove_file(&path).expect("clean up");
13339 }
13340
13341 #[test]
13342 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13343 let path = linked_file("attach_bad_header", 8);
13344 let error = attach(
13345 &path,
13346 "items",
13347 &[section::Attachment {
13348 kind: *section::KEY_MAP,
13349 id: 0,
13350 flags: 0,
13351 header_bytes: 40,
13352 bytes: &[1, 2, 3],
13353 }],
13354 )
13355 .expect_err("a writer's bug stops at the write");
13356 assert!(error.to_string().contains("header is longer"), "{error}");
13357
13358 fs::remove_file(&path).expect("clean up");
13359 }
13360
13361 #[test]
13362 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13363 let path = linked_file("attach_wrong_name", 8);
13364 let error = attach(&path, "orders", &[]).expect_err("no such table");
13365 assert!(error.to_string().contains("orders"), "{error}");
13366 fs::remove_file(&path).expect("clean up");
13367 }
13368
13369 #[test]
13370 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13371 let path = path("frequency_prefix_for_the_planner");
13378 let mut writer =
13379 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13380 .expect("new file");
13381 let mut values = vec![Value::Integer(1); 10_000];
13382 for _ in 0..10 {
13383 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13384 }
13385 for part in values.chunks(8_000) {
13388 let rows = Chunk::new(vec![
13389 Vector::from_values(LogicalType::Integer, part).expect("integers"),
13390 ])
13391 .expect("one column");
13392 writer.append(&rows).expect("a part");
13393 }
13394 writer.finish().expect("commit");
13395 let reader = Reader::open(&path).expect("reopen from disk");
13396 let prefix =
13397 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13398 assert_eq!(prefix.entries.len(), 512);
13401 assert_eq!(prefix.omitted_max, 10);
13402 let common = Common::new(reader);
13403 assert_eq!(common.rows(), 16_000);
13404 let column = common.column("id").expect("the file has that column");
13405 assert_eq!(
13406 common.rows_with(column, &Bound::Int(1)),
13407 Stat::exact(10_000, Provenance::FrequencySynopsis)
13408 );
13409 assert_eq!(
13411 common.rows_with(column, &Bound::Int(1_100)),
13412 Stat::exact(10, Provenance::FrequencySynopsis)
13413 );
13414 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13417 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13420 let remainder = common.remainder(column).expect("the list is a prefix");
13424 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13425 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13426 fs::remove_file(&path).expect("clean up");
13427 }
13428
13429 #[test]
13431 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13432 let path = path("empty");
13433 Writer::empty(&path, &[]).expect("a file with nothing in it");
13434 let catalog = Catalog::open(&path).expect("the empty file opens");
13435 assert_eq!(catalog.len(), 0);
13436 assert!(catalog.is_empty());
13437 assert_eq!(catalog.names().count(), 0);
13438 let mut writer =
13441 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13442 .expect("a table goes into the empty file");
13443 writer.append(&sample_ids()).expect("rows");
13444 writer.finish().expect("commit");
13445 let catalog = Catalog::open(&path).expect("the file opens again");
13446 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13447 fs::remove_file(&path).expect("clean up");
13448 }
13449
13450 #[test]
13460 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13461 let path = path("empty-name");
13462 let field = || vec![Field::required("id", LogicalType::Integer)];
13463 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13464 let catalog = Catalog::open(&path).expect("the file opens");
13465 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13466
13467 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13468 writer.append(&sample_ids()).expect("rows");
13469 writer.finish().expect("commit");
13470 let catalog = Catalog::open(&path).expect("the file opens again");
13471 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13473 let held = catalog.rows().collect::<Vec<_>>();
13474 assert_eq!(held.len(), 1);
13475 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13476
13477 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13479 assert!(error.to_string().contains("same name"), "{error}");
13480 fs::remove_file(&path).expect("clean up");
13481 }
13482
13483 fn sample_view(name: &str) -> ViewEntry {
13485 ViewEntry {
13486 name: name.to_string(),
13487 sql: "SELECT id FROM items WHERE id > 0".to_string(),
13488 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13489 aliases: vec!["n".to_string()],
13490 columns: vec![Field::new("n", LogicalType::Integer)],
13491 }
13492 }
13493
13494 #[test]
13495 fn a_view_written_into_the_catalog_comes_back_whole() {
13496 let path = path("views");
13497 let mut writer =
13498 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13499 .expect("new file");
13500 writer.append(&sample_ids()).expect("rows");
13501 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13502 let catalog = Catalog::open(&path).expect("reopen");
13503 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13504 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13507 fs::remove_file(&path).expect("clean up");
13508 }
13509
13510 #[test]
13512 fn appending_a_table_carries_the_views_forward() {
13513 let path = path("viewscarry");
13514 let mut writer =
13515 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13516 .expect("new file");
13517 writer.append(&sample_ids()).expect("rows");
13518 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13519 let mut writer =
13520 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13521 .expect("a second table");
13522 writer.append(&sample_ids()).expect("rows");
13523 writer.finish().expect("commit");
13524 let catalog = Catalog::open(&path).expect("reopen");
13525 assert_eq!(catalog.views().count(), 1);
13526 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13527 fs::remove_file(&path).expect("clean up");
13528 }
13529
13530 #[test]
13532 fn restating_the_views_leaves_every_table_where_it_was() {
13533 let path = path("restate");
13534 let mut writer =
13535 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13536 .expect("new file");
13537 writer.append(&sample_ids()).expect("rows");
13538 writer.finish().expect("commit");
13539 let before = fs::metadata(&path).expect("the file is there").len();
13540 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13541 let catalog = Catalog::open(&path).expect("reopen");
13542 assert_eq!(catalog.views().count(), 2);
13543 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13544 let after = fs::metadata(&path).expect("the file is there").len();
13547 assert!(after > before, "a generation was written");
13548 assert!(after - before < before, "the table was not written again");
13549 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13552 assert_eq!(reader.table().rows, 3);
13553 Writer::restate(&path, &[]).expect("no views at all");
13556 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13557 fs::remove_file(&path).expect("clean up");
13558 }
13559
13560 #[test]
13562 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13563 let bytes = encode_catalog(
13564 &[Entry {
13565 name: "items".to_string(),
13566 fields: vec![Field::required("id", LogicalType::Integer)],
13567 rows: 1,
13568 directory: Page { offset: HEADER, length: 8, hash: 0 },
13569 nonzero: vec![None],
13570 aggregates: vec![None],
13571 distincts: vec![None],
13572 extremes: vec![None],
13573 frequencies: vec![None],
13574 }],
13575 &[sample_view("items")],
13576 )
13577 .expect("it encodes, because encoding does not look");
13578 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13579 assert!(error.to_string().contains("same name"), "{error}");
13580 }
13581
13582 #[test]
13585 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13586 let rows: usize = 300;
13587 let text: Vec<String> =
13588 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13589 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13590 let mut page = vec![6, 2];
13591 page.extend((0..rows.div_ceil(8)).map(|byte| {
13592 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13593 }));
13594 let compressed = string::encode_only(string::Kind::Fsst, &values)
13595 .expect("encoded")
13596 .expect("text this repetitive compresses");
13597 page.extend_from_slice(&compressed);
13598 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13599 let positions = [0_u32, 3, 8, 13, 200, 299];
13600 let some =
13601 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13602 assert_eq!(some.len(), positions.len());
13603 for (at, &row) in positions.iter().enumerate() {
13604 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13605 }
13606 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13607 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13608 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13609 }
13610
13611 #[test]
13614 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13615 let path = path("rows");
13616 let mut writer = Writer::create(
13617 &path,
13618 "items",
13619 vec![
13620 Field::required("id", LogicalType::Integer),
13621 Field::new("text", LogicalType::Varchar),
13622 ],
13623 )
13624 .expect("new file");
13625 let rows = 2_000;
13626 let chunk = Chunk::new(vec![
13627 Vector::from_values(
13628 LogicalType::Integer,
13629 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13630 )
13631 .expect("integers"),
13632 Vector::from_values(
13633 LogicalType::Varchar,
13634 &(0..rows)
13635 .map(|row| {
13636 if row % 7 == 2 {
13637 Value::Null
13638 } else {
13639 Value::Varchar(format!("a comment about order {}", row * 13))
13640 }
13641 })
13642 .collect::<Vec<_>>(),
13643 )
13644 .expect("strings"),
13645 ])
13646 .expect("matching rows");
13647 writer.append(&chunk).expect("one part");
13648 writer.finish().expect("commit");
13649 let reader = Reader::open(&path).expect("reopen from disk");
13650 let positions = [1_u32, 2, 9, 1_000, 1_999];
13651 for whole in [true, false] {
13652 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13653 let all = reader.read(0, &[0, 1]).expect("the whole part");
13654 assert_eq!(some.len(), positions.len());
13655 for column in 0..2 {
13656 for (at, &row) in positions.iter().enumerate() {
13657 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13658 }
13659 }
13660 }
13661 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13662 }
13663
13664 #[test]
13665 fn committed_file_reopens_and_reads_only_requested_columns() {
13666 let path = path("reopen");
13667 let mut writer = Writer::create(
13668 &path,
13669 "items",
13670 vec![
13671 Field::required("id", LogicalType::Integer),
13672 Field::new("text", LogicalType::Varchar),
13673 ],
13674 )
13675 .expect("new file");
13676 writer.append(&sample()).expect("first part");
13677 writer.append(&sample()).expect("second part");
13678 writer.finish().expect("commit");
13679 let reader = Reader::open(&path).expect("reopen from disk");
13680 assert_eq!(reader.table().rows(), 6);
13681 assert_eq!(reader.table().stripes().len(), 1);
13684 assert_eq!(reader.parts(), 2);
13685 assert_eq!(reader.part_rows(0), 3);
13686 assert_eq!(reader.part_rows(1), 3);
13687 let text = reader.read(1, &[1]).expect("only text page");
13688 assert_eq!(text.width(), 1);
13689 assert_eq!(text.value_at(1, 0), Value::Null);
13690 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13691 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13692 assert_eq!(sparse.width(), 1);
13693 assert_eq!(sparse.value_at(1, 0), Value::Null);
13694 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13695 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13696 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13697 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13698 let count = reader.read(0, &[]).expect("no page is needed for count");
13699 assert_eq!(count.len(), 3);
13700 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13701 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13702 assert_eq!(reader.top_frequencies(0, 1).expect("valid integer synopsis"), None);
13703 let integers = reader.frequency_prefix(0).expect("valid integer synopsis").expect("kept");
13704 assert_eq!(integers.entries, vec![(Value::Integer(-2), 2), (Value::Integer(4), 2)]);
13705 assert_eq!(integers.omitted_max, 2);
13706 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13707 assert_eq!(strings.len(), 3);
13708 assert!(strings.contains(&(Value::Null, 2)));
13709 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13710 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13711 fs::remove_file(path).expect("remove scratch file");
13712 }
13713
13714 #[test]
13722 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13723 let path = path("interleaved-runs");
13724 let mut writer =
13725 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13726 .expect("new file");
13727 for morsel in [2_u64, 0, 3, 1] {
13728 let parts = (0..4_u64)
13729 .map(|chunk| {
13730 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13731 let values =
13732 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13733 let column =
13734 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13735 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13736 })
13737 .collect::<Vec<_>>();
13738 writer.append_stripe(parts).expect("a stripe");
13739 }
13740 writer.finish().expect("commit");
13741
13742 let reader = Reader::open(&path).expect("valid directory");
13743 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13744 assert_eq!(reader.table().rows(), 128);
13745 for part in 0..16_usize {
13746 let read = reader.read(part, &[0]).expect("a part back");
13747 for row in 0..8_usize {
13748 let want = i64::try_from(part * 8 + row).expect("small");
13749 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13750 }
13751 }
13752 fs::remove_file(path).expect("remove scratch file");
13753 }
13754
13755 #[test]
13758 fn runs_that_overlap_each_other_are_refused_at_commit() {
13759 let path = path("overlapping-runs");
13760 let mut writer =
13761 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13762 .expect("new file");
13763 let one = |order: (u64, u64)| {
13764 let column =
13765 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13766 (order, Chunk::new(vec![column]).expect("one column"))
13767 };
13768 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13771 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13772 let error = writer.finish().expect_err("the runs overlap");
13773 assert!(error.message().contains("source order"), "{error}");
13774 fs::remove_file(path).expect("remove scratch file");
13775 }
13776
13777 #[test]
13780 fn a_run_longer_than_a_stripe_is_refused() {
13781 let path = path("overlong-run");
13782 let mut writer =
13783 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13784 .expect("new file");
13785 let parts = (0..=STRIPE_PARTS)
13786 .map(|at| {
13787 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13788 .expect("a column");
13789 let chunk = Chunk::new(vec![column]).expect("one column");
13790 ((0, u64::try_from(at).expect("small")), chunk)
13791 })
13792 .collect::<Vec<_>>();
13793 let error = writer.append_stripe(parts).expect_err("one part too many");
13794 assert!(error.message().contains("more parts than it holds"), "{error}");
13795 fs::remove_file(path).expect("remove scratch file");
13796 }
13797
13798 #[test]
13804 fn parts_past_the_stripe_bound_start_a_new_stripe() {
13805 let path = path("stripe-bound");
13806 let mut writer = Writer::create(
13807 &path,
13808 "items",
13809 vec![
13810 Field::required("id", LogicalType::Integer),
13811 Field::new("text", LogicalType::Varchar),
13812 ],
13813 )
13814 .expect("new file");
13815 let parts = STRIPE_PARTS * 2 + 3;
13816 for part in 0..parts {
13817 let id = part as i32;
13818 let chunk = Chunk::new(vec![
13819 Vector::from_values(
13820 LogicalType::Integer,
13821 &[Value::Integer(id), Value::Integer(-id)],
13822 )
13823 .expect("integers"),
13824 Vector::from_values(
13825 LogicalType::Varchar,
13826 &[Value::Varchar(format!("value {part}")), Value::Null],
13827 )
13828 .expect("strings"),
13829 ])
13830 .expect("matching rows");
13831 writer.append(&chunk).expect("one part");
13832 }
13833 writer.finish().expect("commit");
13834
13835 let reader = Reader::open(&path).expect("reopen from disk");
13836 assert_eq!(reader.parts(), parts);
13837 assert_eq!(reader.table().rows(), parts * 2);
13838 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13839 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13840 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13841 assert_eq!(reader.table().stripes()[2].parts(), 3);
13842 for part in (0..parts).rev() {
13845 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13846 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13847 for chunk in [&dense, &sparse] {
13848 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13849 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13850 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13851 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13852 assert_eq!(chunk.value_at(1, 1), Value::Null);
13853 }
13854 }
13855 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13858 assert!(reader.skips(0, &above), "the first stripe stops at 63");
13859 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13860 fs::remove_file(path).expect("remove scratch file");
13861 }
13862
13863 fn scattered(n: i64) -> i64 {
13865 n.wrapping_mul(-7_046_029_254_386_353_131)
13866 }
13867
13868 #[test]
13874 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13875 let path = path("sieve-skip");
13876 let mut writer =
13877 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13878 .expect("new file");
13879 let parts = STRIPE_PARTS + 3;
13880 let per_part = 128;
13884 for part in 0..parts {
13885 let held: Vec<Value> = (0..per_part)
13886 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13887 .collect();
13888 let chunk =
13889 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13890 .expect("one column");
13891 writer.append(&chunk).expect("one part");
13892 }
13893 writer.finish().expect("commit");
13894
13895 let reader = Reader::open(&path).expect("reopen from disk");
13896 let probe = |value: i64| Probe {
13897 column: 0,
13898 op: Op::Equal,
13899 value: Bound::Int(i128::from(scattered(value))),
13900 };
13901 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13902 let tests = [probe(wanted)];
13903 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13904 let home = wanted as usize / per_part;
13905 assert!(kept.contains(&home), "the part holding {wanted} is read");
13906 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13910 }
13911 let absent = [probe((parts * per_part) as i64 + 1)];
13912 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13913 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13914 let tests = [probe(0)];
13917 assert!(
13918 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13919 "the bounds rule out no stripe at all"
13920 );
13921 fs::remove_file(path).expect("remove scratch file");
13922 }
13923
13924 #[test]
13930 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
13931 let path = path("part-range-skip");
13932 let mut writer =
13933 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13934 .expect("new file");
13935 let parts = STRIPE_PARTS + 3;
13936 let per_part = 128;
13937 for part in 0..parts {
13938 let held: Vec<Value> = (0..per_part)
13942 .map(|row| {
13943 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13944 })
13945 .collect();
13946 let chunk =
13947 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13948 .expect("one column");
13949 writer.append(&chunk).expect("one part");
13950 }
13951 writer.finish().expect("commit");
13952
13953 let reader = Reader::open(&path).expect("reopen from disk");
13954 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13955 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
13956 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
13957 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
13959 fs::remove_file(path).expect("remove scratch file");
13960 }
13961
13962 #[test]
13966 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
13967 let path = path("part-range-certain");
13968 let mut writer =
13969 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13970 .expect("new file");
13971 let parts = STRIPE_PARTS + 3;
13972 let per_part = 128;
13973 for part in 0..parts {
13974 let held: Vec<Value> = (0..per_part)
13975 .map(|row| {
13976 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13977 })
13978 .collect();
13979 let chunk =
13980 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13981 .expect("one column");
13982 writer.append(&chunk).expect("one part");
13983 }
13984 writer.finish().expect("commit");
13985
13986 let reader = Reader::open(&path).expect("reopen from disk");
13987 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13988 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
13989 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
13990 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
13993 fs::remove_file(path).expect("remove scratch file");
13994 }
13995
13996 #[test]
13999 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
14000 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
14001 let path = path("part-range-page");
14002 let mut writer =
14003 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
14004 .expect("new file");
14005 for part in 0..parts {
14006 let held: Vec<Value> = (0..128)
14007 .map(|row| {
14008 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
14009 })
14010 .collect();
14011 let chunk = Chunk::new(vec![
14012 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
14013 ])
14014 .expect("one column");
14015 writer.append(&chunk).expect("one part");
14016 }
14017 writer.finish().expect("commit");
14018 let reader = Reader::open(&path).expect("reopen from disk");
14019 let bytes = reader.layout().columns[0].part_ranges;
14020 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
14021 fs::remove_file(path).expect("remove scratch file");
14022 }
14023 }
14024
14025 #[test]
14028 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
14029 let long = vec![b'a'; PART_BOUND_BYTES * 2];
14030 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
14031 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
14032 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
14033 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
14034 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
14035 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
14036 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
14037 }
14038
14039 #[test]
14042 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
14043 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
14044 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
14045 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
14046 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
14047 }
14048
14049 #[test]
14061 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
14062 let parts = 4;
14063 let per_part = 1024;
14064 let rows = parts * per_part;
14065 let written = |name: &str, keys: &[i64]| {
14066 let path = path(name);
14067 let fields = vec![Field::required("key", LogicalType::BigInt)];
14068 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
14069 for part in 0..parts {
14070 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
14071 .iter()
14072 .map(|key| Value::BigInt(*key))
14073 .collect();
14074 let chunk = Chunk::new(vec![
14075 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
14076 ])
14077 .expect("one column");
14078 writer.append(&chunk).expect("one part");
14079 }
14080 writer.finish().expect("commit");
14081 path
14082 };
14083 let climbing = |step: &dyn Fn(usize) -> i64| {
14086 let mut key = 0;
14087 (0..rows)
14088 .map(|row| {
14089 key += step(row);
14090 key
14091 })
14092 .collect::<Vec<i64>>()
14093 };
14094 let ascending = climbing(&|row| (row % 3) as i64);
14095 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
14099 let near_path = written("stored-near", &ascending);
14100 let far_path = written("stored-far", &sparse);
14101
14102 let one = Reader::open(&near_path).expect("reopen from disk");
14103 let other = Reader::open(&far_path).expect("reopen from disk");
14104 let near = one.stored(0).expect("the column is stored");
14105 let far = other.stored(0).expect("the column is stored");
14106 assert_eq!(near.len(), parts, "one row per part");
14107 assert_eq!(far.len(), parts);
14108 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
14111 assert_eq!(total(&near), one.layout().columns[0].pages);
14112 assert_eq!(total(&far), other.layout().columns[0].pages);
14113 assert!(
14114 total(&near) * 2 < total(&far),
14115 "the sparse keys cost more, {} against {}",
14116 total(&far),
14117 total(&near)
14118 );
14119 for (at, part) in near.iter().enumerate() {
14121 assert_eq!(part.part, at);
14122 assert_eq!(part.row, at * per_part);
14123 assert_eq!(part.rows, per_part);
14124 let held = &ascending[at * per_part..(at + 1) * per_part];
14125 assert_eq!(part.low, Some(Value::BigInt(held[0])));
14126 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
14127 assert_eq!(part.nulls, Some(0));
14128 }
14129 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
14132 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
14133 assert_ne!(near[0].encoding, far[0].encoding);
14134 fs::remove_file(near_path).expect("remove scratch file");
14135 fs::remove_file(far_path).expect("remove scratch file");
14136 }
14137
14138 #[test]
14148 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
14149 let path = path("sieve-pays");
14150 let fields = vec![
14151 Field::required("spread", LogicalType::BigInt),
14152 Field::required("repeated", LogicalType::BigInt),
14153 ];
14154 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
14155 let parts = 3;
14156 let per_part = 1024;
14157 for part in 0..parts {
14158 let base = (part * per_part) as i64;
14159 let spread: Vec<Value> =
14160 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
14161 let repeated: Vec<Value> =
14162 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
14163 let chunk = Chunk::new(vec![
14164 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
14165 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
14166 ])
14167 .expect("two columns");
14168 writer.append(&chunk).expect("one part");
14169 }
14170 writer.finish().expect("commit");
14171
14172 let reader = Reader::open(&path).expect("reopen from disk");
14173 let layout = reader.layout();
14174 let spread = &layout.columns[0];
14175 let repeated = &layout.columns[1];
14176 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
14177 assert_eq!(
14178 repeated.sieves, 0,
14179 "a column whose filter costs more than its parts keeps none"
14180 );
14181 for column in &layout.columns {
14184 assert!(
14185 column.sieves < column.pages,
14186 "{} spends {} on sieves over {} of data",
14187 column.name,
14188 column.sieves,
14189 column.pages
14190 );
14191 }
14192 let absent = [Probe {
14194 column: 0,
14195 op: Op::Equal,
14196 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
14197 }];
14198 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
14199 fs::remove_file(path).expect("remove scratch file");
14200 }
14201
14202 #[test]
14208 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
14209 let path = path("sieve-damaged");
14210 let mut writer =
14211 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
14212 .expect("new file");
14213 let rows = 128;
14214 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
14215 let chunk =
14216 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
14217 .expect("one column");
14218 writer.append(&chunk).expect("one part");
14219 writer.finish().expect("commit");
14220
14221 let page = Reader::open(&path).expect("reopen").table.stripes[0]
14222 .sieves
14223 .get(0)
14224 .expect("a sieve page");
14225 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14226 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14227 file.write_all(&[0xff]).expect("damage one byte");
14228 drop(file);
14229
14230 let reader = Reader::open(&path).expect("reopen the damaged file");
14231 let absent =
14232 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14233 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14234 assert_eq!(
14235 reader.read(0, &[0]).expect("the rows are untouched").len(),
14236 usize::try_from(rows).expect("a small count")
14237 );
14238 fs::remove_file(path).expect("remove scratch file");
14239 }
14240
14241 #[test]
14252 fn workers_that_want_the_same_stripe_read_it_once() {
14253 let path = path("single-flight");
14254 let mut writer =
14255 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14256 .expect("new file");
14257 for part in 0..STRIPE_PARTS {
14258 let id = part as i32;
14259 let chunk = Chunk::new(vec![
14260 Vector::from_values(
14261 LogicalType::Integer,
14262 &[Value::Integer(id), Value::Integer(-id)],
14263 )
14264 .expect("integers"),
14265 ])
14266 .expect("matching rows");
14267 writer.append(&chunk).expect("one part");
14268 }
14269 writer.finish().expect("commit");
14270
14271 let reader = Reader::open(&path).expect("reopen from disk");
14272 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14273 let barrier = std::sync::Barrier::new(8);
14274 std::thread::scope(|scope| {
14275 for worker in 0..8 {
14276 let reader = &reader;
14277 let barrier = &barrier;
14278 scope.spawn(move || {
14279 barrier.wait();
14280 for part in (worker..STRIPE_PARTS).step_by(8) {
14281 let chunk = reader.read(part, &[0]).expect("a whole page read");
14282 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14283 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14284 }
14285 });
14286 }
14287 });
14288 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14289 fs::remove_file(path).expect("remove scratch file");
14290 }
14291
14292 #[test]
14305 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14306 let opened = |label: &str, rows_per_part: i32| {
14307 let path = path(label);
14308 let mut writer =
14309 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14310 .expect("new file");
14311 for part in 0..STRIPE_PARTS * 3 {
14312 let values = (0..rows_per_part)
14316 .map(|row| {
14317 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14318 })
14319 .collect::<Vec<_>>();
14320 let chunk = Chunk::new(vec![
14321 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14322 ])
14323 .expect("matching rows");
14324 writer.append(&chunk).expect("one part");
14325 }
14326 writer.finish().expect("commit");
14327 let reader = Reader::open(&path).expect("reopen from disk");
14328 let size = fs::metadata(&path).expect("the file is there").len();
14329 let out = (reader.reads(), reader.table().stripes().len(), size);
14330 fs::remove_file(path).expect("remove scratch file");
14331 out
14332 };
14333
14334 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14335 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14336 assert_eq!(
14337 thin_stripes, fat_stripes,
14338 "the same stripe count is what makes this a fair ask"
14339 );
14340 assert!(
14341 fat_size > thin_size * 50,
14342 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14343 );
14344
14345 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14346 assert_eq!(thin.pages, 0, "opening read a page");
14347 assert_eq!(fat.pages, 0, "opening read a page");
14348 assert_eq!(thin.indexes, 0, "opening read an index");
14349 assert_eq!(fat.indexes, 0, "opening read an index");
14350 assert!(
14353 fat.opening.bytes < thin.opening.bytes * 2,
14354 "opening the thin file read {} bytes and the fat one read {}",
14355 thin.opening.bytes,
14356 fat.opening.bytes
14357 );
14358 }
14359
14360 #[test]
14368 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14369 let path = path("open-twice");
14370 let mut writer =
14371 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14372 .expect("new file");
14373 for part in 0..STRIPE_PARTS * 3 {
14374 let chunk = Chunk::new(vec![
14375 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14376 .expect("integers"),
14377 ])
14378 .expect("matching rows");
14379 writer.append(&chunk).expect("one part");
14380 }
14381 writer.finish().expect("commit");
14382
14383 let first = Reader::open(&path).expect("open");
14384 for part in 0..first.parts() {
14387 first.read(part, &[0]).expect("a part");
14388 }
14389 assert!(first.reads().pages > 0, "the scan has to have read something");
14390 let second = Reader::open(&path).expect("open again");
14391
14392 assert_eq!(first.reads().opening, second.reads().opening);
14393 assert_eq!(
14394 second.reads().pages,
14395 0,
14396 "the second open read a page off the back of the first"
14397 );
14398 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14399 fs::remove_file(path).expect("remove scratch file");
14400 }
14401
14402 #[test]
14410 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14411 let path = path("index-cache");
14412 let mut writer =
14413 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14414 .expect("new file");
14415 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14416 for part in 0..parts {
14417 let id = part as i32;
14418 let chunk = Chunk::new(vec![
14419 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14420 ])
14421 .expect("matching rows");
14422 writer.append(&chunk).expect("one part");
14423 }
14424 writer.finish().expect("commit");
14425
14426 let reader = Reader::open(&path).expect("reopen from disk");
14427 let stripes = reader.table().stripes().len();
14428 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14429 for _ in 0..2 {
14431 for part in 0..parts {
14432 let chunk = reader.read(part, &[0]).expect("a part");
14433 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14434 }
14435 }
14436 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14437 assert!(
14438 reader.pages.load(Atomic::Relaxed) > stripes,
14439 "the pages are the ones that get read again, which is what makes the index count mean \
14440 something"
14441 );
14442 fs::remove_file(path).expect("remove scratch file");
14443 }
14444
14445 #[test]
14452 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14453 let path = path("page-pool");
14454 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN * 2 + 2);
14455 let fields = || vec![Field::required("id", LogicalType::Integer)];
14456 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14457 for table in ["a", "b"] {
14458 if table == "b" {
14459 writer = writer.next("b".to_string(), fields()).expect("a second table");
14460 }
14461 for part in 0..parts {
14462 let chunk = Chunk::new(vec![
14463 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14464 .expect("integers"),
14465 ])
14466 .expect("matching rows");
14467 writer.append(&chunk).expect("one part");
14468 }
14469 }
14470 writer.finish().expect("commit");
14471
14472 let pool = PagePool::new(usize::MAX);
14473 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14474 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14475 let stripes = a.table().stripes().len();
14476 assert!(
14477 stripes > CACHED_STRIPES_PER_COLUMN * 2,
14478 "the floor has to be smaller than a table"
14479 );
14480 let scan = |reader: &Reader| {
14481 for part in 0..parts {
14482 let chunk = reader.read(part, &[0]).expect("a part");
14483 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14484 }
14485 };
14486 scan(&a);
14489 assert_eq!(pool.bytes(), 0, "a page read once is not the pool's");
14490 scan(&a);
14491 let twice = stripes * 2 - CACHED_STRIPES_PER_COLUMN;
14492 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the second scan reads the rest again");
14493 scan(&a);
14494 assert_eq!(a.pages.load(Atomic::Relaxed), twice, "the third scan reads nothing");
14495 let one = pool.bytes();
14496 assert!(one > 0, "the pool counts what the reader holds");
14497
14498 pool.budget.store(one, Atomic::Relaxed);
14500 scan(&b);
14501 scan(&b);
14502 assert_eq!(b.pages.load(Atomic::Relaxed), twice, "a page is never let go while in use");
14503 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14504 let column = a.cache.columns[0].lock().expect("the column");
14505 let held = column.pages.iter().flatten().count();
14506 assert_eq!(
14507 held,
14508 CACHED_STRIPES_PER_COLUMN + column.passing.len(),
14509 "the count and the slots agree"
14510 );
14511 drop(column);
14512
14513 drop((a, b, catalog));
14515 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14516 scan(&c);
14517 scan(&c);
14518 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14519 fs::remove_file(path).expect("remove scratch file");
14520 }
14521
14522 #[test]
14531 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14532 let workers = CACHED_STRIPES_PER_COLUMN + 4;
14533 let path = path("stripe-per-worker");
14534 let mut writer =
14535 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14536 .expect("new file");
14537 for part in 0..STRIPE_PARTS * workers {
14538 let chunk = Chunk::new(vec![
14539 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14540 .expect("integers"),
14541 ])
14542 .expect("matching rows");
14543 writer.append(&chunk).expect("one part");
14544 }
14545 writer.finish().expect("commit");
14546
14547 let read = |told: bool| {
14548 let reader = Reader::open(&path).expect("reopen from disk");
14549 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14550 if told {
14551 reader.keep_stripes(workers);
14552 }
14553 let barrier = std::sync::Barrier::new(workers);
14554 std::thread::scope(|scope| {
14555 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14556 let reader = &reader;
14557 let barrier = &barrier;
14558 scope.spawn(move || {
14559 for part in run {
14560 barrier.wait();
14561 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14562 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14563 }
14564 assert!(worker < workers);
14565 });
14566 }
14567 });
14568 reader.pages.load(Atomic::Relaxed)
14569 };
14570
14571 assert_eq!(read(true), workers, "one page read per stripe and no more");
14572 assert!(read(false) > workers, "a cache that small is read again on every part");
14573 fs::remove_file(path).expect("remove scratch file");
14574 }
14575
14576 #[test]
14581 fn a_damaged_index_page_is_an_error() {
14582 let path = path("damaged-index");
14583 let mut writer =
14584 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14585 .expect("new file");
14586 writer.append(&sample_ids()).expect("first part");
14587 writer.append(&sample_ids()).expect("second part");
14588 writer.finish().expect("commit");
14589
14590 let reader = Reader::open(&path).expect("valid directory");
14591 let index = reader.table.stripes[0].index;
14592 let mut byte = [0; 1];
14593 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14594 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14595 file.seek(SeekFrom::Start(index.offset)).expect("index start");
14596 file.write_all(&[!byte[0]]).expect("damage the first part length");
14597 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14598 assert!(error.message().contains("index page section checksum differs"), "{error}");
14599 fs::remove_file(path).expect("remove scratch file");
14600 }
14601
14602 #[test]
14609 fn every_integer_width_round_trips_through_a_page() {
14610 let path = path("integer-widths");
14611 let columns = [
14612 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14613 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14614 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14615 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14616 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14617 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14618 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14619 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14620 ];
14621 let fields = columns
14622 .iter()
14623 .enumerate()
14624 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14625 .collect::<Vec<_>>();
14626 let vectors = columns
14627 .iter()
14628 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14629 .collect::<Vec<_>>();
14630 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14631 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14632 writer.finish().expect("commit");
14633
14634 let reader = Reader::open(&path).expect("reopen from disk");
14635 let wanted = (0..columns.len()).collect::<Vec<_>>();
14636 let read = reader.read(0, &wanted).expect("every column");
14637 assert_eq!(read.len(), 2);
14638 for (at, (ty, values)) in columns.iter().enumerate() {
14640 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14641 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14642 }
14643 fs::remove_file(path).expect("remove scratch file");
14644 }
14645
14646 #[test]
14657 fn every_other_type_the_format_knows_round_trips_through_a_page() {
14658 let path = path("other-types");
14659 let columns = [
14660 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14661 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14662 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14663 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14664 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14665 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14666 (
14667 LogicalType::TimestampTz,
14668 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14669 ),
14670 (
14671 LogicalType::Interval,
14672 vec![
14673 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14674 Value::Interval { months: 13, days: -1, micros: 1 },
14675 ],
14676 ),
14677 (
14678 LogicalType::Blob,
14679 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14680 ),
14681 ];
14682 let fields = columns
14683 .iter()
14684 .enumerate()
14685 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14686 .collect::<Vec<_>>();
14687 let vectors = columns
14688 .iter()
14689 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14690 .collect::<Vec<_>>();
14691 let mut writer = Writer::create(&path, "others", fields).expect("new file");
14692 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14693 writer.finish().expect("commit");
14694
14695 let reader = Reader::open(&path).expect("reopen from disk");
14696 let wanted = (0..columns.len()).collect::<Vec<_>>();
14697 let read = reader.read(0, &wanted).expect("every column");
14698 assert_eq!(read.len(), 2);
14699 for (at, (ty, values)) in columns.iter().enumerate() {
14700 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14701 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14702 }
14703 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14706 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14707
14708 fs::remove_file(path).expect("remove scratch file");
14709 }
14710
14711 #[test]
14717 fn a_nan_survives_being_written_down() {
14718 let path = path("nan");
14719 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14720 .expect("a NaN vector");
14721 let mut writer =
14722 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14723 .expect("new file");
14724 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14725 writer.finish().expect("commit");
14726 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14727 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14728 assert!(back.is_nan(), "a NaN came back as {back}");
14729 fs::remove_file(path).expect("remove scratch file");
14730 }
14731
14732 #[test]
14739 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14740 let path = path("uuid-and-bit");
14741 let uuids = vec![0_i128, i128::MIN, -1];
14742 let mut bits = StringColumn::new();
14743 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14744 bits.push_bytes(value);
14745 }
14746 let expected = bits.clone();
14747 let fields =
14748 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14749 let vectors = vec![
14750 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14751 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14752 ];
14753 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14754 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14755 writer.finish().expect("commit");
14756
14757 let reader = Reader::open(&path).expect("reopen from disk");
14758 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14759 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14760 panic!("a uuid column is the 128 bit lane")
14761 };
14762 assert_eq!(back.as_slice(), uuids.as_slice());
14763 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14764 panic!("a bit column is bytes")
14765 };
14766 for row in 0..expected.len() {
14767 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14768 }
14769 fs::remove_file(path).expect("remove scratch file");
14770 }
14771
14772 #[test]
14775 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14776 let mut rows: Vec<Option<u64>> = Vec::new();
14777 let mut state = 0x2545_f491_4f6c_dd1d_u64;
14778 for index in 0..400_000_u64 {
14779 state ^= state << 13;
14780 state ^= state >> 7;
14781 state ^= state << 17;
14782 let times = 1 + (state % 7) as usize;
14783 let bits = match state % 11 {
14784 0 => None,
14785 1..=3 => Some(state % 16),
14786 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14787 };
14788 rows.extend(std::iter::repeat_n(bits, times));
14789 }
14790 let mut by_row = Candidates::default();
14791 for &bits in &rows {
14792 by_row.add(bits, 1);
14793 }
14794 let mut by_run = Candidates::default();
14795 let mut run = Run::default();
14796 let mut runs = 0_usize;
14797 for &bits in &rows {
14798 if let Some((bits, times)) = run.push(bits) {
14799 by_run.add(bits, times);
14800 runs += 1;
14801 }
14802 }
14803 if let Some((bits, times)) = run.take() {
14804 by_run.add(bits, times);
14805 }
14806 assert!(runs < rows.len() / 2, "the rows came in runs");
14807 assert!(by_row.decrements > 0, "the table filled and turned values away");
14808 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14809 assert_eq!(by_run.nulls, by_row.nulls);
14810 assert_eq!(by_run.decrements, by_row.decrements);
14811 }
14812
14813 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14814 let mut pairs = candidates.pairs().collect::<Vec<_>>();
14815 pairs.sort_unstable();
14816 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14817 pairs
14818 }
14819
14820 #[derive(Default)]
14823 struct MapCandidates {
14824 counts: HashMap<u64, u32>,
14825 nulls: u32,
14826 decrements: u64,
14827 }
14828
14829 impl MapCandidates {
14830 fn add(&mut self, bits: Option<u64>, mut times: u32) {
14831 while times > 0 {
14832 let held = match bits {
14833 Some(bits) => self.counts.get_mut(&bits),
14834 None if self.nulls != 0 => Some(&mut self.nulls),
14835 None => None,
14836 };
14837 if let Some(count) = held {
14838 *count = count.saturating_add(times);
14839 return;
14840 }
14841 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14842 match bits {
14843 Some(bits) => {
14844 self.counts.insert(bits, times);
14845 }
14846 None => self.nulls = times,
14847 }
14848 return;
14849 }
14850 self.counts.retain(|_, count| {
14851 *count -= 1;
14852 *count != 0
14853 });
14854 self.nulls = self.nulls.saturating_sub(1);
14855 self.decrements = self.decrements.saturating_add(1);
14856 times -= 1;
14857 }
14858 }
14859 }
14860
14861 #[test]
14865 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14866 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14867 let mut table = Candidates::default();
14868 let mut oracle = MapCandidates::default();
14869 let mut state = seed;
14870 for index in 0..300_000_u64 {
14871 state ^= state << 13;
14872 state ^= state >> 7;
14873 state ^= state << 17;
14874 let bits = match state % 13 {
14875 0 => None,
14876 1..=4 => Some(state % 40),
14877 5 => Some((index % 1000) * 1_000_000),
14878 _ => Some(state),
14879 };
14880 let times = 1 + (state >> 60) as u32 % 3;
14881 table.add(bits, times);
14882 oracle.add(bits, times);
14883 if index % 50_000 == 0 {
14884 let mut expected =
14885 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14886 expected.sort_unstable();
14887 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14888 }
14889 }
14890 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14891 expected.sort_unstable();
14892 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14893 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14894 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14895 assert!(table.decrements > 0, "seed {seed} never filled the table");
14896 for &(bits, _) in &expected {
14897 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14898 }
14899 }
14900 }
14901
14902 #[test]
14903 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14904 let path = path("frequency-ordinals");
14905 let mut writer =
14906 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14907 .expect("new file");
14908 let mut values = Vec::new();
14909 for leader in 0..10_i64 {
14910 values.extend(std::iter::repeat_n(leader, 100));
14911 }
14912 values.extend(1_000_i64..41_000);
14913 for part in values.chunks(1_024) {
14914 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14915 .expect("big integers");
14916 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14917 }
14918 writer.finish().expect("commit");
14919
14920 let reader = Reader::open(&path).expect("reopen from disk");
14921 let occurrences =
14922 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14923 assert!(occurrences.omitted_max < 100);
14924 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14925 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14926 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14927 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14928 assert_eq!(
14929 &occurrences.anchor_indices[..1_000]
14930 .iter()
14931 .map(|&entry| occurrences.anchors[entry as usize].clone())
14932 .collect::<Vec<_>>(),
14933 &(0_i64..10)
14934 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
14935 .collect::<Vec<_>>()
14936 );
14937 fs::remove_file(path).expect("remove scratch file");
14938 }
14939
14940 #[test]
14941 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
14942 let path = path("frequency-bits");
14947 let mut writer = Writer::create(
14948 &path,
14949 "items",
14950 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
14951 )
14952 .expect("new file");
14953 let mut rows = Vec::new();
14954 let mut leaders = Vec::new();
14955 for leader in 0..10_u64 {
14956 let count = 300 - leader * 10;
14957 let (unsigned, signed) = if leader == 0 {
14958 (Value::Null, Value::Null)
14959 } else {
14960 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
14961 };
14962 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
14963 leaders.push(((unsigned, count), (signed, count)));
14964 }
14965 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
14966 for part in rows.chunks(1_024) {
14967 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
14968 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
14969 let chunk = Chunk::new(vec![
14970 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
14971 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
14972 ])
14973 .expect("matching columns");
14974 writer.append(&chunk).expect("rows");
14975 }
14976 writer.finish().expect("commit");
14977
14978 let reader = Reader::open(&path).expect("reopen from disk");
14979 for column in 0..2 {
14980 let prefix =
14981 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14982 let wanted = leaders
14983 .iter()
14984 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
14985 .cloned()
14986 .collect::<Vec<_>>();
14987 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
14988 assert!(prefix.omitted_max < 210, "column {column}");
14989 assert_eq!(
14990 reader.distinct_values(column).expect("valid metadata"),
14991 Some(9 + 40_000),
14992 "column {column}"
14993 );
14994 }
14995 fs::remove_file(path).expect("remove scratch file");
14996 }
14997
14998 #[test]
14999 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
15000 let path = path("frequency-tally");
15006 let types = [
15007 LogicalType::TinyInt,
15008 LogicalType::UInteger,
15009 LogicalType::Date,
15010 LogicalType::Timestamp,
15011 ];
15012 let value = |ty: &LogicalType, at: i64| match ty {
15013 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
15014 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
15015 LogicalType::Date => Value::Date(19_000 - at as i32),
15016 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
15017 };
15018 let fields = types
15019 .iter()
15020 .enumerate()
15021 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
15022 .collect::<Vec<_>>();
15023 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15024 let mut rows = Vec::new();
15025 for at in 0..250_i64 {
15026 for _ in 0..=(at % 37) {
15027 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
15028 }
15029 }
15030 for part in rows.chunks(1_000) {
15031 let columns = types
15032 .iter()
15033 .map(|ty| {
15034 let values = part
15035 .iter()
15036 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
15037 .collect::<Vec<_>>();
15038 Vector::from_values(ty.clone(), &values).expect("a column")
15039 })
15040 .collect();
15041 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
15042 }
15043 writer.finish().expect("commit");
15044
15045 let reader = Reader::open(&path).expect("reopen from disk");
15046 for (column, ty) in types.iter().enumerate() {
15047 let mut counts = HashMap::<Option<i64>, u64>::new();
15048 for row in &rows {
15049 *counts.entry(*row).or_default() += 1;
15050 }
15051 let wanted = counts
15052 .into_iter()
15053 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
15054 .collect::<Vec<_>>();
15055 let prefix =
15056 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
15057 assert_eq!(prefix.entries.len(), 2, "column {column}");
15058 assert!(prefix.omitted_max > 0, "column {column}");
15059 for (value, count) in &prefix.entries {
15060 let held =
15061 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
15062 assert_eq!(held, Some(count), "column {column} value {value:?}");
15063 }
15064 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
15065 assert_eq!(
15066 reader.distinct_values(column).expect("valid metadata"),
15067 Some(wanted.len() as u64 - 1),
15068 "column {column}"
15069 );
15070 }
15071 fs::remove_file(path).expect("remove scratch file");
15072 }
15073
15074 #[test]
15075 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
15076 let edge = FREQUENCY_CANDIDATES as i64;
15081 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
15082 for with_null in [false, true] {
15083 let path = path("distinct-edge");
15084 let mut writer =
15085 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
15086 .expect("new file");
15087 let mut values = Vec::new();
15088 for round in 0..2 {
15089 for value in 0..distinct {
15090 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
15091 values.extend(std::iter::repeat_n(
15092 Value::BigInt(value * 7_919 % distinct),
15093 repeat,
15094 ));
15095 if with_null && value % 1_000 == 0 {
15096 values.push(Value::Null);
15097 }
15098 }
15099 }
15100 if with_null {
15101 values.push(Value::Null);
15102 }
15103 for part in values.chunks(1_024) {
15104 let chunk = Chunk::new(vec![
15105 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
15106 ])
15107 .expect("one column");
15108 writer.append(&chunk).expect("rows");
15109 }
15110 writer.finish().expect("commit");
15111 let reader = Reader::open(&path).expect("reopen from disk");
15112 assert_eq!(
15113 reader.distinct_values(0).expect("valid metadata"),
15114 Some(distinct as u64),
15115 "{distinct} values, null {with_null}"
15116 );
15117 fs::remove_file(path).expect("remove scratch file");
15118 }
15119 }
15120 }
15121
15122 #[test]
15123 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
15124 let path = path("quick-nonzero");
15125 let mut writer = Writer::create(
15126 &path,
15127 "items",
15128 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
15129 )
15130 .expect("create");
15131 for ids in [
15132 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
15133 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
15134 ] {
15135 let labels = vec![Value::Varchar("same".into()); ids.len()];
15136 writer
15137 .append(
15138 &Chunk::new(vec![
15139 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
15140 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
15141 ])
15142 .expect("chunk"),
15143 )
15144 .expect("append");
15145 }
15146 writer.finish().expect("finish");
15147 let catalog = Catalog::open(&path).expect("catalog");
15148 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
15149 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
15150 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
15151 assert_eq!(catalog.exact_numeric_frequencies("items", 1).expect("frequencies"), None);
15152 let prefix = catalog
15153 .table("items")
15154 .expect("reader")
15155 .frequency_prefix(1)
15156 .expect("valid metadata")
15157 .expect("partial frequencies");
15158 assert_eq!(prefix.entries, vec![(Value::Null, 2), (Value::Integer(0), 2)]);
15159 assert_eq!(prefix.omitted_max, 1);
15160 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
15161 assert_eq!(
15162 catalog.integer_extremes("items", 1).expect("extremes"),
15163 Some(IntegerExtremes::Values { low: 0, high: 7 })
15164 );
15165 assert_eq!(
15166 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
15167 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15168 );
15169 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
15170 let mut legacy = catalog.clone();
15171 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
15172 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
15173 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
15174 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
15175 Writer::certify_counts(&path).expect("recertify");
15176 assert_eq!(
15177 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
15178 Some(2)
15179 );
15180 assert_eq!(
15181 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
15182 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
15183 );
15184 assert_eq!(
15185 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
15186 Some(3)
15187 );
15188 assert_eq!(
15189 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
15190 Some(IntegerExtremes::Values { low: 0, high: 7 })
15191 );
15192 assert_eq!(
15193 Catalog::open(&path)
15194 .expect("reopen")
15195 .exact_numeric_frequencies("items", 1)
15196 .expect("frequencies"),
15197 None
15198 );
15199 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
15200 fs::remove_file(path).expect("remove scratch file");
15201 }
15202
15203 #[test]
15204 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
15205 let path = path("pair-frequencies");
15206 let mut pairs = Vec::new();
15207 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
15208 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
15209 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
15210 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
15211 let mut writer = Writer::create(
15212 &path,
15213 "items",
15214 vec![
15215 Field::required("id", LogicalType::BigInt),
15216 Field::required("phrase", LogicalType::Varchar),
15217 ],
15218 )
15219 .expect("new file");
15220 for part in pairs.chunks(1_024) {
15221 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
15222 let phrases =
15223 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
15224 writer
15225 .append(
15226 &Chunk::new(vec![
15227 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
15228 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
15229 ])
15230 .expect("matching columns"),
15231 )
15232 .expect("rows");
15233 }
15234 writer.finish().expect("commit");
15235
15236 let reader = Reader::open(&path).expect("reopen from disk");
15237 assert!(
15238 reader.table.pair_frequencies.is_empty(),
15239 "no query-specific pair result is stored"
15240 );
15241 fs::remove_file(path).expect("remove scratch file");
15242 }
15243
15244 #[test]
15245 fn legacy_group_answers_are_ignored() {
15246 let path = path("legacy-group-answers");
15247 let mut writer = Writer::create(
15248 &path,
15249 "items",
15250 vec![
15251 Field::required("id", LogicalType::BigInt),
15252 Field::required("text", LogicalType::Varchar),
15253 ],
15254 )
15255 .expect("new file");
15256 writer
15257 .append(
15258 &Chunk::new(vec![
15259 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("id"),
15260 Vector::from_values(LogicalType::Varchar, &[Value::Varchar("x".into())])
15261 .expect("text"),
15262 ])
15263 .expect("row"),
15264 )
15265 .expect("append");
15266 writer.finish().expect("commit");
15267 let mut reader = Reader::open(&path).expect("reopen");
15268 let table = Arc::make_mut(&mut reader.table);
15269 table.pair_frequencies.push(PairFrequencySummary {
15270 first: 0,
15271 second: 1,
15272 entries: vec![PairFrequencyEntry { first_entry: 0, second: Some(0), count: 999 }],
15273 omitted_max: 0,
15274 });
15275 table.host_groups = Some(host::HostSummary {
15276 column: 1,
15277 omitted_max: 0,
15278 entries: vec![host::HostEntry {
15279 host: "fake.test".into(),
15280 count: 999,
15281 bytes_sum: 999,
15282 minimum: "x".into(),
15283 }],
15284 });
15285 assert_eq!(reader.top_pair_frequencies(0, 1, 1).expect("legacy pair"), None);
15286 assert_eq!(reader.host_groups(1, 1).expect("legacy host"), None);
15287 fs::remove_file(path).expect("remove scratch file");
15288 }
15289
15290 #[test]
15296 fn a_file_from_another_format_says_which_format_it_is() {
15297 let older = path("older-format");
15298 let mut writer =
15299 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15300 .expect("new file");
15301 let chunk = Chunk::new(vec![
15302 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15303 .expect("integers"),
15304 ])
15305 .expect("chunk");
15306 writer.append(&chunk).expect("page written");
15307 writer.finish().expect("commit");
15308
15309 let unreadable =
15313 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15314 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15315 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15316 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15317 drop(file);
15318 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15319 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15320 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15321
15322 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15323 file.seek(SeekFrom::Start(0)).expect("the magic is first");
15324 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15325 drop(file);
15326 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15327 assert!(complaint.contains("magic"), "{complaint}");
15328 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15329 fs::remove_file(older).expect("remove scratch file");
15330 }
15331
15332 #[test]
15333 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15334 let unfinished = path("unfinished");
15335 let mut writer =
15336 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15337 .expect("new file");
15338 let chunk = Chunk::new(vec![
15339 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15340 .expect("integers"),
15341 ])
15342 .expect("chunk");
15343 writer.append(&chunk).expect("page written");
15344 drop(writer);
15345 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15346 fs::remove_file(unfinished).expect("remove scratch file");
15347
15348 let damaged = path("damaged");
15349 let mut writer =
15350 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15351 .expect("new file");
15352 writer.append(&chunk).expect("page written");
15353 writer.finish().expect("commit");
15354 let reader = Reader::open(&damaged).expect("valid directory");
15355 let mut file =
15356 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15357 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15358 file.write_all(&[255]).expect("damage one byte");
15359 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15360 fs::remove_file(damaged).expect("remove scratch file");
15361 }
15362
15363 #[test]
15364 fn damaged_lazy_dictionary_payload_is_an_error() {
15365 let path = path("damaged-dictionary");
15366 let mut writer = Writer::create(
15367 &path,
15368 "items",
15369 vec![
15370 Field::required("id", LogicalType::Integer),
15371 Field::new("text", LogicalType::Varchar),
15372 ],
15373 )
15374 .expect("new file");
15375 writer.append(&sample()).expect("stripe written");
15376 writer.finish().expect("commit");
15377
15378 let reader = Reader::open(&path).expect("valid directory");
15379 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15380 let mut header = [0; DICTIONARY_HEADER];
15383 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15384 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15387 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15388 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15389 let bits = (width & !DICTIONARY_FLAGS) as usize;
15390 let mut start = [0; 8];
15391 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15392 read_at(&reader.file, at, &mut start).expect("the first block's start");
15393 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15394 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15395 file.write_all(&[255]).expect("damage dictionary payload");
15396
15397 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15398 let error =
15399 chunk.validate_external().expect_err("payload corruption must reach the caller");
15400 assert!(error.message().contains("payload checksum differs"), "{error}");
15401 fs::remove_file(path).expect("remove scratch file");
15402 }
15403
15404 #[test]
15414 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15415 let path = path("dictionary-decide");
15416 let rows = 20_000;
15417 let unique =
15419 |row: usize| format!("{row:09} a value that appears exactly once in the table");
15420 let repeated = |row: usize| unique(row / 40);
15422 let mut writer = Writer::create(
15423 &path,
15424 "items",
15425 vec![
15426 Field::required("unique", LogicalType::Varchar),
15427 Field::required("repeated", LogicalType::Varchar),
15428 ],
15429 )
15430 .expect("new file");
15431 for part in (0..rows).step_by(1_000) {
15432 let span = part..(part + 1_000).min(rows);
15433 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15434 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15435 writer
15436 .append(
15437 &Chunk::new(vec![
15438 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15439 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15440 ])
15441 .expect("two columns"),
15442 )
15443 .expect("a part");
15444 }
15445 writer.finish().expect("commit");
15446
15447 let reader = Reader::open(&path).expect("reopen from disk");
15448 assert!(
15449 reader.table.dictionaries[0].is_none(),
15450 "a column with no repeats has nothing to say twice"
15451 );
15452 assert!(
15453 reader.table.dictionaries[1].is_some(),
15454 "a column whose values come round again keeps its dictionary"
15455 );
15456 let mut first = 0;
15457 for part in 0..reader.parts() {
15458 let chunk = reader.read(part, &[0, 1]).expect("a part");
15459 for row in 0..chunk.len() {
15460 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15461 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15462 }
15463 first += chunk.len();
15464 }
15465 assert_eq!(first, rows, "every row was read back");
15466 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15467 let size = fs::metadata(&path).expect("the file is there").len() as usize;
15468 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15469 fs::remove_file(path).expect("remove scratch file");
15470 }
15471
15472 #[test]
15485 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15486 let path = path("dictionary-blocks");
15487 let value = |row: usize| {
15488 let row = row.saturating_sub(8_000);
15489 format!("{row:07} a value long enough to be worth a payload block")
15490 };
15491 let parts = 40;
15492 let per_part = 1000;
15493 let mut writer =
15494 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15495 .expect("new file");
15496 for part in 0..parts {
15497 let values = (0..per_part)
15498 .map(|row| Value::Varchar(value(part * per_part + row)))
15499 .collect::<Vec<_>>();
15500 let chunk = Chunk::new(vec![
15501 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15502 ])
15503 .expect("matching rows");
15504 writer.append(&chunk).expect("a part");
15505 }
15506 writer.finish().expect("commit");
15507
15508 let reader = Reader::open(&path).expect("reopen from disk");
15509 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15510 assert!(
15511 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15512 "the dictionary has to be several blocks for this to be testing anything"
15513 );
15514 for part in [0, parts - 1] {
15515 let chunk = reader.read(part, &[0]).expect("a part");
15516 chunk.validate_external().expect("every payload block checks out");
15517 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15518 }
15519
15520 let mut header = [0; DICTIONARY_HEADER];
15522 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15523 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15524 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15525 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15526 let bits = (width & !DICTIONARY_FLAGS) as usize;
15527 let mut place = [0; 16];
15528 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15529 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15530 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15531 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15532 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15533 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15534 file.write_all(&[255]).expect("damage the last payload block");
15535 let reader = Reader::open(&path).expect("the directory and the index are untouched");
15536 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15537 let error = chunk.validate_external().expect_err("the damage must reach the caller");
15538 assert!(error.message().contains("payload checksum differs"), "{error}");
15539 fs::remove_file(path).expect("remove scratch file");
15540 }
15541
15542 #[test]
15556 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15557 let path = path("dictionary-offsets");
15558 let value = |row: usize| {
15559 let row = row % 5_000;
15560 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15561 };
15562 let rows = 6_000;
15563 let mut writer =
15564 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15565 .expect("new file");
15566 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15567 for part in values.chunks(1_000) {
15568 let chunk =
15569 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15570 .expect("matching rows");
15571 writer.append(&chunk).expect("a part");
15572 }
15573 writer.finish().expect("commit");
15574
15575 let reader = Reader::open(&path).expect("reopen from disk");
15576 assert!(
15577 rows > TEXT_PAYLOAD_VALUES * 4,
15578 "the dictionary has to be several blocks for this to be testing anything"
15579 );
15580 for part in 0..rows / 1_000 {
15581 let chunk = reader.read(part, &[0]).expect("a part");
15582 for row in 0..1_000 {
15583 let row = part * 1_000 + row;
15584 assert_eq!(
15585 chunk.value_at(row % 1_000, 0),
15586 Value::Varchar(value(row)),
15587 "value {row}"
15588 );
15589 }
15590 }
15591 for _ in 0..2 {
15594 for part in 0..rows / 1_000 {
15595 let chunk = reader.read(part, &[0]).expect("a part");
15596 let mut lens = vec![0_i64; 1_000];
15597 let column = chunk.column(0).expect("one column");
15598 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15599 for (row, &len) in lens.iter().enumerate() {
15600 let row = part * 1_000 + row;
15601 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15602 }
15603 }
15604 }
15605 fs::remove_file(path).expect("remove scratch file");
15606 }
15607
15608 #[test]
15610 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15611 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15612 ends.extend([3, 3, 10]);
15613 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15614 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15615 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15616 let long = [5, 70_005, 70_006];
15618 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15619 assert_eq!(lens, [5, 70_000, 1]);
15620 let mut read = Vec::new();
15621 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15622 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15623 ends.push(9);
15624 assert!(lengths_of(&ends).is_none());
15625 }
15626
15627 #[test]
15639 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15640 let path = path("dictionary-once");
15641 let parts = 8;
15642 let per_part = 500;
15643 let value =
15644 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15645 let mut writer =
15646 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15647 .expect("new file");
15648 for part in 0..parts {
15649 let values = (0..per_part)
15650 .map(|row| Value::Varchar(value(part * per_part + row)))
15651 .collect::<Vec<_>>();
15652 let chunk = Chunk::new(vec![
15653 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15654 ])
15655 .expect("matching rows");
15656 writer.append(&chunk).expect("a part");
15657 }
15658 writer.finish().expect("commit");
15659
15660 let reader = Reader::open(&path).expect("reopen from disk");
15661 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15662 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15663
15664 let workers = 16;
15665 let gate = std::sync::Barrier::new(workers);
15666 std::thread::scope(|scope| {
15667 for worker in 0..workers {
15668 let reader = reader.clone();
15669 let gate = &gate;
15670 scope.spawn(move || {
15671 gate.wait();
15672 let chunk = reader.read(worker % parts, &[0]).expect("a part");
15673 assert_eq!(
15674 chunk.value_at(0, 0),
15675 Value::Varchar(value((worker % parts) * per_part))
15676 );
15677 });
15678 }
15679 });
15680
15681 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15682 fs::remove_file(path).expect("remove scratch file");
15683 }
15684
15685 #[test]
15690 fn a_damaged_sorted_order_is_an_error() {
15691 let path = path("damaged-order");
15692 let mut writer = Writer::create(
15693 &path,
15694 "items",
15695 vec![
15696 Field::required("id", LogicalType::Integer),
15697 Field::new("text", LogicalType::Varchar),
15698 ],
15699 )
15700 .expect("new file");
15701 writer.append(&sample()).expect("stripe written");
15702 writer.finish().expect("commit");
15703
15704 let reader = Reader::open(&path).expect("valid directory");
15705 let page = reader.table.dictionaries[1].expect("string dictionary page");
15706 let mut header = [0; DICTIONARY_HEADER];
15707 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15708 let index_len = dictionary_index_len(&header);
15709 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15710 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15711 file.write_all(&[255]).expect("damage the order");
15712
15713 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15714 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15715 assert!(error.message().contains("rank checksum differs"), "{error}");
15716 fs::remove_file(path).expect("remove scratch file");
15717 }
15718
15719 #[test]
15723 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15724 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15727 let path = path("dictionary-order");
15728 let mut writer =
15729 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15730 .expect("new file");
15731 writer
15732 .append(
15733 &Chunk::new(vec![
15734 Vector::from_values(
15735 LogicalType::Varchar,
15736 &spellings.map(|text| Value::Varchar(text.into())),
15737 )
15738 .expect("strings"),
15739 ])
15740 .expect("one column"),
15741 )
15742 .expect("stripe written");
15743 writer.finish().expect("commit");
15744
15745 let reader = Reader::open(&path).expect("valid directory");
15746 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15747 let count = dictionary.ranks().expect("a v10 file stores one");
15748 assert_eq!(count, spellings.len(), "every distinct value has a rank");
15749 let order = (0..count)
15750 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15751 .collect::<Vec<_>>();
15752 let mut seen = order.clone();
15753 seen.sort_unstable();
15754 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15755
15756 let ranked = order
15757 .iter()
15758 .map(|&code| {
15759 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15760 })
15761 .collect::<Vec<_>>();
15762 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15763 expected.sort();
15764 assert_eq!(ranked, expected, "rank order is value order");
15765
15766 for (rank, value) in expected.iter().enumerate() {
15769 assert_eq!(
15770 dictionary.compare_rank(rank, value).expect("compare"),
15771 Ordering::Equal,
15772 "rank {rank} is its own value"
15773 );
15774 if rank > 0 {
15775 assert_eq!(
15776 dictionary.compare_rank(rank - 1, value).expect("compare"),
15777 Ordering::Less,
15778 "rank {rank} follows the one before it"
15779 );
15780 }
15781 }
15782 fs::remove_file(path).expect("remove scratch file");
15783 }
15784
15785 #[test]
15792 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15793 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15794 let path = path("dictionaries-at-once");
15795 let fields = (0..sizes.len())
15796 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15797 .collect::<Vec<_>>();
15798 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15799 let rows = 10_000_usize;
15800 for start in (0..rows).step_by(1_024) {
15801 let columns = sizes
15802 .iter()
15803 .enumerate()
15804 .map(|(column, &size)| {
15805 let values = (start..(start + 1_024).min(rows))
15806 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15807 .collect::<Vec<_>>();
15808 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15809 })
15810 .collect::<Vec<_>>();
15811 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15812 }
15813 writer.finish().expect("commit");
15814
15815 let reader = Reader::open(&path).expect("valid directory");
15816 for (column, &size) in sizes.iter().enumerate() {
15817 let dictionary =
15818 reader.dictionary(column).expect("read").expect("a string column has one");
15819 let count = dictionary.ranks().expect("a v10 file stores one");
15820 assert_eq!(count, size, "column {column} has its own distinct count");
15821 let ranked = (0..count)
15822 .map(|rank| {
15823 let code = dictionary.code_at_rank(rank).expect("a code");
15824 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15825 })
15826 .collect::<Vec<_>>();
15827 let expected = (0..size)
15828 .map(|value| format!("c{column}-{value:05}").into_bytes())
15829 .collect::<Vec<_>>();
15830 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15831 }
15832 fs::remove_file(path).expect("remove scratch file");
15833 }
15834
15835 #[test]
15843 fn a_large_dictionary_ranks_in_value_order() {
15844 let path = path("dictionary-large-rank");
15845 let value = |row: u64| {
15846 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15847 match row % 3 {
15848 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15849 1 => format!("{mixed}"),
15850 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15851 }
15852 };
15853 let distinct = 70_000;
15854 let parts = 4 * distinct / 1000;
15855 let per_part = 1000;
15856 let mut writer =
15857 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15858 .expect("new file");
15859 for part in 0..parts {
15860 let values = (0..per_part)
15861 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15862 .collect::<Vec<_>>();
15863 let chunk = Chunk::new(vec![
15864 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15865 ])
15866 .expect("matching rows");
15867 writer.append(&chunk).expect("a part");
15868 }
15869 writer.finish().expect("commit");
15870
15871 let reader = Reader::open(&path).expect("reopen from disk");
15872 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15873 let count = dictionary.ranks().expect("a ranked dictionary");
15874 assert_eq!(count, distinct as usize, "every distinct value has a rank");
15875 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15876 let ranked = (0..count)
15877 .map(|rank| {
15878 let code = dictionary.code_at_rank(rank).expect("a code");
15879 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15880 })
15881 .collect::<Vec<_>>();
15882 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15883 expected.sort();
15884 assert_eq!(ranked, expected, "rank order is value order");
15885 fs::remove_file(path).expect("remove scratch file");
15886 }
15887
15888 #[test]
15901 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15902 let path = path("windowed-directory");
15903 let fields = vec![
15904 Field::required("id", LogicalType::BigInt),
15905 Field::required("word", LogicalType::Varchar),
15906 Field::new("score", LogicalType::Double),
15907 ];
15908 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15909 for part in 0..70_i64 {
15910 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15911 let words = (0..100)
15912 .map(|row| Value::Varchar(format!("word {}", row % 13)))
15913 .collect::<Vec<_>>();
15914 let scores = (0..100)
15915 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15916 .collect::<Vec<_>>();
15917 let chunk = Chunk::new(vec![
15918 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15919 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15920 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15921 ])
15922 .expect("three columns");
15923 writer.append(&chunk).expect("a part");
15924 }
15925 writer.finish().expect("commit");
15926
15927 let catalog = Catalog::open(&path).expect("reopen");
15928 let entry = catalog.entries.first().expect("one table").directory;
15929 let (offset, length) = (entry.offset, entry.length as usize);
15930 let mut bytes = vec![0; length];
15931 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
15932 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
15933 let whole = decode_directory(&bytes, catalog.size).expect("whole");
15934 assert!(whole.stripes.len() > 1, "the table should span stripes");
15935 for size in [1, 7, 33, 4_096] {
15936 let mut cursor = Cursor::over(&catalog.file, offset, length);
15937 cursor.window.as_mut().expect("a window").size = size;
15938 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
15939 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
15940 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
15941 let mut stored = 0;
15942 for (column, (left, held)) in
15943 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
15944 {
15945 match (left, held) {
15946 (None, None) => {}
15947 (
15948 Some(super::Frequencies::Stored { span, values }),
15949 Some(super::Frequencies::Held(summary)),
15950 ) => {
15951 let mut one = vec![0; span.length as usize];
15952 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
15953 let read = decode_summary(
15954 &mut Cursor::new(&one),
15955 &whole.fields[column],
15956 whole.rows,
15957 *values,
15958 )
15959 .expect("a valid synopsis")
15960 .expect("one is there");
15961 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
15962 stored += 1;
15963 }
15964 other => panic!("column {column} came back as {other:?}"),
15965 }
15966 }
15967 assert!(stored >= 2, "only {stored} synopses were left in the file");
15968 }
15969 let reader = catalog.table("items").expect("the table");
15970 assert!(reader.frequency_summaries[1].get().is_none());
15971 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
15972 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
15973 let clone = reader.clone();
15974 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
15975 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
15976 fs::remove_file(path).expect("remove scratch file");
15977 }
15978
15979 #[test]
15980 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
15981 let path = path("file-checksum");
15982 let bytes = (0..200_000_u32)
15983 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
15984 .collect::<Vec<_>>();
15985 fs::write(&path, &bytes).expect("scratch file");
15986 let file = File::open(&path).expect("open");
15987 for (offset, length) in [
15988 (0, 0),
15989 (3, 1),
15990 (5, 31),
15991 (0, 32),
15992 (9, 33),
15993 (1, 65_536),
15994 (7, 65_567),
15995 (0, 200_000),
15996 (11, 131_101),
15997 ] {
15998 let whole = checksum(&bytes[offset..offset + length]);
15999 assert_eq!(
16000 file_checksum(&file, offset as u64, length).expect("read"),
16001 whole,
16002 "{offset} {length}"
16003 );
16004 }
16005 fs::remove_file(path).expect("remove scratch file");
16006 }
16007
16008 #[test]
16009 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
16010 let path = path("synopsis-keeps-no-block");
16011 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
16012 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
16013 for _ in 0..3 {
16014 values.extend((0..3_000).step_by(5).map(spelled));
16015 }
16016 let mut writer =
16017 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16018 .expect("new file");
16019 for part in values.chunks(1_024) {
16020 writer
16021 .append(
16022 &Chunk::new(vec![
16023 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16024 ])
16025 .expect("one column"),
16026 )
16027 .expect("a part");
16028 }
16029 writer.finish().expect("commit");
16030
16031 let reader = Reader::open(&path).expect("reopen from disk");
16032 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16033 let resting = dictionary.footprint();
16034 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16035 assert_eq!(prefix.entries.len(), 512);
16036 for (value, count) in &prefix.entries {
16037 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
16038 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
16039 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
16040 }
16041 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
16042 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
16043 assert_eq!(again.entries, prefix.entries);
16044 fs::remove_file(path).expect("remove scratch file");
16045 }
16046
16047 #[test]
16054 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
16055 let path = path("character-lengths");
16056 let spellings = (0..2_500)
16057 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
16058 .collect::<Vec<_>>();
16059 let mut writer =
16060 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
16061 .expect("new file");
16062 for part in spellings.chunks(1_024) {
16063 writer
16064 .append(
16065 &Chunk::new(vec![
16066 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16067 ])
16068 .expect("one column"),
16069 )
16070 .expect("a part");
16071 }
16072 writer.finish().expect("commit");
16073
16074 let reader = Reader::open(&path).expect("reopen from disk");
16075 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16076 let resting = dictionary.footprint();
16077 let mut lens = Vec::new();
16078 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
16079 let counted = dictionary.footprint() - resting;
16080 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16081 assert!(
16082 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16083 "counting kept {counted} bytes, more than a count a value"
16084 );
16085 let expected = (0..dictionary.len())
16086 .map(|code| {
16087 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
16088 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
16089 .expect("small")
16090 })
16091 .collect::<Vec<_>>();
16092 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
16093 let mut again = Vec::new();
16094 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
16095 assert_eq!(again, lens, "the kept counts answer the second time");
16096 fs::remove_file(path).expect("remove scratch file");
16097 }
16098
16099 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
16101 let path = path(label);
16102 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
16103 let mut writer =
16104 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16105 .expect("new file");
16106 for part in values.chunks(1_024) {
16107 writer
16108 .append(
16109 &Chunk::new(vec![
16110 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16111 ])
16112 .expect("one column"),
16113 )
16114 .expect("a part");
16115 }
16116 writer.finish().expect("commit");
16117 let reader = Reader::open(&path).expect("reopen from disk");
16118 (path, reader)
16119 }
16120
16121 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
16127 let codes = (0..len)
16128 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
16129 .collect::<Vec<_>>();
16130 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
16131 (codes, valid)
16132 }
16133
16134 #[test]
16141 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
16142 let spellings = (0..2_500)
16143 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
16144 .collect::<Vec<_>>();
16145 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
16146 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16147 let (codes, valid) = scattered_rows(spellings.len());
16148 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
16149 .expect("every code is inside")
16150 .with_validity(Validity::from_run(&valid));
16151
16152 let resting = dictionary.footprint();
16153 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
16154 .expect("length reads");
16155 let counted = dictionary.footprint() - resting;
16156 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
16157 assert!(
16158 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
16159 "length over a vector with nulls kept {counted} bytes, more than a count a value"
16160 );
16161 let expected = (0..rows.len())
16162 .map(|row| match valid[row] {
16163 true => Value::BigInt(
16164 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
16165 ),
16166 false => Value::Null,
16167 })
16168 .collect::<Vec<_>>();
16169 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
16170 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
16171 fs::remove_file(path).expect("remove scratch file");
16172 }
16173
16174 #[test]
16184 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
16185 let spellings = (0..2_500)
16186 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
16187 .collect::<Vec<_>>();
16188 let (path, reader) = stored_spellings("string-kernels", &spellings);
16189 let page = reader.table.dictionaries[0].expect("a string column has one");
16190 let starved =
16191 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
16192 .expect("a dictionary opens whatever it may keep");
16193 let starved = Arc::new(starved);
16194 let (codes, valid) = scattered_rows(spellings.len());
16195 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
16196 .expect("every code is inside")
16197 .with_validity(Validity::from_run(&valid));
16198 let expected = |each: &dyn Fn(&str) -> String| {
16199 (0..rows.len())
16200 .map(|row| match valid[row] {
16201 true => Value::Varchar(each(&spellings[codes[row] as usize])),
16202 false => Value::Null,
16203 })
16204 .collect::<Vec<_>>()
16205 };
16206 let answers =
16207 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
16208
16209 let resting = starved.footprint();
16212 let ends = spellings.len() * size_of::<u32>();
16213 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
16214 .expect("lower reads");
16215 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
16216 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
16217
16218 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
16219 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
16220 let cut =
16221 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
16222 .expect("substring reads");
16223 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
16224 assert_eq!(answers(&cut), expected(&cut_of), "substring");
16225 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
16226
16227 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16230 .expect("upper reads");
16231 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
16232 let payload = spellings.iter().map(String::len).sum::<usize>();
16233 assert!(
16234 starved.footprint() >= resting + payload,
16235 "a visit that has dropped a column's worth of blocks keeps what it reads"
16236 );
16237 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
16238 .expect("upper reads kept blocks");
16239 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
16240 fs::remove_file(path).expect("remove scratch file");
16241 }
16242
16243 #[test]
16253 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
16254 let path = path("dictionary-sweep");
16255 let spellings = (0..2_500)
16258 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16259 .collect::<Vec<_>>();
16260 let mut writer =
16261 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16262 .expect("new file");
16263 for part in spellings.chunks(1_024) {
16266 writer
16267 .append(
16268 &Chunk::new(vec![
16269 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16270 ])
16271 .expect("one column"),
16272 )
16273 .expect("stripe written");
16274 }
16275 writer.finish().expect("commit");
16276
16277 let reader = Reader::open(&path).expect("valid directory");
16278 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16279 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16280 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
16281 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
16282 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
16283 }
16284
16285 let resting = dictionary.footprint();
16286 let sweep = || {
16287 let mut swept: Vec<Vec<u8>> = Vec::new();
16288 let mut at = 0;
16289 let mut calls = 0;
16290 while at < dictionary.len() {
16291 let stopped = dictionary
16292 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16293 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16294 swept.push(text.to_vec());
16295 Ok(())
16296 })
16297 .expect("a sweep reads");
16298 assert!(stopped > at, "a sweep moves");
16299 at = stopped;
16300 calls += 1;
16301 }
16302 assert_eq!(calls, 3, "a sweep hands over one block at a time");
16303 swept
16304 };
16305 let swept = sweep();
16306 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16307 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16308 let after = dictionary.footprint();
16309 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16310
16311 let read = (0..dictionary.len())
16312 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16313 .collect::<Vec<_>>();
16314 assert_eq!(swept, read, "a sweep answers what a point read answers");
16315 let grown = dictionary.footprint() - after;
16319 assert!(
16320 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16321 "a point read of a kept block decodes nothing, and {grown} bytes grew"
16322 );
16323 fs::remove_file(path).expect("remove scratch file");
16324 }
16325
16326 #[test]
16327 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16328 let path = path("narrow-substring-signature");
16329 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16330 let mut grams = Vec::new();
16331 for text in blocks {
16332 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16333 for gram in text.windows(4) {
16334 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16335 bits[bit / 8] |= 1 << (bit % 8);
16336 }
16337 }
16338 grams.extend(bits);
16339 }
16340 fs::write(&path, &grams).expect("scratch file");
16341 let file = File::open(&path).expect("open scratch file");
16342 let signatures = NativeGrams {
16343 start: 0,
16344 length: grams.len(),
16345 width: NARROW_GRAM_BYTES,
16346 hash: checksum(&grams),
16347 verdicts: Mutex::new(Vec::new()),
16348 };
16349 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16350 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16351 assert!(signatures.footprint() > 0, "a verdict is remembered");
16352 let again = signatures.verdicts(&file, b"google").expect("remembered");
16353 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16354
16355 let damaged = NativeGrams {
16356 hash: signatures.hash ^ 1,
16357 verdicts: Mutex::new(Vec::new()),
16358 ..signatures
16359 };
16360 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16361 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16362 fs::remove_file(path).expect("remove scratch file");
16363 }
16364
16365 #[test]
16366 fn a_damaged_substring_signature_is_checked_only_when_used() {
16367 let path = path("damaged-substring-signature");
16368 let mut writer =
16369 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16370 .expect("new file");
16371 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16372 writer
16373 .append(
16374 &Chunk::new(vec![
16375 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16376 ])
16377 .expect("one column"),
16378 )
16379 .expect("stripe written");
16380 writer.finish().expect("commit");
16381
16382 let reader = Reader::open(&path).expect("valid directory");
16383 let page = reader.table.dictionaries[0].expect("string dictionary page");
16384 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16385 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16386 .expect("last signature byte");
16387 file.write_all(&[255]).expect("damage signature");
16388 let reader = Reader::open(&path).expect("the directory is still valid");
16389 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16390 let error = dictionary
16391 .text_block_might_contain(0, b"goog")
16392 .expect_err("a used signature checks its own checksum");
16393 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16394 fs::remove_file(path).expect("remove scratch file");
16395 }
16396
16397 #[test]
16408 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16409 let path = path("dictionary-sweep-short-run");
16410 let spellings = (0..2_800)
16411 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16412 .collect::<Vec<_>>();
16413 let mut writer =
16414 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16415 .expect("new file");
16416 for part in spellings.chunks(1_024) {
16417 writer
16418 .append(
16419 &Chunk::new(vec![
16420 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16421 ])
16422 .expect("one column"),
16423 )
16424 .expect("stripe written");
16425 }
16426 writer.finish().expect("commit");
16427
16428 let reader = Reader::open(&path).expect("valid directory");
16429 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16430 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16431 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16432 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16433 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16434
16435 let mut swept: Vec<Vec<u8>> = Vec::new();
16436 let mut at = 0;
16437 while at < dictionary.len() {
16438 let stopped = dictionary
16439 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16440 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16441 swept.push(text.to_vec());
16442 Ok(())
16443 })
16444 .expect("a sweep reads");
16445 assert!(stopped > at, "a sweep moves");
16446 at = stopped;
16447 }
16448 let read = (0..dictionary.len())
16449 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16450 .collect::<Vec<_>>();
16451 assert_eq!(swept, read, "a sweep answers what a point read answers");
16452 fs::remove_file(path).expect("remove scratch file");
16453 }
16454
16455 #[test]
16464 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16465 let path = path("dictionary-unpacked-ends");
16466 let spellings = (0..2_800)
16467 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16468 .collect::<Vec<_>>();
16469 let mut writer =
16470 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16471 .expect("new file");
16472 for part in spellings.chunks(1_024) {
16473 writer
16474 .append(
16475 &Chunk::new(vec![
16476 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16477 ])
16478 .expect("one column"),
16479 )
16480 .expect("stripe written");
16481 }
16482 writer.finish().expect("commit");
16483
16484 let reader = Reader::open(&path).expect("valid directory");
16485 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16486 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16487 let wanted = (0..spellings.len())
16488 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16489 .collect::<Vec<_>>();
16490
16491 let pass = |what: &str| {
16492 for (index, value) in wanted.iter().enumerate() {
16493 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16494 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16495 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16496 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16497 }
16498 };
16499 pass("the first pass");
16500 pass("the second pass");
16501
16502 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16506 let mut whole = vec![0i64; wanted.len()];
16507 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16508 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16509 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16510 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16511 let mut through = vec![0i64; codes.len()];
16512 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16513 for (row, &code) in codes.iter().enumerate() {
16514 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16515 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16516 assert_eq!(through[row], one as i64, "row {row} a row at a time");
16517 }
16518
16519 let fresh = Reader::open(&path).expect("valid directory");
16522 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16523 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16524 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16525 let mut short = vec![0i64; few.len()];
16526 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16527 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16528 assert_eq!(short, expected, "the packed ends answer what the table answers");
16529 fs::remove_file(path).expect("remove scratch file");
16530 }
16531
16532 #[test]
16547 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16548 let spellings = (0..3_000)
16549 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16550 .collect::<Vec<_>>();
16551 let mut read = Vec::new();
16552 for layout in ["outside", "inside", "behind"] {
16553 let mut dictionary = GlobalDictionary::new();
16554 for text in &spellings {
16555 dictionary.code(text).expect("a code for every spelling");
16556 }
16557 dictionary.finish_blocks().expect("the last block encodes");
16558 let order = dictionary.ranked(None).expect("a sorted order");
16559 let laid = |from: u64| {
16561 let mut at = from;
16562 dictionary
16563 .blocks
16564 .iter()
16565 .map(|block| {
16566 let place =
16567 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16568 at += block.len() as u64;
16569 place
16570 })
16571 .collect::<Vec<_>>()
16572 };
16573 let payload = dictionary.blocks.concat();
16574 let scattered = layout != "behind";
16575 let (bytes, encoded, offset, length) = if layout == "outside" {
16576 let mut bytes = vec![0; HEADER as usize];
16577 bytes.extend_from_slice(&payload);
16578 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16579 .expect("an encoding");
16580 let offset = bytes.len() as u64;
16581 bytes.extend_from_slice(&encoded.index);
16582 bytes.extend_from_slice(&encoded.ranks);
16583 bytes.extend_from_slice(&encoded.grams);
16584 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16585 (bytes, encoded, offset, length)
16586 } else {
16587 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16590 .expect("an encoding");
16591 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16592 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16593 .expect("an encoding");
16594 let mut bytes = encoded.index.clone();
16595 bytes.extend_from_slice(&encoded.ranks);
16596 bytes.extend_from_slice(&encoded.grams);
16597 bytes.extend_from_slice(&payload);
16598 let length = bytes.len();
16599 (bytes, encoded, 0, length)
16600 };
16601 let path = path(&format!("blocks-{layout}"));
16602 fs::write(&path, &bytes).expect("the dictionary is written on its own");
16603 let file = Arc::new(File::open(&path).expect("it opens again"));
16604 let page = Page {
16605 offset,
16606 length: u32::try_from(length).expect("a test dictionary is small"),
16607 hash: checksum(&encoded.index),
16608 };
16609 let opened =
16610 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16611 .expect("a dictionary laid out either way opens");
16612 let mut swept: Vec<Vec<u8>> = Vec::new();
16613 let mut at = 0;
16614 while at < opened.len() {
16615 at = opened
16616 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16617 swept.push(text.to_vec());
16618 Ok(())
16619 })
16620 .expect("a sweep reads");
16621 }
16622 fs::remove_file(&path).expect("clean up");
16623 read.push(swept);
16624 }
16625 let wanted =
16626 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16627 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16628 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16629 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16630 }
16631
16632 #[test]
16640 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16641 let path = path("dictionary-budget");
16642 let spellings = (0..2_500)
16643 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16644 .collect::<Vec<_>>();
16645 let mut writer =
16646 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16647 .expect("new file");
16648 for part in spellings.chunks(1_024) {
16649 writer
16650 .append(
16651 &Chunk::new(vec![
16652 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16653 ])
16654 .expect("one column"),
16655 )
16656 .expect("stripe written");
16657 }
16658 writer.finish().expect("commit");
16659
16660 let reader = Reader::open(&path).expect("valid directory");
16661 let page = reader.table.dictionaries[0].expect("a string column has one");
16662 let file = Arc::clone(&reader.file);
16663 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16664 .expect("a dictionary opens whatever it may keep");
16665
16666 let resting = starved.footprint();
16667 let mut swept: Vec<Vec<u8>> = Vec::new();
16668 let mut at = 0;
16669 while at < starved.len() {
16670 at = starved
16671 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16672 swept.push(text.to_vec());
16673 Ok(())
16674 })
16675 .expect("a sweep reads");
16676 }
16677 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16678 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16679
16680 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16681 let read = (0..generous.len())
16682 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16683 .collect::<Vec<_>>();
16684 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16685 fs::remove_file(path).expect("remove scratch file");
16686 }
16687
16688 #[test]
16689 fn damaged_membership_cannot_skip_a_string_page() {
16690 let path = path("damaged-membership");
16691 let mut writer = Writer::create(
16692 &path,
16693 "items",
16694 vec![
16695 Field::required("id", LogicalType::Integer),
16696 Field::new("text", LogicalType::Varchar),
16697 ],
16698 )
16699 .expect("new file");
16700 writer.append(&sample()).expect("stripe written");
16701 writer.finish().expect("commit");
16702
16703 let reader = Reader::open(&path).expect("valid directory");
16704 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16705 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16706 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16707 file.write_all(&[255]).expect("damage membership");
16708 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16709 assert!(error.message().contains("membership page checksum differs"), "{error}");
16710 fs::remove_file(path).expect("remove scratch file");
16711 }
16712
16713 #[test]
16714 fn membership_delta_stream_is_sorted_exact_and_bounded() {
16715 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16716 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16717 let encoded = encode_membership(&unique);
16718 assert_eq!(
16719 decode_membership(&encoded).expect("valid membership"),
16720 [4, 9, 72, 900, u32::MAX]
16721 );
16722 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16725 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16726 assert_eq!(
16727 decode_membership(&encode_membership(&merged)).expect("valid membership"),
16728 unique
16729 );
16730 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16731 assert!(
16732 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16733 "a value past u32 is invalid"
16734 );
16735 }
16736
16737 #[test]
16738 fn a_global_dictionary_may_be_larger_than_one_column_page() {
16739 let dictionary = Page {
16740 offset: HEADER,
16741 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16742 hash: 0,
16743 };
16744 let table = Table {
16745 name: "items".to_owned(),
16746 fields: vec![Field::new("text", LogicalType::Varchar)],
16747 stripes: Vec::new(),
16748 rows: 0,
16749 dictionaries: vec![Some(dictionary)],
16750 dictionary_payloads: Vec::new(),
16751 demoted: Vec::new(),
16752 distincts: vec![None],
16753 frequencies: vec![None],
16754 pair_frequencies: Vec::new(),
16755 frequency_texts: Vec::new(),
16756 host_groups: None,
16757 clustering: None,
16758 generation: 1,
16759 sections: Vec::new(),
16760 };
16761 let directory = encode_directory(&table).expect("directory");
16762 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16763
16764 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16765 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16766 }
16767
16768 #[test]
16769 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16770 let path = path("constant-codes");
16771 let mut writer =
16772 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16773 .expect("new file");
16774 let empty = vec![Value::Varchar(String::new()); 1024];
16775 for _ in 0..4 {
16776 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16777 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16778 }
16779 writer.finish().expect("commit");
16780
16781 let reader = Reader::open(&path).expect("valid directory");
16782 let pages = reader.layout().columns.first().expect("one column").pages;
16783 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16787 let read = reader.read(3, &[0]).expect("the last part back");
16788 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16789 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16790 fs::remove_file(path).expect("remove scratch file");
16791 }
16792
16793 #[test]
16794 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16795 let over = integer::encode(&[i64::from(i32::MAX) + 1]).expect("a chunk");
16798 let error = cascade(&LogicalType::Integer, &over, 1).expect_err("a page that disagrees");
16799 assert!(format!("{error}").contains("not of its type"), "{error}");
16800 let low = integer::encode(&[i64::MIN]).expect("a chunk");
16801 assert!(cascade(&LogicalType::BigInt, &low, 1).is_ok(), "bigint holds all of i64");
16802 let zero = integer::encode(&[0]).expect("a chunk");
16803 assert!(cascade(&LogicalType::Varchar, &zero, 1).is_err(), "strings are not integers");
16804 }
16805
16806 #[test]
16807 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16808 let mut state: u32 = 0x9e37_79b9;
16812 let spread: Vec<u32> = (0..1024)
16813 .map(|_| {
16814 state ^= state << 13;
16815 state ^= state >> 17;
16816 state ^= state << 5;
16817 state
16818 })
16819 .collect();
16820 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16821 let near: Vec<u32> = (0..1024).collect();
16822 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16823 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16824 }
16825
16826 #[test]
16832 fn two_writes_of_the_same_rows_give_the_same_bytes() {
16833 fn written(path: &PathBuf) {
16834 let fields = (0..40)
16835 .map(|column| {
16836 let ty =
16837 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16838 Field::new(format!("c{column}"), ty)
16839 })
16840 .collect::<Vec<_>>();
16841 let mut writer = Writer::create(path, "wide", fields).expect("new file");
16842 for part in 0..70_u64 {
16843 let columns = (0..40)
16844 .map(|column| {
16845 let values = (0..64_u64)
16846 .map(|row| {
16847 let seed = part.wrapping_mul(31).wrapping_add(row);
16848 if column % 4 == 0 {
16849 Value::Varchar(format!("v{}", seed % 17))
16850 } else {
16851 Value::BigInt(i64::try_from(seed % 97).expect("small"))
16852 }
16853 })
16854 .collect::<Vec<_>>();
16855 let ty = if column % 4 == 0 {
16856 LogicalType::Varchar
16857 } else {
16858 LogicalType::BigInt
16859 };
16860 Vector::from_values(ty, &values).expect("a column")
16861 })
16862 .collect::<Vec<_>>();
16863 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16864 }
16865 writer.finish().expect("commit");
16866 }
16867
16868 let first = path("repeatable-one");
16869 let second = path("repeatable-two");
16870 written(&first);
16871 written(&second);
16872 let left = fs::read(&first).expect("the first file");
16873 let right = fs::read(&second).expect("the second file");
16874 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16875 assert!(left == right, "two writes of the same rows differ in their bytes");
16876
16877 let reader = Reader::open(&first).expect("valid directory");
16880 assert_eq!(reader.table().rows(), 70 * 64);
16881 let read = reader.read(0, &[0, 1]).expect("the first part back");
16882 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16883 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16884 fs::remove_file(first).expect("remove scratch file");
16885 fs::remove_file(second).expect("remove scratch file");
16886 }
16887
16888 fn three_tables(path: &PathBuf) {
16890 let writer = Writer::create(
16891 path,
16892 "region",
16893 vec![
16894 Field::new("r_key", LogicalType::Integer),
16895 Field::new("r_name", LogicalType::Varchar),
16896 ],
16897 )
16898 .expect("new file");
16899 let mut writer = writer;
16900 writer
16901 .append(
16902 &Chunk::new(vec![
16903 Vector::from_values(
16904 LogicalType::Integer,
16905 &[Value::Integer(0), Value::Integer(1)],
16906 )
16907 .expect("keys"),
16908 Vector::from_values(
16909 LogicalType::Varchar,
16910 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16911 )
16912 .expect("names"),
16913 ])
16914 .expect("two columns"),
16915 )
16916 .expect("a part");
16917 let mut writer = writer
16918 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16919 .expect("a second table");
16920 writer
16921 .append(
16922 &Chunk::new(vec![
16923 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16924 ])
16925 .expect("one column"),
16926 )
16927 .expect("a part");
16928 let mut writer =
16929 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16930 for part in 0..70_i64 {
16931 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
16932 writer
16933 .append(
16934 &Chunk::new(vec![
16935 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
16936 ])
16937 .expect("one column"),
16938 )
16939 .expect("a part");
16940 }
16941 writer.finish().expect("commit");
16942 }
16943
16944 #[test]
16945 fn three_tables_in_one_file_read_back_by_name() {
16946 let file = path("three-tables");
16947 three_tables(&file);
16948 let catalog = Catalog::open(&file).expect("a committed catalog");
16949 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
16950
16951 let region = catalog.table("region").expect("the first table");
16952 assert_eq!(region.table().rows(), 2);
16953 assert_eq!(
16954 region.read(0, &[1]).expect("names").value_at(1, 0),
16955 Value::Varchar("ASIA".to_owned())
16956 );
16957
16958 let wide = catalog.table("wide").expect("the third table");
16959 assert_eq!(wide.table().rows(), 70 * 64);
16960 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
16961
16962 let empty = catalog.table("empty").expect("the second table");
16965 assert_eq!(empty.table().rows(), 1);
16966 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
16967
16968 fs::remove_file(file).expect("remove scratch file");
16969 }
16970
16971 #[test]
16972 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
16973 let file = path("three-tables-missing");
16974 three_tables(&file);
16975 let catalog = Catalog::open(&file).expect("a committed catalog");
16976 let error = catalog.table("nation").expect_err("no such table");
16977 assert!(error.message().contains("nation"), "{}", error.message());
16978 fs::remove_file(file).expect("remove scratch file");
16979 }
16980
16981 #[test]
16982 fn a_file_of_three_tables_will_not_open_as_one() {
16983 let file = path("three-tables-unnamed");
16984 three_tables(&file);
16985 let error = Reader::open(&file).expect_err("more than one table");
16986 assert!(error.message().contains("more than one table"), "{}", error.message());
16987 fs::remove_file(file).expect("remove scratch file");
16988 }
16989
16990 #[test]
16992 fn decimals_of_every_storage_width_round_trip() {
16993 let file = path("decimals");
16994 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
16995 let fields = widths
16996 .iter()
16997 .enumerate()
16998 .map(|(index, (width, scale))| {
16999 Field::new(
17000 format!("d{index}"),
17001 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17002 )
17003 })
17004 .collect::<Vec<_>>();
17005 let mut writer = Writer::create(&file, "money", fields).expect("new file");
17006 let rows: [i128; 3] = [-1234, 0, 999];
17007 let columns = widths
17008 .iter()
17009 .map(|(width, scale)| {
17010 let values = rows
17011 .iter()
17012 .map(|unscaled| Value::Decimal {
17013 unscaled: *unscaled,
17014 width: *width,
17015 scale: *scale,
17016 })
17017 .collect::<Vec<_>>();
17018 Vector::from_values(
17019 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17020 &values,
17021 )
17022 .expect("a decimal column")
17023 })
17024 .collect::<Vec<_>>();
17025 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
17026 writer.finish().expect("commit");
17027
17028 let reader = Reader::open(&file).expect("a committed file");
17029 for (index, (width, scale)) in widths.iter().enumerate() {
17030 assert_eq!(
17031 reader.table().fields()[index].ty,
17032 LogicalType::decimal(*width, *scale).expect("a decimal type"),
17033 "column {index} came back as another type"
17034 );
17035 let column = reader.read(0, &[index]).expect("the column");
17036 for (row, unscaled) in rows.iter().enumerate() {
17037 assert_eq!(
17038 column.value_at(row, 0),
17039 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
17040 "column {index} row {row}"
17041 );
17042 }
17043 }
17044 fs::remove_file(file).expect("remove scratch file");
17045 }
17046
17047 #[test]
17048 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
17049 let file = path("two-of-a-name");
17050 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
17051 .expect("new file");
17052 let error = writer
17053 .next("t", vec![Field::new("a", LogicalType::BigInt)])
17054 .expect_err("the same name twice");
17055 assert!(error.message().contains("same name"), "{}", error.message());
17056 fs::remove_file(file).expect("remove scratch file");
17057 }
17058
17059 #[test]
17060 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
17061 let file = path("integer-tally");
17062 let mut writer =
17063 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
17064 .expect("new file");
17065 let mut values = vec![Value::SmallInt(0); 1024];
17066 values[7] = Value::SmallInt(3);
17067 values[99] = Value::SmallInt(-2);
17068 values[1001] = Value::SmallInt(3);
17069 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
17070 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
17071 values[0] = Value::Null;
17072 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
17073 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
17074 writer.finish().expect("commit");
17075
17076 let reader = Reader::open(&file).expect("read file");
17077 assert_eq!(
17078 reader.integer_tally(0, 0).expect("valid part"),
17079 Some(vec![(-2, 1), (0, 1021), (3, 2)])
17080 );
17081 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
17082 let catalog = Catalog::open(&file).expect("catalog");
17083 assert_eq!(
17084 catalog.integer_tally("events", 0).expect("nullable column"),
17085 Some(vec![(-2, 2), (0, 2041), (3, 4)])
17086 );
17087 fs::remove_file(file).expect("remove scratch file");
17088 }
17089
17090 #[test]
17091 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
17092 let file = path("catalog-integer-tally");
17093 let mut writer = Writer::create(
17094 &file,
17095 "events",
17096 vec![
17097 Field::new("noise", LogicalType::SmallInt),
17098 Field::new("source", LogicalType::SmallInt),
17099 ],
17100 )
17101 .expect("new file");
17102 let noise = vec![Value::SmallInt(9); 1024];
17103 let mut source = vec![Value::SmallInt(0); 1024];
17104 source[7] = Value::SmallInt(3);
17105 source[99] = Value::SmallInt(-2);
17106 let chunk = Chunk::new(vec![
17107 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
17108 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
17109 ])
17110 .expect("two columns");
17111 writer.append(&chunk).expect("append");
17112 writer.finish().expect("commit");
17113
17114 let catalog = Catalog::open(&file).expect("catalog");
17115 assert_eq!(
17116 catalog.integer_tally("events", 1).expect("selected column"),
17117 Some(vec![(-2, 1), (0, 1022), (3, 1)])
17118 );
17119 assert_eq!(
17120 catalog.integer_tally("events", 0).expect("other column"),
17121 Some(vec![(9, 1024)])
17122 );
17123 fs::remove_file(file).expect("remove scratch file");
17124 }
17125
17126 #[test]
17127 fn opening_the_catalog_reads_no_table_directory() {
17128 let file = path("catalog-only");
17129 three_tables(&file);
17130 let catalog = Catalog::open(&file).expect("a committed catalog");
17131 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
17134 assert_eq!(catalog.names().len(), 3);
17135 fs::remove_file(file).expect("remove scratch file");
17136 }
17137
17138 #[test]
17149 fn the_checksum_answers_what_it_has_always_answered() {
17150 let bytes: Vec<u8> =
17151 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
17152 for (length, expected) in [
17153 (0, 0xef46_db37_51d8_e999),
17154 (1, 0xa96c_7f0c_e858_bbb7),
17155 (3, 0x56e6_9576_32a4_87f9),
17156 (4, 0xc60d_15b1_e3ff_8f04),
17157 (5, 0x8088_1585_8624_dd4e),
17158 (7, 0xafbe_fc3d_6c6f_9a8e),
17159 (8, 0x3da5_c7aa_2696_83e0),
17160 (9, 0x465e_c429_b13c_3892),
17161 (15, 0xdee8_9d8a_065a_6233),
17162 (16, 0x1330_489a_7767_9c80),
17163 (31, 0x3391_303d_485e_846e),
17164 (32, 0x40b7_aff7_5d45_bbc8),
17165 (33, 0x4997_cae4_951c_17a5),
17166 (39, 0x5807_28fd_5c14_5739),
17167 (40, 0xf95c_f6f5_c08a_3d3b),
17168 (63, 0x2944_b4da_fc69_b206),
17169 (64, 0xbb76_f6ef_19bd_5a1b),
17170 (65, 0x814e_0c65_4a9f_d640),
17171 (127, 0x00de_aab1_31cf_f89b),
17172 (1000, 0x9e33_00c1_cde3_c58d),
17173 ] {
17174 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
17175 }
17176 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
17177 }
17178 #[test]
17185 fn a_declared_order_comes_back_out_of_the_file() {
17186 let path = path("clustered");
17187 let shipped = vec![
17188 Field::new("key", LogicalType::BigInt),
17189 Field::new("line", LogicalType::Integer),
17190 Field::new("shipdate", LogicalType::Date),
17191 ];
17192 let plain = vec![Field::new("a", LogicalType::Integer)];
17193 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
17194
17195 let mut writer = Writer::create(&path, "lineitem", shipped)
17196 .expect("new file")
17197 .declare(stage_zero.clone())
17198 .expect("the columns are the table's");
17199 let column = |ty: LogicalType, values: &[Value]| {
17200 Vector::from_values(ty, values).expect("the values match the type")
17201 };
17202 writer
17203 .append(
17204 &Chunk::new(vec![
17205 column(
17206 LogicalType::BigInt,
17207 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
17208 ),
17209 column(
17210 LogicalType::Integer,
17211 &[
17212 Value::Integer(1),
17213 Value::Integer(1),
17214 Value::Integer(1),
17215 Value::Integer(1),
17216 ],
17217 ),
17218 column(
17219 LogicalType::Date,
17220 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17221 ),
17222 ])
17223 .expect("three columns"),
17224 )
17225 .expect("four rows");
17226 let mut writer = writer.next("nation", plain).expect("a second table");
17227 writer
17228 .append(
17229 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17230 .expect("one column"),
17231 )
17232 .expect("one row");
17233 writer.finish().expect("commit");
17234
17235 let catalog = Catalog::open(&path).expect("reopen");
17236 let lineitem = catalog.table("lineitem").expect("the clustered table");
17237 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17238 let nation = catalog.table("nation").expect("the plain table");
17239 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17240
17241 assert_eq!(lineitem.table().rows(), 4);
17244 assert_eq!(nation.table().rows(), 1);
17245 fs::remove_file(&path).ok();
17246 }
17247
17248 #[test]
17250 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17251 let path = path("clustered-bad");
17252 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17253 .expect("new file");
17254 let four =
17255 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17256 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17257 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17258 fs::remove_file(&path).ok();
17259 }
17260
17261 #[test]
17267 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17268 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17269 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17270 .collect::<Vec<_>>();
17271 let filled = || {
17272 let mut dictionary = GlobalDictionary::new();
17273 for value in &values {
17274 dictionary.code(value).expect("a code for every value");
17275 }
17276 dictionary.settle().expect("a shape");
17277 dictionary
17278 };
17279 let mut in_place = filled();
17280 in_place.finish_blocks().expect("every block encodes");
17281
17282 let mut handed = filled();
17283 let out = handed.hand_out(3);
17284 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17285 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17286 for job in out.iter().rev() {
17287 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17288 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17289 }
17290 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17291 handed.finish_blocks().expect("the last block encodes");
17292
17293 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17294 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17295 }
17296
17297 #[test]
17299 fn a_block_given_back_twice_is_refused() {
17300 let mut dictionary = GlobalDictionary::new();
17301 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17302 dictionary.code(&format!("value {at}")).expect("a code");
17303 }
17304 dictionary.settle().expect("a shape");
17305 let out = dictionary.hand_out(0);
17306 let last = out.last().expect("blocks went out");
17307 let at = last.place().1;
17308 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17309 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17310 }
17311
17312 #[test]
17318 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17319 let mut values = vec![String::new(), "http://".to_owned()];
17320 for host in 0..7 {
17321 for path in 0..30 {
17322 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17323 values.push(format!("http://example{host}.test/page/{path:04}"));
17324 }
17325 }
17326 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17327
17328 let mut dictionary = GlobalDictionary::new();
17329 for value in &values {
17330 dictionary.code(value).expect("a code for every value");
17331 }
17332 dictionary.finish_blocks().expect("the last block encodes");
17333 let ranked = dictionary.ranked(None).expect("a sorted order");
17334 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17335
17336 let spellings = dictionary_values(&dictionary);
17337 let seen = ranked
17338 .iter()
17339 .map(|&(_, code)| {
17340 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17341 })
17342 .collect::<Vec<_>>();
17343 let mut wanted = values.clone();
17344 wanted.sort_unstable();
17345 assert_eq!(seen, wanted, "the order is the order the bytes give");
17346
17347 for &(carried, code) in &ranked {
17348 let value = &spellings[code as usize];
17349 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17350 }
17351 }
17352
17353 #[test]
17358 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17359 let entry =
17360 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17361 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17362 .map(|code| entry(code, u64::from(code % 7) + 1))
17363 .collect::<Vec<_>>();
17364 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17365
17366 let mut sorted = all.clone();
17367 sorted.sort_unstable_by(|left, right| {
17368 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17369 });
17370 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17371 sorted.truncate(FREQUENCY_ENTRIES);
17372
17373 let mut picked = all.clone();
17374 let omitted = keep_most_frequent(&mut picked);
17375 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17376 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17377 assert!(
17378 picked
17379 .iter()
17380 .zip(&sorted)
17381 .all(|(one, two)| one.value == two.value && one.count == two.count),
17382 "the same entries in the same order"
17383 );
17384
17385 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17386 let omitted = keep_most_frequent(&mut short);
17387 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17388 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17389 }
17390
17391 #[test]
17393 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17394 let empty = GlobalDictionary::new();
17395 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17396
17397 let mut dictionary = GlobalDictionary::new();
17398 for value in ["pear", "apple", "", "apples", "app"] {
17399 dictionary.code(value).expect("a code for every value");
17400 }
17401 dictionary.finish_blocks().expect("the one block encodes");
17402 let spellings = dictionary_values(&dictionary);
17403 let seen = dictionary
17404 .ranked(None)
17405 .expect("a sorted order")
17406 .iter()
17407 .map(|&(_, code)| spellings[code as usize].clone())
17408 .collect::<Vec<_>>();
17409 let wanted: Vec<Vec<u8>> =
17410 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17411 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17412 }
17413
17414 #[test]
17417 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17418 let profile = LoadProfile::begin("demoted");
17419 let mut dictionary = GlobalDictionary::new();
17420 for value in 0..50_000 {
17421 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17422 }
17423 let (_, grown) = dictionary.recharge(Some(&profile));
17424 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17425
17426 dictionary.demote();
17427 let (before, after) = dictionary.recharge(Some(&profile));
17428 assert_eq!(before, grown);
17429 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17432 assert_eq!(profile.held(), after, "the profile was told about the drop");
17433 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17434
17435 dictionary.demote();
17436 assert_eq!(
17437 dictionary.recharge(Some(&profile)),
17438 (after, after),
17439 "demoting twice is a no-op"
17440 );
17441 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17442 }
17443}