1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::File;
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_io::{Filesystem, OpenMode, RealFilesystem};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{DICTIONARY_CAP_BYTES, Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 29;
79
80const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
111
112const HEADER: u64 = 80;
113const SLOT_BYTES: usize = 28;
114const MAX_PAGE: usize = 256 * 1024 * 1024;
115const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
116const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
117const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
118const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
126const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
128const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
134const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
149const DEMOTED: &[u8; 8] = b"RUDBDM1\0";
169const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
177const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
185
186const MAX_SECTIONS: usize = 4096;
193const FREQUENCY_CANDIDATES: usize = 32_768;
194const FREQUENCY_ENTRIES: usize = 512;
195const FREQUENCY_BUILD_RANK: usize = 10;
196const FREQUENCY_ORDINALS: usize = 131_072;
197const MAX_PAIR_FREQUENCIES: usize = 1024;
198const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
203const MAX_FREQUENCY_WORKERS: usize = 32;
210
211fn close_workers() -> usize {
213 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
214}
215
216const CLOSE_BYTES: usize = 1 << 30;
227
228const NUMERIC_CLOSE_BYTES: usize = 4 << 20;
231
232const MAX_ENCODE_WORKERS: usize = 32;
239
240const WRITEBACK_STRETCH: u64 = 32 << 20;
248
249const SIEVE_BUDGET: usize = 8 * 1024;
257
258const PART_BOUND_BYTES: usize = 24;
267
268fn io(error: std::io::Error) -> Error {
269 Error::io(error.to_string())
270}
271
272fn invalid(message: &str) -> Error {
273 Error::invalid_input(format!("invalid rudb native file: {message}"))
274}
275
276fn sum(counts: impl Iterator<Item = u64>) -> u64 {
278 counts.fold(0, u64::saturating_add)
279}
280
281fn span_bytes(spans: &[Span], at: usize) -> u64 {
283 spans.get(at).map_or(0, |span| u64::from(span.length))
284}
285
286fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
288 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
289}
290
291fn dictionary_bytes(table: &Table, at: usize) -> u64 {
293 page_bytes(&table.dictionaries, at)
294 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
295}
296
297fn checksum(bytes: &[u8]) -> u64 {
307 seeded_checksum(bytes, 0)
308}
309
310#[must_use]
317pub fn content_name(bytes: &[u8]) -> u128 {
318 let seed = u64::from(FORMAT);
319 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
320}
321
322#[derive(Debug, Clone)]
328pub struct ContentNamer {
329 seeds: [u64; 2],
330 lanes: [[u64; 4]; 2],
331 held: [u8; 32],
332 filled: usize,
333 length: u64,
334}
335
336impl Default for ContentNamer {
337 fn default() -> Self {
338 let seed = u64::from(FORMAT);
339 let seeds = [seed, !seed];
340 let lanes = seeds.map(|seed| {
341 [
342 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
343 seed.wrapping_add(XXH_P2),
344 seed,
345 seed.wrapping_sub(XXH_P1),
346 ]
347 });
348 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
349 }
350}
351
352impl ContentNamer {
353 pub fn update(&mut self, mut bytes: &[u8]) {
355 self.length += bytes.len() as u64;
356 if self.filled > 0 {
357 let take = (32 - self.filled).min(bytes.len());
358 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
359 self.filled += take;
360 bytes = &bytes[take..];
361 if self.filled < 32 {
362 return;
363 }
364 let block = self.held;
365 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
366 self.filled = 0;
367 }
368 let mut blocks = bytes.chunks_exact(32);
369 for block in blocks.by_ref() {
370 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
371 }
372 let rest = blocks.remainder();
373 self.held[..rest.len()].copy_from_slice(rest);
374 self.filled = rest.len();
375 }
376
377 #[must_use]
379 pub fn finish(&self) -> u128 {
380 let rest = &self.held[..self.filled];
381 let [first, second] = [0, 1].map(|at| {
382 if self.length < 32 {
383 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
384 } else {
385 finish_checksum(self.lanes[at], rest, self.length)
386 }
387 });
388 u128::from(first) << 64 | u128::from(second)
389 }
390}
391
392fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
401 let mut blocks = bytes.chunks_exact(32);
404 let rest = blocks.remainder();
405 if bytes.len() < 32 {
406 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
407 }
408 let mut lanes = [
409 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
410 seed.wrapping_add(XXH_P2),
411 seed,
412 seed.wrapping_sub(XXH_P1),
413 ];
414 for block in blocks.by_ref() {
415 checksum_block(&mut lanes, block);
416 }
417 finish_checksum(lanes, rest, bytes.len() as u64)
418}
419
420const XXH_P1: u64 = 11_400_714_785_074_694_791;
421const XXH_P2: u64 = 14_029_467_366_897_019_727;
422const XXH_P3: u64 = 1_609_587_929_392_839_161;
423const XXH_P4: u64 = 9_650_029_242_287_828_579;
424const XXH_P5: u64 = 2_870_177_450_012_600_261;
425
426fn checksum_round(state: u64, word: u64) -> u64 {
427 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
428}
429
430fn checksum_word(chunk: &[u8]) -> u64 {
431 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
432}
433
434fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
436 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
437 *lane = checksum_round(*lane, checksum_word(chunk));
438 }
439}
440
441fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
443 let merge = |state: u64, lane: u64| {
444 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
445 };
446 let [one, two, three, four] = lanes;
447 let combined = one
448 .rotate_left(1)
449 .wrapping_add(two.rotate_left(7))
450 .wrapping_add(three.rotate_left(12))
451 .wrapping_add(four.rotate_left(18));
452 let hash = merge(merge(merge(merge(combined, one), two), three), four);
453 checksum_tail(hash.wrapping_add(length), rest)
454}
455
456fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
458 let mut words = rest.chunks_exact(8);
459 for chunk in words.by_ref() {
460 hash ^= checksum_round(0, checksum_word(chunk));
461 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
462 }
463 rest = words.remainder();
464 if rest.len() >= 4 {
465 let (head, tail) = rest.split_at(4);
466 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
467 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
468 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
469 rest = tail;
470 }
471 for &byte in rest {
472 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
473 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
474 }
475 hash ^= hash >> 33;
476 hash = hash.wrapping_mul(XXH_P2);
477 hash ^= hash >> 29;
478 hash = hash.wrapping_mul(XXH_P3);
479 hash ^ (hash >> 32)
480}
481
482fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
488 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
489}
490
491fn walk_checksummed(
497 file: &File,
498 offset: u64,
499 length: usize,
500 window: usize,
501 mut each: impl FnMut(&[u8]) -> Result<()>,
502) -> Result<u64> {
503 debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
504 if length < 32 {
505 let mut bytes = vec![0; length];
506 read_at(file, offset, &mut bytes)?;
507 each(&bytes)?;
508 return Ok(checksum(&bytes));
509 }
510 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
511 let mut buffer = vec![0; window.min(length)];
512 let mut read = 0;
513 let (mut whole, mut filled) = (0, 0);
514 while read < length {
515 filled = buffer.len().min(length - read);
516 read_at(file, offset + read as u64, &mut buffer[..filled])?;
517 read += filled;
518 each(&buffer[..filled])?;
519 whole = filled / 32 * 32;
520 for block in buffer[..whole].chunks_exact(32) {
521 checksum_block(&mut lanes, block);
522 }
523 }
524 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
525}
526
527#[derive(Debug, Clone, Copy)]
528struct Slot {
529 offset: u64,
530 length: u32,
531 generation: u64,
532 hash: u64,
533}
534
535impl Slot {
536 fn bytes(self) -> [u8; SLOT_BYTES] {
537 let mut result = [0; SLOT_BYTES];
538 result[..8].copy_from_slice(&self.offset.to_le_bytes());
539 result[8..12].copy_from_slice(&self.length.to_le_bytes());
540 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
541 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
542 result
543 }
544
545 fn read(bytes: &[u8]) -> Self {
546 Self {
547 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
548 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
549 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
550 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
551 }
552 }
553}
554
555#[derive(Debug, Clone, Copy)]
556struct Page {
557 offset: u64,
558 length: u32,
559 hash: u64,
560}
561
562impl Page {
563 fn bytes(&self) -> u64 {
565 u64::from(self.length)
566 }
567}
568
569#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
570enum FrequencyValue {
571 Null,
572 Integer(i128),
573 Code(u32),
574}
575
576type FrequencyMap<V> = HashMap<u64, V, Spread>;
582
583#[derive(Debug)]
597struct Candidates {
598 slots: Vec<Candidate>,
601 held: usize,
602 nulls: u32,
603 decrements: u64,
604 survivors: Vec<Candidate>,
606}
607
608#[derive(Debug, Default, Clone, Copy)]
610struct Candidate {
611 bits: u64,
612 count: u32,
613}
614
615const FIRST_CANDIDATE_SLOTS: usize = 64;
617
618impl Default for Candidates {
619 fn default() -> Self {
620 Self {
621 slots: vec![Candidate::default(); FIRST_CANDIDATE_SLOTS],
622 held: 0,
623 nulls: 0,
624 decrements: 0,
625 survivors: Vec::new(),
626 }
627 }
628}
629
630fn count_from_full(
636 first: &Candidates,
637 distinct: &mut Option<distinct::ExactDistinct>,
638 ended: Option<u64>,
639 counted: bool,
640) {
641 if distinct.is_some() || !first.full() {
642 return;
643 }
644 if !counted {
645 *distinct = Some(distinct::ExactDistinct::declined());
646 return;
647 }
648 let mut set = distinct::ExactDistinct::new();
649 for held in first.held_bits() {
650 set.insert(held);
651 }
652 if let Some(ended) = ended {
653 set.insert(ended);
654 }
655 *distinct = Some(set);
656}
657
658impl Candidates {
659 fn add(&mut self, bits: Option<u64>, mut times: u32) {
666 while times > 0 {
667 let room = self.held + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES;
668 match bits {
669 Some(bits) => {
670 let (at, found) = self.find(bits);
671 if found {
672 self.slots[at].count = self.slots[at].count.saturating_add(times);
673 return;
674 }
675 if room {
676 self.place(at, bits, times);
677 return;
678 }
679 }
680 None if self.nulls != 0 => {
681 self.nulls = self.nulls.saturating_add(times);
682 return;
683 }
684 None if room => {
685 self.nulls = times;
686 return;
687 }
688 None => {}
689 }
690 self.decrement();
691 times -= 1;
692 }
693 }
694
695 fn full(&self) -> bool {
697 self.held + usize::from(self.nulls != 0) >= FREQUENCY_CANDIDATES
698 }
699
700 fn held_bits(&self) -> impl Iterator<Item = u64> + '_ {
702 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| slot.bits)
703 }
704
705 fn find(&self, bits: u64) -> (usize, bool) {
707 let mask = self.slots.len() - 1;
708 let mut at = home(bits, self.slots.len());
709 loop {
710 let slot = self.slots[at];
711 if slot.count == 0 {
712 return (at, false);
713 }
714 if slot.bits == bits {
715 return (at, true);
716 }
717 at = (at + 1) & mask;
718 }
719 }
720
721 fn position(&self, bits: u64) -> Option<usize> {
723 match self.find(bits) {
724 (at, true) => Some(at),
725 (_, false) => None,
726 }
727 }
728
729 fn place(&mut self, at: usize, bits: u64, count: u32) {
732 let at = if (self.held + 1) * 2 > self.slots.len() {
733 let wider = self.slots.len() * 2;
734 let old = std::mem::replace(&mut self.slots, vec![Candidate::default(); wider]);
735 for slot in old.into_iter().filter(|slot| slot.count != 0) {
736 let (to, _) = self.find(slot.bits);
737 self.slots[to] = slot;
738 }
739 self.find(bits).0
740 } else {
741 at
742 };
743 self.slots[at] = Candidate { bits, count };
744 self.held += 1;
745 }
746
747 fn decrement(&mut self) {
749 let mut survivors = std::mem::take(&mut self.survivors);
750 survivors.clear();
751 survivors.extend(
752 self.slots
753 .iter()
754 .filter(|slot| slot.count > 1)
755 .map(|slot| Candidate { bits: slot.bits, count: slot.count - 1 }),
756 );
757 self.slots.fill(Candidate::default());
758 self.held = survivors.len();
759 for &slot in &survivors {
760 let (at, _) = self.find(slot.bits);
761 self.slots[at] = slot;
762 }
763 self.survivors = survivors;
764 self.nulls = self.nulls.saturating_sub(1);
765 self.decrements = self.decrements.saturating_add(1);
766 }
767
768 fn pairs(&self) -> impl Iterator<Item = (u64, u32)> + '_ {
770 self.slots.iter().filter(|slot| slot.count != 0).map(|slot| (slot.bits, slot.count))
771 }
772}
773
774fn home(bits: u64, slots: usize) -> usize {
779 (bits.wrapping_mul(0x9E37_79B9_7F4A_7C15) >> (64 - slots.trailing_zeros())) as usize
780}
781
782#[derive(Debug, Default)]
784struct Run {
785 bits: Option<u64>,
786 times: u32,
787}
788
789impl Run {
790 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
792 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
793 self.times += 1;
794 return None;
795 }
796 let ended = self.take();
797 self.bits = bits;
798 self.times = 1;
799 ended
800 }
801
802 fn take(&mut self) -> Option<(Option<u64>, u32)> {
804 let times = std::mem::take(&mut self.times);
805 (times != 0).then_some((self.bits, times))
806 }
807}
808
809#[derive(Debug, Default, Clone, Copy)]
811struct Spread;
812
813impl std::hash::BuildHasher for Spread {
814 type Hasher = SpreadHasher;
815
816 fn build_hasher(&self) -> SpreadHasher {
817 SpreadHasher(0)
818 }
819}
820
821#[derive(Debug)]
828struct SpreadHasher(u64);
829
830impl SpreadHasher {
831 fn mix(&mut self, word: u64) {
832 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
833 self.0 = (product as u64) ^ ((product >> 64) as u64);
834 }
835}
836
837impl std::hash::Hasher for SpreadHasher {
838 fn write(&mut self, bytes: &[u8]) {
839 for part in bytes.chunks(8) {
840 let mut word = [0; 8];
841 word[..part.len()].copy_from_slice(part);
842 self.mix(u64::from_le_bytes(word));
843 }
844 }
845
846 fn write_u32(&mut self, value: u32) {
847 self.mix(u64::from(value));
848 }
849
850 fn write_u64(&mut self, value: u64) {
851 self.mix(value);
852 }
853
854 fn write_i128(&mut self, value: i128) {
855 self.mix(value as u64);
856 self.mix((value >> 64) as u64);
857 }
858
859 fn write_isize(&mut self, value: isize) {
860 self.mix(value as u64);
861 }
862
863 fn finish(&self) -> u64 {
864 self.0
865 }
866}
867
868#[derive(Debug, Clone)]
869struct FrequencyEntry {
870 value: FrequencyValue,
871 count: u64,
872}
873
874#[derive(Debug, Clone)]
879struct FrequencySummary {
880 entries: Vec<FrequencyEntry>,
881 omitted_max: u64,
882 ordinals: Vec<u64>,
883 ordinal_entries: Vec<u16>,
884}
885
886#[derive(Debug, Clone)]
887struct PairFrequencyEntry {
888 first_entry: u16,
889 second: Option<u32>,
890 count: u64,
891}
892
893#[derive(Debug, Clone)]
899struct PairFrequencySummary {
900 first: u16,
901 second: u16,
902 entries: Vec<PairFrequencyEntry>,
903 omitted_max: u64,
904}
905
906#[derive(Debug, Clone)]
914enum Frequencies {
915 Held(FrequencySummary),
916 Stored {
919 span: Span,
920 values: bool,
921 },
922}
923
924#[derive(Debug, Clone)]
929pub struct FrequencyPrefix {
930 pub entries: Vec<(Value, u64)>,
932 pub omitted_max: u64,
934}
935
936#[derive(Debug, Clone, PartialEq)]
938pub struct FrequencyOccurrences {
939 pub omitted_max: u64,
941 pub ordinals: Vec<u64>,
943 pub anchors: Vec<Value>,
945 pub anchor_indices: Vec<u16>,
947}
948
949pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
951
952#[derive(Debug, Clone, Copy, Default)]
959struct Span {
960 offset: u64,
961 length: u32,
962}
963
964#[derive(Debug, Clone, Default)]
972struct Pages {
973 columns: usize,
974 held: Box<[StripePage]>,
975}
976
977#[derive(Debug, Clone, Copy)]
979struct StripePage {
980 offset: u64,
981 hash: u64,
982 length: u32,
983 column: u32,
984}
985
986impl Pages {
987 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
989 let mut held = Vec::with_capacity(slots.iter().flatten().count());
990 for (column, page) in slots.iter().enumerate() {
991 if let Some(page) = page {
992 let column =
993 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
994 held.push(StripePage {
995 offset: page.offset,
996 hash: page.hash,
997 length: page.length,
998 column,
999 });
1000 }
1001 }
1002 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
1003 }
1004
1005 fn get(&self, column: usize) -> Option<Page> {
1007 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
1008 let placed = self.held[at];
1009 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
1010 }
1011
1012 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
1014 (0..self.columns).map(|column| self.get(column))
1015 }
1016
1017 fn bytes(&self, column: usize) -> u64 {
1019 self.get(column).map_or(0, |page| page.bytes())
1020 }
1021}
1022
1023#[derive(Debug, Clone)]
1025pub struct Stripe {
1026 rows: usize,
1027 parts: Vec<u32>,
1030 index: Span,
1034 pages: Vec<Span>,
1035 memberships: Pages,
1036 sieves: Pages,
1039 part_ranges: Pages,
1050 zone: Zone,
1051}
1052
1053impl Stripe {
1054 #[must_use]
1056 pub fn rows(&self) -> usize {
1057 self.rows
1058 }
1059
1060 #[must_use]
1062 pub fn parts(&self) -> usize {
1063 self.parts.len()
1064 }
1065
1066 #[must_use]
1072 pub fn zone(&self) -> &Zone {
1073 &self.zone
1074 }
1075}
1076
1077#[derive(Debug, Clone)]
1079pub struct Table {
1080 name: String,
1081 fields: Vec<Field>,
1082 stripes: Vec<Stripe>,
1083 rows: usize,
1084 dictionaries: Vec<Option<Page>>,
1085 dictionary_payloads: Vec<u64>,
1091 demoted: Vec<bool>,
1097 frequencies: Vec<Option<Frequencies>>,
1098 pair_frequencies: Vec<PairFrequencySummary>,
1099 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
1104 host_groups: Option<host::HostSummary>,
1106 distincts: Vec<Option<u64>>,
1116 clustering: Option<Clustering>,
1124 generation: u64,
1138 sections: Vec<Section>,
1145}
1146
1147impl Table {
1148 #[must_use]
1150 pub fn name(&self) -> &str {
1151 &self.name
1152 }
1153
1154 #[must_use]
1156 pub fn fields(&self) -> &[Field] {
1157 &self.fields
1158 }
1159
1160 #[must_use]
1162 pub fn rows(&self) -> usize {
1163 self.rows
1164 }
1165
1166 #[must_use]
1168 pub fn stripes(&self) -> &[Stripe] {
1169 &self.stripes
1170 }
1171
1172 #[must_use]
1174 pub fn clustering(&self) -> Option<&Clustering> {
1175 self.clustering.as_ref()
1176 }
1177
1178 #[must_use]
1183 pub fn generation(&self) -> u64 {
1184 self.generation
1185 }
1186
1187 #[must_use]
1194 pub fn sections(&self) -> &[Section] {
1195 &self.sections
1196 }
1197}
1198
1199#[derive(Debug, Clone)]
1211struct Entry {
1212 name: String,
1213 fields: Vec<Field>,
1214 rows: usize,
1215 directory: Page,
1217 nonzero: Vec<Option<u64>>,
1220 aggregates: Vec<Option<(i128, u64)>>,
1222 distincts: Vec<Option<u64>>,
1224 extremes: Vec<StoredIntegerExtremes>,
1226 frequencies: Vec<StoredNumericFrequencies>,
1228}
1229
1230type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1231type StoredNumericFrequencies = Option<NumericFrequencies>;
1232
1233#[derive(Debug, Clone, PartialEq, Eq)]
1246pub struct ViewEntry {
1247 pub name: String,
1249 pub sql: String,
1251 pub statement: String,
1253 pub aliases: Vec<String>,
1255 pub columns: Vec<Field>,
1257}
1258
1259#[derive(Debug, Clone)]
1261pub struct ColumnLayout {
1262 pub name: String,
1264 pub kind: String,
1266 pub pages: u64,
1268 pub memberships: u64,
1270 pub sieves: u64,
1272 pub part_ranges: u64,
1274 pub dictionary: u64,
1276}
1277
1278impl ColumnLayout {
1279 #[must_use]
1281 pub fn total(&self) -> u64 {
1282 self.pages
1283 .saturating_add(self.memberships)
1284 .saturating_add(self.sieves)
1285 .saturating_add(self.part_ranges)
1286 .saturating_add(self.dictionary)
1287 }
1288}
1289
1290#[derive(Debug, Clone)]
1301pub struct Layout {
1302 pub file: u64,
1304 pub rows: usize,
1306 pub stripes: usize,
1308 pub parts: usize,
1310 pub columns: Vec<ColumnLayout>,
1312 pub indexes: u64,
1315 pub directory: u64,
1317 pub header: u64,
1319}
1320
1321impl Layout {
1322 #[must_use]
1324 pub fn columns_total(&self) -> u64 {
1325 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1326 }
1327
1328 #[must_use]
1334 pub fn unaccounted(&self) -> u64 {
1335 self.file
1336 .saturating_sub(self.columns_total())
1337 .saturating_sub(self.indexes)
1338 .saturating_sub(self.directory)
1339 .saturating_sub(self.header)
1340 }
1341}
1342
1343#[derive(Debug, Clone)]
1354pub struct StoredPart {
1355 pub stripe: usize,
1357 pub part: usize,
1359 pub row: usize,
1361 pub rows: usize,
1363 pub encoding: String,
1365 pub bytes: u64,
1367 pub page: u64,
1369 pub offset: u64,
1371 pub low: Option<Value>,
1373 pub high: Option<Value>,
1375 pub nulls: Option<usize>,
1377}
1378
1379const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1386
1387#[derive(Debug)]
1412struct GlobalDictionary {
1413 primary: HashMap<u64, u32, Spread>,
1417 collisions: HashMap<u64, Vec<u32>, Spread>,
1418 checks: Vec<u64>,
1420 ends: Vec<u32>,
1422 counts: Vec<u64>,
1423 nulls: u64,
1424 filling: Vec<u8>,
1426 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1432 waiting: Vec<(usize, Vec<u8>)>,
1437 sample: Vec<(usize, Vec<u8>)>,
1443 stride: usize,
1445 shape: Option<chooser::Settled>,
1447 settled: usize,
1449 blocks: Vec<Vec<u8>>,
1454 early: BTreeMap<usize, EncodedBlock>,
1460 placed: Vec<Placed>,
1462 charged: u64,
1465 demoted: bool,
1467}
1468
1469#[derive(Debug, Clone, Copy)]
1471struct Placed {
1472 start: u64,
1473 length: u64,
1474 hash: u64,
1475}
1476
1477type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1479
1480impl GlobalDictionary {
1481 fn new() -> Self {
1482 Self {
1483 primary: HashMap::default(),
1484 collisions: HashMap::default(),
1485 checks: Vec::new(),
1486 ends: Vec::new(),
1487 counts: Vec::new(),
1488 nulls: 0,
1489 filling: Vec::new(),
1490 grams: Vec::new(),
1491 waiting: Vec::new(),
1492 sample: Vec::new(),
1493 stride: 1,
1494 shape: None,
1495 settled: 0,
1496 blocks: Vec::new(),
1497 early: BTreeMap::new(),
1498 placed: Vec::new(),
1499 charged: 0,
1500 demoted: false,
1501 }
1502 }
1503
1504 fn values(&self) -> usize {
1506 self.ends.len()
1507 }
1508
1509 fn closing_bytes(&self) -> usize {
1512 let values = self.values();
1513 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1514 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1515 .sum::<usize>();
1516 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1517 }
1518
1519 fn held_bytes(&self) -> u64 {
1525 fn table<K, V, S>(map: &HashMap<K, V, S>) -> usize {
1526 (map.capacity() * 8 / 7).next_power_of_two() * (size_of::<(K, V)>() + 1)
1527 }
1528 fn spilled<T>(values: &Vec<T>) -> usize {
1529 values.capacity() * size_of::<T>()
1530 }
1531 let raw = |blocks: &Vec<(usize, Vec<u8>)>| {
1532 spilled(blocks) + blocks.iter().map(|(_, block)| block.capacity()).sum::<usize>()
1533 };
1534 let bytes = table(&self.primary)
1535 + table(&self.collisions)
1536 + self.collisions.values().map(spilled).sum::<usize>()
1537 + spilled(&self.checks)
1538 + spilled(&self.ends)
1539 + spilled(&self.counts)
1540 + self.filling.capacity()
1541 + spilled(&self.grams)
1542 + raw(&self.waiting)
1543 + raw(&self.sample)
1544 + self.blocks.iter().map(Vec::capacity).sum::<usize>()
1545 + spilled(&self.placed);
1546 bytes as u64
1547 }
1548
1549 fn recharge(&mut self, profile: Option<&LoadProfile>) -> (u64, u64) {
1552 let before = self.charged;
1553 let now = self.held_bytes();
1554 if let Some(profile) = profile {
1555 if now >= before {
1556 profile.hold(now - before);
1557 } else {
1558 profile.release(before - now);
1559 }
1560 }
1561 self.charged = now;
1562 (before, now)
1563 }
1564
1565 fn demote(&mut self) {
1573 if self.demoted {
1574 return;
1575 }
1576 self.seal_rest();
1577 self.release_lookup();
1578 self.demoted = true;
1579 }
1580
1581 fn release_lookup(&mut self) {
1588 self.primary = HashMap::default();
1589 self.collisions = HashMap::default();
1590 self.checks = Vec::new();
1591 self.sample = Vec::new();
1592 self.filling = Vec::new();
1593 }
1594
1595 fn encoded(&self) -> usize {
1597 self.placed.len() + self.blocks.len()
1598 }
1599
1600 #[cfg(test)]
1601 fn code(&mut self, text: &str) -> Result<u32> {
1602 let bytes = text.as_bytes();
1603 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1604 }
1605
1606 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1612 if let Some(&code) = self.primary.get(&hash) {
1613 if self.checks.get(code as usize) == Some(&check) {
1614 return Ok(code);
1615 }
1616 if let Some(codes) = self.collisions.get(&hash) {
1617 if let Some(code) =
1618 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1619 {
1620 return Ok(code);
1621 }
1622 }
1623 let code = self.insert(text, check)?;
1624 self.collisions.entry(hash).or_default().push(code);
1625 return Ok(code);
1626 }
1627 let code = self.insert(text, check)?;
1628 self.primary.insert(hash, code);
1629 Ok(code)
1630 }
1631
1632 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1633 if self.demoted {
1634 return Err(Error::internal("a value was coded against a demoted dictionary"));
1635 }
1636 let code = u32::try_from(self.ends.len())
1637 .map_err(|_| invalid("global dictionary has too many values"))?;
1638 self.filling.extend_from_slice(text);
1639 self.ends.push(
1640 u32::try_from(self.filling.len())
1641 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1642 );
1643 self.checks.push(check);
1644 self.counts.push(0);
1645 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1646 self.seal();
1647 }
1648 Ok(code)
1649 }
1650
1651 fn seal(&mut self) {
1657 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1658 let bytes = std::mem::take(&mut self.filling);
1659 if at % self.stride == 0 {
1660 self.sample.push((at, bytes.clone()));
1661 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1662 self.stride *= 2;
1663 let stride = self.stride;
1664 self.sample.retain(|(at, _)| at % stride == 0);
1665 }
1666 }
1667 self.waiting.push((at, bytes));
1668 }
1669
1670 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1672 block_values(self.block_ends(at), bytes)
1673 }
1674
1675 fn block_ends(&self, at: usize) -> &[u32] {
1677 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1678 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1679 &self.ends[first..last]
1680 }
1681
1682 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1689 let Some(shape) = &self.shape else { return Vec::new() };
1690 let waiting = std::mem::take(&mut self.waiting);
1691 waiting
1692 .into_iter()
1693 .map(|(at, bytes)| Unencoded {
1694 column,
1695 at,
1696 ends: self.block_ends(at).to_vec(),
1697 bytes,
1698 shape: shape.clone(),
1699 })
1700 .collect()
1701 }
1702
1703 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1706 if at < self.encoded() || self.early.insert(at, block).is_some() {
1707 return Err(Error::internal("a dictionary block came back twice"));
1708 }
1709 while let Some(block) = self.early.remove(&self.encoded()) {
1710 self.push_block(block);
1711 }
1712 Ok(())
1713 }
1714
1715 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1717 self.blocks.push(bytes);
1718 self.grams.push(*grams);
1719 }
1720
1721 fn settle(&mut self) -> Result<()> {
1729 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1730 return Ok(());
1731 }
1732 self.settle_on_sample()
1733 }
1734
1735 fn settle_rest(&mut self) -> Result<()> {
1743 if self.shape.is_some() || self.sample.is_empty() {
1744 return Ok(());
1745 }
1746 self.settle_on_sample()
1747 }
1748
1749 fn settle_on_sample(&mut self) -> Result<()> {
1750 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1751 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1752 return Ok(());
1753 }
1754 let sample =
1755 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1756 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1757 self.settled = complete;
1758 Ok(())
1759 }
1760
1761 fn seal_rest(&mut self) {
1763 if !self.demoted && self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1767 self.seal();
1768 }
1769 }
1770
1771 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1774 let (block, bytes) = &self.waiting[at];
1775 let values = self.slices(*block, bytes);
1776 let encoded = match &self.shape {
1777 Some(shape) => string::encode_with(&values, shape)?,
1778 None => string::encode(&values)?,
1779 };
1780 Ok((encoded, block_grams(&values)))
1781 }
1782
1783 #[cfg(test)]
1785 fn finish_blocks(&mut self) -> Result<()> {
1786 self.seal_rest();
1787 let made = (0..self.waiting.len())
1788 .map(|at| self.encode_waiting(at))
1789 .collect::<Result<Vec<_>>>()?;
1790 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1791 if self.encoded() != at {
1792 return Err(Error::internal("a dictionary block was encoded out of order"));
1793 }
1794 self.push_block(block);
1795 }
1796 Ok(())
1797 }
1798
1799 fn decoded(&self, file: Option<&dyn rudb_io::File>) -> Result<(Vec<u8>, Vec<u64>)> {
1817 let count = self.placed.len() + self.blocks.len();
1818 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1819 return Err(invalid("global dictionary blocks do not cover its values"));
1820 }
1821 let mut bases = Vec::with_capacity(count);
1822 let mut total = 0_usize;
1823 for block in 0..count {
1824 bases.push(total as u64);
1825 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1826 total = total
1827 .checked_add(self.ends[last] as usize)
1828 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1829 }
1830 let mut flat = vec![0_u8; total];
1831 let mut outs = Vec::with_capacity(count);
1832 let mut rest = flat.as_mut_slice();
1833 for block in 0..count {
1834 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1835 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1836 outs.push((block, out));
1837 rest = after;
1838 }
1839 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1840 let mut stored = Vec::new();
1841 for (block, out) in run {
1842 let encoded = match self.placed.get(*block) {
1843 Some(place) => {
1844 let file = file.ok_or_else(|| {
1845 Error::internal("a written dictionary block has no file")
1846 })?;
1847 let length = usize::try_from(place.length).map_err(|_| {
1848 invalid("global dictionary block does not fit in memory")
1849 })?;
1850 stored.resize(length, 0);
1851 read_at(file, place.start, &mut stored)?;
1852 if checksum(&stored) != place.hash {
1853 return Err(invalid(
1854 "a global dictionary block did not read back as written",
1855 ));
1856 }
1857 stored.as_slice()
1858 }
1859 None => &self.blocks[*block - self.placed.len()],
1860 };
1861 let decoded = string::decode_flat(encoded)?;
1862 if decoded.bytes().len() != out.len() {
1863 return Err(invalid(
1864 "a global dictionary block is not the length its ends say",
1865 ));
1866 }
1867 out.copy_from_slice(decoded.bytes());
1868 }
1869 Ok(())
1870 };
1871 let workers = close_workers().min(count / 16).max(1);
1874 if workers <= 1 {
1875 one(&mut outs)?;
1876 } else {
1877 let per = count.div_ceil(workers);
1878 std::thread::scope(|scope| {
1879 outs.chunks_mut(per)
1880 .map(|run| scope.spawn(|| one(run)))
1881 .collect::<Vec<_>>()
1882 .into_iter()
1883 .try_for_each(|handle| {
1884 handle.join().map_err(|_| {
1885 Error::internal("a global dictionary decode worker panicked")
1886 })?
1887 })
1888 })?;
1889 }
1890 drop(outs);
1891 Ok((flat, bases))
1892 }
1893
1894 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1899 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1900 let Some(&end) = ends.get(code) else { return (0, 0) };
1901 let base = base as usize;
1902 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1903 (base + from, base + end as usize)
1904 }
1905
1906 fn ranked_with_values(&self, file: Option<&dyn rudb_io::File>) -> Result<RankedDictionary> {
1926 let (flat, bases) = self.decoded(file)?;
1927 let value = |code: u32| {
1928 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1929 flat.get(from..to).unwrap_or_default()
1930 };
1931 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1932 sort_by_value_across(&mut codes, value, close_workers());
1933 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1934 Ok((order, flat, bases))
1935 }
1936
1937 #[cfg(test)]
1938 fn ranked(&self, file: Option<&dyn rudb_io::File>) -> Result<Vec<(u64, u32)>> {
1939 self.ranked_with_values(file).map(|(order, _, _)| order)
1940 }
1941}
1942
1943#[derive(Debug)]
1951pub struct Writer {
1952 file: Box<dyn rudb_io::File>,
1955 at: u64,
1963 written_back: u64,
1965 table: Table,
1966 generation: u64,
1967 order: Vec<((u64, u64), (u64, u64))>,
1970 next_order: u64,
1971 dictionaries: Vec<Option<GlobalDictionary>>,
1972 coded: Arc<prepare::Coding>,
1975 gathers: Vec<Option<stats::Gather>>,
1981 lent: Option<Arc<Lent>>,
1984 pending: Vec<PendingChunk>,
1985 closed: Vec<Entry>,
1987 views: Vec<ViewEntry>,
1992 profile: Option<Arc<LoadProfile>>,
1998}
1999
2000#[derive(Debug)]
2008struct PendingChunk {
2009 order: (u64, u64),
2010 chunk: Chunk,
2011}
2012
2013#[derive(Debug, Clone, Copy)]
2019struct Part {
2020 order: (u64, u64),
2021 rows: usize,
2022 footprint: usize,
2023}
2024
2025impl Part {
2026 fn of(pending: &PendingChunk) -> Self {
2027 Self {
2028 order: pending.order,
2029 rows: pending.chunk.len(),
2030 footprint: pending.chunk.footprint(),
2031 }
2032 }
2033}
2034
2035#[derive(Debug)]
2041struct ColumnStripe {
2042 pages: Vec<Vec<u8>>,
2043 codes: Vec<Option<Vec<u32>>>,
2044 sieves: Vec<Option<Sieve>>,
2045 ranges: Vec<Range>,
2046}
2047
2048fn coded_type(ty: &LogicalType) -> bool {
2056 matches!(ty, LogicalType::Varchar | LogicalType::Blob)
2057}
2058
2059fn dictionary_tag(ty: &LogicalType) -> u8 {
2066 if ty == &LogicalType::Blob { 2 } else { 1 }
2067}
2068
2069fn weight(ty: &LogicalType) -> usize {
2077 match ty {
2078 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
2079 LogicalType::HugeInt
2080 | LogicalType::UHugeInt
2081 | LogicalType::Uuid
2082 | LogicalType::Interval => 16,
2083 LogicalType::BigInt
2084 | LogicalType::UBigInt
2085 | LogicalType::Timestamp
2086 | LogicalType::Time
2087 | LogicalType::TimeTz
2088 | LogicalType::TimestampTz
2089 | LogicalType::TimestampS
2090 | LogicalType::TimestampMs
2091 | LogicalType::TimestampNs
2092 | LogicalType::Double
2093 | LogicalType::Decimal { .. } => 8,
2094 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
2095 LogicalType::SmallInt | LogicalType::USmallInt => 2,
2096 _ => 1,
2097 }
2098}
2099
2100pub const STRIPE_PARTS: usize = 64;
2107
2108const DICTIONARY_DECIDE_ROWS: usize = 4_096;
2116
2117const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
2133
2134const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
2136
2137fn index_section(parts: usize) -> Result<usize> {
2139 parts
2140 .checked_mul(INDEX_ENTRY)
2141 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
2142 .ok_or_else(|| invalid("index page length overflow"))
2143}
2144
2145impl Writer {
2146 pub fn open(
2165 path: impl AsRef<Path>,
2166 name: impl Into<String>,
2167 fields: Vec<Field>,
2168 ) -> Result<Self> {
2169 Self::open_in(&RealFilesystem::new(), path, name, fields)
2170 }
2171
2172 pub fn open_in(
2179 fs: &dyn Filesystem,
2180 path: impl AsRef<Path>,
2181 name: impl Into<String>,
2182 fields: Vec<Field>,
2183 ) -> Result<Self> {
2184 for field in &fields {
2185 type_tag(&field.ty)?;
2186 }
2187 let name = name.into();
2188 let file = fs.open(path.as_ref(), OpenMode::ReadWrite)?;
2189 let size = file.len()?;
2190 let (slot, bytes, _) = committed_slot(&*file, size)?;
2191 let (mut closed, views) = decode_catalog(&bytes, size)?;
2192 if let Some(at) = closed.iter().position(|held| held.name == name) {
2203 if closed[at].rows > 0 {
2204 return Err(invalid("two tables in one native file have the same name"));
2205 }
2206 closed.remove(at);
2207 }
2208 let generation = slot
2213 .generation
2214 .checked_add(1)
2215 .ok_or_else(|| invalid("native file generation overflow"))?;
2216 Ok(Self {
2217 file,
2218 at: size,
2221 written_back: size,
2222 dictionaries: fields
2223 .iter()
2224 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2225 .collect(),
2226 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2227 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2228 lent: None,
2229 table: Table {
2230 name,
2231 dictionaries: vec![None; fields.len()],
2232 dictionary_payloads: Vec::new(),
2233 demoted: Vec::new(),
2234 distincts: vec![None; fields.len()],
2235 fields,
2236 stripes: Vec::new(),
2237 rows: 0,
2238 frequencies: Vec::new(),
2239 pair_frequencies: Vec::new(),
2240 frequency_texts: Vec::new(),
2241 host_groups: None,
2242 clustering: None,
2243 generation,
2244 sections: Vec::new(),
2245 },
2246 generation,
2247 order: Vec::new(),
2248 next_order: 0,
2249 pending: Vec::with_capacity(STRIPE_PARTS),
2250 closed,
2251 views,
2252 profile: None,
2253 })
2254 }
2255
2256 pub fn create(
2262 path: impl AsRef<Path>,
2263 name: impl Into<String>,
2264 fields: Vec<Field>,
2265 ) -> Result<Self> {
2266 Self::create_in(&RealFilesystem::new(), path, name, fields)
2267 }
2268
2269 pub fn create_in(
2279 fs: &dyn Filesystem,
2280 path: impl AsRef<Path>,
2281 name: impl Into<String>,
2282 fields: Vec<Field>,
2283 ) -> Result<Self> {
2284 for field in &fields {
2285 type_tag(&field.ty)?;
2286 }
2287 let file = fs.open(path.as_ref(), OpenMode::CreateNew)?;
2288 let mut header = [0; HEADER as usize];
2289 header[..8].copy_from_slice(MAGIC);
2290 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2291 file.write_at(0, &header)?;
2292 Ok(Self {
2293 file,
2294 at: HEADER,
2295 written_back: HEADER,
2296 dictionaries: fields
2297 .iter()
2298 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2299 .collect(),
2300 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2301 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
2302 lent: None,
2303 table: Table {
2304 name: name.into(),
2305 dictionaries: vec![None; fields.len()],
2306 dictionary_payloads: Vec::new(),
2307 demoted: Vec::new(),
2308 distincts: vec![None; fields.len()],
2309 fields,
2310 stripes: Vec::new(),
2311 rows: 0,
2312 frequencies: Vec::new(),
2313 pair_frequencies: Vec::new(),
2314 frequency_texts: Vec::new(),
2315 host_groups: None,
2316 clustering: None,
2317 generation: 1,
2318 sections: Vec::new(),
2319 },
2320 generation: 1,
2321 order: Vec::new(),
2322 next_order: 0,
2323 pending: Vec::with_capacity(STRIPE_PARTS),
2324 closed: Vec::new(),
2325 views: Vec::new(),
2326 profile: None,
2327 })
2328 }
2329
2330 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2352 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::CreateNew)?;
2353 let mut header = [0; HEADER as usize];
2354 header[..8].copy_from_slice(MAGIC);
2355 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2356 file.write_at(0, &header)?;
2357 let catalog = encode_catalog(&[], views)?;
2358 file.write_at(HEADER, &catalog)?;
2359 file.sync()?;
2363 let slot = Slot {
2364 offset: HEADER,
2365 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2366 generation: 1,
2367 hash: checksum(&catalog),
2368 };
2369 file.write_at(slot_offset(1), &slot.bytes())?;
2370 file.sync()?;
2371 Ok(())
2372 }
2373
2374 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2385 for field in &fields {
2386 type_tag(&field.ty)?;
2387 }
2388 let name = name.into();
2389 let entry = self.close()?;
2390 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2391 return Err(invalid("two tables in one native file have the same name"));
2392 }
2393 let Self { file, at, generation, mut closed, views, .. } = self;
2394 closed.push(entry);
2395 Ok(Self {
2396 file,
2397 written_back: at,
2398 at,
2399 generation,
2400 closed,
2401 views,
2402 profile: None,
2403 dictionaries: fields
2404 .iter()
2405 .map(|field| coded_type(&field.ty).then(GlobalDictionary::new))
2406 .collect(),
2407 coded: Arc::new(prepare::Coding::new(fields.iter().map(|field| coded_type(&field.ty)))),
2408 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2409 lent: None,
2410 table: Table {
2411 name,
2412 dictionaries: vec![None; fields.len()],
2413 dictionary_payloads: Vec::new(),
2414 demoted: Vec::new(),
2415 distincts: vec![None; fields.len()],
2416 fields,
2417 stripes: Vec::new(),
2418 rows: 0,
2419 frequencies: Vec::new(),
2420 pair_frequencies: Vec::new(),
2421 frequency_texts: Vec::new(),
2422 host_groups: None,
2423 clustering: None,
2424 generation,
2425 sections: Vec::new(),
2426 },
2427 order: Vec::new(),
2428 next_order: 0,
2429 pending: Vec::with_capacity(STRIPE_PARTS),
2430 })
2431 }
2432
2433 #[must_use]
2443 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2444 self.views = views;
2445 self
2446 }
2447
2448 #[must_use]
2454 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2455 self.profile = Some(profile);
2456 self
2457 }
2458
2459 #[must_use]
2463 pub fn with_dictionary_cap(self, bytes: u64) -> Self {
2464 self.coded.cap(bytes);
2465 self
2466 }
2467
2468 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2483 self.table.clustering = Some(Clustering::new(
2486 clustering.columns().to_vec(),
2487 clustering.width(),
2488 &self.table.fields,
2489 )?);
2490 Ok(self)
2491 }
2492
2493 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2498 self.file.write_at(self.at, bytes)?;
2499 self.at = self
2500 .at
2501 .checked_add(bytes.len() as u64)
2502 .ok_or_else(|| invalid("native file length overflow"))?;
2503 if self.at - self.written_back >= WRITEBACK_STRETCH {
2504 self.file.start_writeback(self.written_back, self.at - self.written_back);
2505 self.written_back = self.at;
2506 }
2507 Ok(())
2508 }
2509
2510 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2516 let order = (self.next_order, 0);
2517 self.next_order = self.next_order.saturating_add(1);
2518 self.append_at(order, chunk)
2519 }
2520
2521 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2532 if chunk.is_empty() {
2533 return Ok(());
2534 }
2535 self.admit(chunk)?;
2536 if self.pending.last().is_some_and(|last| last.order > order) {
2537 self.flush_pending()?;
2538 }
2539 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2544 if self.pending.len() == STRIPE_PARTS {
2545 self.flush_pending()?;
2546 }
2547 Ok(())
2548 }
2549
2550 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2566 if parts.len() > STRIPE_PARTS {
2567 return Err(invalid("a stripe was handed more parts than it holds"));
2568 }
2569 self.flush_pending()?;
2572 for (order, chunk) in parts {
2573 if chunk.is_empty() {
2574 continue;
2575 }
2576 self.admit(&chunk)?;
2577 self.pending.push(PendingChunk { order, chunk });
2578 }
2579 self.flush_pending()
2580 }
2581
2582 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2584 if chunk.width() != self.table.fields.len() {
2585 return Err(invalid("chunk width differs from table schema"));
2586 }
2587 for (index, field) in self.table.fields.iter().enumerate() {
2588 if chunk.column(index)?.logical_type() != &field.ty {
2589 return Err(invalid("chunk type differs from table schema"));
2590 }
2591 }
2592 self.table.rows = self
2593 .table
2594 .rows
2595 .checked_add(chunk.len())
2596 .ok_or_else(|| invalid("row count overflow"))?;
2597 Ok(())
2598 }
2599
2600 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2602 let mut stripe = ColumnStripe {
2603 pages: Vec::with_capacity(columns.len()),
2604 codes: Vec::with_capacity(columns.len()),
2605 sieves: Vec::with_capacity(columns.len()),
2606 ranges: Vec::with_capacity(columns.len()),
2607 };
2608 let mut settling = Settling::default();
2609 for &column in columns {
2610 let bytes = encode(column, &mut settling)?;
2611 if bytes.len() > MAX_PAGE {
2612 return Err(invalid("column page exceeds the configured bound"));
2613 }
2614 let range = Range::of(column);
2617 let sieve =
2628 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2629 stripe.pages.push(bytes);
2630 stripe.codes.push(None);
2631 stripe.sieves.push(sieve);
2632 stripe.ranges.push(range);
2633 }
2634 Ok(stripe)
2635 }
2636
2637 fn place_blocks(&mut self) -> Result<()> {
2642 if let Some(lent) = self.lent.clone() {
2643 return self.place_lent_blocks(&lent);
2644 }
2645 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2646 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2647 for block in std::mem::take(&mut dictionary.blocks) {
2648 let start = self.at;
2649 self.put(&block)?;
2650 dictionary.placed.push(Placed {
2651 start,
2652 length: block.len() as u64,
2653 hash: checksum(&block),
2654 });
2655 }
2656 Ok(())
2657 });
2658 self.dictionaries = dictionaries;
2659 placed
2660 }
2661
2662 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2668 for column in lent.columns() {
2669 let Ok(mut held) = column.try_lock() else { continue };
2670 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2671 for block in std::mem::take(&mut dictionary.blocks) {
2672 let start = self.at;
2673 self.put(&block)?;
2674 dictionary.placed.push(Placed {
2675 start,
2676 length: block.len() as u64,
2677 hash: checksum(&block),
2678 });
2679 }
2680 }
2681 Ok(())
2682 }
2683
2684 fn reclaim(&mut self) -> Result<()> {
2688 let Some(lent) = self.lent.take() else { return Ok(()) };
2689 let (dictionaries, gathers) = lent.reclaim()?;
2690 self.dictionaries = dictionaries;
2691 self.gathers = gathers;
2692 Ok(())
2693 }
2694
2695 fn flush_pending(&mut self) -> Result<()> {
2700 if self.pending.is_empty() {
2701 return Ok(());
2702 }
2703 let held = std::mem::take(&mut self.pending);
2704 let prepared = self.preparer().prepare_held(held)?;
2705 let merged = self.merge_held(prepared)?;
2706 let paged = merged.pages()?;
2707 self.write_paged(paged)
2708 }
2709
2710 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2712 let width = self.table.fields.len();
2713 let parts = held.len();
2714 if encoded.len() != width {
2715 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2716 }
2717 let profile = self.profile.clone();
2718 if let Some(profile) = &profile {
2719 let rows = held.iter().map(|part| part.rows as u64).sum();
2720 let raw = held.iter().map(|part| part.footprint as u64).sum();
2721 let pages =
2722 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2723 profile.moved(Stage::Pages, raw, pages, rows);
2724 }
2725 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2728 let before = self.at;
2729 self.place_blocks()?;
2730 drop(timing);
2731 if let Some(profile) = &profile {
2732 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2733 }
2734 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2735 let before = self.at;
2736 let mut pages = Vec::with_capacity(width);
2737 let mut memberships = vec![None; width];
2738 let mut ranges = Vec::with_capacity(width);
2739 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2740 for stripe in &encoded {
2741 let offset = self.at;
2742 let section = index.len();
2743 let mut length = 0_usize;
2744 for bytes in &stripe.pages {
2745 self.file.write_at(self.at + length as u64, bytes)?;
2746 put_u32(
2747 &mut index,
2748 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2749 );
2750 put_u64(&mut index, checksum(bytes));
2751 length = length
2752 .checked_add(bytes.len())
2753 .ok_or_else(|| invalid("column page length overflow"))?;
2754 }
2755 let hash = checksum(&index[section..]);
2756 put_u64(&mut index, hash);
2757 if length > MAX_PAGE {
2758 return Err(invalid("column page exceeds the configured bound"));
2759 }
2760 self.at = self
2761 .at
2762 .checked_add(length as u64)
2763 .ok_or_else(|| invalid("native file length overflow"))?;
2764 pages.push(Span {
2765 offset,
2766 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2767 });
2768 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2769 }
2770 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2771 if stripe.codes.iter().all(Option::is_none) {
2772 continue;
2773 }
2774 let lists = stripe
2775 .codes
2776 .iter()
2777 .map(|codes| codes.clone().unwrap_or_default())
2778 .collect::<Vec<_>>();
2779 let bytes = encode_membership(&merged_codes(lists));
2780 let offset = self.at;
2781 self.put(&bytes)?;
2782 *membership = Some(Page {
2783 offset,
2784 length: u32::try_from(bytes.len())
2785 .map_err(|_| invalid("membership page length overflow"))?,
2786 hash: checksum(&bytes),
2787 });
2788 }
2789 let mut sieves = vec![None; width];
2790 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2791 if stripe.sieves.iter().all(Option::is_none) {
2792 continue;
2793 }
2794 let bytes = encode_sieves(stripe.sieves.iter())?;
2795 let offset = self.at;
2796 self.put(&bytes)?;
2797 *page = Some(Page {
2798 offset,
2799 length: u32::try_from(bytes.len())
2800 .map_err(|_| invalid("sieve page length overflow"))?,
2801 hash: checksum(&bytes),
2802 });
2803 }
2804 let mut part_ranges = vec![None; width];
2810 if parts > 1 {
2811 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2812 let bytes = encode_part_ranges(&stripe.ranges)?;
2813 if bytes.len() >= span.length as usize {
2814 continue;
2815 }
2816 let offset = self.at;
2817 self.put(&bytes)?;
2818 *page = Some(Page {
2819 offset,
2820 length: u32::try_from(bytes.len())
2821 .map_err(|_| invalid("part range page length overflow"))?,
2822 hash: checksum(&bytes),
2823 });
2824 }
2825 }
2826 let offset = self.at;
2827 self.put(&index)?;
2828 let index = Span {
2829 offset,
2830 length: u32::try_from(index.len())
2831 .map_err(|_| invalid("index page length overflow"))?,
2832 };
2833 let mut rows = 0_usize;
2834 let mut lengths = Vec::with_capacity(parts);
2835 let mut span = None;
2836 for part in held {
2837 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2838 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2839 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2840 }
2841 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2842 self.table.stripes.push(Stripe {
2843 rows,
2844 parts: lengths,
2845 index,
2846 pages,
2847 memberships: Pages::from_slots(memberships)?,
2848 sieves: Pages::from_slots(sieves)?,
2849 part_ranges: Pages::from_slots(part_ranges)?,
2850 zone: Zone::from_ranges(ranges),
2851 });
2852 drop(timing);
2853 if let Some(profile) = &profile {
2854 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2855 }
2856 Ok(())
2857 }
2858
2859 fn numeric_frequency(
2879 &self,
2880 column: usize,
2881 counted: bool,
2882 ) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2883 let signed = match self.table.fields[column].ty {
2884 LogicalType::TinyInt
2885 | LogicalType::SmallInt
2886 | LogicalType::Integer
2887 | LogicalType::BigInt
2888 | LogicalType::Date
2889 | LogicalType::Timestamp => true,
2890 LogicalType::UTinyInt
2891 | LogicalType::USmallInt
2892 | LogicalType::UInteger
2893 | LogicalType::UBigInt => false,
2894 _ => return Ok((None, None)),
2895 };
2896 let value_of = |bits: Option<u64>| match bits {
2897 None => FrequencyValue::Null,
2898 Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2899 Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2900 };
2901 let tallied = self
2906 .gathers
2907 .get(column)
2908 .and_then(Option::as_ref)
2909 .filter(|gather| gather.rows() == self.table.rows as u64)
2910 .and_then(stats::Gather::frequencies)
2911 .and_then(|(values, nulls)| {
2912 let exact = values
2913 .iter()
2914 .map(|(value, count)| Some((frequency_bits(value)?, *count)))
2915 .collect::<Option<FrequencyMap<_>>>()?;
2916 Some((exact, (nulls != 0).then_some(nulls), values.len() as u64))
2917 });
2918 let (exact, null_count, decrements, distinct_count) = match tallied {
2919 Some((exact, null_count, distinct)) => (exact, null_count, 0, Some(distinct)),
2920 None => {
2921 let mut first = Candidates::default();
2932 let mut distinct: Option<distinct::ExactDistinct> = None;
2933 let mut run = Run::default();
2934 self.visit_numeric(column, signed, |_, bits| {
2935 if let Some((ended, times)) = run.push(bits) {
2936 count_from_full(&first, &mut distinct, ended, counted);
2937 first.add(ended, times);
2938 }
2939 if run.times == 1 {
2940 if let (Some(distinct), Some(bits)) = (distinct.as_mut(), bits) {
2941 distinct.insert(bits);
2942 }
2943 }
2944 })?;
2945 if let Some((bits, times)) = run.take() {
2946 count_from_full(&first, &mut distinct, bits, counted);
2947 first.add(bits, times);
2948 }
2949 let distinct_count = match distinct.as_mut() {
2950 Some(distinct) => distinct.count(),
2951 None => Some(first.held as u64),
2952 };
2953 let (nulls, decrements) = (first.nulls, first.decrements);
2954 let (exact, null_count) = if decrements == 0 {
2955 let exact = first
2956 .pairs()
2957 .map(|(bits, count)| (bits, u64::from(count)))
2958 .collect::<FrequencyMap<_>>();
2959 (exact, (nulls != 0).then_some(u64::from(nulls)))
2960 } else {
2961 let mut lower = first.pairs().map(|(_, count)| count).collect::<Vec<_>>();
2962 if nulls != 0 {
2963 lower.push(nulls);
2964 }
2965 lower.sort_unstable_by(|left, right| right.cmp(left));
2966 if lower.len() < FREQUENCY_BUILD_RANK
2967 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2968 {
2969 return Ok((None, distinct_count));
2970 }
2971 let mut recounts = vec![0_u64; first.slots.len()];
2974 let mut null_count = (nulls != 0).then_some(0_u64);
2975 let mut recount = |bits: Option<u64>, times: u32| {
2976 let held = match bits {
2977 Some(bits) => first.position(bits).map(|at| &mut recounts[at]),
2978 None => null_count.as_mut(),
2979 };
2980 if let Some(count) = held {
2981 *count = count.saturating_add(u64::from(times));
2982 }
2983 };
2984 let mut run = Run::default();
2985 self.visit_numeric(column, signed, |_, bits| {
2986 if let Some((bits, times)) = run.push(bits) {
2987 recount(bits, times);
2988 }
2989 })?;
2990 if let Some((bits, times)) = run.take() {
2991 recount(bits, times);
2992 }
2993 let exact = first
2994 .slots
2995 .iter()
2996 .zip(&recounts)
2997 .filter(|(slot, _)| slot.count != 0)
2998 .map(|(slot, &count)| (slot.bits, count))
2999 .collect::<FrequencyMap<_>>();
3000 (exact, null_count)
3001 };
3002 (exact, null_count, decrements, distinct_count)
3003 }
3004 };
3005 let mut entries = exact
3006 .into_iter()
3007 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
3008 .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
3009 .collect::<Vec<_>>();
3010 let omitted_max = keep_most_frequent(&mut entries).max(decrements);
3011 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
3012 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
3013 });
3014 let mut ordinals = Vec::new();
3015 let mut ordinal_entries = Vec::new();
3016 if let Some(kept_rows) = kept_rows {
3017 let mut kept = FrequencyMap::default();
3018 let mut null_kept = None;
3019 for (at, entry) in entries.iter().enumerate() {
3020 let at = u16::try_from(at)
3021 .map_err(|_| invalid("too many retained frequency entries"))?;
3022 match entry.value {
3023 FrequencyValue::Integer(value) => {
3024 kept.insert(value as u64, at);
3025 }
3026 FrequencyValue::Null => null_kept = Some(at),
3027 FrequencyValue::Code(_) => {}
3028 }
3029 }
3030 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3031 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
3032 self.visit_numeric(column, signed, |ordinal, bits| {
3033 let held = match bits {
3034 Some(bits) => kept.get(&bits).copied(),
3035 None => null_kept,
3036 };
3037 if let Some(entry) = held {
3038 ordinals.push(ordinal);
3039 ordinal_entries.push(entry);
3040 }
3041 })?;
3042 }
3043 Ok((
3044 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
3045 distinct_count,
3046 ))
3047 }
3048
3049 fn visit_numeric(
3056 &self,
3057 column: usize,
3058 signed: bool,
3059 mut visit: impl FnMut(u64, Option<u64>),
3060 ) -> Result<()> {
3061 let ty = &self.table.fields[column].ty;
3062 let mut start = 0_u64;
3063 let mut block = Vec::new();
3064 for stripe in &self.table.stripes {
3065 let spans = read_index(&self.file, stripe, column)?;
3066 let page = stripe.pages[column];
3067 let mut bytes = vec![0; page.length as usize];
3068 read_at(&self.file, page.offset, &mut bytes)?;
3069 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3070 let part = part_bytes(&bytes, *span)?;
3071 if checksum(part) != span.hash {
3072 return Err(invalid("column page checksum differs while building frequencies"));
3073 }
3074 let rows = rows as usize;
3075 let vector = decode(ty, rows, part, None)?;
3076 if signed && vector.signed_block(&mut block) && block.len() == rows {
3080 if vector.none_null() {
3081 for (row, &value) in block.iter().enumerate() {
3082 visit(start.saturating_add(row as u64), Some(value as u64));
3083 }
3084 } else {
3085 for (row, &value) in block.iter().enumerate() {
3086 let bits = (!vector.is_null_at(row)).then_some(value as u64);
3087 visit(start.saturating_add(row as u64), bits);
3088 }
3089 }
3090 start = start.saturating_add(rows as u64);
3091 continue;
3092 }
3093 for row in 0..rows {
3095 let bits = if vector.is_null_at(row) {
3096 None
3097 } else {
3098 let widened = match vector.signed_at(row) {
3102 Some(value) => Some(value as u64),
3103 None => match vector.value_at(row) {
3104 Value::UTinyInt(value) => Some(u64::from(value)),
3105 Value::USmallInt(value) => Some(u64::from(value)),
3106 Value::UInteger(value) => Some(u64::from(value)),
3107 Value::UBigInt(value) => Some(value),
3108 _ => None,
3109 },
3110 };
3111 Some(widened.ok_or_else(|| {
3112 invalid("numeric frequency page did not contain an integer value")
3113 })?)
3114 };
3115 visit(start.saturating_add(row as u64), bits);
3116 }
3117 start = start.saturating_add(rows as u64);
3118 }
3119 }
3120 Ok(())
3121 }
3122
3123 fn numeric_columns(&self) -> Vec<usize> {
3125 self.table
3126 .fields
3127 .iter()
3128 .enumerate()
3129 .filter_map(|(column, field)| {
3130 matches!(
3131 field.ty,
3132 LogicalType::TinyInt
3133 | LogicalType::SmallInt
3134 | LogicalType::Integer
3135 | LogicalType::BigInt
3136 | LogicalType::UTinyInt
3137 | LogicalType::USmallInt
3138 | LogicalType::UInteger
3139 | LogicalType::UBigInt
3140 | LogicalType::Date
3141 | LogicalType::Timestamp
3142 )
3143 .then_some(column)
3144 })
3145 .collect()
3146 }
3147
3148 #[allow(dead_code)]
3150 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
3151 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
3152 return Ok(None);
3153 }
3154 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
3155 return Err(invalid("frequency ordinals are not sorted and unique"));
3156 }
3157 let mut out = Vec::with_capacity(ordinals.len());
3158 let mut wanted = 0;
3159 let mut stripe_start = 0_u64;
3160 for stripe in &self.table.stripes {
3161 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
3162 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
3163 stripe_start = stripe_end;
3164 continue;
3165 }
3166 let spans = read_index(&self.file, stripe, column)?;
3167 let page = stripe.pages[column];
3168 let mut bytes = vec![0; page.length as usize];
3169 read_at(&self.file, page.offset, &mut bytes)?;
3170 let mut part_start = stripe_start;
3171 for (span, &rows) in spans.iter().zip(&stripe.parts) {
3172 let part_end = part_start.saturating_add(u64::from(rows));
3173 if wanted < ordinals.len() && ordinals[wanted] < part_end {
3174 let part = part_bytes(&bytes, *span)?;
3175 if checksum(part) != span.hash {
3176 return Err(invalid(
3177 "column page checksum differs while building pair frequencies",
3178 ));
3179 }
3180 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
3181 let positions = ordinals[wanted..upto]
3182 .iter()
3183 .map(|&ordinal| {
3184 usize::try_from(ordinal.saturating_sub(part_start))
3185 .map_err(|_| invalid("frequency row offset does not fit in memory"))
3186 })
3187 .collect::<Result<Vec<_>>>()?;
3188 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
3189 return Ok(None);
3190 }
3191 wanted = upto;
3192 }
3193 part_start = part_end;
3194 }
3195 stripe_start = stripe_end;
3196 }
3197 if wanted != ordinals.len() {
3198 return Err(invalid("frequency ordinal is outside the table"));
3199 }
3200 Ok(Some(out))
3201 }
3202
3203 #[allow(dead_code)]
3205 fn pair_frequencies(
3206 &self,
3207 frequencies: &[Option<Frequencies>],
3208 ) -> Result<Vec<PairFrequencySummary>> {
3209 let anchors = frequencies
3210 .iter()
3211 .enumerate()
3212 .filter_map(|(column, summary)| {
3213 match summary {
3215 Some(Frequencies::Held(summary)) => Some(summary),
3216 _ => None,
3217 }
3218 .filter(|summary| {
3219 !summary.ordinals.is_empty()
3220 && summary.ordinal_entries.len() == summary.ordinals.len()
3221 })
3222 .cloned()
3223 .map(|summary| (column, summary))
3224 })
3225 .collect::<Vec<_>>();
3226 let strings = self
3227 .dictionaries
3228 .iter()
3229 .enumerate()
3230 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
3231 .collect::<Vec<_>>();
3232 let mut summaries = Vec::new();
3233 for (first, anchors) in anchors {
3234 for &second in &strings {
3235 if summaries.len() == MAX_PAIR_FREQUENCIES {
3236 return Ok(summaries);
3237 }
3238 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
3239 continue;
3240 };
3241 if codes.len() != anchors.ordinal_entries.len() {
3242 return Err(invalid("pair frequency columns have different lengths"));
3243 }
3244 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
3245 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
3246 *counts.entry((anchor, code)).or_default() += 1;
3247 }
3248 let mut entries = counts
3249 .into_iter()
3250 .map(|((first_entry, second), count)| PairFrequencyEntry {
3251 first_entry,
3252 second,
3253 count,
3254 })
3255 .collect::<Vec<_>>();
3256 entries.sort_unstable_by(|left, right| {
3257 right
3258 .count
3259 .cmp(&left.count)
3260 .then_with(|| left.first_entry.cmp(&right.first_entry))
3261 .then_with(|| left.second.cmp(&right.second))
3262 });
3263 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
3264 entries.truncate(FREQUENCY_ENTRIES);
3265 summaries.push(PairFrequencySummary {
3266 first: u16::try_from(first)
3267 .map_err(|_| invalid("pair frequency column index overflows"))?,
3268 second: u16::try_from(second)
3269 .map_err(|_| invalid("pair frequency column index overflows"))?,
3270 entries,
3271 omitted_max: anchors.omitted_max.max(pair_omitted),
3272 });
3273 }
3274 }
3275 Ok(summaries)
3276 }
3277
3278 fn close(&mut self) -> Result<Entry> {
3289 self.reclaim()?;
3290 self.flush_pending()?;
3291 let profile = self.profile.clone();
3295 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3296 let before = self.at;
3297 let mut stripes = std::mem::take(&mut self.order)
3298 .into_iter()
3299 .zip(std::mem::take(&mut self.table.stripes))
3300 .collect::<Vec<_>>();
3301 stripes.sort_by_key(|(order, _)| order.0);
3302 let mut previous: Option<(u64, u64)> = None;
3303 for ((first, last), _) in &stripes {
3304 if previous.is_some_and(|previous| previous >= *first) {
3305 return Err(invalid("chunks did not arrive in source order"));
3306 }
3307 previous = Some(*last);
3308 }
3309 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
3310 drop(timing);
3311 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
3312 let placing = self.at;
3313 finish_dictionaries(&mut self.dictionaries)?;
3314 self.place_blocks()?;
3315 for dictionary in self.dictionaries.iter_mut().flatten() {
3316 dictionary.release_lookup();
3317 dictionary.recharge(profile.as_deref());
3318 }
3319 let (numeric, closed) = self.close_columns()?;
3320 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
3321 numeric.into_iter().unzip();
3322 let frequencies =
3323 frequencies.into_iter().map(|held| held.map(Frequencies::Held)).collect::<Vec<_>>();
3324 let pairs = Vec::new();
3326 self.table.frequencies = frequencies;
3327 self.table.distincts = distincts;
3328 self.table.pair_frequencies = pairs;
3329 if let Some(profile) = &profile {
3330 profile.release(self.dictionaries.iter().flatten().map(|held| held.charged).sum());
3331 }
3332 self.table.demoted = self
3333 .dictionaries
3334 .iter()
3335 .map(|dictionary| dictionary.as_ref().is_some_and(|held| held.demoted))
3336 .collect();
3337 if !self.table.demoted.contains(&true) {
3338 self.table.demoted = Vec::new();
3339 }
3340 self.dictionaries = Vec::new();
3341 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3342 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3343 self.table.host_groups = None;
3344 for (index, closed) in closed.into_iter().enumerate() {
3345 let Some(closed) = closed else { continue };
3346 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3347 self.table.distincts[index] = distinct;
3348 self.table.frequencies[index] = frequencies.map(Frequencies::Held);
3349 self.table.frequency_texts[index] = texts;
3350 if hosts.is_some() {
3351 self.table.host_groups = hosts;
3352 }
3353 let offset = self.at;
3354 self.put(&encoded.index)?;
3355 self.put(&encoded.ranks)?;
3356 self.put(&encoded.grams)?;
3357 self.table.dictionary_payloads[index] = payload;
3358 let length = encoded
3359 .index
3360 .len()
3361 .checked_add(encoded.ranks.len())
3362 .and_then(|len| len.checked_add(encoded.grams.len()))
3363 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3364 self.table.dictionaries[index] = Some(Page {
3365 offset,
3366 length: u32::try_from(length)
3367 .map_err(|_| invalid("dictionary page length overflow"))?,
3368 hash: checksum(&encoded.index),
3369 });
3370 }
3371 drop(timing);
3372 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3373 let placed = self.at - placing;
3374 self.write_stats()?;
3375 let directory = encode_directory(&self.table)?;
3376 if directory.len() > MAX_DIRECTORY {
3377 return Err(invalid("directory exceeds the configured bound"));
3378 }
3379 let offset = self.at;
3380 self.put(&directory)?;
3381 drop(timing);
3382 if let Some(profile) = &profile {
3383 profile.moved(Stage::Dictionary, 0, placed, 0);
3384 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3385 }
3386 Ok(Entry {
3387 name: self.table.name.clone(),
3388 fields: self.table.fields.clone(),
3389 rows: self.table.rows,
3390 nonzero: vec![None; self.table.fields.len()],
3391 aggregates: table_aggregate_sums(&self.table),
3392 distincts: self.table.distincts.clone(),
3393 extremes: table_integer_extremes(&self.table),
3394 frequencies: table_complete_numeric_frequencies(&self.table),
3395 directory: Page {
3396 offset,
3397 length: u32::try_from(directory.len())
3398 .map_err(|_| invalid("directory length overflow"))?,
3399 hash: checksum(&directory),
3400 },
3401 })
3402 }
3403
3404 #[allow(clippy::type_complexity)]
3421 fn close_columns(
3422 &self,
3423 ) -> Result<(Vec<(Option<FrequencySummary>, Option<u64>)>, Vec<Option<ClosedDictionary>>)> {
3424 let numeric = self.numeric_columns().into_iter().map(|column| {
3425 let estimate =
3426 self.gathers.get(column).and_then(Option::as_ref).and_then(stats::Gather::distinct);
3427 let counted = !estimate.is_some_and(distinct::beyond);
3428 let set =
3429 if counted { distinct::bytes_for(estimate.unwrap_or(f64::INFINITY)) } else { 0 };
3430 let cost = self.table.rows.saturating_mul(weight(&self.table.fields[column].ty));
3431 (Closing::Numeric { column, counted }, NUMERIC_CLOSE_BYTES + set, cost)
3432 });
3433 let dictionaries =
3434 self.dictionaries.iter().enumerate().filter_map(|(index, dictionary)| {
3435 let dictionary = dictionary.as_ref()?;
3436 let bytes = dictionary.closing_bytes();
3437 Some((Closing::Dictionary { index, dictionary }, bytes, bytes))
3438 });
3439 let mut jobs = numeric.chain(dictionaries).collect::<Vec<_>>();
3440 jobs.sort_by_key(|&(_, _, cost)| cost);
3441 let columns = self.table.fields.len();
3442 let mut frequencies = vec![(None, None); columns];
3443 let mut closed = (0..columns).map(|_| None).collect::<Vec<_>>();
3444 let profile = self.profile.as_deref();
3445 let run = |job: Closing<'_>, bytes: usize| -> Result<Closed> {
3446 let _holding = profile.map(|profile| profile.holding(bytes as u64));
3447 match job {
3448 Closing::Numeric { column, counted } => {
3449 let _timing = profile.map(|profile| profile.span(Stage::Publish));
3450 Ok(Closed::Numeric(column, self.numeric_frequency(column, counted)?))
3451 }
3452 Closing::Dictionary { index, dictionary } => {
3453 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3454 Ok(Closed::Dictionary(index, self.close_dictionary(index, dictionary)?))
3455 }
3456 }
3457 };
3458 let workers = close_workers().min(jobs.len());
3459 let pieces = if workers <= 1 {
3460 jobs.into_iter().map(|(job, bytes, _)| run(job, bytes)).collect::<Result<Vec<_>>>()?
3461 } else {
3462 let state = Mutex::new((jobs, 0_usize));
3464 let finished = Condvar::new();
3465 std::thread::scope(|scope| {
3466 (0..workers)
3467 .map(|_| {
3468 scope.spawn(|| {
3469 let mut mine = Vec::new();
3470 loop {
3471 let mut held = state.lock().map_err(|_| {
3472 Error::internal("a native close worker panicked")
3473 })?;
3474 let (job, bytes) = loop {
3475 let (jobs, busy) = &mut *held;
3476 if jobs.is_empty() {
3477 return Ok(mine);
3478 }
3479 let fits = jobs.iter().rposition(|&(_, bytes, _)| {
3480 *busy == 0 || busy.saturating_add(bytes) <= CLOSE_BYTES
3481 });
3482 if let Some(at) = fits {
3483 let (job, bytes, _) = jobs.remove(at);
3484 *busy += bytes;
3485 break (job, bytes);
3486 }
3487 held = finished.wait(held).map_err(|_| {
3488 Error::internal("a native close worker panicked")
3489 })?;
3490 };
3491 drop(held);
3492 let _room = Room { state: &state, finished: &finished, bytes };
3495 mine.push(run(job, bytes)?);
3496 }
3497 })
3498 })
3499 .collect::<Vec<_>>()
3500 .into_iter()
3501 .map(|handle| {
3502 handle
3503 .join()
3504 .map_err(|_| Error::internal("a native close worker panicked"))?
3505 })
3506 .collect::<Result<Vec<_>>>()
3507 })?
3508 .into_iter()
3509 .flatten()
3510 .collect()
3511 };
3512 for piece in pieces {
3513 match piece {
3514 Closed::Numeric(column, summary) => frequencies[column] = summary,
3515 Closed::Dictionary(index, one) => closed[index] = Some(one),
3516 }
3517 }
3518 Ok((frequencies, closed))
3519 }
3520
3521 fn close_dictionary(
3528 &self,
3529 _index: usize,
3530 dictionary: &GlobalDictionary,
3531 ) -> Result<ClosedDictionary> {
3532 let (order, flat, bases) = dictionary.ranked_with_values(Some(&*self.file))?;
3533 let (distinct, frequencies, texts) = if dictionary.demoted {
3538 (None, None, Vec::new())
3539 } else {
3540 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3541 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3542 (Some(distinct), Some(frequencies), texts)
3543 };
3544 let hosts = None;
3546 drop(flat);
3547 drop(bases);
3548 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3549 let payload = dictionary
3550 .placed
3551 .iter()
3552 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3553 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3554 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3555 }
3556
3557 fn write_stats(&mut self) -> Result<()> {
3569 let gathers = std::mem::take(&mut self.gathers);
3570 let rows = self.table.rows as u64;
3571 let mut payloads = Vec::new();
3572 for (column, gather) in gathers.into_iter().enumerate() {
3573 let Some(gather) = gather else { continue };
3574 if gather.rows() != rows {
3580 continue;
3581 }
3582 let Some(stats) = gather.finish() else { continue };
3583 let mut summary = Vec::new();
3584 stats.summary.encode(&mut summary)?;
3585 let mut sketches = Vec::new();
3586 stats.sketches.encode(&mut sketches)?;
3587 payloads.push((column, summary, sketches));
3588 }
3589 if payloads.is_empty() {
3590 return Ok(());
3591 }
3592 let costs = payloads
3593 .iter()
3594 .map(|(_, summary, sketches)| summary.len() + sketches.len())
3595 .collect::<Vec<_>>();
3596 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3597 let keep = stats::within(&costs, allowance, 0);
3600 for ((column, summary, sketches), _) in
3601 payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3602 {
3603 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3604 for (kind, bytes, header_bytes) in [
3605 (*section::SUMMARY, summary, summary.len() as u32),
3608 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3609 ] {
3610 let written = write_section(
3611 &*self.file,
3612 &mut self.at,
3613 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3614 self.generation,
3615 )?;
3616 self.table.sections.push(written);
3617 }
3618 }
3619 if self.table.sections.len() > MAX_SECTIONS {
3620 return Err(invalid("the table would name more sections than the bound allows"));
3621 }
3622 Ok(())
3623 }
3624
3625 pub fn finish(mut self) -> Result<Table> {
3635 let entry = self.close()?;
3636 let profile = self.profile.take();
3637 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3638 let mut tables = std::mem::take(&mut self.closed);
3639 tables.push(entry);
3640 let catalog = encode_catalog(&tables, &self.views)?;
3641 if catalog.len() > MAX_DIRECTORY {
3642 return Err(invalid("catalog exceeds the configured bound"));
3643 }
3644 let offset = self.at;
3645 self.put(&catalog)?;
3646 if let Some(profile) = &profile {
3647 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3648 }
3649 synced(&*self.file, profile.as_deref())?;
3653 let slot = Slot {
3654 offset,
3655 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3656 generation: self.generation,
3657 hash: checksum(&catalog),
3658 };
3659 self.file.write_at(slot_offset(self.generation), &slot.bytes())?;
3664 synced(&*self.file, profile.as_deref())?;
3665 Ok(self.table)
3666 }
3667
3668 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3685 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3686 let size = file.len()?;
3687 let (slot, bytes, _) = committed_slot(&*file, size)?;
3688 let (closed, _) = decode_catalog(&bytes, size)?;
3689 let generation = slot
3690 .generation
3691 .checked_add(1)
3692 .ok_or_else(|| invalid("native file generation overflow"))?;
3693 let catalog = encode_catalog(&closed, views)?;
3694 if catalog.len() > MAX_DIRECTORY {
3695 return Err(invalid("catalog exceeds the configured bound"));
3696 }
3697 file.write_at(size, &catalog)?;
3698 file.sync()?;
3699 let slot = Slot {
3700 offset: size,
3701 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3702 generation,
3703 hash: checksum(&catalog),
3704 };
3705 file.write_at(slot_offset(generation), &slot.bytes())?;
3706 file.sync()?;
3707 Ok(())
3708 }
3709
3710 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3713 let path = path.as_ref();
3714 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3715 let (mut entries, views) = decode_catalog(&bytes, size)?;
3716 let native = Catalog::open(path)?;
3717 for entry in &mut entries {
3718 let reader = native.table(&entry.name)?;
3719 entry.nonzero.fill(None);
3720 entry.aggregates = reader_aggregate_sums(&reader)?;
3721 entry.distincts = (0..entry.fields.len())
3722 .map(|column| reader.distinct_values(column))
3723 .collect::<Result<Vec<_>>>()?;
3724 entry.extremes = reader_integer_extremes(&reader)?;
3725 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3726 }
3727 let generation = slot
3728 .generation
3729 .checked_add(1)
3730 .ok_or_else(|| invalid("native file generation overflow"))?;
3731 let catalog = encode_catalog(&entries, &views)?;
3732 if catalog.len() > MAX_DIRECTORY {
3733 return Err(invalid("catalog exceeds the configured bound"));
3734 }
3735 let file = RealFilesystem::new().open(path, OpenMode::ReadWrite)?;
3736 file.write_at(size, &catalog)?;
3737 file.sync()?;
3738 let slot = Slot {
3739 offset: size,
3740 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3741 generation,
3742 hash: checksum(&catalog),
3743 };
3744 file.write_at(slot_offset(generation), &slot.bytes())?;
3745 file.sync()?;
3746 Ok(())
3747 }
3748
3749 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3751 Self::certify_summaries(path)
3752 }
3753}
3754
3755fn append(file: &dyn rudb_io::File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3761 let offset = *at;
3762 file.write_at(offset, bytes)?;
3763 *at =
3764 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3765 Ok(offset)
3766}
3767
3768fn write_section(
3774 file: &dyn rudb_io::File,
3775 at: &mut u64,
3776 one: §ion::Attachment<'_>,
3777 generation: u64,
3778) -> Result<Section> {
3779 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3783 return Err(invalid("a section's header is longer than its payload"));
3784 }
3785 let mut extents = Vec::new();
3786 let mut first = 0_u64;
3787 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3788 let offset = append(file, at, chunk)?;
3789 extents.push(section::Extent {
3790 offset,
3791 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3792 hash: checksum(chunk),
3793 first,
3794 });
3795 first += chunk.len() as u64;
3796 }
3797 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3798 section::encode_extents(&extents, &mut table)?;
3799 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3803 Ok(Section {
3804 kind: one.kind,
3805 id: one.id,
3806 generation,
3807 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3808 extent_page,
3809 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3810 hash: checksum(&table),
3811 flags: one.flags,
3812 header_bytes: one.header_bytes,
3813 })
3814}
3815
3816pub fn attach(
3840 path: impl AsRef<Path>,
3841 table: &str,
3842 attachments: &[section::Attachment<'_>],
3843) -> Result<Table> {
3844 let file = RealFilesystem::new().open(path.as_ref(), OpenMode::ReadWrite)?;
3845 let file = &*file;
3846 let size = file.len()?;
3847 let (slot, bytes, _) = committed_slot(file, size)?;
3848 let (mut entries, views) = decode_catalog(&bytes, size)?;
3849 let at = entries
3850 .iter()
3851 .position(|entry| entry.name == table)
3852 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3853 let mut version = [0; 4];
3854 read_at(file, 8, &mut version)?;
3855 let version = u32::from_le_bytes(version);
3856 if version != FORMAT {
3862 return Err(invalid(&format!(
3863 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3864 to be written again"
3865 )));
3866 }
3867 let mut directory = vec![0; entries[at].directory.length as usize];
3868 read_at(file, entries[at].directory.offset, &mut directory)?;
3869 if checksum(&directory) != entries[at].directory.hash {
3870 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3871 }
3872 let mut held = decode_directory(&directory, size)?;
3873 let mut cursor = size;
3874 for one in attachments {
3875 let written = write_section(file, &mut cursor, one, held.generation)?;
3876 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3877 held.sections.push(written);
3878 }
3879 if held.sections.len() > MAX_SECTIONS {
3880 return Err(invalid("the table would name more sections than the bound allows"));
3881 }
3882 let encoded = encode_directory(&held)?;
3883 if encoded.len() > MAX_DIRECTORY {
3884 return Err(invalid("directory exceeds the configured bound"));
3885 }
3886 let offset = append(file, &mut cursor, &encoded)?;
3887 entries[at].directory = Page {
3888 offset,
3889 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3890 hash: checksum(&encoded),
3891 };
3892 let catalog = encode_catalog(&entries, &views)?;
3895 if catalog.len() > MAX_DIRECTORY {
3896 return Err(invalid("catalog exceeds the configured bound"));
3897 }
3898 let offset = append(file, &mut cursor, &catalog)?;
3899 file.sync()?;
3900 let generation =
3901 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3902 let committed = Slot {
3903 offset,
3904 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3905 generation,
3906 hash: checksum(&catalog),
3907 };
3908 file.write_at(slot_offset(generation), &committed.bytes())?;
3909 file.sync()?;
3910 Ok(held)
3911}
3912
3913type Synopsis = Arc<Vec<(Value, u64)>>;
3916
3917#[derive(Debug, Clone)]
3919pub struct Reader {
3920 file: Arc<File>,
3921 table: Arc<Table>,
3922 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3923 loading: Arc<Vec<Mutex<()>>>,
3932 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3935 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3939 opened: Arc<AtomicUsize>,
3943 sieves: Arc<Vec<Vec<SieveSlot>>>,
3947 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3950 places: Arc<Vec<Place>>,
3952 cache: Arc<Shelf>,
3953 pool: PagePool,
3955 pages: Arc<AtomicUsize>,
3958 indexes: Arc<AtomicUsize>,
3961 size: u64,
3963 directory: u64,
3965 opening: Opening,
3967}
3968
3969#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3981pub struct Opening {
3982 pub reads: u32,
3985 pub bytes: u64,
3987}
3988
3989#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3991pub struct Reads {
3992 pub opening: Opening,
3994 pub pages: usize,
3996 pub indexes: usize,
3998 pub dictionaries: usize,
4001}
4002
4003#[derive(Debug, Clone, Copy)]
4005struct Place {
4006 stripe: u32,
4007 part: u32,
4008 rows: u32,
4009}
4010
4011#[derive(Debug, Clone, Copy)]
4013struct PartSpan {
4014 start: usize,
4015 length: usize,
4016 hash: u64,
4017}
4018
4019#[derive(Debug, Clone)]
4025struct CachedColumn {
4026 stripe: usize,
4027 index: Arc<Vec<PartSpan>>,
4028 page: Option<Arc<Vec<u8>>>,
4029}
4030
4031#[derive(Debug, Default)]
4051struct Cached {
4052 pages: Vec<Option<Resident>>,
4053 loading: Vec<usize>,
4054 index: Vec<Option<Arc<Vec<PartSpan>>>>,
4055}
4056
4057#[derive(Debug, Clone)]
4059struct Resident {
4060 page: Arc<Vec<u8>>,
4061 used: Arc<AtomicBool>,
4062}
4063
4064#[derive(Debug)]
4066struct Shelf {
4067 columns: Vec<Mutex<Cached>>,
4068 held: Vec<AtomicUsize>,
4071 kept: AtomicUsize,
4074}
4075
4076#[derive(Debug, Clone, Default)]
4095pub struct PagePool {
4096 ring: Arc<Mutex<Ring>>,
4097 budget: Arc<AtomicUsize>,
4098}
4099
4100#[derive(Debug, Default)]
4101struct Ring {
4102 held: VecDeque<Held>,
4103 bytes: usize,
4104}
4105
4106#[derive(Debug)]
4111struct Held {
4112 shelf: Weak<Shelf>,
4113 column: usize,
4114 stripe: usize,
4115 bytes: usize,
4116 used: Arc<AtomicBool>,
4117}
4118
4119impl PagePool {
4120 #[must_use]
4122 pub fn new(budget: usize) -> Self {
4123 let pool = Self::default();
4124 pool.budget.store(budget, Atomic::Relaxed);
4125 pool
4126 }
4127
4128 #[must_use]
4134 pub fn bytes(&self) -> usize {
4135 self.ring.lock().map_or(0, |ring| ring.bytes)
4136 }
4137
4138 fn admit(&self, held: Held) {
4144 let budget = self.budget.load(Atomic::Relaxed);
4145 let mut gone = Vec::new();
4146 {
4147 let Ok(mut ring) = self.ring.lock() else { return };
4148 ring.bytes += held.bytes;
4149 ring.held.push_back(held);
4150 let mut looked = 0;
4153 let limit = ring.held.len();
4154 while ring.bytes > budget && looked < limit {
4155 looked += 1;
4156 let Some(entry) = ring.held.pop_front() else { break };
4157 let Some(shelf) = entry.shelf.upgrade() else {
4158 ring.bytes -= entry.bytes;
4159 continue;
4160 };
4161 if entry.used.swap(false, Atomic::Relaxed) {
4162 ring.held.push_back(entry);
4163 continue;
4164 }
4165 let count = &shelf.held[entry.column];
4166 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
4167 ring.held.push_back(entry);
4168 continue;
4169 }
4170 count.fetch_sub(1, Atomic::Relaxed);
4171 ring.bytes -= entry.bytes;
4172 gone.push((shelf, entry));
4173 }
4174 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
4177 if let Some(entry) = ring.held.pop_front() {
4178 ring.bytes -= entry.bytes;
4179 }
4180 }
4181 }
4182 for (shelf, entry) in gone {
4183 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
4184 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
4185 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
4186 *slot = None;
4187 }
4188 }
4189 }
4190 }
4191}
4192
4193const CACHED_STRIPES_PER_COLUMN: usize = 4;
4205
4206type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
4208
4209type RangeSlot = OnceLock<Arc<Vec<Range>>>;
4210
4211#[derive(Debug)]
4212struct NativeText {
4213 file: Arc<File>,
4214 values: usize,
4216 offsets: Vec<u8>,
4228 offset_bits: usize,
4231 value_ends: OnceLock<Option<Vec<u32>>>,
4244 value_lens: OnceLock<Option<Lengths>>,
4254 ends_asked: AtomicUsize,
4260 ranks: usize,
4262 rank_at: u64,
4266 rank_ends: Vec<u64>,
4270 rank_hashes: Vec<u64>,
4271 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4272 code_bits: usize,
4275 code_ranks: OnceLock<Option<Vec<u32>>>,
4282 starts: Vec<u64>,
4289 lengths: Vec<u64>,
4290 hashes: Vec<u64>,
4291 grams: Option<NativeGrams>,
4293 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
4295 char_lens: Vec<OnceLock<Box<[u32]>>>,
4304 keep_budget: usize,
4307 payload_kept: AtomicUsize,
4315 swept: Vec<AtomicBool>,
4323 visit_dropped: AtomicUsize,
4338 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
4355}
4356
4357#[derive(Debug)]
4358struct NativeGrams {
4359 start: u64,
4360 length: usize,
4361 width: usize,
4363 hash: u64,
4364 verdicts: Mutex<Vec<Verdict>>,
4371}
4372
4373type Verdict = (Vec<u8>, Arc<[bool]>);
4375
4376const GRAM_VERDICTS: usize = 8;
4378
4379impl NativeGrams {
4380 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
4385 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
4386 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
4387 return Ok(Arc::clone(verdict));
4388 }
4389 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
4390 let mut verdict = Vec::with_capacity(self.length / self.width);
4391 let window = GRAM_WINDOW / self.width * self.width;
4392 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
4393 verdict.extend(bytes.chunks(self.width).map(|bits| {
4394 wanted
4395 .iter()
4396 .flatten()
4397 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
4398 }));
4399 Ok(())
4400 })?;
4401 if hash != self.hash {
4402 return Err(invalid("global dictionary substring signatures checksum differs"));
4403 }
4404 let verdict: Arc<[bool]> = verdict.into();
4405 if held.len() >= GRAM_VERDICTS {
4406 held.remove(0);
4407 }
4408 held.push((literal.to_vec(), Arc::clone(&verdict)));
4409 Ok(verdict)
4410 }
4411
4412 fn footprint(&self) -> usize {
4413 self.verdicts.lock().map_or(0, |held| {
4414 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4415 })
4416 }
4417}
4418
4419const TEXT_SEARCH_MEMO: usize = 64;
4424
4425const TEXT_PAYLOAD_VALUES: usize = 1024;
4441
4442const TEXT_GRAM_BYTES: usize = 8192;
4453
4454const NARROW_GRAM_BYTES: usize = 2048;
4456
4457const GRAM_WINDOW: usize = 256 << 10;
4459
4460fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4463 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4464 let mut first = original ^ (original >> 16);
4465 first = first.wrapping_mul(0x7feb_352d);
4466 first ^= first >> 15;
4467 let mut second = original ^ (original >> 17);
4468 second = second.wrapping_mul(0x846c_a68b);
4469 second ^= second >> 16;
4470 let mask = width * 8 - 1;
4471 [(first as usize) & mask, (second as usize) & mask]
4472}
4473
4474const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4495
4496#[derive(Debug)]
4503enum Lengths {
4504 Narrow(Vec<u16>),
4506 Wide(Vec<u32>),
4508}
4509
4510impl Lengths {
4511 fn extend_at(&self, indices: &[u32], into: &mut Vec<i64>) {
4514 match self {
4515 Lengths::Narrow(lens) => into.extend(
4516 indices
4517 .iter()
4518 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4519 ),
4520 Lengths::Wide(lens) => into.extend(
4521 indices
4522 .iter()
4523 .map(|&index| lens.get(index as usize).map_or(0, |&len| i64::from(len))),
4524 ),
4525 }
4526 }
4527
4528 fn footprint(&self) -> usize {
4530 match self {
4531 Lengths::Narrow(lens) => lens.capacity() * size_of::<u16>(),
4532 Lengths::Wide(lens) => lens.capacity() * size_of::<u32>(),
4533 }
4534 }
4535}
4536
4537fn lengths_of(ends: &[u32]) -> Option<Lengths> {
4546 match lengths_as::<u16>(ends)? {
4547 Some(narrow) => Some(Lengths::Narrow(narrow)),
4548 None => lengths_as::<u32>(ends)?.map(Lengths::Wide),
4549 }
4550}
4551
4552fn lengths_as<T: TryFrom<u32>>(ends: &[u32]) -> Option<Option<Vec<T>>> {
4555 let mut lens = Vec::with_capacity(ends.len());
4556 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4557 let mut start = 0;
4558 for &end in block {
4559 let Ok(len) = T::try_from(end.checked_sub(start)?) else {
4560 return Some(None);
4561 };
4562 lens.push(len);
4563 start = end;
4564 }
4565 }
4566 Some(Some(lens))
4567}
4568
4569const TEXT_OFFSET_RUN: usize = 512;
4576
4577const DICTIONARY_HEADER: usize = 16;
4580
4581const DICTIONARY_SCATTERED: u32 = 1 << 31;
4595const DICTIONARY_GRAMS: u32 = 1 << 30;
4597const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4600const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4602
4603const TEXT_RANK_BLOCK: usize = 512;
4614
4615const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4629
4630impl NativeText {
4631 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4638 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4639 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4640 Ok(Some(bytes.as_slice()))
4641 }
4642
4643 fn block_chars(&self, block: usize) -> Result<&[u32]> {
4650 let slot = self
4651 .char_lens
4652 .get(block)
4653 .ok_or_else(|| invalid("a block past the global dictionary"))?;
4654 if let Some(lens) = slot.get() {
4655 return Ok(lens);
4656 }
4657 let decoded;
4658 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4659 Some(Ok(kept)) => kept,
4660 _ => {
4661 decoded = self.decode_block(block)?;
4662 &decoded
4663 }
4664 };
4665 let first = block * TEXT_PAYLOAD_VALUES;
4666 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4667 let ends = self.ends_within(first, last)?;
4668 if ends.len() != last - first {
4669 return Err(invalid("global dictionary offsets are short"));
4670 }
4671 let mut lens = Vec::with_capacity(ends.len());
4672 let mut start = u64::from(self.start_within(first)?);
4673 for &end in &ends {
4674 let value = usize::try_from(start)
4675 .ok()
4676 .zip(usize::try_from(end).ok())
4677 .and_then(|(from, to)| bytes.get(from..to))
4678 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4679 let characters = value.iter().filter(|byte| (**byte as i8) >= -0x40).count();
4682 lens.push(u32::try_from(characters).unwrap_or(u32::MAX));
4683 start = end;
4684 }
4685 Ok(slot.get_or_init(|| lens.into_boxed_slice()))
4686 }
4687
4688 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4693 let len = self.lengths[block];
4694 let mut stored = vec![
4695 0;
4696 usize::try_from(len).map_err(|_| invalid(
4697 "global dictionary block does not fit in memory"
4698 ))?
4699 ];
4700 read_at(&self.file, self.starts[block], &mut stored)?;
4701 if checksum(&stored) != self.hashes[block] {
4702 return Err(invalid("global dictionary payload checksum differs"));
4703 }
4704 let first = block * TEXT_PAYLOAD_VALUES;
4705 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4706 let want = self.end_within(last - 1)? as usize;
4707 let values = string::decode_flat(&stored)?;
4708 if values.len() != last - first {
4709 return Err(invalid("global dictionary block holds the wrong value count"));
4710 }
4711 let bytes = values.into_bytes();
4712 if bytes.len() != want {
4713 return Err(invalid("global dictionary block decodes to the wrong length"));
4714 }
4715 Ok(bytes)
4716 }
4717
4718 fn loaned_block<'a>(
4727 &'a self,
4728 block: usize,
4729 decoded: &'a mut Vec<u8>,
4730 scattered: bool,
4731 ) -> Result<&'a [u8]> {
4732 let kept = self.blocks.get(block).and_then(OnceLock::get);
4733 if let Some(Ok(kept)) = kept {
4734 return Ok(kept);
4735 }
4736 let again = kept.is_none()
4737 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4738 let keep = again
4739 && (self.payload_kept.load(Atomic::Relaxed) < self.keep_budget
4740 || (scattered && self.visit_dropped.load(Atomic::Relaxed) >= self.blocks.len()));
4741 if keep {
4742 let kept = self
4743 .payload_block(block)?
4744 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4745 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4746 return Ok(kept);
4747 }
4748 *decoded = self.decode_block(block)?;
4749 if scattered && again {
4750 self.visit_dropped.fetch_add(1, Atomic::Relaxed);
4751 }
4752 Ok(decoded)
4753 }
4754
4755 fn ends_worth_unpacking(&self) -> usize {
4772 self.values.max(TEXT_PAYLOAD_VALUES)
4773 }
4774
4775 fn value_ends(&self) -> Option<&[u32]> {
4777 if let Some(built) = self.value_ends.get() {
4778 return built.as_deref();
4779 }
4780 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4781 return None;
4782 }
4783 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4784 }
4785
4786 fn unpack_ends(&self) -> Option<Vec<u32>> {
4792 let mut ends = vec![0u32; self.values];
4793 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4794 let bytes = self.packed().get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4795 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4796 u32::try_from(bits).unwrap_or(u32::MAX)
4797 })
4798 .ok()?;
4799 }
4800 if ends.contains(&u32::MAX) { None } else { Some(ends) }
4803 }
4804
4805 fn packed(&self) -> &[u8] {
4807 self.offsets.get(DICTIONARY_HEADER..).unwrap_or_default()
4808 }
4809
4810 fn end_within(&self, index: usize) -> Result<u32> {
4812 if let Some(ends) = self.value_ends() {
4813 return ends
4814 .get(index)
4815 .copied()
4816 .ok_or_else(|| invalid("global dictionary offsets are short"));
4817 }
4818 let run = index / TEXT_OFFSET_RUN;
4819 let bytes = self
4820 .packed()
4821 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4822 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4823 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4824 .map_err(|_| invalid("global dictionary offsets are short"))?;
4825 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4826 }
4827
4828 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4846 let mut ends = vec![0u64; last.saturating_sub(first)];
4847 let mut scratch = Vec::new();
4848 let mut at = first;
4849 while at < last {
4850 let run = at / TEXT_OFFSET_RUN;
4851 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4852 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4853 let bytes = self
4854 .packed()
4855 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4856 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4857 let from = at % TEXT_OFFSET_RUN;
4858 let upto = stop - run * TEXT_OFFSET_RUN;
4859 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4860 return Err(invalid("global dictionary offsets are short"));
4861 }
4862 let into = &mut ends[at - first..stop - first];
4863 if from == 0 {
4864 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4865 .map_err(|_| invalid("global dictionary offsets are short"))?;
4866 } else {
4867 scratch.resize(held, 0);
4868 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4869 .map_err(|_| invalid("global dictionary offsets are short"))?;
4870 into.copy_from_slice(&scratch[from..upto]);
4871 }
4872 at = stop;
4873 }
4874 Ok(ends)
4875 }
4876
4877 fn start_within(&self, index: usize) -> Result<u32> {
4880 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4881 }
4882
4883 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4891 if let Some(ends) = self.value_ends() {
4892 let end =
4893 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4894 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4897 if start > end {
4898 return Err(invalid("global dictionary value ends before it starts"));
4899 }
4900 return Ok((start, end));
4901 }
4902 let within = index % TEXT_OFFSET_RUN;
4903 let (start, end) = if within == 0 {
4904 (self.start_within(index)?, self.end_within(index)?)
4905 } else {
4906 let run = index / TEXT_OFFSET_RUN;
4907 let bytes = self
4908 .packed()
4909 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4910 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4911 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4912 .map_err(|_| invalid("global dictionary offsets are short"))?;
4913 let ends = u32::try_from(end)
4914 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4915 let starts = u32::try_from(start)
4916 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4917 (starts, ends)
4918 };
4919 if start > end {
4920 return Err(invalid("global dictionary value ends before it starts"));
4921 }
4922 Ok((start, end))
4923 }
4924
4925 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4932 let slot = self
4933 .rank_blocks
4934 .get(rank / TEXT_RANK_BLOCK)
4935 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4936 let block = slot
4937 .get_or_init(|| {
4938 let mut bytes = Vec::new();
4939 self.read_rank_block(rank / TEXT_RANK_BLOCK, &mut bytes)?;
4940 Ok(bytes)
4941 })
4942 .as_ref()
4943 .map_err(Clone::clone)?;
4944 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4945 }
4946
4947 fn read_rank_block(&self, which: usize, bytes: &mut Vec<u8>) -> Result<()> {
4950 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4951 let end = self.rank_ends[which];
4952 bytes.clear();
4953 bytes.resize((end - start) as usize, 0);
4954 read_at(&self.file, self.rank_at + start, bytes)?;
4955 let expected = self
4956 .rank_hashes
4957 .get(which)
4958 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?;
4959 if checksum(bytes) != *expected {
4960 return Err(invalid("global dictionary rank checksum differs"));
4961 }
4962 Ok(())
4963 }
4964
4965 fn head_at(&self, rank: usize) -> Result<u64> {
4967 let (block, within) = self.rank_parts(rank)?;
4968 let (base, width, packed) = rank_heads(block)?;
4969 let above = bitpack::tail_at(packed, width, within)
4970 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4971 Ok(base.wrapping_add(above))
4972 }
4973
4974 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4976 let (_, width, packed) = rank_heads(block)?;
4977 packed
4978 .get(bitpack::tail_len(count, width)..)
4979 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4980 }
4981
4982 fn rank_block_len(&self, rank: usize) -> usize {
4984 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4985 TEXT_RANK_BLOCK.min(self.ranks - first)
4986 }
4987}
4988
4989fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4991 let header = block
4992 .get(..RANK_BLOCK_HEADER)
4993 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4994 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4995 let width = header[8] as usize;
4996 if width > 64 {
4997 return Err(invalid("global dictionary rank block packs heads past a word"));
4998 }
4999 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
5000}
5001
5002fn offset_width(ends: &[u32]) -> usize {
5009 let span = ends.iter().copied().max().unwrap_or(0);
5013 (u32::BITS - span.leading_zeros()) as usize
5014}
5015
5016fn offset_bytes(values: usize, bits: usize) -> usize {
5019 let full = values / TEXT_OFFSET_RUN;
5020 let rest = values % TEXT_OFFSET_RUN;
5021 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
5022}
5023
5024fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
5028 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
5029 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
5030 run.clear();
5031 run.extend(chunk.iter().map(|&end| u64::from(end)));
5032 bitpack::pack_tail(&run, bits, out)
5033 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
5034 }
5035 Ok(())
5036}
5037
5038fn code_width(values: usize) -> usize {
5040 match u64::try_from(values).unwrap_or(u64::MAX) {
5041 0 | 1 => 0,
5042 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
5043 }
5044}
5045
5046impl TextSource for NativeText {
5047 fn len(&self) -> usize {
5048 self.values
5049 }
5050
5051 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
5052 let Some(grams) = &self.grams else { return Ok(true) };
5053 if literal.len() < 4 || first >= self.values {
5054 return Ok(true);
5055 }
5056 let verdict = grams.verdicts(&self.file, literal)?;
5057 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
5058 }
5059
5060 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
5061 if index >= self.values {
5062 return Ok(None);
5063 }
5064 let (start, end) = self.span_within(index)?;
5065 if start == end {
5066 return Ok(Some(&[]));
5067 }
5068 let block = index / TEXT_PAYLOAD_VALUES;
5071 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
5072 Ok(bytes.get(start as usize..end as usize))
5073 }
5074
5075 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
5076 if index >= self.values {
5077 return Ok(None);
5078 }
5079 let (start, end) = self.span_within(index)?;
5080 Ok(Some((end - start) as usize))
5081 }
5082
5083 fn bytes_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5090 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
5091 into.reserve(indices.len());
5092 let Some(ends) = self.value_ends() else {
5093 for &index in indices {
5094 into.push(
5095 self.bytes_len_at(index as usize)?
5096 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX)),
5097 );
5098 }
5099 return Ok(());
5100 };
5101 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
5102 lens.extend_at(indices, into);
5103 return Ok(());
5104 }
5105 for &index in indices {
5106 let index = index as usize;
5107 let Some(&end) = ends.get(index) else {
5109 into.push(0);
5110 continue;
5111 };
5112 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
5113 if start > end {
5114 return Err(invalid("global dictionary value ends before it starts"));
5115 }
5116 into.push(i64::from(end - start));
5117 }
5118 Ok(())
5119 }
5120
5121 fn chars_lens_at(&self, indices: &[u32], into: &mut Vec<i64>) -> Result<()> {
5124 into.reserve(indices.len());
5125 for &index in indices {
5126 let index = index as usize;
5127 if index >= self.values {
5129 into.push(0);
5130 continue;
5131 }
5132 let lens = self.block_chars(index / TEXT_PAYLOAD_VALUES)?;
5133 let len = lens
5134 .get(index % TEXT_PAYLOAD_VALUES)
5135 .ok_or_else(|| invalid("global dictionary block holds the wrong value count"))?;
5136 into.push(i64::from(*len));
5137 }
5138 Ok(())
5139 }
5140
5141 fn sweep(
5154 &self,
5155 first: usize,
5156 limit: usize,
5157 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5158 ) -> Result<usize> {
5159 let limit = limit.min(self.values);
5160 if first >= limit {
5161 return Ok(first);
5162 }
5163 let block = first / TEXT_PAYLOAD_VALUES;
5164 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
5165 let mut decoded = Vec::new();
5166 let bytes = self.loaned_block(block, &mut decoded, false)?;
5167 let ends = self.ends_within(first, last)?;
5168 if ends.len() != last - first {
5169 return Err(invalid("global dictionary offsets are short"));
5170 }
5171 let mut start = u64::from(self.start_within(first)?);
5172 for (index, &end) in (first..last).zip(&ends) {
5175 let value = usize::try_from(start)
5176 .ok()
5177 .zip(usize::try_from(end).ok())
5178 .and_then(|(from, to)| bytes.get(from..to))
5179 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5180 body(index, value)?;
5181 start = end;
5182 }
5183 Ok(last)
5184 }
5185
5186 fn visit_at(
5195 &self,
5196 indices: &[u32],
5197 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5198 ) -> Result<()> {
5199 let mut order = (0..indices.len()).collect::<Vec<_>>();
5200 order.sort_unstable_by_key(|&at| indices[at]);
5201 let block_of = |at: usize| {
5202 let index = indices[at] as usize;
5203 (index < self.values).then_some(index / TEXT_PAYLOAD_VALUES)
5204 };
5205 let mut decoded = Vec::new();
5206 let mut run = 0;
5207 while run < order.len() {
5208 let Some(block) = block_of(order[run]) else {
5209 for &at in &order[run..] {
5211 body(at, &[])?;
5212 }
5213 break;
5214 };
5215 let upto = run + order[run..].partition_point(|&at| block_of(at) == Some(block));
5216 let bytes = self.loaned_block(block, &mut decoded, true)?;
5217 for &at in &order[run..upto] {
5218 let (start, end) = self.span_within(indices[at] as usize)?;
5219 let value = bytes
5220 .get(start as usize..end as usize)
5221 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5222 body(at, value)?;
5223 }
5224 run = upto;
5225 }
5226 Ok(())
5227 }
5228
5229 fn visit(
5235 &self,
5236 indices: &[usize],
5237 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
5238 ) -> Result<()> {
5239 let mut at = 0;
5240 while at < indices.len() {
5241 let block = indices[at] / TEXT_PAYLOAD_VALUES;
5242 let upto =
5243 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
5244 let wanted = &indices[at..upto];
5245 if wanted.iter().any(|&index| index >= self.values) {
5246 return Err(invalid("a visited value is past the global dictionary"));
5247 }
5248 let decoded;
5249 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
5250 Some(Ok(kept)) => kept,
5251 _ => {
5252 decoded = self.decode_block(block)?;
5253 &decoded
5254 }
5255 };
5256 for (offset, &index) in wanted.iter().enumerate() {
5257 let (start, end) = self.span_within(index)?;
5258 let value = bytes
5259 .get(start as usize..end as usize)
5260 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
5261 body(at + offset, value)?;
5262 }
5263 at = upto;
5264 }
5265 Ok(())
5266 }
5267
5268 fn ranks(&self) -> Option<usize> {
5269 (self.ranks > 0).then_some(self.ranks)
5270 }
5271
5272 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
5280 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
5281 if let Some(&answer) = memo.get(wanted) {
5282 return Ok(answer);
5283 }
5284 let answer = search_below(self, ranks, wanted)?;
5285 if memo.len() >= TEXT_SEARCH_MEMO {
5286 memo.clear();
5287 }
5288 memo.insert(wanted.to_vec(), answer);
5289 Ok(answer)
5290 }
5291
5292 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
5293 let settled = self.head_at(rank)?.cmp(&head(wanted));
5297 if settled != Ordering::Equal {
5298 return Ok(settled);
5299 }
5300 let code = self.code_at_rank(rank)?;
5301 let bytes = self
5302 .bytes_at(code as usize)?
5303 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
5304 Ok(bytes.cmp(wanted))
5305 }
5306
5307 fn code_at_rank(&self, rank: usize) -> Result<u32> {
5308 let (block, within) = self.rank_parts(rank)?;
5309 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
5310 let code = bitpack::tail_at(codes, self.code_bits, within)
5311 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
5312 let code = u32::try_from(code)
5313 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
5314 if code as usize >= self.len() {
5315 return Err(invalid("global dictionary order names a code it does not have"));
5316 }
5317 Ok(code)
5318 }
5319
5320 fn code_ranks(&self) -> Option<&[u32]> {
5321 if self.ranks == 0 || self.ranks != self.len() {
5325 return None;
5326 }
5327 self.code_ranks
5328 .get_or_init(|| {
5329 let mut ranks = vec![u32::MAX; self.ranks];
5330 let mut scratch = Vec::new();
5338 let mut codes = vec![0u64; TEXT_RANK_BLOCK];
5339 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
5340 let which = first / TEXT_RANK_BLOCK;
5341 let block = match self.rank_blocks.get(which)?.get() {
5342 Some(kept) => kept.as_ref().ok()?.as_slice(),
5343 None => {
5344 self.read_rank_block(which, &mut scratch).ok()?;
5345 scratch.as_slice()
5346 }
5347 };
5348 let count = self.rank_block_len(first);
5349 let packed = self.rank_codes(block, count).ok()?;
5350 let codes = codes.get_mut(..count)?;
5351 bitpack::unpack_tail_into(packed, self.code_bits, codes, |bits| bits).ok()?;
5352 for (within, &code) in codes.iter().enumerate() {
5353 let code = usize::try_from(code).ok()?;
5354 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
5355 }
5356 }
5357 if ranks.contains(&u32::MAX) {
5358 return None;
5359 }
5360 Some(ranks)
5361 })
5362 .as_deref()
5363 }
5364
5365 fn footprint(&self) -> usize {
5366 self.offsets.capacity()
5367 + self
5368 .value_ends
5369 .get()
5370 .and_then(Option::as_ref)
5371 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
5372 + self.value_lens.get().and_then(Option::as_ref).map_or(0, Lengths::footprint)
5373 + self
5374 .code_ranks
5375 .get()
5376 .and_then(Option::as_ref)
5377 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
5378 + self.rank_hashes.capacity() * size_of::<u64>()
5379 + self.rank_ends.capacity() * size_of::<u64>()
5380 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5381 + self
5382 .rank_blocks
5383 .iter()
5384 .filter_map(OnceLock::get)
5385 .filter_map(|result| result.as_ref().ok())
5386 .map(Vec::capacity)
5387 .sum::<usize>()
5388 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
5389 + self.char_lens.capacity() * size_of::<OnceLock<Box<[u32]>>>()
5390 + self
5391 .char_lens
5392 .iter()
5393 .filter_map(OnceLock::get)
5394 .map(|lens| lens.len() * size_of::<u32>())
5395 .sum::<usize>()
5396 + self.hashes.capacity() * size_of::<u64>()
5397 + self.starts.capacity() * size_of::<u64>()
5398 + self.lengths.capacity() * size_of::<u64>()
5399 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
5400 + self
5401 .blocks
5402 .iter()
5403 .filter_map(OnceLock::get)
5404 .filter_map(|result| result.as_ref().ok())
5405 .map(Vec::capacity)
5406 .sum::<usize>()
5407 }
5408}
5409
5410fn places(table: &Table) -> Result<Vec<Place>> {
5412 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
5413 for (at, stripe) in table.stripes.iter().enumerate() {
5414 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
5415 for (part, &rows) in stripe.parts.iter().enumerate() {
5416 places.push(Place {
5417 stripe: index,
5418 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
5419 rows,
5420 });
5421 }
5422 }
5423 Ok(places)
5424}
5425
5426fn read_index<F: Positional + ?Sized>(
5431 file: &F,
5432 stripe: &Stripe,
5433 column: usize,
5434) -> Result<Vec<PartSpan>> {
5435 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5436 read_index_span(file, stripe.index, *page, stripe.parts.len(), column)
5437}
5438
5439fn read_index_span<F: Positional + ?Sized>(
5440 file: &F,
5441 index: Span,
5442 page: Span,
5443 parts: usize,
5444 column: usize,
5445) -> Result<Vec<PartSpan>> {
5446 let section = index_section(parts)?;
5447 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
5448 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
5449 if end > index.length as usize {
5450 return Err(invalid("index page is shorter than its columns"));
5451 }
5452 let mut bytes = vec![0; section];
5453 let offset =
5454 index.offset.checked_add(at as u64).ok_or_else(|| invalid("index page offset overflow"))?;
5455 read_at(file, offset, &mut bytes)?;
5456 let entries = section - size_of::<u64>();
5457 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
5458 if checksum(&bytes[..entries]) != stored {
5459 return Err(invalid(&format!(
5462 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
5463 wanted {stored:016x} and got {:016x}",
5464 checksum(&bytes[..entries]),
5465 )));
5466 }
5467 let mut spans = Vec::with_capacity(parts);
5468 let mut start = 0_usize;
5469 for part in 0..parts {
5470 let at = part * INDEX_ENTRY;
5471 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
5472 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
5473 spans.push(PartSpan { start, length, hash });
5474 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
5475 }
5476 if start != page.length as usize {
5477 return Err(invalid("column page length differs from its index"));
5478 }
5479 Ok(spans)
5480}
5481
5482fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
5484 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
5485 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
5486}
5487
5488fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
5494 if let Some(slot) = cached.index.get_mut(held.stripe) {
5495 if slot.is_none() {
5496 *slot = Some(Arc::clone(&held.index));
5497 }
5498 }
5499 let page = held.page.clone()?;
5500 let slot = cached.pages.get_mut(held.stripe)?;
5501 if slot.is_some() {
5502 return None;
5503 }
5504 let bytes = page.len();
5505 let used = Arc::new(AtomicBool::new(true));
5508 *slot = Some(Resident { page, used: Arc::clone(&used) });
5509 Some((bytes, used))
5510}
5511
5512#[derive(Debug, Clone)]
5521pub struct Catalog {
5522 file: Arc<File>,
5523 size: u64,
5524 entries: Arc<Vec<Entry>>,
5525 views: Arc<Vec<ViewEntry>>,
5527 opening: Opening,
5528 pool: PagePool,
5530}
5531
5532#[derive(Debug, Clone, PartialEq, Eq)]
5534pub struct CertifiedSums {
5535 pub columns: Vec<(i128, u64)>,
5536 pub rows: u64,
5537}
5538
5539#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5541pub enum IntegerExtremes {
5542 Null,
5543 Values { low: i128, high: i128 },
5544}
5545
5546pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
5548
5549impl Catalog {
5550 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5559 Self::open_in(path, &PagePool::default())
5560 }
5561
5562 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
5568 let (file, size, _, bytes, opening) = slot_bytes(path)?;
5569 let (entries, views) = decode_catalog(&bytes, size)?;
5570 Ok(Self {
5571 file: Arc::new(file),
5572 size,
5573 entries: Arc::new(entries),
5574 views: Arc::new(views),
5575 opening,
5576 pool: pool.clone(),
5577 })
5578 }
5579
5580 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
5582 self.entries.iter().map(|entry| entry.name.as_str())
5583 }
5584
5585 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
5592 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
5593 }
5594
5595 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
5601 self.views.iter()
5602 }
5603
5604 #[must_use]
5606 pub fn len(&self) -> usize {
5607 self.entries.len()
5608 }
5609
5610 #[must_use]
5613 pub fn is_empty(&self) -> bool {
5614 self.entries.is_empty()
5615 }
5616
5617 pub fn table(&self, name: &str) -> Result<Reader> {
5623 let entry = self
5624 .entries
5625 .iter()
5626 .find(|entry| entry.name == name)
5627 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5628 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5632 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5633 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5634 }
5635 let mut opening = self.opening;
5636 opening.reads += 1;
5637 opening.bytes += u64::from(entry.directory.length);
5638 Reader::build(
5639 Arc::clone(&self.file),
5640 self.size,
5641 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5642 u64::from(entry.directory.length),
5643 opening,
5644 self.pool.clone(),
5645 )
5646 }
5647
5648 pub fn integer_tally(&self, name: &str, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
5656 let mut counts = BTreeMap::<i64, u64>::new();
5657 let Some(()) = self.integer_fold(name, column, |value, count| {
5658 let held = counts.entry(value).or_default();
5659 *held = held.checked_add(count).ok_or_else(|| invalid("integer count overflow"))?;
5660 Ok(())
5661 })?
5662 else {
5663 return Ok(None);
5664 };
5665 Ok(Some(counts.into_iter().collect()))
5666 }
5667
5668 pub fn integer_fold(
5675 &self,
5676 name: &str,
5677 column: usize,
5678 mut emit: impl FnMut(i64, u64) -> Result<()>,
5679 ) -> Result<Option<()>> {
5680 let entry = self
5681 .entries
5682 .iter()
5683 .find(|entry| entry.name == name)
5684 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5685 let field =
5686 entry.fields.get(column).ok_or_else(|| invalid("integer column index out of range"))?;
5687 if !signed_integer(&field.ty) {
5688 return Ok(None);
5689 }
5690 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5691 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5692 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5693 }
5694 quick_integer_fold(
5695 &self.file,
5696 Cursor::over(&self.file, offset, length),
5697 entry,
5698 self.size,
5699 column,
5700 &mut emit,
5701 )?;
5702 Ok(Some(()))
5703 }
5704
5705 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5710 let entry = self
5711 .entries
5712 .iter()
5713 .find(|entry| entry.name == name)
5714 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5715 let Some(field) = entry.fields.get(column) else {
5716 return Err(invalid("frequency column index out of range"));
5717 };
5718 if !matches!(
5719 field.ty,
5720 LogicalType::TinyInt
5721 | LogicalType::SmallInt
5722 | LogicalType::Integer
5723 | LogicalType::BigInt
5724 | LogicalType::UTinyInt
5725 | LogicalType::USmallInt
5726 | LogicalType::UInteger
5727 | LogicalType::UBigInt
5728 ) {
5729 return Ok(None);
5730 }
5731 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5732 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5733 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5734 }
5735 if let Some(Some(frequencies)) = entry.frequencies.get(column) {
5736 return frequencies
5737 .iter()
5738 .filter(|(value, _)| value.is_some_and(|value| value != 0))
5739 .try_fold(0_u64, |total, (_, count)| total.checked_add(*count))
5740 .map(Some)
5741 .ok_or_else(|| invalid("numeric frequency count overflow"));
5742 }
5743 quick_nonzero(
5744 Cursor::over(&self.file, offset, length),
5745 &entry.name,
5746 &entry.fields,
5747 entry.rows,
5748 column,
5749 )
5750 }
5751
5752 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5755 let entry = self
5756 .entries
5757 .iter()
5758 .find(|entry| entry.name == name)
5759 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5760 let mut sums = Vec::with_capacity(columns.len());
5761 for &column in columns {
5762 let Some(field) = entry.fields.get(column) else {
5763 return Err(invalid("aggregate column index out of range"));
5764 };
5765 if !signed_integer(&field.ty) {
5766 return Ok(None);
5767 }
5768 let Some(sum) = entry.aggregates[column] else {
5769 return Ok(None);
5770 };
5771 sums.push(sum);
5772 }
5773 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5774 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5775 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5776 }
5777 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5778 }
5779
5780 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5782 let entry = self
5783 .entries
5784 .iter()
5785 .find(|entry| entry.name == name)
5786 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5787 let Some(count) = entry.distincts.get(column).copied() else {
5788 return Err(invalid("distinct column index out of range"));
5789 };
5790 let Some(count) = count else { return Ok(None) };
5791 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5792 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5793 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5794 }
5795 Ok(Some(count))
5796 }
5797
5798 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5800 let entry = self
5801 .entries
5802 .iter()
5803 .find(|entry| entry.name == name)
5804 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5805 let Some(extremes) = entry.extremes.get(column).copied() else {
5806 return Err(invalid("extremes column index out of range"));
5807 };
5808 let Some(extremes) = extremes else { return Ok(None) };
5809 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5810 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5811 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5812 }
5813 Ok(Some(match extremes {
5814 None => IntegerExtremes::Null,
5815 Some((low, high)) => IntegerExtremes::Values { low, high },
5816 }))
5817 }
5818
5819 pub fn exact_numeric_frequencies(
5821 &self,
5822 name: &str,
5823 column: usize,
5824 ) -> Result<Option<NumericFrequencies>> {
5825 let entry = self
5826 .entries
5827 .iter()
5828 .find(|entry| entry.name == name)
5829 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5830 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5831 return Err(invalid("numeric frequency column index out of range"));
5832 };
5833 let Some(frequencies) = frequencies else { return Ok(None) };
5834 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5835 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5836 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5837 }
5838 Ok(Some(frequencies))
5839 }
5840
5841 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5843 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5844 }
5845}
5846
5847fn slot_offset(generation: u64) -> u64 {
5852 16 + (generation - 1) % 2 * SLOT_BYTES as u64
5853}
5854
5855fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5860 let file = File::open(path).map_err(io)?;
5861 let size = file.metadata().map_err(io)?.len();
5862 let (slot, bytes, opening) = committed_slot(&file, size)?;
5863 Ok((file, size, slot, bytes, opening))
5864}
5865
5866fn committed_slot<F: Positional + ?Sized>(file: &F, size: u64) -> Result<(Slot, Vec<u8>, Opening)> {
5872 if size < HEADER {
5873 return Err(invalid("file is shorter than its header"));
5874 }
5875 let mut header = [0; HEADER as usize];
5876 read_at(file, 0, &mut header)?;
5877 let mut opening = Opening { reads: 1, bytes: HEADER };
5878 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5879 if &header[..8] != MAGIC {
5884 return Err(invalid("the header does not begin with a rudb native magic"));
5885 }
5886 if !READABLE.contains(&version) {
5887 return Err(invalid(&format!(
5888 "the file is format {version} and this build reads format {FORMAT}, so it has to \
5889 be written again"
5890 )));
5891 }
5892 let mut selected = None;
5893 for start in [16, 16 + SLOT_BYTES] {
5894 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5895 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5896 continue;
5897 }
5898 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5899 if slot.offset < HEADER || end > size {
5900 continue;
5901 }
5902 let mut bytes = vec![0; slot.length as usize];
5903 read_at(file, slot.offset, &mut bytes)?;
5904 opening.reads += 1;
5905 opening.bytes += u64::from(slot.length);
5906 if checksum(&bytes) == slot.hash
5907 && selected
5908 .as_ref()
5909 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5910 {
5911 selected = Some((slot, bytes));
5912 }
5913 }
5914 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5915 Ok((slot, bytes, opening))
5916}
5917
5918impl Reader {
5919 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5926 let catalog = Catalog::open(path)?;
5927 let mut names = catalog.names();
5928 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5929 if names.next().is_some() {
5930 return Err(invalid(
5931 "the file holds more than one table, so it has to be opened by name",
5932 ));
5933 }
5934 catalog.table(&name)
5935 }
5936
5937 fn build(
5939 file: Arc<File>,
5940 size: u64,
5941 table: Table,
5942 directory: u64,
5943 opening: Opening,
5944 pool: PagePool,
5945 ) -> Result<Self> {
5946 let places = places(&table)?;
5947 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5948 let table_fields = table.fields.len();
5949 let stripes = table.stripes.len();
5950 let columns = (0..table.fields.len())
5951 .map(|_| {
5952 Mutex::new(Cached {
5953 pages: (0..stripes).map(|_| None).collect(),
5954 index: (0..stripes).map(|_| None).collect(),
5955 ..Cached::default()
5956 })
5957 })
5958 .collect::<Vec<_>>();
5959 let cache = Shelf {
5960 columns,
5961 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5962 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5963 };
5964 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5965 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5966 .collect();
5967 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5968 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5969 .collect();
5970 Ok(Self {
5971 file,
5972 table: Arc::new(table),
5973 dictionaries: Arc::new(dictionaries),
5974 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5975 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5976 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5977 opened: Arc::new(AtomicUsize::new(0)),
5978 sieves: Arc::new(sieves),
5979 part_ranges: Arc::new(part_ranges),
5980 places: Arc::new(places),
5981 cache: Arc::new(cache),
5982 pool,
5983 pages: Arc::new(AtomicUsize::new(0)),
5984 indexes: Arc::new(AtomicUsize::new(0)),
5985 size,
5986 directory,
5987 opening,
5988 })
5989 }
5990
5991 #[must_use]
5998 pub fn reads(&self) -> Reads {
5999 Reads {
6000 opening: self.opening,
6001 pages: self.pages.load(Atomic::Relaxed),
6002 indexes: self.indexes.load(Atomic::Relaxed),
6003 dictionaries: self.opened.load(Atomic::Relaxed),
6004 }
6005 }
6006
6007 #[must_use]
6012 pub fn layout(&self) -> Layout {
6013 let table = &self.table;
6014 let stripes = table.stripes.as_slice();
6015 let columns = table
6016 .fields
6017 .iter()
6018 .enumerate()
6019 .map(|(at, field)| ColumnLayout {
6020 name: field.name.clone(),
6021 kind: field.ty.to_string(),
6022 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
6023 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
6024 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
6025 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
6026 dictionary: dictionary_bytes(table, at),
6027 })
6028 .collect();
6029 Layout {
6030 file: self.size,
6031 rows: table.rows,
6032 stripes: stripes.len(),
6033 parts: self.places.len(),
6034 columns,
6035 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
6036 directory: self.directory,
6037 header: HEADER,
6038 }
6039 }
6040
6041 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
6058 let field = self
6059 .table
6060 .fields
6061 .get(column)
6062 .ok_or_else(|| invalid("stored column index out of range"))?;
6063 let mut stored = Vec::with_capacity(self.places.len());
6064 let mut row = 0;
6065 for (at, stripe) in self.table.stripes.iter().enumerate() {
6066 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6067 let index = read_index(&self.file, stripe, column)?;
6068 let mut bytes = vec![0; page.length as usize];
6069 read_at(&self.file, page.offset, &mut bytes)?;
6070 let ranges = self.stripe_part_ranges(at, column);
6071 for (part, &rows) in stripe.parts.iter().enumerate() {
6072 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
6073 let held = part_bytes(&bytes, span)?;
6074 let range = ranges.and_then(|held| held.get(part));
6075 stored.push(StoredPart {
6076 stripe: at,
6077 part,
6078 row,
6079 rows: rows as usize,
6080 encoding: page_encoding(&field.ty, rows as usize, held),
6081 bytes: span.length as u64,
6082 page: page.offset,
6083 offset: span.start as u64,
6084 low: range
6085 .and_then(|range| range.low.clone())
6086 .and_then(|bound| bound.into_value(&field.ty)),
6087 high: range
6088 .and_then(|range| range.high.clone())
6089 .and_then(|bound| bound.into_value(&field.ty)),
6090 nulls: range.map(|range| range.nulls),
6091 });
6092 row += rows as usize;
6093 }
6094 }
6095 Ok(stored)
6096 }
6097
6098 #[must_use]
6100 pub fn parts(&self) -> usize {
6101 self.places.len()
6102 }
6103
6104 #[must_use]
6111 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
6112 let mut runs = Vec::with_capacity(self.table.stripes.len());
6113 let mut start = 0;
6114 for stripe in &self.table.stripes {
6115 let end = start + stripe.parts.len();
6116 runs.push(start..end);
6117 start = end;
6118 }
6119 runs
6120 }
6121
6122 #[must_use]
6127 pub fn stripe_rows(&self, stripe: usize) -> usize {
6128 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
6129 }
6130
6131 pub fn keep_stripes(&self, stripes: usize) {
6138 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
6139 }
6140
6141 #[must_use]
6143 pub fn part_rows(&self, at: usize) -> usize {
6144 self.places.get(at).map_or(0, |place| place.rows as usize)
6145 }
6146
6147 #[must_use]
6149 pub fn table(&self) -> &Table {
6150 &self.table
6151 }
6152
6153 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
6162 let field = self
6163 .table
6164 .fields
6165 .get(column)
6166 .ok_or_else(|| invalid("frequency column index out of range"))?;
6167 let Some(summary) = self.frequency_summary(column)? else {
6168 return Ok(None);
6169 };
6170 if top == 0 || summary.entries.len() < top {
6171 return Ok(None);
6172 }
6173 let boundary = summary.entries[top - 1].count;
6174 if boundary <= summary.omitted_max {
6175 return Ok(None);
6176 }
6177 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
6178 }
6179
6180 pub fn top_pair_frequencies(
6191 &self,
6192 first: usize,
6193 second: usize,
6194 top: usize,
6195 ) -> Result<Option<PairFrequencyCounts>> {
6196 if first >= self.table.fields.len() || second >= self.table.fields.len() {
6197 return Err(invalid("pair frequency column index out of range"));
6198 }
6199 let Some(summary) =
6200 self.table.pair_frequencies.iter().find(|summary| {
6201 summary.first as usize == first && summary.second as usize == second
6202 })
6203 else {
6204 return Ok(None);
6205 };
6206 if top == 0 || summary.entries.len() < top {
6207 return Ok(None);
6208 }
6209 let boundary = summary.entries[top - 1].count;
6210 if boundary <= summary.omitted_max {
6211 return Ok(None);
6212 }
6213 let first_summary = self
6214 .frequency_summary(first)?
6215 .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
6216 let anchors = self
6217 .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
6218 .into_iter()
6219 .map(|(value, _)| value)
6220 .collect::<Vec<_>>();
6221 let dictionary = self
6222 .dictionary(second)?
6223 .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
6224 let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
6225 codes.sort_unstable();
6226 codes.dedup();
6227 let texts = dictionary
6228 .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
6229 let mut out = Vec::with_capacity(summary.entries.len());
6230 for entry in &summary.entries {
6231 if entry.count < boundary {
6232 break;
6233 }
6234 let first = anchors
6235 .get(entry.first_entry as usize)
6236 .cloned()
6237 .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
6238 let second = match entry.second {
6239 None => Value::Null,
6240 Some(code) => {
6241 let at = codes
6242 .binary_search(&code)
6243 .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
6244 texts[at].clone()
6245 }
6246 };
6247 out.push((vec![first, second], entry.count));
6248 }
6249 Ok(Some(out))
6250 }
6251
6252 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
6272 let Some(prefix) = self.frequency_prefix(column)? else {
6273 return Ok(None);
6274 };
6275 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
6276 }
6277
6278 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
6301 let field = self
6302 .table
6303 .fields
6304 .get(column)
6305 .ok_or_else(|| invalid("frequency column index out of range"))?;
6306 let Some(summary) = self.frequency_summary(column)? else {
6307 return Ok(None);
6308 };
6309 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6310 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
6311 }
6312
6313 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
6315 Ok(match self.table.frequencies.get(column) {
6316 None | Some(None) => None,
6317 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
6318 Some(Some(Frequencies::Stored { span, values })) => {
6319 let slot = self
6320 .frequency_summaries
6321 .get(column)
6322 .ok_or_else(|| invalid("frequency column index out of range"))?;
6323 if let Some(summary) = slot.get() {
6324 return Ok(Some(Cow::Borrowed(summary.as_ref())));
6325 }
6326 let field = self
6327 .table
6328 .fields
6329 .get(column)
6330 .ok_or_else(|| invalid("frequency column index out of range"))?;
6331 let mut bytes = vec![0; span.length as usize];
6332 read_at(&self.file, span.offset, &mut bytes)?;
6333 let summary =
6334 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
6335 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
6336 let _ = slot.set(Arc::new(summary));
6337 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
6338 }
6339 })
6340 }
6341
6342 fn decode_frequencies(
6350 &self,
6351 column: usize,
6352 ty: &LogicalType,
6353 entries: &[FrequencyEntry],
6354 ) -> Result<Vec<(Value, u64)>> {
6355 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
6356 return Ok(values.as_ref().clone());
6357 }
6358 let values = self.decode_frequencies_once(column, ty, entries)?;
6359 if let Some(slot) = self.frequency_values.get(column) {
6360 let _ = slot.set(Arc::new(values.clone()));
6361 }
6362 Ok(values)
6363 }
6364
6365 fn decode_frequencies_once(
6366 &self,
6367 column: usize,
6368 ty: &LogicalType,
6369 entries: &[FrequencyEntry],
6370 ) -> Result<Vec<(Value, u64)>> {
6371 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
6372 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
6373 return Err(invalid("frequency text count differs from its synopsis"));
6374 }
6375 let dictionary =
6376 if coded_type(ty) && stored_texts.is_none() { self.dictionary(column)? } else { None };
6377 let mut codes = entries
6378 .iter()
6379 .filter_map(|entry| match entry.value {
6380 FrequencyValue::Code(code) => Some(code as usize),
6381 _ => None,
6382 })
6383 .collect::<Vec<_>>();
6384 codes.sort_unstable();
6385 codes.dedup();
6386 let texts = match &dictionary {
6387 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
6388 _ => Vec::new(),
6389 };
6390 let mut out = Vec::with_capacity(entries.len());
6391 for (entry_at, entry) in entries.iter().enumerate() {
6392 let value = match entry.value {
6393 FrequencyValue::Null => {
6394 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
6395 return Err(invalid("a null frequency entry has text"));
6396 }
6397 Value::Null
6398 }
6399 FrequencyValue::Integer(value) => match *ty {
6400 LogicalType::TinyInt => Value::TinyInt(
6401 i8::try_from(value)
6402 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
6403 ),
6404 LogicalType::UTinyInt => Value::UTinyInt(
6405 u8::try_from(value)
6406 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
6407 ),
6408 LogicalType::USmallInt => Value::USmallInt(
6409 u16::try_from(value)
6410 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
6411 ),
6412 LogicalType::UInteger => Value::UInteger(
6413 u32::try_from(value)
6414 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
6415 ),
6416 LogicalType::UBigInt => Value::UBigInt(
6417 u64::try_from(value)
6418 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
6419 ),
6420 LogicalType::SmallInt => Value::SmallInt(
6421 i16::try_from(value)
6422 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
6423 ),
6424 LogicalType::Integer => Value::Integer(
6425 i32::try_from(value)
6426 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
6427 ),
6428 LogicalType::BigInt => Value::BigInt(
6429 i64::try_from(value)
6430 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
6431 ),
6432 LogicalType::Date => Value::Date(
6433 i32::try_from(value)
6434 .map_err(|_| invalid("frequency DATE is out of range"))?,
6435 ),
6436 LogicalType::Timestamp => Value::Timestamp(
6437 i64::try_from(value)
6438 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
6439 ),
6440 _ => return Err(invalid("integer frequency belongs to another type")),
6441 },
6442 FrequencyValue::Code(code) => {
6443 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
6444 if *ty == LogicalType::Blob {
6445 Value::Blob(text.clone())
6446 } else {
6447 Value::Varchar(
6448 String::from_utf8(text.clone())
6449 .map_err(|_| invalid("frequency text is not UTF-8"))?,
6450 )
6451 }
6452 } else {
6453 if dictionary.is_none() {
6454 return Err(invalid("frequency code has no dictionary or stored text"));
6455 }
6456 let at = codes
6457 .binary_search(&(code as usize))
6458 .map_err(|_| invalid("frequency code was not among the codes read"))?;
6459 texts[at].clone()
6460 }
6461 }
6462 };
6463 out.push((value, entry.count));
6464 }
6465 Ok(out)
6466 }
6467
6468 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
6478 let field = self
6479 .table
6480 .fields
6481 .get(column)
6482 .ok_or_else(|| invalid("frequency column index out of range"))?;
6483 let Some(summary) = self.frequency_summary(column)? else {
6484 return Ok(None);
6485 };
6486 if summary.ordinals.is_empty() {
6487 return Ok(None);
6488 }
6489 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
6490 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
6491 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
6492 } else {
6493 (Vec::new(), Vec::new())
6494 };
6495 Ok(Some(FrequencyOccurrences {
6496 omitted_max: summary.omitted_max,
6497 ordinals: summary.ordinals.clone(),
6498 anchors,
6499 anchor_indices,
6500 }))
6501 }
6502
6503 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
6529 self.table
6530 .distincts
6531 .get(column)
6532 .copied()
6533 .ok_or_else(|| invalid("distinct column index out of range"))
6534 }
6535
6536 pub fn null_count(&self, column: usize) -> Result<u64> {
6547 if column >= self.table.fields.len() {
6548 return Err(invalid("null count column index out of range"));
6549 }
6550 let mut nulls = 0_u64;
6551 for stripe in &self.table.stripes {
6552 let range = stripe
6553 .zone
6554 .column(column)
6555 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6556 nulls = nulls
6557 .checked_add(range.nulls as u64)
6558 .ok_or_else(|| invalid("null count overflow"))?;
6559 }
6560 Ok(nulls)
6561 }
6562
6563 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
6578 if self.null_count(column)? > 0 || self.demoted(column) {
6579 return Ok(None);
6580 }
6581 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
6582 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
6583 if ranks == 0 {
6584 return Ok(None);
6585 }
6586 let low = text_at_rank(&dictionary, 0)?;
6587 let high = text_at_rank(&dictionary, ranks - 1)?;
6588 Ok(Some((low, high)))
6589 }
6590
6591 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
6614 if column >= self.table.fields.len() {
6615 return Err(invalid("extremes column index out of range"));
6616 }
6617 let mut low: Option<Bound> = None;
6618 let mut high: Option<Bound> = None;
6619 for stripe in &self.table.stripes {
6620 let range = stripe
6621 .zone
6622 .column(column)
6623 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6624 if !range.exact {
6625 return Ok(None);
6626 }
6627 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
6632 if stripe.rows > range.nulls {
6633 return Ok(None);
6634 }
6635 continue;
6636 };
6637 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
6638 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
6639 }
6640 Ok(low.zip(high))
6641 }
6642
6643 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
6656 if column >= self.table.fields.len() {
6657 return Err(invalid("sum column index out of range"));
6658 }
6659 let mut total = 0_i128;
6660 let mut rows = 0_u64;
6661 for stripe in &self.table.stripes {
6662 let range = stripe
6663 .zone
6664 .column(column)
6665 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
6666 let Some(part) = range.sum else { return Ok(None) };
6667 let Some(sum) = total.checked_add(part) else { return Ok(None) };
6668 total = sum;
6669 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
6670 }
6671 Ok(Some((total, rows)))
6672 }
6673
6674 pub fn host_groups(
6677 &self,
6678 column: usize,
6679 minimum_count: u64,
6680 ) -> Result<Option<Vec<host::HostEntry>>> {
6681 if column >= self.table.fields.len() {
6682 return Err(invalid("host group column index out of range"));
6683 }
6684 let Some(summary) = &self.table.host_groups else { return Ok(None) };
6685 if summary.column != column || minimum_count <= summary.omitted_max {
6686 return Ok(None);
6687 }
6688 Ok(Some(summary.entries.clone()))
6689 }
6690
6691 #[must_use]
6695 pub fn demoted(&self, column: usize) -> bool {
6696 self.table.demoted.get(column).copied().unwrap_or(false)
6697 }
6698
6699 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
6708 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
6709 if let Some(dictionary) = self.dictionaries[column].get() {
6710 return Ok(Some(Arc::clone(dictionary)));
6711 }
6712 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6713 if let Some(dictionary) = self.dictionaries[column].get() {
6714 return Ok(Some(Arc::clone(dictionary)));
6715 }
6716 self.opened.fetch_add(1, Atomic::Relaxed);
6717 let dictionary = Arc::new(open_global_dictionary(
6718 Arc::clone(&self.file),
6719 page,
6720 &self.table.fields[column].ty,
6721 TEXT_KEEP_BUDGET,
6722 )?);
6723 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6724 Ok(Some(dictionary))
6725 }
6726
6727 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6734 if of.extent_bytes == 0 {
6735 return Ok(Vec::new());
6736 }
6737 let mut bytes = vec![0; of.extent_bytes as usize];
6738 read_at(&self.file, of.extent_page, &mut bytes)?;
6739 if checksum(&bytes) != of.hash {
6740 return Err(invalid("a section's extent table does not checksum"));
6741 }
6742 let extents = section::decode_extents(&bytes)?;
6743 if extents.len() != of.extents as usize {
6744 return Err(invalid("a section's extent table is not the length the entry says"));
6745 }
6746 Ok(extents)
6747 }
6748
6749 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
6759 let end = of
6760 .offset
6761 .checked_add(u64::from(of.length))
6762 .ok_or_else(|| invalid("an extent overflows the file"))?;
6763 if of.offset < HEADER || end > self.size {
6764 return Err(invalid("an extent is outside the file"));
6765 }
6766 let mut bytes = vec![0; of.length as usize];
6767 read_at(&self.file, of.offset, &mut bytes)?;
6768 if checksum(&bytes) != of.hash {
6769 return Err(invalid("an extent does not checksum"));
6770 }
6771 Ok(bytes)
6772 }
6773
6774 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6783 let extents = self.extents(of)?;
6784 let mut bytes =
6785 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6786 for one in &extents {
6787 if one.first != bytes.len() as u64 {
6788 return Err(invalid("a section's extents do not join up"));
6789 }
6790 bytes.extend_from_slice(&self.extent(one)?);
6791 }
6792 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6795 return Err(invalid("a section's header is longer than its payload"));
6796 }
6797 Ok(bytes)
6798 }
6799
6800 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6809 self.read_impl(part, columns, true, None)
6810 }
6811
6812 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6822 self.read_impl(part, columns, false, None)
6823 }
6824
6825 pub fn integer_tally(&self, part: usize, column: usize) -> Result<Option<Vec<(i64, u64)>>> {
6833 let place = *self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6834 let field =
6835 self.table.fields.get(column).ok_or_else(|| invalid("column index out of range"))?;
6836 if !matches!(
6837 field.ty,
6838 LogicalType::TinyInt
6839 | LogicalType::SmallInt
6840 | LogicalType::Integer
6841 | LogicalType::BigInt
6842 ) {
6843 return Ok(None);
6844 }
6845 let stripe_index = place.stripe as usize;
6846 let stripe = self
6847 .table
6848 .stripes
6849 .get(stripe_index)
6850 .ok_or_else(|| invalid("stripe index out of range"))?;
6851 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6852 let held = self.held(stripe_index, stripe, column, true)?;
6853 let span = *held
6854 .index
6855 .get(place.part as usize)
6856 .ok_or_else(|| invalid("part index out of range"))?;
6857 let owned;
6858 let bytes = match &held.page {
6859 Some(page) => part_bytes(page, span)?,
6860 None => {
6861 let offset = page
6862 .offset
6863 .checked_add(span.start as u64)
6864 .ok_or_else(|| invalid("part range overflow"))?;
6865 let mut bytes = vec![0; span.length];
6866 read_at(&self.file, offset, &mut bytes)?;
6867 owned = bytes;
6868 &owned
6869 }
6870 };
6871 if checksum(bytes) != span.hash {
6872 return Err(invalid("integer part checksum differs"));
6873 }
6874 if bytes.first() != Some(&5) || bytes.get(1) != Some(&0) {
6875 return Ok(None);
6876 }
6877 let (rows, counts) = integer::tally(&bytes[2..])?;
6878 if rows != place.rows as usize {
6879 return Err(invalid("encoded integer part holds the wrong number of rows"));
6880 }
6881 for &(value, _) in &counts {
6882 let fits = match field.ty {
6883 LogicalType::TinyInt => i8::try_from(value).is_ok(),
6884 LogicalType::SmallInt => i16::try_from(value).is_ok(),
6885 LogicalType::Integer => i32::try_from(value).is_ok(),
6886 LogicalType::BigInt => true,
6887 _ => false,
6888 };
6889 if !fits {
6890 return Err(invalid("encoded integer value is outside its column type"));
6891 }
6892 }
6893 Ok(Some(counts))
6894 }
6895
6896 pub fn read_rows(
6909 &self,
6910 part: usize,
6911 columns: &[usize],
6912 positions: &[u32],
6913 whole: bool,
6914 ) -> Result<Chunk> {
6915 self.read_impl(part, columns, whole, Some(positions))
6916 }
6917
6918 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6925 if self.demoted(column) {
6928 return Ok(false);
6929 }
6930 if candidates.is_empty() {
6931 return Ok(true);
6932 }
6933 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6934 return Err(Error::internal("native code candidates are not sorted and unique"));
6935 }
6936 let stripe = self.stripe_of(part)?;
6937 let Some(page) = stripe.memberships.get(column) else {
6938 return Ok(false);
6939 };
6940 let mut bytes = vec![0; page.length as usize];
6941 read_at(&self.file, page.offset, &mut bytes)?;
6942 if checksum(&bytes) != page.hash {
6943 return Err(invalid("membership page checksum differs"));
6944 }
6945 let codes = decode_membership(&bytes)?;
6946 let mut left = 0;
6947 let mut right = 0;
6948 while left < codes.len() && right < candidates.len() {
6949 match codes[left].cmp(&candidates[right]) {
6950 Ordering::Less => left += 1,
6951 Ordering::Greater => right += 1,
6952 Ordering::Equal => return Ok(false),
6953 }
6954 }
6955 Ok(true)
6956 }
6957
6958 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6959 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6960 self.table
6961 .stripes
6962 .get(place.stripe as usize)
6963 .ok_or_else(|| invalid("stripe index out of range"))
6964 }
6965
6966 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6983 let cache =
6984 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6985 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6986 let known = cached.index.get(at).and_then(Clone::clone);
6987 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6988 slot.used.store(true, Atomic::Relaxed);
6989 Arc::clone(&slot.page)
6990 });
6991 if let Some(index) = known.clone() {
6992 if !whole || page.is_some() {
6993 return Ok(CachedColumn { stripe: at, index, page });
6994 }
6995 }
6996 if cached.loading.contains(&at) {
6997 drop(cached);
6998 if let Some(index) = known {
7002 return Ok(CachedColumn { stripe: at, index, page: None });
7003 }
7004 let held = self.page_of(stripe, column, at, false, None)?;
7005 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7006 remember(&mut cached, &held);
7007 return Ok(held);
7008 }
7009 cached.loading.push(at);
7010 drop(cached);
7011
7012 let read = self.page_of(stripe, column, at, whole, known);
7013
7014 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
7018 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
7019 cached.loading.remove(position);
7020 }
7021 let held = read?;
7022 let taken = remember(&mut cached, &held);
7023 drop(cached);
7024 if let Some((bytes, used)) = taken {
7025 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
7026 self.pool.admit(Held {
7027 shelf: Arc::downgrade(&self.cache),
7028 column,
7029 stripe: at,
7030 bytes,
7031 used,
7032 });
7033 }
7034 Ok(held)
7035 }
7036
7037 fn page_of(
7043 &self,
7044 stripe: &Stripe,
7045 column: usize,
7046 at: usize,
7047 whole: bool,
7048 known: Option<Arc<Vec<PartSpan>>>,
7049 ) -> Result<CachedColumn> {
7050 let index = match known {
7051 Some(index) => index,
7052 None => {
7053 self.indexes.fetch_add(1, Atomic::Relaxed);
7054 Arc::new(read_index(&self.file, stripe, column)?)
7055 }
7056 };
7057 let page = if whole {
7058 self.pages.fetch_add(1, Atomic::Relaxed);
7059 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7060 let mut bytes = vec![0; span.length as usize];
7061 read_at(&self.file, span.offset, &mut bytes)?;
7062 Some(Arc::new(bytes))
7063 } else {
7064 None
7065 };
7066 Ok(CachedColumn { stripe: at, index, page })
7067 }
7068
7069 fn read_impl(
7070 &self,
7071 at: usize,
7072 columns: &[usize],
7073 whole: bool,
7074 positions: Option<&[u32]>,
7075 ) -> Result<Chunk> {
7076 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
7077 let index = place.stripe as usize;
7078 let stripe =
7079 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
7080 let rows = place.rows as usize;
7081 let mut picked = Vec::with_capacity(columns.len());
7082 for &column in columns {
7083 let field = self
7084 .table
7085 .fields
7086 .get(column)
7087 .ok_or_else(|| invalid("column index out of range"))?;
7088 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
7089 let held = self.held(index, stripe, column, whole)?;
7090 let span = *held
7091 .index
7092 .get(place.part as usize)
7093 .ok_or_else(|| invalid("part index out of range"))?;
7094 let owned;
7095 let bytes = match &held.page {
7096 Some(held) => part_bytes(held, span)?,
7097 None => {
7098 let offset = page
7099 .offset
7100 .checked_add(span.start as u64)
7101 .ok_or_else(|| invalid("part range overflow"))?;
7102 let mut bytes = vec![0; span.length];
7103 read_at(&self.file, offset, &mut bytes)?;
7104 owned = bytes;
7105 &owned
7106 }
7107 };
7108 if checksum(bytes) != span.hash {
7109 return Err(invalid(&format!(
7110 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
7111 wanted {:016x} and got {:016x}",
7112 place.part,
7113 page.offset,
7114 span.start,
7115 span.length,
7116 span.hash,
7117 checksum(bytes),
7118 )));
7119 }
7120 let dictionary = self.dictionary(column)?;
7121 let mut vector = match positions {
7127 None => decode(&field.ty, rows, bytes, dictionary)?,
7128 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
7129 };
7130 if self.demoted(column) && vector.stable_dictionary_parts().is_some() {
7134 vector = vector.flatten()?;
7135 }
7136 picked.push(vector.into_pages());
7137 }
7138 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
7139 }
7140
7141 #[must_use]
7157 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
7158 let Some(place) = self.places.get(part).copied() else { return false };
7159 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7160 if stripe.zone.skips(probes) {
7161 return true;
7162 }
7163 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
7164 }
7165
7166 fn outside(&self, place: Place, probe: &Probe) -> bool {
7172 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7173 Some(ranges) => ranges
7174 .get(place.part as usize)
7175 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
7176 None => false,
7177 }
7178 }
7179
7180 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
7186 let slot = self.part_ranges.get(column)?.get(stripe)?;
7187 if let Some(held) = slot.get() {
7188 return Some(held);
7189 }
7190 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
7191 let mut bytes = vec![0; page.length as usize];
7192 read_at(&self.file, page.offset, &mut bytes).ok()?;
7193 if checksum(&bytes) != page.hash {
7194 return None;
7195 }
7196 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
7197 let _ = slot.set(ranges);
7198 slot.get().map(|held| held.as_slice())
7199 }
7200
7201 #[must_use]
7218 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
7219 let Some(place) = self.places.get(part).copied() else { return false };
7220 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
7221 if stripe.zone.certain(probes) {
7222 return true;
7223 }
7224 probes
7225 .iter()
7226 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
7227 }
7228
7229 fn inside(&self, place: Place, probe: &Probe) -> bool {
7235 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
7236 Some(ranges) => ranges
7237 .get(place.part as usize)
7238 .is_some_and(|range| range.certain(probe.op, &probe.value)),
7239 None => false,
7240 }
7241 }
7242
7243 #[must_use]
7254 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
7255 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
7256 }
7257
7258 fn sifted(&self, place: Place, probe: &Probe) -> bool {
7264 if probe.op != Op::Equal {
7265 return false;
7266 }
7267 match self.stripe_sieves(place.stripe as usize, probe.column) {
7268 Some(sieves) => sieves
7269 .get(place.part as usize)
7270 .and_then(Option::as_ref)
7271 .is_some_and(|sieve| sieve.excludes(&probe.value)),
7272 None => false,
7273 }
7274 }
7275
7276 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
7283 let slot = self.sieves.get(column)?.get(stripe)?;
7284 if let Some(held) = slot.get() {
7285 return Some(held);
7286 }
7287 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
7288 let mut bytes = vec![0; page.length as usize];
7289 read_at(&self.file, page.offset, &mut bytes).ok()?;
7290 if checksum(&bytes) != page.hash {
7291 return None;
7292 }
7293 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
7294 let _ = slot.set(sieves);
7295 slot.get().map(|held| held.as_slice())
7296 }
7297}
7298
7299fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
7301 let code = dictionary.code_at_rank(rank)? as usize;
7302 if dictionary.logical_type() == &LogicalType::Blob {
7303 let bytes = dictionary
7304 .try_bytes_at(code)?
7305 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7306 return Ok(Value::Blob(bytes.to_vec()));
7307 }
7308 let text = dictionary
7309 .try_text_at(code)?
7310 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
7311 Ok(Value::Varchar(text.into()))
7312}
7313
7314fn read_at<F: Positional + ?Sized>(file: &F, offset: u64, bytes: &mut [u8]) -> Result<()> {
7324 file.fill_at(offset, bytes)
7325}
7326
7327trait Positional {
7335 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()>;
7340}
7341
7342impl<T: Positional + ?Sized> Positional for &T {
7343 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7344 (**self).fill_at(offset, bytes)
7345 }
7346}
7347
7348impl<T: Positional + ?Sized> Positional for Arc<T> {
7349 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7350 (**self).fill_at(offset, bytes)
7351 }
7352}
7353
7354impl<T: Positional + ?Sized> Positional for Box<T> {
7355 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7356 (**self).fill_at(offset, bytes)
7357 }
7358}
7359
7360impl Positional for dyn rudb_io::File + '_ {
7361 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7362 while !bytes.is_empty() {
7363 let read = self.read_at(offset, bytes)?;
7364 if read == 0 {
7365 return Err(invalid("column page ends before its declared length"));
7366 }
7367 offset += read as u64;
7368 bytes = &mut bytes[read..];
7369 }
7370 Ok(())
7371 }
7372}
7373
7374impl Positional for File {
7375 #[cfg(unix)]
7376 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7377 use std::os::unix::fs::FileExt;
7378 while !bytes.is_empty() {
7379 let read = self.read_at(bytes, offset).map_err(io)?;
7380 if read == 0 {
7381 return Err(invalid("column page ends before its declared length"));
7382 }
7383 offset += read as u64;
7384 bytes = &mut bytes[read..];
7385 }
7386 Ok(())
7387 }
7388
7389 #[cfg(windows)]
7395 fn fill_at(&self, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
7396 use std::os::windows::fs::FileExt;
7397 while !bytes.is_empty() {
7398 let read = self.seek_read(bytes, offset).map_err(io)?;
7399 if read == 0 {
7400 return Err(invalid("column page ends before its declared length"));
7401 }
7402 offset += read as u64;
7403 bytes = &mut bytes[read..];
7404 }
7405 Ok(())
7406 }
7407
7408 #[cfg(not(any(unix, windows)))]
7413 fn fill_at(&self, offset: u64, bytes: &mut [u8]) -> Result<()> {
7414 use std::io::{Read, Seek, SeekFrom};
7415 let mut file = self.try_clone().map_err(io)?;
7416 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7417 file.read_exact(bytes).map_err(io)
7418 }
7419}
7420
7421#[cfg(test)]
7426fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
7427 use std::io::{Seek, SeekFrom, Write};
7428 let mut file = file;
7429 file.seek(SeekFrom::Start(offset)).map_err(io)?;
7430 file.write_all(bytes).map_err(io)
7431}
7432
7433fn type_tag(ty: &LogicalType) -> Result<u8> {
7440 match ty {
7441 LogicalType::SmallInt => Ok(1),
7442 LogicalType::Integer => Ok(2),
7443 LogicalType::BigInt => Ok(3),
7444 LogicalType::Varchar => Ok(4),
7445 LogicalType::Date => Ok(5),
7446 LogicalType::Timestamp => Ok(6),
7447 LogicalType::Boolean => Ok(7),
7448 LogicalType::TinyInt => Ok(8),
7449 LogicalType::UTinyInt => Ok(9),
7450 LogicalType::USmallInt => Ok(10),
7451 LogicalType::UInteger => Ok(11),
7452 LogicalType::UBigInt => Ok(12),
7453 LogicalType::Decimal { .. } => Ok(13),
7454 LogicalType::Float => Ok(14),
7455 LogicalType::Double => Ok(15),
7456 LogicalType::HugeInt => Ok(16),
7457 LogicalType::UHugeInt => Ok(17),
7458 LogicalType::Time => Ok(18),
7459 LogicalType::TimeTz => Ok(19),
7460 LogicalType::TimestampTz => Ok(20),
7461 LogicalType::Interval => Ok(21),
7462 LogicalType::Uuid => Ok(22),
7463 LogicalType::Blob => Ok(23),
7464 LogicalType::Bit => Ok(24),
7465 LogicalType::TimestampS => Ok(25),
7466 LogicalType::TimestampMs => Ok(26),
7467 LogicalType::TimestampNs => Ok(27),
7468 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
7469 }
7470}
7471
7472fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
7478 out.push(type_tag(ty)?);
7479 if let LogicalType::Decimal { width, scale } = ty {
7480 out.push(*width);
7481 out.push(*scale);
7482 }
7483 Ok(())
7484}
7485
7486fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
7488 let tag = cur.u8()?;
7489 if tag == 13 {
7490 let width = cur.u8()?;
7491 let scale = cur.u8()?;
7492 return LogicalType::decimal(width, scale)
7493 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
7494 }
7495 tag_type(tag)
7496}
7497
7498fn tag_type(tag: u8) -> Result<LogicalType> {
7499 match tag {
7500 1 => Ok(LogicalType::SmallInt),
7501 2 => Ok(LogicalType::Integer),
7502 3 => Ok(LogicalType::BigInt),
7503 4 => Ok(LogicalType::Varchar),
7504 5 => Ok(LogicalType::Date),
7505 6 => Ok(LogicalType::Timestamp),
7506 7 => Ok(LogicalType::Boolean),
7507 8 => Ok(LogicalType::TinyInt),
7508 9 => Ok(LogicalType::UTinyInt),
7509 10 => Ok(LogicalType::USmallInt),
7510 11 => Ok(LogicalType::UInteger),
7511 12 => Ok(LogicalType::UBigInt),
7512 14 => Ok(LogicalType::Float),
7513 15 => Ok(LogicalType::Double),
7514 16 => Ok(LogicalType::HugeInt),
7515 17 => Ok(LogicalType::UHugeInt),
7516 18 => Ok(LogicalType::Time),
7517 19 => Ok(LogicalType::TimeTz),
7518 20 => Ok(LogicalType::TimestampTz),
7519 21 => Ok(LogicalType::Interval),
7520 22 => Ok(LogicalType::Uuid),
7521 23 => Ok(LogicalType::Blob),
7522 24 => Ok(LogicalType::Bit),
7523 25 => Ok(LogicalType::TimestampS),
7524 26 => Ok(LogicalType::TimestampMs),
7525 27 => Ok(LogicalType::TimestampNs),
7526 _ => Err(invalid("column type tag is unknown")),
7527 }
7528}
7529
7530fn put_u16(out: &mut Vec<u8>, value: u16) {
7531 out.extend_from_slice(&value.to_le_bytes());
7532}
7533fn put_u32(out: &mut Vec<u8>, value: u32) {
7534 out.extend_from_slice(&value.to_le_bytes());
7535}
7536fn put_u64(out: &mut Vec<u8>, value: u64) {
7537 out.extend_from_slice(&value.to_le_bytes());
7538}
7539fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
7540 while value >= 0x80 {
7541 out.push((value as u8 & 0x7f) | 0x80);
7542 value >>= 7;
7543 }
7544 out.push(value as u8);
7545}
7546
7547fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
7548 match (left, right) {
7549 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
7550 (FrequencyValue::Null, _) => Ordering::Less,
7551 (_, FrequencyValue::Null) => Ordering::Greater,
7552 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
7553 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
7554 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
7555 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
7556 }
7557}
7558
7559fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
7572 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
7573 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
7574 };
7575 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
7576 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
7577 let omitted_max = next.count;
7578 entries.truncate(FREQUENCY_ENTRIES);
7579 omitted_max
7580 } else {
7581 0
7582 };
7583 entries.sort_unstable_by(order);
7584 omitted_max
7585}
7586
7587fn code_frequency(
7588 dictionary: &GlobalDictionary,
7589 flat: &[u8],
7590 bases: &[u64],
7591) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
7592 let mut entries = dictionary
7593 .counts
7594 .iter()
7595 .enumerate()
7596 .filter(|(_, count)| **count != 0)
7597 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
7598 .collect::<Vec<_>>();
7599 if dictionary.nulls != 0 {
7600 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
7601 }
7602 let omitted_max = keep_most_frequent(&mut entries);
7603 let mut spans = Vec::with_capacity(entries.len());
7604 let mut text_bytes = 0_usize;
7605 for entry in &entries {
7606 let span = match entry.value {
7607 FrequencyValue::Code(code) => {
7608 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
7609 let bytes = flat
7610 .get(span.0..span.1)
7611 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
7612 text_bytes = text_bytes.saturating_add(bytes.len());
7613 Some(span)
7614 }
7615 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
7616 };
7617 spans.push(span);
7618 }
7619 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
7620 Vec::new()
7621 } else {
7622 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
7623 };
7624 Ok((
7625 FrequencySummary {
7626 entries,
7627 omitted_max,
7628 ordinals: Vec::new(),
7629 ordinal_entries: Vec::new(),
7630 },
7631 texts,
7632 ))
7633}
7634
7635fn encode_directory(table: &Table) -> Result<Vec<u8>> {
7636 let mut out = DIRECTORY.to_vec();
7637 let name = table.name.as_bytes();
7638 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7639 out.extend_from_slice(name);
7640 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
7641 for field in &table.fields {
7642 let name = field.name.as_bytes();
7643 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
7644 out.extend_from_slice(name);
7645 put_type(&mut out, &field.ty)?;
7646 out.push(u8::from(field.not_null));
7647 }
7648 for (field, dictionary) in table.fields.iter().zip(&table.dictionaries) {
7649 match dictionary {
7650 None => out.push(0),
7651 Some(page) => {
7652 out.push(dictionary_tag(&field.ty));
7653 put_u64(&mut out, page.offset);
7654 put_u32(&mut out, page.length);
7655 put_u64(&mut out, page.hash);
7656 }
7657 }
7658 }
7659 for distinct in &table.distincts {
7660 match distinct {
7661 None => out.push(0),
7662 Some(count) => {
7663 out.push(1);
7664 put_u64(&mut out, *count);
7665 }
7666 }
7667 }
7668 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
7669 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
7670 for stripe in &table.stripes {
7671 put_u32(
7672 &mut out,
7673 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
7674 );
7675 for &rows in &stripe.parts {
7676 put_u32(&mut out, rows);
7677 }
7678 put_u64(&mut out, stripe.index.offset);
7679 put_u32(&mut out, stripe.index.length);
7680 for page in &stripe.pages {
7681 put_u64(&mut out, page.offset);
7682 put_u32(&mut out, page.length);
7683 }
7684 for (column, ((field, dictionary), membership)) in
7689 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots()).enumerate()
7690 {
7691 if !coded_type(&field.ty) || dictionary.is_none() {
7692 continue;
7693 }
7694 let page = match membership {
7695 Some(page) => page,
7696 None if table.demoted.get(column).copied().unwrap_or(false) => {
7697 Page { offset: HEADER, length: 0, hash: 0 }
7698 }
7699 None => return Err(invalid("string page has no code membership index")),
7700 };
7701 put_u64(&mut out, page.offset);
7702 put_u32(&mut out, page.length);
7703 put_u64(&mut out, page.hash);
7704 }
7705 for sieve in stripe.sieves.slots() {
7706 match sieve {
7707 None => out.push(0),
7708 Some(page) => {
7709 out.push(1);
7710 put_u64(&mut out, page.offset);
7711 put_u32(&mut out, page.length);
7712 put_u64(&mut out, page.hash);
7713 }
7714 }
7715 }
7716 for held in stripe.part_ranges.slots() {
7717 match held {
7718 None => out.push(0),
7719 Some(page) => {
7720 out.push(1);
7721 put_u64(&mut out, page.offset);
7722 put_u32(&mut out, page.length);
7723 put_u64(&mut out, page.hash);
7724 }
7725 }
7726 }
7727 for range in stripe.zone.columns() {
7728 put_bound(&mut out, range.low.as_ref())?;
7729 put_bound(&mut out, range.high.as_ref())?;
7730 put_u32(
7731 &mut out,
7732 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
7733 );
7734 out.push(u8::from(range.exact));
7735 match range.sum {
7736 None => out.push(0),
7737 Some(total) => {
7738 out.push(1);
7739 out.extend_from_slice(&total.to_le_bytes());
7740 }
7741 }
7742 }
7743 }
7744 out.extend_from_slice(FREQUENCIES);
7745 put_u16(
7746 &mut out,
7747 u16::try_from(table.frequencies.len())
7748 .map_err(|_| invalid("too many frequency columns"))?,
7749 );
7750 for summary in &table.frequencies {
7751 let summary = match summary {
7752 None => {
7753 out.push(0);
7754 continue;
7755 }
7756 Some(Frequencies::Held(summary)) => summary,
7757 Some(Frequencies::Stored { .. }) => {
7759 return Err(invalid("a synopsis left in the file cannot be written back"));
7760 }
7761 };
7762 out.push(1);
7763 put_u64(&mut out, summary.omitted_max);
7764 put_u32(
7765 &mut out,
7766 u32::try_from(summary.entries.len())
7767 .map_err(|_| invalid("too many frequency entries"))?,
7768 );
7769 for entry in &summary.entries {
7770 match entry.value {
7771 FrequencyValue::Null => out.push(0),
7772 FrequencyValue::Integer(value) => {
7773 out.push(1);
7774 out.extend_from_slice(&value.to_le_bytes());
7775 }
7776 FrequencyValue::Code(value) => {
7777 out.push(2);
7778 put_u32(&mut out, value);
7779 }
7780 }
7781 put_u64(&mut out, entry.count);
7782 }
7783 put_u32(
7784 &mut out,
7785 u32::try_from(summary.ordinals.len())
7786 .map_err(|_| invalid("too many frequency ordinals"))?,
7787 );
7788 let mut previous = 0_u64;
7789 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
7790 let delta = if at == 0 {
7791 ordinal
7792 } else {
7793 ordinal
7794 .checked_sub(previous)
7795 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
7796 };
7797 if at != 0 && delta == 0 {
7798 return Err(invalid("frequency ordinals are not unique"));
7799 }
7800 put_var_u64(&mut out, delta);
7801 previous = ordinal;
7802 }
7803 if summary.ordinal_entries.len() != summary.ordinals.len() {
7804 return Err(invalid("frequency ordinal values have a different length"));
7805 }
7806 for &entry in &summary.ordinal_entries {
7807 if entry as usize >= summary.entries.len() {
7808 return Err(invalid("frequency ordinal value is outside its entries"));
7809 }
7810 put_u16(&mut out, entry);
7811 }
7812 }
7813 if !table.pair_frequencies.is_empty() {
7814 out.extend_from_slice(PAIR_FREQUENCIES);
7815 put_u16(
7816 &mut out,
7817 u16::try_from(table.pair_frequencies.len())
7818 .map_err(|_| invalid("too many pair frequency summaries"))?,
7819 );
7820 for summary in &table.pair_frequencies {
7821 put_u16(&mut out, summary.first);
7822 put_u16(&mut out, summary.second);
7823 put_u64(&mut out, summary.omitted_max);
7824 put_u16(
7825 &mut out,
7826 u16::try_from(summary.entries.len())
7827 .map_err(|_| invalid("too many pair frequency entries"))?,
7828 );
7829 for entry in &summary.entries {
7830 put_u16(&mut out, entry.first_entry);
7831 match entry.second {
7832 None => out.push(0),
7833 Some(code) => {
7834 out.push(1);
7835 put_u32(&mut out, code);
7836 }
7837 }
7838 put_u64(&mut out, entry.count);
7839 }
7840 }
7841 }
7842 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7843 if text_columns != 0 {
7844 out.extend_from_slice(FREQUENCY_TEXTS);
7845 put_u16(
7846 &mut out,
7847 u16::try_from(text_columns)
7848 .map_err(|_| invalid("too many string frequency columns"))?,
7849 );
7850 for (column, texts) in table.frequency_texts.iter().enumerate() {
7851 if texts.is_empty() {
7852 continue;
7853 }
7854 put_u16(
7855 &mut out,
7856 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7857 );
7858 put_u16(
7859 &mut out,
7860 u16::try_from(texts.len())
7861 .map_err(|_| invalid("too many frequency text entries"))?,
7862 );
7863 for text in texts {
7864 match text {
7865 None => out.push(0),
7866 Some(text) => {
7867 out.push(1);
7868 put_u32(
7869 &mut out,
7870 u32::try_from(text.len())
7871 .map_err(|_| invalid("frequency text is too long"))?,
7872 );
7873 out.extend_from_slice(text);
7874 }
7875 }
7876 }
7877 }
7878 }
7879 if let Some(summary) = &table.host_groups {
7880 out.extend_from_slice(HOST_GROUPS);
7881 put_u16(
7882 &mut out,
7883 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7884 );
7885 put_u64(&mut out, summary.omitted_max);
7886 put_u16(
7887 &mut out,
7888 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7889 );
7890 for entry in &summary.entries {
7891 put_u32(
7892 &mut out,
7893 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7894 );
7895 out.extend_from_slice(entry.host.as_bytes());
7896 put_u64(&mut out, entry.count);
7897 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7898 put_u32(
7899 &mut out,
7900 u32::try_from(entry.minimum.len())
7901 .map_err(|_| invalid("host minimum is too long"))?,
7902 );
7903 out.extend_from_slice(entry.minimum.as_bytes());
7904 }
7905 }
7906 if let Some(clustering) = &table.clustering {
7909 out.extend_from_slice(CLUSTERING);
7910 out.push(clustering.width().tag());
7911 put_u16(
7912 &mut out,
7913 u16::try_from(clustering.columns().len())
7914 .map_err(|_| invalid("too many clustering columns"))?,
7915 );
7916 for &column in clustering.columns() {
7917 put_u16(
7918 &mut out,
7919 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7920 );
7921 }
7922 }
7923 let demoted = (0..table.fields.len())
7924 .filter(|&column| table.demoted.get(column).copied().unwrap_or(false))
7925 .collect::<Vec<_>>();
7926 if !demoted.is_empty() {
7927 out.extend_from_slice(DEMOTED);
7928 put_u16(
7929 &mut out,
7930 u16::try_from(demoted.len()).map_err(|_| invalid("too many demoted columns"))?,
7931 );
7932 for column in demoted {
7933 put_u16(
7934 &mut out,
7935 u16::try_from(column).map_err(|_| invalid("demoted column index overflow"))?,
7936 );
7937 }
7938 }
7939 out.extend_from_slice(SECTIONS);
7945 put_u64(&mut out, table.generation);
7946 put_u16(
7947 &mut out,
7948 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7949 );
7950 for held in &table.sections {
7951 held.encode(&mut out)?;
7952 }
7953 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7954 out.extend_from_slice(DICTIONARY_PAYLOADS);
7955 put_u16(
7956 &mut out,
7957 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7958 );
7959 for at in 0..table.fields.len() {
7960 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7961 }
7962 }
7963 Ok(out)
7964}
7965
7966fn signed_integer(ty: &LogicalType) -> bool {
7975 matches!(
7976 ty,
7977 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7978 )
7979}
7980
7981fn integer_or_date(ty: &LogicalType) -> bool {
7982 matches!(
7983 ty,
7984 LogicalType::TinyInt
7985 | LogicalType::SmallInt
7986 | LogicalType::Integer
7987 | LogicalType::BigInt
7988 | LogicalType::UTinyInt
7989 | LogicalType::USmallInt
7990 | LogicalType::UInteger
7991 | LogicalType::UBigInt
7992 | LogicalType::Date
7993 )
7994}
7995
7996fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7997 table
7998 .fields
7999 .iter()
8000 .enumerate()
8001 .map(|(column, field)| {
8002 if !integer_or_date(&field.ty) {
8003 return None;
8004 }
8005 let mut low: Option<i128> = None;
8006 let mut high: Option<i128> = None;
8007 for stripe in &table.stripes {
8008 let range = stripe.zone.column(column)?;
8009 if !range.exact {
8010 return None;
8011 }
8012 match (range.low.as_ref(), range.high.as_ref()) {
8013 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
8014 low = Some(low.map_or(*small, |held| held.min(*small)));
8015 high = Some(high.map_or(*large, |held| held.max(*large)));
8016 }
8017 (None, None) if stripe.rows == range.nulls => {}
8018 _ => return None,
8019 }
8020 }
8021 Some(low.zip(high))
8022 })
8023 .collect()
8024}
8025
8026fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
8027 reader
8028 .table
8029 .fields
8030 .iter()
8031 .enumerate()
8032 .map(|(column, field)| {
8033 if !integer_or_date(&field.ty) {
8034 return Ok(None);
8035 }
8036 match reader.exact_extremes(column)? {
8037 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
8038 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
8039 _ => Ok(None),
8040 }
8041 })
8042 .collect()
8043}
8044
8045fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
8046 table
8047 .fields
8048 .iter()
8049 .enumerate()
8050 .map(|(column, field)| {
8051 if !integer_or_date(&field.ty) {
8052 return None;
8053 }
8054 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
8055 return None;
8056 };
8057 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8058 return None;
8059 }
8060 let entries = summary
8061 .entries
8062 .iter()
8063 .map(|entry| {
8064 let value = match entry.value {
8065 FrequencyValue::Null => None,
8066 FrequencyValue::Integer(value) => Some(value),
8067 FrequencyValue::Code(_) => return None,
8068 };
8069 Some((value, entry.count))
8070 })
8071 .collect::<Option<Vec<_>>>()?;
8072 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
8073 (rows == table.rows as u64).then_some(entries)
8074 })
8075 .collect()
8076}
8077
8078fn frequency_bits(value: &Value) -> Option<u64> {
8084 Some(match value {
8085 Value::TinyInt(value) => i64::from(*value) as u64,
8086 Value::SmallInt(value) => i64::from(*value) as u64,
8087 Value::Integer(value) | Value::Date(value) => i64::from(*value) as u64,
8088 Value::BigInt(value) | Value::Timestamp(value) => *value as u64,
8089 Value::UTinyInt(value) => u64::from(*value),
8090 Value::USmallInt(value) => u64::from(*value),
8091 Value::UInteger(value) => u64::from(*value),
8092 Value::UBigInt(value) => *value,
8093 _ => return None,
8094 })
8095}
8096
8097fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
8098 Some(match value {
8099 Value::Null => None,
8100 Value::TinyInt(value) => Some(i128::from(*value)),
8101 Value::SmallInt(value) => Some(i128::from(*value)),
8102 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
8103 Value::BigInt(value) => Some(i128::from(*value)),
8104 Value::UTinyInt(value) => Some(i128::from(*value)),
8105 Value::USmallInt(value) => Some(i128::from(*value)),
8106 Value::UInteger(value) => Some(i128::from(*value)),
8107 Value::UBigInt(value) => Some(i128::from(*value)),
8108 _ => return None,
8109 })
8110}
8111
8112fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
8113 reader
8114 .table
8115 .fields
8116 .iter()
8117 .enumerate()
8118 .map(|(column, field)| {
8119 if !integer_or_date(&field.ty) {
8120 return Ok(None);
8121 }
8122 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
8123 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
8124 return Ok(None);
8125 }
8126 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
8127 let Some(entries) = entries
8128 .iter()
8129 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
8130 .collect::<Option<Vec<_>>>()
8131 else {
8132 return Ok(None);
8133 };
8134 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
8135 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
8136 })
8137 .collect()
8138}
8139
8140fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
8141 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
8142 let range = stripe.zone.column(column)?;
8143 let sum = sum.checked_add(range.sum?)?;
8144 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
8145 Some((sum, count.checked_add(nonnull)?))
8146 })
8147}
8148
8149fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
8150 table
8151 .fields
8152 .iter()
8153 .enumerate()
8154 .map(|(column, field)| {
8155 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
8156 })
8157 .collect()
8158}
8159
8160fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
8161 reader
8162 .table
8163 .fields
8164 .iter()
8165 .enumerate()
8166 .map(
8167 |(column, field)| {
8168 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
8169 },
8170 )
8171 .collect()
8172}
8173
8174fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
8175 let mut out = CATALOG.to_vec();
8176 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
8177 for entry in entries {
8178 let name = entry.name.as_bytes();
8179 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
8180 out.extend_from_slice(name);
8181 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
8182 put_u16(
8183 &mut out,
8184 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
8185 );
8186 for field in &entry.fields {
8187 let name = field.name.as_bytes();
8188 put_u16(
8189 &mut out,
8190 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8191 );
8192 out.extend_from_slice(name);
8193 put_type(&mut out, &field.ty)?;
8194 out.push(u8::from(field.not_null));
8195 }
8196 put_u64(&mut out, entry.directory.offset);
8197 put_u32(&mut out, entry.directory.length);
8198 put_u64(&mut out, entry.directory.hash);
8199 }
8200 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
8201 for view in views {
8202 let name = view.name.as_bytes();
8203 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
8204 out.extend_from_slice(name);
8205 put_long_text(&mut out, &view.sql, "view body")?;
8206 put_long_text(&mut out, &view.statement, "view statement")?;
8207 put_u16(
8208 &mut out,
8209 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
8210 );
8211 for alias in &view.aliases {
8212 let alias = alias.as_bytes();
8213 put_u16(
8214 &mut out,
8215 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
8216 );
8217 out.extend_from_slice(alias);
8218 }
8219 put_u16(
8220 &mut out,
8221 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
8222 );
8223 for field in &view.columns {
8224 let name = field.name.as_bytes();
8225 put_u16(
8226 &mut out,
8227 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
8228 );
8229 out.extend_from_slice(name);
8230 put_type(&mut out, &field.ty)?;
8231 out.push(u8::from(field.not_null));
8232 }
8233 }
8234 out.extend_from_slice(NONZERO_COUNTS);
8235 for entry in entries {
8236 if entry.nonzero.len() != entry.fields.len() {
8237 return Err(invalid("nonzero count width differs from schema"));
8238 }
8239 for count in &entry.nonzero {
8240 match count {
8241 None => out.push(0),
8242 Some(count) => {
8243 out.push(1);
8244 put_u64(&mut out, *count);
8245 }
8246 }
8247 }
8248 }
8249 out.extend_from_slice(AGGREGATE_SUMS);
8250 for entry in entries {
8251 if entry.aggregates.len() != entry.fields.len() {
8252 return Err(invalid("aggregate sum width differs from schema"));
8253 }
8254 for summary in &entry.aggregates {
8255 match summary {
8256 None => out.push(0),
8257 Some((sum, count)) => {
8258 out.push(1);
8259 out.extend_from_slice(&sum.to_le_bytes());
8260 put_u64(&mut out, *count);
8261 }
8262 }
8263 }
8264 }
8265 out.extend_from_slice(DISTINCT_COUNTS);
8266 for entry in entries {
8267 if entry.distincts.len() != entry.fields.len() {
8268 return Err(invalid("distinct count width differs from schema"));
8269 }
8270 for count in &entry.distincts {
8271 match count {
8272 None => out.push(0),
8273 Some(count) => {
8274 if *count > entry.rows as u64 {
8275 return Err(invalid("distinct count exceeds table rows"));
8276 }
8277 out.push(1);
8278 put_u64(&mut out, *count);
8279 }
8280 }
8281 }
8282 }
8283 out.extend_from_slice(INTEGER_EXTREMES);
8284 for entry in entries {
8285 if entry.extremes.len() != entry.fields.len() {
8286 return Err(invalid("integer extremes width differs from schema"));
8287 }
8288 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
8289 match extremes {
8290 None => out.push(0),
8291 Some(None) if integer_or_date(&field.ty) => out.push(1),
8292 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
8293 out.push(2);
8294 out.extend_from_slice(&low.to_le_bytes());
8295 out.extend_from_slice(&high.to_le_bytes());
8296 }
8297 _ => return Err(invalid("integer extremes type or range differs")),
8298 }
8299 }
8300 }
8301 out.extend_from_slice(COMPLETE_FREQUENCIES);
8302 for entry in entries {
8303 if entry.frequencies.len() != entry.fields.len() {
8304 return Err(invalid("numeric frequency width differs from schema"));
8305 }
8306 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
8307 match frequencies {
8308 None => out.push(0),
8309 Some(entries)
8310 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
8311 {
8312 let mut total = 0_u64;
8313 for (at, (value, count)) in entries.iter().enumerate() {
8314 if entries[..at].iter().any(|(held, _)| held == value) {
8315 return Err(invalid("numeric frequency value repeats"));
8316 }
8317 total = total
8318 .checked_add(*count)
8319 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8320 }
8321 if total != entry.rows as u64 {
8322 return Err(invalid("numeric frequencies do not cover table rows"));
8323 }
8324 out.push(1);
8325 out.push(entries.len() as u8);
8326 for (value, count) in entries {
8327 match value {
8328 None => out.push(0),
8329 Some(value) => {
8330 out.push(1);
8331 out.extend_from_slice(&value.to_le_bytes());
8332 }
8333 }
8334 put_u64(&mut out, *count);
8335 }
8336 }
8337 _ => return Err(invalid("numeric frequency type or width differs")),
8338 }
8339 }
8340 }
8341 Ok(out)
8342}
8343
8344fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
8346 let bytes = text.as_bytes();
8347 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
8348 out.extend_from_slice(bytes);
8349 Ok(())
8350}
8351
8352fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
8355 let mut cur = Cursor::new(bytes);
8356 if cur.take(8)? != CATALOG {
8357 return Err(invalid("catalog magic differs"));
8358 }
8359 let count = cur.u32()? as usize;
8360 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
8361 for _ in 0..count {
8362 let name = cur.text()?;
8363 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8364 let width = cur.u16()? as usize;
8365 let mut fields = Vec::with_capacity(width);
8366 for _ in 0..width {
8367 let name = cur.text()?;
8368 let ty = read_type(&mut cur)?;
8369 let not_null = match cur.u8()? {
8370 0 => false,
8371 1 => true,
8372 _ => return Err(invalid("nullability flag differs")),
8373 };
8374 fields.push(Field { name, ty, not_null });
8375 }
8376 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8377 let end = directory
8378 .offset
8379 .checked_add(u64::from(directory.length))
8380 .ok_or_else(|| invalid("table directory offset overflow"))?;
8381 if directory.offset < HEADER
8382 || end > size
8383 || directory.length as usize > MAX_DIRECTORY
8384 || directory.length == 0
8385 {
8386 return Err(invalid("table directory range is outside the file"));
8387 }
8388 if entries.iter().any(|held| held.name == name) {
8389 return Err(invalid("two tables in the catalog have the same name"));
8390 }
8391 let nonzero = vec![None; fields.len()];
8392 let aggregates = vec![None; fields.len()];
8393 let distincts = vec![None; fields.len()];
8394 let extremes = vec![None; fields.len()];
8395 let frequencies = vec![None; fields.len()];
8396 entries.push(Entry {
8397 name,
8398 fields,
8399 rows,
8400 directory,
8401 nonzero,
8402 aggregates,
8403 distincts,
8404 extremes,
8405 frequencies,
8406 });
8407 }
8408 let count = if cur.done() { 0 } else { cur.u32()? as usize };
8413 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
8414 for _ in 0..count {
8415 let name = cur.text()?;
8416 let sql = cur.long_text()?;
8417 let statement = cur.long_text()?;
8418 let width = cur.u16()? as usize;
8419 let mut aliases = Vec::with_capacity(width);
8420 for _ in 0..width {
8421 aliases.push(cur.text()?);
8422 }
8423 let width = cur.u16()? as usize;
8424 let mut columns = Vec::with_capacity(width);
8425 for _ in 0..width {
8426 let name = cur.text()?;
8427 let ty = read_type(&mut cur)?;
8428 let not_null = match cur.u8()? {
8429 0 => false,
8430 1 => true,
8431 _ => return Err(invalid("nullability flag differs")),
8432 };
8433 columns.push(Field { name, ty, not_null });
8434 }
8435 if views.iter().any(|held| held.name == name) {
8439 return Err(invalid("two views in the catalog have the same name"));
8440 }
8441 if entries.iter().any(|held| held.name == name) {
8442 return Err(invalid("a table and a view in the catalog have the same name"));
8443 }
8444 views.push(ViewEntry { name, sql, statement, aliases, columns });
8445 }
8446 if !cur.done() {
8447 if cur.take(8)? != NONZERO_COUNTS {
8448 return Err(invalid("catalog extension magic differs"));
8449 }
8450 for entry in &mut entries {
8451 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
8452 *count = match cur.u8()? {
8453 0 => None,
8454 1 if matches!(
8455 field.ty,
8456 LogicalType::TinyInt
8457 | LogicalType::SmallInt
8458 | LogicalType::Integer
8459 | LogicalType::BigInt
8460 | LogicalType::UTinyInt
8461 | LogicalType::USmallInt
8462 | LogicalType::UInteger
8463 | LogicalType::UBigInt
8464 ) =>
8465 {
8466 let value = cur.u64()?;
8467 if value > entry.rows as u64 {
8468 return Err(invalid("nonzero count exceeds rows"));
8469 }
8470 Some(value)
8471 }
8472 _ => return Err(invalid("nonzero count tag or column type differs")),
8473 };
8474 }
8475 }
8476 }
8477 if !cur.done() {
8478 if cur.take(8)? != AGGREGATE_SUMS {
8479 return Err(invalid("aggregate catalog extension magic differs"));
8480 }
8481 for entry in &mut entries {
8482 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
8483 *summary = match cur.u8()? {
8484 0 => None,
8485 1 if signed_integer(&field.ty) => {
8486 let sum = i128::from_le_bytes(
8487 cur.take(16)?
8488 .try_into()
8489 .map_err(|_| invalid("aggregate sum is truncated"))?,
8490 );
8491 let count = cur.u64()?;
8492 if count > entry.rows as u64 {
8493 return Err(invalid("aggregate count exceeds table rows"));
8494 }
8495 Some((sum, count))
8496 }
8497 _ => return Err(invalid("aggregate sum tag or column type differs")),
8498 };
8499 }
8500 }
8501 }
8502 if !cur.done() {
8503 if cur.take(8)? != DISTINCT_COUNTS {
8504 return Err(invalid("distinct catalog extension magic differs"));
8505 }
8506 for entry in &mut entries {
8507 for count in &mut entry.distincts {
8508 *count = match cur.u8()? {
8509 0 => None,
8510 1 => {
8511 let value = cur.u64()?;
8512 if value > entry.rows as u64 {
8513 return Err(invalid("distinct count exceeds table rows"));
8514 }
8515 Some(value)
8516 }
8517 _ => return Err(invalid("distinct count tag differs")),
8518 };
8519 }
8520 }
8521 }
8522 if !cur.done() {
8523 if cur.take(8)? != INTEGER_EXTREMES {
8524 return Err(invalid("integer extremes catalog extension magic differs"));
8525 }
8526 for entry in &mut entries {
8527 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
8528 *extremes = match cur.u8()? {
8529 0 => None,
8530 1 if integer_or_date(&field.ty) => Some(None),
8531 2 if integer_or_date(&field.ty) => {
8532 let low = i128::from_le_bytes(
8533 cur.take(16)?
8534 .try_into()
8535 .map_err(|_| invalid("minimum is truncated"))?,
8536 );
8537 let high = i128::from_le_bytes(
8538 cur.take(16)?
8539 .try_into()
8540 .map_err(|_| invalid("maximum is truncated"))?,
8541 );
8542 if low > high {
8543 return Err(invalid("integer extremes are reversed"));
8544 }
8545 Some(Some((low, high)))
8546 }
8547 _ => return Err(invalid("integer extremes tag or type differs")),
8548 };
8549 }
8550 }
8551 }
8552 if !cur.done() {
8553 if cur.take(8)? != COMPLETE_FREQUENCIES {
8554 return Err(invalid("numeric frequency catalog extension magic differs"));
8555 }
8556 for entry in &mut entries {
8557 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
8558 *frequencies = match cur.u8()? {
8559 0 => None,
8560 1 if integer_or_date(&field.ty) => {
8561 let len = cur.u8()? as usize;
8562 if len > MAX_CATALOG_FREQUENCIES {
8563 return Err(invalid("too many catalog numeric frequencies"));
8564 }
8565 let mut values = Vec::with_capacity(len);
8566 let mut total = 0_u64;
8567 for _ in 0..len {
8568 let value = match cur.u8()? {
8569 0 => None,
8570 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
8571 |_| invalid("numeric frequency value is truncated"),
8572 )?)),
8573 _ => return Err(invalid("numeric frequency value tag differs")),
8574 };
8575 if values.iter().any(|(held, _)| *held == value) {
8576 return Err(invalid("numeric frequency value repeats"));
8577 }
8578 let count = cur.u64()?;
8579 total = total
8580 .checked_add(count)
8581 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
8582 values.push((value, count));
8583 }
8584 if total != entry.rows as u64 {
8585 return Err(invalid("numeric frequencies do not cover table rows"));
8586 }
8587 Some(values)
8588 }
8589 _ => return Err(invalid("numeric frequency tag or type differs")),
8590 };
8591 }
8592 }
8593 }
8594 if !cur.done() {
8595 return Err(invalid("catalog has trailing bytes"));
8596 }
8597 Ok((entries, views))
8598}
8599
8600struct Cursor<'a> {
8608 bytes: &'a [u8],
8609 at: usize,
8610 window: Option<Window<'a>>,
8611}
8612
8613struct Window<'a> {
8615 file: &'a File,
8616 offset: u64,
8617 length: usize,
8618 start: usize,
8620 held: Vec<u8>,
8621 size: usize,
8623}
8624
8625const DIRECTORY_WINDOW: usize = 64 << 10;
8627
8628impl<'a> Cursor<'a> {
8629 fn new(bytes: &'a [u8]) -> Self {
8630 Self { bytes, at: 0, window: None }
8631 }
8632
8633 fn over(file: &'a File, offset: u64, length: usize) -> Self {
8635 let window =
8636 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
8637 Self { bytes: &[], at: 0, window: Some(window) }
8638 }
8639
8640 fn len(&self) -> usize {
8642 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
8643 }
8644
8645 fn ensure(&mut self, len: usize) -> Result<()> {
8647 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8648 if end > self.len() {
8649 return Err(invalid("directory is truncated"));
8650 }
8651 let Some(window) = &mut self.window else { return Ok(()) };
8652 if self.at < window.start || end > window.start + window.held.len() {
8653 let want = len.max(window.size).min(window.length - self.at);
8654 window.start = self.at;
8655 window.held.resize(want, 0);
8656 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
8657 }
8658 Ok(())
8659 }
8660
8661 fn held(&self, at: usize, len: usize) -> &[u8] {
8663 match &self.window {
8664 Some(window) => &window.held[at - window.start..at - window.start + len],
8665 None => &self.bytes[at..at + len],
8666 }
8667 }
8668
8669 #[inline]
8671 fn peek(&mut self, len: usize) -> Result<&[u8]> {
8672 if self.window.is_none() {
8673 let bytes = self.bytes;
8674 return Ok(&bytes[self.at..self.end(len)?]);
8675 }
8676 self.ensure(len)?;
8677 Ok(self.held(self.at, len))
8678 }
8679
8680 #[inline]
8686 fn take(&mut self, len: usize) -> Result<&[u8]> {
8687 if self.window.is_none() {
8688 let bytes = self.bytes;
8689 let (at, end) = (self.at, self.end(len)?);
8690 self.at = end;
8691 return Ok(&bytes[at..end]);
8692 }
8693 self.take_windowed(len)
8694 }
8695
8696 fn skip(&mut self, len: usize) -> Result<()> {
8698 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8699 if end > self.len() {
8700 return Err(invalid("directory is truncated"));
8701 }
8702 self.at = end;
8703 Ok(())
8704 }
8705
8706 fn skip_bound(&mut self) -> Result<()> {
8707 match self.u8()? {
8708 0 => Ok(()),
8709 1 => self.skip(16),
8710 2 => self.skip(8),
8711 3 => {
8712 let length = self.u32()? as usize;
8713 self.skip(length)
8714 }
8715 4 => self.skip(17),
8716 _ => Err(invalid("a stored bound has an unknown tag")),
8717 }
8718 }
8719
8720 #[inline]
8722 fn end(&self, len: usize) -> Result<usize> {
8723 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
8724 if end > self.bytes.len() {
8725 return Err(invalid("directory is truncated"));
8726 }
8727 Ok(end)
8728 }
8729
8730 #[inline(never)]
8732 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
8733 self.ensure(len)?;
8734 self.at += len;
8735 Ok(self.held(self.at - len, len))
8736 }
8737 #[inline]
8738 fn u8(&mut self) -> Result<u8> {
8739 Ok(self.take(1)?[0])
8740 }
8741 #[inline]
8742 fn u16(&mut self) -> Result<u16> {
8743 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
8744 }
8745 #[inline]
8746 fn u32(&mut self) -> Result<u32> {
8747 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
8748 }
8749 #[inline]
8750 fn u64(&mut self) -> Result<u64> {
8751 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
8752 }
8753 fn var_u64(&mut self) -> Result<u64> {
8754 let mut value = 0_u64;
8755 for shift in (0..=63).step_by(7) {
8756 let byte = self.u8()?;
8757 let part = u64::from(byte & 0x7f);
8758 if shift == 63 && part > 1 {
8759 return Err(invalid("frequency ordinal varint overflows"));
8760 }
8761 value |= part << shift;
8762 if byte & 0x80 == 0 {
8763 return Ok(value);
8764 }
8765 }
8766 Err(invalid("frequency ordinal varint is too long"))
8767 }
8768 fn bound(&mut self) -> Result<Option<Bound>> {
8777 let rest = self.len().saturating_sub(self.at);
8778 let mut want = 32;
8779 loop {
8780 let offered = self.peek(want.min(rest))?;
8781 let mut used = 0;
8782 match bounds::get(offered, &mut used) {
8783 Ok(bound) => {
8784 self.at += used;
8785 return Ok(bound);
8786 }
8787 Err(_) if want < rest => want *= 2,
8788 Err(error) => return Err(error),
8789 }
8790 }
8791 }
8792 fn text(&mut self) -> Result<String> {
8793 let len = self.u16()? as usize;
8794 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8795 }
8796 fn done(&self) -> bool {
8799 self.at >= self.len()
8800 }
8801 fn long_text(&mut self) -> Result<String> {
8808 let len = self.u32()? as usize;
8809 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8810 }
8811}
8812
8813fn decode_summary(
8815 cur: &mut Cursor<'_>,
8816 field: &Field,
8817 rows: usize,
8818 values: bool,
8819) -> Result<Option<FrequencySummary>> {
8820 Ok(match cur.u8()? {
8821 0 => None,
8822 1 => {
8823 let omitted_max = cur.u64()?;
8824 let count = cur.u32()? as usize;
8825 if count > FREQUENCY_ENTRIES {
8826 return Err(invalid("frequency entry count exceeds its bound"));
8827 }
8828 let mut entries = Vec::with_capacity(count);
8829 for _ in 0..count {
8831 let value = match cur.u8()? {
8832 0 => FrequencyValue::Null,
8833 1 => FrequencyValue::Integer(i128::from_le_bytes(
8834 cur.take(16)?.try_into().expect("sixteen bytes"),
8835 )),
8836 2 => FrequencyValue::Code(cur.u32()?),
8837 _ => return Err(invalid("frequency value tag differs")),
8838 };
8839 let valid = matches!(
8840 (&field.ty, value),
8841 (_, FrequencyValue::Null)
8842 | (LogicalType::Varchar | LogicalType::Blob, FrequencyValue::Code(_))
8843 | (
8844 LogicalType::TinyInt
8845 | LogicalType::SmallInt
8846 | LogicalType::Integer
8847 | LogicalType::BigInt
8848 | LogicalType::UTinyInt
8849 | LogicalType::USmallInt
8850 | LogicalType::UInteger
8851 | LogicalType::UBigInt
8852 | LogicalType::Date
8853 | LogicalType::Timestamp,
8854 FrequencyValue::Integer(_),
8855 )
8856 );
8857 if !valid {
8858 return Err(invalid("frequency value does not match its column"));
8859 }
8860 let count = cur.u64()?;
8861 if count == 0 || count > rows as u64 {
8862 return Err(invalid("frequency count is outside the table"));
8863 }
8864 entries.push(FrequencyEntry { value, count });
8865 }
8866 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8867 return Err(invalid("frequency entries are not descending"));
8868 }
8869 let ordinals = {
8870 let ordinal_count = cur.u32()? as usize;
8871 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8872 return Err(invalid("frequency ordinal count exceeds its bound"));
8873 }
8874 let mut ordinals = Vec::with_capacity(ordinal_count);
8875 let mut previous = 0_u64;
8876 for at in 0..ordinal_count {
8877 let delta = cur.var_u64()?;
8878 if at != 0 && delta == 0 {
8879 return Err(invalid("frequency ordinals are not increasing"));
8880 }
8881 let ordinal = if at == 0 {
8882 delta
8883 } else {
8884 previous
8885 .checked_add(delta)
8886 .ok_or_else(|| invalid("frequency ordinal overflows"))?
8887 };
8888 if ordinal >= rows as u64 {
8889 return Err(invalid("frequency ordinal is outside the table"));
8890 }
8891 ordinals.push(ordinal);
8892 previous = ordinal;
8893 }
8894 ordinals
8895 };
8896 let ordinal_entries = if values {
8897 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8898 for _ in 0..ordinals.len() {
8899 let entry = cur.u16()?;
8900 if entry as usize >= entries.len() {
8901 return Err(invalid("frequency ordinal value is outside its entries"));
8902 }
8903 ordinal_entries.push(entry);
8904 }
8905 ordinal_entries
8906 } else {
8907 Vec::new()
8908 };
8909 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8910 }
8911 _ => return Err(invalid("frequency summary tag differs")),
8912 })
8913}
8914
8915fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8918 match cur.u8()? {
8919 0 => Ok(()),
8920 1 => {
8921 cur.skip(8)?;
8922 let entries = cur.u32()? as usize;
8923 if entries > FREQUENCY_ENTRIES {
8924 return Err(invalid("frequency entry count exceeds its bound"));
8925 }
8926 for _ in 0..entries {
8927 match cur.u8()? {
8928 0 => {}
8929 1 => cur.skip(16)?,
8930 2 => cur.skip(4)?,
8931 _ => return Err(invalid("frequency value tag differs")),
8932 }
8933 cur.skip(8)?;
8934 }
8935 let ordinals = cur.u32()? as usize;
8936 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8937 return Err(invalid("frequency ordinal count exceeds its bound"));
8938 }
8939 for _ in 0..ordinals {
8940 cur.var_u64()?;
8941 }
8942 if values {
8943 cur.skip(ordinals * 2)?;
8944 }
8945 Ok(())
8946 }
8947 _ => Err(invalid("frequency summary tag differs")),
8948 }
8949}
8950
8951fn quick_nonzero(
8955 mut cur: Cursor<'_>,
8956 name: &str,
8957 fields: &[Field],
8958 rows: usize,
8959 wanted: usize,
8960) -> Result<Option<u64>> {
8961 if cur.take(8)? != DIRECTORY || cur.text()? != name {
8962 return Err(invalid("table directory differs from the catalog"));
8963 }
8964 let width = cur.u16()? as usize;
8965 if width != fields.len() {
8966 return Err(invalid("table directory width differs from the catalog"));
8967 }
8968 for field in fields {
8969 let stored =
8970 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8971 if &stored != field {
8972 return Err(invalid("table directory schema differs from the catalog"));
8973 }
8974 }
8975 let mut dictionaries = Vec::with_capacity(width);
8976 for field in fields {
8977 let held = match cur.u8()? {
8978 0 => false,
8979 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
8980 cur.skip(20)?;
8981 true
8982 }
8983 _ => return Err(invalid("dictionary page tag differs")),
8984 };
8985 dictionaries.push(held);
8986 }
8987 for _ in 0..width {
8988 match cur.u8()? {
8989 0 => {}
8990 1 => cur.skip(8)?,
8991 _ => return Err(invalid("distinct count tag differs")),
8992 }
8993 }
8994 if cur.u64()? != rows as u64 {
8995 return Err(invalid("table row count differs from the catalog"));
8996 }
8997 let stripes = cur.u32()? as usize;
8998 let mut total = 0_usize;
8999 let mut nulls = 0_u64;
9000 for _ in 0..stripes {
9001 let parts = cur.u32()? as usize;
9002 if parts == 0 || parts > STRIPE_PARTS {
9003 return Err(invalid("stripe part count is outside its bound"));
9004 }
9005 let mut stripe_rows = 0_usize;
9006 for _ in 0..parts {
9007 stripe_rows = stripe_rows
9008 .checked_add(cur.u32()? as usize)
9009 .ok_or_else(|| invalid("stripe row count overflow"))?;
9010 }
9011 total =
9012 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9013 cur.skip(12 + width * 12)?;
9014 for (field, held) in fields.iter().zip(&dictionaries) {
9015 if coded_type(&field.ty) && *held {
9016 cur.skip(20)?;
9017 }
9018 }
9019 for _ in 0..width * 2 {
9020 match cur.u8()? {
9021 0 => {}
9022 1 => cur.skip(20)?,
9023 _ => return Err(invalid("stripe page tag differs")),
9024 }
9025 }
9026 for column in 0..width {
9027 cur.skip_bound()?;
9028 cur.skip_bound()?;
9029 let count = cur.u32()? as u64;
9030 if count > stripe_rows as u64 {
9031 return Err(invalid("null count exceeds stripe rows"));
9032 }
9033 if column == wanted {
9034 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
9035 }
9036 cur.skip(1)?;
9037 match cur.u8()? {
9038 0 => {}
9039 1 => cur.skip(16)?,
9040 _ => return Err(invalid("a stripe sum has an unknown tag")),
9041 }
9042 }
9043 }
9044 if total != rows {
9045 return Err(invalid("table row count differs from stripes"));
9046 }
9047 if cur.done() {
9048 return Ok(None);
9049 }
9050 let magic = cur.take(8)?;
9051 let values = magic == FREQUENCIES;
9052 if !values && magic != FREQUENCIES_V2 {
9053 return Err(invalid("directory extension magic differs"));
9054 }
9055 if cur.u16()? as usize != width {
9056 return Err(invalid("frequency column count differs"));
9057 }
9058 for _ in 0..wanted {
9059 skip_summary(&mut cur, values, rows)?;
9060 }
9061 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
9062 return Ok(None);
9063 };
9064 let zero = summary
9065 .entries
9066 .iter()
9067 .find(|entry| entry.value == FrequencyValue::Integer(0))
9068 .map(|entry| entry.count)
9069 .or_else(|| (summary.omitted_max == 0).then_some(0));
9070 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
9071}
9072
9073fn quick_integer_fold(
9076 file: &File,
9077 mut cur: Cursor<'_>,
9078 entry: &Entry,
9079 size: u64,
9080 wanted: usize,
9081 emit: &mut impl FnMut(i64, u64) -> Result<()>,
9082) -> Result<()> {
9083 let name = &entry.name;
9084 let fields = &entry.fields;
9085 let rows = entry.rows;
9086 if cur.take(8)? != DIRECTORY || cur.text()? != name.as_str() {
9087 return Err(invalid("table directory differs from the catalog"));
9088 }
9089 let width = cur.u16()? as usize;
9090 if width != fields.len() {
9091 return Err(invalid("table directory width differs from the catalog"));
9092 }
9093 for field in fields {
9094 let stored =
9095 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
9096 if &stored != field {
9097 return Err(invalid("table directory schema differs from the catalog"));
9098 }
9099 }
9100 let mut dictionaries = Vec::with_capacity(width);
9101 for field in fields {
9102 dictionaries.push(match cur.u8()? {
9103 0 => false,
9104 tag if coded_type(&field.ty) && tag == dictionary_tag(&field.ty) => {
9105 cur.skip(20)?;
9106 true
9107 }
9108 _ => return Err(invalid("dictionary page tag differs")),
9109 });
9110 }
9111 for _ in 0..width {
9112 match cur.u8()? {
9113 0 => {}
9114 1 => cur.skip(8)?,
9115 _ => return Err(invalid("distinct count tag differs")),
9116 }
9117 }
9118 if cur.u64()? != rows as u64 {
9119 return Err(invalid("table row count differs from the catalog"));
9120 }
9121 let stripes = cur.u32()? as usize;
9122 let mut total = 0_usize;
9123 let mut bytes = Vec::new();
9124 for _ in 0..stripes {
9125 let parts = cur.u32()? as usize;
9126 if parts == 0 || parts > STRIPE_PARTS {
9127 return Err(invalid("stripe part count is outside its bound"));
9128 }
9129 let mut part_rows = Vec::with_capacity(parts);
9130 for _ in 0..parts {
9131 let count = cur.u32()? as usize;
9132 if count == 0 {
9133 return Err(invalid("empty part"));
9134 }
9135 total = total.checked_add(count).ok_or_else(|| invalid("stripe row count overflow"))?;
9136 part_rows.push(count);
9137 }
9138 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9139 let section = index_section(parts)?;
9140 let index_length =
9141 section.checked_mul(width).ok_or_else(|| invalid("index page length overflow"))?;
9142 if index.offset < HEADER
9143 || index.offset.checked_add(u64::from(index.length)).is_none_or(|end| end > size)
9144 || index.length as usize != index_length
9145 {
9146 return Err(invalid("index page range is outside the file"));
9147 }
9148 cur.skip(wanted * 12)?;
9149 let page = Span { offset: cur.u64()?, length: cur.u32()? };
9150 if page.offset < HEADER
9151 || page.offset.checked_add(u64::from(page.length)).is_none_or(|end| end > size)
9152 || page.length as usize > MAX_PAGE
9153 {
9154 return Err(invalid("column page range is outside the file"));
9155 }
9156 cur.skip((width - wanted - 1) * 12)?;
9157 for (field, held) in fields.iter().zip(&dictionaries) {
9158 if coded_type(&field.ty) && *held {
9159 cur.skip(20)?;
9160 }
9161 }
9162 for _ in 0..width * 2 {
9163 match cur.u8()? {
9164 0 => {}
9165 1 => cur.skip(20)?,
9166 _ => return Err(invalid("stripe page tag differs")),
9167 }
9168 }
9169 for _ in 0..width {
9170 cur.skip_bound()?;
9171 cur.skip_bound()?;
9172 cur.skip(5)?;
9173 match cur.u8()? {
9174 0 => {}
9175 1 => cur.skip(16)?,
9176 _ => return Err(invalid("a stripe sum has an unknown tag")),
9177 }
9178 }
9179 let spans = read_index_span(file, index, page, parts, wanted)?;
9180 for (span, expected_rows) in spans.into_iter().zip(part_rows) {
9181 bytes.resize(span.length, 0);
9182 let at = page
9183 .offset
9184 .checked_add(span.start as u64)
9185 .ok_or_else(|| invalid("part range overflow"))?;
9186 read_at(file, at, &mut bytes)?;
9187 if checksum(&bytes) != span.hash {
9188 return Err(invalid("integer part checksum differs"));
9189 }
9190 if bytes.first() == Some(&5) && bytes.get(1) == Some(&0) {
9191 let decoded_rows = integer::fold(&bytes[2..], |value, count| {
9192 check_integer_tally_value(value, &fields[wanted].ty)?;
9193 emit(value, count)
9194 })?;
9195 if decoded_rows != expected_rows {
9196 return Err(invalid("encoded integer part holds the wrong number of rows"));
9197 }
9198 } else {
9199 let column = decode(&fields[wanted].ty, expected_rows, &bytes, None)?;
9200 if let Some(packed) = column.packed_parts() {
9201 let validity = column.validity();
9202 let all_valid = column.none_null();
9203 let base = packed.base();
9204 let mut codes = [0_u64; 64];
9205 for from in (0..expected_rows).step_by(codes.len()) {
9206 let count = (expected_rows - from).min(codes.len());
9207 packed.unpack(from, &mut codes[..count]);
9208 for (offset, &code) in codes[..count].iter().enumerate() {
9209 if all_valid || validity.is_valid(from + offset) {
9210 emit((base + i128::from(code)) as i64, 1)?;
9212 }
9213 }
9214 }
9215 continue;
9216 }
9217 let column = column.into_flat()?;
9218 let validity = column.validity();
9219 macro_rules! count_decoded {
9220 ($values:expr) => {
9221 for (row, &value) in $values.as_slice().iter().enumerate() {
9222 if validity.is_valid(row) {
9223 emit(i64::from(value), 1)?;
9224 }
9225 }
9226 };
9227 }
9228 match column.data() {
9229 Some(Data::Int8(values)) => count_decoded!(values),
9230 Some(Data::Int16(values)) => count_decoded!(values),
9231 Some(Data::Int32(values)) => count_decoded!(values),
9232 Some(Data::Int64(values)) => count_decoded!(values),
9233 _ => return Err(invalid("decoded integer part has the wrong type")),
9234 }
9235 }
9236 }
9237 }
9238 if total != rows {
9239 return Err(invalid("table row count differs from stripes"));
9240 }
9241 Ok(())
9242}
9243
9244fn check_integer_tally_value(value: i64, ty: &LogicalType) -> Result<()> {
9245 let fits = match ty {
9246 LogicalType::TinyInt => i8::try_from(value).is_ok(),
9247 LogicalType::SmallInt => i16::try_from(value).is_ok(),
9248 LogicalType::Integer => i32::try_from(value).is_ok(),
9249 LogicalType::BigInt => true,
9250 _ => false,
9251 };
9252 if fits { Ok(()) } else { Err(invalid("encoded integer value is outside its column type")) }
9253}
9254
9255fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
9256 read_directory(Cursor::new(bytes), size, None)
9257}
9258
9259fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
9264 if cur.take(8)? != DIRECTORY {
9265 return Err(invalid("directory magic differs"));
9266 }
9267 let name = cur.text()?;
9268 let width = cur.u16()? as usize;
9269 let mut fields = Vec::with_capacity(width);
9270 for _ in 0..width {
9271 let name = cur.text()?;
9272 let ty = read_type(&mut cur)?;
9273 let not_null = match cur.u8()? {
9274 0 => false,
9275 1 => true,
9276 _ => return Err(invalid("nullability flag differs")),
9277 };
9278 fields.push(Field { name, ty, not_null });
9279 }
9280 let mut dictionaries = Vec::with_capacity(width);
9281 for field in &fields {
9282 dictionaries.push(match cur.u8()? {
9283 0 => None,
9284 tag if tag == dictionary_tag(&field.ty) => {
9285 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9286 let end = page
9287 .offset
9288 .checked_add(u64::from(page.length))
9289 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
9290 if page.offset < HEADER || end > size {
9295 return Err(invalid("dictionary page range is outside the file"));
9296 }
9297 Some(page)
9298 }
9299 _ => return Err(invalid("dictionary page tag differs")),
9300 });
9301 }
9302 let mut distincts = Vec::with_capacity(width);
9303 for _ in 0..width {
9304 distincts.push(match cur.u8()? {
9305 0 => None,
9306 1 => Some(cur.u64()?),
9307 _ => return Err(invalid("distinct count tag differs")),
9308 });
9309 }
9310 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
9311 let count = cur.u32()? as usize;
9312 let mut stripes = Vec::with_capacity(count);
9313 let mut total = 0_usize;
9314 for _ in 0..count {
9315 let count = cur.u32()? as usize;
9316 if count == 0 || count > STRIPE_PARTS {
9317 return Err(invalid("stripe part count is outside its bound"));
9318 }
9319 let mut parts = Vec::with_capacity(count);
9320 let mut stripe_rows = 0_usize;
9321 for _ in 0..count {
9322 let rows = cur.u32()?;
9323 if rows == 0 {
9324 return Err(invalid("empty part"));
9325 }
9326 parts.push(rows);
9327 stripe_rows = stripe_rows
9328 .checked_add(rows as usize)
9329 .ok_or_else(|| invalid("stripe row count overflow"))?;
9330 }
9331 total =
9332 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
9333 let index = Span { offset: cur.u64()?, length: cur.u32()? };
9334 let section = index_section(count)?;
9335 let wanted = section
9336 .checked_mul(width)
9337 .and_then(|bytes| u32::try_from(bytes).ok())
9338 .ok_or_else(|| invalid("index page length overflow"))?;
9339 let end = index
9340 .offset
9341 .checked_add(u64::from(index.length))
9342 .ok_or_else(|| invalid("index page offset overflow"))?;
9343 if index.offset < HEADER || end > size || index.length != wanted {
9344 return Err(invalid("index page range is outside the file"));
9345 }
9346 let mut pages = Vec::with_capacity(width);
9347 for _ in 0..width {
9348 let offset = cur.u64()?;
9349 let length = cur.u32()?;
9350 let end = offset
9351 .checked_add(u64::from(length))
9352 .ok_or_else(|| invalid("page offset overflow"))?;
9353 if offset < HEADER || end > size || length as usize > MAX_PAGE {
9354 return Err(invalid("page range is outside the file"));
9355 }
9356 pages.push(Span { offset, length });
9357 }
9358 let mut memberships = vec![None; width];
9359 for (column, field) in fields.iter().enumerate() {
9360 if !coded_type(&field.ty) || dictionaries[column].is_none() {
9361 continue;
9362 }
9363 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9364 let end = page
9365 .offset
9366 .checked_add(u64::from(page.length))
9367 .ok_or_else(|| invalid("membership page offset overflow"))?;
9368 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9369 return Err(invalid("membership page range is outside the file"));
9370 }
9371 if page.length != 0 {
9374 memberships[column] = Some(page);
9375 }
9376 }
9377 let mut sieves = vec![None; width];
9378 for sieve in sieves.iter_mut().take(width) {
9379 match cur.u8()? {
9380 0 => continue,
9381 1 => {}
9382 _ => return Err(invalid("a sieve page has an unknown tag")),
9383 }
9384 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9385 let end = page
9386 .offset
9387 .checked_add(u64::from(page.length))
9388 .ok_or_else(|| invalid("sieve page offset overflow"))?;
9389 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9390 return Err(invalid("sieve page range is outside the file"));
9391 }
9392 *sieve = Some(page);
9393 }
9394 let mut part_ranges = vec![None; width];
9395 for held in part_ranges.iter_mut().take(width) {
9396 match cur.u8()? {
9397 0 => continue,
9398 1 => {}
9399 _ => return Err(invalid("a part range page has an unknown tag")),
9400 }
9401 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
9402 let end = page
9403 .offset
9404 .checked_add(u64::from(page.length))
9405 .ok_or_else(|| invalid("part range page offset overflow"))?;
9406 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
9407 return Err(invalid("part range page range is outside the file"));
9408 }
9409 *held = Some(page);
9410 }
9411 let mut ranges = Vec::with_capacity(width);
9412 for column in 0..width {
9413 let low = cur.bound()?;
9414 let high = cur.bound()?;
9415 let nulls = cur.u32()? as usize;
9416 if nulls > stripe_rows {
9417 return Err(invalid("null count exceeds stripe rows"));
9418 }
9419 let exact = cur.u8()? != 0;
9420 let sum = match cur.u8()? {
9421 0 => None,
9422 1 => Some(i128::from_le_bytes(
9423 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
9424 )),
9425 _ => return Err(invalid("a stripe sum has an unknown tag")),
9426 };
9427 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
9433 let low = low.map(|bound| scaled_as(bound, ty));
9434 let high = high.map(|bound| scaled_as(bound, ty));
9435 ranges.push(Range { low, high, nulls, exact, sum });
9436 }
9437 stripes.push(Stripe {
9438 rows: stripe_rows,
9439 parts,
9440 index,
9441 pages,
9442 memberships: Pages::from_slots(memberships)?,
9443 sieves: Pages::from_slots(sieves)?,
9444 part_ranges: Pages::from_slots(part_ranges)?,
9445 zone: Zone::from_ranges(ranges),
9446 });
9447 }
9448 if total != rows {
9449 return Err(invalid("table row count differs from stripes"));
9450 }
9451 let mut entry_counts = vec![0; width];
9454 let frequencies = if cur.done() {
9455 vec![None; width]
9456 } else {
9457 let frequency_magic = cur.take(8)?;
9458 let frequency_values = frequency_magic == FREQUENCIES;
9459 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
9460 return Err(invalid("directory extension magic differs"));
9461 }
9462 if cur.u16()? as usize != width {
9463 return Err(invalid("frequency column count differs"));
9464 }
9465 let mut frequencies = Vec::with_capacity(width);
9466 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
9467 let start = cur.at;
9468 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
9469 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
9470 frequencies.push(match (summary, stored_at) {
9471 (None, _) => None,
9472 (Some(summary), None) => Some(Frequencies::Held(summary)),
9473 (Some(_), Some(offset)) => Some(Frequencies::Stored {
9474 span: Span {
9475 offset: offset + start as u64,
9476 length: u32::try_from(cur.at - start)
9477 .map_err(|_| invalid("a frequency synopsis is too long"))?,
9478 },
9479 values: frequency_values,
9480 }),
9481 });
9482 }
9483 frequencies
9484 };
9485 let mut clustering = None;
9495 let mut sections = Vec::new();
9496 let mut pair_frequencies = Vec::new();
9497 let mut seen_pair_frequencies = false;
9498 let mut frequency_texts = vec![Vec::new(); width];
9499 let mut seen_frequency_texts = false;
9500 let mut host_groups = None;
9501 let mut demoted = Vec::new();
9502 let mut seen_sections = false;
9503 let mut dictionary_payloads = Vec::new();
9504 let mut seen_payloads = false;
9505 let mut generation = 0;
9508 while !cur.done() {
9509 let mut tag = [0u8; 8];
9510 tag.copy_from_slice(cur.take(8)?);
9511 if &tag == PAIR_FREQUENCIES {
9512 if seen_pair_frequencies {
9513 return Err(invalid("directory names two pair frequency blocks"));
9514 }
9515 seen_pair_frequencies = true;
9516 let count = cur.u16()? as usize;
9517 if count > MAX_PAIR_FREQUENCIES {
9518 return Err(invalid("pair frequency count exceeds its bound"));
9519 }
9520 pair_frequencies = Vec::with_capacity(count);
9521 for _ in 0..count {
9522 let first = cur.u16()?;
9523 let second = cur.u16()?;
9524 let first_at = first as usize;
9525 let second_at = second as usize;
9526 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
9527 return Err(invalid("pair frequency first column has no synopsis"));
9528 }
9529 let first_entries = entry_counts[first_at];
9530 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
9531 || dictionaries.get(second_at).copied().flatten().is_none()
9532 {
9533 return Err(invalid("pair frequency second column has no stable dictionary"));
9534 }
9535 if pair_frequencies
9536 .iter()
9537 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
9538 {
9539 return Err(invalid("directory repeats a pair frequency summary"));
9540 }
9541 let omitted_max = cur.u64()?;
9542 if omitted_max > rows as u64 {
9543 return Err(invalid("pair frequency omitted count exceeds the table"));
9544 }
9545 let entries_count = cur.u16()? as usize;
9546 if entries_count > FREQUENCY_ENTRIES {
9547 return Err(invalid("pair frequency entry count exceeds its bound"));
9548 }
9549 let mut entries = Vec::with_capacity(entries_count);
9550 for _ in 0..entries_count {
9551 let first_entry = cur.u16()?;
9552 if first_entry as usize >= first_entries {
9553 return Err(invalid("pair frequency anchor is outside its synopsis"));
9554 }
9555 let second = match cur.u8()? {
9556 0 => None,
9557 1 => Some(cur.u32()?),
9558 _ => return Err(invalid("pair frequency string tag differs")),
9559 };
9560 let count = cur.u64()?;
9561 if count == 0 || count > rows as u64 {
9562 return Err(invalid("pair frequency count is outside the table"));
9563 }
9564 entries.push(PairFrequencyEntry { first_entry, second, count });
9565 }
9566 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
9567 return Err(invalid("pair frequency entries are not descending"));
9568 }
9569 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
9570 }
9571 } else if &tag == FREQUENCY_TEXTS {
9572 if seen_frequency_texts {
9573 return Err(invalid("directory names two frequency text blocks"));
9574 }
9575 seen_frequency_texts = true;
9576 let columns = cur.u16()? as usize;
9577 if columns > width {
9578 return Err(invalid("frequency text column count exceeds the schema"));
9579 }
9580 for _ in 0..columns {
9581 let column = cur.u16()? as usize;
9582 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
9583 return Err(invalid("frequency text column is repeated or out of range"));
9584 }
9585 if !matches!(fields.get(column), Some(field) if coded_type(&field.ty))
9586 || dictionaries.get(column).copied().flatten().is_none()
9587 || frequencies.get(column).and_then(Option::as_ref).is_none()
9588 {
9589 return Err(invalid("frequency texts belong to a non-string synopsis"));
9590 }
9591 let count = cur.u16()? as usize;
9592 if count == 0 || count != entry_counts[column] {
9593 return Err(invalid("frequency text count differs from its synopsis"));
9594 }
9595 let mut texts = Vec::with_capacity(count);
9596 for _ in 0..count {
9597 texts.push(match cur.u8()? {
9598 0 => None,
9599 1 => {
9600 let length = cur.u32()? as usize;
9601 let bytes = cur.take(length)?.to_vec();
9602 if fields[column].ty == LogicalType::Varchar {
9603 std::str::from_utf8(&bytes)
9604 .map_err(|_| invalid("frequency text is not UTF-8"))?;
9605 }
9606 Some(bytes)
9607 }
9608 _ => return Err(invalid("frequency text tag differs")),
9609 });
9610 }
9611 frequency_texts[column] = texts;
9612 }
9613 } else if &tag == HOST_GROUPS {
9614 if host_groups.is_some() {
9615 return Err(invalid("directory names two host group blocks"));
9616 }
9617 let column = cur.u16()? as usize;
9618 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
9619 || dictionaries.get(column).copied().flatten().is_none()
9620 {
9621 return Err(invalid("host groups belong to a non-string dictionary"));
9622 }
9623 let omitted_max = cur.u64()?;
9624 if omitted_max > rows as u64 {
9625 return Err(invalid("host group bound exceeds the table"));
9626 }
9627 let count = cur.u16()? as usize;
9628 if count > host::CAPACITY {
9629 return Err(invalid("host group count exceeds its bound"));
9630 }
9631 let mut entries = Vec::with_capacity(count);
9632 let mut bytes = 0_usize;
9633 for _ in 0..count {
9634 let host_len = cur.u32()? as usize;
9635 bytes =
9636 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
9637 if bytes > host::BYTE_BUDGET {
9638 return Err(invalid("host groups exceed their byte budget"));
9639 }
9640 let host = std::str::from_utf8(cur.take(host_len)?)
9641 .map_err(|_| invalid("host is not UTF-8"))?
9642 .to_owned();
9643 let count = cur.u64()?;
9644 if count == 0 || count > rows as u64 {
9645 return Err(invalid("host group count exceeds the table"));
9646 }
9647 let bytes_sum = i128::from_le_bytes(
9648 cur.take(16)?
9649 .try_into()
9650 .map_err(|_| invalid("host length sum is truncated"))?,
9651 );
9652 if bytes_sum < 0 {
9653 return Err(invalid("host length sum is negative"));
9654 }
9655 let minimum_len = cur.u32()? as usize;
9656 bytes =
9657 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
9658 if bytes > host::BYTE_BUDGET {
9659 return Err(invalid("host groups exceed their byte budget"));
9660 }
9661 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
9662 .map_err(|_| invalid("host minimum is not UTF-8"))?
9663 .to_owned();
9664 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
9665 }
9666 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
9667 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
9668 {
9669 return Err(invalid("host groups are not in certified order"));
9670 }
9671 host_groups = Some(host::HostSummary { column, omitted_max, entries });
9672 } else if &tag == CLUSTERING {
9673 if clustering.is_some() {
9674 return Err(invalid("directory names two clustering declarations"));
9675 }
9676 let bucket = Width::from_tag(cur.u8()?)
9677 .ok_or_else(|| invalid("clustering width tag differs"))?;
9678 let count = cur.u16()? as usize;
9679 let mut columns = Vec::with_capacity(count.min(fields.len()));
9680 for _ in 0..count {
9681 columns.push(u32::from(cur.u16()?));
9682 }
9683 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
9686 invalid("stored clustering declaration does not match the table it is on")
9687 })?);
9688 } else if &tag == DEMOTED {
9689 if !demoted.is_empty() {
9690 return Err(invalid("directory names two demoted column blocks"));
9691 }
9692 let count = cur.u16()? as usize;
9693 if count == 0 || count > width {
9694 return Err(invalid("demoted column count is outside the schema"));
9695 }
9696 demoted = vec![false; width];
9697 for _ in 0..count {
9698 let column = cur.u16()? as usize;
9699 if dictionaries.get(column).copied().flatten().is_none() || demoted[column] {
9700 return Err(invalid("a demoted column is repeated or has no dictionary"));
9701 }
9702 demoted[column] = true;
9703 }
9704 } else if &tag == SECTIONS {
9705 if seen_sections {
9706 return Err(invalid("directory names two section tables"));
9707 }
9708 seen_sections = true;
9709 generation = cur.u64()?;
9710 let count = cur.u16()? as usize;
9711 if count > MAX_SECTIONS {
9712 return Err(invalid("section count exceeds its bound"));
9713 }
9714 sections = Vec::with_capacity(count);
9715 for _ in 0..count {
9718 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
9719 }
9720 for held in §ions {
9721 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
9722 return Err(invalid("a section's extent table overflows the file"));
9723 };
9724 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
9728 return Err(invalid("a section's extent table is outside the file"));
9729 }
9730 if held.extents == 0 && held.extent_bytes != 0 {
9731 return Err(invalid("a section with no extents names an extent table"));
9732 }
9733 }
9734 } else if &tag == DICTIONARY_PAYLOADS {
9735 if seen_payloads {
9736 return Err(invalid("directory names two dictionary payload blocks"));
9737 }
9738 seen_payloads = true;
9739 let count = cur.u16()? as usize;
9740 if count != fields.len() {
9741 return Err(invalid("dictionary payload block does not match the table's columns"));
9742 }
9743 dictionary_payloads = Vec::with_capacity(count);
9744 for _ in 0..count {
9745 let bytes = cur.u64()?;
9746 if bytes > size {
9747 return Err(invalid("a dictionary payload is larger than the file"));
9748 }
9749 dictionary_payloads.push(bytes);
9750 }
9751 } else {
9752 return Err(invalid("directory extension magic differs"));
9753 }
9754 }
9755 if !cur.done() {
9756 return Err(invalid("directory has trailing bytes"));
9757 }
9758 for stripe in &stripes {
9759 for (column, field) in fields.iter().enumerate() {
9760 if coded_type(&field.ty)
9761 && dictionaries[column].is_some()
9762 && stripe.memberships.get(column).is_none()
9763 && !demoted.get(column).copied().unwrap_or(false)
9764 {
9765 return Err(invalid("string page has no code membership index"));
9766 }
9767 }
9768 }
9769 Ok(Table {
9770 name,
9771 fields,
9772 stripes,
9773 rows,
9774 dictionaries,
9775 dictionary_payloads,
9776 demoted,
9777 distincts,
9778 frequencies,
9779 pair_frequencies,
9780 frequency_texts,
9781 host_groups,
9782 clustering,
9783 generation,
9784 sections,
9785 })
9786}
9787
9788fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
9790 bounds::put(out, bound)
9791}
9792
9793#[derive(Debug)]
9810struct Codes;
9811
9812impl chooser::Chooser for Codes {
9813 fn name(&self) -> &'static str {
9814 "codes"
9815 }
9816
9817 fn narrow_strings(
9818 &self,
9819 _values: &[&[u8]],
9820 offered: &[string::Kind],
9821 _depth: u8,
9822 ) -> Vec<string::Kind> {
9823 offered.to_vec()
9826 }
9827
9828 fn narrow_integers(
9829 &self,
9830 _values: &[i64],
9831 offered: &[integer::Kind],
9832 depth: u8,
9833 ) -> Vec<integer::Kind> {
9834 narrowed_to(Codes::keep(depth), offered)
9837 }
9838
9839 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9840 Codes::keep(depth).contains(&kind)
9841 }
9842}
9843
9844impl Codes {
9845 fn keep(depth: u8) -> &'static [integer::Kind] {
9846 if depth == 0 {
9847 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
9848 } else {
9849 &[integer::Kind::Constant, integer::Kind::Packed]
9850 }
9851 }
9852}
9853
9854fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
9862 let narrowed: Vec<integer::Kind> =
9863 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
9864 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
9865}
9866
9867#[derive(Debug)]
9879struct Fixed;
9880
9881impl chooser::Chooser for Fixed {
9882 fn name(&self) -> &'static str {
9883 "fixed"
9884 }
9885
9886 fn narrow_strings(
9887 &self,
9888 _values: &[&[u8]],
9889 offered: &[string::Kind],
9890 _depth: u8,
9891 ) -> Vec<string::Kind> {
9892 offered.to_vec()
9893 }
9894
9895 fn narrow_integers(
9896 &self,
9897 _values: &[i64],
9898 offered: &[integer::Kind],
9899 depth: u8,
9900 ) -> Vec<integer::Kind> {
9901 narrowed_to(Fixed::keep(depth), offered)
9902 }
9903
9904 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
9905 Fixed::keep(depth).contains(&kind)
9906 }
9907}
9908
9909impl Fixed {
9910 fn keep(depth: u8) -> &'static [integer::Kind] {
9911 if depth == 0 {
9912 &[
9913 integer::Kind::Constant,
9914 integer::Kind::Packed,
9915 integer::Kind::Delta,
9916 integer::Kind::Rle,
9917 integer::Kind::Sparse,
9918 integer::Kind::Strided,
9919 ]
9920 } else {
9921 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
9922 }
9923 }
9924}
9925
9926fn widened(data: &Data) -> Option<Vec<i64>> {
9933 match data {
9934 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9935 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9936 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9937 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9938 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9939 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
9940 Data::Int64(values) => Some(values.to_vec()),
9941 _ => None,
9942 }
9943}
9944
9945trait Narrow: Copy {
9952 const BIASED: (u32, u64);
9957
9958 fn narrow(value: i64) -> Self;
9960}
9961
9962#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
9979fn residue<T: Narrow>(value: i64) -> u64 {
9980 let (bits, bias) = T::BIASED;
9981 (value as u64).wrapping_add(bias) >> bits
9982}
9983
9984macro_rules! narrows {
9989 ($($ty:ty => $bias:expr),* $(,)?) => {$(
9990 impl Narrow for $ty {
9991 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
9992
9993 #[allow(
9994 clippy::cast_possible_truncation,
9995 clippy::cast_sign_loss,
9996 reason = "the caller has checked the bits this truncates away"
9997 )]
9998 fn narrow(value: i64) -> Self {
9999 value as Self
10000 }
10001 }
10002 )*};
10003}
10004
10005narrows! {
10006 i8 => 1 << 7,
10007 u8 => 0,
10008 i16 => 1 << 15,
10009 u16 => 0,
10010 i32 => 1 << 31,
10011 u32 => 0,
10012}
10013
10014fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
10027 let mut spilled = 0u64;
10028 for value in values {
10029 spilled |= residue::<T>(*value);
10030 }
10031 if spilled != 0 {
10032 return Err(invalid("page value is not of its type"));
10033 }
10034 Ok(values.iter().map(|value| T::narrow(*value)).collect())
10035}
10036
10037fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
10042 Ok(match ty {
10043 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
10044 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
10045 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
10046 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
10047 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
10048 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
10049 LogicalType::BigInt
10050 | LogicalType::Timestamp
10051 | LogicalType::Time
10052 | LogicalType::TimeTz
10053 | LogicalType::TimestampTz
10054 | LogicalType::TimestampS
10055 | LogicalType::TimestampMs
10056 | LogicalType::TimestampNs => Data::Int64(values.into()),
10057 LogicalType::Decimal { .. } => match ty.physical() {
10060 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
10061 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
10062 PhysicalType::Int64 => Data::Int64(values.into()),
10063 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
10064 },
10065 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
10066 })
10067}
10068
10069fn plain_width(ty: &LogicalType) -> Option<usize> {
10072 Some(match ty {
10073 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
10074 LogicalType::SmallInt | LogicalType::USmallInt => 2,
10075 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
10076 LogicalType::BigInt
10077 | LogicalType::Timestamp
10078 | LogicalType::Time
10079 | LogicalType::TimeTz
10080 | LogicalType::TimestampTz
10081 | LogicalType::TimestampS
10082 | LogicalType::TimestampMs
10083 | LogicalType::TimestampNs => 8,
10084 LogicalType::Decimal { .. } => match ty.physical() {
10085 PhysicalType::Int16 => 2,
10086 PhysicalType::Int32 => 4,
10087 PhysicalType::Int64 => 8,
10088 _ => return None,
10091 },
10092 _ => return None,
10093 })
10094}
10095
10096fn cascaded(
10102 flat: &Vector,
10103 ty: &LogicalType,
10104 packed: Option<&Packed<'_>>,
10105 settling: &mut Settling,
10106) -> Result<Option<Vec<u8>>> {
10107 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
10108 let Some(values) = widened(data) else { return Ok(None) };
10109 let plain = values.len().saturating_mul(width);
10110 let best = match packed {
10111 Some(packed) => plain.min(21 + size_of_val(packed.words())),
10113 None => plain,
10114 };
10115 let out = settling.encode(&values)?;
10116 Ok((out.len() < best).then_some(out))
10117}
10118
10119const SEARCH_EVERY: usize = 16;
10126
10127#[derive(Debug, Default)]
10132struct Settling {
10133 shape: Option<Shape>,
10136 since: usize,
10138}
10139
10140impl Settling {
10141 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
10148 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
10149 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
10150 let out = integer::encode_with(values, &replay)?;
10151 if !replay.held() {
10152 self.settle(&out, values.len(), replay.first_offered())?;
10153 return Ok(out);
10154 }
10155 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
10156 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
10157 self.since += 1;
10158 return Ok(out);
10159 }
10160 }
10161 let search = chooser::Replay::new(&[], &Fixed);
10163 let out = integer::encode_with(values, &search)?;
10164 self.settle(&out, values.len(), search.first_offered())?;
10165 Ok(out)
10166 }
10167
10168 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
10169 let kinds = integer::shape(out)?;
10170 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
10171 self.since = 0;
10172 Ok(())
10173 }
10174}
10175
10176#[derive(Debug)]
10178struct Shape {
10179 kinds: Vec<integer::Kind>,
10180 offered: Vec<integer::Kind>,
10181 len: usize,
10182 rows: usize,
10183}
10184
10185fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
10223 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
10224 let mut payload = 0_usize;
10225 for row in 0..flat.len() {
10226 let text = flat.bytes_at(row).unwrap_or(b"");
10229 payload = payload.saturating_add(text.len());
10230 values.push(text);
10231 }
10232 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
10234 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
10235 return Ok(None);
10236 };
10237 Ok((out.len() < plain).then_some(out))
10238}
10239
10240fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
10241 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
10242 let coded = integer::encode_with(&wide, &Codes)?;
10243 let plain = codes.len().saturating_mul(size_of::<u32>());
10244 Ok((coded.len() < plain).then_some(coded))
10245}
10246
10247fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
10250 let flag = match flat.validity() {
10251 Validity::AllValid => 0,
10252 Validity::AllInvalid => 1,
10253 Validity::Mask(_) => 2,
10254 };
10255 out.push(flag);
10256 if flag == 2 {
10257 for group in (0..flat.len()).step_by(8) {
10258 let mut bits = 0_u8;
10259 for bit in 0..8 {
10260 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
10261 bits |= 1 << bit;
10262 }
10263 }
10264 out.push(bits);
10265 }
10266 }
10267}
10268
10269fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
10276 let coded = encoded_codes(codes)?;
10277 let mut out = Vec::with_capacity(
10278 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
10279 );
10280 out.push(if coded.is_some() { 4 } else { 3 });
10281 out.extend_from_slice(validity);
10282 match coded {
10283 Some(coded) => out.extend_from_slice(&coded),
10284 None => {
10285 for &code in codes {
10286 put_u32(&mut out, code);
10287 }
10288 }
10289 }
10290 Ok(out)
10291}
10292
10293fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
10296 let ty = vector.logical_type();
10297 let flat = vector.flatten()?;
10299 let mut out = Vec::new();
10300 let dictionary = if coded_type(ty) { string_dictionary(&flat)? } else { None };
10301 let compressed_text =
10302 if dictionary.is_none() && coded_type(ty) { text_compressed(&flat)? } else { None };
10303 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
10304 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
10305 let cascade =
10309 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
10310 out.push(if cascade.is_some() {
10311 5
10312 } else if dictionary.is_some() {
10313 1
10314 } else if compressed_text.is_some() {
10315 6
10316 } else if packed.is_some() {
10317 2
10318 } else {
10319 0
10320 });
10321 push_validity(&mut out, &flat);
10322 if let Some(cascade) = cascade {
10323 out.extend_from_slice(&cascade);
10324 return Ok(out);
10325 }
10326 if let Some(dictionary) = dictionary {
10327 out.extend_from_slice(&dictionary);
10328 return Ok(out);
10329 }
10330 if let Some(compressed_text) = compressed_text {
10331 out.extend_from_slice(&compressed_text);
10332 return Ok(out);
10333 }
10334 if let Some(packed) = packed {
10335 if packed.offset() != 0 {
10336 return Err(invalid("writer received a sliced packed vector"));
10337 }
10338 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
10339 out.extend_from_slice(&packed.base().to_le_bytes());
10340 put_u32(
10341 &mut out,
10342 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
10343 );
10344 for word in packed.words() {
10345 put_u64(&mut out, *word);
10346 }
10347 return Ok(out);
10348 }
10349 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
10350 match (ty, data) {
10351 (LogicalType::TinyInt, Data::Int8(values)) => {
10352 for value in &**values {
10353 out.extend_from_slice(&value.to_le_bytes());
10354 }
10355 }
10356 (LogicalType::UTinyInt, Data::UInt8(values)) => {
10357 for value in &**values {
10358 out.extend_from_slice(&value.to_le_bytes());
10359 }
10360 }
10361 (LogicalType::SmallInt, Data::Int16(values)) => {
10362 for value in &**values {
10363 out.extend_from_slice(&value.to_le_bytes());
10364 }
10365 }
10366 (LogicalType::USmallInt, Data::UInt16(values)) => {
10367 for value in &**values {
10368 out.extend_from_slice(&value.to_le_bytes());
10369 }
10370 }
10371 (LogicalType::UInteger, Data::UInt32(values)) => {
10372 for value in &**values {
10373 out.extend_from_slice(&value.to_le_bytes());
10374 }
10375 }
10376 (LogicalType::UBigInt, Data::UInt64(values)) => {
10377 for value in &**values {
10378 out.extend_from_slice(&value.to_le_bytes());
10379 }
10380 }
10381 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
10382 for value in &**values {
10383 out.extend_from_slice(&value.to_le_bytes());
10384 }
10385 }
10386 (
10387 LogicalType::BigInt
10388 | LogicalType::Timestamp
10389 | LogicalType::Time
10390 | LogicalType::TimeTz
10391 | LogicalType::TimestampTz
10392 | LogicalType::TimestampS
10393 | LogicalType::TimestampMs
10394 | LogicalType::TimestampNs,
10395 Data::Int64(values),
10396 ) => {
10397 for value in &**values {
10398 out.extend_from_slice(&value.to_le_bytes());
10399 }
10400 }
10401 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
10404 for value in &**values {
10405 out.extend_from_slice(&value.to_le_bytes());
10406 }
10407 }
10408 (LogicalType::UHugeInt, Data::UInt128(values)) => {
10409 for value in &**values {
10410 out.extend_from_slice(&value.to_le_bytes());
10411 }
10412 }
10413 (LogicalType::Float, Data::Float32(values)) => {
10416 for value in &**values {
10417 out.extend_from_slice(&value.to_le_bytes());
10418 }
10419 }
10420 (LogicalType::Double, Data::Float64(values)) => {
10421 for value in &**values {
10422 out.extend_from_slice(&value.to_le_bytes());
10423 }
10424 }
10425 (LogicalType::Interval, Data::Interval(values)) => {
10429 for (months, days, micros) in &**values {
10430 out.extend_from_slice(&months.to_le_bytes());
10431 out.extend_from_slice(&days.to_le_bytes());
10432 out.extend_from_slice(µs.to_le_bytes());
10433 }
10434 }
10435 (LogicalType::Boolean, Data::Bool(values)) => {
10436 for value in &**values {
10437 out.push(u8::from(*value));
10438 }
10439 }
10440 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
10443 for value in &**values {
10444 out.extend_from_slice(&value.to_le_bytes());
10445 }
10446 }
10447 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
10448 for value in &**values {
10449 out.extend_from_slice(&value.to_le_bytes());
10450 }
10451 }
10452 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
10453 for value in &**values {
10454 out.extend_from_slice(&value.to_le_bytes());
10455 }
10456 }
10457 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
10458 for value in &**values {
10459 out.extend_from_slice(&value.to_le_bytes());
10460 }
10461 }
10462 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
10467 let mut bytes = Vec::new();
10468 put_u32(&mut out, 0);
10469 for row in 0..vector.len() {
10470 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
10471 bytes.extend_from_slice(value);
10472 put_u32(
10473 &mut out,
10474 u32::try_from(bytes.len())
10475 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
10476 );
10477 }
10478 out.extend_from_slice(&bytes);
10479 }
10480 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10481 }
10482 Ok(out)
10483}
10484
10485fn put_varint(out: &mut Vec<u8>, mut value: u32) {
10486 while value >= 0x80 {
10487 out.push((value as u8 & 0x7f) | 0x80);
10488 value >>= 7;
10489 }
10490 out.push(value as u8);
10491}
10492
10493fn unique_codes(codes: &[u32]) -> Vec<u32> {
10495 let mut unique = codes.to_vec();
10496 unique.sort_unstable();
10497 unique.dedup();
10498 unique
10499}
10500
10501fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
10507 let mut lists = lists;
10508 while lists.len() > 1 {
10509 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
10510 for pair in lists.chunks(2) {
10511 match pair {
10512 [left, right] => next.push(merged_pair(left, right)),
10513 [only] => next.push(only.clone()),
10514 _ => {}
10515 }
10516 }
10517 lists = next;
10518 }
10519 lists.pop().unwrap_or_default()
10520}
10521
10522fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
10523 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
10524 let mut at = 0;
10525 let mut to = 0;
10526 while at < left.len() && to < right.len() {
10527 match left[at].cmp(&right[to]) {
10528 Ordering::Less => {
10529 out.push(left[at]);
10530 at += 1;
10531 }
10532 Ordering::Greater => {
10533 out.push(right[to]);
10534 to += 1;
10535 }
10536 Ordering::Equal => {
10537 out.push(left[at]);
10538 at += 1;
10539 to += 1;
10540 }
10541 }
10542 }
10543 out.extend_from_slice(&left[at..]);
10544 out.extend_from_slice(&right[to..]);
10545 out
10546}
10547
10548fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
10553 let mut merged = Range::default();
10554 let mut first = true;
10555 for range in ranges {
10556 merged.nulls = merged.nulls.saturating_add(range.nulls);
10557 merged.sum = match (merged.sum.take(), range.sum) {
10561 (Some(held), Some(next)) if !first => held.checked_add(next),
10562 (_, next) if first => next,
10563 _ => None,
10564 };
10565 merged.exact = if first { range.exact } else { merged.exact && range.exact };
10566 if first {
10567 merged.low = range.low;
10568 merged.high = range.high;
10569 first = false;
10570 continue;
10571 }
10572 merged.low = match (merged.low.take(), range.low) {
10573 (Some(held), Some(next)) => Some(held.smaller(next)),
10574 _ => None,
10575 };
10576 merged.high = match (merged.high.take(), range.high) {
10577 (Some(held), Some(next)) => Some(held.larger(next)),
10578 _ => None,
10579 };
10580 }
10581 merged
10582}
10583
10584fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
10597 match bound {
10598 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
10599 value.truncate(PART_BOUND_BYTES);
10600 if !high {
10601 return Some(Bound::Bytes(value));
10602 }
10603 while let Some(last) = value.pop() {
10604 if last < u8::MAX {
10605 value.push(last + 1);
10606 return Some(Bound::Bytes(value));
10607 }
10608 }
10609 None
10610 }
10611 other => other,
10612 }
10613}
10614
10615fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
10623 let mut out = Vec::new();
10624 put_u32(
10625 &mut out,
10626 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10627 );
10628 for range in ranges {
10629 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
10630 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
10631 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
10632 }
10633 Ok(out)
10634}
10635
10636fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
10638 let mut cur = Cursor::new(bytes);
10639 let parts = cur.u32()? as usize;
10640 let mut out = Vec::new();
10641 for _ in 0..parts {
10642 let low = cur.bound()?;
10643 let high = cur.bound()?;
10644 let nulls = cur.u32()? as usize;
10645 out.push(Range { low, high, nulls, exact: false, sum: None });
10646 }
10647 Ok(out)
10648}
10649
10650fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
10651 let held: Vec<&Option<Sieve>> = sieves.collect();
10652 let mut out = Vec::new();
10653 put_u32(
10654 &mut out,
10655 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
10656 );
10657 for sieve in &held {
10658 let length = sieve.as_ref().map_or(0, Sieve::len);
10659 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
10660 }
10661 for sieve in held.into_iter().flatten() {
10663 out.extend_from_slice(&sieve.to_bytes());
10664 }
10665 Ok(out)
10666}
10667
10668fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
10674 let parts = u32::from_le_bytes(
10675 bytes
10676 .get(..4)
10677 .ok_or_else(|| invalid("sieve page is truncated"))?
10678 .try_into()
10679 .map_err(|_| invalid("sieve page is truncated"))?,
10680 ) as usize;
10681 let mut lengths = Vec::with_capacity(parts);
10682 for part in 0..parts {
10683 let at = 4 + part * 4;
10684 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
10685 lengths.push(u32::from_le_bytes(
10686 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
10687 ) as usize);
10688 }
10689 let mut at = 4 + parts * 4;
10690 let mut out = Vec::with_capacity(parts);
10691 for length in lengths {
10692 if length == 0 {
10693 out.push(None);
10694 continue;
10695 }
10696 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
10697 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
10698 out.push(Sieve::from_bytes(field));
10699 at = end;
10700 }
10701 if at != bytes.len() {
10702 return Err(invalid("sieve page has trailing bytes"));
10703 }
10704 Ok(out)
10705}
10706
10707fn encode_membership(unique: &[u32]) -> Vec<u8> {
10713 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
10714 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
10715 let mut previous = 0;
10716 for (at, &code) in unique.iter().enumerate() {
10717 put_varint(&mut out, if at == 0 { code } else { code - previous });
10718 previous = code;
10719 }
10720 out
10721}
10722
10723fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
10724 let mut value = 0_u32;
10725 for shift in (0..35).step_by(7) {
10726 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
10727 *at += 1;
10728 let part = u32::from(byte & 0x7f);
10729 if shift == 28 && part > 0x0f {
10730 return Err(invalid("membership varint overflow"));
10731 }
10732 value = value
10733 .checked_add(
10734 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
10735 )
10736 .ok_or_else(|| invalid("membership varint overflow"))?;
10737 if byte & 0x80 == 0 {
10738 return Ok(value);
10739 }
10740 }
10741 Err(invalid("membership varint is too long"))
10742}
10743
10744fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
10745 let mut at = 0;
10746 let count = take_varint(bytes, &mut at)? as usize;
10747 let mut codes = Vec::with_capacity(count);
10748 let mut previous = 0_u32;
10749 for index in 0..count {
10750 let delta = take_varint(bytes, &mut at)?;
10751 let code = if index == 0 {
10752 delta
10753 } else {
10754 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
10755 };
10756 if index > 0 && code <= previous {
10757 return Err(invalid("membership codes are not increasing"));
10758 }
10759 codes.push(code);
10760 previous = code;
10761 }
10762 if at != bytes.len() {
10763 return Err(invalid("membership page has trailing bytes"));
10764 }
10765 Ok(codes)
10766}
10767
10768fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
10769 let mut by_text = HashMap::new();
10770 let mut values = Vec::new();
10771 let mut codes = Vec::with_capacity(vector.len());
10772 let mut plain_bytes = 0_usize;
10773 for row in 0..vector.len() {
10774 let text = vector.bytes_at(row).unwrap_or(b"");
10775 plain_bytes = plain_bytes.saturating_add(text.len());
10776 let code = match by_text.get(text) {
10777 Some(&code) => code,
10778 None => {
10779 let code = u32::try_from(values.len())
10780 .map_err(|_| invalid("too many dictionary values"))?;
10781 by_text.insert(text, code);
10782 values.push(text);
10783 code
10784 }
10785 };
10786 codes.push(code);
10787 }
10788 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
10789 let encoded = 8_usize
10790 .saturating_add((values.len() + 1).saturating_mul(4))
10791 .saturating_add(dictionary_bytes)
10792 .saturating_add(codes.len().saturating_mul(4));
10793 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
10794 if encoded >= plain {
10795 return Ok(None);
10796 }
10797 let mut out = Vec::with_capacity(encoded);
10798 put_u32(
10799 &mut out,
10800 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
10801 );
10802 put_u32(
10803 &mut out,
10804 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
10805 );
10806 let mut offset = 0_u32;
10807 put_u32(&mut out, offset);
10808 for value in &values {
10809 offset = offset
10810 .checked_add(
10811 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
10812 )
10813 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
10814 put_u32(&mut out, offset);
10815 }
10816 for value in values {
10817 out.extend_from_slice(value);
10818 }
10819 for code in codes {
10820 put_u32(&mut out, code);
10821 }
10822 Ok(Some(out))
10823}
10824
10825struct Room<'a, T> {
10827 state: &'a Mutex<(T, usize)>,
10828 finished: &'a Condvar,
10829 bytes: usize,
10830}
10831
10832impl<T> Drop for Room<'_, T> {
10833 fn drop(&mut self) {
10834 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
10835 held.1 -= self.bytes;
10836 drop(held);
10837 self.finished.notify_all();
10838 }
10839}
10840
10841enum Closing<'a> {
10843 Numeric {
10845 column: usize,
10846 counted: bool,
10847 },
10848 Dictionary {
10849 index: usize,
10850 dictionary: &'a GlobalDictionary,
10851 },
10852}
10853
10854enum Closed {
10856 Numeric(usize, (Option<FrequencySummary>, Option<u64>)),
10857 Dictionary(usize, ClosedDictionary),
10858}
10859
10860struct ClosedDictionary {
10862 distinct: Option<u64>,
10864 frequencies: Option<FrequencySummary>,
10865 texts: Vec<Option<Vec<u8>>>,
10866 hosts: Option<host::HostSummary>,
10867 encoded: EncodedDictionary,
10868 payload: u64,
10870}
10871
10872struct EncodedDictionary {
10873 index: Vec<u8>,
10874 ranks: Vec<u8>,
10875 grams: Vec<u8>,
10876}
10877
10878fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
10919 let mut work = vec![(0, codes.len(), 0)];
10920 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
10921 while let Some((from, to, depth)) = work.pop() {
10922 let part = &mut codes[from..to];
10923 keyed.clear();
10924 keyed.extend(part.iter().map(|&code| {
10925 let value = values(code);
10926 let rest = value.get(depth..).unwrap_or_default();
10927 (head(rest), rest.len().min(8) as u8, code)
10928 }));
10929 keyed.sort_unstable();
10930 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
10931 *slot = entry.2;
10932 }
10933 let mut start = 0;
10934 while start < keyed.len() {
10935 let (key, taken, _) = keyed[start];
10936 let mut end = start + 1;
10937 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
10938 end += 1;
10939 }
10940 if taken == 8 && end - start > 1 {
10941 work.push((from + start, from + end, depth + 8));
10942 }
10943 start = end;
10944 }
10945 }
10946}
10947
10948const PARALLEL_SORT_MIN: usize = 1 << 16;
10950
10951const BUCKETS_PER_WORKER: usize = 4;
10954
10955const SAMPLES_PER_BUCKET: usize = 32;
10957
10958fn sort_by_value_across<'a>(
10976 codes: &mut [u32],
10977 values: impl Fn(u32) -> &'a [u8] + Sync,
10978 workers: usize,
10979) {
10980 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
10981 sort_by_value(codes, values);
10982 return;
10983 }
10984 let buckets = workers * BUCKETS_PER_WORKER;
10985 let wanted = buckets * SAMPLES_PER_BUCKET;
10986 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
10987 sort_by_value(&mut sample, &values);
10988 let splitters =
10989 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
10990 let values = &values;
10991 let splitters = &splitters;
10992 let per = codes.len().div_ceil(workers);
10993 let places = std::thread::scope(|scope| {
10995 codes
10996 .chunks(per)
10997 .map(|run| {
10998 scope.spawn(move || {
10999 run.iter()
11000 .map(|&code| {
11001 let value = values(code);
11002 splitters.partition_point(|splitter| *splitter <= value) as u32
11003 })
11004 .collect::<Vec<_>>()
11005 })
11006 })
11007 .collect::<Vec<_>>()
11008 .into_iter()
11009 .flat_map(|handle| {
11010 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
11011 })
11012 .collect::<Vec<_>>()
11013 });
11014 let mut starts = vec![0_usize; buckets + 1];
11015 for &place in &places {
11016 starts[place as usize + 1] += 1;
11017 }
11018 for bucket in 0..buckets {
11019 starts[bucket + 1] += starts[bucket];
11020 }
11021 let mut laid = vec![0_u32; codes.len()];
11022 let mut next = starts.clone();
11023 for (&code, &place) in codes.iter().zip(&places) {
11024 laid[next[place as usize]] = code;
11025 next[place as usize] += 1;
11026 }
11027 drop(places);
11028 let mut runs = Vec::with_capacity(buckets);
11029 let mut rest = laid.as_mut_slice();
11030 for bucket in 0..buckets {
11031 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
11032 runs.push(run);
11033 rest = after;
11034 }
11035 runs.sort_by_key(|run| run.len());
11037 let queue = Mutex::new(runs);
11038 std::thread::scope(|scope| {
11039 for _ in 0..workers {
11040 scope.spawn(|| {
11041 loop {
11042 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
11043 let Some(run) = taken else { break };
11044 sort_by_value(run, values);
11045 }
11046 });
11047 }
11048 });
11049 codes.copy_from_slice(&laid);
11050}
11051
11052fn head(bytes: &[u8]) -> u64 {
11054 let mut word = [0; 8];
11055 let take = bytes.len().min(8);
11056 word[..take].copy_from_slice(&bytes[..take]);
11057 u64::from_be_bytes(word)
11058}
11059
11060fn encode_global_dictionary(
11071 dictionary: &GlobalDictionary,
11072 order: &[(u64, u32)],
11073 places: &[Placed],
11074 scattered: bool,
11075) -> Result<EncodedDictionary> {
11076 let values = dictionary.values();
11077 if order.len() != values {
11078 return Err(invalid("global dictionary order does not cover its values"));
11079 }
11080 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
11081 if places.len() != blocks {
11082 return Err(invalid("global dictionary payload is not the blocks it says it is"));
11083 }
11084 if dictionary.grams.len() != blocks {
11085 return Err(invalid("global dictionary signatures do not cover its blocks"));
11086 }
11087 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
11088 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
11089 let offset_bits = offset_width(&dictionary.ends);
11090 let payload_words = if scattered { 3 } else { 2 };
11091 let index_len = DICTIONARY_HEADER
11092 .checked_add(offset_bytes(values, offset_bits))
11093 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
11094 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11095 .and_then(|len| len.checked_add(8))
11096 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
11097 let mut index = Vec::with_capacity(index_len);
11098 put_u32(
11099 &mut index,
11100 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
11101 );
11102 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
11103 put_u32(
11104 &mut index,
11105 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
11106 );
11107 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
11108 | DICTIONARY_GRAMS
11109 | DICTIONARY_WIDE_GRAMS;
11110 put_u32(&mut index, offset_bits as u32 | flag);
11111 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
11112 let mut end = 0_u64;
11117 for place in places {
11118 if scattered {
11119 put_u64(&mut index, place.start);
11120 put_u64(&mut index, place.length);
11121 } else {
11122 end = end
11123 .checked_add(place.length)
11124 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
11125 put_u64(&mut index, end);
11126 }
11127 }
11128 for place in places {
11129 put_u64(&mut index, place.hash);
11130 }
11131 if rank_ends.len() != rank_blocks {
11134 return Err(invalid("global dictionary order is not the blocks it says it is"));
11135 }
11136 for end in &rank_ends {
11137 put_u64(&mut index, *end);
11138 }
11139 let mut at = 0_usize;
11140 for end in &rank_ends {
11141 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
11142 put_u64(&mut index, checksum(&ranks[at..end]));
11143 at = end;
11144 }
11145 let gram_len = blocks
11146 .checked_mul(TEXT_GRAM_BYTES)
11147 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
11148 let mut grams = Vec::with_capacity(gram_len);
11149 for block in &dictionary.grams {
11150 grams.extend_from_slice(block);
11151 }
11152 put_u64(&mut index, checksum(&grams));
11153 if index.len() != index_len {
11154 return Err(invalid("global dictionary index is not the length it was laid out for"));
11155 }
11156 Ok(EncodedDictionary { index, ranks, grams })
11157}
11158
11159const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
11166
11167fn payload_shapes() -> Vec<chooser::Settled> {
11193 let integers = vec![integer::Kind::Packed];
11194 [
11195 vec![string::Kind::Front, string::Kind::Lz],
11196 vec![string::Kind::Lz, string::Kind::Fsst],
11197 vec![string::Kind::Lz, string::Kind::Plain],
11198 vec![string::Kind::Fsst],
11199 vec![string::Kind::Plain],
11200 ]
11201 .into_iter()
11202 .map(|strings| chooser::Settled::new(strings, integers.clone()))
11203 .collect()
11204}
11205
11206fn synced(file: &dyn rudb_io::File, profile: Option<&LoadProfile>) -> Result<()> {
11213 let started = profile.map(|_| std::time::Instant::now());
11214 file.sync()?;
11215 if let (Some(profile), Some(started)) = (profile, started) {
11216 profile.waited(
11217 Stage::Publish,
11218 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
11219 );
11220 }
11221 Ok(())
11222}
11223
11224#[derive(Debug)]
11229pub(crate) struct Unencoded {
11230 column: usize,
11231 at: usize,
11232 ends: Vec<u32>,
11233 bytes: Vec<u8>,
11234 shape: chooser::Settled,
11235}
11236
11237impl Unencoded {
11238 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
11240 let values = block_values(&self.ends, &self.bytes);
11241 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
11242 }
11243
11244 pub(crate) fn place(&self) -> (usize, usize) {
11246 (self.column, self.at)
11247 }
11248}
11249
11250pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
11254
11255fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
11257 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
11258 for value in values {
11259 for gram in value.windows(4) {
11260 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
11261 grams[bit / 8] |= 1 << (bit % 8);
11262 }
11263 }
11264 }
11265 grams
11266}
11267
11268fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
11270 let mut out = Vec::with_capacity(ends.len());
11271 let mut from = 0;
11272 for &to in ends {
11273 out.push(&bytes[from..to as usize]);
11274 from = to as usize;
11275 }
11276 out
11277}
11278
11279fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11286 for dictionary in dictionaries.iter_mut().flatten() {
11287 if !dictionary.early.is_empty() {
11288 return Err(Error::internal("a dictionary block handed out never came back"));
11289 }
11290 dictionary.seal_rest();
11291 dictionary.settle_rest()?;
11292 }
11293 encode_waiting(dictionaries)?;
11294 if dictionaries
11297 .iter()
11298 .flatten()
11299 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
11300 {
11301 return Err(Error::internal("a dictionary block handed out never came back"));
11302 }
11303 Ok(())
11304}
11305
11306fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
11309 let jobs = dictionaries
11310 .iter()
11311 .enumerate()
11312 .flat_map(|(column, held)| {
11313 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
11314 })
11315 .collect::<Vec<_>>();
11316 if jobs.is_empty() {
11317 return Ok(());
11318 }
11319 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
11320 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
11321 Ok((column, at, held.encode_waiting(at)?))
11322 };
11323 let workers = std::thread::available_parallelism()
11324 .map_or(1, usize::from)
11325 .min(MAX_FREQUENCY_WORKERS)
11326 .min(jobs.len());
11327 let made = if workers <= 1 {
11328 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
11329 } else {
11330 let next = AtomicUsize::new(0);
11331 let jobs = &jobs;
11332 let pieces = std::thread::scope(|scope| {
11333 (0..workers)
11334 .map(|_| {
11335 scope.spawn(|| {
11336 let mut mine = Vec::new();
11337 loop {
11338 let job = next.fetch_add(1, Atomic::Relaxed);
11339 let Some(&(column, at)) = jobs.get(job) else { break };
11340 mine.push(one(column, at)?);
11341 }
11342 Ok(mine)
11343 })
11344 })
11345 .collect::<Vec<_>>()
11346 .into_iter()
11347 .map(|handle| {
11348 handle
11349 .join()
11350 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
11351 })
11352 .collect::<Result<Vec<_>>>()
11353 })?;
11354 pieces.into_iter().flatten().collect()
11355 };
11356 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
11357 (0..dictionaries.len()).map(|_| Vec::new()).collect();
11358 for (column, at, bytes) in made {
11359 done[column].push((at, bytes));
11360 }
11361 for (column, mut made) in done.into_iter().enumerate() {
11362 if made.is_empty() {
11363 continue;
11364 }
11365 let Some(held) = dictionaries[column].as_mut() else { continue };
11366 made.sort_by_key(|(at, _)| *at);
11367 let waiting = std::mem::take(&mut held.waiting);
11368 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
11369 if held.encoded() != at {
11370 return Err(Error::internal("a dictionary block was encoded out of order"));
11371 }
11372 held.push_block(block);
11373 }
11374 }
11375 Ok(())
11376}
11377
11378fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
11388 let mut best: Option<(chooser::Settled, usize)> = None;
11389 for shape in payload_shapes() {
11390 let mut size = 0;
11391 for block in sample {
11392 size += string::encode_with(block, &shape)?.len();
11393 }
11394 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
11395 best = Some((shape, size));
11396 }
11397 }
11398 best.map(|(shape, _)| shape)
11399 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
11400}
11401
11402fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
11409 let mut out = Vec::with_capacity(order.len() * 4);
11410 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
11411 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
11412 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
11413 for block in order.chunks(TEXT_RANK_BLOCK) {
11414 let base = block.first().map_or(0, |&(head, _)| head);
11417 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
11418 let width = (u64::BITS - span.leading_zeros()) as usize;
11419 heads.clear();
11420 codes.clear();
11421 for &(head, code) in block {
11422 heads.push(head.wrapping_sub(base));
11423 codes.push(u64::from(code));
11424 }
11425 put_u64(&mut out, base);
11426 out.push(width as u8);
11427 bitpack::pack_tail(&heads, width, &mut out)
11428 .map_err(|_| invalid("global dictionary heads do not pack"))?;
11429 bitpack::pack_tail(&codes, code_bits, &mut out)
11430 .map_err(|_| invalid("global dictionary codes do not pack"))?;
11431 ends.push(out.len() as u64);
11432 }
11433 Ok((out, ends))
11434}
11435
11436fn open_global_dictionary(
11443 file: Arc<File>,
11444 page: Page,
11445 ty: &LogicalType,
11446 keep_budget: usize,
11447) -> Result<Vector> {
11448 if !coded_type(ty) {
11449 return Err(invalid("global dictionary belongs to a non-string column"));
11450 }
11451 let mut header = [0; DICTIONARY_HEADER];
11452 read_at(&file, page.offset, &mut header)?;
11453 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
11454 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
11455 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
11456 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11457 let scattered = width & DICTIONARY_SCATTERED != 0;
11458 let has_grams = width & DICTIONARY_GRAMS != 0;
11459 let gram_width =
11460 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
11461 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
11462 if per_block != TEXT_PAYLOAD_VALUES {
11463 return Err(invalid("global dictionary block width differs"));
11464 }
11465 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
11466 return Err(invalid("global dictionary block count differs from its value count"));
11467 }
11468 if offset_bits > u32::BITS as usize {
11469 return Err(invalid("global dictionary packs offsets past a payload"));
11470 }
11471 let offset_len = offset_bytes(count, offset_bits);
11472 let ranks = count;
11477 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
11478 let payload_words = if scattered { 3 } else { 2 };
11482 let hash_len = blocks
11483 .checked_mul(payload_words * 8)
11484 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
11485 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
11486 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
11487 let gram_len = if has_grams {
11488 blocks
11489 .checked_mul(gram_width)
11490 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
11491 } else {
11492 0
11493 };
11494 let index_len = DICTIONARY_HEADER
11495 .checked_add(offset_len)
11496 .and_then(|len| len.checked_add(hash_len))
11497 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11498 if index_len > page.length as usize {
11499 return Err(invalid("global dictionary offset index exceeds its page"));
11500 }
11501 let mut index = vec![0; index_len];
11502 index[..DICTIONARY_HEADER].copy_from_slice(&header);
11503 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
11504 if checksum(&index) != page.hash {
11505 return Err(invalid("global dictionary index checksum differs"));
11506 }
11507 let word_end = index_len - usize::from(has_grams) * 8;
11508 let gram_hash = has_grams
11509 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
11510 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
11511 .chunks_exact(8)
11512 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
11513 .collect::<Vec<_>>();
11514 let mut rest = words.split_off(blocks * payload_words);
11515 let rank_hashes = rest.split_off(rank_blocks);
11516 let rank_ends = rest;
11517 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
11520 return Err(invalid("global dictionary order blocks do not rise"));
11521 }
11522 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
11523 .map_err(|_| invalid("global dictionary rank overflow"))?;
11524 let body_len = index_len
11525 .checked_add(rank_len)
11526 .ok_or_else(|| invalid("global dictionary header overflow"))?;
11527 if body_len > page.length as usize {
11528 return Err(invalid("global dictionary order exceeds its page"));
11529 }
11530 let gram_end = body_len
11531 .checked_add(gram_len)
11532 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
11533 if gram_end > page.length as usize {
11534 return Err(invalid("global dictionary signatures exceed their page"));
11535 }
11536 let grams = gram_hash.map(|hash| NativeGrams {
11537 start: page.offset + body_len as u64,
11538 length: gram_len,
11539 width: gram_width,
11540 hash,
11541 verdicts: Mutex::new(Vec::new()),
11542 });
11543 let mut offsets = index;
11547 offsets.truncate(DICTIONARY_HEADER + offset_len);
11548 let hashes = words.split_off(blocks * (payload_words - 1));
11549 let (starts, lengths) = if scattered {
11550 let mut starts = Vec::with_capacity(blocks);
11551 let mut lengths = Vec::with_capacity(blocks);
11552 for pair in words.chunks_exact(2) {
11553 starts.push(pair[0]);
11554 lengths.push(pair[1]);
11555 }
11556 (starts, lengths)
11557 } else {
11558 let base = page.offset + gram_end as u64;
11562 let mut starts = Vec::with_capacity(blocks);
11563 let mut lengths = Vec::with_capacity(blocks);
11564 let mut at = 0_u64;
11565 for &end in &words {
11566 let len = end
11567 .checked_sub(at)
11568 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
11569 starts.push(base + at);
11570 lengths.push(len);
11571 at = end;
11572 }
11573 (starts, lengths)
11574 };
11575 let stored_len = page.length as u64 - gram_end as u64;
11581 if scattered && stored_len == 0 {
11582 let size = file.metadata().map_err(io)?.len();
11583 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
11584 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
11585 });
11586 if !inside {
11587 return Err(invalid("global dictionary block lies outside the file"));
11588 }
11589 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
11590 return Err(invalid("global dictionary blocks do not bound the payload"));
11591 }
11592 Vector::external_text(
11593 ty.clone(),
11594 Arc::new(NativeText {
11595 file,
11596 values: count,
11597 offsets,
11598 offset_bits,
11599 value_ends: OnceLock::new(),
11600 value_lens: OnceLock::new(),
11601 ends_asked: AtomicUsize::new(0),
11602 ranks,
11603 rank_at: page.offset + index_len as u64,
11604 rank_ends,
11605 rank_hashes,
11606 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
11607 code_bits: code_width(count),
11608 code_ranks: OnceLock::new(),
11609 starts,
11610 lengths,
11611 hashes,
11612 grams,
11613 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
11614 char_lens: (0..blocks).map(|_| OnceLock::new()).collect(),
11615 keep_budget,
11616 payload_kept: AtomicUsize::new(0),
11617 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
11618 visit_dropped: AtomicUsize::new(0),
11619 searched: Mutex::new(HashMap::new()),
11620 }),
11621 )
11622}
11623
11624fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
11637 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
11639 let mut cur = Cursor::new(bytes);
11640 let codec = cur.u8()?;
11641 if cur.u8()? == 2 {
11642 cur.take(rows.div_ceil(8))?;
11643 }
11644 Ok((codec, cur.at))
11645 }
11646 let Ok((codec, at)) = cascade_at(rows, bytes) else {
11647 return "UNREADABLE".to_string();
11648 };
11649 let tail = &bytes[at..];
11650 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
11651 match codec {
11652 0 => match ty {
11653 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
11654 _ => "FIXED".to_string(),
11655 },
11656 1 => "DICT(PLAIN)".to_string(),
11657 2 => "FOR+BITPACK".to_string(),
11658 3 => "TABLE DICT".to_string(),
11659 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
11660 5 => described(integer::describe(tail)),
11661 6 => described(string::describe(tail)),
11662 other => format!("CODEC {other}"),
11663 }
11664}
11665
11666fn decode_selected_stable_codes(
11671 rows: usize,
11672 bytes: &[u8],
11673 positions: &[usize],
11674 out: &mut Vec<Option<u32>>,
11675) -> Result<bool> {
11676 if positions.windows(2).any(|pair| pair[0] >= pair[1])
11677 || positions.last().is_some_and(|&position| position >= rows)
11678 {
11679 return Err(invalid("selected code positions are not sorted and in range"));
11680 }
11681 let mut cur = Cursor::new(bytes);
11682 let codec = cur.u8()?;
11683 if codec != 3 && codec != 4 {
11684 return Ok(false);
11685 }
11686 let flag = cur.u8()?;
11687 let mask = match flag {
11688 0 | 1 => None,
11689 2 => {
11690 let at = cur.at;
11691 let len = rows.div_ceil(8);
11692 cur.take(len)?;
11693 Some((at, len))
11694 }
11695 _ => return Err(invalid("page validity tag differs")),
11696 };
11697 let valid = |row: usize| match flag {
11698 0 => true,
11699 1 => false,
11700 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
11701 _ => unreachable!("the validity tag was checked"),
11702 };
11703 if codec == 4 {
11704 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
11705 for (&row, code) in positions.iter().zip(wide) {
11706 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
11707 out.push(valid(row).then_some(code));
11708 }
11709 return Ok(true);
11710 }
11711 let codes_at = cur.at;
11712 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
11713 cur.take(codes_len)?;
11714 if cur.at != bytes.len() {
11715 return Err(invalid("global code page has trailing bytes"));
11716 }
11717 let codes = &bytes[codes_at..codes_at + codes_len];
11718 for &row in positions {
11719 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
11720 let code = u32::from_le_bytes(
11721 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
11722 );
11723 out.push(valid(row).then_some(code));
11724 }
11725 Ok(true)
11726}
11727
11728fn decode_at(
11734 ty: &LogicalType,
11735 rows: usize,
11736 bytes: &[u8],
11737 global: Option<Arc<Vector>>,
11738 positions: &[u32],
11739) -> Result<Vector> {
11740 if positions.last().is_some_and(|&last| last as usize >= rows) {
11741 return Err(invalid("a position is past the end of the part"));
11742 }
11743 if bytes.first() != Some(&6) {
11744 return decode(ty, rows, bytes, global)?.gather(positions);
11745 }
11746 if !coded_type(ty) {
11747 return Err(invalid("compressed text codec belongs to a non-string page"));
11748 }
11749 let mut cur = Cursor::new(bytes);
11750 cur.u8()?;
11751 let validity = match cur.u8()? {
11752 0 => Validity::AllValid,
11753 1 => Validity::AllInvalid,
11754 2 => {
11755 let mask = cur.take(rows.div_ceil(8))?;
11756 Validity::from_iter(positions.len(), |at| {
11757 let row = positions[at] as usize;
11758 mask[row / 8] >> (row % 8) & 1 == 1
11759 })
11760 }
11761 _ => return Err(invalid("page validity tag differs")),
11762 };
11763 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
11764 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11765 let mut start = 0;
11766 for end in ends {
11767 let len = end
11768 .checked_sub(start)
11769 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
11770 push_value(&mut values, ty, start, len)?;
11771 start = end;
11772 }
11773 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
11774}
11775
11776fn push_value(values: &mut StringColumn, ty: &LogicalType, at: usize, len: usize) -> Result<()> {
11779 if ty == &LogicalType::Varchar {
11780 values.push_in_place(at, len)?;
11781 } else {
11782 values.push_bytes_in_place(at, len)?;
11783 }
11784 Ok(())
11785}
11786
11787fn decode(
11788 ty: &LogicalType,
11789 rows: usize,
11790 bytes: &[u8],
11791 global: Option<Arc<Vector>>,
11792) -> Result<Vector> {
11793 let mut cur = Cursor::new(bytes);
11794 let codec = cur.u8()?;
11795 let flag = cur.u8()?;
11796 let validity = match flag {
11797 0 => Validity::AllValid,
11798 1 => Validity::AllInvalid,
11799 2 => {
11800 let mask = cur.take(rows.div_ceil(8))?;
11801 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
11802 }
11803 _ => return Err(invalid("page validity tag differs")),
11804 };
11805 if codec == 1 {
11806 if !coded_type(ty) {
11807 return Err(invalid("dictionary codec belongs to a non-string page"));
11808 }
11809 let count = cur.u32()? as usize;
11810 let payload_len = cur.u32()? as usize;
11811 let offset_bytes = cur.take(
11812 (count + 1)
11813 .checked_mul(4)
11814 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
11815 )?;
11816 let offsets = offset_bytes
11817 .chunks_exact(4)
11818 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11819 .collect::<Vec<_>>();
11820 let payload = cur.take(payload_len)?.to_vec();
11821 if offsets.first() != Some(&0)
11822 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11823 || offsets.windows(2).any(|pair| pair[0] > pair[1])
11824 {
11825 return Err(invalid("dictionary offsets do not bound the payload"));
11826 }
11827 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
11830 for pair in offsets.windows(2) {
11831 push_value(&mut strings, ty, pair[0] as usize, (pair[1] - pair[0]) as usize)?;
11832 }
11833 let mut codes = Vec::with_capacity(rows);
11834 for _ in 0..rows {
11835 codes.push(cur.u32()?);
11836 }
11837 if codes.iter().any(|code| *code as usize >= count) {
11838 return Err(invalid("dictionary code is out of range"));
11839 }
11840 if cur.at != bytes.len() {
11841 return Err(invalid("dictionary page has trailing bytes"));
11842 }
11843 let dictionary = Vector::flat(ty.clone(), Data::Varlen(strings))?;
11844 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
11845 }
11846 if codec == 3 || codec == 4 {
11847 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
11848 let codes = if codec == 4 {
11849 let wide = integer::decode(&bytes[cur.at..])?;
11852 if wide.len() != rows {
11853 return Err(invalid("encoded code page holds the wrong number of rows"));
11854 }
11855 let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
11862 if seen < 0 || seen > i64::from(u32::MAX) {
11863 return Err(invalid("code is not a code"));
11864 }
11865 wide.iter().map(|&code| code as u32).collect()
11866 } else {
11867 let mut codes = Vec::with_capacity(rows);
11868 for _ in 0..rows {
11869 codes.push(cur.u32()?);
11870 }
11871 if cur.at != bytes.len() {
11872 return Err(invalid("global code page has trailing bytes"));
11873 }
11874 codes
11875 };
11876 let highest = codes.iter().copied().max();
11877 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
11878 .with_validity(validity));
11879 }
11880 if codec == 6 {
11881 if !coded_type(ty) {
11882 return Err(invalid("compressed text codec belongs to a non-string page"));
11883 }
11884 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
11888 if ends.len() != rows {
11889 return Err(invalid("compressed text page holds the wrong number of rows"));
11890 }
11891 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11894 let mut start = 0;
11895 for end in ends {
11896 let len = end
11897 .checked_sub(start)
11898 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
11899 push_value(&mut values, ty, start, len)?;
11900 start = end;
11901 }
11902 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
11903 }
11904 if codec == 5 {
11905 let values = integer::decode(&bytes[cur.at..])?;
11907 if values.len() != rows {
11908 return Err(invalid("cascade page holds the wrong number of rows"));
11909 }
11910 let data = narrowed(ty, values)?;
11911 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
11912 }
11913 if codec == 2 {
11914 let width = u32::from(cur.u8()?);
11915 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
11916 let count = cur.u32()? as usize;
11917 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
11918 let words: Vec<u64> = cur
11919 .take(length)?
11920 .chunks_exact(8)
11921 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
11922 .collect();
11923 if cur.at != bytes.len() {
11924 return Err(invalid("packed page has trailing bytes"));
11925 }
11926 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
11927 }
11928 if codec != 0 {
11929 return Err(invalid("page codec is unknown"));
11930 }
11931 let data = match ty {
11932 LogicalType::TinyInt => {
11933 let values = cur.take(rows)?;
11934 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
11935 }
11936 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
11937 LogicalType::SmallInt => {
11938 let values =
11939 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11940 Data::Int16(
11941 values
11942 .chunks_exact(2)
11943 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
11944 .collect::<Vec<_>>()
11945 .into(),
11946 )
11947 }
11948 LogicalType::USmallInt => {
11949 let values =
11950 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11951 Data::UInt16(
11952 values
11953 .chunks_exact(2)
11954 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
11955 .collect::<Vec<_>>()
11956 .into(),
11957 )
11958 }
11959 LogicalType::UInteger => {
11960 let values =
11961 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11962 Data::UInt32(
11963 values
11964 .chunks_exact(4)
11965 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
11966 .collect::<Vec<_>>()
11967 .into(),
11968 )
11969 }
11970 LogicalType::UBigInt => {
11971 let values =
11972 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11973 Data::UInt64(
11974 values
11975 .chunks_exact(8)
11976 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
11977 .collect::<Vec<_>>()
11978 .into(),
11979 )
11980 }
11981 LogicalType::Integer | LogicalType::Date => {
11982 let values =
11983 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11984 Data::Int32(
11985 values
11986 .chunks_exact(4)
11987 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
11988 .collect::<Vec<_>>()
11989 .into(),
11990 )
11991 }
11992 LogicalType::BigInt
11993 | LogicalType::Timestamp
11994 | LogicalType::Time
11995 | LogicalType::TimeTz
11996 | LogicalType::TimestampTz
11997 | LogicalType::TimestampS
11998 | LogicalType::TimestampMs
11999 | LogicalType::TimestampNs => {
12000 let values =
12001 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12002 Data::Int64(
12003 values
12004 .chunks_exact(8)
12005 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12006 .collect::<Vec<_>>()
12007 .into(),
12008 )
12009 }
12010 LogicalType::HugeInt | LogicalType::Uuid => {
12011 let values =
12012 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12013 Data::Int128(
12014 values
12015 .chunks_exact(16)
12016 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12017 .collect::<Vec<_>>()
12018 .into(),
12019 )
12020 }
12021 LogicalType::UHugeInt => {
12022 let values =
12023 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12024 Data::UInt128(
12025 values
12026 .chunks_exact(16)
12027 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12028 .collect::<Vec<_>>()
12029 .into(),
12030 )
12031 }
12032 LogicalType::Float => {
12033 let values =
12034 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12035 Data::Float32(
12036 values
12037 .chunks_exact(4)
12038 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
12039 .collect::<Vec<_>>()
12040 .into(),
12041 )
12042 }
12043 LogicalType::Double => {
12044 let values =
12045 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12046 Data::Float64(
12047 values
12048 .chunks_exact(8)
12049 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
12050 .collect::<Vec<_>>()
12051 .into(),
12052 )
12053 }
12054 LogicalType::Interval => {
12055 let values =
12056 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12057 Data::Interval(
12058 values
12059 .chunks_exact(16)
12060 .map(|item| {
12061 (
12062 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
12063 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
12064 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
12065 )
12066 })
12067 .collect::<Vec<_>>()
12068 .into(),
12069 )
12070 }
12071 LogicalType::Boolean => {
12072 let values = cur.take(rows)?;
12073 if values.iter().any(|value| *value > 1) {
12074 return Err(invalid("boolean page has another value"));
12075 }
12076 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
12077 }
12078 LogicalType::Decimal { .. } => match ty.physical() {
12081 PhysicalType::Int16 => {
12082 let values =
12083 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
12084 Data::Int16(
12085 values
12086 .chunks_exact(2)
12087 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
12088 .collect::<Vec<_>>()
12089 .into(),
12090 )
12091 }
12092 PhysicalType::Int32 => {
12093 let values =
12094 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
12095 Data::Int32(
12096 values
12097 .chunks_exact(4)
12098 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
12099 .collect::<Vec<_>>()
12100 .into(),
12101 )
12102 }
12103 PhysicalType::Int64 => {
12104 let values =
12105 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
12106 Data::Int64(
12107 values
12108 .chunks_exact(8)
12109 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
12110 .collect::<Vec<_>>()
12111 .into(),
12112 )
12113 }
12114 _ => {
12115 let values =
12116 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
12117 Data::Int128(
12118 values
12119 .chunks_exact(16)
12120 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
12121 .collect::<Vec<_>>()
12122 .into(),
12123 )
12124 }
12125 },
12126 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
12127 let offset_bytes = cur
12128 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
12129 let offsets = offset_bytes
12130 .chunks_exact(4)
12131 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
12132 .collect::<Vec<_>>();
12133 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
12134 if offsets.first() != Some(&0)
12135 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
12136 || offsets.windows(2).any(|pair| pair[0] > pair[1])
12137 {
12138 return Err(invalid("string offsets do not bound the payload"));
12139 }
12140 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
12148 let text = ty == &LogicalType::Varchar;
12149 for pair in offsets.windows(2) {
12150 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
12151 if text {
12152 values.push_in_place(at, len)?;
12153 } else {
12154 values.push_bytes_in_place(at, len)?;
12155 }
12156 }
12157 Data::Varlen(values)
12158 }
12159 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
12160 };
12161 if cur.at != bytes.len() {
12162 return Err(invalid("page has trailing bytes"));
12163 }
12164 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
12165}
12166
12167#[cfg(test)]
12168mod tests {
12169 use std::fs::{self, OpenOptions};
12170 use std::io::{Seek, SeekFrom, Write};
12171 use std::path::PathBuf;
12172 use std::time::{SystemTime, UNIX_EPOCH};
12173
12174 use rudb_common::Stat;
12175 use rudb_common::Value;
12176 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
12177 use rudb_common::stat::Provenance;
12178
12179 use super::*;
12180
12181 #[test]
12182 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
12183 let bytes: Vec<u8> =
12184 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
12185 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
12186 let whole = content_name(&bytes[..length]);
12187 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
12188 let mut namer = ContentNamer::default();
12189 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
12190 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
12191 }
12192 }
12193 }
12194
12195 #[derive(Debug)]
12198 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
12199
12200 impl chooser::Chooser for TestsEverything<'_> {
12201 fn name(&self) -> &'static str {
12202 "tests everything"
12203 }
12204
12205 fn narrow_strings(
12206 &self,
12207 values: &[&[u8]],
12208 offered: &[string::Kind],
12209 depth: u8,
12210 ) -> Vec<string::Kind> {
12211 self.0.narrow_strings(values, offered, depth)
12212 }
12213
12214 fn narrow_integers(
12215 &self,
12216 values: &[i64],
12217 offered: &[integer::Kind],
12218 depth: u8,
12219 ) -> Vec<integer::Kind> {
12220 self.0.narrow_integers(values, offered, depth)
12221 }
12222 }
12223
12224 #[test]
12225 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
12226 let columns: Vec<Vec<i64>> = vec![
12227 vec![],
12228 vec![5; 1000],
12229 (0..1000).collect(),
12230 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
12231 (0..1000).map(|row| row / 50).collect(),
12232 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
12233 (0..1000).map(|row| (row * 7919) % 13).collect(),
12234 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
12235 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
12236 (0..1000).map(|row| i64::MIN + row % 3).collect(),
12237 ];
12238 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
12239 for column in &columns {
12240 for chooser in choosers {
12241 let quick = integer::encode_with(column, chooser).unwrap();
12242 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
12243 assert_eq!(
12244 quick,
12245 full,
12246 "{} on {:?}",
12247 chooser.name(),
12248 &column[..column.len().min(8)]
12249 );
12250 }
12251 }
12252 }
12253
12254 #[test]
12257 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
12258 let mut settling = Settling::default();
12259 for part in 0..STRIPE_PARTS as i64 {
12260 let values: Vec<i64> = (0..2048)
12261 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
12262 .collect();
12263 let searched = integer::encode_with(&values, &Fixed).unwrap();
12264 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
12265 }
12266 }
12267
12268 #[test]
12272 fn a_column_that_changes_under_the_shape_is_searched_again() {
12273 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12274 let mut noise = move || {
12275 state ^= state << 13;
12276 state ^= state >> 7;
12277 state ^= state << 17;
12278 (state % 1_000_000) as i64
12279 };
12280 let mut settling = Settling::default();
12281 for part in 0..STRIPE_PARTS as i64 {
12282 let values: Vec<i64> = match part / 16 {
12283 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
12284 1 => (0..2048).map(|_| noise()).collect(),
12285 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
12286 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
12287 };
12288 let settled = settling.encode(&values).unwrap();
12289 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
12290 let searched = integer::encode_with(&values, &Fixed).unwrap();
12291 assert!(
12292 settled.len() * 4 <= searched.len() * 5,
12293 "part {part}: {} settled against {} searched, {} against {}",
12294 settled.len(),
12295 searched.len(),
12296 integer::describe(&settled).unwrap(),
12297 integer::describe(&searched).unwrap(),
12298 );
12299 }
12300 }
12301
12302 #[test]
12303 fn checksum_matches_fixed_vectors() {
12304 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
12305 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
12306 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
12307 }
12308
12309 #[test]
12310 fn sorting_across_threads_matches_sorting_on_one() {
12311 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
12312 let mut next = move || {
12313 state ^= state << 13;
12314 state ^= state >> 7;
12315 state ^= state << 17;
12316 state
12317 };
12318 let mut values = Vec::new();
12319 for at in 0..150_000_u64 {
12320 let value = match next() % 6 {
12321 0 => Vec::new(),
12322 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
12323 2 => format!("https://example.com/path/{at}").into_bytes(),
12324 3 => b"same".to_vec(),
12325 4 => vec![0xff; (next() % 12) as usize],
12326 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
12327 };
12328 values.push(value);
12329 }
12330 let value = |code: u32| values[code as usize].as_slice();
12331 for workers in [1, 2, 3, 8, 32] {
12332 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
12333 let mut across = one.clone();
12334 sort_by_value(&mut one, value);
12335 sort_by_value_across(&mut across, value, workers);
12336 assert_eq!(one, across, "{workers} workers");
12337 }
12338 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
12339 sort_by_value_across(&mut sorted, value, 8);
12340 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
12341 }
12342
12343 fn path(label: &str) -> PathBuf {
12344 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
12345 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
12346 }
12347
12348 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
12353 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
12354 (0..dictionary.values())
12355 .map(|code| {
12356 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
12357 flat[from..to].to_vec()
12358 })
12359 .collect()
12360 }
12361
12362 fn attached(table: &Table) -> Vec<&Section> {
12369 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
12370 }
12371
12372 #[test]
12374 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
12375 const SPANS: usize = 64;
12376 const SPAN: usize = 512;
12377 let path = path("positional");
12378 let content: Vec<u8> =
12379 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
12380 fs::write(&path, &content).expect("the file is written");
12381 let file = Arc::new(File::open(&path).expect("the file opens"));
12382 std::thread::scope(|scope| {
12383 for _ in 0..8 {
12384 let file = Arc::clone(&file);
12385 scope.spawn(move || {
12386 for _ in 0..64 {
12387 for span in 0..SPANS {
12388 let mut bytes = [0_u8; SPAN];
12389 read_at(&file, (span * SPAN) as u64, &mut bytes)
12390 .expect("the span reads");
12391 assert!(
12392 bytes.iter().all(|byte| *byte == span as u8),
12393 "span {span} came back as {}",
12394 bytes[0],
12395 );
12396 }
12397 }
12398 });
12399 }
12400 });
12401 let mut past = [0_u8; SPAN];
12402 let end = (SPANS * SPAN) as u64;
12403 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
12404 assert!(error.message().contains("ends before its declared length"), "{error}");
12405 drop(file);
12406 let _ = fs::remove_file(&path);
12407 }
12408
12409 #[test]
12416 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
12417 let path = path("cursor");
12418 let mut writer = Writer::create(
12419 &path,
12420 "items",
12421 vec![
12422 Field::required("id", LogicalType::Integer),
12423 Field::new("text", LogicalType::Varchar),
12424 ],
12425 )
12426 .expect("new file");
12427 writer.append(&sample()).expect("first part");
12428 writer.append(&sample()).expect("second part");
12429 writer.finish().expect("commit");
12430 let reader = Reader::open(&path).expect("reopen from disk");
12431 assert_eq!(reader.table().rows(), 6);
12432 let ids = reader.read(0, &[0]).expect("the integer page reads back");
12433 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
12434 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
12435 let text = reader.read(1, &[1]).expect("the text page reads back");
12436 assert_eq!(text.value_at(1, 0), Value::Null);
12437 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12438 let end = reader.table().stripes().iter().flat_map(|stripe| {
12441 stripe
12442 .pages
12443 .iter()
12444 .map(|page| page.offset + u64::from(page.length))
12445 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
12446 });
12447 let last = end.fold(HEADER, u64::max);
12448 let directory = fs::metadata(&path).expect("the file is there").len();
12449 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
12450 fs::remove_file(path).expect("remove scratch file");
12451 }
12452
12453 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
12459 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
12460 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
12461 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
12462 let bits = (width & !DICTIONARY_FLAGS) as usize;
12463 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
12464 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
12465 DICTIONARY_HEADER as u64
12466 + offset_bytes(count as usize, bits) as u64
12467 + blocks * payload_words * 8
12468 + rank_blocks * 16
12469 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
12470 }
12471
12472 fn sample() -> Chunk {
12473 Chunk::new(vec![
12474 Vector::from_values(
12475 LogicalType::Integer,
12476 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
12477 )
12478 .expect("integers"),
12479 Vector::from_values(
12480 LogicalType::Varchar,
12481 &[
12482 Value::Varchar("alpha".into()),
12483 Value::Null,
12484 Value::Varchar("long text after a slash".into()),
12485 ],
12486 )
12487 .expect("strings"),
12488 ])
12489 .expect("matching rows")
12490 }
12491
12492 fn sample_ids() -> Chunk {
12493 Chunk::new(vec![
12494 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
12495 .expect("integers"),
12496 ])
12497 .expect("one column")
12498 }
12499
12500 #[test]
12501 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
12502 let path = path("nulls_for_the_planner");
12505 let mut writer =
12506 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
12507 .expect("new file");
12508 let rows = Chunk::new(vec![
12509 Vector::from_values(
12510 LogicalType::Integer,
12511 &[
12512 Value::Integer(4),
12513 Value::Null,
12514 Value::Integer(9),
12515 Value::Null,
12516 Value::Integer(1),
12517 Value::Integer(2),
12518 ],
12519 )
12520 .expect("integers"),
12521 ])
12522 .expect("one column");
12523 writer.append(&rows).expect("the only part");
12524 writer.finish().expect("commit");
12525 let reader = Reader::open(&path).expect("reopen from disk");
12526 let stripes = Stripes::new(reader);
12527 let column = stripes.column("a").expect("the file has that column");
12528 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
12529 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
12532 fs::remove_file(&path).expect("clean up");
12533 }
12534
12535 #[test]
12536 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
12537 let path = path("frequencies_for_the_planner");
12542 let mut writer =
12543 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12544 .expect("new file");
12545 let rows = Chunk::new(vec![
12546 Vector::from_values(
12547 LogicalType::Integer,
12548 &[
12549 Value::Integer(4),
12550 Value::Integer(4),
12551 Value::Integer(4),
12552 Value::Integer(9),
12553 Value::Integer(9),
12554 Value::Integer(1),
12555 ],
12556 )
12557 .expect("integers"),
12558 ])
12559 .expect("one column");
12560 writer.append(&rows).expect("the only part");
12561 writer.finish().expect("commit");
12562 let reader = Reader::open(&path).expect("reopen from disk");
12563 let common = Common::new(reader);
12564 assert_eq!(common.rows(), 6);
12565 let column = common.column("id").expect("the file has that column");
12566 assert_eq!(common.column("nothing"), None);
12567 assert_eq!(
12568 common.rows_with(column, &Bound::Int(4)),
12569 Stat::exact(3, Provenance::FrequencySynopsis)
12570 );
12571 assert_eq!(
12573 common.rows_with(column, &Bound::Int(7)),
12574 Stat::exact(0, Provenance::FrequencySynopsis)
12575 );
12576 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
12579 assert_eq!(common.remainder(column), None);
12582 fs::remove_file(&path).expect("clean up");
12583 }
12584
12585 #[test]
12586 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
12587 let path = path("string_frequencies_for_the_planner");
12588 let mut writer =
12589 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
12590 .expect("new file");
12591 let rows = Chunk::new(vec![
12592 Vector::from_values(
12593 LogicalType::Varchar,
12594 &[
12595 Value::Varchar(String::new()),
12596 Value::Varchar("alpha".into()),
12597 Value::Varchar(String::new()),
12598 Value::Varchar("beta".into()),
12599 Value::Varchar(String::new()),
12600 ],
12601 )
12602 .expect("strings"),
12603 ])
12604 .expect("one column");
12605 writer.append(&rows).expect("the only part");
12606 writer.finish().expect("commit");
12607
12608 let reader = Reader::open(&path).expect("reopen from disk");
12609 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
12610 let common = Common::new(reader.clone());
12611 let column = common.column("text").expect("the file has that column");
12612 assert_eq!(
12613 common.rows_with(column, &Bound::Bytes(Vec::new())),
12614 Stat::exact(3, Provenance::FrequencySynopsis)
12615 );
12616 assert_eq!(
12617 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
12618 Stat::exact(0, Provenance::FrequencySynopsis)
12619 );
12620 assert_eq!(
12621 reader.reads().dictionaries,
12622 0,
12623 "the bounded spellings answer without opening the dictionary index"
12624 );
12625 fs::remove_file(&path).expect("clean up");
12626 }
12627
12628 #[test]
12629 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
12630 let path = path("certified_host_groups");
12631 let mut writer =
12632 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
12633 .expect("new file");
12634 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
12635 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
12636 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
12637 values.push(Value::Varchar(String::new()));
12638 for part in values.chunks(512) {
12639 writer
12640 .append(
12641 &Chunk::new(vec![
12642 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
12643 ])
12644 .expect("one column"),
12645 )
12646 .expect("part written");
12647 }
12648 writer.finish().expect("commit");
12649 let reader = Reader::open(&path).expect("reopen");
12650 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
12651 fs::remove_file(&path).expect("clean up");
12652 }
12653
12654 fn bare_table(sections: Vec<Section>) -> Table {
12659 Table {
12660 name: "linked".to_owned(),
12661 fields: vec![Field::required("id", LogicalType::Integer)],
12662 stripes: Vec::new(),
12663 rows: 0,
12664 dictionaries: vec![None],
12665 dictionary_payloads: Vec::new(),
12666 demoted: Vec::new(),
12667 distincts: vec![None],
12668 frequencies: vec![None],
12669 pair_frequencies: Vec::new(),
12670 frequency_texts: Vec::new(),
12671 host_groups: None,
12672 clustering: None,
12673 generation: 1,
12674 sections,
12675 }
12676 }
12677
12678 fn a_key_map_section() -> Section {
12679 Section {
12680 kind: *section::KEY_MAP,
12681 id: 1,
12682 generation: 3,
12683 extents: 1,
12684 extent_page: HEADER,
12685 extent_bytes: section::EXTENT_BYTES as u32,
12686 hash: 0x1234_5678_9abc_def0,
12687 flags: 0,
12688 header_bytes: 24,
12689 }
12690 }
12691
12692 #[test]
12693 fn a_section_table_round_trips_through_a_directory() {
12694 let mut later = a_key_map_section();
12695 later.kind = *b"RUDBZZ9\0";
12696 later.id = 2;
12697 let table = bare_table(vec![a_key_map_section(), later]);
12698 let directory = encode_directory(&table).expect("directory");
12699 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12700 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
12701 assert!(decoded.sections()[0].known());
12705 assert!(!decoded.sections()[1].known());
12706 }
12707
12708 #[test]
12709 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
12710 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12714 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
12715 let older = &directory[..directory.len() - block];
12716 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
12717 assert!(decoded.sections().is_empty());
12718 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
12719 assert_eq!(decoded.name(), "linked");
12720 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
12721 }
12722
12723 #[test]
12724 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
12725 let path = path("format_twenty_two");
12732 let mut writer =
12733 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12734 .expect("new file");
12735 let rows = Chunk::new(vec![
12736 Vector::from_values(
12737 LogicalType::Integer,
12738 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
12739 )
12740 .expect("integers"),
12741 ])
12742 .expect("one column");
12743 writer.append(&rows).expect("the only part");
12744 writer.finish().expect("commit");
12745
12746 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12747 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12748 drop(file);
12749
12750 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
12751 assert_eq!(reader.table().rows(), 3);
12752 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
12757
12758 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12761 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
12762 drop(file);
12763 let error = Reader::open(&path).expect_err("format 21 is not readable");
12764 assert!(error.to_string().contains("format 21"), "{error}");
12765
12766 fs::remove_file(&path).expect("clean up");
12767 }
12768
12769 #[test]
12770 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
12771 let mut past = a_key_map_section();
12776 past.extent_page = 1 << 30;
12777 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
12778 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
12779 assert!(error.to_string().contains("outside the file"), "{error}");
12780
12781 let mut inside_the_header = a_key_map_section();
12782 inside_the_header.extent_page = 8;
12783 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
12784 assert!(
12785 decode_directory(&directory, 1 << 20).is_err(),
12786 "a section may not overlap a header"
12787 );
12788 }
12789
12790 #[test]
12791 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
12792 let not_built = Section {
12796 kind: *section::FORWARD_LINK,
12797 id: 9,
12798 generation: 3,
12799 extents: 0,
12800 extent_page: 0,
12801 extent_bytes: 0,
12802 hash: 0,
12803 flags: 0,
12804 header_bytes: 0,
12805 };
12806 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
12807 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
12808 assert_eq!(decoded.sections(), &[not_built]);
12809
12810 let mut incoherent = not_built;
12813 incoherent.extent_bytes = 28;
12814 incoherent.extent_page = HEADER;
12815 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
12816 assert!(decode_directory(&directory, 1 << 20).is_err());
12817 }
12818
12819 #[test]
12820 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
12821 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
12822 let mut torn = directory.clone();
12823 let count_at = torn.len() - size_of::<u16>();
12824 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
12825 assert!(decode_directory(&torn, 1 << 20).is_err());
12828 }
12829
12830 fn linked_file(label: &str, rows: i32) -> PathBuf {
12832 let path = path(label);
12833 let mut writer =
12834 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12835 .expect("new file");
12836 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
12837 let chunk =
12838 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
12839 .expect("one column");
12840 writer.append(&chunk).expect("the only part");
12841 writer.finish().expect("commit");
12842 path
12843 }
12844
12845 fn a_key_map_payload() -> Vec<u8> {
12846 (0..512_u32).flat_map(u32::to_le_bytes).collect()
12849 }
12850
12851 #[test]
12852 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
12853 let path = linked_file("attach", 64);
12854 let payload = a_key_map_payload();
12855 let table = attach(
12856 &path,
12857 "items",
12858 &[section::Attachment {
12859 kind: *section::KEY_MAP,
12860 id: 0,
12861 flags: 2,
12862 header_bytes: 40,
12863 bytes: &payload,
12864 }],
12865 )
12866 .expect("attach a key map");
12867 assert_eq!(attached(&table).len(), 1);
12868
12869 let reader = Reader::open(&path).expect("reopen after the attach");
12870 let held = attached(reader.table());
12871 assert_eq!(held.len(), 1);
12872 assert_eq!(held[0].kind, *section::KEY_MAP);
12873 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
12874 assert_eq!(held[0].header_bytes, 40);
12875 assert_eq!(held[0].generation, 1);
12879 assert!(held[0].usable(reader.table().generation()));
12880 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
12881 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
12882
12883 fs::remove_file(&path).expect("clean up");
12884 }
12885
12886 #[test]
12887 fn attaching_a_section_answers_every_row_exactly_as_before() {
12888 let path = linked_file("attach_changes_nothing", 300);
12893 let before = Reader::open(&path).expect("open before");
12894 let rows = before.table().rows();
12895 let first = before.read(0, &[0]).expect("read before");
12896 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
12897 let layout = before.layout().columns_total();
12898 drop(before);
12899
12900 let payload = a_key_map_payload();
12901 attach(
12902 &path,
12903 "items",
12904 &[section::Attachment {
12905 kind: *section::KEY_MAP,
12906 id: 0,
12907 flags: 0,
12908 header_bytes: 0,
12909 bytes: &payload,
12910 }],
12911 )
12912 .expect("attach");
12913
12914 let after = Reader::open(&path).expect("open after");
12915 assert_eq!(after.table().rows(), rows);
12916 let read = after.read(0, &[0]).expect("read after");
12917 for (at, value) in values.iter().enumerate() {
12918 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
12919 }
12920 assert_eq!(
12921 after.layout().columns_total(),
12922 layout,
12923 "an attach appends and does not rewrite a column page"
12924 );
12925
12926 fs::remove_file(&path).expect("clean up");
12927 }
12928
12929 #[test]
12930 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
12931 let path = linked_file("attach_twice", 32);
12935 let one = a_key_map_payload();
12936 let two = vec![7_u8; 1024];
12937 let entry = |bytes| section::Attachment {
12938 kind: *section::KEY_MAP,
12939 id: 4,
12940 flags: 1,
12941 header_bytes: 0,
12942 bytes,
12943 };
12944 attach(&path, "items", &[entry(&one)]).expect("first build");
12945 attach(&path, "items", &[entry(&two)]).expect("rebuild");
12946
12947 let reader = Reader::open(&path).expect("reopen");
12948 let held = attached(reader.table());
12949 assert_eq!(held.len(), 1, "one map per column and not one per build");
12950 assert_eq!(reader.payload(held[0]).expect("payload"), two);
12951
12952 fs::remove_file(&path).expect("clean up");
12953 }
12954
12955 #[test]
12956 fn an_attach_carries_through_a_kind_it_does_not_know() {
12957 let path = linked_file("attach_unknown", 16);
12961 let payload = vec![3_u8; 96];
12962 attach(
12963 &path,
12964 "items",
12965 &[section::Attachment {
12966 kind: *b"RUDBZZ9\0",
12967 id: 1,
12968 flags: 0,
12969 header_bytes: 0,
12970 bytes: &payload,
12971 }],
12972 )
12973 .expect("a kind this build does not know still writes");
12974 let key_map = a_key_map_payload();
12975 attach(
12976 &path,
12977 "items",
12978 &[section::Attachment {
12979 kind: *section::KEY_MAP,
12980 id: 0,
12981 flags: 0,
12982 header_bytes: 0,
12983 bytes: &key_map,
12984 }],
12985 )
12986 .expect("attach beside it");
12987
12988 let reader = Reader::open(&path).expect("reopen");
12989 let held = attached(reader.table());
12990 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
12991 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
12992 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
12993
12994 fs::remove_file(&path).expect("clean up");
12995 }
12996
12997 #[test]
12998 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
12999 let path = linked_file("attach_not_built", 8);
13000 attach(
13001 &path,
13002 "items",
13003 &[section::Attachment {
13004 kind: *section::FORWARD_LINK,
13005 id: 2,
13006 flags: 0,
13007 header_bytes: 0,
13008 bytes: &[],
13009 }],
13010 )
13011 .expect("record a link that did not fit the budget");
13012
13013 let reader = Reader::open(&path).expect("reopen");
13014 let held = attached(reader.table());
13015 assert_eq!(held.len(), 1);
13016 assert_eq!(held[0].extents, 0);
13017 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
13018 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
13019 assert!(reader.payload(held[0]).expect("no payload").is_empty());
13020
13021 fs::remove_file(&path).expect("clean up");
13022 }
13023
13024 #[test]
13025 fn a_payload_past_one_extent_is_split_and_joined_back() {
13026 let path = linked_file("attach_two_extents", 8);
13030 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
13031 attach(
13032 &path,
13033 "items",
13034 &[section::Attachment {
13035 kind: *section::KEY_MAP,
13036 id: 0,
13037 flags: 0,
13038 header_bytes: 0,
13039 bytes: &payload,
13040 }],
13041 )
13042 .expect("attach a payload past the bound");
13043
13044 let reader = Reader::open(&path).expect("reopen");
13045 let held = attached(reader.table());
13046 let extents = reader.extents(held[0]).expect("extent table");
13047 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
13048 assert_eq!(extents[0].length, section::MAX_EXTENT);
13049 assert_eq!(extents[1].length, 1);
13050 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
13051 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
13053 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
13054
13055 fs::remove_file(&path).expect("clean up");
13056 }
13057
13058 #[test]
13059 fn a_torn_extent_is_refused_rather_than_decoded() {
13060 let path = linked_file("attach_torn", 8);
13061 let payload = a_key_map_payload();
13062 attach(
13063 &path,
13064 "items",
13065 &[section::Attachment {
13066 kind: *section::KEY_MAP,
13067 id: 0,
13068 flags: 0,
13069 header_bytes: 0,
13070 bytes: &payload,
13071 }],
13072 )
13073 .expect("attach");
13074
13075 let reader = Reader::open(&path).expect("reopen");
13076 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
13077 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
13078 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
13079 drop(file);
13080
13081 let reader = Reader::open(&path).expect("the table still opens");
13082 let error = reader
13083 .payload(&reader.table().sections()[0])
13084 .expect_err("a corrupt payload is not handed out");
13085 assert!(error.to_string().contains("checksum"), "{error}");
13086 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
13089
13090 fs::remove_file(&path).expect("clean up");
13091 }
13092
13093 #[test]
13094 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
13095 let path = linked_file("attach_old_format", 8);
13098 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
13099 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
13100 drop(file);
13101
13102 let payload = a_key_map_payload();
13103 let error = attach(
13104 &path,
13105 "items",
13106 &[section::Attachment {
13107 kind: *section::KEY_MAP,
13108 id: 0,
13109 flags: 0,
13110 header_bytes: 0,
13111 bytes: &payload,
13112 }],
13113 )
13114 .expect_err("format 22 cannot gain a section");
13115 assert!(error.to_string().contains("format 22"), "{error}");
13116 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
13117
13118 fs::remove_file(&path).expect("clean up");
13119 }
13120
13121 #[test]
13122 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
13123 let path = linked_file("attach_bad_header", 8);
13124 let error = attach(
13125 &path,
13126 "items",
13127 &[section::Attachment {
13128 kind: *section::KEY_MAP,
13129 id: 0,
13130 flags: 0,
13131 header_bytes: 40,
13132 bytes: &[1, 2, 3],
13133 }],
13134 )
13135 .expect_err("a writer's bug stops at the write");
13136 assert!(error.to_string().contains("header is longer"), "{error}");
13137
13138 fs::remove_file(&path).expect("clean up");
13139 }
13140
13141 #[test]
13142 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
13143 let path = linked_file("attach_wrong_name", 8);
13144 let error = attach(&path, "orders", &[]).expect_err("no such table");
13145 assert!(error.to_string().contains("orders"), "{error}");
13146 fs::remove_file(&path).expect("clean up");
13147 }
13148
13149 #[test]
13150 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
13151 let path = path("frequency_prefix_for_the_planner");
13158 let mut writer =
13159 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13160 .expect("new file");
13161 let mut values = vec![Value::Integer(1); 10_000];
13162 for _ in 0..10 {
13163 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
13164 }
13165 for part in values.chunks(8_000) {
13168 let rows = Chunk::new(vec![
13169 Vector::from_values(LogicalType::Integer, part).expect("integers"),
13170 ])
13171 .expect("one column");
13172 writer.append(&rows).expect("a part");
13173 }
13174 writer.finish().expect("commit");
13175 let reader = Reader::open(&path).expect("reopen from disk");
13176 let prefix =
13177 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
13178 assert_eq!(prefix.entries.len(), 512);
13181 assert_eq!(prefix.omitted_max, 10);
13182 let common = Common::new(reader);
13183 assert_eq!(common.rows(), 16_000);
13184 let column = common.column("id").expect("the file has that column");
13185 assert_eq!(
13186 common.rows_with(column, &Bound::Int(1)),
13187 Stat::exact(10_000, Provenance::FrequencySynopsis)
13188 );
13189 assert_eq!(
13191 common.rows_with(column, &Bound::Int(1_100)),
13192 Stat::exact(10, Provenance::FrequencySynopsis)
13193 );
13194 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
13197 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
13200 let remainder = common.remainder(column).expect("the list is a prefix");
13204 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
13205 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
13206 fs::remove_file(&path).expect("clean up");
13207 }
13208
13209 #[test]
13211 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
13212 let path = path("empty");
13213 Writer::empty(&path, &[]).expect("a file with nothing in it");
13214 let catalog = Catalog::open(&path).expect("the empty file opens");
13215 assert_eq!(catalog.len(), 0);
13216 assert!(catalog.is_empty());
13217 assert_eq!(catalog.names().count(), 0);
13218 let mut writer =
13221 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13222 .expect("a table goes into the empty file");
13223 writer.append(&sample_ids()).expect("rows");
13224 writer.finish().expect("commit");
13225 let catalog = Catalog::open(&path).expect("the file opens again");
13226 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13227 fs::remove_file(&path).expect("clean up");
13228 }
13229
13230 #[test]
13240 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
13241 let path = path("empty-name");
13242 let field = || vec![Field::required("id", LogicalType::Integer)];
13243 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
13244 let catalog = Catalog::open(&path).expect("the file opens");
13245 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
13246
13247 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
13248 writer.append(&sample_ids()).expect("rows");
13249 writer.finish().expect("commit");
13250 let catalog = Catalog::open(&path).expect("the file opens again");
13251 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13253 let held = catalog.rows().collect::<Vec<_>>();
13254 assert_eq!(held.len(), 1);
13255 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
13256
13257 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
13259 assert!(error.to_string().contains("same name"), "{error}");
13260 fs::remove_file(&path).expect("clean up");
13261 }
13262
13263 fn sample_view(name: &str) -> ViewEntry {
13265 ViewEntry {
13266 name: name.to_string(),
13267 sql: "SELECT id FROM items WHERE id > 0".to_string(),
13268 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
13269 aliases: vec!["n".to_string()],
13270 columns: vec![Field::new("n", LogicalType::Integer)],
13271 }
13272 }
13273
13274 #[test]
13275 fn a_view_written_into_the_catalog_comes_back_whole() {
13276 let path = path("views");
13277 let mut writer =
13278 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13279 .expect("new file");
13280 writer.append(&sample_ids()).expect("rows");
13281 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13282 let catalog = Catalog::open(&path).expect("reopen");
13283 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
13284 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13287 fs::remove_file(&path).expect("clean up");
13288 }
13289
13290 #[test]
13292 fn appending_a_table_carries_the_views_forward() {
13293 let path = path("viewscarry");
13294 let mut writer =
13295 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13296 .expect("new file");
13297 writer.append(&sample_ids()).expect("rows");
13298 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
13299 let mut writer =
13300 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
13301 .expect("a second table");
13302 writer.append(&sample_ids()).expect("rows");
13303 writer.finish().expect("commit");
13304 let catalog = Catalog::open(&path).expect("reopen");
13305 assert_eq!(catalog.views().count(), 1);
13306 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
13307 fs::remove_file(&path).expect("clean up");
13308 }
13309
13310 #[test]
13312 fn restating_the_views_leaves_every_table_where_it_was() {
13313 let path = path("restate");
13314 let mut writer =
13315 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13316 .expect("new file");
13317 writer.append(&sample_ids()).expect("rows");
13318 writer.finish().expect("commit");
13319 let before = fs::metadata(&path).expect("the file is there").len();
13320 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
13321 let catalog = Catalog::open(&path).expect("reopen");
13322 assert_eq!(catalog.views().count(), 2);
13323 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
13324 let after = fs::metadata(&path).expect("the file is there").len();
13327 assert!(after > before, "a generation was written");
13328 assert!(after - before < before, "the table was not written again");
13329 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
13332 assert_eq!(reader.table().rows, 3);
13333 Writer::restate(&path, &[]).expect("no views at all");
13336 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
13337 fs::remove_file(&path).expect("clean up");
13338 }
13339
13340 #[test]
13342 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
13343 let bytes = encode_catalog(
13344 &[Entry {
13345 name: "items".to_string(),
13346 fields: vec![Field::required("id", LogicalType::Integer)],
13347 rows: 1,
13348 directory: Page { offset: HEADER, length: 8, hash: 0 },
13349 nonzero: vec![None],
13350 aggregates: vec![None],
13351 distincts: vec![None],
13352 extremes: vec![None],
13353 frequencies: vec![None],
13354 }],
13355 &[sample_view("items")],
13356 )
13357 .expect("it encodes, because encoding does not look");
13358 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
13359 assert!(error.to_string().contains("same name"), "{error}");
13360 }
13361
13362 #[test]
13365 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
13366 let rows: usize = 300;
13367 let text: Vec<String> =
13368 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
13369 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
13370 let mut page = vec![6, 2];
13371 page.extend((0..rows.div_ceil(8)).map(|byte| {
13372 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
13373 }));
13374 let compressed = string::encode_only(string::Kind::Fsst, &values)
13375 .expect("encoded")
13376 .expect("text this repetitive compresses");
13377 page.extend_from_slice(&compressed);
13378 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
13379 let positions = [0_u32, 3, 8, 13, 200, 299];
13380 let some =
13381 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
13382 assert_eq!(some.len(), positions.len());
13383 for (at, &row) in positions.iter().enumerate() {
13384 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
13385 }
13386 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
13387 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
13388 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
13389 }
13390
13391 #[test]
13394 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
13395 let path = path("rows");
13396 let mut writer = Writer::create(
13397 &path,
13398 "items",
13399 vec![
13400 Field::required("id", LogicalType::Integer),
13401 Field::new("text", LogicalType::Varchar),
13402 ],
13403 )
13404 .expect("new file");
13405 let rows = 2_000;
13406 let chunk = Chunk::new(vec![
13407 Vector::from_values(
13408 LogicalType::Integer,
13409 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
13410 )
13411 .expect("integers"),
13412 Vector::from_values(
13413 LogicalType::Varchar,
13414 &(0..rows)
13415 .map(|row| {
13416 if row % 7 == 2 {
13417 Value::Null
13418 } else {
13419 Value::Varchar(format!("a comment about order {}", row * 13))
13420 }
13421 })
13422 .collect::<Vec<_>>(),
13423 )
13424 .expect("strings"),
13425 ])
13426 .expect("matching rows");
13427 writer.append(&chunk).expect("one part");
13428 writer.finish().expect("commit");
13429 let reader = Reader::open(&path).expect("reopen from disk");
13430 let positions = [1_u32, 2, 9, 1_000, 1_999];
13431 for whole in [true, false] {
13432 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
13433 let all = reader.read(0, &[0, 1]).expect("the whole part");
13434 assert_eq!(some.len(), positions.len());
13435 for column in 0..2 {
13436 for (at, &row) in positions.iter().enumerate() {
13437 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
13438 }
13439 }
13440 }
13441 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
13442 }
13443
13444 #[test]
13445 fn committed_file_reopens_and_reads_only_requested_columns() {
13446 let path = path("reopen");
13447 let mut writer = Writer::create(
13448 &path,
13449 "items",
13450 vec![
13451 Field::required("id", LogicalType::Integer),
13452 Field::new("text", LogicalType::Varchar),
13453 ],
13454 )
13455 .expect("new file");
13456 writer.append(&sample()).expect("first part");
13457 writer.append(&sample()).expect("second part");
13458 writer.finish().expect("commit");
13459 let reader = Reader::open(&path).expect("reopen from disk");
13460 assert_eq!(reader.table().rows(), 6);
13461 assert_eq!(reader.table().stripes().len(), 1);
13464 assert_eq!(reader.parts(), 2);
13465 assert_eq!(reader.part_rows(0), 3);
13466 assert_eq!(reader.part_rows(1), 3);
13467 let text = reader.read(1, &[1]).expect("only text page");
13468 assert_eq!(text.width(), 1);
13469 assert_eq!(text.value_at(1, 0), Value::Null);
13470 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13471 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
13472 assert_eq!(sparse.width(), 1);
13473 assert_eq!(sparse.value_at(1, 0), Value::Null);
13474 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
13475 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
13476 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
13477 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
13478 let count = reader.read(0, &[]).expect("no page is needed for count");
13479 assert_eq!(count.len(), 3);
13480 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
13481 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
13482 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
13483 assert_eq!(
13484 integers,
13485 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
13486 );
13487 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
13488 assert_eq!(strings.len(), 3);
13489 assert!(strings.contains(&(Value::Null, 2)));
13490 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
13491 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
13492 fs::remove_file(path).expect("remove scratch file");
13493 }
13494
13495 #[test]
13503 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
13504 let path = path("interleaved-runs");
13505 let mut writer =
13506 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
13507 .expect("new file");
13508 for morsel in [2_u64, 0, 3, 1] {
13509 let parts = (0..4_u64)
13510 .map(|chunk| {
13511 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
13512 let values =
13513 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
13514 let column =
13515 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
13516 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
13517 })
13518 .collect::<Vec<_>>();
13519 writer.append_stripe(parts).expect("a stripe");
13520 }
13521 writer.finish().expect("commit");
13522
13523 let reader = Reader::open(&path).expect("valid directory");
13524 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
13525 assert_eq!(reader.table().rows(), 128);
13526 for part in 0..16_usize {
13527 let read = reader.read(part, &[0]).expect("a part back");
13528 for row in 0..8_usize {
13529 let want = i64::try_from(part * 8 + row).expect("small");
13530 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
13531 }
13532 }
13533 fs::remove_file(path).expect("remove scratch file");
13534 }
13535
13536 #[test]
13539 fn runs_that_overlap_each_other_are_refused_at_commit() {
13540 let path = path("overlapping-runs");
13541 let mut writer =
13542 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
13543 .expect("new file");
13544 let one = |order: (u64, u64)| {
13545 let column =
13546 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
13547 (order, Chunk::new(vec![column]).expect("one column"))
13548 };
13549 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
13552 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
13553 let error = writer.finish().expect_err("the runs overlap");
13554 assert!(error.message().contains("source order"), "{error}");
13555 fs::remove_file(path).expect("remove scratch file");
13556 }
13557
13558 #[test]
13561 fn a_run_longer_than_a_stripe_is_refused() {
13562 let path = path("overlong-run");
13563 let mut writer =
13564 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
13565 .expect("new file");
13566 let parts = (0..=STRIPE_PARTS)
13567 .map(|at| {
13568 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
13569 .expect("a column");
13570 let chunk = Chunk::new(vec![column]).expect("one column");
13571 ((0, u64::try_from(at).expect("small")), chunk)
13572 })
13573 .collect::<Vec<_>>();
13574 let error = writer.append_stripe(parts).expect_err("one part too many");
13575 assert!(error.message().contains("more parts than it holds"), "{error}");
13576 fs::remove_file(path).expect("remove scratch file");
13577 }
13578
13579 #[test]
13585 fn parts_past_the_stripe_bound_start_a_new_stripe() {
13586 let path = path("stripe-bound");
13587 let mut writer = Writer::create(
13588 &path,
13589 "items",
13590 vec![
13591 Field::required("id", LogicalType::Integer),
13592 Field::new("text", LogicalType::Varchar),
13593 ],
13594 )
13595 .expect("new file");
13596 let parts = STRIPE_PARTS * 2 + 3;
13597 for part in 0..parts {
13598 let id = part as i32;
13599 let chunk = Chunk::new(vec![
13600 Vector::from_values(
13601 LogicalType::Integer,
13602 &[Value::Integer(id), Value::Integer(-id)],
13603 )
13604 .expect("integers"),
13605 Vector::from_values(
13606 LogicalType::Varchar,
13607 &[Value::Varchar(format!("value {part}")), Value::Null],
13608 )
13609 .expect("strings"),
13610 ])
13611 .expect("matching rows");
13612 writer.append(&chunk).expect("one part");
13613 }
13614 writer.finish().expect("commit");
13615
13616 let reader = Reader::open(&path).expect("reopen from disk");
13617 assert_eq!(reader.parts(), parts);
13618 assert_eq!(reader.table().rows(), parts * 2);
13619 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
13620 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
13621 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
13622 assert_eq!(reader.table().stripes()[2].parts(), 3);
13623 for part in (0..parts).rev() {
13626 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
13627 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
13628 for chunk in [&dense, &sparse] {
13629 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
13630 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13631 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13632 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
13633 assert_eq!(chunk.value_at(1, 1), Value::Null);
13634 }
13635 }
13636 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
13639 assert!(reader.skips(0, &above), "the first stripe stops at 63");
13640 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
13641 fs::remove_file(path).expect("remove scratch file");
13642 }
13643
13644 fn scattered(n: i64) -> i64 {
13646 n.wrapping_mul(-7_046_029_254_386_353_131)
13647 }
13648
13649 #[test]
13655 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
13656 let path = path("sieve-skip");
13657 let mut writer =
13658 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13659 .expect("new file");
13660 let parts = STRIPE_PARTS + 3;
13661 let per_part = 128;
13665 for part in 0..parts {
13666 let held: Vec<Value> = (0..per_part)
13667 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
13668 .collect();
13669 let chunk =
13670 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13671 .expect("one column");
13672 writer.append(&chunk).expect("one part");
13673 }
13674 writer.finish().expect("commit");
13675
13676 let reader = Reader::open(&path).expect("reopen from disk");
13677 let probe = |value: i64| Probe {
13678 column: 0,
13679 op: Op::Equal,
13680 value: Bound::Int(i128::from(scattered(value))),
13681 };
13682 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
13683 let tests = [probe(wanted)];
13684 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
13685 let home = wanted as usize / per_part;
13686 assert!(kept.contains(&home), "the part holding {wanted} is read");
13687 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
13691 }
13692 let absent = [probe((parts * per_part) as i64 + 1)];
13693 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
13694 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
13695 let tests = [probe(0)];
13698 assert!(
13699 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
13700 "the bounds rule out no stripe at all"
13701 );
13702 fs::remove_file(path).expect("remove scratch file");
13703 }
13704
13705 #[test]
13711 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
13712 let path = path("part-range-skip");
13713 let mut writer =
13714 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13715 .expect("new file");
13716 let parts = STRIPE_PARTS + 3;
13717 let per_part = 128;
13718 for part in 0..parts {
13719 let held: Vec<Value> = (0..per_part)
13723 .map(|row| {
13724 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13725 })
13726 .collect();
13727 let chunk =
13728 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13729 .expect("one column");
13730 writer.append(&chunk).expect("one part");
13731 }
13732 writer.finish().expect("commit");
13733
13734 let reader = Reader::open(&path).expect("reopen from disk");
13735 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13736 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
13737 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
13738 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
13740 fs::remove_file(path).expect("remove scratch file");
13741 }
13742
13743 #[test]
13747 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
13748 let path = path("part-range-certain");
13749 let mut writer =
13750 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13751 .expect("new file");
13752 let parts = STRIPE_PARTS + 3;
13753 let per_part = 128;
13754 for part in 0..parts {
13755 let held: Vec<Value> = (0..per_part)
13756 .map(|row| {
13757 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
13758 })
13759 .collect();
13760 let chunk =
13761 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13762 .expect("one column");
13763 writer.append(&chunk).expect("one part");
13764 }
13765 writer.finish().expect("commit");
13766
13767 let reader = Reader::open(&path).expect("reopen from disk");
13768 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
13769 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
13770 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
13771 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
13774 fs::remove_file(path).expect("remove scratch file");
13775 }
13776
13777 #[test]
13780 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
13781 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
13782 let path = path("part-range-page");
13783 let mut writer =
13784 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
13785 .expect("new file");
13786 for part in 0..parts {
13787 let held: Vec<Value> = (0..128)
13788 .map(|row| {
13789 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
13790 })
13791 .collect();
13792 let chunk = Chunk::new(vec![
13793 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
13794 ])
13795 .expect("one column");
13796 writer.append(&chunk).expect("one part");
13797 }
13798 writer.finish().expect("commit");
13799 let reader = Reader::open(&path).expect("reopen from disk");
13800 let bytes = reader.layout().columns[0].part_ranges;
13801 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
13802 fs::remove_file(path).expect("remove scratch file");
13803 }
13804 }
13805
13806 #[test]
13809 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
13810 let long = vec![b'a'; PART_BOUND_BYTES * 2];
13811 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
13812 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
13813 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
13814 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
13815 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
13816 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
13817 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
13818 }
13819
13820 #[test]
13823 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
13824 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
13825 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
13826 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
13827 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
13828 }
13829
13830 #[test]
13842 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
13843 let parts = 4;
13844 let per_part = 1024;
13845 let rows = parts * per_part;
13846 let written = |name: &str, keys: &[i64]| {
13847 let path = path(name);
13848 let fields = vec![Field::required("key", LogicalType::BigInt)];
13849 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
13850 for part in 0..parts {
13851 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
13852 .iter()
13853 .map(|key| Value::BigInt(*key))
13854 .collect();
13855 let chunk = Chunk::new(vec![
13856 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
13857 ])
13858 .expect("one column");
13859 writer.append(&chunk).expect("one part");
13860 }
13861 writer.finish().expect("commit");
13862 path
13863 };
13864 let climbing = |step: &dyn Fn(usize) -> i64| {
13867 let mut key = 0;
13868 (0..rows)
13869 .map(|row| {
13870 key += step(row);
13871 key
13872 })
13873 .collect::<Vec<i64>>()
13874 };
13875 let ascending = climbing(&|row| (row % 3) as i64);
13876 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
13880 let near_path = written("stored-near", &ascending);
13881 let far_path = written("stored-far", &sparse);
13882
13883 let one = Reader::open(&near_path).expect("reopen from disk");
13884 let other = Reader::open(&far_path).expect("reopen from disk");
13885 let near = one.stored(0).expect("the column is stored");
13886 let far = other.stored(0).expect("the column is stored");
13887 assert_eq!(near.len(), parts, "one row per part");
13888 assert_eq!(far.len(), parts);
13889 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
13892 assert_eq!(total(&near), one.layout().columns[0].pages);
13893 assert_eq!(total(&far), other.layout().columns[0].pages);
13894 assert!(
13895 total(&near) * 2 < total(&far),
13896 "the sparse keys cost more, {} against {}",
13897 total(&far),
13898 total(&near)
13899 );
13900 for (at, part) in near.iter().enumerate() {
13902 assert_eq!(part.part, at);
13903 assert_eq!(part.row, at * per_part);
13904 assert_eq!(part.rows, per_part);
13905 let held = &ascending[at * per_part..(at + 1) * per_part];
13906 assert_eq!(part.low, Some(Value::BigInt(held[0])));
13907 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
13908 assert_eq!(part.nulls, Some(0));
13909 }
13910 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
13913 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
13914 assert_ne!(near[0].encoding, far[0].encoding);
13915 fs::remove_file(near_path).expect("remove scratch file");
13916 fs::remove_file(far_path).expect("remove scratch file");
13917 }
13918
13919 #[test]
13929 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
13930 let path = path("sieve-pays");
13931 let fields = vec![
13932 Field::required("spread", LogicalType::BigInt),
13933 Field::required("repeated", LogicalType::BigInt),
13934 ];
13935 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
13936 let parts = 3;
13937 let per_part = 1024;
13938 for part in 0..parts {
13939 let base = (part * per_part) as i64;
13940 let spread: Vec<Value> =
13941 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
13942 let repeated: Vec<Value> =
13943 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
13944 let chunk = Chunk::new(vec![
13945 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
13946 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
13947 ])
13948 .expect("two columns");
13949 writer.append(&chunk).expect("one part");
13950 }
13951 writer.finish().expect("commit");
13952
13953 let reader = Reader::open(&path).expect("reopen from disk");
13954 let layout = reader.layout();
13955 let spread = &layout.columns[0];
13956 let repeated = &layout.columns[1];
13957 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
13958 assert_eq!(
13959 repeated.sieves, 0,
13960 "a column whose filter costs more than its parts keeps none"
13961 );
13962 for column in &layout.columns {
13965 assert!(
13966 column.sieves < column.pages,
13967 "{} spends {} on sieves over {} of data",
13968 column.name,
13969 column.sieves,
13970 column.pages
13971 );
13972 }
13973 let absent = [Probe {
13975 column: 0,
13976 op: Op::Equal,
13977 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
13978 }];
13979 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
13980 fs::remove_file(path).expect("remove scratch file");
13981 }
13982
13983 #[test]
13989 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
13990 let path = path("sieve-damaged");
13991 let mut writer =
13992 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
13993 .expect("new file");
13994 let rows = 128;
13995 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
13996 let chunk =
13997 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
13998 .expect("one column");
13999 writer.append(&chunk).expect("one part");
14000 writer.finish().expect("commit");
14001
14002 let page = Reader::open(&path).expect("reopen").table.stripes[0]
14003 .sieves
14004 .get(0)
14005 .expect("a sieve page");
14006 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
14007 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
14008 file.write_all(&[0xff]).expect("damage one byte");
14009 drop(file);
14010
14011 let reader = Reader::open(&path).expect("reopen the damaged file");
14012 let absent =
14013 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
14014 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
14015 assert_eq!(
14016 reader.read(0, &[0]).expect("the rows are untouched").len(),
14017 usize::try_from(rows).expect("a small count")
14018 );
14019 fs::remove_file(path).expect("remove scratch file");
14020 }
14021
14022 #[test]
14033 fn workers_that_want_the_same_stripe_read_it_once() {
14034 let path = path("single-flight");
14035 let mut writer =
14036 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14037 .expect("new file");
14038 for part in 0..STRIPE_PARTS {
14039 let id = part as i32;
14040 let chunk = Chunk::new(vec![
14041 Vector::from_values(
14042 LogicalType::Integer,
14043 &[Value::Integer(id), Value::Integer(-id)],
14044 )
14045 .expect("integers"),
14046 ])
14047 .expect("matching rows");
14048 writer.append(&chunk).expect("one part");
14049 }
14050 writer.finish().expect("commit");
14051
14052 let reader = Reader::open(&path).expect("reopen from disk");
14053 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
14054 let barrier = std::sync::Barrier::new(8);
14055 std::thread::scope(|scope| {
14056 for worker in 0..8 {
14057 let reader = &reader;
14058 let barrier = &barrier;
14059 scope.spawn(move || {
14060 barrier.wait();
14061 for part in (worker..STRIPE_PARTS).step_by(8) {
14062 let chunk = reader.read(part, &[0]).expect("a whole page read");
14063 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14064 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
14065 }
14066 });
14067 }
14068 });
14069 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
14070 fs::remove_file(path).expect("remove scratch file");
14071 }
14072
14073 #[test]
14086 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
14087 let opened = |label: &str, rows_per_part: i32| {
14088 let path = path(label);
14089 let mut writer =
14090 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14091 .expect("new file");
14092 for part in 0..STRIPE_PARTS * 3 {
14093 let values = (0..rows_per_part)
14097 .map(|row| {
14098 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
14099 })
14100 .collect::<Vec<_>>();
14101 let chunk = Chunk::new(vec![
14102 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
14103 ])
14104 .expect("matching rows");
14105 writer.append(&chunk).expect("one part");
14106 }
14107 writer.finish().expect("commit");
14108 let reader = Reader::open(&path).expect("reopen from disk");
14109 let size = fs::metadata(&path).expect("the file is there").len();
14110 let out = (reader.reads(), reader.table().stripes().len(), size);
14111 fs::remove_file(path).expect("remove scratch file");
14112 out
14113 };
14114
14115 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
14116 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
14117 assert_eq!(
14118 thin_stripes, fat_stripes,
14119 "the same stripe count is what makes this a fair ask"
14120 );
14121 assert!(
14122 fat_size > thin_size * 50,
14123 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
14124 );
14125
14126 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
14127 assert_eq!(thin.pages, 0, "opening read a page");
14128 assert_eq!(fat.pages, 0, "opening read a page");
14129 assert_eq!(thin.indexes, 0, "opening read an index");
14130 assert_eq!(fat.indexes, 0, "opening read an index");
14131 assert!(
14134 fat.opening.bytes < thin.opening.bytes * 2,
14135 "opening the thin file read {} bytes and the fat one read {}",
14136 thin.opening.bytes,
14137 fat.opening.bytes
14138 );
14139 }
14140
14141 #[test]
14149 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
14150 let path = path("open-twice");
14151 let mut writer =
14152 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14153 .expect("new file");
14154 for part in 0..STRIPE_PARTS * 3 {
14155 let chunk = Chunk::new(vec![
14156 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14157 .expect("integers"),
14158 ])
14159 .expect("matching rows");
14160 writer.append(&chunk).expect("one part");
14161 }
14162 writer.finish().expect("commit");
14163
14164 let first = Reader::open(&path).expect("open");
14165 for part in 0..first.parts() {
14168 first.read(part, &[0]).expect("a part");
14169 }
14170 assert!(first.reads().pages > 0, "the scan has to have read something");
14171 let second = Reader::open(&path).expect("open again");
14172
14173 assert_eq!(first.reads().opening, second.reads().opening);
14174 assert_eq!(
14175 second.reads().pages,
14176 0,
14177 "the second open read a page off the back of the first"
14178 );
14179 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
14180 fs::remove_file(path).expect("remove scratch file");
14181 }
14182
14183 #[test]
14191 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
14192 let path = path("index-cache");
14193 let mut writer =
14194 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14195 .expect("new file");
14196 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14197 for part in 0..parts {
14198 let id = part as i32;
14199 let chunk = Chunk::new(vec![
14200 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
14201 ])
14202 .expect("matching rows");
14203 writer.append(&chunk).expect("one part");
14204 }
14205 writer.finish().expect("commit");
14206
14207 let reader = Reader::open(&path).expect("reopen from disk");
14208 let stripes = reader.table().stripes().len();
14209 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
14210 for _ in 0..2 {
14212 for part in 0..parts {
14213 let chunk = reader.read(part, &[0]).expect("a part");
14214 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14215 }
14216 }
14217 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
14218 assert!(
14219 reader.pages.load(Atomic::Relaxed) > stripes,
14220 "the pages are the ones that get read again, which is what makes the index count mean \
14221 something"
14222 );
14223 fs::remove_file(path).expect("remove scratch file");
14224 }
14225
14226 #[test]
14233 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
14234 let path = path("page-pool");
14235 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
14236 let fields = || vec![Field::required("id", LogicalType::Integer)];
14237 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
14238 for table in ["a", "b"] {
14239 if table == "b" {
14240 writer = writer.next("b".to_string(), fields()).expect("a second table");
14241 }
14242 for part in 0..parts {
14243 let chunk = Chunk::new(vec![
14244 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14245 .expect("integers"),
14246 ])
14247 .expect("matching rows");
14248 writer.append(&chunk).expect("one part");
14249 }
14250 }
14251 writer.finish().expect("commit");
14252
14253 let pool = PagePool::new(usize::MAX);
14254 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
14255 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
14256 let stripes = a.table().stripes().len();
14257 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
14258 let scan = |reader: &Reader| {
14259 for part in 0..parts {
14260 let chunk = reader.read(part, &[0]).expect("a part");
14261 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14262 }
14263 };
14264 scan(&a);
14265 scan(&a);
14266 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
14267 let one = pool.bytes();
14268 assert!(one > 0, "the pool counts what the reader holds");
14269
14270 pool.budget.store(one, Atomic::Relaxed);
14272 scan(&b);
14273 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
14274 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
14275 let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
14276 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
14277
14278 drop((a, b, catalog));
14280 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
14281 scan(&c);
14282 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
14283 fs::remove_file(path).expect("remove scratch file");
14284 }
14285
14286 #[test]
14295 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
14296 let workers = CACHED_STRIPES_PER_COLUMN + 4;
14297 let path = path("stripe-per-worker");
14298 let mut writer =
14299 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14300 .expect("new file");
14301 for part in 0..STRIPE_PARTS * workers {
14302 let chunk = Chunk::new(vec![
14303 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
14304 .expect("integers"),
14305 ])
14306 .expect("matching rows");
14307 writer.append(&chunk).expect("one part");
14308 }
14309 writer.finish().expect("commit");
14310
14311 let read = |told: bool| {
14312 let reader = Reader::open(&path).expect("reopen from disk");
14313 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
14314 if told {
14315 reader.keep_stripes(workers);
14316 }
14317 let barrier = std::sync::Barrier::new(workers);
14318 std::thread::scope(|scope| {
14319 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
14320 let reader = &reader;
14321 let barrier = &barrier;
14322 scope.spawn(move || {
14323 for part in run {
14324 barrier.wait();
14325 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
14326 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
14327 }
14328 assert!(worker < workers);
14329 });
14330 }
14331 });
14332 reader.pages.load(Atomic::Relaxed)
14333 };
14334
14335 assert_eq!(read(true), workers, "one page read per stripe and no more");
14336 assert!(read(false) > workers, "a cache that small is read again on every part");
14337 fs::remove_file(path).expect("remove scratch file");
14338 }
14339
14340 #[test]
14345 fn a_damaged_index_page_is_an_error() {
14346 let path = path("damaged-index");
14347 let mut writer =
14348 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
14349 .expect("new file");
14350 writer.append(&sample_ids()).expect("first part");
14351 writer.append(&sample_ids()).expect("second part");
14352 writer.finish().expect("commit");
14353
14354 let reader = Reader::open(&path).expect("valid directory");
14355 let index = reader.table.stripes[0].index;
14356 let mut byte = [0; 1];
14357 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
14358 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
14359 file.seek(SeekFrom::Start(index.offset)).expect("index start");
14360 file.write_all(&[!byte[0]]).expect("damage the first part length");
14361 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
14362 assert!(error.message().contains("index page section checksum differs"), "{error}");
14363 fs::remove_file(path).expect("remove scratch file");
14364 }
14365
14366 #[test]
14373 fn every_integer_width_round_trips_through_a_page() {
14374 let path = path("integer-widths");
14375 let columns = [
14376 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
14377 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
14378 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
14379 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
14380 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
14381 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
14382 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
14383 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
14384 ];
14385 let fields = columns
14386 .iter()
14387 .enumerate()
14388 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14389 .collect::<Vec<_>>();
14390 let vectors = columns
14391 .iter()
14392 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14393 .collect::<Vec<_>>();
14394 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
14395 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14396 writer.finish().expect("commit");
14397
14398 let reader = Reader::open(&path).expect("reopen from disk");
14399 let wanted = (0..columns.len()).collect::<Vec<_>>();
14400 let read = reader.read(0, &wanted).expect("every column");
14401 assert_eq!(read.len(), 2);
14402 for (at, (ty, values)) in columns.iter().enumerate() {
14404 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14405 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14406 }
14407 fs::remove_file(path).expect("remove scratch file");
14408 }
14409
14410 #[test]
14421 fn every_other_type_the_format_knows_round_trips_through_a_page() {
14422 let path = path("other-types");
14423 let columns = [
14424 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
14425 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
14426 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
14427 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
14428 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
14429 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
14430 (
14431 LogicalType::TimestampTz,
14432 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
14433 ),
14434 (
14435 LogicalType::Interval,
14436 vec![
14437 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
14438 Value::Interval { months: 13, days: -1, micros: 1 },
14439 ],
14440 ),
14441 (
14442 LogicalType::Blob,
14443 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
14444 ),
14445 ];
14446 let fields = columns
14447 .iter()
14448 .enumerate()
14449 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
14450 .collect::<Vec<_>>();
14451 let vectors = columns
14452 .iter()
14453 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
14454 .collect::<Vec<_>>();
14455 let mut writer = Writer::create(&path, "others", fields).expect("new file");
14456 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14457 writer.finish().expect("commit");
14458
14459 let reader = Reader::open(&path).expect("reopen from disk");
14460 let wanted = (0..columns.len()).collect::<Vec<_>>();
14461 let read = reader.read(0, &wanted).expect("every column");
14462 assert_eq!(read.len(), 2);
14463 for (at, (ty, values)) in columns.iter().enumerate() {
14464 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
14465 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
14466 }
14467 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
14470 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
14471
14472 fs::remove_file(path).expect("remove scratch file");
14473 }
14474
14475 #[test]
14481 fn a_nan_survives_being_written_down() {
14482 let path = path("nan");
14483 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
14484 .expect("a NaN vector");
14485 let mut writer =
14486 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
14487 .expect("new file");
14488 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
14489 writer.finish().expect("commit");
14490 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
14491 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
14492 assert!(back.is_nan(), "a NaN came back as {back}");
14493 fs::remove_file(path).expect("remove scratch file");
14494 }
14495
14496 #[test]
14503 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
14504 let path = path("uuid-and-bit");
14505 let uuids = vec![0_i128, i128::MIN, -1];
14506 let mut bits = StringColumn::new();
14507 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
14508 bits.push_bytes(value);
14509 }
14510 let expected = bits.clone();
14511 let fields =
14512 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
14513 let vectors = vec![
14514 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
14515 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
14516 ];
14517 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
14518 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
14519 writer.finish().expect("commit");
14520
14521 let reader = Reader::open(&path).expect("reopen from disk");
14522 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
14523 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
14524 panic!("a uuid column is the 128 bit lane")
14525 };
14526 assert_eq!(back.as_slice(), uuids.as_slice());
14527 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
14528 panic!("a bit column is bytes")
14529 };
14530 for row in 0..expected.len() {
14531 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
14532 }
14533 fs::remove_file(path).expect("remove scratch file");
14534 }
14535
14536 #[test]
14539 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
14540 let mut rows: Vec<Option<u64>> = Vec::new();
14541 let mut state = 0x2545_f491_4f6c_dd1d_u64;
14542 for index in 0..400_000_u64 {
14543 state ^= state << 13;
14544 state ^= state >> 7;
14545 state ^= state << 17;
14546 let times = 1 + (state % 7) as usize;
14547 let bits = match state % 11 {
14548 0 => None,
14549 1..=3 => Some(state % 16),
14550 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
14551 };
14552 rows.extend(std::iter::repeat_n(bits, times));
14553 }
14554 let mut by_row = Candidates::default();
14555 for &bits in &rows {
14556 by_row.add(bits, 1);
14557 }
14558 let mut by_run = Candidates::default();
14559 let mut run = Run::default();
14560 let mut runs = 0_usize;
14561 for &bits in &rows {
14562 if let Some((bits, times)) = run.push(bits) {
14563 by_run.add(bits, times);
14564 runs += 1;
14565 }
14566 }
14567 if let Some((bits, times)) = run.take() {
14568 by_run.add(bits, times);
14569 }
14570 assert!(runs < rows.len() / 2, "the rows came in runs");
14571 assert!(by_row.decrements > 0, "the table filled and turned values away");
14572 assert_eq!(sorted_candidates(&by_run), sorted_candidates(&by_row));
14573 assert_eq!(by_run.nulls, by_row.nulls);
14574 assert_eq!(by_run.decrements, by_row.decrements);
14575 }
14576
14577 fn sorted_candidates(candidates: &Candidates) -> Vec<(u64, u32)> {
14578 let mut pairs = candidates.pairs().collect::<Vec<_>>();
14579 pairs.sort_unstable();
14580 assert_eq!(pairs.len(), candidates.held, "the count of held slots drifted");
14581 pairs
14582 }
14583
14584 #[derive(Default)]
14587 struct MapCandidates {
14588 counts: HashMap<u64, u32>,
14589 nulls: u32,
14590 decrements: u64,
14591 }
14592
14593 impl MapCandidates {
14594 fn add(&mut self, bits: Option<u64>, mut times: u32) {
14595 while times > 0 {
14596 let held = match bits {
14597 Some(bits) => self.counts.get_mut(&bits),
14598 None if self.nulls != 0 => Some(&mut self.nulls),
14599 None => None,
14600 };
14601 if let Some(count) = held {
14602 *count = count.saturating_add(times);
14603 return;
14604 }
14605 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
14606 match bits {
14607 Some(bits) => {
14608 self.counts.insert(bits, times);
14609 }
14610 None => self.nulls = times,
14611 }
14612 return;
14613 }
14614 self.counts.retain(|_, count| {
14615 *count -= 1;
14616 *count != 0
14617 });
14618 self.nulls = self.nulls.saturating_sub(1);
14619 self.decrements = self.decrements.saturating_add(1);
14620 times -= 1;
14621 }
14622 }
14623 }
14624
14625 #[test]
14629 fn the_open_addressed_candidates_agree_with_the_map_they_replaced() {
14630 for seed in [0x2545_f491_4f6c_dd1d_u64, 0x9e37_79b9_7f4a_7c15, 7] {
14631 let mut table = Candidates::default();
14632 let mut oracle = MapCandidates::default();
14633 let mut state = seed;
14634 for index in 0..300_000_u64 {
14635 state ^= state << 13;
14636 state ^= state >> 7;
14637 state ^= state << 17;
14638 let bits = match state % 13 {
14639 0 => None,
14640 1..=4 => Some(state % 40),
14641 5 => Some((index % 1000) * 1_000_000),
14642 _ => Some(state),
14643 };
14644 let times = 1 + (state >> 60) as u32 % 3;
14645 table.add(bits, times);
14646 oracle.add(bits, times);
14647 if index % 50_000 == 0 {
14648 let mut expected =
14649 oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14650 expected.sort_unstable();
14651 assert_eq!(sorted_candidates(&table), expected, "seed {seed} row {index}");
14652 }
14653 }
14654 let mut expected = oracle.counts.iter().map(|(&b, &c)| (b, c)).collect::<Vec<_>>();
14655 expected.sort_unstable();
14656 assert_eq!(sorted_candidates(&table), expected, "seed {seed}");
14657 assert_eq!(table.nulls, oracle.nulls, "seed {seed}");
14658 assert_eq!(table.decrements, oracle.decrements, "seed {seed}");
14659 assert!(table.decrements > 0, "seed {seed} never filled the table");
14660 for &(bits, _) in &expected {
14661 assert!(table.position(bits).is_some(), "seed {seed} lost {bits}");
14662 }
14663 }
14664 }
14665
14666 #[test]
14667 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
14668 let path = path("frequency-ordinals");
14669 let mut writer =
14670 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
14671 .expect("new file");
14672 let mut values = Vec::new();
14673 for leader in 0..10_i64 {
14674 values.extend(std::iter::repeat_n(leader, 100));
14675 }
14676 values.extend(1_000_i64..41_000);
14677 for part in values.chunks(1_024) {
14678 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
14679 .expect("big integers");
14680 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
14681 }
14682 writer.finish().expect("commit");
14683
14684 let reader = Reader::open(&path).expect("reopen from disk");
14685 let occurrences =
14686 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
14687 assert!(occurrences.omitted_max < 100);
14688 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
14689 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
14690 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
14691 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
14692 assert_eq!(
14693 &occurrences.anchor_indices[..1_000]
14694 .iter()
14695 .map(|&entry| occurrences.anchors[entry as usize].clone())
14696 .collect::<Vec<_>>(),
14697 &(0_i64..10)
14698 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
14699 .collect::<Vec<_>>()
14700 );
14701 fs::remove_file(path).expect("remove scratch file");
14702 }
14703
14704 #[test]
14705 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
14706 let path = path("frequency-bits");
14711 let mut writer = Writer::create(
14712 &path,
14713 "items",
14714 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
14715 )
14716 .expect("new file");
14717 let mut rows = Vec::new();
14718 let mut leaders = Vec::new();
14719 for leader in 0..10_u64 {
14720 let count = 300 - leader * 10;
14721 let (unsigned, signed) = if leader == 0 {
14722 (Value::Null, Value::Null)
14723 } else {
14724 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
14725 };
14726 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
14727 leaders.push(((unsigned, count), (signed, count)));
14728 }
14729 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
14730 for part in rows.chunks(1_024) {
14731 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
14732 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
14733 let chunk = Chunk::new(vec![
14734 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
14735 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
14736 ])
14737 .expect("matching columns");
14738 writer.append(&chunk).expect("rows");
14739 }
14740 writer.finish().expect("commit");
14741
14742 let reader = Reader::open(&path).expect("reopen from disk");
14743 for column in 0..2 {
14744 let prefix =
14745 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14746 let wanted = leaders
14747 .iter()
14748 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
14749 .cloned()
14750 .collect::<Vec<_>>();
14751 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
14752 assert!(prefix.omitted_max < 210, "column {column}");
14753 assert_eq!(
14754 reader.distinct_values(column).expect("valid metadata"),
14755 Some(9 + 40_000),
14756 "column {column}"
14757 );
14758 }
14759 fs::remove_file(path).expect("remove scratch file");
14760 }
14761
14762 #[test]
14763 fn a_narrow_column_takes_its_frequencies_from_the_tally_and_they_match_the_rows() {
14764 let path = path("frequency-tally");
14770 let types = [
14771 LogicalType::TinyInt,
14772 LogicalType::UInteger,
14773 LogicalType::Date,
14774 LogicalType::Timestamp,
14775 ];
14776 let value = |ty: &LogicalType, at: i64| match ty {
14777 LogicalType::TinyInt => Value::TinyInt((at % 250 - 125) as i8),
14778 LogicalType::UInteger => Value::UInteger(u32::MAX - at as u32),
14779 LogicalType::Date => Value::Date(19_000 - at as i32),
14780 _ => Value::Timestamp(1_700_000_000_000_000 - at * 1_000_003),
14781 };
14782 let fields = types
14783 .iter()
14784 .enumerate()
14785 .map(|(at, ty)| Field::new(format!("c{at}"), ty.clone()))
14786 .collect::<Vec<_>>();
14787 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14788 let mut rows = Vec::new();
14789 for at in 0..250_i64 {
14790 for _ in 0..=(at % 37) {
14791 rows.push(if rows.len() % 13 == 0 { None } else { Some(at) });
14792 }
14793 }
14794 for part in rows.chunks(1_000) {
14795 let columns = types
14796 .iter()
14797 .map(|ty| {
14798 let values = part
14799 .iter()
14800 .map(|row| row.map_or(Value::Null, |at| value(ty, at)))
14801 .collect::<Vec<_>>();
14802 Vector::from_values(ty.clone(), &values).expect("a column")
14803 })
14804 .collect();
14805 writer.append(&Chunk::new(columns).expect("matching columns")).expect("rows");
14806 }
14807 writer.finish().expect("commit");
14808
14809 let reader = Reader::open(&path).expect("reopen from disk");
14810 for (column, ty) in types.iter().enumerate() {
14811 let mut counts = HashMap::<Option<i64>, u64>::new();
14812 for row in &rows {
14813 *counts.entry(*row).or_default() += 1;
14814 }
14815 let wanted = counts
14816 .into_iter()
14817 .map(|(row, count)| (row.map_or(Value::Null, |at| value(ty, at)), count))
14818 .collect::<Vec<_>>();
14819 let prefix =
14820 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
14821 assert_eq!(prefix.entries.len(), wanted.len(), "column {column}");
14822 assert_eq!(prefix.omitted_max, 0, "column {column}");
14823 for (value, count) in &prefix.entries {
14824 let held =
14825 wanted.iter().find(|(wanted, _)| wanted == value).map(|(_, count)| count);
14826 assert_eq!(held, Some(count), "column {column} value {value:?}");
14827 }
14828 assert!(prefix.entries.windows(2).all(|pair| pair[0].1 >= pair[1].1));
14829 assert_eq!(
14830 reader.distinct_values(column).expect("valid metadata"),
14831 Some(wanted.len() as u64 - 1),
14832 "column {column}"
14833 );
14834 }
14835 fs::remove_file(path).expect("remove scratch file");
14836 }
14837
14838 #[test]
14839 fn distinct_counts_are_exact_either_side_of_a_full_candidate_table() {
14840 let edge = FREQUENCY_CANDIDATES as i64;
14845 for distinct in [0, 1, 7, edge - 2, edge - 1, edge, edge + 1, edge + 2, 3 * edge] {
14846 for with_null in [false, true] {
14847 let path = path("distinct-edge");
14848 let mut writer =
14849 Writer::create(&path, "items", vec![Field::new("id", LogicalType::BigInt)])
14850 .expect("new file");
14851 let mut values = Vec::new();
14852 for round in 0..2 {
14853 for value in 0..distinct {
14854 let repeat = if round == 0 { 1 + (value % 3) as usize } else { 1 };
14855 values.extend(std::iter::repeat_n(
14856 Value::BigInt(value * 7_919 % distinct),
14857 repeat,
14858 ));
14859 if with_null && value % 1_000 == 0 {
14860 values.push(Value::Null);
14861 }
14862 }
14863 }
14864 if with_null {
14865 values.push(Value::Null);
14866 }
14867 for part in values.chunks(1_024) {
14868 let chunk = Chunk::new(vec![
14869 Vector::from_values(LogicalType::BigInt, part).expect("ids"),
14870 ])
14871 .expect("one column");
14872 writer.append(&chunk).expect("rows");
14873 }
14874 writer.finish().expect("commit");
14875 let reader = Reader::open(&path).expect("reopen from disk");
14876 assert_eq!(
14877 reader.distinct_values(0).expect("valid metadata"),
14878 Some(distinct as u64),
14879 "{distinct} values, null {with_null}"
14880 );
14881 fs::remove_file(path).expect("remove scratch file");
14882 }
14883 }
14884 }
14885
14886 #[test]
14887 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
14888 let path = path("quick-nonzero");
14889 let mut writer = Writer::create(
14890 &path,
14891 "items",
14892 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
14893 )
14894 .expect("create");
14895 for ids in [
14896 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
14897 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
14898 ] {
14899 let labels = vec![Value::Varchar("same".into()); ids.len()];
14900 writer
14901 .append(
14902 &Chunk::new(vec![
14903 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
14904 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
14905 ])
14906 .expect("chunk"),
14907 )
14908 .expect("append");
14909 }
14910 writer.finish().expect("finish");
14911 let catalog = Catalog::open(&path).expect("catalog");
14912 assert_eq!(catalog.entries[0].nonzero, vec![None, None]);
14913 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
14914 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
14915 let frequencies =
14916 catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
14917 assert_eq!(frequencies.len(), 4);
14918 for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
14919 assert!(frequencies.contains(&pair), "missing {pair:?}");
14920 }
14921 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
14922 assert_eq!(
14923 catalog.integer_extremes("items", 1).expect("extremes"),
14924 Some(IntegerExtremes::Values { low: 0, high: 7 })
14925 );
14926 assert_eq!(
14927 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
14928 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
14929 );
14930 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
14931 let mut legacy = catalog.clone();
14932 Arc::make_mut(&mut legacy.entries)[0].nonzero[1] = Some(999);
14933 assert_eq!(legacy.nonzero_count("items", 1).expect("ignore legacy count"), Some(2));
14934 Arc::make_mut(&mut legacy.entries)[0].frequencies[1] = None;
14935 assert_eq!(legacy.nonzero_count("items", 1).expect("directory fallback"), Some(2));
14936 Writer::certify_counts(&path).expect("recertify");
14937 assert_eq!(
14938 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
14939 Some(2)
14940 );
14941 assert_eq!(
14942 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
14943 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
14944 );
14945 assert_eq!(
14946 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
14947 Some(3)
14948 );
14949 assert_eq!(
14950 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
14951 Some(IntegerExtremes::Values { low: 0, high: 7 })
14952 );
14953 assert_eq!(
14954 Catalog::open(&path)
14955 .expect("reopen")
14956 .exact_numeric_frequencies("items", 1)
14957 .expect("frequencies"),
14958 Some(frequencies)
14959 );
14960 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
14961 fs::remove_file(path).expect("remove scratch file");
14962 }
14963
14964 #[test]
14965 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
14966 let path = path("pair-frequencies");
14967 let mut pairs = Vec::new();
14968 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
14969 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
14970 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
14971 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
14972 let mut writer = Writer::create(
14973 &path,
14974 "items",
14975 vec![
14976 Field::required("id", LogicalType::BigInt),
14977 Field::required("phrase", LogicalType::Varchar),
14978 ],
14979 )
14980 .expect("new file");
14981 for part in pairs.chunks(1_024) {
14982 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
14983 let phrases =
14984 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
14985 writer
14986 .append(
14987 &Chunk::new(vec![
14988 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
14989 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
14990 ])
14991 .expect("matching columns"),
14992 )
14993 .expect("rows");
14994 }
14995 writer.finish().expect("commit");
14996
14997 let reader = Reader::open(&path).expect("reopen from disk");
14998 assert!(
14999 reader.table.pair_frequencies.is_empty(),
15000 "no query-specific pair result is stored"
15001 );
15002 fs::remove_file(path).expect("remove scratch file");
15003 }
15004
15005 #[test]
15011 fn a_file_from_another_format_says_which_format_it_is() {
15012 let older = path("older-format");
15013 let mut writer =
15014 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
15015 .expect("new file");
15016 let chunk = Chunk::new(vec![
15017 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15018 .expect("integers"),
15019 ])
15020 .expect("chunk");
15021 writer.append(&chunk).expect("page written");
15022 writer.finish().expect("commit");
15023
15024 let unreadable =
15028 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
15029 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15030 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
15031 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
15032 drop(file);
15033 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
15034 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
15035 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
15036
15037 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
15038 file.seek(SeekFrom::Start(0)).expect("the magic is first");
15039 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
15040 drop(file);
15041 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
15042 assert!(complaint.contains("magic"), "{complaint}");
15043 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
15044 fs::remove_file(older).expect("remove scratch file");
15045 }
15046
15047 #[test]
15048 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
15049 let unfinished = path("unfinished");
15050 let mut writer =
15051 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
15052 .expect("new file");
15053 let chunk = Chunk::new(vec![
15054 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
15055 .expect("integers"),
15056 ])
15057 .expect("chunk");
15058 writer.append(&chunk).expect("page written");
15059 drop(writer);
15060 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
15061 fs::remove_file(unfinished).expect("remove scratch file");
15062
15063 let damaged = path("damaged");
15064 let mut writer =
15065 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
15066 .expect("new file");
15067 writer.append(&chunk).expect("page written");
15068 writer.finish().expect("commit");
15069 let reader = Reader::open(&damaged).expect("valid directory");
15070 let mut file =
15071 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
15072 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
15073 file.write_all(&[255]).expect("damage one byte");
15074 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
15075 fs::remove_file(damaged).expect("remove scratch file");
15076 }
15077
15078 #[test]
15079 fn damaged_lazy_dictionary_payload_is_an_error() {
15080 let path = path("damaged-dictionary");
15081 let mut writer = Writer::create(
15082 &path,
15083 "items",
15084 vec![
15085 Field::required("id", LogicalType::Integer),
15086 Field::new("text", LogicalType::Varchar),
15087 ],
15088 )
15089 .expect("new file");
15090 writer.append(&sample()).expect("stripe written");
15091 writer.finish().expect("commit");
15092
15093 let reader = Reader::open(&path).expect("valid directory");
15094 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
15095 let mut header = [0; DICTIONARY_HEADER];
15098 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15099 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15102 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15103 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
15104 let bits = (width & !DICTIONARY_FLAGS) as usize;
15105 let mut start = [0; 8];
15106 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
15107 read_at(&reader.file, at, &mut start).expect("the first block's start");
15108 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15109 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
15110 file.write_all(&[255]).expect("damage dictionary payload");
15111
15112 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
15113 let error =
15114 chunk.validate_external().expect_err("payload corruption must reach the caller");
15115 assert!(error.message().contains("payload checksum differs"), "{error}");
15116 fs::remove_file(path).expect("remove scratch file");
15117 }
15118
15119 #[test]
15129 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
15130 let path = path("dictionary-decide");
15131 let rows = 20_000;
15132 let unique =
15134 |row: usize| format!("{row:09} a value that appears exactly once in the table");
15135 let repeated = |row: usize| unique(row / 40);
15137 let mut writer = Writer::create(
15138 &path,
15139 "items",
15140 vec![
15141 Field::required("unique", LogicalType::Varchar),
15142 Field::required("repeated", LogicalType::Varchar),
15143 ],
15144 )
15145 .expect("new file");
15146 for part in (0..rows).step_by(1_000) {
15147 let span = part..(part + 1_000).min(rows);
15148 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
15149 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
15150 writer
15151 .append(
15152 &Chunk::new(vec![
15153 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
15154 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
15155 ])
15156 .expect("two columns"),
15157 )
15158 .expect("a part");
15159 }
15160 writer.finish().expect("commit");
15161
15162 let reader = Reader::open(&path).expect("reopen from disk");
15163 assert!(
15164 reader.table.dictionaries[0].is_none(),
15165 "a column with no repeats has nothing to say twice"
15166 );
15167 assert!(
15168 reader.table.dictionaries[1].is_some(),
15169 "a column whose values come round again keeps its dictionary"
15170 );
15171 let mut first = 0;
15172 for part in 0..reader.parts() {
15173 let chunk = reader.read(part, &[0, 1]).expect("a part");
15174 for row in 0..chunk.len() {
15175 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
15176 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
15177 }
15178 first += chunk.len();
15179 }
15180 assert_eq!(first, rows, "every row was read back");
15181 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
15182 let size = fs::metadata(&path).expect("the file is there").len() as usize;
15183 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
15184 fs::remove_file(path).expect("remove scratch file");
15185 }
15186
15187 #[test]
15200 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
15201 let path = path("dictionary-blocks");
15202 let value = |row: usize| {
15203 let row = row.saturating_sub(8_000);
15204 format!("{row:07} a value long enough to be worth a payload block")
15205 };
15206 let parts = 40;
15207 let per_part = 1000;
15208 let mut writer =
15209 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15210 .expect("new file");
15211 for part in 0..parts {
15212 let values = (0..per_part)
15213 .map(|row| Value::Varchar(value(part * per_part + row)))
15214 .collect::<Vec<_>>();
15215 let chunk = Chunk::new(vec![
15216 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15217 ])
15218 .expect("matching rows");
15219 writer.append(&chunk).expect("a part");
15220 }
15221 writer.finish().expect("commit");
15222
15223 let reader = Reader::open(&path).expect("reopen from disk");
15224 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
15225 assert!(
15226 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
15227 "the dictionary has to be several blocks for this to be testing anything"
15228 );
15229 for part in [0, parts - 1] {
15230 let chunk = reader.read(part, &[0]).expect("a part");
15231 chunk.validate_external().expect("every payload block checks out");
15232 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
15233 }
15234
15235 let mut header = [0; DICTIONARY_HEADER];
15237 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
15238 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
15239 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
15240 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
15241 let bits = (width & !DICTIONARY_FLAGS) as usize;
15242 let mut place = [0; 16];
15243 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
15244 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
15245 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
15246 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
15247 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15248 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
15249 file.write_all(&[255]).expect("damage the last payload block");
15250 let reader = Reader::open(&path).expect("the directory and the index are untouched");
15251 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
15252 let error = chunk.validate_external().expect_err("the damage must reach the caller");
15253 assert!(error.message().contains("payload checksum differs"), "{error}");
15254 fs::remove_file(path).expect("remove scratch file");
15255 }
15256
15257 #[test]
15271 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
15272 let path = path("dictionary-offsets");
15273 let value = |row: usize| {
15274 let row = row % 5_000;
15275 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
15276 };
15277 let rows = 6_000;
15278 let mut writer =
15279 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15280 .expect("new file");
15281 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
15282 for part in values.chunks(1_000) {
15283 let chunk =
15284 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
15285 .expect("matching rows");
15286 writer.append(&chunk).expect("a part");
15287 }
15288 writer.finish().expect("commit");
15289
15290 let reader = Reader::open(&path).expect("reopen from disk");
15291 assert!(
15292 rows > TEXT_PAYLOAD_VALUES * 4,
15293 "the dictionary has to be several blocks for this to be testing anything"
15294 );
15295 for part in 0..rows / 1_000 {
15296 let chunk = reader.read(part, &[0]).expect("a part");
15297 for row in 0..1_000 {
15298 let row = part * 1_000 + row;
15299 assert_eq!(
15300 chunk.value_at(row % 1_000, 0),
15301 Value::Varchar(value(row)),
15302 "value {row}"
15303 );
15304 }
15305 }
15306 for _ in 0..2 {
15309 for part in 0..rows / 1_000 {
15310 let chunk = reader.read(part, &[0]).expect("a part");
15311 let mut lens = vec![0_i64; 1_000];
15312 let column = chunk.column(0).expect("one column");
15313 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
15314 for (row, &len) in lens.iter().enumerate() {
15315 let row = part * 1_000 + row;
15316 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
15317 }
15318 }
15319 }
15320 fs::remove_file(path).expect("remove scratch file");
15321 }
15322
15323 #[test]
15325 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
15326 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
15327 ends.extend([3, 3, 10]);
15328 let Some(Lengths::Narrow(lens)) = lengths_of(&ends) else { panic!("short ordered ends") };
15329 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
15330 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
15331 let long = [5, 70_005, 70_006];
15333 let Some(Lengths::Wide(lens)) = lengths_of(&long) else { panic!("long ordered ends") };
15334 assert_eq!(lens, [5, 70_000, 1]);
15335 let mut read = Vec::new();
15336 Lengths::Wide(lens).extend_at(&[1, 9, 0], &mut read);
15337 assert_eq!(read, [70_000, 0, 5], "a position past the end is no length");
15338 ends.push(9);
15339 assert!(lengths_of(&ends).is_none());
15340 }
15341
15342 #[test]
15354 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
15355 let path = path("dictionary-once");
15356 let parts = 8;
15357 let per_part = 500;
15358 let value =
15359 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
15360 let mut writer =
15361 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15362 .expect("new file");
15363 for part in 0..parts {
15364 let values = (0..per_part)
15365 .map(|row| Value::Varchar(value(part * per_part + row)))
15366 .collect::<Vec<_>>();
15367 let chunk = Chunk::new(vec![
15368 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15369 ])
15370 .expect("matching rows");
15371 writer.append(&chunk).expect("a part");
15372 }
15373 writer.finish().expect("commit");
15374
15375 let reader = Reader::open(&path).expect("reopen from disk");
15376 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
15377 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
15378
15379 let workers = 16;
15380 let gate = std::sync::Barrier::new(workers);
15381 std::thread::scope(|scope| {
15382 for worker in 0..workers {
15383 let reader = reader.clone();
15384 let gate = &gate;
15385 scope.spawn(move || {
15386 gate.wait();
15387 let chunk = reader.read(worker % parts, &[0]).expect("a part");
15388 assert_eq!(
15389 chunk.value_at(0, 0),
15390 Value::Varchar(value((worker % parts) * per_part))
15391 );
15392 });
15393 }
15394 });
15395
15396 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
15397 fs::remove_file(path).expect("remove scratch file");
15398 }
15399
15400 #[test]
15405 fn a_damaged_sorted_order_is_an_error() {
15406 let path = path("damaged-order");
15407 let mut writer = Writer::create(
15408 &path,
15409 "items",
15410 vec![
15411 Field::required("id", LogicalType::Integer),
15412 Field::new("text", LogicalType::Varchar),
15413 ],
15414 )
15415 .expect("new file");
15416 writer.append(&sample()).expect("stripe written");
15417 writer.finish().expect("commit");
15418
15419 let reader = Reader::open(&path).expect("valid directory");
15420 let page = reader.table.dictionaries[1].expect("string dictionary page");
15421 let mut header = [0; DICTIONARY_HEADER];
15422 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
15423 let index_len = dictionary_index_len(&header);
15424 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
15425 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
15426 file.write_all(&[255]).expect("damage the order");
15427
15428 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
15429 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
15430 assert!(error.message().contains("rank checksum differs"), "{error}");
15431 fs::remove_file(path).expect("remove scratch file");
15432 }
15433
15434 #[test]
15438 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
15439 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
15442 let path = path("dictionary-order");
15443 let mut writer =
15444 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15445 .expect("new file");
15446 writer
15447 .append(
15448 &Chunk::new(vec![
15449 Vector::from_values(
15450 LogicalType::Varchar,
15451 &spellings.map(|text| Value::Varchar(text.into())),
15452 )
15453 .expect("strings"),
15454 ])
15455 .expect("one column"),
15456 )
15457 .expect("stripe written");
15458 writer.finish().expect("commit");
15459
15460 let reader = Reader::open(&path).expect("valid directory");
15461 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15462 let count = dictionary.ranks().expect("a v10 file stores one");
15463 assert_eq!(count, spellings.len(), "every distinct value has a rank");
15464 let order = (0..count)
15465 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
15466 .collect::<Vec<_>>();
15467 let mut seen = order.clone();
15468 seen.sort_unstable();
15469 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
15470
15471 let ranked = order
15472 .iter()
15473 .map(|&code| {
15474 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15475 })
15476 .collect::<Vec<_>>();
15477 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
15478 expected.sort();
15479 assert_eq!(ranked, expected, "rank order is value order");
15480
15481 for (rank, value) in expected.iter().enumerate() {
15484 assert_eq!(
15485 dictionary.compare_rank(rank, value).expect("compare"),
15486 Ordering::Equal,
15487 "rank {rank} is its own value"
15488 );
15489 if rank > 0 {
15490 assert_eq!(
15491 dictionary.compare_rank(rank - 1, value).expect("compare"),
15492 Ordering::Less,
15493 "rank {rank} follows the one before it"
15494 );
15495 }
15496 }
15497 fs::remove_file(path).expect("remove scratch file");
15498 }
15499
15500 #[test]
15507 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
15508 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
15509 let path = path("dictionaries-at-once");
15510 let fields = (0..sizes.len())
15511 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
15512 .collect::<Vec<_>>();
15513 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15514 let rows = 10_000_usize;
15515 for start in (0..rows).step_by(1_024) {
15516 let columns = sizes
15517 .iter()
15518 .enumerate()
15519 .map(|(column, &size)| {
15520 let values = (start..(start + 1_024).min(rows))
15521 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
15522 .collect::<Vec<_>>();
15523 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
15524 })
15525 .collect::<Vec<_>>();
15526 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
15527 }
15528 writer.finish().expect("commit");
15529
15530 let reader = Reader::open(&path).expect("valid directory");
15531 for (column, &size) in sizes.iter().enumerate() {
15532 let dictionary =
15533 reader.dictionary(column).expect("read").expect("a string column has one");
15534 let count = dictionary.ranks().expect("a v10 file stores one");
15535 assert_eq!(count, size, "column {column} has its own distinct count");
15536 let ranked = (0..count)
15537 .map(|rank| {
15538 let code = dictionary.code_at_rank(rank).expect("a code");
15539 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15540 })
15541 .collect::<Vec<_>>();
15542 let expected = (0..size)
15543 .map(|value| format!("c{column}-{value:05}").into_bytes())
15544 .collect::<Vec<_>>();
15545 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
15546 }
15547 fs::remove_file(path).expect("remove scratch file");
15548 }
15549
15550 #[test]
15558 fn a_large_dictionary_ranks_in_value_order() {
15559 let path = path("dictionary-large-rank");
15560 let value = |row: u64| {
15561 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
15562 match row % 3 {
15563 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
15564 1 => format!("{mixed}"),
15565 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
15566 }
15567 };
15568 let distinct = 70_000;
15569 let parts = 4 * distinct / 1000;
15570 let per_part = 1000;
15571 let mut writer =
15572 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15573 .expect("new file");
15574 for part in 0..parts {
15575 let values = (0..per_part)
15576 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
15577 .collect::<Vec<_>>();
15578 let chunk = Chunk::new(vec![
15579 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
15580 ])
15581 .expect("matching rows");
15582 writer.append(&chunk).expect("a part");
15583 }
15584 writer.finish().expect("commit");
15585
15586 let reader = Reader::open(&path).expect("reopen from disk");
15587 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15588 let count = dictionary.ranks().expect("a ranked dictionary");
15589 assert_eq!(count, distinct as usize, "every distinct value has a rank");
15590 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
15591 let ranked = (0..count)
15592 .map(|rank| {
15593 let code = dictionary.code_at_rank(rank).expect("a code");
15594 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
15595 })
15596 .collect::<Vec<_>>();
15597 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
15598 expected.sort();
15599 assert_eq!(ranked, expected, "rank order is value order");
15600 fs::remove_file(path).expect("remove scratch file");
15601 }
15602
15603 #[test]
15616 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
15617 let path = path("windowed-directory");
15618 let fields = vec![
15619 Field::required("id", LogicalType::BigInt),
15620 Field::required("word", LogicalType::Varchar),
15621 Field::new("score", LogicalType::Double),
15622 ];
15623 let mut writer = Writer::create(&path, "items", fields).expect("new file");
15624 for part in 0..70_i64 {
15625 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
15626 let words = (0..100)
15627 .map(|row| Value::Varchar(format!("word {}", row % 13)))
15628 .collect::<Vec<_>>();
15629 let scores = (0..100)
15630 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
15631 .collect::<Vec<_>>();
15632 let chunk = Chunk::new(vec![
15633 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
15634 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
15635 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
15636 ])
15637 .expect("three columns");
15638 writer.append(&chunk).expect("a part");
15639 }
15640 writer.finish().expect("commit");
15641
15642 let catalog = Catalog::open(&path).expect("reopen");
15643 let entry = catalog.entries.first().expect("one table").directory;
15644 let (offset, length) = (entry.offset, entry.length as usize);
15645 let mut bytes = vec![0; length];
15646 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
15647 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
15648 let whole = decode_directory(&bytes, catalog.size).expect("whole");
15649 assert!(whole.stripes.len() > 1, "the table should span stripes");
15650 for size in [1, 7, 33, 4_096] {
15651 let mut cursor = Cursor::over(&catalog.file, offset, length);
15652 cursor.window.as_mut().expect("a window").size = size;
15653 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
15654 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
15655 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
15656 let mut stored = 0;
15657 for (column, (left, held)) in
15658 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
15659 {
15660 match (left, held) {
15661 (None, None) => {}
15662 (
15663 Some(super::Frequencies::Stored { span, values }),
15664 Some(super::Frequencies::Held(summary)),
15665 ) => {
15666 let mut one = vec![0; span.length as usize];
15667 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
15668 let read = decode_summary(
15669 &mut Cursor::new(&one),
15670 &whole.fields[column],
15671 whole.rows,
15672 *values,
15673 )
15674 .expect("a valid synopsis")
15675 .expect("one is there");
15676 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
15677 stored += 1;
15678 }
15679 other => panic!("column {column} came back as {other:?}"),
15680 }
15681 }
15682 assert!(stored >= 2, "only {stored} synopses were left in the file");
15683 }
15684 let reader = catalog.table("items").expect("the table");
15685 assert!(reader.frequency_summaries[1].get().is_none());
15686 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
15687 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
15688 let clone = reader.clone();
15689 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
15690 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
15691 fs::remove_file(path).expect("remove scratch file");
15692 }
15693
15694 #[test]
15695 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
15696 let path = path("file-checksum");
15697 let bytes = (0..200_000_u32)
15698 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
15699 .collect::<Vec<_>>();
15700 fs::write(&path, &bytes).expect("scratch file");
15701 let file = File::open(&path).expect("open");
15702 for (offset, length) in [
15703 (0, 0),
15704 (3, 1),
15705 (5, 31),
15706 (0, 32),
15707 (9, 33),
15708 (1, 65_536),
15709 (7, 65_567),
15710 (0, 200_000),
15711 (11, 131_101),
15712 ] {
15713 let whole = checksum(&bytes[offset..offset + length]);
15714 assert_eq!(
15715 file_checksum(&file, offset as u64, length).expect("read"),
15716 whole,
15717 "{offset} {length}"
15718 );
15719 }
15720 fs::remove_file(path).expect("remove scratch file");
15721 }
15722
15723 #[test]
15724 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
15725 let path = path("synopsis-keeps-no-block");
15726 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
15727 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
15728 for _ in 0..3 {
15729 values.extend((0..3_000).step_by(5).map(spelled));
15730 }
15731 let mut writer =
15732 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15733 .expect("new file");
15734 for part in values.chunks(1_024) {
15735 writer
15736 .append(
15737 &Chunk::new(vec![
15738 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15739 ])
15740 .expect("one column"),
15741 )
15742 .expect("a part");
15743 }
15744 writer.finish().expect("commit");
15745
15746 let reader = Reader::open(&path).expect("reopen from disk");
15747 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15748 let resting = dictionary.footprint();
15749 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15750 assert_eq!(prefix.entries.len(), 512);
15751 for (value, count) in &prefix.entries {
15752 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
15753 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
15754 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
15755 }
15756 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
15757 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
15758 assert_eq!(again.entries, prefix.entries);
15759 fs::remove_file(path).expect("remove scratch file");
15760 }
15761
15762 #[test]
15769 fn character_lengths_are_counted_without_keeping_the_dictionary_blocks() {
15770 let path = path("character-lengths");
15771 let spellings = (0..2_500)
15772 .map(|index| Value::Varchar(format!("héllo {index:05} {}", "ü".repeat(index % 30))))
15773 .collect::<Vec<_>>();
15774 let mut writer =
15775 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
15776 .expect("new file");
15777 for part in spellings.chunks(1_024) {
15778 writer
15779 .append(
15780 &Chunk::new(vec![
15781 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15782 ])
15783 .expect("one column"),
15784 )
15785 .expect("a part");
15786 }
15787 writer.finish().expect("commit");
15788
15789 let reader = Reader::open(&path).expect("reopen from disk");
15790 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15791 let resting = dictionary.footprint();
15792 let mut lens = Vec::new();
15793 assert!(dictionary.try_chars_lens(&mut lens).expect("counted"), "a stored source counts");
15794 let counted = dictionary.footprint() - resting;
15795 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15796 assert!(
15797 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15798 "counting kept {counted} bytes, more than a count a value"
15799 );
15800 let expected = (0..dictionary.len())
15801 .map(|code| {
15802 let bytes = dictionary.try_bytes_at(code).expect("read").expect("a value");
15803 i64::try_from(std::str::from_utf8(bytes).expect("utf-8").chars().count())
15804 .expect("small")
15805 })
15806 .collect::<Vec<_>>();
15807 assert_eq!(lens, expected, "a count is the number of characters, not of bytes");
15808 let mut again = Vec::new();
15809 assert!(dictionary.try_chars_lens(&mut again).expect("counted"));
15810 assert_eq!(again, lens, "the kept counts answer the second time");
15811 fs::remove_file(path).expect("remove scratch file");
15812 }
15813
15814 fn stored_spellings(label: &str, spellings: &[String]) -> (PathBuf, Reader) {
15816 let path = path(label);
15817 let values = spellings.iter().map(|text| Value::Varchar(text.clone())).collect::<Vec<_>>();
15818 let mut writer =
15819 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15820 .expect("new file");
15821 for part in values.chunks(1_024) {
15822 writer
15823 .append(
15824 &Chunk::new(vec![
15825 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15826 ])
15827 .expect("one column"),
15828 )
15829 .expect("a part");
15830 }
15831 writer.finish().expect("commit");
15832 let reader = Reader::open(&path).expect("reopen from disk");
15833 (path, reader)
15834 }
15835
15836 fn scattered_rows(len: usize) -> (Vec<u32>, Vec<bool>) {
15842 let codes = (0..len)
15843 .map(|row| u32::try_from(row * 7_919 % len).expect("a small dictionary"))
15844 .collect::<Vec<_>>();
15845 let valid = (0..len).map(|row| row % 7 != 3).collect::<Vec<_>>();
15846 (codes, valid)
15847 }
15848
15849 #[test]
15856 fn character_lengths_with_nulls_are_counted_without_keeping_the_dictionary_blocks() {
15857 let spellings = (0..2_500)
15858 .map(|index| format!("héllo {index:05} {}", "ü".repeat(index % 30)))
15859 .collect::<Vec<_>>();
15860 let (path, reader) = stored_spellings("character-lengths-nulls", &spellings);
15861 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15862 let (codes, valid) = scattered_rows(spellings.len());
15863 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&dictionary))
15864 .expect("every code is inside")
15865 .with_validity(Validity::from_run(&valid));
15866
15867 let resting = dictionary.footprint();
15868 let lens = rudb_kernels::call("length", &[&rows], &LogicalType::BigInt, None)
15869 .expect("length reads");
15870 let counted = dictionary.footprint() - resting;
15871 let blocks = dictionary.len().div_ceil(TEXT_PAYLOAD_VALUES);
15872 assert!(
15873 counted <= blocks * TEXT_PAYLOAD_VALUES * size_of::<u32>(),
15874 "length over a vector with nulls kept {counted} bytes, more than a count a value"
15875 );
15876 let expected = (0..rows.len())
15877 .map(|row| match valid[row] {
15878 true => Value::BigInt(
15879 i64::try_from(spellings[codes[row] as usize].chars().count()).expect("small"),
15880 ),
15881 false => Value::Null,
15882 })
15883 .collect::<Vec<_>>();
15884 let answers = (0..lens.len()).map(|row| lens.value_at(row)).collect::<Vec<_>>();
15885 assert_eq!(answers, expected, "a count of characters where a row has one, null elsewhere");
15886 fs::remove_file(path).expect("remove scratch file");
15887 }
15888
15889 #[test]
15899 fn string_kernels_read_a_stored_dictionary_without_keeping_its_blocks() {
15900 let spellings = (0..2_500)
15901 .map(|index| format!("HéLLo {index:05} {}", "Üß".repeat(index % 30)))
15902 .collect::<Vec<_>>();
15903 let (path, reader) = stored_spellings("string-kernels", &spellings);
15904 let page = reader.table.dictionaries[0].expect("a string column has one");
15905 let starved =
15906 open_global_dictionary(Arc::clone(&reader.file), page, &LogicalType::Varchar, 0)
15907 .expect("a dictionary opens whatever it may keep");
15908 let starved = Arc::new(starved);
15909 let (codes, valid) = scattered_rows(spellings.len());
15910 let rows = Vector::dictionary_over(codes.clone(), Arc::clone(&starved))
15911 .expect("every code is inside")
15912 .with_validity(Validity::from_run(&valid));
15913 let expected = |each: &dyn Fn(&str) -> String| {
15914 (0..rows.len())
15915 .map(|row| match valid[row] {
15916 true => Value::Varchar(each(&spellings[codes[row] as usize])),
15917 false => Value::Null,
15918 })
15919 .collect::<Vec<_>>()
15920 };
15921 let answers =
15922 |vector: &Vector| (0..vector.len()).map(|row| vector.value_at(row)).collect::<Vec<_>>();
15923
15924 let resting = starved.footprint();
15927 let ends = spellings.len() * size_of::<u32>();
15928 let lowered = rudb_kernels::call("lower", &[&rows], &LogicalType::Varchar, None)
15929 .expect("lower reads");
15930 assert_eq!(answers(&lowered), expected(&|text| text.to_lowercase()), "lower");
15931 assert!(starved.footprint() <= resting + ends, "lower kept a block it read");
15932
15933 let start = Vector::constant(LogicalType::BigInt, Value::BigInt(3), rows.len());
15934 let length = Vector::constant(LogicalType::BigInt, Value::BigInt(9), rows.len());
15935 let cut =
15936 rudb_kernels::call("substring", &[&rows, &start, &length], &LogicalType::Varchar, None)
15937 .expect("substring reads");
15938 let cut_of = |text: &str| text.chars().skip(2).take(9).collect::<String>();
15939 assert_eq!(answers(&cut), expected(&cut_of), "substring");
15940 assert!(starved.footprint() <= resting + ends, "substring kept a block it read");
15941
15942 let raised = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
15945 .expect("upper reads");
15946 assert_eq!(answers(&raised), expected(&|text| text.to_uppercase()), "upper");
15947 let payload = spellings.iter().map(String::len).sum::<usize>();
15948 assert!(
15949 starved.footprint() >= resting + payload,
15950 "a visit that has dropped a column's worth of blocks keeps what it reads"
15951 );
15952 let again = rudb_kernels::call("upper", &[&rows], &LogicalType::Varchar, None)
15953 .expect("upper reads kept blocks");
15954 assert_eq!(answers(&again), answers(&raised), "the kept blocks answer the same");
15955 fs::remove_file(path).expect("remove scratch file");
15956 }
15957
15958 #[test]
15968 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
15969 let path = path("dictionary-sweep");
15970 let spellings = (0..2_500)
15973 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
15974 .collect::<Vec<_>>();
15975 let mut writer =
15976 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15977 .expect("new file");
15978 for part in spellings.chunks(1_024) {
15981 writer
15982 .append(
15983 &Chunk::new(vec![
15984 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
15985 ])
15986 .expect("one column"),
15987 )
15988 .expect("stripe written");
15989 }
15990 writer.finish().expect("commit");
15991
15992 let reader = Reader::open(&path).expect("valid directory");
15993 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
15994 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
15995 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
15996 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
15997 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
15998 }
15999
16000 let resting = dictionary.footprint();
16001 let sweep = || {
16002 let mut swept: Vec<Vec<u8>> = Vec::new();
16003 let mut at = 0;
16004 let mut calls = 0;
16005 while at < dictionary.len() {
16006 let stopped = dictionary
16007 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16008 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16009 swept.push(text.to_vec());
16010 Ok(())
16011 })
16012 .expect("a sweep reads");
16013 assert!(stopped > at, "a sweep moves");
16014 at = stopped;
16015 calls += 1;
16016 }
16017 assert_eq!(calls, 3, "a sweep hands over one block at a time");
16018 swept
16019 };
16020 let swept = sweep();
16021 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
16022 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
16023 let after = dictionary.footprint();
16024 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
16025
16026 let read = (0..dictionary.len())
16027 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16028 .collect::<Vec<_>>();
16029 assert_eq!(swept, read, "a sweep answers what a point read answers");
16030 let grown = dictionary.footprint() - after;
16034 assert!(
16035 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
16036 "a point read of a kept block decodes nothing, and {grown} bytes grew"
16037 );
16038 fs::remove_file(path).expect("remove scratch file");
16039 }
16040
16041 #[test]
16042 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
16043 let path = path("narrow-substring-signature");
16044 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
16045 let mut grams = Vec::new();
16046 for text in blocks {
16047 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
16048 for gram in text.windows(4) {
16049 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
16050 bits[bit / 8] |= 1 << (bit % 8);
16051 }
16052 }
16053 grams.extend(bits);
16054 }
16055 fs::write(&path, &grams).expect("scratch file");
16056 let file = File::open(&path).expect("open scratch file");
16057 let signatures = NativeGrams {
16058 start: 0,
16059 length: grams.len(),
16060 width: NARROW_GRAM_BYTES,
16061 hash: checksum(&grams),
16062 verdicts: Mutex::new(Vec::new()),
16063 };
16064 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
16065 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
16066 assert!(signatures.footprint() > 0, "a verdict is remembered");
16067 let again = signatures.verdicts(&file, b"google").expect("remembered");
16068 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
16069
16070 let damaged = NativeGrams {
16071 hash: signatures.hash ^ 1,
16072 verdicts: Mutex::new(Vec::new()),
16073 ..signatures
16074 };
16075 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
16076 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
16077 fs::remove_file(path).expect("remove scratch file");
16078 }
16079
16080 #[test]
16081 fn a_damaged_substring_signature_is_checked_only_when_used() {
16082 let path = path("damaged-substring-signature");
16083 let mut writer =
16084 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16085 .expect("new file");
16086 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
16087 writer
16088 .append(
16089 &Chunk::new(vec![
16090 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
16091 ])
16092 .expect("one column"),
16093 )
16094 .expect("stripe written");
16095 writer.finish().expect("commit");
16096
16097 let reader = Reader::open(&path).expect("valid directory");
16098 let page = reader.table.dictionaries[0].expect("string dictionary page");
16099 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
16100 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
16101 .expect("last signature byte");
16102 file.write_all(&[255]).expect("damage signature");
16103 let reader = Reader::open(&path).expect("the directory is still valid");
16104 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
16105 let error = dictionary
16106 .text_block_might_contain(0, b"goog")
16107 .expect_err("a used signature checks its own checksum");
16108 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
16109 fs::remove_file(path).expect("remove scratch file");
16110 }
16111
16112 #[test]
16123 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
16124 let path = path("dictionary-sweep-short-run");
16125 let spellings = (0..2_800)
16126 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16127 .collect::<Vec<_>>();
16128 let mut writer =
16129 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16130 .expect("new file");
16131 for part in spellings.chunks(1_024) {
16132 writer
16133 .append(
16134 &Chunk::new(vec![
16135 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16136 ])
16137 .expect("one column"),
16138 )
16139 .expect("stripe written");
16140 }
16141 writer.finish().expect("commit");
16142
16143 let reader = Reader::open(&path).expect("valid directory");
16144 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16145 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16146 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
16147 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
16148 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
16149
16150 let mut swept: Vec<Vec<u8>> = Vec::new();
16151 let mut at = 0;
16152 while at < dictionary.len() {
16153 let stopped = dictionary
16154 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
16155 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
16156 swept.push(text.to_vec());
16157 Ok(())
16158 })
16159 .expect("a sweep reads");
16160 assert!(stopped > at, "a sweep moves");
16161 at = stopped;
16162 }
16163 let read = (0..dictionary.len())
16164 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
16165 .collect::<Vec<_>>();
16166 assert_eq!(swept, read, "a sweep answers what a point read answers");
16167 fs::remove_file(path).expect("remove scratch file");
16168 }
16169
16170 #[test]
16179 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
16180 let path = path("dictionary-unpacked-ends");
16181 let spellings = (0..2_800)
16182 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
16183 .collect::<Vec<_>>();
16184 let mut writer =
16185 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16186 .expect("new file");
16187 for part in spellings.chunks(1_024) {
16188 writer
16189 .append(
16190 &Chunk::new(vec![
16191 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16192 ])
16193 .expect("one column"),
16194 )
16195 .expect("stripe written");
16196 }
16197 writer.finish().expect("commit");
16198
16199 let reader = Reader::open(&path).expect("valid directory");
16200 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
16201 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
16202 let wanted = (0..spellings.len())
16203 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
16204 .collect::<Vec<_>>();
16205
16206 let pass = |what: &str| {
16207 for (index, value) in wanted.iter().enumerate() {
16208 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
16209 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
16210 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
16211 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
16212 }
16213 };
16214 pass("the first pass");
16215 pass("the second pass");
16216
16217 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
16221 let mut whole = vec![0i64; wanted.len()];
16222 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
16223 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
16224 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
16225 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
16226 let mut through = vec![0i64; codes.len()];
16227 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
16228 for (row, &code) in codes.iter().enumerate() {
16229 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
16230 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
16231 assert_eq!(through[row], one as i64, "row {row} a row at a time");
16232 }
16233
16234 let fresh = Reader::open(&path).expect("valid directory");
16237 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
16238 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
16239 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
16240 let mut short = vec![0i64; few.len()];
16241 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
16242 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
16243 assert_eq!(short, expected, "the packed ends answer what the table answers");
16244 fs::remove_file(path).expect("remove scratch file");
16245 }
16246
16247 #[test]
16257 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
16258 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
16259 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
16260 fit::<i8>(&[128]).expect_err("one past the top does not fit");
16261 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
16262 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
16263 fit::<u8>(&[256]).expect_err("one past the top does not fit");
16264 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
16265 assert_eq!(
16266 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
16267 vec![-32_768_i16, 0, 32_767]
16268 );
16269 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
16270 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
16271 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
16272 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
16273 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
16274 assert_eq!(
16275 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
16276 vec![i32::MIN, 0, i32::MAX]
16277 );
16278 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
16279 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
16280 assert_eq!(
16281 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
16282 vec![0_u32, 4_294_967_295]
16283 );
16284 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
16285 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
16286
16287 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
16290 }
16291
16292 #[test]
16299 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
16300 for value in -70_000_i64..70_000 {
16301 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
16302 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
16303 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
16304 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
16305 }
16306 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
16307 for edge in wide {
16308 for step in -2_i64..=2 {
16309 let value = edge.saturating_add(step);
16310 assert_eq!(
16311 fit::<i32>(&[value]).is_ok(),
16312 i32::try_from(value).is_ok(),
16313 "{value} as i32"
16314 );
16315 assert_eq!(
16316 fit::<u32>(&[value]).is_ok(),
16317 u32::try_from(value).is_ok(),
16318 "{value} as u32"
16319 );
16320 }
16321 }
16322 }
16323
16324 #[test]
16339 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
16340 let spellings = (0..3_000)
16341 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
16342 .collect::<Vec<_>>();
16343 let mut read = Vec::new();
16344 for layout in ["outside", "inside", "behind"] {
16345 let mut dictionary = GlobalDictionary::new();
16346 for text in &spellings {
16347 dictionary.code(text).expect("a code for every spelling");
16348 }
16349 dictionary.finish_blocks().expect("the last block encodes");
16350 let order = dictionary.ranked(None).expect("a sorted order");
16351 let laid = |from: u64| {
16353 let mut at = from;
16354 dictionary
16355 .blocks
16356 .iter()
16357 .map(|block| {
16358 let place =
16359 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
16360 at += block.len() as u64;
16361 place
16362 })
16363 .collect::<Vec<_>>()
16364 };
16365 let payload = dictionary.blocks.concat();
16366 let scattered = layout != "behind";
16367 let (bytes, encoded, offset, length) = if layout == "outside" {
16368 let mut bytes = vec![0; HEADER as usize];
16369 bytes.extend_from_slice(&payload);
16370 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
16371 .expect("an encoding");
16372 let offset = bytes.len() as u64;
16373 bytes.extend_from_slice(&encoded.index);
16374 bytes.extend_from_slice(&encoded.ranks);
16375 bytes.extend_from_slice(&encoded.grams);
16376 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
16377 (bytes, encoded, offset, length)
16378 } else {
16379 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
16382 .expect("an encoding");
16383 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
16384 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
16385 .expect("an encoding");
16386 let mut bytes = encoded.index.clone();
16387 bytes.extend_from_slice(&encoded.ranks);
16388 bytes.extend_from_slice(&encoded.grams);
16389 bytes.extend_from_slice(&payload);
16390 let length = bytes.len();
16391 (bytes, encoded, 0, length)
16392 };
16393 let path = path(&format!("blocks-{layout}"));
16394 fs::write(&path, &bytes).expect("the dictionary is written on its own");
16395 let file = Arc::new(File::open(&path).expect("it opens again"));
16396 let page = Page {
16397 offset,
16398 length: u32::try_from(length).expect("a test dictionary is small"),
16399 hash: checksum(&encoded.index),
16400 };
16401 let opened =
16402 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
16403 .expect("a dictionary laid out either way opens");
16404 let mut swept: Vec<Vec<u8>> = Vec::new();
16405 let mut at = 0;
16406 while at < opened.len() {
16407 at = opened
16408 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
16409 swept.push(text.to_vec());
16410 Ok(())
16411 })
16412 .expect("a sweep reads");
16413 }
16414 fs::remove_file(&path).expect("clean up");
16415 read.push(swept);
16416 }
16417 let wanted =
16418 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
16419 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
16420 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
16421 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
16422 }
16423
16424 #[test]
16432 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
16433 let path = path("dictionary-budget");
16434 let spellings = (0..2_500)
16435 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
16436 .collect::<Vec<_>>();
16437 let mut writer =
16438 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16439 .expect("new file");
16440 for part in spellings.chunks(1_024) {
16441 writer
16442 .append(
16443 &Chunk::new(vec![
16444 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
16445 ])
16446 .expect("one column"),
16447 )
16448 .expect("stripe written");
16449 }
16450 writer.finish().expect("commit");
16451
16452 let reader = Reader::open(&path).expect("valid directory");
16453 let page = reader.table.dictionaries[0].expect("a string column has one");
16454 let file = Arc::clone(&reader.file);
16455 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
16456 .expect("a dictionary opens whatever it may keep");
16457
16458 let resting = starved.footprint();
16459 let mut swept: Vec<Vec<u8>> = Vec::new();
16460 let mut at = 0;
16461 while at < starved.len() {
16462 at = starved
16463 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
16464 swept.push(text.to_vec());
16465 Ok(())
16466 })
16467 .expect("a sweep reads");
16468 }
16469 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
16470 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
16471
16472 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
16473 let read = (0..generous.len())
16474 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
16475 .collect::<Vec<_>>();
16476 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
16477 fs::remove_file(path).expect("remove scratch file");
16478 }
16479
16480 #[test]
16481 fn damaged_membership_cannot_skip_a_string_page() {
16482 let path = path("damaged-membership");
16483 let mut writer = Writer::create(
16484 &path,
16485 "items",
16486 vec![
16487 Field::required("id", LogicalType::Integer),
16488 Field::new("text", LogicalType::Varchar),
16489 ],
16490 )
16491 .expect("new file");
16492 writer.append(&sample()).expect("stripe written");
16493 writer.finish().expect("commit");
16494
16495 let reader = Reader::open(&path).expect("valid directory");
16496 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
16497 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
16498 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
16499 file.write_all(&[255]).expect("damage membership");
16500 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
16501 assert!(error.message().contains("membership page checksum differs"), "{error}");
16502 fs::remove_file(path).expect("remove scratch file");
16503 }
16504
16505 #[test]
16506 fn membership_delta_stream_is_sorted_exact_and_bounded() {
16507 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
16508 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
16509 let encoded = encode_membership(&unique);
16510 assert_eq!(
16511 decode_membership(&encoded).expect("valid membership"),
16512 [4, 9, 72, 900, u32::MAX]
16513 );
16514 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
16517 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
16518 assert_eq!(
16519 decode_membership(&encode_membership(&merged)).expect("valid membership"),
16520 unique
16521 );
16522 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
16523 assert!(
16524 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
16525 "a value past u32 is invalid"
16526 );
16527 }
16528
16529 #[test]
16530 fn a_global_dictionary_may_be_larger_than_one_column_page() {
16531 let dictionary = Page {
16532 offset: HEADER,
16533 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
16534 hash: 0,
16535 };
16536 let table = Table {
16537 name: "items".to_owned(),
16538 fields: vec![Field::new("text", LogicalType::Varchar)],
16539 stripes: Vec::new(),
16540 rows: 0,
16541 dictionaries: vec![Some(dictionary)],
16542 dictionary_payloads: Vec::new(),
16543 demoted: Vec::new(),
16544 distincts: vec![None],
16545 frequencies: vec![None],
16546 pair_frequencies: Vec::new(),
16547 frequency_texts: Vec::new(),
16548 host_groups: None,
16549 clustering: None,
16550 generation: 1,
16551 sections: Vec::new(),
16552 };
16553 let directory = encode_directory(&table).expect("directory");
16554 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
16555
16556 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
16557 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
16558 }
16559
16560 #[test]
16561 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
16562 let path = path("constant-codes");
16563 let mut writer =
16564 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
16565 .expect("new file");
16566 let empty = vec![Value::Varchar(String::new()); 1024];
16567 for _ in 0..4 {
16568 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
16569 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
16570 }
16571 writer.finish().expect("commit");
16572
16573 let reader = Reader::open(&path).expect("valid directory");
16574 let pages = reader.layout().columns.first().expect("one column").pages;
16575 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
16579 let read = reader.read(3, &[0]).expect("the last part back");
16580 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
16581 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
16582 fs::remove_file(path).expect("remove scratch file");
16583 }
16584
16585 #[test]
16586 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
16587 let over = vec![i64::from(i32::MAX) + 1];
16590 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
16591 assert!(format!("{error}").contains("not of its type"), "{error}");
16592 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
16593 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
16594 }
16595
16596 #[test]
16597 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
16598 let mut state: u32 = 0x9e37_79b9;
16602 let spread: Vec<u32> = (0..1024)
16603 .map(|_| {
16604 state ^= state << 13;
16605 state ^= state >> 17;
16606 state ^= state << 5;
16607 state
16608 })
16609 .collect();
16610 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
16611 let near: Vec<u32> = (0..1024).collect();
16612 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
16613 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
16614 }
16615
16616 #[test]
16622 fn two_writes_of_the_same_rows_give_the_same_bytes() {
16623 fn written(path: &PathBuf) {
16624 let fields = (0..40)
16625 .map(|column| {
16626 let ty =
16627 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
16628 Field::new(format!("c{column}"), ty)
16629 })
16630 .collect::<Vec<_>>();
16631 let mut writer = Writer::create(path, "wide", fields).expect("new file");
16632 for part in 0..70_u64 {
16633 let columns = (0..40)
16634 .map(|column| {
16635 let values = (0..64_u64)
16636 .map(|row| {
16637 let seed = part.wrapping_mul(31).wrapping_add(row);
16638 if column % 4 == 0 {
16639 Value::Varchar(format!("v{}", seed % 17))
16640 } else {
16641 Value::BigInt(i64::try_from(seed % 97).expect("small"))
16642 }
16643 })
16644 .collect::<Vec<_>>();
16645 let ty = if column % 4 == 0 {
16646 LogicalType::Varchar
16647 } else {
16648 LogicalType::BigInt
16649 };
16650 Vector::from_values(ty, &values).expect("a column")
16651 })
16652 .collect::<Vec<_>>();
16653 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
16654 }
16655 writer.finish().expect("commit");
16656 }
16657
16658 let first = path("repeatable-one");
16659 let second = path("repeatable-two");
16660 written(&first);
16661 written(&second);
16662 let left = fs::read(&first).expect("the first file");
16663 let right = fs::read(&second).expect("the second file");
16664 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
16665 assert!(left == right, "two writes of the same rows differ in their bytes");
16666
16667 let reader = Reader::open(&first).expect("valid directory");
16670 assert_eq!(reader.table().rows(), 70 * 64);
16671 let read = reader.read(0, &[0, 1]).expect("the first part back");
16672 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
16673 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
16674 fs::remove_file(first).expect("remove scratch file");
16675 fs::remove_file(second).expect("remove scratch file");
16676 }
16677
16678 fn three_tables(path: &PathBuf) {
16680 let writer = Writer::create(
16681 path,
16682 "region",
16683 vec![
16684 Field::new("r_key", LogicalType::Integer),
16685 Field::new("r_name", LogicalType::Varchar),
16686 ],
16687 )
16688 .expect("new file");
16689 let mut writer = writer;
16690 writer
16691 .append(
16692 &Chunk::new(vec![
16693 Vector::from_values(
16694 LogicalType::Integer,
16695 &[Value::Integer(0), Value::Integer(1)],
16696 )
16697 .expect("keys"),
16698 Vector::from_values(
16699 LogicalType::Varchar,
16700 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
16701 )
16702 .expect("names"),
16703 ])
16704 .expect("two columns"),
16705 )
16706 .expect("a part");
16707 let mut writer = writer
16708 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
16709 .expect("a second table");
16710 writer
16711 .append(
16712 &Chunk::new(vec![
16713 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
16714 ])
16715 .expect("one column"),
16716 )
16717 .expect("a part");
16718 let mut writer =
16719 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
16720 for part in 0..70_i64 {
16721 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
16722 writer
16723 .append(
16724 &Chunk::new(vec![
16725 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
16726 ])
16727 .expect("one column"),
16728 )
16729 .expect("a part");
16730 }
16731 writer.finish().expect("commit");
16732 }
16733
16734 #[test]
16735 fn three_tables_in_one_file_read_back_by_name() {
16736 let file = path("three-tables");
16737 three_tables(&file);
16738 let catalog = Catalog::open(&file).expect("a committed catalog");
16739 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
16740
16741 let region = catalog.table("region").expect("the first table");
16742 assert_eq!(region.table().rows(), 2);
16743 assert_eq!(
16744 region.read(0, &[1]).expect("names").value_at(1, 0),
16745 Value::Varchar("ASIA".to_owned())
16746 );
16747
16748 let wide = catalog.table("wide").expect("the third table");
16749 assert_eq!(wide.table().rows(), 70 * 64);
16750 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
16751
16752 let empty = catalog.table("empty").expect("the second table");
16755 assert_eq!(empty.table().rows(), 1);
16756 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
16757
16758 fs::remove_file(file).expect("remove scratch file");
16759 }
16760
16761 #[test]
16762 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
16763 let file = path("three-tables-missing");
16764 three_tables(&file);
16765 let catalog = Catalog::open(&file).expect("a committed catalog");
16766 let error = catalog.table("nation").expect_err("no such table");
16767 assert!(error.message().contains("nation"), "{}", error.message());
16768 fs::remove_file(file).expect("remove scratch file");
16769 }
16770
16771 #[test]
16772 fn a_file_of_three_tables_will_not_open_as_one() {
16773 let file = path("three-tables-unnamed");
16774 three_tables(&file);
16775 let error = Reader::open(&file).expect_err("more than one table");
16776 assert!(error.message().contains("more than one table"), "{}", error.message());
16777 fs::remove_file(file).expect("remove scratch file");
16778 }
16779
16780 #[test]
16782 fn decimals_of_every_storage_width_round_trip() {
16783 let file = path("decimals");
16784 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
16785 let fields = widths
16786 .iter()
16787 .enumerate()
16788 .map(|(index, (width, scale))| {
16789 Field::new(
16790 format!("d{index}"),
16791 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16792 )
16793 })
16794 .collect::<Vec<_>>();
16795 let mut writer = Writer::create(&file, "money", fields).expect("new file");
16796 let rows: [i128; 3] = [-1234, 0, 999];
16797 let columns = widths
16798 .iter()
16799 .map(|(width, scale)| {
16800 let values = rows
16801 .iter()
16802 .map(|unscaled| Value::Decimal {
16803 unscaled: *unscaled,
16804 width: *width,
16805 scale: *scale,
16806 })
16807 .collect::<Vec<_>>();
16808 Vector::from_values(
16809 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16810 &values,
16811 )
16812 .expect("a decimal column")
16813 })
16814 .collect::<Vec<_>>();
16815 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
16816 writer.finish().expect("commit");
16817
16818 let reader = Reader::open(&file).expect("a committed file");
16819 for (index, (width, scale)) in widths.iter().enumerate() {
16820 assert_eq!(
16821 reader.table().fields()[index].ty,
16822 LogicalType::decimal(*width, *scale).expect("a decimal type"),
16823 "column {index} came back as another type"
16824 );
16825 let column = reader.read(0, &[index]).expect("the column");
16826 for (row, unscaled) in rows.iter().enumerate() {
16827 assert_eq!(
16828 column.value_at(row, 0),
16829 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
16830 "column {index} row {row}"
16831 );
16832 }
16833 }
16834 fs::remove_file(file).expect("remove scratch file");
16835 }
16836
16837 #[test]
16838 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
16839 let file = path("two-of-a-name");
16840 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
16841 .expect("new file");
16842 let error = writer
16843 .next("t", vec![Field::new("a", LogicalType::BigInt)])
16844 .expect_err("the same name twice");
16845 assert!(error.message().contains("same name"), "{}", error.message());
16846 fs::remove_file(file).expect("remove scratch file");
16847 }
16848
16849 #[test]
16850 fn integer_tally_counts_encoded_rows_and_declines_null_parts() {
16851 let file = path("integer-tally");
16852 let mut writer =
16853 Writer::create(&file, "events", vec![Field::new("source", LogicalType::SmallInt)])
16854 .expect("new file");
16855 let mut values = vec![Value::SmallInt(0); 1024];
16856 values[7] = Value::SmallInt(3);
16857 values[99] = Value::SmallInt(-2);
16858 values[1001] = Value::SmallInt(3);
16859 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("integer values");
16860 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("first part");
16861 values[0] = Value::Null;
16862 let column = Vector::from_values(LogicalType::SmallInt, &values).expect("nullable values");
16863 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("second part");
16864 writer.finish().expect("commit");
16865
16866 let reader = Reader::open(&file).expect("read file");
16867 assert_eq!(
16868 reader.integer_tally(0, 0).expect("valid part"),
16869 Some(vec![(-2, 1), (0, 1021), (3, 2)])
16870 );
16871 assert!(reader.integer_tally(1, 0).expect("valid null part").is_none());
16872 let catalog = Catalog::open(&file).expect("catalog");
16873 assert_eq!(
16874 catalog.integer_tally("events", 0).expect("nullable column"),
16875 Some(vec![(-2, 2), (0, 2041), (3, 4)])
16876 );
16877 fs::remove_file(file).expect("remove scratch file");
16878 }
16879
16880 #[test]
16881 fn catalog_tallies_one_integer_column_without_opening_the_whole_table() {
16882 let file = path("catalog-integer-tally");
16883 let mut writer = Writer::create(
16884 &file,
16885 "events",
16886 vec![
16887 Field::new("noise", LogicalType::SmallInt),
16888 Field::new("source", LogicalType::SmallInt),
16889 ],
16890 )
16891 .expect("new file");
16892 let noise = vec![Value::SmallInt(9); 1024];
16893 let mut source = vec![Value::SmallInt(0); 1024];
16894 source[7] = Value::SmallInt(3);
16895 source[99] = Value::SmallInt(-2);
16896 let chunk = Chunk::new(vec![
16897 Vector::from_values(LogicalType::SmallInt, &noise).expect("noise"),
16898 Vector::from_values(LogicalType::SmallInt, &source).expect("source"),
16899 ])
16900 .expect("two columns");
16901 writer.append(&chunk).expect("append");
16902 writer.finish().expect("commit");
16903
16904 let catalog = Catalog::open(&file).expect("catalog");
16905 assert_eq!(
16906 catalog.integer_tally("events", 1).expect("selected column"),
16907 Some(vec![(-2, 1), (0, 1022), (3, 1)])
16908 );
16909 assert_eq!(
16910 catalog.integer_tally("events", 0).expect("other column"),
16911 Some(vec![(9, 1024)])
16912 );
16913 fs::remove_file(file).expect("remove scratch file");
16914 }
16915
16916 #[test]
16917 fn opening_the_catalog_reads_no_table_directory() {
16918 let file = path("catalog-only");
16919 three_tables(&file);
16920 let catalog = Catalog::open(&file).expect("a committed catalog");
16921 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
16924 assert_eq!(catalog.names().len(), 3);
16925 fs::remove_file(file).expect("remove scratch file");
16926 }
16927
16928 #[test]
16939 fn the_checksum_answers_what_it_has_always_answered() {
16940 let bytes: Vec<u8> =
16941 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
16942 for (length, expected) in [
16943 (0, 0xef46_db37_51d8_e999),
16944 (1, 0xa96c_7f0c_e858_bbb7),
16945 (3, 0x56e6_9576_32a4_87f9),
16946 (4, 0xc60d_15b1_e3ff_8f04),
16947 (5, 0x8088_1585_8624_dd4e),
16948 (7, 0xafbe_fc3d_6c6f_9a8e),
16949 (8, 0x3da5_c7aa_2696_83e0),
16950 (9, 0x465e_c429_b13c_3892),
16951 (15, 0xdee8_9d8a_065a_6233),
16952 (16, 0x1330_489a_7767_9c80),
16953 (31, 0x3391_303d_485e_846e),
16954 (32, 0x40b7_aff7_5d45_bbc8),
16955 (33, 0x4997_cae4_951c_17a5),
16956 (39, 0x5807_28fd_5c14_5739),
16957 (40, 0xf95c_f6f5_c08a_3d3b),
16958 (63, 0x2944_b4da_fc69_b206),
16959 (64, 0xbb76_f6ef_19bd_5a1b),
16960 (65, 0x814e_0c65_4a9f_d640),
16961 (127, 0x00de_aab1_31cf_f89b),
16962 (1000, 0x9e33_00c1_cde3_c58d),
16963 ] {
16964 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
16965 }
16966 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
16967 }
16968 #[test]
16975 fn a_declared_order_comes_back_out_of_the_file() {
16976 let path = path("clustered");
16977 let shipped = vec![
16978 Field::new("key", LogicalType::BigInt),
16979 Field::new("line", LogicalType::Integer),
16980 Field::new("shipdate", LogicalType::Date),
16981 ];
16982 let plain = vec![Field::new("a", LogicalType::Integer)];
16983 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
16984
16985 let mut writer = Writer::create(&path, "lineitem", shipped)
16986 .expect("new file")
16987 .declare(stage_zero.clone())
16988 .expect("the columns are the table's");
16989 let column = |ty: LogicalType, values: &[Value]| {
16990 Vector::from_values(ty, values).expect("the values match the type")
16991 };
16992 writer
16993 .append(
16994 &Chunk::new(vec![
16995 column(
16996 LogicalType::BigInt,
16997 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
16998 ),
16999 column(
17000 LogicalType::Integer,
17001 &[
17002 Value::Integer(1),
17003 Value::Integer(1),
17004 Value::Integer(1),
17005 Value::Integer(1),
17006 ],
17007 ),
17008 column(
17009 LogicalType::Date,
17010 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
17011 ),
17012 ])
17013 .expect("three columns"),
17014 )
17015 .expect("four rows");
17016 let mut writer = writer.next("nation", plain).expect("a second table");
17017 writer
17018 .append(
17019 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
17020 .expect("one column"),
17021 )
17022 .expect("one row");
17023 writer.finish().expect("commit");
17024
17025 let catalog = Catalog::open(&path).expect("reopen");
17026 let lineitem = catalog.table("lineitem").expect("the clustered table");
17027 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
17028 let nation = catalog.table("nation").expect("the plain table");
17029 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
17030
17031 assert_eq!(lineitem.table().rows(), 4);
17034 assert_eq!(nation.table().rows(), 1);
17035 fs::remove_file(&path).ok();
17036 }
17037
17038 #[test]
17040 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
17041 let path = path("clustered-bad");
17042 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
17043 .expect("new file");
17044 let four =
17045 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
17046 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
17047 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
17048 fs::remove_file(&path).ok();
17049 }
17050
17051 #[test]
17057 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
17058 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
17059 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
17060 .collect::<Vec<_>>();
17061 let filled = || {
17062 let mut dictionary = GlobalDictionary::new();
17063 for value in &values {
17064 dictionary.code(value).expect("a code for every value");
17065 }
17066 dictionary.settle().expect("a shape");
17067 dictionary
17068 };
17069 let mut in_place = filled();
17070 in_place.finish_blocks().expect("every block encodes");
17071
17072 let mut handed = filled();
17073 let out = handed.hand_out(3);
17074 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
17075 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
17076 for job in out.iter().rev() {
17077 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
17078 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
17079 }
17080 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
17081 handed.finish_blocks().expect("the last block encodes");
17082
17083 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
17084 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
17085 }
17086
17087 #[test]
17089 fn a_block_given_back_twice_is_refused() {
17090 let mut dictionary = GlobalDictionary::new();
17091 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
17092 dictionary.code(&format!("value {at}")).expect("a code");
17093 }
17094 dictionary.settle().expect("a shape");
17095 let out = dictionary.hand_out(0);
17096 let last = out.last().expect("blocks went out");
17097 let at = last.place().1;
17098 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
17099 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
17100 }
17101
17102 #[test]
17108 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
17109 let mut values = vec![String::new(), "http://".to_owned()];
17110 for host in 0..7 {
17111 for path in 0..30 {
17112 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
17113 values.push(format!("http://example{host}.test/page/{path:04}"));
17114 }
17115 }
17116 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
17117
17118 let mut dictionary = GlobalDictionary::new();
17119 for value in &values {
17120 dictionary.code(value).expect("a code for every value");
17121 }
17122 dictionary.finish_blocks().expect("the last block encodes");
17123 let ranked = dictionary.ranked(None).expect("a sorted order");
17124 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
17125
17126 let spellings = dictionary_values(&dictionary);
17127 let seen = ranked
17128 .iter()
17129 .map(|&(_, code)| {
17130 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
17131 })
17132 .collect::<Vec<_>>();
17133 let mut wanted = values.clone();
17134 wanted.sort_unstable();
17135 assert_eq!(seen, wanted, "the order is the order the bytes give");
17136
17137 for &(carried, code) in &ranked {
17138 let value = &spellings[code as usize];
17139 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
17140 }
17141 }
17142
17143 #[test]
17148 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
17149 let entry =
17150 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
17151 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
17152 .map(|code| entry(code, u64::from(code % 7) + 1))
17153 .collect::<Vec<_>>();
17154 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
17155
17156 let mut sorted = all.clone();
17157 sorted.sort_unstable_by(|left, right| {
17158 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
17159 });
17160 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
17161 sorted.truncate(FREQUENCY_ENTRIES);
17162
17163 let mut picked = all.clone();
17164 let omitted = keep_most_frequent(&mut picked);
17165 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
17166 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
17167 assert!(
17168 picked
17169 .iter()
17170 .zip(&sorted)
17171 .all(|(one, two)| one.value == two.value && one.count == two.count),
17172 "the same entries in the same order"
17173 );
17174
17175 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
17176 let omitted = keep_most_frequent(&mut short);
17177 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
17178 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
17179 }
17180
17181 #[test]
17183 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
17184 let empty = GlobalDictionary::new();
17185 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
17186
17187 let mut dictionary = GlobalDictionary::new();
17188 for value in ["pear", "apple", "", "apples", "app"] {
17189 dictionary.code(value).expect("a code for every value");
17190 }
17191 dictionary.finish_blocks().expect("the one block encodes");
17192 let spellings = dictionary_values(&dictionary);
17193 let seen = dictionary
17194 .ranked(None)
17195 .expect("a sorted order")
17196 .iter()
17197 .map(|&(_, code)| spellings[code as usize].clone())
17198 .collect::<Vec<_>>();
17199 let wanted: Vec<Vec<u8>> =
17200 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
17201 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
17202 }
17203
17204 #[test]
17207 fn a_demoted_dictionary_holds_less_and_takes_no_more_values() {
17208 let profile = LoadProfile::begin("demoted");
17209 let mut dictionary = GlobalDictionary::new();
17210 for value in 0..50_000 {
17211 dictionary.code(&format!("https://example.com/page/{value}")).expect("a code");
17212 }
17213 let (_, grown) = dictionary.recharge(Some(&profile));
17214 assert_eq!(profile.held(), grown, "the profile holds what the dictionary does");
17215
17216 dictionary.demote();
17217 let (before, after) = dictionary.recharge(Some(&profile));
17218 assert_eq!(before, grown);
17219 assert!(after < grown - grown / 4, "the lookup is let go of: {after} of {grown}");
17222 assert_eq!(profile.held(), after, "the profile was told about the drop");
17223 assert!(dictionary.code("one more").is_err(), "a demoted dictionary takes no values");
17224
17225 dictionary.demote();
17226 assert_eq!(
17227 dictionary.recharge(Some(&profile)),
17228 (after, after),
17229 "demoting twice is a no-op"
17230 );
17231 assert_eq!(dictionary.values(), 50_000, "the values coded before stay");
17232 }
17233}