1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 28;
79
80const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, FORMAT];
109
110const HEADER: u64 = 80;
111const SLOT_BYTES: usize = 28;
112const MAX_PAGE: usize = 256 * 1024 * 1024;
113const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
114const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
115const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
116const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
124const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
126const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
132const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
147const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
155const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
163
164const MAX_SECTIONS: usize = 4096;
171const FREQUENCY_CANDIDATES: usize = 32_768;
172const FREQUENCY_ENTRIES: usize = 512;
173const FREQUENCY_BUILD_RANK: usize = 10;
174const FREQUENCY_ORDINALS: usize = 131_072;
175const MAX_PAIR_FREQUENCIES: usize = 1024;
176const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
181const MAX_FREQUENCY_WORKERS: usize = 32;
188
189fn close_workers() -> usize {
191 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
192}
193
194const CLOSE_DICTIONARY_BYTES: usize = 1 << 30;
204
205const MAX_ENCODE_WORKERS: usize = 32;
212
213const WRITEBACK_STRETCH: u64 = 32 << 20;
221
222const SIEVE_BUDGET: usize = 8 * 1024;
230
231const PART_BOUND_BYTES: usize = 24;
240
241fn io(error: std::io::Error) -> Error {
242 Error::io(error.to_string())
243}
244
245fn invalid(message: &str) -> Error {
246 Error::invalid_input(format!("invalid rudb native file: {message}"))
247}
248
249fn sum(counts: impl Iterator<Item = u64>) -> u64 {
251 counts.fold(0, u64::saturating_add)
252}
253
254fn span_bytes(spans: &[Span], at: usize) -> u64 {
256 spans.get(at).map_or(0, |span| u64::from(span.length))
257}
258
259fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
261 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
262}
263
264fn dictionary_bytes(table: &Table, at: usize) -> u64 {
266 page_bytes(&table.dictionaries, at)
267 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
268}
269
270fn checksum(bytes: &[u8]) -> u64 {
280 seeded_checksum(bytes, 0)
281}
282
283#[must_use]
290pub fn content_name(bytes: &[u8]) -> u128 {
291 let seed = u64::from(FORMAT);
292 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
293}
294
295#[derive(Debug, Clone)]
301pub struct ContentNamer {
302 seeds: [u64; 2],
303 lanes: [[u64; 4]; 2],
304 held: [u8; 32],
305 filled: usize,
306 length: u64,
307}
308
309impl Default for ContentNamer {
310 fn default() -> Self {
311 let seed = u64::from(FORMAT);
312 let seeds = [seed, !seed];
313 let lanes = seeds.map(|seed| {
314 [
315 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
316 seed.wrapping_add(XXH_P2),
317 seed,
318 seed.wrapping_sub(XXH_P1),
319 ]
320 });
321 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
322 }
323}
324
325impl ContentNamer {
326 pub fn update(&mut self, mut bytes: &[u8]) {
328 self.length += bytes.len() as u64;
329 if self.filled > 0 {
330 let take = (32 - self.filled).min(bytes.len());
331 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
332 self.filled += take;
333 bytes = &bytes[take..];
334 if self.filled < 32 {
335 return;
336 }
337 let block = self.held;
338 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
339 self.filled = 0;
340 }
341 let mut blocks = bytes.chunks_exact(32);
342 for block in blocks.by_ref() {
343 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
344 }
345 let rest = blocks.remainder();
346 self.held[..rest.len()].copy_from_slice(rest);
347 self.filled = rest.len();
348 }
349
350 #[must_use]
352 pub fn finish(&self) -> u128 {
353 let rest = &self.held[..self.filled];
354 let [first, second] = [0, 1].map(|at| {
355 if self.length < 32 {
356 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
357 } else {
358 finish_checksum(self.lanes[at], rest, self.length)
359 }
360 });
361 u128::from(first) << 64 | u128::from(second)
362 }
363}
364
365fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
374 let mut blocks = bytes.chunks_exact(32);
377 let rest = blocks.remainder();
378 if bytes.len() < 32 {
379 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
380 }
381 let mut lanes = [
382 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
383 seed.wrapping_add(XXH_P2),
384 seed,
385 seed.wrapping_sub(XXH_P1),
386 ];
387 for block in blocks.by_ref() {
388 checksum_block(&mut lanes, block);
389 }
390 finish_checksum(lanes, rest, bytes.len() as u64)
391}
392
393const XXH_P1: u64 = 11_400_714_785_074_694_791;
394const XXH_P2: u64 = 14_029_467_366_897_019_727;
395const XXH_P3: u64 = 1_609_587_929_392_839_161;
396const XXH_P4: u64 = 9_650_029_242_287_828_579;
397const XXH_P5: u64 = 2_870_177_450_012_600_261;
398
399fn checksum_round(state: u64, word: u64) -> u64 {
400 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
401}
402
403fn checksum_word(chunk: &[u8]) -> u64 {
404 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
405}
406
407fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
409 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
410 *lane = checksum_round(*lane, checksum_word(chunk));
411 }
412}
413
414fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
416 let merge = |state: u64, lane: u64| {
417 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
418 };
419 let [one, two, three, four] = lanes;
420 let combined = one
421 .rotate_left(1)
422 .wrapping_add(two.rotate_left(7))
423 .wrapping_add(three.rotate_left(12))
424 .wrapping_add(four.rotate_left(18));
425 let hash = merge(merge(merge(merge(combined, one), two), three), four);
426 checksum_tail(hash.wrapping_add(length), rest)
427}
428
429fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
431 let mut words = rest.chunks_exact(8);
432 for chunk in words.by_ref() {
433 hash ^= checksum_round(0, checksum_word(chunk));
434 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
435 }
436 rest = words.remainder();
437 if rest.len() >= 4 {
438 let (head, tail) = rest.split_at(4);
439 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
440 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
441 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
442 rest = tail;
443 }
444 for &byte in rest {
445 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
446 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
447 }
448 hash ^= hash >> 33;
449 hash = hash.wrapping_mul(XXH_P2);
450 hash ^= hash >> 29;
451 hash = hash.wrapping_mul(XXH_P3);
452 hash ^ (hash >> 32)
453}
454
455fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
461 if length < 32 {
462 let mut bytes = vec![0; length];
463 read_at(file, offset, &mut bytes)?;
464 return Ok(checksum(&bytes));
465 }
466 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
467 let mut buffer = vec![0; DIRECTORY_WINDOW.min(length)];
468 let mut kept = 0;
469 let mut read = 0;
470 while read < length {
471 let want = (buffer.len() - kept).min(length - read);
472 read_at(file, offset + read as u64, &mut buffer[kept..kept + want])?;
473 read += want;
474 let filled = kept + want;
475 let whole = filled / 32 * 32;
476 for block in buffer[..whole].chunks_exact(32) {
477 checksum_block(&mut lanes, block);
478 }
479 buffer.copy_within(whole..filled, 0);
480 kept = filled - whole;
481 }
482 Ok(finish_checksum(lanes, &buffer[..kept], length as u64))
483}
484
485#[derive(Debug, Clone, Copy)]
486struct Slot {
487 offset: u64,
488 length: u32,
489 generation: u64,
490 hash: u64,
491}
492
493impl Slot {
494 fn bytes(self) -> [u8; SLOT_BYTES] {
495 let mut result = [0; SLOT_BYTES];
496 result[..8].copy_from_slice(&self.offset.to_le_bytes());
497 result[8..12].copy_from_slice(&self.length.to_le_bytes());
498 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
499 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
500 result
501 }
502
503 fn read(bytes: &[u8]) -> Self {
504 Self {
505 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
506 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
507 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
508 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
509 }
510 }
511}
512
513#[derive(Debug, Clone, Copy)]
514struct Page {
515 offset: u64,
516 length: u32,
517 hash: u64,
518}
519
520impl Page {
521 fn bytes(&self) -> u64 {
523 u64::from(self.length)
524 }
525}
526
527#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
528enum FrequencyValue {
529 Null,
530 Integer(i128),
531 Code(u32),
532}
533
534type FrequencyMap<V> = HashMap<u64, V, Spread>;
540
541#[derive(Debug, Default)]
544struct Candidates {
545 counts: FrequencyMap<u32>,
546 nulls: u32,
547 decrements: u64,
548}
549
550impl Candidates {
551 fn add(&mut self, bits: Option<u64>, mut times: u32) {
558 while times > 0 {
559 let held = match bits {
560 Some(bits) => self.counts.get_mut(&bits),
561 None if self.nulls != 0 => Some(&mut self.nulls),
562 None => None,
563 };
564 if let Some(count) = held {
565 *count = count.saturating_add(times);
566 return;
567 }
568 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
569 match bits {
570 Some(bits) => {
571 self.counts.insert(bits, times);
572 }
573 None => self.nulls = times,
574 }
575 return;
576 }
577 self.counts.retain(|_, count| {
578 *count -= 1;
579 *count != 0
580 });
581 self.nulls = self.nulls.saturating_sub(1);
582 self.decrements = self.decrements.saturating_add(1);
583 times -= 1;
584 }
585 }
586}
587
588#[derive(Debug, Default)]
590struct Run {
591 bits: Option<u64>,
592 times: u32,
593}
594
595impl Run {
596 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
598 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
599 self.times += 1;
600 return None;
601 }
602 let ended = self.take();
603 self.bits = bits;
604 self.times = 1;
605 ended
606 }
607
608 fn take(&mut self) -> Option<(Option<u64>, u32)> {
610 let times = std::mem::take(&mut self.times);
611 (times != 0).then_some((self.bits, times))
612 }
613}
614
615#[derive(Debug, Default, Clone, Copy)]
617struct Spread;
618
619impl std::hash::BuildHasher for Spread {
620 type Hasher = SpreadHasher;
621
622 fn build_hasher(&self) -> SpreadHasher {
623 SpreadHasher(0)
624 }
625}
626
627#[derive(Debug)]
634struct SpreadHasher(u64);
635
636impl SpreadHasher {
637 fn mix(&mut self, word: u64) {
638 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
639 self.0 = (product as u64) ^ ((product >> 64) as u64);
640 }
641}
642
643impl std::hash::Hasher for SpreadHasher {
644 fn write(&mut self, bytes: &[u8]) {
645 for part in bytes.chunks(8) {
646 let mut word = [0; 8];
647 word[..part.len()].copy_from_slice(part);
648 self.mix(u64::from_le_bytes(word));
649 }
650 }
651
652 fn write_u32(&mut self, value: u32) {
653 self.mix(u64::from(value));
654 }
655
656 fn write_u64(&mut self, value: u64) {
657 self.mix(value);
658 }
659
660 fn write_i128(&mut self, value: i128) {
661 self.mix(value as u64);
662 self.mix((value >> 64) as u64);
663 }
664
665 fn write_isize(&mut self, value: isize) {
666 self.mix(value as u64);
667 }
668
669 fn finish(&self) -> u64 {
670 self.0
671 }
672}
673
674#[derive(Debug, Clone)]
675struct FrequencyEntry {
676 value: FrequencyValue,
677 count: u64,
678}
679
680#[derive(Debug, Clone)]
685struct FrequencySummary {
686 entries: Vec<FrequencyEntry>,
687 omitted_max: u64,
688 ordinals: Vec<u64>,
689 ordinal_entries: Vec<u16>,
690}
691
692#[derive(Debug, Clone)]
693struct PairFrequencyEntry {
694 first_entry: u16,
695 second: Option<u32>,
696 count: u64,
697}
698
699#[derive(Debug, Clone)]
705struct PairFrequencySummary {
706 first: u16,
707 second: u16,
708 entries: Vec<PairFrequencyEntry>,
709 omitted_max: u64,
710}
711
712#[derive(Debug, Clone)]
720enum Frequencies {
721 Held(FrequencySummary),
722 Stored {
725 span: Span,
726 values: bool,
727 },
728}
729
730#[derive(Debug, Clone)]
735pub struct FrequencyPrefix {
736 pub entries: Vec<(Value, u64)>,
738 pub omitted_max: u64,
740}
741
742#[derive(Debug, Clone, PartialEq)]
744pub struct FrequencyOccurrences {
745 pub omitted_max: u64,
747 pub ordinals: Vec<u64>,
749 pub anchors: Vec<Value>,
751 pub anchor_indices: Vec<u16>,
753}
754
755pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
757
758#[derive(Debug, Clone, Copy, Default)]
765struct Span {
766 offset: u64,
767 length: u32,
768}
769
770#[derive(Debug, Clone, Default)]
778struct Pages {
779 columns: usize,
780 held: Box<[StripePage]>,
781}
782
783#[derive(Debug, Clone, Copy)]
785struct StripePage {
786 offset: u64,
787 hash: u64,
788 length: u32,
789 column: u32,
790}
791
792impl Pages {
793 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
795 let mut held = Vec::with_capacity(slots.iter().flatten().count());
796 for (column, page) in slots.iter().enumerate() {
797 if let Some(page) = page {
798 let column =
799 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
800 held.push(StripePage {
801 offset: page.offset,
802 hash: page.hash,
803 length: page.length,
804 column,
805 });
806 }
807 }
808 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
809 }
810
811 fn get(&self, column: usize) -> Option<Page> {
813 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
814 let placed = self.held[at];
815 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
816 }
817
818 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
820 (0..self.columns).map(|column| self.get(column))
821 }
822
823 fn bytes(&self, column: usize) -> u64 {
825 self.get(column).map_or(0, |page| page.bytes())
826 }
827}
828
829#[derive(Debug, Clone)]
831pub struct Stripe {
832 rows: usize,
833 parts: Vec<u32>,
836 index: Span,
840 pages: Vec<Span>,
841 memberships: Pages,
842 sieves: Pages,
845 part_ranges: Pages,
856 zone: Zone,
857}
858
859impl Stripe {
860 #[must_use]
862 pub fn rows(&self) -> usize {
863 self.rows
864 }
865
866 #[must_use]
868 pub fn parts(&self) -> usize {
869 self.parts.len()
870 }
871
872 #[must_use]
878 pub fn zone(&self) -> &Zone {
879 &self.zone
880 }
881}
882
883#[derive(Debug, Clone)]
885pub struct Table {
886 name: String,
887 fields: Vec<Field>,
888 stripes: Vec<Stripe>,
889 rows: usize,
890 dictionaries: Vec<Option<Page>>,
891 dictionary_payloads: Vec<u64>,
897 frequencies: Vec<Option<Frequencies>>,
898 pair_frequencies: Vec<PairFrequencySummary>,
899 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
904 host_groups: Option<host::HostSummary>,
906 distincts: Vec<Option<u64>>,
916 clustering: Option<Clustering>,
924 generation: u64,
938 sections: Vec<Section>,
945}
946
947impl Table {
948 #[must_use]
950 pub fn name(&self) -> &str {
951 &self.name
952 }
953
954 #[must_use]
956 pub fn fields(&self) -> &[Field] {
957 &self.fields
958 }
959
960 #[must_use]
962 pub fn rows(&self) -> usize {
963 self.rows
964 }
965
966 #[must_use]
968 pub fn stripes(&self) -> &[Stripe] {
969 &self.stripes
970 }
971
972 #[must_use]
974 pub fn clustering(&self) -> Option<&Clustering> {
975 self.clustering.as_ref()
976 }
977
978 #[must_use]
983 pub fn generation(&self) -> u64 {
984 self.generation
985 }
986
987 #[must_use]
994 pub fn sections(&self) -> &[Section] {
995 &self.sections
996 }
997}
998
999#[derive(Debug, Clone)]
1011struct Entry {
1012 name: String,
1013 fields: Vec<Field>,
1014 rows: usize,
1015 directory: Page,
1017 nonzero: Vec<Option<u64>>,
1019 aggregates: Vec<Option<(i128, u64)>>,
1021 distincts: Vec<Option<u64>>,
1023 extremes: Vec<StoredIntegerExtremes>,
1025 frequencies: Vec<StoredNumericFrequencies>,
1027}
1028
1029type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1030type StoredNumericFrequencies = Option<NumericFrequencies>;
1031
1032#[derive(Debug, Clone, PartialEq, Eq)]
1045pub struct ViewEntry {
1046 pub name: String,
1048 pub sql: String,
1050 pub statement: String,
1052 pub aliases: Vec<String>,
1054 pub columns: Vec<Field>,
1056}
1057
1058#[derive(Debug, Clone)]
1060pub struct ColumnLayout {
1061 pub name: String,
1063 pub kind: String,
1065 pub pages: u64,
1067 pub memberships: u64,
1069 pub sieves: u64,
1071 pub part_ranges: u64,
1073 pub dictionary: u64,
1075}
1076
1077impl ColumnLayout {
1078 #[must_use]
1080 pub fn total(&self) -> u64 {
1081 self.pages
1082 .saturating_add(self.memberships)
1083 .saturating_add(self.sieves)
1084 .saturating_add(self.part_ranges)
1085 .saturating_add(self.dictionary)
1086 }
1087}
1088
1089#[derive(Debug, Clone)]
1100pub struct Layout {
1101 pub file: u64,
1103 pub rows: usize,
1105 pub stripes: usize,
1107 pub parts: usize,
1109 pub columns: Vec<ColumnLayout>,
1111 pub indexes: u64,
1114 pub directory: u64,
1116 pub header: u64,
1118}
1119
1120impl Layout {
1121 #[must_use]
1123 pub fn columns_total(&self) -> u64 {
1124 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1125 }
1126
1127 #[must_use]
1133 pub fn unaccounted(&self) -> u64 {
1134 self.file
1135 .saturating_sub(self.columns_total())
1136 .saturating_sub(self.indexes)
1137 .saturating_sub(self.directory)
1138 .saturating_sub(self.header)
1139 }
1140}
1141
1142#[derive(Debug, Clone)]
1153pub struct StoredPart {
1154 pub stripe: usize,
1156 pub part: usize,
1158 pub row: usize,
1160 pub rows: usize,
1162 pub encoding: String,
1164 pub bytes: u64,
1166 pub page: u64,
1168 pub offset: u64,
1170 pub low: Option<Value>,
1172 pub high: Option<Value>,
1174 pub nulls: Option<usize>,
1176}
1177
1178const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1185
1186#[derive(Debug)]
1211struct GlobalDictionary {
1212 primary: HashMap<u64, u32>,
1213 collisions: HashMap<u64, Vec<u32>>,
1214 checks: Vec<u64>,
1216 ends: Vec<u32>,
1218 counts: Vec<u64>,
1219 nulls: u64,
1220 filling: Vec<u8>,
1222 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1228 waiting: Vec<(usize, Vec<u8>)>,
1233 sample: Vec<(usize, Vec<u8>)>,
1239 stride: usize,
1241 shape: Option<chooser::Settled>,
1243 settled: usize,
1245 blocks: Vec<Vec<u8>>,
1250 early: BTreeMap<usize, EncodedBlock>,
1256 placed: Vec<Placed>,
1258}
1259
1260#[derive(Debug, Clone, Copy)]
1262struct Placed {
1263 start: u64,
1264 length: u64,
1265 hash: u64,
1266}
1267
1268type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1270
1271impl GlobalDictionary {
1272 fn new() -> Self {
1273 Self {
1274 primary: HashMap::new(),
1275 collisions: HashMap::new(),
1276 checks: Vec::new(),
1277 ends: Vec::new(),
1278 counts: Vec::new(),
1279 nulls: 0,
1280 filling: Vec::new(),
1281 grams: Vec::new(),
1282 waiting: Vec::new(),
1283 sample: Vec::new(),
1284 stride: 1,
1285 shape: None,
1286 settled: 0,
1287 blocks: Vec::new(),
1288 early: BTreeMap::new(),
1289 placed: Vec::new(),
1290 }
1291 }
1292
1293 fn values(&self) -> usize {
1295 self.ends.len()
1296 }
1297
1298 fn closing_bytes(&self) -> usize {
1301 let values = self.values();
1302 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1303 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1304 .sum::<usize>();
1305 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1306 }
1307
1308 fn encoded(&self) -> usize {
1310 self.placed.len() + self.blocks.len()
1311 }
1312
1313 #[cfg(test)]
1314 fn code(&mut self, text: &str) -> Result<u32> {
1315 let bytes = text.as_bytes();
1316 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1317 }
1318
1319 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1325 if let Some(&code) = self.primary.get(&hash) {
1326 if self.checks.get(code as usize) == Some(&check) {
1327 return Ok(code);
1328 }
1329 if let Some(codes) = self.collisions.get(&hash) {
1330 if let Some(code) =
1331 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1332 {
1333 return Ok(code);
1334 }
1335 }
1336 let code = self.insert(text, check)?;
1337 self.collisions.entry(hash).or_default().push(code);
1338 return Ok(code);
1339 }
1340 let code = self.insert(text, check)?;
1341 self.primary.insert(hash, code);
1342 Ok(code)
1343 }
1344
1345 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1346 let code = u32::try_from(self.ends.len())
1347 .map_err(|_| invalid("global dictionary has too many values"))?;
1348 self.filling.extend_from_slice(text);
1349 self.ends.push(
1350 u32::try_from(self.filling.len())
1351 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1352 );
1353 self.checks.push(check);
1354 self.counts.push(0);
1355 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1356 self.seal();
1357 }
1358 Ok(code)
1359 }
1360
1361 fn seal(&mut self) {
1367 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1368 let bytes = std::mem::take(&mut self.filling);
1369 if at % self.stride == 0 {
1370 self.sample.push((at, bytes.clone()));
1371 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1372 self.stride *= 2;
1373 let stride = self.stride;
1374 self.sample.retain(|(at, _)| at % stride == 0);
1375 }
1376 }
1377 self.waiting.push((at, bytes));
1378 }
1379
1380 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1382 block_values(self.block_ends(at), bytes)
1383 }
1384
1385 fn block_ends(&self, at: usize) -> &[u32] {
1387 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1388 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1389 &self.ends[first..last]
1390 }
1391
1392 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1399 let Some(shape) = &self.shape else { return Vec::new() };
1400 let waiting = std::mem::take(&mut self.waiting);
1401 waiting
1402 .into_iter()
1403 .map(|(at, bytes)| Unencoded {
1404 column,
1405 at,
1406 ends: self.block_ends(at).to_vec(),
1407 bytes,
1408 shape: shape.clone(),
1409 })
1410 .collect()
1411 }
1412
1413 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1416 if at < self.encoded() || self.early.insert(at, block).is_some() {
1417 return Err(Error::internal("a dictionary block came back twice"));
1418 }
1419 while let Some(block) = self.early.remove(&self.encoded()) {
1420 self.push_block(block);
1421 }
1422 Ok(())
1423 }
1424
1425 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1427 self.blocks.push(bytes);
1428 self.grams.push(*grams);
1429 }
1430
1431 fn settle(&mut self) -> Result<()> {
1439 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1440 return Ok(());
1441 }
1442 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1443 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1444 return Ok(());
1445 }
1446 let sample =
1447 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1448 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1449 self.settled = complete;
1450 Ok(())
1451 }
1452
1453 fn seal_rest(&mut self) {
1455 if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1458 self.seal();
1459 }
1460 }
1461
1462 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1465 let (block, bytes) = &self.waiting[at];
1466 let values = self.slices(*block, bytes);
1467 let encoded = match &self.shape {
1468 Some(shape) => string::encode_with(&values, shape)?,
1469 None => string::encode(&values)?,
1470 };
1471 Ok((encoded, block_grams(&values)))
1472 }
1473
1474 #[cfg(test)]
1476 fn finish_blocks(&mut self) -> Result<()> {
1477 self.seal_rest();
1478 let made = (0..self.waiting.len())
1479 .map(|at| self.encode_waiting(at))
1480 .collect::<Result<Vec<_>>>()?;
1481 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1482 if self.encoded() != at {
1483 return Err(Error::internal("a dictionary block was encoded out of order"));
1484 }
1485 self.push_block(block);
1486 }
1487 Ok(())
1488 }
1489
1490 fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1508 let count = self.placed.len() + self.blocks.len();
1509 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1510 return Err(invalid("global dictionary blocks do not cover its values"));
1511 }
1512 let mut bases = Vec::with_capacity(count);
1513 let mut total = 0_usize;
1514 for block in 0..count {
1515 bases.push(total as u64);
1516 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1517 total = total
1518 .checked_add(self.ends[last] as usize)
1519 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1520 }
1521 let mut flat = vec![0_u8; total];
1522 let mut outs = Vec::with_capacity(count);
1523 let mut rest = flat.as_mut_slice();
1524 for block in 0..count {
1525 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1526 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1527 outs.push((block, out));
1528 rest = after;
1529 }
1530 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1531 let mut stored = Vec::new();
1532 for (block, out) in run {
1533 let encoded = match self.placed.get(*block) {
1534 Some(place) => {
1535 let file = file.ok_or_else(|| {
1536 Error::internal("a written dictionary block has no file")
1537 })?;
1538 let length = usize::try_from(place.length).map_err(|_| {
1539 invalid("global dictionary block does not fit in memory")
1540 })?;
1541 stored.resize(length, 0);
1542 read_at(file, place.start, &mut stored)?;
1543 if checksum(&stored) != place.hash {
1544 return Err(invalid(
1545 "a global dictionary block did not read back as written",
1546 ));
1547 }
1548 stored.as_slice()
1549 }
1550 None => &self.blocks[*block - self.placed.len()],
1551 };
1552 let decoded = string::decode_flat(encoded)?;
1553 if decoded.bytes().len() != out.len() {
1554 return Err(invalid(
1555 "a global dictionary block is not the length its ends say",
1556 ));
1557 }
1558 out.copy_from_slice(decoded.bytes());
1559 }
1560 Ok(())
1561 };
1562 let workers = close_workers().min(count / 16).max(1);
1565 if workers <= 1 {
1566 one(&mut outs)?;
1567 } else {
1568 let per = count.div_ceil(workers);
1569 std::thread::scope(|scope| {
1570 outs.chunks_mut(per)
1571 .map(|run| scope.spawn(|| one(run)))
1572 .collect::<Vec<_>>()
1573 .into_iter()
1574 .try_for_each(|handle| {
1575 handle.join().map_err(|_| {
1576 Error::internal("a global dictionary decode worker panicked")
1577 })?
1578 })
1579 })?;
1580 }
1581 drop(outs);
1582 Ok((flat, bases))
1583 }
1584
1585 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1590 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1591 let Some(&end) = ends.get(code) else { return (0, 0) };
1592 let base = base as usize;
1593 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1594 (base + from, base + end as usize)
1595 }
1596
1597 fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1617 let (flat, bases) = self.decoded(file)?;
1618 let value = |code: u32| {
1619 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1620 flat.get(from..to).unwrap_or_default()
1621 };
1622 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1623 sort_by_value_across(&mut codes, value, close_workers());
1624 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1625 Ok((order, flat, bases))
1626 }
1627
1628 #[cfg(test)]
1629 fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1630 self.ranked_with_values(file).map(|(order, _, _)| order)
1631 }
1632}
1633
1634#[derive(Debug)]
1642pub struct Writer {
1643 file: File,
1644 at: u64,
1652 written_back: u64,
1654 table: Table,
1655 generation: u64,
1656 order: Vec<((u64, u64), (u64, u64))>,
1659 next_order: u64,
1660 dictionaries: Vec<Option<GlobalDictionary>>,
1661 coded: Arc<[AtomicBool]>,
1664 gathers: Vec<Option<stats::Gather>>,
1670 lent: Option<Arc<Lent>>,
1673 pending: Vec<PendingChunk>,
1674 closed: Vec<Entry>,
1676 views: Vec<ViewEntry>,
1681 profile: Option<Arc<LoadProfile>>,
1687}
1688
1689#[derive(Debug)]
1697struct PendingChunk {
1698 order: (u64, u64),
1699 chunk: Chunk,
1700}
1701
1702#[derive(Debug, Clone, Copy)]
1708struct Part {
1709 order: (u64, u64),
1710 rows: usize,
1711 footprint: usize,
1712}
1713
1714impl Part {
1715 fn of(pending: &PendingChunk) -> Self {
1716 Self {
1717 order: pending.order,
1718 rows: pending.chunk.len(),
1719 footprint: pending.chunk.footprint(),
1720 }
1721 }
1722}
1723
1724#[derive(Debug)]
1730struct ColumnStripe {
1731 pages: Vec<Vec<u8>>,
1732 codes: Vec<Option<Vec<u32>>>,
1733 sieves: Vec<Option<Sieve>>,
1734 ranges: Vec<Range>,
1735}
1736
1737fn weight(ty: &LogicalType) -> usize {
1745 match ty {
1746 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1747 LogicalType::HugeInt
1748 | LogicalType::UHugeInt
1749 | LogicalType::Uuid
1750 | LogicalType::Interval => 16,
1751 LogicalType::BigInt
1752 | LogicalType::UBigInt
1753 | LogicalType::Timestamp
1754 | LogicalType::Time
1755 | LogicalType::TimeTz
1756 | LogicalType::TimestampTz
1757 | LogicalType::TimestampS
1758 | LogicalType::TimestampMs
1759 | LogicalType::TimestampNs
1760 | LogicalType::Double
1761 | LogicalType::Decimal { .. } => 8,
1762 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1763 LogicalType::SmallInt | LogicalType::USmallInt => 2,
1764 _ => 1,
1765 }
1766}
1767
1768pub const STRIPE_PARTS: usize = 64;
1775
1776const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1784
1785const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1801
1802const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1804
1805fn index_section(parts: usize) -> Result<usize> {
1807 parts
1808 .checked_mul(INDEX_ENTRY)
1809 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1810 .ok_or_else(|| invalid("index page length overflow"))
1811}
1812
1813impl Writer {
1814 pub fn open(
1833 path: impl AsRef<Path>,
1834 name: impl Into<String>,
1835 fields: Vec<Field>,
1836 ) -> Result<Self> {
1837 for field in &fields {
1838 type_tag(&field.ty)?;
1839 }
1840 let name = name.into();
1841 let path = path.as_ref();
1842 let (_, size, slot, bytes, _) = slot_bytes(path)?;
1843 let (mut closed, views) = decode_catalog(&bytes, size)?;
1844 if let Some(at) = closed.iter().position(|held| held.name == name) {
1855 if closed[at].rows > 0 {
1856 return Err(invalid("two tables in one native file have the same name"));
1857 }
1858 closed.remove(at);
1859 }
1860 let generation = slot
1865 .generation
1866 .checked_add(1)
1867 .ok_or_else(|| invalid("native file generation overflow"))?;
1868 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1869 Ok(Self {
1870 file,
1871 at: size,
1874 written_back: size,
1875 dictionaries: fields
1876 .iter()
1877 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1878 .collect(),
1879 coded: fields
1880 .iter()
1881 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1882 .collect(),
1883 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1884 lent: None,
1885 table: Table {
1886 name,
1887 dictionaries: vec![None; fields.len()],
1888 dictionary_payloads: Vec::new(),
1889 distincts: vec![None; fields.len()],
1890 fields,
1891 stripes: Vec::new(),
1892 rows: 0,
1893 frequencies: Vec::new(),
1894 pair_frequencies: Vec::new(),
1895 frequency_texts: Vec::new(),
1896 host_groups: None,
1897 clustering: None,
1898 generation,
1899 sections: Vec::new(),
1900 },
1901 generation,
1902 order: Vec::new(),
1903 next_order: 0,
1904 pending: Vec::with_capacity(STRIPE_PARTS),
1905 closed,
1906 views,
1907 profile: None,
1908 })
1909 }
1910
1911 pub fn create(
1917 path: impl AsRef<Path>,
1918 name: impl Into<String>,
1919 fields: Vec<Field>,
1920 ) -> Result<Self> {
1921 for field in &fields {
1922 type_tag(&field.ty)?;
1923 }
1924 let file =
1925 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1926 let mut header = [0; HEADER as usize];
1927 header[..8].copy_from_slice(MAGIC);
1928 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1929 write_at(&file, 0, &header)?;
1930 Ok(Self {
1931 file,
1932 at: HEADER,
1933 written_back: HEADER,
1934 dictionaries: fields
1935 .iter()
1936 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1937 .collect(),
1938 coded: fields
1939 .iter()
1940 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1941 .collect(),
1942 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1943 lent: None,
1944 table: Table {
1945 name: name.into(),
1946 dictionaries: vec![None; fields.len()],
1947 dictionary_payloads: Vec::new(),
1948 distincts: vec![None; fields.len()],
1949 fields,
1950 stripes: Vec::new(),
1951 rows: 0,
1952 frequencies: Vec::new(),
1953 pair_frequencies: Vec::new(),
1954 frequency_texts: Vec::new(),
1955 host_groups: None,
1956 clustering: None,
1957 generation: 1,
1958 sections: Vec::new(),
1959 },
1960 generation: 1,
1961 order: Vec::new(),
1962 next_order: 0,
1963 pending: Vec::with_capacity(STRIPE_PARTS),
1964 closed: Vec::new(),
1965 views: Vec::new(),
1966 profile: None,
1967 })
1968 }
1969
1970 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1992 let file =
1993 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1994 let mut header = [0; HEADER as usize];
1995 header[..8].copy_from_slice(MAGIC);
1996 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1997 write_at(&file, 0, &header)?;
1998 let catalog = encode_catalog(&[], views)?;
1999 write_at(&file, HEADER, &catalog)?;
2000 file.sync_all().map_err(io)?;
2004 let slot = Slot {
2005 offset: HEADER,
2006 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2007 generation: 1,
2008 hash: checksum(&catalog),
2009 };
2010 write_at(&file, slot_offset(1), &slot.bytes())?;
2011 file.sync_all().map_err(io)?;
2012 Ok(())
2013 }
2014
2015 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2026 for field in &fields {
2027 type_tag(&field.ty)?;
2028 }
2029 let name = name.into();
2030 let entry = self.close()?;
2031 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2032 return Err(invalid("two tables in one native file have the same name"));
2033 }
2034 let Self { file, at, generation, mut closed, views, .. } = self;
2035 closed.push(entry);
2036 Ok(Self {
2037 file,
2038 written_back: at,
2039 at,
2040 generation,
2041 closed,
2042 views,
2043 profile: None,
2044 dictionaries: fields
2045 .iter()
2046 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
2047 .collect(),
2048 coded: fields
2049 .iter()
2050 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
2051 .collect(),
2052 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2053 lent: None,
2054 table: Table {
2055 name,
2056 dictionaries: vec![None; fields.len()],
2057 dictionary_payloads: Vec::new(),
2058 distincts: vec![None; fields.len()],
2059 fields,
2060 stripes: Vec::new(),
2061 rows: 0,
2062 frequencies: Vec::new(),
2063 pair_frequencies: Vec::new(),
2064 frequency_texts: Vec::new(),
2065 host_groups: None,
2066 clustering: None,
2067 generation,
2068 sections: Vec::new(),
2069 },
2070 order: Vec::new(),
2071 next_order: 0,
2072 pending: Vec::with_capacity(STRIPE_PARTS),
2073 })
2074 }
2075
2076 #[must_use]
2086 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2087 self.views = views;
2088 self
2089 }
2090
2091 #[must_use]
2097 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2098 self.profile = Some(profile);
2099 self
2100 }
2101
2102 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2117 self.table.clustering = Some(Clustering::new(
2120 clustering.columns().to_vec(),
2121 clustering.width(),
2122 &self.table.fields,
2123 )?);
2124 Ok(self)
2125 }
2126
2127 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2132 write_at(&self.file, self.at, bytes)?;
2133 self.at = self
2134 .at
2135 .checked_add(bytes.len() as u64)
2136 .ok_or_else(|| invalid("native file length overflow"))?;
2137 if self.at - self.written_back >= WRITEBACK_STRETCH {
2138 rudb_io::start_writeback(&self.file, self.written_back, self.at - self.written_back);
2139 self.written_back = self.at;
2140 }
2141 Ok(())
2142 }
2143
2144 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2150 let order = (self.next_order, 0);
2151 self.next_order = self.next_order.saturating_add(1);
2152 self.append_at(order, chunk)
2153 }
2154
2155 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2166 if chunk.is_empty() {
2167 return Ok(());
2168 }
2169 self.admit(chunk)?;
2170 if self.pending.last().is_some_and(|last| last.order > order) {
2171 self.flush_pending()?;
2172 }
2173 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2178 if self.pending.len() == STRIPE_PARTS {
2179 self.flush_pending()?;
2180 }
2181 Ok(())
2182 }
2183
2184 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2200 if parts.len() > STRIPE_PARTS {
2201 return Err(invalid("a stripe was handed more parts than it holds"));
2202 }
2203 self.flush_pending()?;
2206 for (order, chunk) in parts {
2207 if chunk.is_empty() {
2208 continue;
2209 }
2210 self.admit(&chunk)?;
2211 self.pending.push(PendingChunk { order, chunk });
2212 }
2213 self.flush_pending()
2214 }
2215
2216 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2218 if chunk.width() != self.table.fields.len() {
2219 return Err(invalid("chunk width differs from table schema"));
2220 }
2221 for (index, field) in self.table.fields.iter().enumerate() {
2222 if chunk.column(index)?.logical_type() != &field.ty {
2223 return Err(invalid("chunk type differs from table schema"));
2224 }
2225 }
2226 self.table.rows = self
2227 .table
2228 .rows
2229 .checked_add(chunk.len())
2230 .ok_or_else(|| invalid("row count overflow"))?;
2231 Ok(())
2232 }
2233
2234 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2236 let mut stripe = ColumnStripe {
2237 pages: Vec::with_capacity(columns.len()),
2238 codes: Vec::with_capacity(columns.len()),
2239 sieves: Vec::with_capacity(columns.len()),
2240 ranges: Vec::with_capacity(columns.len()),
2241 };
2242 let mut settling = Settling::default();
2243 for &column in columns {
2244 let bytes = encode(column, &mut settling)?;
2245 if bytes.len() > MAX_PAGE {
2246 return Err(invalid("column page exceeds the configured bound"));
2247 }
2248 let range = Range::of(column);
2251 let sieve =
2262 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2263 stripe.pages.push(bytes);
2264 stripe.codes.push(None);
2265 stripe.sieves.push(sieve);
2266 stripe.ranges.push(range);
2267 }
2268 Ok(stripe)
2269 }
2270
2271 fn place_blocks(&mut self) -> Result<()> {
2276 if let Some(lent) = self.lent.clone() {
2277 return self.place_lent_blocks(&lent);
2278 }
2279 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2280 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2281 for block in std::mem::take(&mut dictionary.blocks) {
2282 let start = self.at;
2283 self.put(&block)?;
2284 dictionary.placed.push(Placed {
2285 start,
2286 length: block.len() as u64,
2287 hash: checksum(&block),
2288 });
2289 }
2290 Ok(())
2291 });
2292 self.dictionaries = dictionaries;
2293 placed
2294 }
2295
2296 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2302 for column in lent.columns() {
2303 let Ok(mut held) = column.try_lock() else { continue };
2304 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2305 for block in std::mem::take(&mut dictionary.blocks) {
2306 let start = self.at;
2307 self.put(&block)?;
2308 dictionary.placed.push(Placed {
2309 start,
2310 length: block.len() as u64,
2311 hash: checksum(&block),
2312 });
2313 }
2314 }
2315 Ok(())
2316 }
2317
2318 fn reclaim(&mut self) -> Result<()> {
2322 let Some(lent) = self.lent.take() else { return Ok(()) };
2323 let (dictionaries, gathers) = lent.reclaim()?;
2324 self.dictionaries = dictionaries;
2325 self.gathers = gathers;
2326 Ok(())
2327 }
2328
2329 fn flush_pending(&mut self) -> Result<()> {
2334 if self.pending.is_empty() {
2335 return Ok(());
2336 }
2337 let held = std::mem::take(&mut self.pending);
2338 let prepared = self.preparer().prepare_held(held)?;
2339 let merged = self.merge_held(prepared)?;
2340 let paged = merged.pages()?;
2341 self.write_paged(paged)
2342 }
2343
2344 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2346 let width = self.table.fields.len();
2347 let parts = held.len();
2348 if encoded.len() != width {
2349 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2350 }
2351 let profile = self.profile.clone();
2352 if let Some(profile) = &profile {
2353 let rows = held.iter().map(|part| part.rows as u64).sum();
2354 let raw = held.iter().map(|part| part.footprint as u64).sum();
2355 let pages =
2356 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2357 profile.moved(Stage::Pages, raw, pages, rows);
2358 }
2359 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2362 let before = self.at;
2363 self.place_blocks()?;
2364 drop(timing);
2365 if let Some(profile) = &profile {
2366 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2367 }
2368 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2369 let before = self.at;
2370 let mut pages = Vec::with_capacity(width);
2371 let mut memberships = vec![None; width];
2372 let mut ranges = Vec::with_capacity(width);
2373 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2374 for stripe in &encoded {
2375 let offset = self.at;
2376 let section = index.len();
2377 let mut length = 0_usize;
2378 for bytes in &stripe.pages {
2379 write_at(&self.file, self.at + length as u64, bytes)?;
2380 put_u32(
2381 &mut index,
2382 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2383 );
2384 put_u64(&mut index, checksum(bytes));
2385 length = length
2386 .checked_add(bytes.len())
2387 .ok_or_else(|| invalid("column page length overflow"))?;
2388 }
2389 let hash = checksum(&index[section..]);
2390 put_u64(&mut index, hash);
2391 if length > MAX_PAGE {
2392 return Err(invalid("column page exceeds the configured bound"));
2393 }
2394 self.at = self
2395 .at
2396 .checked_add(length as u64)
2397 .ok_or_else(|| invalid("native file length overflow"))?;
2398 pages.push(Span {
2399 offset,
2400 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2401 });
2402 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2403 }
2404 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2405 if stripe.codes.iter().all(Option::is_none) {
2406 continue;
2407 }
2408 let lists = stripe
2409 .codes
2410 .iter()
2411 .map(|codes| codes.clone().unwrap_or_default())
2412 .collect::<Vec<_>>();
2413 let bytes = encode_membership(&merged_codes(lists));
2414 let offset = self.at;
2415 self.put(&bytes)?;
2416 *membership = Some(Page {
2417 offset,
2418 length: u32::try_from(bytes.len())
2419 .map_err(|_| invalid("membership page length overflow"))?,
2420 hash: checksum(&bytes),
2421 });
2422 }
2423 let mut sieves = vec![None; width];
2424 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2425 if stripe.sieves.iter().all(Option::is_none) {
2426 continue;
2427 }
2428 let bytes = encode_sieves(stripe.sieves.iter())?;
2429 let offset = self.at;
2430 self.put(&bytes)?;
2431 *page = Some(Page {
2432 offset,
2433 length: u32::try_from(bytes.len())
2434 .map_err(|_| invalid("sieve page length overflow"))?,
2435 hash: checksum(&bytes),
2436 });
2437 }
2438 let mut part_ranges = vec![None; width];
2444 if parts > 1 {
2445 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2446 let bytes = encode_part_ranges(&stripe.ranges)?;
2447 if bytes.len() >= span.length as usize {
2448 continue;
2449 }
2450 let offset = self.at;
2451 self.put(&bytes)?;
2452 *page = Some(Page {
2453 offset,
2454 length: u32::try_from(bytes.len())
2455 .map_err(|_| invalid("part range page length overflow"))?,
2456 hash: checksum(&bytes),
2457 });
2458 }
2459 }
2460 let offset = self.at;
2461 self.put(&index)?;
2462 let index = Span {
2463 offset,
2464 length: u32::try_from(index.len())
2465 .map_err(|_| invalid("index page length overflow"))?,
2466 };
2467 let mut rows = 0_usize;
2468 let mut lengths = Vec::with_capacity(parts);
2469 let mut span = None;
2470 for part in held {
2471 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2472 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2473 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2474 }
2475 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2476 self.table.stripes.push(Stripe {
2477 rows,
2478 parts: lengths,
2479 index,
2480 pages,
2481 memberships: Pages::from_slots(memberships)?,
2482 sieves: Pages::from_slots(sieves)?,
2483 part_ranges: Pages::from_slots(part_ranges)?,
2484 zone: Zone::from_ranges(ranges),
2485 });
2486 drop(timing);
2487 if let Some(profile) = &profile {
2488 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2489 }
2490 Ok(())
2491 }
2492
2493 fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2509 let signed = match self.table.fields[column].ty {
2510 LogicalType::TinyInt
2511 | LogicalType::SmallInt
2512 | LogicalType::Integer
2513 | LogicalType::BigInt
2514 | LogicalType::Date
2515 | LogicalType::Timestamp => true,
2516 LogicalType::UTinyInt
2517 | LogicalType::USmallInt
2518 | LogicalType::UInteger
2519 | LogicalType::UBigInt => false,
2520 _ => return Ok((None, None)),
2521 };
2522 let value_of = |bits: Option<u64>| match bits {
2523 None => FrequencyValue::Null,
2524 Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2525 Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2526 };
2527 let mut first = Candidates::default();
2530 let mut distinct = distinct::ExactDistinct::new();
2531 let mut run = Run::default();
2532 self.visit_numeric(column, signed, |_, bits| {
2533 if let Some((bits, times)) = run.push(bits) {
2534 first.add(bits, times);
2535 }
2536 if run.times == 1 {
2537 if let Some(bits) = bits {
2538 distinct.insert(bits);
2539 }
2540 }
2541 })?;
2542 if let Some((bits, times)) = run.take() {
2543 first.add(bits, times);
2544 }
2545 let Candidates { counts: candidates, nulls, decrements } = first;
2546 let (exact, null_count) = if decrements == 0 {
2547 let exact = candidates
2548 .into_iter()
2549 .map(|(bits, count)| (bits, u64::from(count)))
2550 .collect::<FrequencyMap<_>>();
2551 (exact, (nulls != 0).then_some(u64::from(nulls)))
2552 } else {
2553 let mut lower = candidates.values().copied().collect::<Vec<_>>();
2554 if nulls != 0 {
2555 lower.push(nulls);
2556 }
2557 lower.sort_unstable_by(|left, right| right.cmp(left));
2558 if lower.len() < FREQUENCY_BUILD_RANK
2559 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2560 {
2561 return Ok((None, distinct.count()));
2562 }
2563 let mut exact =
2564 candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2565 let mut null_count = (nulls != 0).then_some(0_u64);
2566 let mut recount = |bits: Option<u64>, times: u32| {
2567 let held = match bits {
2568 Some(bits) => exact.get_mut(&bits),
2569 None => null_count.as_mut(),
2570 };
2571 if let Some(count) = held {
2572 *count = count.saturating_add(u64::from(times));
2573 }
2574 };
2575 let mut run = Run::default();
2576 self.visit_numeric(column, signed, |_, bits| {
2577 if let Some((bits, times)) = run.push(bits) {
2578 recount(bits, times);
2579 }
2580 })?;
2581 if let Some((bits, times)) = run.take() {
2582 recount(bits, times);
2583 }
2584 (exact, null_count)
2585 };
2586 let mut entries = exact
2587 .into_iter()
2588 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2589 .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2590 .collect::<Vec<_>>();
2591 let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2592 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2593 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2594 });
2595 let mut ordinals = Vec::new();
2596 let mut ordinal_entries = Vec::new();
2597 if let Some(kept_rows) = kept_rows {
2598 let mut kept = FrequencyMap::default();
2599 let mut null_kept = None;
2600 for (at, entry) in entries.iter().enumerate() {
2601 let at = u16::try_from(at)
2602 .map_err(|_| invalid("too many retained frequency entries"))?;
2603 match entry.value {
2604 FrequencyValue::Integer(value) => {
2605 kept.insert(value as u64, at);
2606 }
2607 FrequencyValue::Null => null_kept = Some(at),
2608 FrequencyValue::Code(_) => {}
2609 }
2610 }
2611 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2612 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2613 self.visit_numeric(column, signed, |ordinal, bits| {
2614 let held = match bits {
2615 Some(bits) => kept.get(&bits).copied(),
2616 None => null_kept,
2617 };
2618 if let Some(entry) = held {
2619 ordinals.push(ordinal);
2620 ordinal_entries.push(entry);
2621 }
2622 })?;
2623 }
2624 Ok((
2625 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2626 distinct.count(),
2627 ))
2628 }
2629
2630 fn visit_numeric(
2637 &self,
2638 column: usize,
2639 signed: bool,
2640 mut visit: impl FnMut(u64, Option<u64>),
2641 ) -> Result<()> {
2642 let ty = &self.table.fields[column].ty;
2643 let mut start = 0_u64;
2644 let mut block = Vec::new();
2645 for stripe in &self.table.stripes {
2646 let spans = read_index(&self.file, stripe, column)?;
2647 let page = stripe.pages[column];
2648 let mut bytes = vec![0; page.length as usize];
2649 read_at(&self.file, page.offset, &mut bytes)?;
2650 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2651 let part = part_bytes(&bytes, *span)?;
2652 if checksum(part) != span.hash {
2653 return Err(invalid("column page checksum differs while building frequencies"));
2654 }
2655 let rows = rows as usize;
2656 let vector = decode(ty, rows, part, None)?;
2657 if signed && vector.signed_block(&mut block) && block.len() == rows {
2661 if vector.none_null() {
2662 for (row, &value) in block.iter().enumerate() {
2663 visit(start.saturating_add(row as u64), Some(value as u64));
2664 }
2665 } else {
2666 for (row, &value) in block.iter().enumerate() {
2667 let bits = (!vector.is_null_at(row)).then_some(value as u64);
2668 visit(start.saturating_add(row as u64), bits);
2669 }
2670 }
2671 start = start.saturating_add(rows as u64);
2672 continue;
2673 }
2674 for row in 0..rows {
2676 let bits = if vector.is_null_at(row) {
2677 None
2678 } else {
2679 let widened = match vector.signed_at(row) {
2683 Some(value) => Some(value as u64),
2684 None => match vector.value_at(row) {
2685 Value::UTinyInt(value) => Some(u64::from(value)),
2686 Value::USmallInt(value) => Some(u64::from(value)),
2687 Value::UInteger(value) => Some(u64::from(value)),
2688 Value::UBigInt(value) => Some(value),
2689 _ => None,
2690 },
2691 };
2692 Some(widened.ok_or_else(|| {
2693 invalid("numeric frequency page did not contain an integer value")
2694 })?)
2695 };
2696 visit(start.saturating_add(row as u64), bits);
2697 }
2698 start = start.saturating_add(rows as u64);
2699 }
2700 }
2701 Ok(())
2702 }
2703
2704 fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2712 let mut columns = self
2713 .table
2714 .fields
2715 .iter()
2716 .enumerate()
2717 .filter_map(|(column, field)| {
2718 matches!(
2719 field.ty,
2720 LogicalType::TinyInt
2721 | LogicalType::SmallInt
2722 | LogicalType::Integer
2723 | LogicalType::BigInt
2724 | LogicalType::UTinyInt
2725 | LogicalType::USmallInt
2726 | LogicalType::UInteger
2727 | LogicalType::UBigInt
2728 | LogicalType::Date
2729 | LogicalType::Timestamp
2730 )
2731 .then_some(column)
2732 })
2733 .collect::<Vec<_>>();
2734 let workers = std::thread::available_parallelism()
2735 .map_or(1, usize::from)
2736 .min(MAX_FREQUENCY_WORKERS)
2737 .min(columns.len());
2738 let profile = self.profile.as_deref();
2739 if workers <= 1 {
2740 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2741 let mut frequencies = vec![(None, None); self.table.fields.len()];
2742 for column in columns {
2743 frequencies[column] = self.numeric_frequency(column)?;
2744 }
2745 return Ok(frequencies);
2746 }
2747 columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2750 let queue = Mutex::new(columns);
2751 let pieces = std::thread::scope(|scope| {
2752 (0..workers)
2753 .map(|_| {
2754 scope.spawn(|| {
2755 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2756 let mut mine = Vec::new();
2757 loop {
2758 let taken = queue
2759 .lock()
2760 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2761 .pop();
2762 let Some(column) = taken else { break };
2763 mine.push((column, self.numeric_frequency(column)?));
2764 }
2765 Ok(mine)
2766 })
2767 })
2768 .collect::<Vec<_>>()
2769 .into_iter()
2770 .map(|handle| {
2771 handle
2772 .join()
2773 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2774 })
2775 .collect::<Result<Vec<_>>>()
2776 })?;
2777 let mut frequencies = vec![(None, None); self.table.fields.len()];
2778 for piece in pieces {
2779 for (column, summary) in piece {
2780 frequencies[column] = summary;
2781 }
2782 }
2783 Ok(frequencies)
2784 }
2785
2786 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2788 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2789 return Ok(None);
2790 }
2791 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2792 return Err(invalid("frequency ordinals are not sorted and unique"));
2793 }
2794 let mut out = Vec::with_capacity(ordinals.len());
2795 let mut wanted = 0;
2796 let mut stripe_start = 0_u64;
2797 for stripe in &self.table.stripes {
2798 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2799 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2800 stripe_start = stripe_end;
2801 continue;
2802 }
2803 let spans = read_index(&self.file, stripe, column)?;
2804 let page = stripe.pages[column];
2805 let mut bytes = vec![0; page.length as usize];
2806 read_at(&self.file, page.offset, &mut bytes)?;
2807 let mut part_start = stripe_start;
2808 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2809 let part_end = part_start.saturating_add(u64::from(rows));
2810 if wanted < ordinals.len() && ordinals[wanted] < part_end {
2811 let part = part_bytes(&bytes, *span)?;
2812 if checksum(part) != span.hash {
2813 return Err(invalid(
2814 "column page checksum differs while building pair frequencies",
2815 ));
2816 }
2817 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2818 let positions = ordinals[wanted..upto]
2819 .iter()
2820 .map(|&ordinal| {
2821 usize::try_from(ordinal.saturating_sub(part_start))
2822 .map_err(|_| invalid("frequency row offset does not fit in memory"))
2823 })
2824 .collect::<Result<Vec<_>>>()?;
2825 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2826 return Ok(None);
2827 }
2828 wanted = upto;
2829 }
2830 part_start = part_end;
2831 }
2832 stripe_start = stripe_end;
2833 }
2834 if wanted != ordinals.len() {
2835 return Err(invalid("frequency ordinal is outside the table"));
2836 }
2837 Ok(Some(out))
2838 }
2839
2840 fn pair_frequencies(
2842 &self,
2843 frequencies: &[Option<Frequencies>],
2844 ) -> Result<Vec<PairFrequencySummary>> {
2845 let anchors = frequencies
2846 .iter()
2847 .enumerate()
2848 .filter_map(|(column, summary)| {
2849 match summary {
2851 Some(Frequencies::Held(summary)) => Some(summary),
2852 _ => None,
2853 }
2854 .filter(|summary| {
2855 !summary.ordinals.is_empty()
2856 && summary.ordinal_entries.len() == summary.ordinals.len()
2857 })
2858 .cloned()
2859 .map(|summary| (column, summary))
2860 })
2861 .collect::<Vec<_>>();
2862 let strings = self
2863 .dictionaries
2864 .iter()
2865 .enumerate()
2866 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2867 .collect::<Vec<_>>();
2868 let mut summaries = Vec::new();
2869 for (first, anchors) in anchors {
2870 for &second in &strings {
2871 if summaries.len() == MAX_PAIR_FREQUENCIES {
2872 return Ok(summaries);
2873 }
2874 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2875 continue;
2876 };
2877 if codes.len() != anchors.ordinal_entries.len() {
2878 return Err(invalid("pair frequency columns have different lengths"));
2879 }
2880 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2881 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2882 *counts.entry((anchor, code)).or_default() += 1;
2883 }
2884 let mut entries = counts
2885 .into_iter()
2886 .map(|((first_entry, second), count)| PairFrequencyEntry {
2887 first_entry,
2888 second,
2889 count,
2890 })
2891 .collect::<Vec<_>>();
2892 entries.sort_unstable_by(|left, right| {
2893 right
2894 .count
2895 .cmp(&left.count)
2896 .then_with(|| left.first_entry.cmp(&right.first_entry))
2897 .then_with(|| left.second.cmp(&right.second))
2898 });
2899 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2900 entries.truncate(FREQUENCY_ENTRIES);
2901 summaries.push(PairFrequencySummary {
2902 first: u16::try_from(first)
2903 .map_err(|_| invalid("pair frequency column index overflows"))?,
2904 second: u16::try_from(second)
2905 .map_err(|_| invalid("pair frequency column index overflows"))?,
2906 entries,
2907 omitted_max: anchors.omitted_max.max(pair_omitted),
2908 });
2909 }
2910 }
2911 Ok(summaries)
2912 }
2913
2914 fn close(&mut self) -> Result<Entry> {
2925 self.reclaim()?;
2926 self.flush_pending()?;
2927 let profile = self.profile.clone();
2931 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2932 let before = self.at;
2933 let mut stripes = std::mem::take(&mut self.order)
2934 .into_iter()
2935 .zip(std::mem::take(&mut self.table.stripes))
2936 .collect::<Vec<_>>();
2937 stripes.sort_by_key(|(order, _)| order.0);
2938 let mut previous: Option<(u64, u64)> = None;
2939 for ((first, last), _) in &stripes {
2940 if previous.is_some_and(|previous| previous >= *first) {
2941 return Err(invalid("chunks did not arrive in source order"));
2942 }
2943 previous = Some(*last);
2944 }
2945 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2946 drop(timing);
2947 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2948 let placing = self.at;
2949 finish_dictionaries(&mut self.dictionaries)?;
2950 self.place_blocks()?;
2951 let this = &*self;
2958 let (numeric, closed) = std::thread::scope(|scope| {
2959 let numeric = scope.spawn(|| {
2960 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
2961 this.numeric_frequencies()?.into_iter().unzip();
2962 let frequencies = frequencies
2963 .into_iter()
2964 .map(|held| held.map(Frequencies::Held))
2965 .collect::<Vec<_>>();
2966 let pairs = this.pair_frequencies(&frequencies)?;
2967 Ok::<_, Error>((frequencies, distincts, pairs))
2968 });
2969 let closed = this.close_dictionaries();
2970 let numeric =
2971 numeric.join().map_err(|_| Error::internal("the native frequency thread panicked"));
2972 (numeric, closed)
2973 });
2974 let (frequencies, distincts, pairs) = numeric??;
2975 let closed = closed?;
2976 self.table.frequencies = frequencies;
2977 self.table.distincts = distincts;
2978 self.table.pair_frequencies = pairs;
2979 self.dictionaries = Vec::new();
2980 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
2981 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
2982 self.table.host_groups = None;
2983 for (index, closed) in closed.into_iter().enumerate() {
2984 let Some(closed) = closed else { continue };
2985 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
2986 self.table.distincts[index] = Some(distinct);
2987 self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
2988 self.table.frequency_texts[index] = texts;
2989 if hosts.is_some() {
2990 self.table.host_groups = hosts;
2991 }
2992 let offset = self.at;
2993 self.put(&encoded.index)?;
2994 self.put(&encoded.ranks)?;
2995 self.put(&encoded.grams)?;
2996 self.table.dictionary_payloads[index] = payload;
2997 let length = encoded
2998 .index
2999 .len()
3000 .checked_add(encoded.ranks.len())
3001 .and_then(|len| len.checked_add(encoded.grams.len()))
3002 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3003 self.table.dictionaries[index] = Some(Page {
3004 offset,
3005 length: u32::try_from(length)
3006 .map_err(|_| invalid("dictionary page length overflow"))?,
3007 hash: checksum(&encoded.index),
3008 });
3009 }
3010 drop(timing);
3011 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3012 let placed = self.at - placing;
3013 self.write_stats()?;
3014 let directory = encode_directory(&self.table)?;
3015 if directory.len() > MAX_DIRECTORY {
3016 return Err(invalid("directory exceeds the configured bound"));
3017 }
3018 let offset = self.at;
3019 self.put(&directory)?;
3020 drop(timing);
3021 if let Some(profile) = &profile {
3022 profile.moved(Stage::Dictionary, 0, placed, 0);
3023 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3024 }
3025 Ok(Entry {
3026 name: self.table.name.clone(),
3027 fields: self.table.fields.clone(),
3028 rows: self.table.rows,
3029 nonzero: table_nonzero_counts(&self.table),
3030 aggregates: table_aggregate_sums(&self.table),
3031 distincts: self.table.distincts.clone(),
3032 extremes: table_integer_extremes(&self.table),
3033 frequencies: table_complete_numeric_frequencies(&self.table),
3034 directory: Page {
3035 offset,
3036 length: u32::try_from(directory.len())
3037 .map_err(|_| invalid("directory length overflow"))?,
3038 hash: checksum(&directory),
3039 },
3040 })
3041 }
3042
3043 fn close_dictionaries(&self) -> Result<Vec<Option<ClosedDictionary>>> {
3050 let mut jobs = self
3051 .dictionaries
3052 .iter()
3053 .enumerate()
3054 .filter_map(|(index, dictionary)| {
3055 dictionary
3056 .as_ref()
3057 .map(|dictionary| (index, dictionary, dictionary.closing_bytes()))
3058 })
3059 .collect::<Vec<_>>();
3060 jobs.sort_by_key(|&(_, _, bytes)| bytes);
3061 let mut closed = (0..self.dictionaries.len()).map(|_| None).collect::<Vec<_>>();
3062 let workers = close_workers().min(jobs.len());
3063 if workers <= 1 {
3064 for (index, dictionary, _) in jobs {
3065 closed[index] = Some(self.close_dictionary(index, dictionary)?);
3066 }
3067 return Ok(closed);
3068 }
3069 let state = Mutex::new((jobs, 0_usize));
3071 let finished = Condvar::new();
3072 let profile = self.profile.as_deref();
3073 let pieces = std::thread::scope(|scope| {
3074 (0..workers)
3075 .map(|_| {
3076 scope.spawn(|| {
3077 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3078 let mut mine = Vec::new();
3079 loop {
3080 let mut held = state.lock().map_err(|_| {
3081 Error::internal("a native dictionary worker panicked")
3082 })?;
3083 let (index, dictionary, bytes) = loop {
3084 let (jobs, busy) = &mut *held;
3085 if jobs.is_empty() {
3086 return Ok(mine);
3087 }
3088 let fits = jobs.iter().rposition(|&(_, _, bytes)| {
3089 *busy == 0
3090 || busy.saturating_add(bytes) <= CLOSE_DICTIONARY_BYTES
3091 });
3092 if let Some(at) = fits {
3093 let job = jobs.remove(at);
3094 *busy += job.2;
3095 break job;
3096 }
3097 held = finished.wait(held).map_err(|_| {
3098 Error::internal("a native dictionary worker panicked")
3099 })?;
3100 };
3101 drop(held);
3102 let _room = Room { state: &state, finished: &finished, bytes };
3105 mine.push((index, self.close_dictionary(index, dictionary)?));
3106 }
3107 })
3108 })
3109 .collect::<Vec<_>>()
3110 .into_iter()
3111 .map(|handle| {
3112 handle
3113 .join()
3114 .map_err(|_| Error::internal("a native dictionary worker panicked"))?
3115 })
3116 .collect::<Result<Vec<_>>>()
3117 })?;
3118 for (index, one) in pieces.into_iter().flatten() {
3119 closed[index] = Some(one);
3120 }
3121 Ok(closed)
3122 }
3123
3124 fn close_dictionary(
3131 &self,
3132 index: usize,
3133 dictionary: &GlobalDictionary,
3134 ) -> Result<ClosedDictionary> {
3135 let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
3136 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3140 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3141 let hosts = if self.table.fields[index].name.eq_ignore_ascii_case("Referer") {
3142 host::build(index, dictionary, &flat, &bases)?
3143 } else {
3144 None
3145 };
3146 drop(flat);
3147 drop(bases);
3148 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3149 let payload = dictionary
3150 .placed
3151 .iter()
3152 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3153 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3154 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3155 }
3156
3157 fn write_stats(&mut self) -> Result<()> {
3169 let gathers = std::mem::take(&mut self.gathers);
3170 let rows = self.table.rows as u64;
3171 let mut payloads = Vec::new();
3172 for (column, gather) in gathers.into_iter().enumerate() {
3173 let Some(gather) = gather else { continue };
3174 if gather.rows() != rows {
3180 continue;
3181 }
3182 let Some(stats) = gather.finish() else { continue };
3183 let mut summary = Vec::new();
3184 stats.summary.encode(&mut summary)?;
3185 let mut sketches = Vec::new();
3186 stats.sketches.encode(&mut sketches)?;
3187 payloads.push((column, summary, sketches));
3188 }
3189 if payloads.is_empty() {
3190 return Ok(());
3191 }
3192 let costs = payloads
3193 .iter()
3194 .map(|(_, summary, sketches)| summary.len() + sketches.len())
3195 .collect::<Vec<_>>();
3196 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3197 let keep = stats::within(&costs, allowance, 0);
3200 for ((column, summary, sketches), _) in
3201 payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3202 {
3203 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3204 for (kind, bytes, header_bytes) in [
3205 (*section::SUMMARY, summary, summary.len() as u32),
3208 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3209 ] {
3210 let written = write_section(
3211 &self.file,
3212 &mut self.at,
3213 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3214 self.generation,
3215 )?;
3216 self.table.sections.push(written);
3217 }
3218 }
3219 if self.table.sections.len() > MAX_SECTIONS {
3220 return Err(invalid("the table would name more sections than the bound allows"));
3221 }
3222 Ok(())
3223 }
3224
3225 pub fn finish(mut self) -> Result<Table> {
3235 let entry = self.close()?;
3236 let profile = self.profile.take();
3237 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3238 let mut tables = std::mem::take(&mut self.closed);
3239 tables.push(entry);
3240 let catalog = encode_catalog(&tables, &self.views)?;
3241 if catalog.len() > MAX_DIRECTORY {
3242 return Err(invalid("catalog exceeds the configured bound"));
3243 }
3244 let offset = self.at;
3245 self.put(&catalog)?;
3246 if let Some(profile) = &profile {
3247 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3248 }
3249 synced(&self.file, profile.as_deref())?;
3253 let slot = Slot {
3254 offset,
3255 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3256 generation: self.generation,
3257 hash: checksum(&catalog),
3258 };
3259 write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
3264 synced(&self.file, profile.as_deref())?;
3265 Ok(self.table)
3266 }
3267
3268 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3285 let path = path.as_ref();
3286 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3287 let (closed, _) = decode_catalog(&bytes, size)?;
3288 let generation = slot
3289 .generation
3290 .checked_add(1)
3291 .ok_or_else(|| invalid("native file generation overflow"))?;
3292 let catalog = encode_catalog(&closed, views)?;
3293 if catalog.len() > MAX_DIRECTORY {
3294 return Err(invalid("catalog exceeds the configured bound"));
3295 }
3296 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3297 write_at(&file, size, &catalog)?;
3298 file.sync_all().map_err(io)?;
3299 let slot = Slot {
3300 offset: size,
3301 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3302 generation,
3303 hash: checksum(&catalog),
3304 };
3305 write_at(&file, slot_offset(generation), &slot.bytes())?;
3306 file.sync_all().map_err(io)?;
3307 Ok(())
3308 }
3309
3310 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3313 let path = path.as_ref();
3314 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3315 let (mut entries, views) = decode_catalog(&bytes, size)?;
3316 let native = Catalog::open(path)?;
3317 for entry in &mut entries {
3318 let reader = native.table(&entry.name)?;
3319 entry.nonzero = reader_nonzero_counts(&reader)?;
3320 entry.aggregates = reader_aggregate_sums(&reader)?;
3321 entry.distincts = (0..entry.fields.len())
3322 .map(|column| reader.distinct_values(column))
3323 .collect::<Result<Vec<_>>>()?;
3324 entry.extremes = reader_integer_extremes(&reader)?;
3325 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3326 }
3327 let generation = slot
3328 .generation
3329 .checked_add(1)
3330 .ok_or_else(|| invalid("native file generation overflow"))?;
3331 let catalog = encode_catalog(&entries, &views)?;
3332 if catalog.len() > MAX_DIRECTORY {
3333 return Err(invalid("catalog exceeds the configured bound"));
3334 }
3335 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3336 write_at(&file, size, &catalog)?;
3337 file.sync_all().map_err(io)?;
3338 let slot = Slot {
3339 offset: size,
3340 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3341 generation,
3342 hash: checksum(&catalog),
3343 };
3344 write_at(&file, slot_offset(generation), &slot.bytes())?;
3345 file.sync_all().map_err(io)?;
3346 Ok(())
3347 }
3348
3349 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3351 Self::certify_summaries(path)
3352 }
3353}
3354
3355fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3361 let offset = *at;
3362 write_at(file, offset, bytes)?;
3363 *at =
3364 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3365 Ok(offset)
3366}
3367
3368fn write_section(
3374 file: &File,
3375 at: &mut u64,
3376 one: §ion::Attachment<'_>,
3377 generation: u64,
3378) -> Result<Section> {
3379 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3383 return Err(invalid("a section's header is longer than its payload"));
3384 }
3385 let mut extents = Vec::new();
3386 let mut first = 0_u64;
3387 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3388 let offset = append(file, at, chunk)?;
3389 extents.push(section::Extent {
3390 offset,
3391 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3392 hash: checksum(chunk),
3393 first,
3394 });
3395 first += chunk.len() as u64;
3396 }
3397 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3398 section::encode_extents(&extents, &mut table)?;
3399 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3403 Ok(Section {
3404 kind: one.kind,
3405 id: one.id,
3406 generation,
3407 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3408 extent_page,
3409 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3410 hash: checksum(&table),
3411 flags: one.flags,
3412 header_bytes: one.header_bytes,
3413 })
3414}
3415
3416pub fn attach(
3440 path: impl AsRef<Path>,
3441 table: &str,
3442 attachments: &[section::Attachment<'_>],
3443) -> Result<Table> {
3444 let path = path.as_ref();
3445 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3446 let (mut entries, views) = decode_catalog(&bytes, size)?;
3447 let at = entries
3448 .iter()
3449 .position(|entry| entry.name == table)
3450 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3451 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3452 let mut version = [0; 4];
3453 read_at(&file, 8, &mut version)?;
3454 let version = u32::from_le_bytes(version);
3455 if version != FORMAT {
3461 return Err(invalid(&format!(
3462 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3463 to be written again"
3464 )));
3465 }
3466 let mut directory = vec![0; entries[at].directory.length as usize];
3467 read_at(&file, entries[at].directory.offset, &mut directory)?;
3468 if checksum(&directory) != entries[at].directory.hash {
3469 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3470 }
3471 let mut held = decode_directory(&directory, size)?;
3472 let mut cursor = size;
3473 for one in attachments {
3474 let written = write_section(&file, &mut cursor, one, held.generation)?;
3475 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3476 held.sections.push(written);
3477 }
3478 if held.sections.len() > MAX_SECTIONS {
3479 return Err(invalid("the table would name more sections than the bound allows"));
3480 }
3481 let encoded = encode_directory(&held)?;
3482 if encoded.len() > MAX_DIRECTORY {
3483 return Err(invalid("directory exceeds the configured bound"));
3484 }
3485 let offset = append(&file, &mut cursor, &encoded)?;
3486 entries[at].directory = Page {
3487 offset,
3488 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3489 hash: checksum(&encoded),
3490 };
3491 let catalog = encode_catalog(&entries, &views)?;
3494 if catalog.len() > MAX_DIRECTORY {
3495 return Err(invalid("catalog exceeds the configured bound"));
3496 }
3497 let offset = append(&file, &mut cursor, &catalog)?;
3498 file.sync_all().map_err(io)?;
3499 let generation =
3500 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3501 let committed = Slot {
3502 offset,
3503 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3504 generation,
3505 hash: checksum(&catalog),
3506 };
3507 write_at(&file, slot_offset(generation), &committed.bytes())?;
3508 file.sync_all().map_err(io)?;
3509 Ok(held)
3510}
3511
3512type Synopsis = Arc<Vec<(Value, u64)>>;
3515
3516#[derive(Debug, Clone)]
3518pub struct Reader {
3519 file: Arc<File>,
3520 table: Arc<Table>,
3521 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3522 loading: Arc<Vec<Mutex<()>>>,
3531 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3534 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3538 opened: Arc<AtomicUsize>,
3542 sieves: Arc<Vec<Vec<SieveSlot>>>,
3546 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3549 places: Arc<Vec<Place>>,
3551 cache: Arc<Shelf>,
3552 pool: PagePool,
3554 pages: Arc<AtomicUsize>,
3557 indexes: Arc<AtomicUsize>,
3560 size: u64,
3562 directory: u64,
3564 opening: Opening,
3566}
3567
3568#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3580pub struct Opening {
3581 pub reads: u32,
3584 pub bytes: u64,
3586}
3587
3588#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3590pub struct Reads {
3591 pub opening: Opening,
3593 pub pages: usize,
3595 pub indexes: usize,
3597 pub dictionaries: usize,
3600}
3601
3602#[derive(Debug, Clone, Copy)]
3604struct Place {
3605 stripe: u32,
3606 part: u32,
3607 rows: u32,
3608}
3609
3610#[derive(Debug, Clone, Copy)]
3612struct PartSpan {
3613 start: usize,
3614 length: usize,
3615 hash: u64,
3616}
3617
3618#[derive(Debug, Clone)]
3624struct CachedColumn {
3625 stripe: usize,
3626 index: Arc<Vec<PartSpan>>,
3627 page: Option<Arc<Vec<u8>>>,
3628}
3629
3630#[derive(Debug, Default)]
3650struct Cached {
3651 pages: Vec<Option<Resident>>,
3652 loading: Vec<usize>,
3653 index: Vec<Option<Arc<Vec<PartSpan>>>>,
3654}
3655
3656#[derive(Debug, Clone)]
3658struct Resident {
3659 page: Arc<Vec<u8>>,
3660 used: Arc<AtomicBool>,
3661}
3662
3663#[derive(Debug)]
3665struct Shelf {
3666 columns: Vec<Mutex<Cached>>,
3667 held: Vec<AtomicUsize>,
3670 kept: AtomicUsize,
3673}
3674
3675#[derive(Debug, Clone, Default)]
3694pub struct PagePool {
3695 ring: Arc<Mutex<Ring>>,
3696 budget: Arc<AtomicUsize>,
3697}
3698
3699#[derive(Debug, Default)]
3700struct Ring {
3701 held: VecDeque<Held>,
3702 bytes: usize,
3703}
3704
3705#[derive(Debug)]
3710struct Held {
3711 shelf: Weak<Shelf>,
3712 column: usize,
3713 stripe: usize,
3714 bytes: usize,
3715 used: Arc<AtomicBool>,
3716}
3717
3718impl PagePool {
3719 #[must_use]
3721 pub fn new(budget: usize) -> Self {
3722 let pool = Self::default();
3723 pool.budget.store(budget, Atomic::Relaxed);
3724 pool
3725 }
3726
3727 #[must_use]
3733 pub fn bytes(&self) -> usize {
3734 self.ring.lock().map_or(0, |ring| ring.bytes)
3735 }
3736
3737 fn admit(&self, held: Held) {
3743 let budget = self.budget.load(Atomic::Relaxed);
3744 let mut gone = Vec::new();
3745 {
3746 let Ok(mut ring) = self.ring.lock() else { return };
3747 ring.bytes += held.bytes;
3748 ring.held.push_back(held);
3749 let mut looked = 0;
3752 let limit = ring.held.len();
3753 while ring.bytes > budget && looked < limit {
3754 looked += 1;
3755 let Some(entry) = ring.held.pop_front() else { break };
3756 let Some(shelf) = entry.shelf.upgrade() else {
3757 ring.bytes -= entry.bytes;
3758 continue;
3759 };
3760 if entry.used.swap(false, Atomic::Relaxed) {
3761 ring.held.push_back(entry);
3762 continue;
3763 }
3764 let count = &shelf.held[entry.column];
3765 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3766 ring.held.push_back(entry);
3767 continue;
3768 }
3769 count.fetch_sub(1, Atomic::Relaxed);
3770 ring.bytes -= entry.bytes;
3771 gone.push((shelf, entry));
3772 }
3773 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3776 if let Some(entry) = ring.held.pop_front() {
3777 ring.bytes -= entry.bytes;
3778 }
3779 }
3780 }
3781 for (shelf, entry) in gone {
3782 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3783 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3784 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3785 *slot = None;
3786 }
3787 }
3788 }
3789 }
3790}
3791
3792const CACHED_STRIPES_PER_COLUMN: usize = 4;
3804
3805type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3807
3808type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3809
3810#[derive(Debug)]
3811struct NativeText {
3812 file: Arc<File>,
3813 values: usize,
3815 offsets: Vec<u8>,
3824 offset_bits: usize,
3827 value_ends: OnceLock<Option<Vec<u32>>>,
3840 value_lens: OnceLock<Option<Vec<u32>>>,
3850 ends_asked: AtomicUsize,
3856 ranks: usize,
3858 rank_at: u64,
3862 rank_ends: Vec<u64>,
3866 rank_hashes: Vec<u64>,
3867 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3868 code_bits: usize,
3871 code_ranks: OnceLock<Option<Vec<u32>>>,
3878 starts: Vec<u64>,
3885 lengths: Vec<u64>,
3886 hashes: Vec<u64>,
3887 grams: Option<NativeGrams>,
3889 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3891 keep_budget: usize,
3894 payload_kept: AtomicUsize,
3902 swept: Vec<AtomicBool>,
3910 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3927}
3928
3929#[derive(Debug)]
3930struct NativeGrams {
3931 start: u64,
3932 length: usize,
3933 hash: u64,
3934 loaded: OnceLock<Result<Vec<u8>>>,
3935}
3936
3937const TEXT_SEARCH_MEMO: usize = 64;
3942
3943const TEXT_PAYLOAD_VALUES: usize = 1024;
3959
3960const TEXT_GRAM_BYTES: usize = 2048;
3963
3964fn gram_bits(bytes: &[u8]) -> [usize; 2] {
3966 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
3967 let mut first = original ^ (original >> 16);
3968 first = first.wrapping_mul(0x7feb_352d);
3969 first ^= first >> 15;
3970 let mut second = original ^ (original >> 17);
3971 second = second.wrapping_mul(0x846c_a68b);
3972 second ^= second >> 16;
3973 let mask = TEXT_GRAM_BYTES * 8 - 1;
3974 [(first as usize) & mask, (second as usize) & mask]
3975}
3976
3977const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
3998
3999fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
4005 let mut lens = Vec::with_capacity(ends.len());
4006 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4007 let mut start = 0;
4008 for &end in block {
4009 lens.push(end.checked_sub(start)?);
4010 start = end;
4011 }
4012 }
4013 Some(lens)
4014}
4015
4016const TEXT_OFFSET_RUN: usize = 512;
4023
4024const DICTIONARY_HEADER: usize = 16;
4027
4028const DICTIONARY_SCATTERED: u32 = 1 << 31;
4042const DICTIONARY_GRAMS: u32 = 1 << 30;
4044
4045const TEXT_RANK_BLOCK: usize = 512;
4056
4057const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4071
4072impl NativeText {
4073 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4080 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4081 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4082 Ok(Some(bytes.as_slice()))
4083 }
4084
4085 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4090 let len = self.lengths[block];
4091 let mut stored = vec![
4092 0;
4093 usize::try_from(len).map_err(|_| invalid(
4094 "global dictionary block does not fit in memory"
4095 ))?
4096 ];
4097 read_at(&self.file, self.starts[block], &mut stored)?;
4098 if checksum(&stored) != self.hashes[block] {
4099 return Err(invalid("global dictionary payload checksum differs"));
4100 }
4101 let first = block * TEXT_PAYLOAD_VALUES;
4102 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4103 let want = self.end_within(last - 1)? as usize;
4104 let values = string::decode_flat(&stored)?;
4105 if values.len() != last - first {
4106 return Err(invalid("global dictionary block holds the wrong value count"));
4107 }
4108 let bytes = values.into_bytes();
4109 if bytes.len() != want {
4110 return Err(invalid("global dictionary block decodes to the wrong length"));
4111 }
4112 Ok(bytes)
4113 }
4114
4115 fn ends_worth_unpacking(&self) -> usize {
4132 self.values.max(TEXT_PAYLOAD_VALUES)
4133 }
4134
4135 fn value_ends(&self) -> Option<&[u32]> {
4137 if let Some(built) = self.value_ends.get() {
4138 return built.as_deref();
4139 }
4140 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4141 return None;
4142 }
4143 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4144 }
4145
4146 fn unpack_ends(&self) -> Option<Vec<u32>> {
4152 let mut ends = vec![0u32; self.values];
4153 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4154 let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4155 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4156 u32::try_from(bits).unwrap_or(u32::MAX)
4157 })
4158 .ok()?;
4159 }
4160 if ends.contains(&u32::MAX) { None } else { Some(ends) }
4163 }
4164
4165 fn end_within(&self, index: usize) -> Result<u32> {
4167 if let Some(ends) = self.value_ends() {
4168 return ends
4169 .get(index)
4170 .copied()
4171 .ok_or_else(|| invalid("global dictionary offsets are short"));
4172 }
4173 let run = index / TEXT_OFFSET_RUN;
4174 let bytes = self
4175 .offsets
4176 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4177 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4178 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4179 .map_err(|_| invalid("global dictionary offsets are short"))?;
4180 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4181 }
4182
4183 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4201 let mut ends = vec![0u64; last.saturating_sub(first)];
4202 let mut scratch = Vec::new();
4203 let mut at = first;
4204 while at < last {
4205 let run = at / TEXT_OFFSET_RUN;
4206 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4207 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4208 let bytes = self
4209 .offsets
4210 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4211 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4212 let from = at % TEXT_OFFSET_RUN;
4213 let upto = stop - run * TEXT_OFFSET_RUN;
4214 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4215 return Err(invalid("global dictionary offsets are short"));
4216 }
4217 let into = &mut ends[at - first..stop - first];
4218 if from == 0 {
4219 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4220 .map_err(|_| invalid("global dictionary offsets are short"))?;
4221 } else {
4222 scratch.resize(held, 0);
4223 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4224 .map_err(|_| invalid("global dictionary offsets are short"))?;
4225 into.copy_from_slice(&scratch[from..upto]);
4226 }
4227 at = stop;
4228 }
4229 Ok(ends)
4230 }
4231
4232 fn start_within(&self, index: usize) -> Result<u32> {
4235 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4236 }
4237
4238 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4246 if let Some(ends) = self.value_ends() {
4247 let end =
4248 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4249 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4252 if start > end {
4253 return Err(invalid("global dictionary value ends before it starts"));
4254 }
4255 return Ok((start, end));
4256 }
4257 let within = index % TEXT_OFFSET_RUN;
4258 let (start, end) = if within == 0 {
4259 (self.start_within(index)?, self.end_within(index)?)
4260 } else {
4261 let run = index / TEXT_OFFSET_RUN;
4262 let bytes = self
4263 .offsets
4264 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4265 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4266 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4267 .map_err(|_| invalid("global dictionary offsets are short"))?;
4268 let ends = u32::try_from(end)
4269 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4270 let starts = u32::try_from(start)
4271 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4272 (starts, ends)
4273 };
4274 if start > end {
4275 return Err(invalid("global dictionary value ends before it starts"));
4276 }
4277 Ok((start, end))
4278 }
4279
4280 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4287 let slot = self
4288 .rank_blocks
4289 .get(rank / TEXT_RANK_BLOCK)
4290 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4291 let block = slot
4292 .get_or_init(|| {
4293 let which = rank / TEXT_RANK_BLOCK;
4294 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4295 let end = self.rank_ends[which];
4296 let mut bytes = vec![0; (end - start) as usize];
4297 read_at(&self.file, self.rank_at + start, &mut bytes)?;
4298 if checksum(&bytes)
4299 != *self
4300 .rank_hashes
4301 .get(rank / TEXT_RANK_BLOCK)
4302 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
4303 {
4304 return Err(invalid("global dictionary rank checksum differs"));
4305 }
4306 Ok(bytes)
4307 })
4308 .as_ref()
4309 .map_err(Clone::clone)?;
4310 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4311 }
4312
4313 fn head_at(&self, rank: usize) -> Result<u64> {
4315 let (block, within) = self.rank_parts(rank)?;
4316 let (base, width, packed) = rank_heads(block)?;
4317 let above = bitpack::tail_at(packed, width, within)
4318 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4319 Ok(base.wrapping_add(above))
4320 }
4321
4322 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4324 let (_, width, packed) = rank_heads(block)?;
4325 packed
4326 .get(bitpack::tail_len(count, width)..)
4327 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4328 }
4329
4330 fn rank_block_len(&self, rank: usize) -> usize {
4332 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4333 TEXT_RANK_BLOCK.min(self.ranks - first)
4334 }
4335}
4336
4337fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4339 let header = block
4340 .get(..RANK_BLOCK_HEADER)
4341 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4342 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4343 let width = header[8] as usize;
4344 if width > 64 {
4345 return Err(invalid("global dictionary rank block packs heads past a word"));
4346 }
4347 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4348}
4349
4350fn offset_width(ends: &[u32]) -> usize {
4357 let span = ends.iter().copied().max().unwrap_or(0);
4361 (u32::BITS - span.leading_zeros()) as usize
4362}
4363
4364fn offset_bytes(values: usize, bits: usize) -> usize {
4367 let full = values / TEXT_OFFSET_RUN;
4368 let rest = values % TEXT_OFFSET_RUN;
4369 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4370}
4371
4372fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4376 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4377 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4378 run.clear();
4379 run.extend(chunk.iter().map(|&end| u64::from(end)));
4380 bitpack::pack_tail(&run, bits, out)
4381 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4382 }
4383 Ok(())
4384}
4385
4386fn code_width(values: usize) -> usize {
4388 match u64::try_from(values).unwrap_or(u64::MAX) {
4389 0 | 1 => 0,
4390 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4391 }
4392}
4393
4394impl TextSource for NativeText {
4395 fn len(&self) -> usize {
4396 self.values
4397 }
4398
4399 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4400 let Some(grams) = &self.grams else { return Ok(true) };
4401 if literal.len() < 4 || first >= self.values {
4402 return Ok(true);
4403 }
4404 let bytes = grams
4405 .loaded
4406 .get_or_init(|| {
4407 let mut bytes = vec![0; grams.length];
4408 read_at(&self.file, grams.start, &mut bytes)?;
4409 if checksum(&bytes) != grams.hash {
4410 return Err(invalid("global dictionary substring signatures checksum differs"));
4411 }
4412 Ok(bytes)
4413 })
4414 .as_ref()
4415 .map_err(Clone::clone)?;
4416 let block = first / TEXT_PAYLOAD_VALUES;
4417 let Some(bits) = bytes.get(block * TEXT_GRAM_BYTES..(block + 1) * TEXT_GRAM_BYTES) else {
4418 return Ok(true);
4419 };
4420 Ok(literal.windows(4).all(|gram| {
4421 gram_bits(gram).into_iter().all(|bit| bits[bit / 8] & (1 << (bit % 8)) != 0)
4422 }))
4423 }
4424
4425 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4426 if index >= self.values {
4427 return Ok(None);
4428 }
4429 let (start, end) = self.span_within(index)?;
4430 if start == end {
4431 return Ok(Some(&[]));
4432 }
4433 let block = index / TEXT_PAYLOAD_VALUES;
4436 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4437 Ok(bytes.get(start as usize..end as usize))
4438 }
4439
4440 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4441 if index >= self.values {
4442 return Ok(None);
4443 }
4444 let (start, end) = self.span_within(index)?;
4445 Ok(Some((end - start) as usize))
4446 }
4447
4448 fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4455 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4456 let Some(ends) = self.value_ends() else {
4457 for (slot, &index) in into.iter_mut().zip(indices) {
4458 *slot = self
4459 .bytes_len_at(index as usize)?
4460 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4461 }
4462 return Ok(());
4463 };
4464 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4465 for (slot, &index) in into.iter_mut().zip(indices) {
4466 *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4469 }
4470 return Ok(());
4471 }
4472 for (slot, &index) in into.iter_mut().zip(indices) {
4473 let index = index as usize;
4474 let Some(&end) = ends.get(index) else {
4476 *slot = 0;
4477 continue;
4478 };
4479 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4480 if start > end {
4481 return Err(invalid("global dictionary value ends before it starts"));
4482 }
4483 *slot = i64::from(end - start);
4484 }
4485 Ok(())
4486 }
4487
4488 fn sweep(
4501 &self,
4502 first: usize,
4503 limit: usize,
4504 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4505 ) -> Result<usize> {
4506 let limit = limit.min(self.values);
4507 if first >= limit {
4508 return Ok(first);
4509 }
4510 let block = first / TEXT_PAYLOAD_VALUES;
4511 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4512 let decoded;
4513 let kept = self.blocks.get(block).and_then(OnceLock::get);
4514 let again = kept.is_none()
4515 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4516 let bytes: &[u8] = match kept {
4517 Some(Ok(kept)) => kept,
4518 _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4519 let kept = self
4520 .payload_block(block)?
4521 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4522 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4523 kept
4524 }
4525 _ => {
4526 decoded = self.decode_block(block)?;
4527 &decoded
4528 }
4529 };
4530 let ends = self.ends_within(first, last)?;
4531 if ends.len() != last - first {
4532 return Err(invalid("global dictionary offsets are short"));
4533 }
4534 let mut start = u64::from(self.start_within(first)?);
4535 for (index, &end) in (first..last).zip(&ends) {
4538 let value = usize::try_from(start)
4539 .ok()
4540 .zip(usize::try_from(end).ok())
4541 .and_then(|(from, to)| bytes.get(from..to))
4542 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4543 body(index, value)?;
4544 start = end;
4545 }
4546 Ok(last)
4547 }
4548
4549 fn visit(
4555 &self,
4556 indices: &[usize],
4557 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4558 ) -> Result<()> {
4559 let mut at = 0;
4560 while at < indices.len() {
4561 let block = indices[at] / TEXT_PAYLOAD_VALUES;
4562 let upto =
4563 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4564 let wanted = &indices[at..upto];
4565 if wanted.iter().any(|&index| index >= self.values) {
4566 return Err(invalid("a visited value is past the global dictionary"));
4567 }
4568 let decoded;
4569 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4570 Some(Ok(kept)) => kept,
4571 _ => {
4572 decoded = self.decode_block(block)?;
4573 &decoded
4574 }
4575 };
4576 for (offset, &index) in wanted.iter().enumerate() {
4577 let (start, end) = self.span_within(index)?;
4578 let value = bytes
4579 .get(start as usize..end as usize)
4580 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4581 body(at + offset, value)?;
4582 }
4583 at = upto;
4584 }
4585 Ok(())
4586 }
4587
4588 fn ranks(&self) -> Option<usize> {
4589 (self.ranks > 0).then_some(self.ranks)
4590 }
4591
4592 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4600 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4601 if let Some(&answer) = memo.get(wanted) {
4602 return Ok(answer);
4603 }
4604 let answer = search_below(self, ranks, wanted)?;
4605 if memo.len() >= TEXT_SEARCH_MEMO {
4606 memo.clear();
4607 }
4608 memo.insert(wanted.to_vec(), answer);
4609 Ok(answer)
4610 }
4611
4612 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4613 let settled = self.head_at(rank)?.cmp(&head(wanted));
4617 if settled != Ordering::Equal {
4618 return Ok(settled);
4619 }
4620 let code = self.code_at_rank(rank)?;
4621 let bytes = self
4622 .bytes_at(code as usize)?
4623 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4624 Ok(bytes.cmp(wanted))
4625 }
4626
4627 fn code_at_rank(&self, rank: usize) -> Result<u32> {
4628 let (block, within) = self.rank_parts(rank)?;
4629 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4630 let code = bitpack::tail_at(codes, self.code_bits, within)
4631 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4632 let code = u32::try_from(code)
4633 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4634 if code as usize >= self.len() {
4635 return Err(invalid("global dictionary order names a code it does not have"));
4636 }
4637 Ok(code)
4638 }
4639
4640 fn code_ranks(&self) -> Option<&[u32]> {
4641 if self.ranks == 0 || self.ranks != self.len() {
4645 return None;
4646 }
4647 self.code_ranks
4648 .get_or_init(|| {
4649 let mut ranks = vec![u32::MAX; self.ranks];
4650 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4653 let (block, _) = self.rank_parts(first).ok()?;
4654 let count = self.rank_block_len(first);
4655 let codes = self.rank_codes(block, count).ok()?;
4656 for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4657 .ok()?
4658 .into_iter()
4659 .enumerate()
4660 {
4661 let code = usize::try_from(code).ok()?;
4662 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4663 }
4664 }
4665 if ranks.contains(&u32::MAX) {
4666 return None;
4667 }
4668 Some(ranks)
4669 })
4670 .as_deref()
4671 }
4672
4673 fn footprint(&self) -> usize {
4674 self.offsets.capacity()
4675 + self
4676 .value_ends
4677 .get()
4678 .and_then(Option::as_ref)
4679 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4680 + self
4681 .value_lens
4682 .get()
4683 .and_then(Option::as_ref)
4684 .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4685 + self
4686 .code_ranks
4687 .get()
4688 .and_then(Option::as_ref)
4689 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4690 + self.rank_hashes.capacity() * size_of::<u64>()
4691 + self.rank_ends.capacity() * size_of::<u64>()
4692 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4693 + self
4694 .rank_blocks
4695 .iter()
4696 .filter_map(OnceLock::get)
4697 .filter_map(|result| result.as_ref().ok())
4698 .map(Vec::capacity)
4699 .sum::<usize>()
4700 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4701 + self.hashes.capacity() * size_of::<u64>()
4702 + self.starts.capacity() * size_of::<u64>()
4703 + self.lengths.capacity() * size_of::<u64>()
4704 + self
4705 .grams
4706 .as_ref()
4707 .and_then(|grams| grams.loaded.get())
4708 .and_then(|result| result.as_ref().ok())
4709 .map_or(0, Vec::capacity)
4710 + self
4711 .blocks
4712 .iter()
4713 .filter_map(OnceLock::get)
4714 .filter_map(|result| result.as_ref().ok())
4715 .map(Vec::capacity)
4716 .sum::<usize>()
4717 }
4718}
4719
4720fn places(table: &Table) -> Result<Vec<Place>> {
4722 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4723 for (at, stripe) in table.stripes.iter().enumerate() {
4724 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4725 for (part, &rows) in stripe.parts.iter().enumerate() {
4726 places.push(Place {
4727 stripe: index,
4728 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4729 rows,
4730 });
4731 }
4732 }
4733 Ok(places)
4734}
4735
4736fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4741 let parts = stripe.parts.len();
4742 let section = index_section(parts)?;
4743 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4744 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4745 if end > stripe.index.length as usize {
4746 return Err(invalid("index page is shorter than its columns"));
4747 }
4748 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4749 let mut bytes = vec![0; section];
4750 let offset = stripe
4751 .index
4752 .offset
4753 .checked_add(at as u64)
4754 .ok_or_else(|| invalid("index page offset overflow"))?;
4755 read_at(file, offset, &mut bytes)?;
4756 let entries = section - size_of::<u64>();
4757 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4758 if checksum(&bytes[..entries]) != stored {
4759 return Err(invalid(&format!(
4762 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4763 wanted {stored:016x} and got {:016x}",
4764 checksum(&bytes[..entries]),
4765 )));
4766 }
4767 let mut spans = Vec::with_capacity(parts);
4768 let mut start = 0_usize;
4769 for part in 0..parts {
4770 let at = part * INDEX_ENTRY;
4771 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4772 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4773 spans.push(PartSpan { start, length, hash });
4774 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4775 }
4776 if start != page.length as usize {
4777 return Err(invalid("column page length differs from its index"));
4778 }
4779 Ok(spans)
4780}
4781
4782fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4784 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4785 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4786}
4787
4788fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4794 if let Some(slot) = cached.index.get_mut(held.stripe) {
4795 if slot.is_none() {
4796 *slot = Some(Arc::clone(&held.index));
4797 }
4798 }
4799 let page = held.page.clone()?;
4800 let slot = cached.pages.get_mut(held.stripe)?;
4801 if slot.is_some() {
4802 return None;
4803 }
4804 let bytes = page.len();
4805 let used = Arc::new(AtomicBool::new(true));
4808 *slot = Some(Resident { page, used: Arc::clone(&used) });
4809 Some((bytes, used))
4810}
4811
4812#[derive(Debug, Clone)]
4821pub struct Catalog {
4822 file: Arc<File>,
4823 size: u64,
4824 entries: Arc<Vec<Entry>>,
4825 views: Arc<Vec<ViewEntry>>,
4827 opening: Opening,
4828 pool: PagePool,
4830}
4831
4832#[derive(Debug, Clone, PartialEq, Eq)]
4834pub struct CertifiedSums {
4835 pub columns: Vec<(i128, u64)>,
4836 pub rows: u64,
4837}
4838
4839#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4841pub enum IntegerExtremes {
4842 Null,
4843 Values { low: i128, high: i128 },
4844}
4845
4846pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
4848
4849impl Catalog {
4850 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4859 Self::open_in(path, &PagePool::default())
4860 }
4861
4862 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4868 let (file, size, _, bytes, opening) = slot_bytes(path)?;
4869 let (entries, views) = decode_catalog(&bytes, size)?;
4870 Ok(Self {
4871 file: Arc::new(file),
4872 size,
4873 entries: Arc::new(entries),
4874 views: Arc::new(views),
4875 opening,
4876 pool: pool.clone(),
4877 })
4878 }
4879
4880 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4882 self.entries.iter().map(|entry| entry.name.as_str())
4883 }
4884
4885 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4892 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4893 }
4894
4895 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4901 self.views.iter()
4902 }
4903
4904 #[must_use]
4906 pub fn len(&self) -> usize {
4907 self.entries.len()
4908 }
4909
4910 #[must_use]
4913 pub fn is_empty(&self) -> bool {
4914 self.entries.is_empty()
4915 }
4916
4917 pub fn table(&self, name: &str) -> Result<Reader> {
4923 let entry = self
4924 .entries
4925 .iter()
4926 .find(|entry| entry.name == name)
4927 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4928 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4932 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4933 return Err(invalid(&format!("the directory of table {name} does not checksum")));
4934 }
4935 let mut opening = self.opening;
4936 opening.reads += 1;
4937 opening.bytes += u64::from(entry.directory.length);
4938 Reader::build(
4939 Arc::clone(&self.file),
4940 self.size,
4941 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
4942 u64::from(entry.directory.length),
4943 opening,
4944 self.pool.clone(),
4945 )
4946 }
4947
4948 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
4952 let entry = self
4953 .entries
4954 .iter()
4955 .find(|entry| entry.name == name)
4956 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4957 let Some(field) = entry.fields.get(column) else {
4958 return Err(invalid("frequency column index out of range"));
4959 };
4960 if !matches!(
4961 field.ty,
4962 LogicalType::TinyInt
4963 | LogicalType::SmallInt
4964 | LogicalType::Integer
4965 | LogicalType::BigInt
4966 | LogicalType::UTinyInt
4967 | LogicalType::USmallInt
4968 | LogicalType::UInteger
4969 | LogicalType::UBigInt
4970 ) {
4971 return Ok(None);
4972 }
4973 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
4974 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
4975 return Err(invalid(&format!("the directory of table {name} does not checksum")));
4976 }
4977 if let Some(count) = entry.nonzero.get(column).copied().flatten() {
4978 return Ok(Some(count));
4979 }
4980 quick_nonzero(
4981 Cursor::over(&self.file, offset, length),
4982 &entry.name,
4983 &entry.fields,
4984 entry.rows,
4985 column,
4986 )
4987 }
4988
4989 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
4992 let entry = self
4993 .entries
4994 .iter()
4995 .find(|entry| entry.name == name)
4996 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4997 let mut sums = Vec::with_capacity(columns.len());
4998 for &column in columns {
4999 let Some(field) = entry.fields.get(column) else {
5000 return Err(invalid("aggregate column index out of range"));
5001 };
5002 if !signed_integer(&field.ty) {
5003 return Ok(None);
5004 }
5005 let Some(sum) = entry.aggregates[column] else {
5006 return Ok(None);
5007 };
5008 sums.push(sum);
5009 }
5010 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5011 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5012 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5013 }
5014 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5015 }
5016
5017 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5019 let entry = self
5020 .entries
5021 .iter()
5022 .find(|entry| entry.name == name)
5023 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5024 let Some(count) = entry.distincts.get(column).copied() else {
5025 return Err(invalid("distinct column index out of range"));
5026 };
5027 let Some(count) = count else { return Ok(None) };
5028 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5029 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5030 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5031 }
5032 Ok(Some(count))
5033 }
5034
5035 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5037 let entry = self
5038 .entries
5039 .iter()
5040 .find(|entry| entry.name == name)
5041 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5042 let Some(extremes) = entry.extremes.get(column).copied() else {
5043 return Err(invalid("extremes column index out of range"));
5044 };
5045 let Some(extremes) = extremes else { return Ok(None) };
5046 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5047 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5048 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5049 }
5050 Ok(Some(match extremes {
5051 None => IntegerExtremes::Null,
5052 Some((low, high)) => IntegerExtremes::Values { low, high },
5053 }))
5054 }
5055
5056 pub fn exact_numeric_frequencies(
5058 &self,
5059 name: &str,
5060 column: usize,
5061 ) -> Result<Option<NumericFrequencies>> {
5062 let entry = self
5063 .entries
5064 .iter()
5065 .find(|entry| entry.name == name)
5066 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5067 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5068 return Err(invalid("numeric frequency column index out of range"));
5069 };
5070 let Some(frequencies) = frequencies else { return Ok(None) };
5071 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5072 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5073 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5074 }
5075 Ok(Some(frequencies))
5076 }
5077
5078 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5080 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5081 }
5082}
5083
5084fn slot_offset(generation: u64) -> u64 {
5089 16 + (generation - 1) % 2 * SLOT_BYTES as u64
5090}
5091
5092fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5097 let mut file = File::open(path).map_err(io)?;
5098 let size = file.metadata().map_err(io)?.len();
5099 if size < HEADER {
5100 return Err(invalid("file is shorter than its header"));
5101 }
5102 let mut header = [0; HEADER as usize];
5103 file.read_exact(&mut header).map_err(io)?;
5104 let mut opening = Opening { reads: 1, bytes: HEADER };
5105 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5106 if &header[..8] != MAGIC {
5111 return Err(invalid("the header does not begin with a rudb native magic"));
5112 }
5113 if !READABLE.contains(&version) {
5114 return Err(invalid(&format!(
5115 "the file is format {version} and this build reads format {FORMAT}, so it has to \
5116 be written again"
5117 )));
5118 }
5119 let mut selected = None;
5120 for start in [16, 16 + SLOT_BYTES] {
5121 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5122 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5123 continue;
5124 }
5125 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5126 if slot.offset < HEADER || end > size {
5127 continue;
5128 }
5129 let mut bytes = vec![0; slot.length as usize];
5130 file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
5131 file.read_exact(&mut bytes).map_err(io)?;
5132 opening.reads += 1;
5133 opening.bytes += u64::from(slot.length);
5134 if checksum(&bytes) == slot.hash
5135 && selected
5136 .as_ref()
5137 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5138 {
5139 selected = Some((slot, bytes));
5140 }
5141 }
5142 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5143 Ok((file, size, slot, bytes, opening))
5144}
5145
5146impl Reader {
5147 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5154 let catalog = Catalog::open(path)?;
5155 let mut names = catalog.names();
5156 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5157 if names.next().is_some() {
5158 return Err(invalid(
5159 "the file holds more than one table, so it has to be opened by name",
5160 ));
5161 }
5162 catalog.table(&name)
5163 }
5164
5165 fn build(
5167 file: Arc<File>,
5168 size: u64,
5169 table: Table,
5170 directory: u64,
5171 opening: Opening,
5172 pool: PagePool,
5173 ) -> Result<Self> {
5174 let places = places(&table)?;
5175 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5176 let table_fields = table.fields.len();
5177 let stripes = table.stripes.len();
5178 let columns = (0..table.fields.len())
5179 .map(|_| {
5180 Mutex::new(Cached {
5181 pages: (0..stripes).map(|_| None).collect(),
5182 index: (0..stripes).map(|_| None).collect(),
5183 ..Cached::default()
5184 })
5185 })
5186 .collect::<Vec<_>>();
5187 let cache = Shelf {
5188 columns,
5189 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5190 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5191 };
5192 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5193 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5194 .collect();
5195 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5196 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5197 .collect();
5198 Ok(Self {
5199 file,
5200 table: Arc::new(table),
5201 dictionaries: Arc::new(dictionaries),
5202 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5203 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5204 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5205 opened: Arc::new(AtomicUsize::new(0)),
5206 sieves: Arc::new(sieves),
5207 part_ranges: Arc::new(part_ranges),
5208 places: Arc::new(places),
5209 cache: Arc::new(cache),
5210 pool,
5211 pages: Arc::new(AtomicUsize::new(0)),
5212 indexes: Arc::new(AtomicUsize::new(0)),
5213 size,
5214 directory,
5215 opening,
5216 })
5217 }
5218
5219 #[must_use]
5226 pub fn reads(&self) -> Reads {
5227 Reads {
5228 opening: self.opening,
5229 pages: self.pages.load(Atomic::Relaxed),
5230 indexes: self.indexes.load(Atomic::Relaxed),
5231 dictionaries: self.opened.load(Atomic::Relaxed),
5232 }
5233 }
5234
5235 #[must_use]
5240 pub fn layout(&self) -> Layout {
5241 let table = &self.table;
5242 let stripes = table.stripes.as_slice();
5243 let columns = table
5244 .fields
5245 .iter()
5246 .enumerate()
5247 .map(|(at, field)| ColumnLayout {
5248 name: field.name.clone(),
5249 kind: field.ty.to_string(),
5250 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
5251 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
5252 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
5253 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
5254 dictionary: dictionary_bytes(table, at),
5255 })
5256 .collect();
5257 Layout {
5258 file: self.size,
5259 rows: table.rows,
5260 stripes: stripes.len(),
5261 parts: self.places.len(),
5262 columns,
5263 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
5264 directory: self.directory,
5265 header: HEADER,
5266 }
5267 }
5268
5269 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
5286 let field = self
5287 .table
5288 .fields
5289 .get(column)
5290 .ok_or_else(|| invalid("stored column index out of range"))?;
5291 let mut stored = Vec::with_capacity(self.places.len());
5292 let mut row = 0;
5293 for (at, stripe) in self.table.stripes.iter().enumerate() {
5294 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5295 let index = read_index(&self.file, stripe, column)?;
5296 let mut bytes = vec![0; page.length as usize];
5297 read_at(&self.file, page.offset, &mut bytes)?;
5298 let ranges = self.stripe_part_ranges(at, column);
5299 for (part, &rows) in stripe.parts.iter().enumerate() {
5300 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
5301 let held = part_bytes(&bytes, span)?;
5302 let range = ranges.and_then(|held| held.get(part));
5303 stored.push(StoredPart {
5304 stripe: at,
5305 part,
5306 row,
5307 rows: rows as usize,
5308 encoding: page_encoding(&field.ty, rows as usize, held),
5309 bytes: span.length as u64,
5310 page: page.offset,
5311 offset: span.start as u64,
5312 low: range
5313 .and_then(|range| range.low.clone())
5314 .and_then(|bound| bound.into_value(&field.ty)),
5315 high: range
5316 .and_then(|range| range.high.clone())
5317 .and_then(|bound| bound.into_value(&field.ty)),
5318 nulls: range.map(|range| range.nulls),
5319 });
5320 row += rows as usize;
5321 }
5322 }
5323 Ok(stored)
5324 }
5325
5326 #[must_use]
5328 pub fn parts(&self) -> usize {
5329 self.places.len()
5330 }
5331
5332 #[must_use]
5339 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
5340 let mut runs = Vec::with_capacity(self.table.stripes.len());
5341 let mut start = 0;
5342 for stripe in &self.table.stripes {
5343 let end = start + stripe.parts.len();
5344 runs.push(start..end);
5345 start = end;
5346 }
5347 runs
5348 }
5349
5350 #[must_use]
5355 pub fn stripe_rows(&self, stripe: usize) -> usize {
5356 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
5357 }
5358
5359 pub fn keep_stripes(&self, stripes: usize) {
5366 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
5367 }
5368
5369 #[must_use]
5371 pub fn part_rows(&self, at: usize) -> usize {
5372 self.places.get(at).map_or(0, |place| place.rows as usize)
5373 }
5374
5375 #[must_use]
5377 pub fn table(&self) -> &Table {
5378 &self.table
5379 }
5380
5381 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
5390 let field = self
5391 .table
5392 .fields
5393 .get(column)
5394 .ok_or_else(|| invalid("frequency column index out of range"))?;
5395 let Some(summary) = self.frequency_summary(column)? else {
5396 return Ok(None);
5397 };
5398 if top == 0 || summary.entries.len() < top {
5399 return Ok(None);
5400 }
5401 let boundary = summary.entries[top - 1].count;
5402 if boundary <= summary.omitted_max {
5403 return Ok(None);
5404 }
5405 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5406 }
5407
5408 pub fn top_pair_frequencies(
5419 &self,
5420 first: usize,
5421 second: usize,
5422 top: usize,
5423 ) -> Result<Option<PairFrequencyCounts>> {
5424 if first >= self.table.fields.len() || second >= self.table.fields.len() {
5425 return Err(invalid("pair frequency column index out of range"));
5426 }
5427 let Some(summary) =
5428 self.table.pair_frequencies.iter().find(|summary| {
5429 summary.first as usize == first && summary.second as usize == second
5430 })
5431 else {
5432 return Ok(None);
5433 };
5434 if top == 0 || summary.entries.len() < top {
5435 return Ok(None);
5436 }
5437 let boundary = summary.entries[top - 1].count;
5438 if boundary <= summary.omitted_max {
5439 return Ok(None);
5440 }
5441 let first_summary = self
5442 .frequency_summary(first)?
5443 .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5444 let anchors = self
5445 .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5446 .into_iter()
5447 .map(|(value, _)| value)
5448 .collect::<Vec<_>>();
5449 let dictionary = self
5450 .dictionary(second)?
5451 .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5452 let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5453 codes.sort_unstable();
5454 codes.dedup();
5455 let texts = dictionary
5456 .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5457 let mut out = Vec::with_capacity(summary.entries.len());
5458 for entry in &summary.entries {
5459 if entry.count < boundary {
5460 break;
5461 }
5462 let first = anchors
5463 .get(entry.first_entry as usize)
5464 .cloned()
5465 .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5466 let second = match entry.second {
5467 None => Value::Null,
5468 Some(code) => {
5469 let at = codes
5470 .binary_search(&code)
5471 .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5472 texts[at].clone()
5473 }
5474 };
5475 out.push((vec![first, second], entry.count));
5476 }
5477 Ok(Some(out))
5478 }
5479
5480 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5500 let Some(prefix) = self.frequency_prefix(column)? else {
5501 return Ok(None);
5502 };
5503 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5504 }
5505
5506 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5529 let field = self
5530 .table
5531 .fields
5532 .get(column)
5533 .ok_or_else(|| invalid("frequency column index out of range"))?;
5534 let Some(summary) = self.frequency_summary(column)? else {
5535 return Ok(None);
5536 };
5537 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5538 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5539 }
5540
5541 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5543 Ok(match self.table.frequencies.get(column) {
5544 None | Some(None) => None,
5545 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5546 Some(Some(Frequencies::Stored { span, values })) => {
5547 let slot = self
5548 .frequency_summaries
5549 .get(column)
5550 .ok_or_else(|| invalid("frequency column index out of range"))?;
5551 if let Some(summary) = slot.get() {
5552 return Ok(Some(Cow::Borrowed(summary.as_ref())));
5553 }
5554 let field = self
5555 .table
5556 .fields
5557 .get(column)
5558 .ok_or_else(|| invalid("frequency column index out of range"))?;
5559 let mut bytes = vec![0; span.length as usize];
5560 read_at(&self.file, span.offset, &mut bytes)?;
5561 let summary =
5562 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5563 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5564 let _ = slot.set(Arc::new(summary));
5565 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5566 }
5567 })
5568 }
5569
5570 fn decode_frequencies(
5578 &self,
5579 column: usize,
5580 ty: &LogicalType,
5581 entries: &[FrequencyEntry],
5582 ) -> Result<Vec<(Value, u64)>> {
5583 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5584 return Ok(values.as_ref().clone());
5585 }
5586 let values = self.decode_frequencies_once(column, ty, entries)?;
5587 if let Some(slot) = self.frequency_values.get(column) {
5588 let _ = slot.set(Arc::new(values.clone()));
5589 }
5590 Ok(values)
5591 }
5592
5593 fn decode_frequencies_once(
5594 &self,
5595 column: usize,
5596 ty: &LogicalType,
5597 entries: &[FrequencyEntry],
5598 ) -> Result<Vec<(Value, u64)>> {
5599 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5600 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5601 return Err(invalid("frequency text count differs from its synopsis"));
5602 }
5603 let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5604 self.dictionary(column)?
5605 } else {
5606 None
5607 };
5608 let mut codes = entries
5609 .iter()
5610 .filter_map(|entry| match entry.value {
5611 FrequencyValue::Code(code) => Some(code as usize),
5612 _ => None,
5613 })
5614 .collect::<Vec<_>>();
5615 codes.sort_unstable();
5616 codes.dedup();
5617 let texts = match &dictionary {
5618 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5619 _ => Vec::new(),
5620 };
5621 let mut out = Vec::with_capacity(entries.len());
5622 for (entry_at, entry) in entries.iter().enumerate() {
5623 let value = match entry.value {
5624 FrequencyValue::Null => {
5625 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5626 return Err(invalid("a null frequency entry has text"));
5627 }
5628 Value::Null
5629 }
5630 FrequencyValue::Integer(value) => match *ty {
5631 LogicalType::TinyInt => Value::TinyInt(
5632 i8::try_from(value)
5633 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5634 ),
5635 LogicalType::UTinyInt => Value::UTinyInt(
5636 u8::try_from(value)
5637 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5638 ),
5639 LogicalType::USmallInt => Value::USmallInt(
5640 u16::try_from(value)
5641 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5642 ),
5643 LogicalType::UInteger => Value::UInteger(
5644 u32::try_from(value)
5645 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5646 ),
5647 LogicalType::UBigInt => Value::UBigInt(
5648 u64::try_from(value)
5649 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5650 ),
5651 LogicalType::SmallInt => Value::SmallInt(
5652 i16::try_from(value)
5653 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5654 ),
5655 LogicalType::Integer => Value::Integer(
5656 i32::try_from(value)
5657 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5658 ),
5659 LogicalType::BigInt => Value::BigInt(
5660 i64::try_from(value)
5661 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5662 ),
5663 LogicalType::Date => Value::Date(
5664 i32::try_from(value)
5665 .map_err(|_| invalid("frequency DATE is out of range"))?,
5666 ),
5667 LogicalType::Timestamp => Value::Timestamp(
5668 i64::try_from(value)
5669 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5670 ),
5671 _ => return Err(invalid("integer frequency belongs to another type")),
5672 },
5673 FrequencyValue::Code(code) => {
5674 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5675 Value::Varchar(
5676 String::from_utf8(text.clone())
5677 .map_err(|_| invalid("frequency text is not UTF-8"))?,
5678 )
5679 } else {
5680 if dictionary.is_none() {
5681 return Err(invalid("frequency code has no dictionary or stored text"));
5682 }
5683 let at = codes
5684 .binary_search(&(code as usize))
5685 .map_err(|_| invalid("frequency code was not among the codes read"))?;
5686 texts[at].clone()
5687 }
5688 }
5689 };
5690 out.push((value, entry.count));
5691 }
5692 Ok(out)
5693 }
5694
5695 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5705 let field = self
5706 .table
5707 .fields
5708 .get(column)
5709 .ok_or_else(|| invalid("frequency column index out of range"))?;
5710 let Some(summary) = self.frequency_summary(column)? else {
5711 return Ok(None);
5712 };
5713 if summary.ordinals.is_empty() {
5714 return Ok(None);
5715 }
5716 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5717 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5718 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5719 } else {
5720 (Vec::new(), Vec::new())
5721 };
5722 Ok(Some(FrequencyOccurrences {
5723 omitted_max: summary.omitted_max,
5724 ordinals: summary.ordinals.clone(),
5725 anchors,
5726 anchor_indices,
5727 }))
5728 }
5729
5730 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5756 self.table
5757 .distincts
5758 .get(column)
5759 .copied()
5760 .ok_or_else(|| invalid("distinct column index out of range"))
5761 }
5762
5763 pub fn null_count(&self, column: usize) -> Result<u64> {
5774 if column >= self.table.fields.len() {
5775 return Err(invalid("null count column index out of range"));
5776 }
5777 let mut nulls = 0_u64;
5778 for stripe in &self.table.stripes {
5779 let range = stripe
5780 .zone
5781 .column(column)
5782 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5783 nulls = nulls
5784 .checked_add(range.nulls as u64)
5785 .ok_or_else(|| invalid("null count overflow"))?;
5786 }
5787 Ok(nulls)
5788 }
5789
5790 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5805 if self.null_count(column)? > 0 {
5806 return Ok(None);
5807 }
5808 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5809 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5810 if ranks == 0 {
5811 return Ok(None);
5812 }
5813 let low = text_at_rank(&dictionary, 0)?;
5814 let high = text_at_rank(&dictionary, ranks - 1)?;
5815 Ok(Some((low, high)))
5816 }
5817
5818 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5841 if column >= self.table.fields.len() {
5842 return Err(invalid("extremes column index out of range"));
5843 }
5844 let mut low: Option<Bound> = None;
5845 let mut high: Option<Bound> = None;
5846 for stripe in &self.table.stripes {
5847 let range = stripe
5848 .zone
5849 .column(column)
5850 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5851 if !range.exact {
5852 return Ok(None);
5853 }
5854 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5859 if stripe.rows > range.nulls {
5860 return Ok(None);
5861 }
5862 continue;
5863 };
5864 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5865 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5866 }
5867 Ok(low.zip(high))
5868 }
5869
5870 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5883 if column >= self.table.fields.len() {
5884 return Err(invalid("sum column index out of range"));
5885 }
5886 let mut total = 0_i128;
5887 let mut rows = 0_u64;
5888 for stripe in &self.table.stripes {
5889 let range = stripe
5890 .zone
5891 .column(column)
5892 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5893 let Some(part) = range.sum else { return Ok(None) };
5894 let Some(sum) = total.checked_add(part) else { return Ok(None) };
5895 total = sum;
5896 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5897 }
5898 Ok(Some((total, rows)))
5899 }
5900
5901 pub fn host_groups(
5904 &self,
5905 column: usize,
5906 minimum_count: u64,
5907 ) -> Result<Option<Vec<host::HostEntry>>> {
5908 if column >= self.table.fields.len() {
5909 return Err(invalid("host group column index out of range"));
5910 }
5911 let Some(summary) = &self.table.host_groups else { return Ok(None) };
5912 if summary.column != column || minimum_count <= summary.omitted_max {
5913 return Ok(None);
5914 }
5915 Ok(Some(summary.entries.clone()))
5916 }
5917
5918 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5927 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5928 if let Some(dictionary) = self.dictionaries[column].get() {
5929 return Ok(Some(Arc::clone(dictionary)));
5930 }
5931 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
5932 if let Some(dictionary) = self.dictionaries[column].get() {
5933 return Ok(Some(Arc::clone(dictionary)));
5934 }
5935 self.opened.fetch_add(1, Atomic::Relaxed);
5936 let dictionary = Arc::new(open_global_dictionary(
5937 Arc::clone(&self.file),
5938 page,
5939 &self.table.fields[column].ty,
5940 TEXT_KEEP_BUDGET,
5941 )?);
5942 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
5943 Ok(Some(dictionary))
5944 }
5945
5946 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
5953 if of.extent_bytes == 0 {
5954 return Ok(Vec::new());
5955 }
5956 let mut bytes = vec![0; of.extent_bytes as usize];
5957 read_at(&self.file, of.extent_page, &mut bytes)?;
5958 if checksum(&bytes) != of.hash {
5959 return Err(invalid("a section's extent table does not checksum"));
5960 }
5961 let extents = section::decode_extents(&bytes)?;
5962 if extents.len() != of.extents as usize {
5963 return Err(invalid("a section's extent table is not the length the entry says"));
5964 }
5965 Ok(extents)
5966 }
5967
5968 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
5978 let end = of
5979 .offset
5980 .checked_add(u64::from(of.length))
5981 .ok_or_else(|| invalid("an extent overflows the file"))?;
5982 if of.offset < HEADER || end > self.size {
5983 return Err(invalid("an extent is outside the file"));
5984 }
5985 let mut bytes = vec![0; of.length as usize];
5986 read_at(&self.file, of.offset, &mut bytes)?;
5987 if checksum(&bytes) != of.hash {
5988 return Err(invalid("an extent does not checksum"));
5989 }
5990 Ok(bytes)
5991 }
5992
5993 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6002 let extents = self.extents(of)?;
6003 let mut bytes =
6004 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6005 for one in &extents {
6006 if one.first != bytes.len() as u64 {
6007 return Err(invalid("a section's extents do not join up"));
6008 }
6009 bytes.extend_from_slice(&self.extent(one)?);
6010 }
6011 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6014 return Err(invalid("a section's header is longer than its payload"));
6015 }
6016 Ok(bytes)
6017 }
6018
6019 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6028 self.read_impl(part, columns, true)
6029 }
6030
6031 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6041 self.read_impl(part, columns, false)
6042 }
6043
6044 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6051 if candidates.is_empty() {
6052 return Ok(true);
6053 }
6054 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6055 return Err(Error::internal("native code candidates are not sorted and unique"));
6056 }
6057 let stripe = self.stripe_of(part)?;
6058 let Some(page) = stripe.memberships.get(column) else {
6059 return Ok(false);
6060 };
6061 let mut bytes = vec![0; page.length as usize];
6062 read_at(&self.file, page.offset, &mut bytes)?;
6063 if checksum(&bytes) != page.hash {
6064 return Err(invalid("membership page checksum differs"));
6065 }
6066 let codes = decode_membership(&bytes)?;
6067 let mut left = 0;
6068 let mut right = 0;
6069 while left < codes.len() && right < candidates.len() {
6070 match codes[left].cmp(&candidates[right]) {
6071 Ordering::Less => left += 1,
6072 Ordering::Greater => right += 1,
6073 Ordering::Equal => return Ok(false),
6074 }
6075 }
6076 Ok(true)
6077 }
6078
6079 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6080 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6081 self.table
6082 .stripes
6083 .get(place.stripe as usize)
6084 .ok_or_else(|| invalid("stripe index out of range"))
6085 }
6086
6087 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6104 let cache =
6105 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6106 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6107 let known = cached.index.get(at).and_then(Clone::clone);
6108 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6109 slot.used.store(true, Atomic::Relaxed);
6110 Arc::clone(&slot.page)
6111 });
6112 if let Some(index) = known.clone() {
6113 if !whole || page.is_some() {
6114 return Ok(CachedColumn { stripe: at, index, page });
6115 }
6116 }
6117 if cached.loading.contains(&at) {
6118 drop(cached);
6119 if let Some(index) = known {
6123 return Ok(CachedColumn { stripe: at, index, page: None });
6124 }
6125 let held = self.page_of(stripe, column, at, false, None)?;
6126 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6127 remember(&mut cached, &held);
6128 return Ok(held);
6129 }
6130 cached.loading.push(at);
6131 drop(cached);
6132
6133 let read = self.page_of(stripe, column, at, whole, known);
6134
6135 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6139 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
6140 cached.loading.remove(position);
6141 }
6142 let held = read?;
6143 let taken = remember(&mut cached, &held);
6144 drop(cached);
6145 if let Some((bytes, used)) = taken {
6146 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
6147 self.pool.admit(Held {
6148 shelf: Arc::downgrade(&self.cache),
6149 column,
6150 stripe: at,
6151 bytes,
6152 used,
6153 });
6154 }
6155 Ok(held)
6156 }
6157
6158 fn page_of(
6164 &self,
6165 stripe: &Stripe,
6166 column: usize,
6167 at: usize,
6168 whole: bool,
6169 known: Option<Arc<Vec<PartSpan>>>,
6170 ) -> Result<CachedColumn> {
6171 let index = match known {
6172 Some(index) => index,
6173 None => {
6174 self.indexes.fetch_add(1, Atomic::Relaxed);
6175 Arc::new(read_index(&self.file, stripe, column)?)
6176 }
6177 };
6178 let page = if whole {
6179 self.pages.fetch_add(1, Atomic::Relaxed);
6180 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6181 let mut bytes = vec![0; span.length as usize];
6182 read_at(&self.file, span.offset, &mut bytes)?;
6183 Some(Arc::new(bytes))
6184 } else {
6185 None
6186 };
6187 Ok(CachedColumn { stripe: at, index, page })
6188 }
6189
6190 fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
6191 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
6192 let index = place.stripe as usize;
6193 let stripe =
6194 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
6195 let rows = place.rows as usize;
6196 let mut picked = Vec::with_capacity(columns.len());
6197 for &column in columns {
6198 let field = self
6199 .table
6200 .fields
6201 .get(column)
6202 .ok_or_else(|| invalid("column index out of range"))?;
6203 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6204 let held = self.held(index, stripe, column, whole)?;
6205 let span = *held
6206 .index
6207 .get(place.part as usize)
6208 .ok_or_else(|| invalid("part index out of range"))?;
6209 let owned;
6210 let bytes = match &held.page {
6211 Some(held) => part_bytes(held, span)?,
6212 None => {
6213 let offset = page
6214 .offset
6215 .checked_add(span.start as u64)
6216 .ok_or_else(|| invalid("part range overflow"))?;
6217 let mut bytes = vec![0; span.length];
6218 read_at(&self.file, offset, &mut bytes)?;
6219 owned = bytes;
6220 &owned
6221 }
6222 };
6223 if checksum(bytes) != span.hash {
6224 return Err(invalid(&format!(
6225 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
6226 wanted {:016x} and got {:016x}",
6227 place.part,
6228 page.offset,
6229 span.start,
6230 span.length,
6231 span.hash,
6232 checksum(bytes),
6233 )));
6234 }
6235 let dictionary = self.dictionary(column)?;
6236 picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
6242 }
6243 Chunk::with_rows(picked, rows)
6244 }
6245
6246 #[must_use]
6262 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
6263 let Some(place) = self.places.get(part).copied() else { return false };
6264 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6265 if stripe.zone.skips(probes) {
6266 return true;
6267 }
6268 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
6269 }
6270
6271 fn outside(&self, place: Place, probe: &Probe) -> bool {
6277 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6278 Some(ranges) => ranges
6279 .get(place.part as usize)
6280 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
6281 None => false,
6282 }
6283 }
6284
6285 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
6291 let slot = self.part_ranges.get(column)?.get(stripe)?;
6292 if let Some(held) = slot.get() {
6293 return Some(held);
6294 }
6295 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
6296 let mut bytes = vec![0; page.length as usize];
6297 read_at(&self.file, page.offset, &mut bytes).ok()?;
6298 if checksum(&bytes) != page.hash {
6299 return None;
6300 }
6301 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
6302 let _ = slot.set(ranges);
6303 slot.get().map(|held| held.as_slice())
6304 }
6305
6306 #[must_use]
6323 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
6324 let Some(place) = self.places.get(part).copied() else { return false };
6325 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6326 if stripe.zone.certain(probes) {
6327 return true;
6328 }
6329 probes
6330 .iter()
6331 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
6332 }
6333
6334 fn inside(&self, place: Place, probe: &Probe) -> bool {
6340 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6341 Some(ranges) => ranges
6342 .get(place.part as usize)
6343 .is_some_and(|range| range.certain(probe.op, &probe.value)),
6344 None => false,
6345 }
6346 }
6347
6348 #[must_use]
6359 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
6360 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
6361 }
6362
6363 fn sifted(&self, place: Place, probe: &Probe) -> bool {
6369 if probe.op != Op::Equal {
6370 return false;
6371 }
6372 match self.stripe_sieves(place.stripe as usize, probe.column) {
6373 Some(sieves) => sieves
6374 .get(place.part as usize)
6375 .and_then(Option::as_ref)
6376 .is_some_and(|sieve| sieve.excludes(&probe.value)),
6377 None => false,
6378 }
6379 }
6380
6381 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
6388 let slot = self.sieves.get(column)?.get(stripe)?;
6389 if let Some(held) = slot.get() {
6390 return Some(held);
6391 }
6392 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
6393 let mut bytes = vec![0; page.length as usize];
6394 read_at(&self.file, page.offset, &mut bytes).ok()?;
6395 if checksum(&bytes) != page.hash {
6396 return None;
6397 }
6398 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
6399 let _ = slot.set(sieves);
6400 slot.get().map(|held| held.as_slice())
6401 }
6402}
6403
6404fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6406 let code = dictionary.code_at_rank(rank)? as usize;
6407 let text = dictionary
6408 .try_text_at(code)?
6409 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6410 Ok(Value::Varchar(text.into()))
6411}
6412
6413#[cfg(unix)]
6418fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6419 use std::os::unix::fs::FileExt;
6420 while !bytes.is_empty() {
6421 let written = file.write_at(bytes, offset).map_err(io)?;
6422 if written == 0 {
6423 return Err(invalid("a write to the native file wrote nothing"));
6424 }
6425 offset += written as u64;
6426 bytes = &bytes[written..];
6427 }
6428 Ok(())
6429}
6430
6431#[cfg(windows)]
6433fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6434 use std::os::windows::fs::FileExt;
6435 while !bytes.is_empty() {
6436 let written = file.seek_write(bytes, offset).map_err(io)?;
6437 if written == 0 {
6438 return Err(invalid("a write to the native file wrote nothing"));
6439 }
6440 offset += written as u64;
6441 bytes = &bytes[written..];
6442 }
6443 Ok(())
6444}
6445
6446#[cfg(not(any(unix, windows)))]
6448fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6449 use std::io::Write;
6450 let mut file = file.try_clone().map_err(io)?;
6451 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6452 file.write_all(bytes).map_err(io)
6453}
6454
6455#[cfg(unix)]
6465fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6466 use std::os::unix::fs::FileExt;
6467 while !bytes.is_empty() {
6468 let read = file.read_at(bytes, offset).map_err(io)?;
6469 if read == 0 {
6470 return Err(invalid("column page ends before its declared length"));
6471 }
6472 offset += read as u64;
6473 bytes = &mut bytes[read..];
6474 }
6475 Ok(())
6476}
6477
6478#[cfg(windows)]
6484fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6485 use std::os::windows::fs::FileExt;
6486 while !bytes.is_empty() {
6487 let read = file.seek_read(bytes, offset).map_err(io)?;
6488 if read == 0 {
6489 return Err(invalid("column page ends before its declared length"));
6490 }
6491 offset += read as u64;
6492 bytes = &mut bytes[read..];
6493 }
6494 Ok(())
6495}
6496
6497#[cfg(not(any(unix, windows)))]
6502fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6503 let mut file = file.try_clone().map_err(io)?;
6504 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6505 file.read_exact(bytes).map_err(io)
6506}
6507
6508fn type_tag(ty: &LogicalType) -> Result<u8> {
6515 match ty {
6516 LogicalType::SmallInt => Ok(1),
6517 LogicalType::Integer => Ok(2),
6518 LogicalType::BigInt => Ok(3),
6519 LogicalType::Varchar => Ok(4),
6520 LogicalType::Date => Ok(5),
6521 LogicalType::Timestamp => Ok(6),
6522 LogicalType::Boolean => Ok(7),
6523 LogicalType::TinyInt => Ok(8),
6524 LogicalType::UTinyInt => Ok(9),
6525 LogicalType::USmallInt => Ok(10),
6526 LogicalType::UInteger => Ok(11),
6527 LogicalType::UBigInt => Ok(12),
6528 LogicalType::Decimal { .. } => Ok(13),
6529 LogicalType::Float => Ok(14),
6530 LogicalType::Double => Ok(15),
6531 LogicalType::HugeInt => Ok(16),
6532 LogicalType::UHugeInt => Ok(17),
6533 LogicalType::Time => Ok(18),
6534 LogicalType::TimeTz => Ok(19),
6535 LogicalType::TimestampTz => Ok(20),
6536 LogicalType::Interval => Ok(21),
6537 LogicalType::Uuid => Ok(22),
6538 LogicalType::Blob => Ok(23),
6539 LogicalType::Bit => Ok(24),
6540 LogicalType::TimestampS => Ok(25),
6541 LogicalType::TimestampMs => Ok(26),
6542 LogicalType::TimestampNs => Ok(27),
6543 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6544 }
6545}
6546
6547fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6553 out.push(type_tag(ty)?);
6554 if let LogicalType::Decimal { width, scale } = ty {
6555 out.push(*width);
6556 out.push(*scale);
6557 }
6558 Ok(())
6559}
6560
6561fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6563 let tag = cur.u8()?;
6564 if tag == 13 {
6565 let width = cur.u8()?;
6566 let scale = cur.u8()?;
6567 return LogicalType::decimal(width, scale)
6568 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6569 }
6570 tag_type(tag)
6571}
6572
6573fn tag_type(tag: u8) -> Result<LogicalType> {
6574 match tag {
6575 1 => Ok(LogicalType::SmallInt),
6576 2 => Ok(LogicalType::Integer),
6577 3 => Ok(LogicalType::BigInt),
6578 4 => Ok(LogicalType::Varchar),
6579 5 => Ok(LogicalType::Date),
6580 6 => Ok(LogicalType::Timestamp),
6581 7 => Ok(LogicalType::Boolean),
6582 8 => Ok(LogicalType::TinyInt),
6583 9 => Ok(LogicalType::UTinyInt),
6584 10 => Ok(LogicalType::USmallInt),
6585 11 => Ok(LogicalType::UInteger),
6586 12 => Ok(LogicalType::UBigInt),
6587 14 => Ok(LogicalType::Float),
6588 15 => Ok(LogicalType::Double),
6589 16 => Ok(LogicalType::HugeInt),
6590 17 => Ok(LogicalType::UHugeInt),
6591 18 => Ok(LogicalType::Time),
6592 19 => Ok(LogicalType::TimeTz),
6593 20 => Ok(LogicalType::TimestampTz),
6594 21 => Ok(LogicalType::Interval),
6595 22 => Ok(LogicalType::Uuid),
6596 23 => Ok(LogicalType::Blob),
6597 24 => Ok(LogicalType::Bit),
6598 25 => Ok(LogicalType::TimestampS),
6599 26 => Ok(LogicalType::TimestampMs),
6600 27 => Ok(LogicalType::TimestampNs),
6601 _ => Err(invalid("column type tag is unknown")),
6602 }
6603}
6604
6605fn put_u16(out: &mut Vec<u8>, value: u16) {
6606 out.extend_from_slice(&value.to_le_bytes());
6607}
6608fn put_u32(out: &mut Vec<u8>, value: u32) {
6609 out.extend_from_slice(&value.to_le_bytes());
6610}
6611fn put_u64(out: &mut Vec<u8>, value: u64) {
6612 out.extend_from_slice(&value.to_le_bytes());
6613}
6614fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6615 while value >= 0x80 {
6616 out.push((value as u8 & 0x7f) | 0x80);
6617 value >>= 7;
6618 }
6619 out.push(value as u8);
6620}
6621
6622fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6623 match (left, right) {
6624 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6625 (FrequencyValue::Null, _) => Ordering::Less,
6626 (_, FrequencyValue::Null) => Ordering::Greater,
6627 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6628 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6629 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6630 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6631 }
6632}
6633
6634fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6647 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6648 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6649 };
6650 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6651 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6652 let omitted_max = next.count;
6653 entries.truncate(FREQUENCY_ENTRIES);
6654 omitted_max
6655 } else {
6656 0
6657 };
6658 entries.sort_unstable_by(order);
6659 omitted_max
6660}
6661
6662fn code_frequency(
6663 dictionary: &GlobalDictionary,
6664 flat: &[u8],
6665 bases: &[u64],
6666) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6667 let mut entries = dictionary
6668 .counts
6669 .iter()
6670 .enumerate()
6671 .filter(|(_, count)| **count != 0)
6672 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6673 .collect::<Vec<_>>();
6674 if dictionary.nulls != 0 {
6675 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6676 }
6677 let omitted_max = keep_most_frequent(&mut entries);
6678 let mut spans = Vec::with_capacity(entries.len());
6679 let mut text_bytes = 0_usize;
6680 for entry in &entries {
6681 let span = match entry.value {
6682 FrequencyValue::Code(code) => {
6683 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6684 let bytes = flat
6685 .get(span.0..span.1)
6686 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6687 text_bytes = text_bytes.saturating_add(bytes.len());
6688 Some(span)
6689 }
6690 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6691 };
6692 spans.push(span);
6693 }
6694 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6695 Vec::new()
6696 } else {
6697 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6698 };
6699 Ok((
6700 FrequencySummary {
6701 entries,
6702 omitted_max,
6703 ordinals: Vec::new(),
6704 ordinal_entries: Vec::new(),
6705 },
6706 texts,
6707 ))
6708}
6709
6710fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6711 let mut out = DIRECTORY.to_vec();
6712 let name = table.name.as_bytes();
6713 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6714 out.extend_from_slice(name);
6715 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6716 for field in &table.fields {
6717 let name = field.name.as_bytes();
6718 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6719 out.extend_from_slice(name);
6720 put_type(&mut out, &field.ty)?;
6721 out.push(u8::from(field.not_null));
6722 }
6723 for dictionary in &table.dictionaries {
6724 match dictionary {
6725 None => out.push(0),
6726 Some(page) => {
6727 out.push(1);
6728 put_u64(&mut out, page.offset);
6729 put_u32(&mut out, page.length);
6730 put_u64(&mut out, page.hash);
6731 }
6732 }
6733 }
6734 for distinct in &table.distincts {
6735 match distinct {
6736 None => out.push(0),
6737 Some(count) => {
6738 out.push(1);
6739 put_u64(&mut out, *count);
6740 }
6741 }
6742 }
6743 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6744 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6745 for stripe in &table.stripes {
6746 put_u32(
6747 &mut out,
6748 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6749 );
6750 for &rows in &stripe.parts {
6751 put_u32(&mut out, rows);
6752 }
6753 put_u64(&mut out, stripe.index.offset);
6754 put_u32(&mut out, stripe.index.length);
6755 for page in &stripe.pages {
6756 put_u64(&mut out, page.offset);
6757 put_u32(&mut out, page.length);
6758 }
6759 for ((field, dictionary), membership) in
6764 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6765 {
6766 if field.ty != LogicalType::Varchar || dictionary.is_none() {
6767 continue;
6768 }
6769 let page =
6770 membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6771 put_u64(&mut out, page.offset);
6772 put_u32(&mut out, page.length);
6773 put_u64(&mut out, page.hash);
6774 }
6775 for sieve in stripe.sieves.slots() {
6776 match sieve {
6777 None => out.push(0),
6778 Some(page) => {
6779 out.push(1);
6780 put_u64(&mut out, page.offset);
6781 put_u32(&mut out, page.length);
6782 put_u64(&mut out, page.hash);
6783 }
6784 }
6785 }
6786 for held in stripe.part_ranges.slots() {
6787 match held {
6788 None => out.push(0),
6789 Some(page) => {
6790 out.push(1);
6791 put_u64(&mut out, page.offset);
6792 put_u32(&mut out, page.length);
6793 put_u64(&mut out, page.hash);
6794 }
6795 }
6796 }
6797 for range in stripe.zone.columns() {
6798 put_bound(&mut out, range.low.as_ref())?;
6799 put_bound(&mut out, range.high.as_ref())?;
6800 put_u32(
6801 &mut out,
6802 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6803 );
6804 out.push(u8::from(range.exact));
6805 match range.sum {
6806 None => out.push(0),
6807 Some(total) => {
6808 out.push(1);
6809 out.extend_from_slice(&total.to_le_bytes());
6810 }
6811 }
6812 }
6813 }
6814 out.extend_from_slice(FREQUENCIES);
6815 put_u16(
6816 &mut out,
6817 u16::try_from(table.frequencies.len())
6818 .map_err(|_| invalid("too many frequency columns"))?,
6819 );
6820 for summary in &table.frequencies {
6821 let summary = match summary {
6822 None => {
6823 out.push(0);
6824 continue;
6825 }
6826 Some(Frequencies::Held(summary)) => summary,
6827 Some(Frequencies::Stored { .. }) => {
6829 return Err(invalid("a synopsis left in the file cannot be written back"));
6830 }
6831 };
6832 out.push(1);
6833 put_u64(&mut out, summary.omitted_max);
6834 put_u32(
6835 &mut out,
6836 u32::try_from(summary.entries.len())
6837 .map_err(|_| invalid("too many frequency entries"))?,
6838 );
6839 for entry in &summary.entries {
6840 match entry.value {
6841 FrequencyValue::Null => out.push(0),
6842 FrequencyValue::Integer(value) => {
6843 out.push(1);
6844 out.extend_from_slice(&value.to_le_bytes());
6845 }
6846 FrequencyValue::Code(value) => {
6847 out.push(2);
6848 put_u32(&mut out, value);
6849 }
6850 }
6851 put_u64(&mut out, entry.count);
6852 }
6853 put_u32(
6854 &mut out,
6855 u32::try_from(summary.ordinals.len())
6856 .map_err(|_| invalid("too many frequency ordinals"))?,
6857 );
6858 let mut previous = 0_u64;
6859 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6860 let delta = if at == 0 {
6861 ordinal
6862 } else {
6863 ordinal
6864 .checked_sub(previous)
6865 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6866 };
6867 if at != 0 && delta == 0 {
6868 return Err(invalid("frequency ordinals are not unique"));
6869 }
6870 put_var_u64(&mut out, delta);
6871 previous = ordinal;
6872 }
6873 if summary.ordinal_entries.len() != summary.ordinals.len() {
6874 return Err(invalid("frequency ordinal values have a different length"));
6875 }
6876 for &entry in &summary.ordinal_entries {
6877 if entry as usize >= summary.entries.len() {
6878 return Err(invalid("frequency ordinal value is outside its entries"));
6879 }
6880 put_u16(&mut out, entry);
6881 }
6882 }
6883 if !table.pair_frequencies.is_empty() {
6884 out.extend_from_slice(PAIR_FREQUENCIES);
6885 put_u16(
6886 &mut out,
6887 u16::try_from(table.pair_frequencies.len())
6888 .map_err(|_| invalid("too many pair frequency summaries"))?,
6889 );
6890 for summary in &table.pair_frequencies {
6891 put_u16(&mut out, summary.first);
6892 put_u16(&mut out, summary.second);
6893 put_u64(&mut out, summary.omitted_max);
6894 put_u16(
6895 &mut out,
6896 u16::try_from(summary.entries.len())
6897 .map_err(|_| invalid("too many pair frequency entries"))?,
6898 );
6899 for entry in &summary.entries {
6900 put_u16(&mut out, entry.first_entry);
6901 match entry.second {
6902 None => out.push(0),
6903 Some(code) => {
6904 out.push(1);
6905 put_u32(&mut out, code);
6906 }
6907 }
6908 put_u64(&mut out, entry.count);
6909 }
6910 }
6911 }
6912 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
6913 if text_columns != 0 {
6914 out.extend_from_slice(FREQUENCY_TEXTS);
6915 put_u16(
6916 &mut out,
6917 u16::try_from(text_columns)
6918 .map_err(|_| invalid("too many string frequency columns"))?,
6919 );
6920 for (column, texts) in table.frequency_texts.iter().enumerate() {
6921 if texts.is_empty() {
6922 continue;
6923 }
6924 put_u16(
6925 &mut out,
6926 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
6927 );
6928 put_u16(
6929 &mut out,
6930 u16::try_from(texts.len())
6931 .map_err(|_| invalid("too many frequency text entries"))?,
6932 );
6933 for text in texts {
6934 match text {
6935 None => out.push(0),
6936 Some(text) => {
6937 out.push(1);
6938 put_u32(
6939 &mut out,
6940 u32::try_from(text.len())
6941 .map_err(|_| invalid("frequency text is too long"))?,
6942 );
6943 out.extend_from_slice(text);
6944 }
6945 }
6946 }
6947 }
6948 }
6949 if let Some(summary) = &table.host_groups {
6950 out.extend_from_slice(HOST_GROUPS);
6951 put_u16(
6952 &mut out,
6953 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
6954 );
6955 put_u64(&mut out, summary.omitted_max);
6956 put_u16(
6957 &mut out,
6958 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
6959 );
6960 for entry in &summary.entries {
6961 put_u32(
6962 &mut out,
6963 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
6964 );
6965 out.extend_from_slice(entry.host.as_bytes());
6966 put_u64(&mut out, entry.count);
6967 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
6968 put_u32(
6969 &mut out,
6970 u32::try_from(entry.minimum.len())
6971 .map_err(|_| invalid("host minimum is too long"))?,
6972 );
6973 out.extend_from_slice(entry.minimum.as_bytes());
6974 }
6975 }
6976 if let Some(clustering) = &table.clustering {
6979 out.extend_from_slice(CLUSTERING);
6980 out.push(clustering.width().tag());
6981 put_u16(
6982 &mut out,
6983 u16::try_from(clustering.columns().len())
6984 .map_err(|_| invalid("too many clustering columns"))?,
6985 );
6986 for &column in clustering.columns() {
6987 put_u16(
6988 &mut out,
6989 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
6990 );
6991 }
6992 }
6993 out.extend_from_slice(SECTIONS);
6999 put_u64(&mut out, table.generation);
7000 put_u16(
7001 &mut out,
7002 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7003 );
7004 for held in &table.sections {
7005 held.encode(&mut out)?;
7006 }
7007 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7008 out.extend_from_slice(DICTIONARY_PAYLOADS);
7009 put_u16(
7010 &mut out,
7011 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7012 );
7013 for at in 0..table.fields.len() {
7014 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7015 }
7016 }
7017 Ok(out)
7018}
7019
7020fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
7029 table
7030 .fields
7031 .iter()
7032 .enumerate()
7033 .map(|(column, field)| {
7034 if !matches!(
7035 field.ty,
7036 LogicalType::TinyInt
7037 | LogicalType::SmallInt
7038 | LogicalType::Integer
7039 | LogicalType::BigInt
7040 | LogicalType::UTinyInt
7041 | LogicalType::USmallInt
7042 | LogicalType::UInteger
7043 | LogicalType::UBigInt
7044 ) {
7045 return None;
7046 }
7047 let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
7048 return None;
7049 };
7050 let zero = summary
7051 .entries
7052 .iter()
7053 .find(|entry| entry.value == FrequencyValue::Integer(0))
7054 .map(|entry| entry.count)
7055 .or_else(|| (summary.omitted_max == 0).then_some(0))?;
7056 let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
7057 count.checked_add(stripe.zone.column(column)?.nulls as u64)
7058 })?;
7059 (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
7060 })
7061 .collect()
7062}
7063
7064fn signed_integer(ty: &LogicalType) -> bool {
7065 matches!(
7066 ty,
7067 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7068 )
7069}
7070
7071fn integer_or_date(ty: &LogicalType) -> bool {
7072 matches!(
7073 ty,
7074 LogicalType::TinyInt
7075 | LogicalType::SmallInt
7076 | LogicalType::Integer
7077 | LogicalType::BigInt
7078 | LogicalType::UTinyInt
7079 | LogicalType::USmallInt
7080 | LogicalType::UInteger
7081 | LogicalType::UBigInt
7082 | LogicalType::Date
7083 )
7084}
7085
7086fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7087 table
7088 .fields
7089 .iter()
7090 .enumerate()
7091 .map(|(column, field)| {
7092 if !integer_or_date(&field.ty) {
7093 return None;
7094 }
7095 let mut low: Option<i128> = None;
7096 let mut high: Option<i128> = None;
7097 for stripe in &table.stripes {
7098 let range = stripe.zone.column(column)?;
7099 if !range.exact {
7100 return None;
7101 }
7102 match (range.low.as_ref(), range.high.as_ref()) {
7103 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
7104 low = Some(low.map_or(*small, |held| held.min(*small)));
7105 high = Some(high.map_or(*large, |held| held.max(*large)));
7106 }
7107 (None, None) if stripe.rows == range.nulls => {}
7108 _ => return None,
7109 }
7110 }
7111 Some(low.zip(high))
7112 })
7113 .collect()
7114}
7115
7116fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
7117 reader
7118 .table
7119 .fields
7120 .iter()
7121 .enumerate()
7122 .map(|(column, field)| {
7123 if !integer_or_date(&field.ty) {
7124 return Ok(None);
7125 }
7126 match reader.exact_extremes(column)? {
7127 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
7128 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
7129 _ => Ok(None),
7130 }
7131 })
7132 .collect()
7133}
7134
7135fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
7136 table
7137 .fields
7138 .iter()
7139 .enumerate()
7140 .map(|(column, field)| {
7141 if !integer_or_date(&field.ty) {
7142 return None;
7143 }
7144 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
7145 return None;
7146 };
7147 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7148 return None;
7149 }
7150 let entries = summary
7151 .entries
7152 .iter()
7153 .map(|entry| {
7154 let value = match entry.value {
7155 FrequencyValue::Null => None,
7156 FrequencyValue::Integer(value) => Some(value),
7157 FrequencyValue::Code(_) => return None,
7158 };
7159 Some((value, entry.count))
7160 })
7161 .collect::<Option<Vec<_>>>()?;
7162 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
7163 (rows == table.rows as u64).then_some(entries)
7164 })
7165 .collect()
7166}
7167
7168fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
7169 Some(match value {
7170 Value::Null => None,
7171 Value::TinyInt(value) => Some(i128::from(*value)),
7172 Value::SmallInt(value) => Some(i128::from(*value)),
7173 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
7174 Value::BigInt(value) => Some(i128::from(*value)),
7175 Value::UTinyInt(value) => Some(i128::from(*value)),
7176 Value::USmallInt(value) => Some(i128::from(*value)),
7177 Value::UInteger(value) => Some(i128::from(*value)),
7178 Value::UBigInt(value) => Some(i128::from(*value)),
7179 _ => return None,
7180 })
7181}
7182
7183fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
7184 reader
7185 .table
7186 .fields
7187 .iter()
7188 .enumerate()
7189 .map(|(column, field)| {
7190 if !integer_or_date(&field.ty) {
7191 return Ok(None);
7192 }
7193 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
7194 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7195 return Ok(None);
7196 }
7197 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
7198 let Some(entries) = entries
7199 .iter()
7200 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
7201 .collect::<Option<Vec<_>>>()
7202 else {
7203 return Ok(None);
7204 };
7205 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
7206 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
7207 })
7208 .collect()
7209}
7210
7211fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
7212 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
7213 let range = stripe.zone.column(column)?;
7214 let sum = sum.checked_add(range.sum?)?;
7215 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
7216 Some((sum, count.checked_add(nonnull)?))
7217 })
7218}
7219
7220fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
7221 table
7222 .fields
7223 .iter()
7224 .enumerate()
7225 .map(|(column, field)| {
7226 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
7227 })
7228 .collect()
7229}
7230
7231fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
7232 reader
7233 .table
7234 .fields
7235 .iter()
7236 .enumerate()
7237 .map(|(column, field)| {
7238 if !matches!(
7239 field.ty,
7240 LogicalType::TinyInt
7241 | LogicalType::SmallInt
7242 | LogicalType::Integer
7243 | LogicalType::BigInt
7244 | LogicalType::UTinyInt
7245 | LogicalType::USmallInt
7246 | LogicalType::UInteger
7247 | LogicalType::UBigInt
7248 ) {
7249 return Ok(None);
7250 }
7251 let Some(summary) = reader.frequency_summary(column)? else {
7252 return Ok(None);
7253 };
7254 let zero = summary
7255 .entries
7256 .iter()
7257 .find(|entry| entry.value == FrequencyValue::Integer(0))
7258 .map(|entry| entry.count)
7259 .or_else(|| (summary.omitted_max == 0).then_some(0));
7260 let Some(zero) = zero else { return Ok(None) };
7261 let nulls = reader.null_count(column)?;
7262 Ok((reader.table.rows as u64)
7263 .checked_sub(nulls)
7264 .and_then(|count| count.checked_sub(zero)))
7265 })
7266 .collect()
7267}
7268
7269fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
7270 reader
7271 .table
7272 .fields
7273 .iter()
7274 .enumerate()
7275 .map(
7276 |(column, field)| {
7277 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
7278 },
7279 )
7280 .collect()
7281}
7282
7283fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
7284 let mut out = CATALOG.to_vec();
7285 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
7286 for entry in entries {
7287 let name = entry.name.as_bytes();
7288 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7289 out.extend_from_slice(name);
7290 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
7291 put_u16(
7292 &mut out,
7293 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
7294 );
7295 for field in &entry.fields {
7296 let name = field.name.as_bytes();
7297 put_u16(
7298 &mut out,
7299 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7300 );
7301 out.extend_from_slice(name);
7302 put_type(&mut out, &field.ty)?;
7303 out.push(u8::from(field.not_null));
7304 }
7305 put_u64(&mut out, entry.directory.offset);
7306 put_u32(&mut out, entry.directory.length);
7307 put_u64(&mut out, entry.directory.hash);
7308 }
7309 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
7310 for view in views {
7311 let name = view.name.as_bytes();
7312 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
7313 out.extend_from_slice(name);
7314 put_long_text(&mut out, &view.sql, "view body")?;
7315 put_long_text(&mut out, &view.statement, "view statement")?;
7316 put_u16(
7317 &mut out,
7318 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
7319 );
7320 for alias in &view.aliases {
7321 let alias = alias.as_bytes();
7322 put_u16(
7323 &mut out,
7324 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
7325 );
7326 out.extend_from_slice(alias);
7327 }
7328 put_u16(
7329 &mut out,
7330 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
7331 );
7332 for field in &view.columns {
7333 let name = field.name.as_bytes();
7334 put_u16(
7335 &mut out,
7336 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7337 );
7338 out.extend_from_slice(name);
7339 put_type(&mut out, &field.ty)?;
7340 out.push(u8::from(field.not_null));
7341 }
7342 }
7343 out.extend_from_slice(NONZERO_COUNTS);
7344 for entry in entries {
7345 if entry.nonzero.len() != entry.fields.len() {
7346 return Err(invalid("nonzero count width differs from schema"));
7347 }
7348 for count in &entry.nonzero {
7349 match count {
7350 None => out.push(0),
7351 Some(count) => {
7352 out.push(1);
7353 put_u64(&mut out, *count);
7354 }
7355 }
7356 }
7357 }
7358 out.extend_from_slice(AGGREGATE_SUMS);
7359 for entry in entries {
7360 if entry.aggregates.len() != entry.fields.len() {
7361 return Err(invalid("aggregate sum width differs from schema"));
7362 }
7363 for summary in &entry.aggregates {
7364 match summary {
7365 None => out.push(0),
7366 Some((sum, count)) => {
7367 out.push(1);
7368 out.extend_from_slice(&sum.to_le_bytes());
7369 put_u64(&mut out, *count);
7370 }
7371 }
7372 }
7373 }
7374 out.extend_from_slice(DISTINCT_COUNTS);
7375 for entry in entries {
7376 if entry.distincts.len() != entry.fields.len() {
7377 return Err(invalid("distinct count width differs from schema"));
7378 }
7379 for count in &entry.distincts {
7380 match count {
7381 None => out.push(0),
7382 Some(count) => {
7383 if *count > entry.rows as u64 {
7384 return Err(invalid("distinct count exceeds table rows"));
7385 }
7386 out.push(1);
7387 put_u64(&mut out, *count);
7388 }
7389 }
7390 }
7391 }
7392 out.extend_from_slice(INTEGER_EXTREMES);
7393 for entry in entries {
7394 if entry.extremes.len() != entry.fields.len() {
7395 return Err(invalid("integer extremes width differs from schema"));
7396 }
7397 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
7398 match extremes {
7399 None => out.push(0),
7400 Some(None) if integer_or_date(&field.ty) => out.push(1),
7401 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
7402 out.push(2);
7403 out.extend_from_slice(&low.to_le_bytes());
7404 out.extend_from_slice(&high.to_le_bytes());
7405 }
7406 _ => return Err(invalid("integer extremes type or range differs")),
7407 }
7408 }
7409 }
7410 out.extend_from_slice(COMPLETE_FREQUENCIES);
7411 for entry in entries {
7412 if entry.frequencies.len() != entry.fields.len() {
7413 return Err(invalid("numeric frequency width differs from schema"));
7414 }
7415 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
7416 match frequencies {
7417 None => out.push(0),
7418 Some(entries)
7419 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
7420 {
7421 let mut total = 0_u64;
7422 for (at, (value, count)) in entries.iter().enumerate() {
7423 if entries[..at].iter().any(|(held, _)| held == value) {
7424 return Err(invalid("numeric frequency value repeats"));
7425 }
7426 total = total
7427 .checked_add(*count)
7428 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7429 }
7430 if total != entry.rows as u64 {
7431 return Err(invalid("numeric frequencies do not cover table rows"));
7432 }
7433 out.push(1);
7434 out.push(entries.len() as u8);
7435 for (value, count) in entries {
7436 match value {
7437 None => out.push(0),
7438 Some(value) => {
7439 out.push(1);
7440 out.extend_from_slice(&value.to_le_bytes());
7441 }
7442 }
7443 put_u64(&mut out, *count);
7444 }
7445 }
7446 _ => return Err(invalid("numeric frequency type or width differs")),
7447 }
7448 }
7449 }
7450 Ok(out)
7451}
7452
7453fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
7455 let bytes = text.as_bytes();
7456 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
7457 out.extend_from_slice(bytes);
7458 Ok(())
7459}
7460
7461fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
7464 let mut cur = Cursor::new(bytes);
7465 if cur.take(8)? != CATALOG {
7466 return Err(invalid("catalog magic differs"));
7467 }
7468 let count = cur.u32()? as usize;
7469 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
7470 for _ in 0..count {
7471 let name = cur.text()?;
7472 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7473 let width = cur.u16()? as usize;
7474 let mut fields = Vec::with_capacity(width);
7475 for _ in 0..width {
7476 let name = cur.text()?;
7477 let ty = read_type(&mut cur)?;
7478 let not_null = match cur.u8()? {
7479 0 => false,
7480 1 => true,
7481 _ => return Err(invalid("nullability flag differs")),
7482 };
7483 fields.push(Field { name, ty, not_null });
7484 }
7485 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7486 let end = directory
7487 .offset
7488 .checked_add(u64::from(directory.length))
7489 .ok_or_else(|| invalid("table directory offset overflow"))?;
7490 if directory.offset < HEADER
7491 || end > size
7492 || directory.length as usize > MAX_DIRECTORY
7493 || directory.length == 0
7494 {
7495 return Err(invalid("table directory range is outside the file"));
7496 }
7497 if entries.iter().any(|held| held.name == name) {
7498 return Err(invalid("two tables in the catalog have the same name"));
7499 }
7500 let nonzero = vec![None; fields.len()];
7501 let aggregates = vec![None; fields.len()];
7502 let distincts = vec![None; fields.len()];
7503 let extremes = vec![None; fields.len()];
7504 let frequencies = vec![None; fields.len()];
7505 entries.push(Entry {
7506 name,
7507 fields,
7508 rows,
7509 directory,
7510 nonzero,
7511 aggregates,
7512 distincts,
7513 extremes,
7514 frequencies,
7515 });
7516 }
7517 let count = if cur.done() { 0 } else { cur.u32()? as usize };
7522 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
7523 for _ in 0..count {
7524 let name = cur.text()?;
7525 let sql = cur.long_text()?;
7526 let statement = cur.long_text()?;
7527 let width = cur.u16()? as usize;
7528 let mut aliases = Vec::with_capacity(width);
7529 for _ in 0..width {
7530 aliases.push(cur.text()?);
7531 }
7532 let width = cur.u16()? as usize;
7533 let mut columns = Vec::with_capacity(width);
7534 for _ in 0..width {
7535 let name = cur.text()?;
7536 let ty = read_type(&mut cur)?;
7537 let not_null = match cur.u8()? {
7538 0 => false,
7539 1 => true,
7540 _ => return Err(invalid("nullability flag differs")),
7541 };
7542 columns.push(Field { name, ty, not_null });
7543 }
7544 if views.iter().any(|held| held.name == name) {
7548 return Err(invalid("two views in the catalog have the same name"));
7549 }
7550 if entries.iter().any(|held| held.name == name) {
7551 return Err(invalid("a table and a view in the catalog have the same name"));
7552 }
7553 views.push(ViewEntry { name, sql, statement, aliases, columns });
7554 }
7555 if !cur.done() {
7556 if cur.take(8)? != NONZERO_COUNTS {
7557 return Err(invalid("catalog extension magic differs"));
7558 }
7559 for entry in &mut entries {
7560 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
7561 *count = match cur.u8()? {
7562 0 => None,
7563 1 if matches!(
7564 field.ty,
7565 LogicalType::TinyInt
7566 | LogicalType::SmallInt
7567 | LogicalType::Integer
7568 | LogicalType::BigInt
7569 | LogicalType::UTinyInt
7570 | LogicalType::USmallInt
7571 | LogicalType::UInteger
7572 | LogicalType::UBigInt
7573 ) =>
7574 {
7575 let value = cur.u64()?;
7576 if value > entry.rows as u64 {
7577 return Err(invalid("nonzero count exceeds rows"));
7578 }
7579 Some(value)
7580 }
7581 _ => return Err(invalid("nonzero count tag or column type differs")),
7582 };
7583 }
7584 }
7585 }
7586 if !cur.done() {
7587 if cur.take(8)? != AGGREGATE_SUMS {
7588 return Err(invalid("aggregate catalog extension magic differs"));
7589 }
7590 for entry in &mut entries {
7591 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
7592 *summary = match cur.u8()? {
7593 0 => None,
7594 1 if signed_integer(&field.ty) => {
7595 let sum = i128::from_le_bytes(
7596 cur.take(16)?
7597 .try_into()
7598 .map_err(|_| invalid("aggregate sum is truncated"))?,
7599 );
7600 let count = cur.u64()?;
7601 if count > entry.rows as u64 {
7602 return Err(invalid("aggregate count exceeds table rows"));
7603 }
7604 Some((sum, count))
7605 }
7606 _ => return Err(invalid("aggregate sum tag or column type differs")),
7607 };
7608 }
7609 }
7610 }
7611 if !cur.done() {
7612 if cur.take(8)? != DISTINCT_COUNTS {
7613 return Err(invalid("distinct catalog extension magic differs"));
7614 }
7615 for entry in &mut entries {
7616 for count in &mut entry.distincts {
7617 *count = match cur.u8()? {
7618 0 => None,
7619 1 => {
7620 let value = cur.u64()?;
7621 if value > entry.rows as u64 {
7622 return Err(invalid("distinct count exceeds table rows"));
7623 }
7624 Some(value)
7625 }
7626 _ => return Err(invalid("distinct count tag differs")),
7627 };
7628 }
7629 }
7630 }
7631 if !cur.done() {
7632 if cur.take(8)? != INTEGER_EXTREMES {
7633 return Err(invalid("integer extremes catalog extension magic differs"));
7634 }
7635 for entry in &mut entries {
7636 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
7637 *extremes = match cur.u8()? {
7638 0 => None,
7639 1 if integer_or_date(&field.ty) => Some(None),
7640 2 if integer_or_date(&field.ty) => {
7641 let low = i128::from_le_bytes(
7642 cur.take(16)?
7643 .try_into()
7644 .map_err(|_| invalid("minimum is truncated"))?,
7645 );
7646 let high = i128::from_le_bytes(
7647 cur.take(16)?
7648 .try_into()
7649 .map_err(|_| invalid("maximum is truncated"))?,
7650 );
7651 if low > high {
7652 return Err(invalid("integer extremes are reversed"));
7653 }
7654 Some(Some((low, high)))
7655 }
7656 _ => return Err(invalid("integer extremes tag or type differs")),
7657 };
7658 }
7659 }
7660 }
7661 if !cur.done() {
7662 if cur.take(8)? != COMPLETE_FREQUENCIES {
7663 return Err(invalid("numeric frequency catalog extension magic differs"));
7664 }
7665 for entry in &mut entries {
7666 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
7667 *frequencies = match cur.u8()? {
7668 0 => None,
7669 1 if integer_or_date(&field.ty) => {
7670 let len = cur.u8()? as usize;
7671 if len > MAX_CATALOG_FREQUENCIES {
7672 return Err(invalid("too many catalog numeric frequencies"));
7673 }
7674 let mut values = Vec::with_capacity(len);
7675 let mut total = 0_u64;
7676 for _ in 0..len {
7677 let value = match cur.u8()? {
7678 0 => None,
7679 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
7680 |_| invalid("numeric frequency value is truncated"),
7681 )?)),
7682 _ => return Err(invalid("numeric frequency value tag differs")),
7683 };
7684 if values.iter().any(|(held, _)| *held == value) {
7685 return Err(invalid("numeric frequency value repeats"));
7686 }
7687 let count = cur.u64()?;
7688 total = total
7689 .checked_add(count)
7690 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7691 values.push((value, count));
7692 }
7693 if total != entry.rows as u64 {
7694 return Err(invalid("numeric frequencies do not cover table rows"));
7695 }
7696 Some(values)
7697 }
7698 _ => return Err(invalid("numeric frequency tag or type differs")),
7699 };
7700 }
7701 }
7702 }
7703 if !cur.done() {
7704 return Err(invalid("catalog has trailing bytes"));
7705 }
7706 Ok((entries, views))
7707}
7708
7709struct Cursor<'a> {
7717 bytes: &'a [u8],
7718 at: usize,
7719 window: Option<Window<'a>>,
7720}
7721
7722struct Window<'a> {
7724 file: &'a File,
7725 offset: u64,
7726 length: usize,
7727 start: usize,
7729 held: Vec<u8>,
7730 size: usize,
7732}
7733
7734const DIRECTORY_WINDOW: usize = 64 << 10;
7736
7737impl<'a> Cursor<'a> {
7738 fn new(bytes: &'a [u8]) -> Self {
7739 Self { bytes, at: 0, window: None }
7740 }
7741
7742 fn over(file: &'a File, offset: u64, length: usize) -> Self {
7744 let window =
7745 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7746 Self { bytes: &[], at: 0, window: Some(window) }
7747 }
7748
7749 fn len(&self) -> usize {
7751 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7752 }
7753
7754 fn ensure(&mut self, len: usize) -> Result<()> {
7756 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7757 if end > self.len() {
7758 return Err(invalid("directory is truncated"));
7759 }
7760 let Some(window) = &mut self.window else { return Ok(()) };
7761 if self.at < window.start || end > window.start + window.held.len() {
7762 let want = len.max(window.size).min(window.length - self.at);
7763 window.start = self.at;
7764 window.held.resize(want, 0);
7765 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7766 }
7767 Ok(())
7768 }
7769
7770 fn held(&self, at: usize, len: usize) -> &[u8] {
7772 match &self.window {
7773 Some(window) => &window.held[at - window.start..at - window.start + len],
7774 None => &self.bytes[at..at + len],
7775 }
7776 }
7777
7778 #[inline]
7780 fn peek(&mut self, len: usize) -> Result<&[u8]> {
7781 if self.window.is_none() {
7782 let bytes = self.bytes;
7783 return Ok(&bytes[self.at..self.end(len)?]);
7784 }
7785 self.ensure(len)?;
7786 Ok(self.held(self.at, len))
7787 }
7788
7789 #[inline]
7795 fn take(&mut self, len: usize) -> Result<&[u8]> {
7796 if self.window.is_none() {
7797 let bytes = self.bytes;
7798 let (at, end) = (self.at, self.end(len)?);
7799 self.at = end;
7800 return Ok(&bytes[at..end]);
7801 }
7802 self.take_windowed(len)
7803 }
7804
7805 fn skip(&mut self, len: usize) -> Result<()> {
7807 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7808 if end > self.len() {
7809 return Err(invalid("directory is truncated"));
7810 }
7811 self.at = end;
7812 Ok(())
7813 }
7814
7815 fn skip_bound(&mut self) -> Result<()> {
7816 match self.u8()? {
7817 0 => Ok(()),
7818 1 => self.skip(16),
7819 2 => self.skip(8),
7820 3 => {
7821 let length = self.u32()? as usize;
7822 self.skip(length)
7823 }
7824 4 => self.skip(17),
7825 _ => Err(invalid("a stored bound has an unknown tag")),
7826 }
7827 }
7828
7829 #[inline]
7831 fn end(&self, len: usize) -> Result<usize> {
7832 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7833 if end > self.bytes.len() {
7834 return Err(invalid("directory is truncated"));
7835 }
7836 Ok(end)
7837 }
7838
7839 #[inline(never)]
7841 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7842 self.ensure(len)?;
7843 self.at += len;
7844 Ok(self.held(self.at - len, len))
7845 }
7846 #[inline]
7847 fn u8(&mut self) -> Result<u8> {
7848 Ok(self.take(1)?[0])
7849 }
7850 #[inline]
7851 fn u16(&mut self) -> Result<u16> {
7852 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7853 }
7854 #[inline]
7855 fn u32(&mut self) -> Result<u32> {
7856 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7857 }
7858 #[inline]
7859 fn u64(&mut self) -> Result<u64> {
7860 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7861 }
7862 fn var_u64(&mut self) -> Result<u64> {
7863 let mut value = 0_u64;
7864 for shift in (0..=63).step_by(7) {
7865 let byte = self.u8()?;
7866 let part = u64::from(byte & 0x7f);
7867 if shift == 63 && part > 1 {
7868 return Err(invalid("frequency ordinal varint overflows"));
7869 }
7870 value |= part << shift;
7871 if byte & 0x80 == 0 {
7872 return Ok(value);
7873 }
7874 }
7875 Err(invalid("frequency ordinal varint is too long"))
7876 }
7877 fn bound(&mut self) -> Result<Option<Bound>> {
7886 let rest = self.len().saturating_sub(self.at);
7887 let mut want = 32;
7888 loop {
7889 let offered = self.peek(want.min(rest))?;
7890 let mut used = 0;
7891 match bounds::get(offered, &mut used) {
7892 Ok(bound) => {
7893 self.at += used;
7894 return Ok(bound);
7895 }
7896 Err(_) if want < rest => want *= 2,
7897 Err(error) => return Err(error),
7898 }
7899 }
7900 }
7901 fn text(&mut self) -> Result<String> {
7902 let len = self.u16()? as usize;
7903 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
7904 }
7905 fn done(&self) -> bool {
7908 self.at >= self.len()
7909 }
7910 fn long_text(&mut self) -> Result<String> {
7917 let len = self.u32()? as usize;
7918 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
7919 }
7920}
7921
7922fn decode_summary(
7924 cur: &mut Cursor<'_>,
7925 field: &Field,
7926 rows: usize,
7927 values: bool,
7928) -> Result<Option<FrequencySummary>> {
7929 Ok(match cur.u8()? {
7930 0 => None,
7931 1 => {
7932 let omitted_max = cur.u64()?;
7933 let count = cur.u32()? as usize;
7934 if count > FREQUENCY_ENTRIES {
7935 return Err(invalid("frequency entry count exceeds its bound"));
7936 }
7937 let mut entries = Vec::with_capacity(count);
7938 for _ in 0..count {
7940 let value = match cur.u8()? {
7941 0 => FrequencyValue::Null,
7942 1 => FrequencyValue::Integer(i128::from_le_bytes(
7943 cur.take(16)?.try_into().expect("sixteen bytes"),
7944 )),
7945 2 => FrequencyValue::Code(cur.u32()?),
7946 _ => return Err(invalid("frequency value tag differs")),
7947 };
7948 let valid = matches!(
7949 (&field.ty, value),
7950 (_, FrequencyValue::Null)
7951 | (LogicalType::Varchar, FrequencyValue::Code(_))
7952 | (
7953 LogicalType::TinyInt
7954 | LogicalType::SmallInt
7955 | LogicalType::Integer
7956 | LogicalType::BigInt
7957 | LogicalType::UTinyInt
7958 | LogicalType::USmallInt
7959 | LogicalType::UInteger
7960 | LogicalType::UBigInt
7961 | LogicalType::Date
7962 | LogicalType::Timestamp,
7963 FrequencyValue::Integer(_),
7964 )
7965 );
7966 if !valid {
7967 return Err(invalid("frequency value does not match its column"));
7968 }
7969 let count = cur.u64()?;
7970 if count == 0 || count > rows as u64 {
7971 return Err(invalid("frequency count is outside the table"));
7972 }
7973 entries.push(FrequencyEntry { value, count });
7974 }
7975 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
7976 return Err(invalid("frequency entries are not descending"));
7977 }
7978 let ordinals = {
7979 let ordinal_count = cur.u32()? as usize;
7980 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
7981 return Err(invalid("frequency ordinal count exceeds its bound"));
7982 }
7983 let mut ordinals = Vec::with_capacity(ordinal_count);
7984 let mut previous = 0_u64;
7985 for at in 0..ordinal_count {
7986 let delta = cur.var_u64()?;
7987 if at != 0 && delta == 0 {
7988 return Err(invalid("frequency ordinals are not increasing"));
7989 }
7990 let ordinal = if at == 0 {
7991 delta
7992 } else {
7993 previous
7994 .checked_add(delta)
7995 .ok_or_else(|| invalid("frequency ordinal overflows"))?
7996 };
7997 if ordinal >= rows as u64 {
7998 return Err(invalid("frequency ordinal is outside the table"));
7999 }
8000 ordinals.push(ordinal);
8001 previous = ordinal;
8002 }
8003 ordinals
8004 };
8005 let ordinal_entries = if values {
8006 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8007 for _ in 0..ordinals.len() {
8008 let entry = cur.u16()?;
8009 if entry as usize >= entries.len() {
8010 return Err(invalid("frequency ordinal value is outside its entries"));
8011 }
8012 ordinal_entries.push(entry);
8013 }
8014 ordinal_entries
8015 } else {
8016 Vec::new()
8017 };
8018 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8019 }
8020 _ => return Err(invalid("frequency summary tag differs")),
8021 })
8022}
8023
8024fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8027 match cur.u8()? {
8028 0 => Ok(()),
8029 1 => {
8030 cur.skip(8)?;
8031 let entries = cur.u32()? as usize;
8032 if entries > FREQUENCY_ENTRIES {
8033 return Err(invalid("frequency entry count exceeds its bound"));
8034 }
8035 for _ in 0..entries {
8036 match cur.u8()? {
8037 0 => {}
8038 1 => cur.skip(16)?,
8039 2 => cur.skip(4)?,
8040 _ => return Err(invalid("frequency value tag differs")),
8041 }
8042 cur.skip(8)?;
8043 }
8044 let ordinals = cur.u32()? as usize;
8045 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8046 return Err(invalid("frequency ordinal count exceeds its bound"));
8047 }
8048 for _ in 0..ordinals {
8049 cur.var_u64()?;
8050 }
8051 if values {
8052 cur.skip(ordinals * 2)?;
8053 }
8054 Ok(())
8055 }
8056 _ => Err(invalid("frequency summary tag differs")),
8057 }
8058}
8059
8060fn quick_nonzero(
8064 mut cur: Cursor<'_>,
8065 name: &str,
8066 fields: &[Field],
8067 rows: usize,
8068 wanted: usize,
8069) -> Result<Option<u64>> {
8070 if cur.take(8)? != DIRECTORY || cur.text()? != name {
8071 return Err(invalid("table directory differs from the catalog"));
8072 }
8073 let width = cur.u16()? as usize;
8074 if width != fields.len() {
8075 return Err(invalid("table directory width differs from the catalog"));
8076 }
8077 for field in fields {
8078 let stored =
8079 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8080 if &stored != field {
8081 return Err(invalid("table directory schema differs from the catalog"));
8082 }
8083 }
8084 let mut dictionaries = Vec::with_capacity(width);
8085 for _ in 0..width {
8086 let held = match cur.u8()? {
8087 0 => false,
8088 1 => {
8089 cur.skip(20)?;
8090 true
8091 }
8092 _ => return Err(invalid("dictionary page tag differs")),
8093 };
8094 dictionaries.push(held);
8095 }
8096 for _ in 0..width {
8097 match cur.u8()? {
8098 0 => {}
8099 1 => cur.skip(8)?,
8100 _ => return Err(invalid("distinct count tag differs")),
8101 }
8102 }
8103 if cur.u64()? != rows as u64 {
8104 return Err(invalid("table row count differs from the catalog"));
8105 }
8106 let stripes = cur.u32()? as usize;
8107 let mut total = 0_usize;
8108 let mut nulls = 0_u64;
8109 for _ in 0..stripes {
8110 let parts = cur.u32()? as usize;
8111 if parts == 0 || parts > STRIPE_PARTS {
8112 return Err(invalid("stripe part count is outside its bound"));
8113 }
8114 let mut stripe_rows = 0_usize;
8115 for _ in 0..parts {
8116 stripe_rows = stripe_rows
8117 .checked_add(cur.u32()? as usize)
8118 .ok_or_else(|| invalid("stripe row count overflow"))?;
8119 }
8120 total =
8121 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8122 cur.skip(12 + width * 12)?;
8123 for (field, held) in fields.iter().zip(&dictionaries) {
8124 if field.ty == LogicalType::Varchar && *held {
8125 cur.skip(20)?;
8126 }
8127 }
8128 for _ in 0..width * 2 {
8129 match cur.u8()? {
8130 0 => {}
8131 1 => cur.skip(20)?,
8132 _ => return Err(invalid("stripe page tag differs")),
8133 }
8134 }
8135 for column in 0..width {
8136 cur.skip_bound()?;
8137 cur.skip_bound()?;
8138 let count = cur.u32()? as u64;
8139 if count > stripe_rows as u64 {
8140 return Err(invalid("null count exceeds stripe rows"));
8141 }
8142 if column == wanted {
8143 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
8144 }
8145 cur.skip(1)?;
8146 match cur.u8()? {
8147 0 => {}
8148 1 => cur.skip(16)?,
8149 _ => return Err(invalid("a stripe sum has an unknown tag")),
8150 }
8151 }
8152 }
8153 if total != rows {
8154 return Err(invalid("table row count differs from stripes"));
8155 }
8156 if cur.done() {
8157 return Ok(None);
8158 }
8159 let magic = cur.take(8)?;
8160 let values = magic == FREQUENCIES;
8161 if !values && magic != FREQUENCIES_V2 {
8162 return Err(invalid("directory extension magic differs"));
8163 }
8164 if cur.u16()? as usize != width {
8165 return Err(invalid("frequency column count differs"));
8166 }
8167 for _ in 0..wanted {
8168 skip_summary(&mut cur, values, rows)?;
8169 }
8170 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
8171 return Ok(None);
8172 };
8173 let zero = summary
8174 .entries
8175 .iter()
8176 .find(|entry| entry.value == FrequencyValue::Integer(0))
8177 .map(|entry| entry.count)
8178 .or_else(|| (summary.omitted_max == 0).then_some(0));
8179 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
8180}
8181
8182fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
8183 read_directory(Cursor::new(bytes), size, None)
8184}
8185
8186fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
8191 if cur.take(8)? != DIRECTORY {
8192 return Err(invalid("directory magic differs"));
8193 }
8194 let name = cur.text()?;
8195 let width = cur.u16()? as usize;
8196 let mut fields = Vec::with_capacity(width);
8197 for _ in 0..width {
8198 let name = cur.text()?;
8199 let ty = read_type(&mut cur)?;
8200 let not_null = match cur.u8()? {
8201 0 => false,
8202 1 => true,
8203 _ => return Err(invalid("nullability flag differs")),
8204 };
8205 fields.push(Field { name, ty, not_null });
8206 }
8207 let mut dictionaries = Vec::with_capacity(width);
8208 for _ in 0..width {
8209 dictionaries.push(match cur.u8()? {
8210 0 => None,
8211 1 => {
8212 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8213 let end = page
8214 .offset
8215 .checked_add(u64::from(page.length))
8216 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
8217 if page.offset < HEADER || end > size {
8222 return Err(invalid("dictionary page range is outside the file"));
8223 }
8224 Some(page)
8225 }
8226 _ => return Err(invalid("dictionary page tag differs")),
8227 });
8228 }
8229 let mut distincts = Vec::with_capacity(width);
8230 for _ in 0..width {
8231 distincts.push(match cur.u8()? {
8232 0 => None,
8233 1 => Some(cur.u64()?),
8234 _ => return Err(invalid("distinct count tag differs")),
8235 });
8236 }
8237 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8238 let count = cur.u32()? as usize;
8239 let mut stripes = Vec::with_capacity(count);
8240 let mut total = 0_usize;
8241 for _ in 0..count {
8242 let count = cur.u32()? as usize;
8243 if count == 0 || count > STRIPE_PARTS {
8244 return Err(invalid("stripe part count is outside its bound"));
8245 }
8246 let mut parts = Vec::with_capacity(count);
8247 let mut stripe_rows = 0_usize;
8248 for _ in 0..count {
8249 let rows = cur.u32()?;
8250 if rows == 0 {
8251 return Err(invalid("empty part"));
8252 }
8253 parts.push(rows);
8254 stripe_rows = stripe_rows
8255 .checked_add(rows as usize)
8256 .ok_or_else(|| invalid("stripe row count overflow"))?;
8257 }
8258 total =
8259 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8260 let index = Span { offset: cur.u64()?, length: cur.u32()? };
8261 let section = index_section(count)?;
8262 let wanted = section
8263 .checked_mul(width)
8264 .and_then(|bytes| u32::try_from(bytes).ok())
8265 .ok_or_else(|| invalid("index page length overflow"))?;
8266 let end = index
8267 .offset
8268 .checked_add(u64::from(index.length))
8269 .ok_or_else(|| invalid("index page offset overflow"))?;
8270 if index.offset < HEADER || end > size || index.length != wanted {
8271 return Err(invalid("index page range is outside the file"));
8272 }
8273 let mut pages = Vec::with_capacity(width);
8274 for _ in 0..width {
8275 let offset = cur.u64()?;
8276 let length = cur.u32()?;
8277 let end = offset
8278 .checked_add(u64::from(length))
8279 .ok_or_else(|| invalid("page offset overflow"))?;
8280 if offset < HEADER || end > size || length as usize > MAX_PAGE {
8281 return Err(invalid("page range is outside the file"));
8282 }
8283 pages.push(Span { offset, length });
8284 }
8285 let mut memberships = vec![None; width];
8286 for (column, field) in fields.iter().enumerate() {
8287 if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
8288 continue;
8289 }
8290 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8291 let end = page
8292 .offset
8293 .checked_add(u64::from(page.length))
8294 .ok_or_else(|| invalid("membership page offset overflow"))?;
8295 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8296 return Err(invalid("membership page range is outside the file"));
8297 }
8298 memberships[column] = Some(page);
8299 }
8300 let mut sieves = vec![None; width];
8301 for sieve in sieves.iter_mut().take(width) {
8302 match cur.u8()? {
8303 0 => continue,
8304 1 => {}
8305 _ => return Err(invalid("a sieve page has an unknown tag")),
8306 }
8307 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8308 let end = page
8309 .offset
8310 .checked_add(u64::from(page.length))
8311 .ok_or_else(|| invalid("sieve page offset overflow"))?;
8312 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8313 return Err(invalid("sieve page range is outside the file"));
8314 }
8315 *sieve = Some(page);
8316 }
8317 let mut part_ranges = vec![None; width];
8318 for held in part_ranges.iter_mut().take(width) {
8319 match cur.u8()? {
8320 0 => continue,
8321 1 => {}
8322 _ => return Err(invalid("a part range page has an unknown tag")),
8323 }
8324 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8325 let end = page
8326 .offset
8327 .checked_add(u64::from(page.length))
8328 .ok_or_else(|| invalid("part range page offset overflow"))?;
8329 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8330 return Err(invalid("part range page range is outside the file"));
8331 }
8332 *held = Some(page);
8333 }
8334 let mut ranges = Vec::with_capacity(width);
8335 for column in 0..width {
8336 let low = cur.bound()?;
8337 let high = cur.bound()?;
8338 let nulls = cur.u32()? as usize;
8339 if nulls > stripe_rows {
8340 return Err(invalid("null count exceeds stripe rows"));
8341 }
8342 let exact = cur.u8()? != 0;
8343 let sum = match cur.u8()? {
8344 0 => None,
8345 1 => Some(i128::from_le_bytes(
8346 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
8347 )),
8348 _ => return Err(invalid("a stripe sum has an unknown tag")),
8349 };
8350 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
8356 let low = low.map(|bound| scaled_as(bound, ty));
8357 let high = high.map(|bound| scaled_as(bound, ty));
8358 ranges.push(Range { low, high, nulls, exact, sum });
8359 }
8360 stripes.push(Stripe {
8361 rows: stripe_rows,
8362 parts,
8363 index,
8364 pages,
8365 memberships: Pages::from_slots(memberships)?,
8366 sieves: Pages::from_slots(sieves)?,
8367 part_ranges: Pages::from_slots(part_ranges)?,
8368 zone: Zone::from_ranges(ranges),
8369 });
8370 }
8371 if total != rows {
8372 return Err(invalid("table row count differs from stripes"));
8373 }
8374 let mut entry_counts = vec![0; width];
8377 let frequencies = if cur.done() {
8378 vec![None; width]
8379 } else {
8380 let frequency_magic = cur.take(8)?;
8381 let frequency_values = frequency_magic == FREQUENCIES;
8382 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
8383 return Err(invalid("directory extension magic differs"));
8384 }
8385 if cur.u16()? as usize != width {
8386 return Err(invalid("frequency column count differs"));
8387 }
8388 let mut frequencies = Vec::with_capacity(width);
8389 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
8390 let start = cur.at;
8391 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
8392 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
8393 frequencies.push(match (summary, stored_at) {
8394 (None, _) => None,
8395 (Some(summary), None) => Some(Frequencies::Held(summary)),
8396 (Some(_), Some(offset)) => Some(Frequencies::Stored {
8397 span: Span {
8398 offset: offset + start as u64,
8399 length: u32::try_from(cur.at - start)
8400 .map_err(|_| invalid("a frequency synopsis is too long"))?,
8401 },
8402 values: frequency_values,
8403 }),
8404 });
8405 }
8406 frequencies
8407 };
8408 let mut clustering = None;
8418 let mut sections = Vec::new();
8419 let mut pair_frequencies = Vec::new();
8420 let mut seen_pair_frequencies = false;
8421 let mut frequency_texts = vec![Vec::new(); width];
8422 let mut seen_frequency_texts = false;
8423 let mut host_groups = None;
8424 let mut seen_sections = false;
8425 let mut dictionary_payloads = Vec::new();
8426 let mut seen_payloads = false;
8427 let mut generation = 0;
8430 while !cur.done() {
8431 let mut tag = [0u8; 8];
8432 tag.copy_from_slice(cur.take(8)?);
8433 if &tag == PAIR_FREQUENCIES {
8434 if seen_pair_frequencies {
8435 return Err(invalid("directory names two pair frequency blocks"));
8436 }
8437 seen_pair_frequencies = true;
8438 let count = cur.u16()? as usize;
8439 if count > MAX_PAIR_FREQUENCIES {
8440 return Err(invalid("pair frequency count exceeds its bound"));
8441 }
8442 pair_frequencies = Vec::with_capacity(count);
8443 for _ in 0..count {
8444 let first = cur.u16()?;
8445 let second = cur.u16()?;
8446 let first_at = first as usize;
8447 let second_at = second as usize;
8448 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
8449 return Err(invalid("pair frequency first column has no synopsis"));
8450 }
8451 let first_entries = entry_counts[first_at];
8452 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
8453 || dictionaries.get(second_at).copied().flatten().is_none()
8454 {
8455 return Err(invalid("pair frequency second column has no stable dictionary"));
8456 }
8457 if pair_frequencies
8458 .iter()
8459 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
8460 {
8461 return Err(invalid("directory repeats a pair frequency summary"));
8462 }
8463 let omitted_max = cur.u64()?;
8464 if omitted_max > rows as u64 {
8465 return Err(invalid("pair frequency omitted count exceeds the table"));
8466 }
8467 let entries_count = cur.u16()? as usize;
8468 if entries_count > FREQUENCY_ENTRIES {
8469 return Err(invalid("pair frequency entry count exceeds its bound"));
8470 }
8471 let mut entries = Vec::with_capacity(entries_count);
8472 for _ in 0..entries_count {
8473 let first_entry = cur.u16()?;
8474 if first_entry as usize >= first_entries {
8475 return Err(invalid("pair frequency anchor is outside its synopsis"));
8476 }
8477 let second = match cur.u8()? {
8478 0 => None,
8479 1 => Some(cur.u32()?),
8480 _ => return Err(invalid("pair frequency string tag differs")),
8481 };
8482 let count = cur.u64()?;
8483 if count == 0 || count > rows as u64 {
8484 return Err(invalid("pair frequency count is outside the table"));
8485 }
8486 entries.push(PairFrequencyEntry { first_entry, second, count });
8487 }
8488 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8489 return Err(invalid("pair frequency entries are not descending"));
8490 }
8491 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
8492 }
8493 } else if &tag == FREQUENCY_TEXTS {
8494 if seen_frequency_texts {
8495 return Err(invalid("directory names two frequency text blocks"));
8496 }
8497 seen_frequency_texts = true;
8498 let columns = cur.u16()? as usize;
8499 if columns > width {
8500 return Err(invalid("frequency text column count exceeds the schema"));
8501 }
8502 for _ in 0..columns {
8503 let column = cur.u16()? as usize;
8504 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
8505 return Err(invalid("frequency text column is repeated or out of range"));
8506 }
8507 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8508 || dictionaries.get(column).copied().flatten().is_none()
8509 || frequencies.get(column).and_then(Option::as_ref).is_none()
8510 {
8511 return Err(invalid("frequency texts belong to a non-string synopsis"));
8512 }
8513 let count = cur.u16()? as usize;
8514 if count == 0 || count != entry_counts[column] {
8515 return Err(invalid("frequency text count differs from its synopsis"));
8516 }
8517 let mut texts = Vec::with_capacity(count);
8518 for _ in 0..count {
8519 texts.push(match cur.u8()? {
8520 0 => None,
8521 1 => {
8522 let length = cur.u32()? as usize;
8523 let bytes = cur.take(length)?.to_vec();
8524 std::str::from_utf8(&bytes)
8525 .map_err(|_| invalid("frequency text is not UTF-8"))?;
8526 Some(bytes)
8527 }
8528 _ => return Err(invalid("frequency text tag differs")),
8529 });
8530 }
8531 frequency_texts[column] = texts;
8532 }
8533 } else if &tag == HOST_GROUPS {
8534 if host_groups.is_some() {
8535 return Err(invalid("directory names two host group blocks"));
8536 }
8537 let column = cur.u16()? as usize;
8538 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8539 || dictionaries.get(column).copied().flatten().is_none()
8540 {
8541 return Err(invalid("host groups belong to a non-string dictionary"));
8542 }
8543 let omitted_max = cur.u64()?;
8544 if omitted_max > rows as u64 {
8545 return Err(invalid("host group bound exceeds the table"));
8546 }
8547 let count = cur.u16()? as usize;
8548 if count > host::CAPACITY {
8549 return Err(invalid("host group count exceeds its bound"));
8550 }
8551 let mut entries = Vec::with_capacity(count);
8552 let mut bytes = 0_usize;
8553 for _ in 0..count {
8554 let host_len = cur.u32()? as usize;
8555 bytes =
8556 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
8557 if bytes > host::BYTE_BUDGET {
8558 return Err(invalid("host groups exceed their byte budget"));
8559 }
8560 let host = std::str::from_utf8(cur.take(host_len)?)
8561 .map_err(|_| invalid("host is not UTF-8"))?
8562 .to_owned();
8563 let count = cur.u64()?;
8564 if count == 0 || count > rows as u64 {
8565 return Err(invalid("host group count exceeds the table"));
8566 }
8567 let bytes_sum = i128::from_le_bytes(
8568 cur.take(16)?
8569 .try_into()
8570 .map_err(|_| invalid("host length sum is truncated"))?,
8571 );
8572 if bytes_sum < 0 {
8573 return Err(invalid("host length sum is negative"));
8574 }
8575 let minimum_len = cur.u32()? as usize;
8576 bytes =
8577 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
8578 if bytes > host::BYTE_BUDGET {
8579 return Err(invalid("host groups exceed their byte budget"));
8580 }
8581 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
8582 .map_err(|_| invalid("host minimum is not UTF-8"))?
8583 .to_owned();
8584 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
8585 }
8586 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
8587 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
8588 {
8589 return Err(invalid("host groups are not in certified order"));
8590 }
8591 host_groups = Some(host::HostSummary { column, omitted_max, entries });
8592 } else if &tag == CLUSTERING {
8593 if clustering.is_some() {
8594 return Err(invalid("directory names two clustering declarations"));
8595 }
8596 let bucket = Width::from_tag(cur.u8()?)
8597 .ok_or_else(|| invalid("clustering width tag differs"))?;
8598 let count = cur.u16()? as usize;
8599 let mut columns = Vec::with_capacity(count.min(fields.len()));
8600 for _ in 0..count {
8601 columns.push(u32::from(cur.u16()?));
8602 }
8603 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
8606 invalid("stored clustering declaration does not match the table it is on")
8607 })?);
8608 } else if &tag == SECTIONS {
8609 if seen_sections {
8610 return Err(invalid("directory names two section tables"));
8611 }
8612 seen_sections = true;
8613 generation = cur.u64()?;
8614 let count = cur.u16()? as usize;
8615 if count > MAX_SECTIONS {
8616 return Err(invalid("section count exceeds its bound"));
8617 }
8618 sections = Vec::with_capacity(count);
8619 for _ in 0..count {
8622 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
8623 }
8624 for held in §ions {
8625 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
8626 return Err(invalid("a section's extent table overflows the file"));
8627 };
8628 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
8632 return Err(invalid("a section's extent table is outside the file"));
8633 }
8634 if held.extents == 0 && held.extent_bytes != 0 {
8635 return Err(invalid("a section with no extents names an extent table"));
8636 }
8637 }
8638 } else if &tag == DICTIONARY_PAYLOADS {
8639 if seen_payloads {
8640 return Err(invalid("directory names two dictionary payload blocks"));
8641 }
8642 seen_payloads = true;
8643 let count = cur.u16()? as usize;
8644 if count != fields.len() {
8645 return Err(invalid("dictionary payload block does not match the table's columns"));
8646 }
8647 dictionary_payloads = Vec::with_capacity(count);
8648 for _ in 0..count {
8649 let bytes = cur.u64()?;
8650 if bytes > size {
8651 return Err(invalid("a dictionary payload is larger than the file"));
8652 }
8653 dictionary_payloads.push(bytes);
8654 }
8655 } else {
8656 return Err(invalid("directory extension magic differs"));
8657 }
8658 }
8659 if !cur.done() {
8660 return Err(invalid("directory has trailing bytes"));
8661 }
8662 Ok(Table {
8663 name,
8664 fields,
8665 stripes,
8666 rows,
8667 dictionaries,
8668 dictionary_payloads,
8669 distincts,
8670 frequencies,
8671 pair_frequencies,
8672 frequency_texts,
8673 host_groups,
8674 clustering,
8675 generation,
8676 sections,
8677 })
8678}
8679
8680fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
8682 bounds::put(out, bound)
8683}
8684
8685#[derive(Debug)]
8702struct Codes;
8703
8704impl chooser::Chooser for Codes {
8705 fn name(&self) -> &'static str {
8706 "codes"
8707 }
8708
8709 fn narrow_strings(
8710 &self,
8711 _values: &[&[u8]],
8712 offered: &[string::Kind],
8713 _depth: u8,
8714 ) -> Vec<string::Kind> {
8715 offered.to_vec()
8718 }
8719
8720 fn narrow_integers(
8721 &self,
8722 _values: &[i64],
8723 offered: &[integer::Kind],
8724 depth: u8,
8725 ) -> Vec<integer::Kind> {
8726 narrowed_to(Codes::keep(depth), offered)
8729 }
8730
8731 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8732 Codes::keep(depth).contains(&kind)
8733 }
8734}
8735
8736impl Codes {
8737 fn keep(depth: u8) -> &'static [integer::Kind] {
8738 if depth == 0 {
8739 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8740 } else {
8741 &[integer::Kind::Constant, integer::Kind::Packed]
8742 }
8743 }
8744}
8745
8746fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8754 let narrowed: Vec<integer::Kind> =
8755 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8756 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8757}
8758
8759#[derive(Debug)]
8771struct Fixed;
8772
8773impl chooser::Chooser for Fixed {
8774 fn name(&self) -> &'static str {
8775 "fixed"
8776 }
8777
8778 fn narrow_strings(
8779 &self,
8780 _values: &[&[u8]],
8781 offered: &[string::Kind],
8782 _depth: u8,
8783 ) -> Vec<string::Kind> {
8784 offered.to_vec()
8785 }
8786
8787 fn narrow_integers(
8788 &self,
8789 _values: &[i64],
8790 offered: &[integer::Kind],
8791 depth: u8,
8792 ) -> Vec<integer::Kind> {
8793 narrowed_to(Fixed::keep(depth), offered)
8794 }
8795
8796 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8797 Fixed::keep(depth).contains(&kind)
8798 }
8799}
8800
8801impl Fixed {
8802 fn keep(depth: u8) -> &'static [integer::Kind] {
8803 if depth == 0 {
8804 &[
8805 integer::Kind::Constant,
8806 integer::Kind::Packed,
8807 integer::Kind::Delta,
8808 integer::Kind::Rle,
8809 integer::Kind::Sparse,
8810 integer::Kind::Strided,
8811 ]
8812 } else {
8813 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8814 }
8815 }
8816}
8817
8818fn widened(data: &Data) -> Option<Vec<i64>> {
8825 match data {
8826 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8827 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8828 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8829 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8830 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8831 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8832 Data::Int64(values) => Some(values.to_vec()),
8833 _ => None,
8834 }
8835}
8836
8837trait Narrow: Copy {
8844 const BIASED: (u32, u64);
8849
8850 fn narrow(value: i64) -> Self;
8852}
8853
8854#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8871fn residue<T: Narrow>(value: i64) -> u64 {
8872 let (bits, bias) = T::BIASED;
8873 (value as u64).wrapping_add(bias) >> bits
8874}
8875
8876macro_rules! narrows {
8881 ($($ty:ty => $bias:expr),* $(,)?) => {$(
8882 impl Narrow for $ty {
8883 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8884
8885 #[allow(
8886 clippy::cast_possible_truncation,
8887 clippy::cast_sign_loss,
8888 reason = "the caller has checked the bits this truncates away"
8889 )]
8890 fn narrow(value: i64) -> Self {
8891 value as Self
8892 }
8893 }
8894 )*};
8895}
8896
8897narrows! {
8898 i8 => 1 << 7,
8899 u8 => 0,
8900 i16 => 1 << 15,
8901 u16 => 0,
8902 i32 => 1 << 31,
8903 u32 => 0,
8904}
8905
8906fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
8919 let mut spilled = 0u64;
8920 for value in values {
8921 spilled |= residue::<T>(*value);
8922 }
8923 if spilled != 0 {
8924 return Err(invalid("page value is not of its type"));
8925 }
8926 Ok(values.iter().map(|value| T::narrow(*value)).collect())
8927}
8928
8929fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
8934 Ok(match ty {
8935 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
8936 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
8937 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
8938 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
8939 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
8940 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
8941 LogicalType::BigInt
8942 | LogicalType::Timestamp
8943 | LogicalType::Time
8944 | LogicalType::TimeTz
8945 | LogicalType::TimestampTz
8946 | LogicalType::TimestampS
8947 | LogicalType::TimestampMs
8948 | LogicalType::TimestampNs => Data::Int64(values.into()),
8949 LogicalType::Decimal { .. } => match ty.physical() {
8952 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
8953 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
8954 PhysicalType::Int64 => Data::Int64(values.into()),
8955 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
8956 },
8957 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
8958 })
8959}
8960
8961fn plain_width(ty: &LogicalType) -> Option<usize> {
8964 Some(match ty {
8965 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
8966 LogicalType::SmallInt | LogicalType::USmallInt => 2,
8967 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
8968 LogicalType::BigInt
8969 | LogicalType::Timestamp
8970 | LogicalType::Time
8971 | LogicalType::TimeTz
8972 | LogicalType::TimestampTz
8973 | LogicalType::TimestampS
8974 | LogicalType::TimestampMs
8975 | LogicalType::TimestampNs => 8,
8976 LogicalType::Decimal { .. } => match ty.physical() {
8977 PhysicalType::Int16 => 2,
8978 PhysicalType::Int32 => 4,
8979 PhysicalType::Int64 => 8,
8980 _ => return None,
8983 },
8984 _ => return None,
8985 })
8986}
8987
8988fn cascaded(
8994 flat: &Vector,
8995 ty: &LogicalType,
8996 packed: Option<&Packed<'_>>,
8997 settling: &mut Settling,
8998) -> Result<Option<Vec<u8>>> {
8999 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
9000 let Some(values) = widened(data) else { return Ok(None) };
9001 let plain = values.len().saturating_mul(width);
9002 let best = match packed {
9003 Some(packed) => plain.min(21 + size_of_val(packed.words())),
9005 None => plain,
9006 };
9007 let out = settling.encode(&values)?;
9008 Ok((out.len() < best).then_some(out))
9009}
9010
9011const SEARCH_EVERY: usize = 16;
9018
9019#[derive(Debug, Default)]
9024struct Settling {
9025 shape: Option<Shape>,
9028 since: usize,
9030}
9031
9032impl Settling {
9033 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
9040 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
9041 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
9042 let out = integer::encode_with(values, &replay)?;
9043 if !replay.held() {
9044 self.settle(&out, values.len(), replay.first_offered())?;
9045 return Ok(out);
9046 }
9047 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
9048 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
9049 self.since += 1;
9050 return Ok(out);
9051 }
9052 }
9053 let search = chooser::Replay::new(&[], &Fixed);
9055 let out = integer::encode_with(values, &search)?;
9056 self.settle(&out, values.len(), search.first_offered())?;
9057 Ok(out)
9058 }
9059
9060 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
9061 let kinds = integer::shape(out)?;
9062 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
9063 self.since = 0;
9064 Ok(())
9065 }
9066}
9067
9068#[derive(Debug)]
9070struct Shape {
9071 kinds: Vec<integer::Kind>,
9072 offered: Vec<integer::Kind>,
9073 len: usize,
9074 rows: usize,
9075}
9076
9077fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
9115 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
9116 let mut payload = 0_usize;
9117 for row in 0..flat.len() {
9118 let text = flat.text_at(row).unwrap_or("").as_bytes();
9119 payload = payload.saturating_add(text.len());
9120 values.push(text);
9121 }
9122 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
9124 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
9125 return Ok(None);
9126 };
9127 Ok((out.len() < plain).then_some(out))
9128}
9129
9130fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
9131 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
9132 let coded = integer::encode_with(&wide, &Codes)?;
9133 let plain = codes.len().saturating_mul(size_of::<u32>());
9134 Ok((coded.len() < plain).then_some(coded))
9135}
9136
9137fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
9140 let flag = match flat.validity() {
9141 Validity::AllValid => 0,
9142 Validity::AllInvalid => 1,
9143 Validity::Mask(_) => 2,
9144 };
9145 out.push(flag);
9146 if flag == 2 {
9147 for group in (0..flat.len()).step_by(8) {
9148 let mut bits = 0_u8;
9149 for bit in 0..8 {
9150 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
9151 bits |= 1 << bit;
9152 }
9153 }
9154 out.push(bits);
9155 }
9156 }
9157}
9158
9159fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
9166 let coded = encoded_codes(codes)?;
9167 let mut out = Vec::with_capacity(
9168 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
9169 );
9170 out.push(if coded.is_some() { 4 } else { 3 });
9171 out.extend_from_slice(validity);
9172 match coded {
9173 Some(coded) => out.extend_from_slice(&coded),
9174 None => {
9175 for &code in codes {
9176 put_u32(&mut out, code);
9177 }
9178 }
9179 }
9180 Ok(out)
9181}
9182
9183fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
9186 let ty = vector.logical_type();
9187 let flat = vector.flatten()?;
9189 let mut out = Vec::new();
9190 let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
9191 let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
9192 text_compressed(&flat)?
9193 } else {
9194 None
9195 };
9196 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
9197 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
9198 let cascade =
9202 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
9203 out.push(if cascade.is_some() {
9204 5
9205 } else if dictionary.is_some() {
9206 1
9207 } else if compressed_text.is_some() {
9208 6
9209 } else if packed.is_some() {
9210 2
9211 } else {
9212 0
9213 });
9214 push_validity(&mut out, &flat);
9215 if let Some(cascade) = cascade {
9216 out.extend_from_slice(&cascade);
9217 return Ok(out);
9218 }
9219 if let Some(dictionary) = dictionary {
9220 out.extend_from_slice(&dictionary);
9221 return Ok(out);
9222 }
9223 if let Some(compressed_text) = compressed_text {
9224 out.extend_from_slice(&compressed_text);
9225 return Ok(out);
9226 }
9227 if let Some(packed) = packed {
9228 if packed.offset() != 0 {
9229 return Err(invalid("writer received a sliced packed vector"));
9230 }
9231 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
9232 out.extend_from_slice(&packed.base().to_le_bytes());
9233 put_u32(
9234 &mut out,
9235 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
9236 );
9237 for word in packed.words() {
9238 put_u64(&mut out, *word);
9239 }
9240 return Ok(out);
9241 }
9242 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
9243 match (ty, data) {
9244 (LogicalType::TinyInt, Data::Int8(values)) => {
9245 for value in &**values {
9246 out.extend_from_slice(&value.to_le_bytes());
9247 }
9248 }
9249 (LogicalType::UTinyInt, Data::UInt8(values)) => {
9250 for value in &**values {
9251 out.extend_from_slice(&value.to_le_bytes());
9252 }
9253 }
9254 (LogicalType::SmallInt, Data::Int16(values)) => {
9255 for value in &**values {
9256 out.extend_from_slice(&value.to_le_bytes());
9257 }
9258 }
9259 (LogicalType::USmallInt, Data::UInt16(values)) => {
9260 for value in &**values {
9261 out.extend_from_slice(&value.to_le_bytes());
9262 }
9263 }
9264 (LogicalType::UInteger, Data::UInt32(values)) => {
9265 for value in &**values {
9266 out.extend_from_slice(&value.to_le_bytes());
9267 }
9268 }
9269 (LogicalType::UBigInt, Data::UInt64(values)) => {
9270 for value in &**values {
9271 out.extend_from_slice(&value.to_le_bytes());
9272 }
9273 }
9274 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
9275 for value in &**values {
9276 out.extend_from_slice(&value.to_le_bytes());
9277 }
9278 }
9279 (
9280 LogicalType::BigInt
9281 | LogicalType::Timestamp
9282 | LogicalType::Time
9283 | LogicalType::TimeTz
9284 | LogicalType::TimestampTz
9285 | LogicalType::TimestampS
9286 | LogicalType::TimestampMs
9287 | LogicalType::TimestampNs,
9288 Data::Int64(values),
9289 ) => {
9290 for value in &**values {
9291 out.extend_from_slice(&value.to_le_bytes());
9292 }
9293 }
9294 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
9297 for value in &**values {
9298 out.extend_from_slice(&value.to_le_bytes());
9299 }
9300 }
9301 (LogicalType::UHugeInt, Data::UInt128(values)) => {
9302 for value in &**values {
9303 out.extend_from_slice(&value.to_le_bytes());
9304 }
9305 }
9306 (LogicalType::Float, Data::Float32(values)) => {
9309 for value in &**values {
9310 out.extend_from_slice(&value.to_le_bytes());
9311 }
9312 }
9313 (LogicalType::Double, Data::Float64(values)) => {
9314 for value in &**values {
9315 out.extend_from_slice(&value.to_le_bytes());
9316 }
9317 }
9318 (LogicalType::Interval, Data::Interval(values)) => {
9322 for (months, days, micros) in &**values {
9323 out.extend_from_slice(&months.to_le_bytes());
9324 out.extend_from_slice(&days.to_le_bytes());
9325 out.extend_from_slice(µs.to_le_bytes());
9326 }
9327 }
9328 (LogicalType::Boolean, Data::Bool(values)) => {
9329 for value in &**values {
9330 out.push(u8::from(*value));
9331 }
9332 }
9333 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
9336 for value in &**values {
9337 out.extend_from_slice(&value.to_le_bytes());
9338 }
9339 }
9340 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
9341 for value in &**values {
9342 out.extend_from_slice(&value.to_le_bytes());
9343 }
9344 }
9345 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
9346 for value in &**values {
9347 out.extend_from_slice(&value.to_le_bytes());
9348 }
9349 }
9350 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
9351 for value in &**values {
9352 out.extend_from_slice(&value.to_le_bytes());
9353 }
9354 }
9355 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
9360 let mut bytes = Vec::new();
9361 put_u32(&mut out, 0);
9362 for row in 0..vector.len() {
9363 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
9364 bytes.extend_from_slice(value);
9365 put_u32(
9366 &mut out,
9367 u32::try_from(bytes.len())
9368 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
9369 );
9370 }
9371 out.extend_from_slice(&bytes);
9372 }
9373 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
9374 }
9375 Ok(out)
9376}
9377
9378fn put_varint(out: &mut Vec<u8>, mut value: u32) {
9379 while value >= 0x80 {
9380 out.push((value as u8 & 0x7f) | 0x80);
9381 value >>= 7;
9382 }
9383 out.push(value as u8);
9384}
9385
9386fn unique_codes(codes: &[u32]) -> Vec<u32> {
9388 let mut unique = codes.to_vec();
9389 unique.sort_unstable();
9390 unique.dedup();
9391 unique
9392}
9393
9394fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
9400 let mut lists = lists;
9401 while lists.len() > 1 {
9402 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
9403 for pair in lists.chunks(2) {
9404 match pair {
9405 [left, right] => next.push(merged_pair(left, right)),
9406 [only] => next.push(only.clone()),
9407 _ => {}
9408 }
9409 }
9410 lists = next;
9411 }
9412 lists.pop().unwrap_or_default()
9413}
9414
9415fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
9416 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
9417 let mut at = 0;
9418 let mut to = 0;
9419 while at < left.len() && to < right.len() {
9420 match left[at].cmp(&right[to]) {
9421 Ordering::Less => {
9422 out.push(left[at]);
9423 at += 1;
9424 }
9425 Ordering::Greater => {
9426 out.push(right[to]);
9427 to += 1;
9428 }
9429 Ordering::Equal => {
9430 out.push(left[at]);
9431 at += 1;
9432 to += 1;
9433 }
9434 }
9435 }
9436 out.extend_from_slice(&left[at..]);
9437 out.extend_from_slice(&right[to..]);
9438 out
9439}
9440
9441fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
9446 let mut merged = Range::default();
9447 let mut first = true;
9448 for range in ranges {
9449 merged.nulls = merged.nulls.saturating_add(range.nulls);
9450 merged.sum = match (merged.sum.take(), range.sum) {
9454 (Some(held), Some(next)) if !first => held.checked_add(next),
9455 (_, next) if first => next,
9456 _ => None,
9457 };
9458 merged.exact = if first { range.exact } else { merged.exact && range.exact };
9459 if first {
9460 merged.low = range.low;
9461 merged.high = range.high;
9462 first = false;
9463 continue;
9464 }
9465 merged.low = match (merged.low.take(), range.low) {
9466 (Some(held), Some(next)) => Some(held.smaller(next)),
9467 _ => None,
9468 };
9469 merged.high = match (merged.high.take(), range.high) {
9470 (Some(held), Some(next)) => Some(held.larger(next)),
9471 _ => None,
9472 };
9473 }
9474 merged
9475}
9476
9477fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
9490 match bound {
9491 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
9492 value.truncate(PART_BOUND_BYTES);
9493 if !high {
9494 return Some(Bound::Bytes(value));
9495 }
9496 while let Some(last) = value.pop() {
9497 if last < u8::MAX {
9498 value.push(last + 1);
9499 return Some(Bound::Bytes(value));
9500 }
9501 }
9502 None
9503 }
9504 other => other,
9505 }
9506}
9507
9508fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
9516 let mut out = Vec::new();
9517 put_u32(
9518 &mut out,
9519 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9520 );
9521 for range in ranges {
9522 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
9523 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
9524 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
9525 }
9526 Ok(out)
9527}
9528
9529fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
9531 let mut cur = Cursor::new(bytes);
9532 let parts = cur.u32()? as usize;
9533 let mut out = Vec::new();
9534 for _ in 0..parts {
9535 let low = cur.bound()?;
9536 let high = cur.bound()?;
9537 let nulls = cur.u32()? as usize;
9538 out.push(Range { low, high, nulls, exact: false, sum: None });
9539 }
9540 Ok(out)
9541}
9542
9543fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
9544 let held: Vec<&Option<Sieve>> = sieves.collect();
9545 let mut out = Vec::new();
9546 put_u32(
9547 &mut out,
9548 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9549 );
9550 for sieve in &held {
9551 let length = sieve.as_ref().map_or(0, Sieve::len);
9552 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
9553 }
9554 for sieve in held.into_iter().flatten() {
9556 out.extend_from_slice(&sieve.to_bytes());
9557 }
9558 Ok(out)
9559}
9560
9561fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
9567 let parts = u32::from_le_bytes(
9568 bytes
9569 .get(..4)
9570 .ok_or_else(|| invalid("sieve page is truncated"))?
9571 .try_into()
9572 .map_err(|_| invalid("sieve page is truncated"))?,
9573 ) as usize;
9574 let mut lengths = Vec::with_capacity(parts);
9575 for part in 0..parts {
9576 let at = 4 + part * 4;
9577 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
9578 lengths.push(u32::from_le_bytes(
9579 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
9580 ) as usize);
9581 }
9582 let mut at = 4 + parts * 4;
9583 let mut out = Vec::with_capacity(parts);
9584 for length in lengths {
9585 if length == 0 {
9586 out.push(None);
9587 continue;
9588 }
9589 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
9590 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
9591 out.push(Sieve::from_bytes(field));
9592 at = end;
9593 }
9594 if at != bytes.len() {
9595 return Err(invalid("sieve page has trailing bytes"));
9596 }
9597 Ok(out)
9598}
9599
9600fn encode_membership(unique: &[u32]) -> Vec<u8> {
9606 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
9607 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
9608 let mut previous = 0;
9609 for (at, &code) in unique.iter().enumerate() {
9610 put_varint(&mut out, if at == 0 { code } else { code - previous });
9611 previous = code;
9612 }
9613 out
9614}
9615
9616fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
9617 let mut value = 0_u32;
9618 for shift in (0..35).step_by(7) {
9619 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
9620 *at += 1;
9621 let part = u32::from(byte & 0x7f);
9622 if shift == 28 && part > 0x0f {
9623 return Err(invalid("membership varint overflow"));
9624 }
9625 value = value
9626 .checked_add(
9627 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
9628 )
9629 .ok_or_else(|| invalid("membership varint overflow"))?;
9630 if byte & 0x80 == 0 {
9631 return Ok(value);
9632 }
9633 }
9634 Err(invalid("membership varint is too long"))
9635}
9636
9637fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
9638 let mut at = 0;
9639 let count = take_varint(bytes, &mut at)? as usize;
9640 let mut codes = Vec::with_capacity(count);
9641 let mut previous = 0_u32;
9642 for index in 0..count {
9643 let delta = take_varint(bytes, &mut at)?;
9644 let code = if index == 0 {
9645 delta
9646 } else {
9647 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
9648 };
9649 if index > 0 && code <= previous {
9650 return Err(invalid("membership codes are not increasing"));
9651 }
9652 codes.push(code);
9653 previous = code;
9654 }
9655 if at != bytes.len() {
9656 return Err(invalid("membership page has trailing bytes"));
9657 }
9658 Ok(codes)
9659}
9660
9661fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
9662 let mut by_text = HashMap::new();
9663 let mut values = Vec::new();
9664 let mut codes = Vec::with_capacity(vector.len());
9665 let mut plain_bytes = 0_usize;
9666 for row in 0..vector.len() {
9667 let text = vector.text_at(row).unwrap_or("");
9668 plain_bytes = plain_bytes.saturating_add(text.len());
9669 let code = match by_text.get(text) {
9670 Some(&code) => code,
9671 None => {
9672 let code = u32::try_from(values.len())
9673 .map_err(|_| invalid("too many dictionary values"))?;
9674 by_text.insert(text, code);
9675 values.push(text);
9676 code
9677 }
9678 };
9679 codes.push(code);
9680 }
9681 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
9682 let encoded = 8_usize
9683 .saturating_add((values.len() + 1).saturating_mul(4))
9684 .saturating_add(dictionary_bytes)
9685 .saturating_add(codes.len().saturating_mul(4));
9686 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
9687 if encoded >= plain {
9688 return Ok(None);
9689 }
9690 let mut out = Vec::with_capacity(encoded);
9691 put_u32(
9692 &mut out,
9693 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
9694 );
9695 put_u32(
9696 &mut out,
9697 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
9698 );
9699 let mut offset = 0_u32;
9700 put_u32(&mut out, offset);
9701 for value in &values {
9702 offset = offset
9703 .checked_add(
9704 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
9705 )
9706 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
9707 put_u32(&mut out, offset);
9708 }
9709 for value in values {
9710 out.extend_from_slice(value.as_bytes());
9711 }
9712 for code in codes {
9713 put_u32(&mut out, code);
9714 }
9715 Ok(Some(out))
9716}
9717
9718struct Room<'a, T> {
9720 state: &'a Mutex<(T, usize)>,
9721 finished: &'a Condvar,
9722 bytes: usize,
9723}
9724
9725impl<T> Drop for Room<'_, T> {
9726 fn drop(&mut self) {
9727 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
9728 held.1 -= self.bytes;
9729 drop(held);
9730 self.finished.notify_all();
9731 }
9732}
9733
9734struct ClosedDictionary {
9736 distinct: u64,
9737 frequencies: FrequencySummary,
9738 texts: Vec<Option<Vec<u8>>>,
9739 hosts: Option<host::HostSummary>,
9740 encoded: EncodedDictionary,
9741 payload: u64,
9743}
9744
9745struct EncodedDictionary {
9746 index: Vec<u8>,
9747 ranks: Vec<u8>,
9748 grams: Vec<u8>,
9749}
9750
9751fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9792 let mut work = vec![(0, codes.len(), 0)];
9793 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9794 while let Some((from, to, depth)) = work.pop() {
9795 let part = &mut codes[from..to];
9796 keyed.clear();
9797 keyed.extend(part.iter().map(|&code| {
9798 let value = values(code);
9799 let rest = value.get(depth..).unwrap_or_default();
9800 (head(rest), rest.len().min(8) as u8, code)
9801 }));
9802 keyed.sort_unstable();
9803 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9804 *slot = entry.2;
9805 }
9806 let mut start = 0;
9807 while start < keyed.len() {
9808 let (key, taken, _) = keyed[start];
9809 let mut end = start + 1;
9810 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9811 end += 1;
9812 }
9813 if taken == 8 && end - start > 1 {
9814 work.push((from + start, from + end, depth + 8));
9815 }
9816 start = end;
9817 }
9818 }
9819}
9820
9821const PARALLEL_SORT_MIN: usize = 1 << 16;
9823
9824const BUCKETS_PER_WORKER: usize = 4;
9827
9828const SAMPLES_PER_BUCKET: usize = 32;
9830
9831fn sort_by_value_across<'a>(
9849 codes: &mut [u32],
9850 values: impl Fn(u32) -> &'a [u8] + Sync,
9851 workers: usize,
9852) {
9853 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9854 sort_by_value(codes, values);
9855 return;
9856 }
9857 let buckets = workers * BUCKETS_PER_WORKER;
9858 let wanted = buckets * SAMPLES_PER_BUCKET;
9859 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9860 sort_by_value(&mut sample, &values);
9861 let splitters =
9862 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9863 let values = &values;
9864 let splitters = &splitters;
9865 let per = codes.len().div_ceil(workers);
9866 let places = std::thread::scope(|scope| {
9868 codes
9869 .chunks(per)
9870 .map(|run| {
9871 scope.spawn(move || {
9872 run.iter()
9873 .map(|&code| {
9874 let value = values(code);
9875 splitters.partition_point(|splitter| *splitter <= value) as u32
9876 })
9877 .collect::<Vec<_>>()
9878 })
9879 })
9880 .collect::<Vec<_>>()
9881 .into_iter()
9882 .flat_map(|handle| {
9883 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9884 })
9885 .collect::<Vec<_>>()
9886 });
9887 let mut starts = vec![0_usize; buckets + 1];
9888 for &place in &places {
9889 starts[place as usize + 1] += 1;
9890 }
9891 for bucket in 0..buckets {
9892 starts[bucket + 1] += starts[bucket];
9893 }
9894 let mut laid = vec![0_u32; codes.len()];
9895 let mut next = starts.clone();
9896 for (&code, &place) in codes.iter().zip(&places) {
9897 laid[next[place as usize]] = code;
9898 next[place as usize] += 1;
9899 }
9900 drop(places);
9901 let mut runs = Vec::with_capacity(buckets);
9902 let mut rest = laid.as_mut_slice();
9903 for bucket in 0..buckets {
9904 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
9905 runs.push(run);
9906 rest = after;
9907 }
9908 runs.sort_by_key(|run| run.len());
9910 let queue = Mutex::new(runs);
9911 std::thread::scope(|scope| {
9912 for _ in 0..workers {
9913 scope.spawn(|| {
9914 loop {
9915 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
9916 let Some(run) = taken else { break };
9917 sort_by_value(run, values);
9918 }
9919 });
9920 }
9921 });
9922 codes.copy_from_slice(&laid);
9923}
9924
9925fn head(bytes: &[u8]) -> u64 {
9927 let mut word = [0; 8];
9928 let take = bytes.len().min(8);
9929 word[..take].copy_from_slice(&bytes[..take]);
9930 u64::from_be_bytes(word)
9931}
9932
9933fn encode_global_dictionary(
9944 dictionary: &GlobalDictionary,
9945 order: &[(u64, u32)],
9946 places: &[Placed],
9947 scattered: bool,
9948) -> Result<EncodedDictionary> {
9949 let values = dictionary.values();
9950 if order.len() != values {
9951 return Err(invalid("global dictionary order does not cover its values"));
9952 }
9953 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
9954 if places.len() != blocks {
9955 return Err(invalid("global dictionary payload is not the blocks it says it is"));
9956 }
9957 if dictionary.grams.len() != blocks {
9958 return Err(invalid("global dictionary signatures do not cover its blocks"));
9959 }
9960 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
9961 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
9962 let offset_bits = offset_width(&dictionary.ends);
9963 let payload_words = if scattered { 3 } else { 2 };
9964 let index_len = DICTIONARY_HEADER
9965 .checked_add(offset_bytes(values, offset_bits))
9966 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
9967 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
9968 .and_then(|len| len.checked_add(8))
9969 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
9970 let mut index = Vec::with_capacity(index_len);
9971 put_u32(
9972 &mut index,
9973 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
9974 );
9975 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
9976 put_u32(
9977 &mut index,
9978 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
9979 );
9980 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 }) | DICTIONARY_GRAMS;
9981 put_u32(&mut index, offset_bits as u32 | flag);
9982 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
9983 let mut end = 0_u64;
9988 for place in places {
9989 if scattered {
9990 put_u64(&mut index, place.start);
9991 put_u64(&mut index, place.length);
9992 } else {
9993 end = end
9994 .checked_add(place.length)
9995 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
9996 put_u64(&mut index, end);
9997 }
9998 }
9999 for place in places {
10000 put_u64(&mut index, place.hash);
10001 }
10002 if rank_ends.len() != rank_blocks {
10005 return Err(invalid("global dictionary order is not the blocks it says it is"));
10006 }
10007 for end in &rank_ends {
10008 put_u64(&mut index, *end);
10009 }
10010 let mut at = 0_usize;
10011 for end in &rank_ends {
10012 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
10013 put_u64(&mut index, checksum(&ranks[at..end]));
10014 at = end;
10015 }
10016 let gram_len = blocks
10017 .checked_mul(TEXT_GRAM_BYTES)
10018 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
10019 let mut grams = Vec::with_capacity(gram_len);
10020 for block in &dictionary.grams {
10021 grams.extend_from_slice(block);
10022 }
10023 put_u64(&mut index, checksum(&grams));
10024 if index.len() != index_len {
10025 return Err(invalid("global dictionary index is not the length it was laid out for"));
10026 }
10027 Ok(EncodedDictionary { index, ranks, grams })
10028}
10029
10030const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
10037
10038fn payload_shapes() -> Vec<chooser::Settled> {
10064 let integers = vec![integer::Kind::Packed];
10065 [
10066 vec![string::Kind::Front, string::Kind::Lz],
10067 vec![string::Kind::Lz, string::Kind::Fsst],
10068 vec![string::Kind::Lz, string::Kind::Plain],
10069 vec![string::Kind::Fsst],
10070 vec![string::Kind::Plain],
10071 ]
10072 .into_iter()
10073 .map(|strings| chooser::Settled::new(strings, integers.clone()))
10074 .collect()
10075}
10076
10077fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
10084 let started = profile.map(|_| std::time::Instant::now());
10085 file.sync_all().map_err(io)?;
10086 if let (Some(profile), Some(started)) = (profile, started) {
10087 profile.waited(
10088 Stage::Publish,
10089 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
10090 );
10091 }
10092 Ok(())
10093}
10094
10095#[derive(Debug)]
10100pub(crate) struct Unencoded {
10101 column: usize,
10102 at: usize,
10103 ends: Vec<u32>,
10104 bytes: Vec<u8>,
10105 shape: chooser::Settled,
10106}
10107
10108impl Unencoded {
10109 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
10111 let values = block_values(&self.ends, &self.bytes);
10112 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
10113 }
10114
10115 pub(crate) fn place(&self) -> (usize, usize) {
10117 (self.column, self.at)
10118 }
10119}
10120
10121pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
10125
10126fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
10128 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
10129 for value in values {
10130 for gram in value.windows(4) {
10131 for bit in gram_bits(gram) {
10132 grams[bit / 8] |= 1 << (bit % 8);
10133 }
10134 }
10135 }
10136 grams
10137}
10138
10139fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
10141 let mut out = Vec::with_capacity(ends.len());
10142 let mut from = 0;
10143 for &to in ends {
10144 out.push(&bytes[from..to as usize]);
10145 from = to as usize;
10146 }
10147 out
10148}
10149
10150fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10157 for dictionary in dictionaries.iter_mut().flatten() {
10158 if !dictionary.early.is_empty() {
10159 return Err(Error::internal("a dictionary block handed out never came back"));
10160 }
10161 dictionary.seal_rest();
10162 }
10163 encode_waiting(dictionaries)?;
10164 if dictionaries
10167 .iter()
10168 .flatten()
10169 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
10170 {
10171 return Err(Error::internal("a dictionary block handed out never came back"));
10172 }
10173 Ok(())
10174}
10175
10176fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10179 let jobs = dictionaries
10180 .iter()
10181 .enumerate()
10182 .flat_map(|(column, held)| {
10183 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
10184 })
10185 .collect::<Vec<_>>();
10186 if jobs.is_empty() {
10187 return Ok(());
10188 }
10189 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
10190 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
10191 Ok((column, at, held.encode_waiting(at)?))
10192 };
10193 let workers = std::thread::available_parallelism()
10194 .map_or(1, usize::from)
10195 .min(MAX_FREQUENCY_WORKERS)
10196 .min(jobs.len());
10197 let made = if workers <= 1 {
10198 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
10199 } else {
10200 let next = AtomicUsize::new(0);
10201 let jobs = &jobs;
10202 let pieces = std::thread::scope(|scope| {
10203 (0..workers)
10204 .map(|_| {
10205 scope.spawn(|| {
10206 let mut mine = Vec::new();
10207 loop {
10208 let job = next.fetch_add(1, Atomic::Relaxed);
10209 let Some(&(column, at)) = jobs.get(job) else { break };
10210 mine.push(one(column, at)?);
10211 }
10212 Ok(mine)
10213 })
10214 })
10215 .collect::<Vec<_>>()
10216 .into_iter()
10217 .map(|handle| {
10218 handle
10219 .join()
10220 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
10221 })
10222 .collect::<Result<Vec<_>>>()
10223 })?;
10224 pieces.into_iter().flatten().collect()
10225 };
10226 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
10227 (0..dictionaries.len()).map(|_| Vec::new()).collect();
10228 for (column, at, bytes) in made {
10229 done[column].push((at, bytes));
10230 }
10231 for (column, mut made) in done.into_iter().enumerate() {
10232 if made.is_empty() {
10233 continue;
10234 }
10235 let Some(held) = dictionaries[column].as_mut() else { continue };
10236 made.sort_by_key(|(at, _)| *at);
10237 let waiting = std::mem::take(&mut held.waiting);
10238 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
10239 if held.encoded() != at {
10240 return Err(Error::internal("a dictionary block was encoded out of order"));
10241 }
10242 held.push_block(block);
10243 }
10244 }
10245 Ok(())
10246}
10247
10248fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
10258 let mut best: Option<(chooser::Settled, usize)> = None;
10259 for shape in payload_shapes() {
10260 let mut size = 0;
10261 for block in sample {
10262 size += string::encode_with(block, &shape)?.len();
10263 }
10264 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
10265 best = Some((shape, size));
10266 }
10267 }
10268 best.map(|(shape, _)| shape)
10269 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
10270}
10271
10272fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
10279 let mut out = Vec::with_capacity(order.len() * 4);
10280 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
10281 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
10282 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
10283 for block in order.chunks(TEXT_RANK_BLOCK) {
10284 let base = block.first().map_or(0, |&(head, _)| head);
10287 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
10288 let width = (u64::BITS - span.leading_zeros()) as usize;
10289 heads.clear();
10290 codes.clear();
10291 for &(head, code) in block {
10292 heads.push(head.wrapping_sub(base));
10293 codes.push(u64::from(code));
10294 }
10295 put_u64(&mut out, base);
10296 out.push(width as u8);
10297 bitpack::pack_tail(&heads, width, &mut out)
10298 .map_err(|_| invalid("global dictionary heads do not pack"))?;
10299 bitpack::pack_tail(&codes, code_bits, &mut out)
10300 .map_err(|_| invalid("global dictionary codes do not pack"))?;
10301 ends.push(out.len() as u64);
10302 }
10303 Ok((out, ends))
10304}
10305
10306fn open_global_dictionary(
10313 file: Arc<File>,
10314 page: Page,
10315 ty: &LogicalType,
10316 keep_budget: usize,
10317) -> Result<Vector> {
10318 if ty != &LogicalType::Varchar {
10319 return Err(invalid("global dictionary belongs to a non-string column"));
10320 }
10321 let mut header = [0; DICTIONARY_HEADER];
10322 read_at(&file, page.offset, &mut header)?;
10323 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
10324 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
10325 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
10326 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10327 let scattered = width & DICTIONARY_SCATTERED != 0;
10328 let has_grams = width & DICTIONARY_GRAMS != 0;
10329 let offset_bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
10330 if per_block != TEXT_PAYLOAD_VALUES {
10331 return Err(invalid("global dictionary block width differs"));
10332 }
10333 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
10334 return Err(invalid("global dictionary block count differs from its value count"));
10335 }
10336 if offset_bits > u32::BITS as usize {
10337 return Err(invalid("global dictionary packs offsets past a payload"));
10338 }
10339 let offset_len = offset_bytes(count, offset_bits);
10340 let ranks = count;
10345 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
10346 let payload_words = if scattered { 3 } else { 2 };
10350 let hash_len = blocks
10351 .checked_mul(payload_words * 8)
10352 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10353 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
10354 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
10355 let gram_len = if has_grams {
10356 blocks
10357 .checked_mul(TEXT_GRAM_BYTES)
10358 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
10359 } else {
10360 0
10361 };
10362 let index_len = DICTIONARY_HEADER
10363 .checked_add(offset_len)
10364 .and_then(|len| len.checked_add(hash_len))
10365 .ok_or_else(|| invalid("global dictionary header overflow"))?;
10366 if index_len > page.length as usize {
10367 return Err(invalid("global dictionary offset index exceeds its page"));
10368 }
10369 let mut index = vec![0; index_len];
10370 index[..DICTIONARY_HEADER].copy_from_slice(&header);
10371 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
10372 if checksum(&index) != page.hash {
10373 return Err(invalid("global dictionary index checksum differs"));
10374 }
10375 let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
10376 let word_end = index_len - usize::from(has_grams) * 8;
10377 let gram_hash = has_grams
10378 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
10379 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
10380 .chunks_exact(8)
10381 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
10382 .collect::<Vec<_>>();
10383 let mut rest = words.split_off(blocks * payload_words);
10384 let rank_hashes = rest.split_off(rank_blocks);
10385 let rank_ends = rest;
10386 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
10389 return Err(invalid("global dictionary order blocks do not rise"));
10390 }
10391 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
10392 .map_err(|_| invalid("global dictionary rank overflow"))?;
10393 let body_len = index_len
10394 .checked_add(rank_len)
10395 .ok_or_else(|| invalid("global dictionary header overflow"))?;
10396 if body_len > page.length as usize {
10397 return Err(invalid("global dictionary order exceeds its page"));
10398 }
10399 let gram_end = body_len
10400 .checked_add(gram_len)
10401 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
10402 if gram_end > page.length as usize {
10403 return Err(invalid("global dictionary signatures exceed their page"));
10404 }
10405 let grams = gram_hash.map(|hash| NativeGrams {
10406 start: page.offset + body_len as u64,
10407 length: gram_len,
10408 hash,
10409 loaded: OnceLock::new(),
10410 });
10411 let hashes = words.split_off(blocks * (payload_words - 1));
10412 let (starts, lengths) = if scattered {
10413 let mut starts = Vec::with_capacity(blocks);
10414 let mut lengths = Vec::with_capacity(blocks);
10415 for pair in words.chunks_exact(2) {
10416 starts.push(pair[0]);
10417 lengths.push(pair[1]);
10418 }
10419 (starts, lengths)
10420 } else {
10421 let base = page.offset + gram_end as u64;
10425 let mut starts = Vec::with_capacity(blocks);
10426 let mut lengths = Vec::with_capacity(blocks);
10427 let mut at = 0_u64;
10428 for &end in &words {
10429 let len = end
10430 .checked_sub(at)
10431 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
10432 starts.push(base + at);
10433 lengths.push(len);
10434 at = end;
10435 }
10436 (starts, lengths)
10437 };
10438 let stored_len = page.length as u64 - gram_end as u64;
10444 if scattered && stored_len == 0 {
10445 let size = file.metadata().map_err(io)?.len();
10446 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
10447 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
10448 });
10449 if !inside {
10450 return Err(invalid("global dictionary block lies outside the file"));
10451 }
10452 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
10453 return Err(invalid("global dictionary blocks do not bound the payload"));
10454 }
10455 Vector::external_text(
10456 LogicalType::Varchar,
10457 Arc::new(NativeText {
10458 file,
10459 values: count,
10460 offsets,
10461 offset_bits,
10462 value_ends: OnceLock::new(),
10463 value_lens: OnceLock::new(),
10464 ends_asked: AtomicUsize::new(0),
10465 ranks,
10466 rank_at: page.offset + index_len as u64,
10467 rank_ends,
10468 rank_hashes,
10469 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
10470 code_bits: code_width(count),
10471 code_ranks: OnceLock::new(),
10472 starts,
10473 lengths,
10474 hashes,
10475 grams,
10476 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
10477 keep_budget,
10478 payload_kept: AtomicUsize::new(0),
10479 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
10480 searched: Mutex::new(HashMap::new()),
10481 }),
10482 )
10483}
10484
10485fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
10498 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
10500 let mut cur = Cursor::new(bytes);
10501 let codec = cur.u8()?;
10502 if cur.u8()? == 2 {
10503 cur.take(rows.div_ceil(8))?;
10504 }
10505 Ok((codec, cur.at))
10506 }
10507 let Ok((codec, at)) = cascade_at(rows, bytes) else {
10508 return "UNREADABLE".to_string();
10509 };
10510 let tail = &bytes[at..];
10511 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
10512 match codec {
10513 0 => match ty {
10514 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
10515 _ => "FIXED".to_string(),
10516 },
10517 1 => "DICT(PLAIN)".to_string(),
10518 2 => "FOR+BITPACK".to_string(),
10519 3 => "TABLE DICT".to_string(),
10520 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
10521 5 => described(integer::describe(tail)),
10522 6 => described(string::describe(tail)),
10523 other => format!("CODEC {other}"),
10524 }
10525}
10526
10527fn decode_selected_stable_codes(
10532 rows: usize,
10533 bytes: &[u8],
10534 positions: &[usize],
10535 out: &mut Vec<Option<u32>>,
10536) -> Result<bool> {
10537 if positions.windows(2).any(|pair| pair[0] >= pair[1])
10538 || positions.last().is_some_and(|&position| position >= rows)
10539 {
10540 return Err(invalid("selected code positions are not sorted and in range"));
10541 }
10542 let mut cur = Cursor::new(bytes);
10543 let codec = cur.u8()?;
10544 if codec != 3 && codec != 4 {
10545 return Ok(false);
10546 }
10547 let flag = cur.u8()?;
10548 let mask = match flag {
10549 0 | 1 => None,
10550 2 => {
10551 let at = cur.at;
10552 let len = rows.div_ceil(8);
10553 cur.take(len)?;
10554 Some((at, len))
10555 }
10556 _ => return Err(invalid("page validity tag differs")),
10557 };
10558 let valid = |row: usize| match flag {
10559 0 => true,
10560 1 => false,
10561 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
10562 _ => unreachable!("the validity tag was checked"),
10563 };
10564 if codec == 4 {
10565 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
10566 for (&row, code) in positions.iter().zip(wide) {
10567 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
10568 out.push(valid(row).then_some(code));
10569 }
10570 return Ok(true);
10571 }
10572 let codes_at = cur.at;
10573 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
10574 cur.take(codes_len)?;
10575 if cur.at != bytes.len() {
10576 return Err(invalid("global code page has trailing bytes"));
10577 }
10578 let codes = &bytes[codes_at..codes_at + codes_len];
10579 for &row in positions {
10580 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
10581 let code = u32::from_le_bytes(
10582 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
10583 );
10584 out.push(valid(row).then_some(code));
10585 }
10586 Ok(true)
10587}
10588
10589fn decode(
10590 ty: &LogicalType,
10591 rows: usize,
10592 bytes: &[u8],
10593 global: Option<Arc<Vector>>,
10594) -> Result<Vector> {
10595 let mut cur = Cursor::new(bytes);
10596 let codec = cur.u8()?;
10597 let flag = cur.u8()?;
10598 let validity = match flag {
10599 0 => Validity::AllValid,
10600 1 => Validity::AllInvalid,
10601 2 => {
10602 let mask = cur.take(rows.div_ceil(8))?;
10603 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
10604 }
10605 _ => return Err(invalid("page validity tag differs")),
10606 };
10607 if codec == 1 {
10608 if ty != &LogicalType::Varchar {
10609 return Err(invalid("dictionary codec belongs to a non-string page"));
10610 }
10611 let count = cur.u32()? as usize;
10612 let payload_len = cur.u32()? as usize;
10613 let offset_bytes = cur.take(
10614 (count + 1)
10615 .checked_mul(4)
10616 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
10617 )?;
10618 let offsets = offset_bytes
10619 .chunks_exact(4)
10620 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10621 .collect::<Vec<_>>();
10622 let payload = cur.take(payload_len)?.to_vec();
10623 if offsets.first() != Some(&0)
10624 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10625 || offsets.windows(2).any(|pair| pair[0] > pair[1])
10626 {
10627 return Err(invalid("dictionary offsets do not bound the payload"));
10628 }
10629 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
10632 for pair in offsets.windows(2) {
10633 strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
10634 }
10635 let mut codes = Vec::with_capacity(rows);
10636 for _ in 0..rows {
10637 codes.push(cur.u32()?);
10638 }
10639 if codes.iter().any(|code| *code as usize >= count) {
10640 return Err(invalid("dictionary code is out of range"));
10641 }
10642 if cur.at != bytes.len() {
10643 return Err(invalid("dictionary page has trailing bytes"));
10644 }
10645 let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
10646 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
10647 }
10648 if codec == 3 || codec == 4 {
10649 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
10650 let codes = if codec == 4 {
10651 let wide = integer::decode(&bytes[cur.at..])?;
10654 if wide.len() != rows {
10655 return Err(invalid("encoded code page holds the wrong number of rows"));
10656 }
10657 let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
10664 if seen < 0 || seen > i64::from(u32::MAX) {
10665 return Err(invalid("code is not a code"));
10666 }
10667 wide.iter().map(|&code| code as u32).collect()
10668 } else {
10669 let mut codes = Vec::with_capacity(rows);
10670 for _ in 0..rows {
10671 codes.push(cur.u32()?);
10672 }
10673 if cur.at != bytes.len() {
10674 return Err(invalid("global code page has trailing bytes"));
10675 }
10676 codes
10677 };
10678 let highest = codes.iter().copied().max();
10679 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
10680 .with_validity(validity));
10681 }
10682 if codec == 6 {
10683 if ty != &LogicalType::Varchar {
10684 return Err(invalid("compressed text codec belongs to a non-string page"));
10685 }
10686 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
10690 if ends.len() != rows {
10691 return Err(invalid("compressed text page holds the wrong number of rows"));
10692 }
10693 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10696 let mut start = 0;
10697 for end in ends {
10698 let len = end
10699 .checked_sub(start)
10700 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10701 values.push_in_place(start, len)?;
10702 start = end;
10703 }
10704 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
10705 }
10706 if codec == 5 {
10707 let values = integer::decode(&bytes[cur.at..])?;
10709 if values.len() != rows {
10710 return Err(invalid("cascade page holds the wrong number of rows"));
10711 }
10712 let data = narrowed(ty, values)?;
10713 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
10714 }
10715 if codec == 2 {
10716 let width = u32::from(cur.u8()?);
10717 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
10718 let count = cur.u32()? as usize;
10719 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
10720 let words: Vec<u64> = cur
10721 .take(length)?
10722 .chunks_exact(8)
10723 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
10724 .collect();
10725 if cur.at != bytes.len() {
10726 return Err(invalid("packed page has trailing bytes"));
10727 }
10728 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
10729 }
10730 if codec != 0 {
10731 return Err(invalid("page codec is unknown"));
10732 }
10733 let data = match ty {
10734 LogicalType::TinyInt => {
10735 let values = cur.take(rows)?;
10736 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
10737 }
10738 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
10739 LogicalType::SmallInt => {
10740 let values =
10741 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10742 Data::Int16(
10743 values
10744 .chunks_exact(2)
10745 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10746 .collect::<Vec<_>>()
10747 .into(),
10748 )
10749 }
10750 LogicalType::USmallInt => {
10751 let values =
10752 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10753 Data::UInt16(
10754 values
10755 .chunks_exact(2)
10756 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
10757 .collect::<Vec<_>>()
10758 .into(),
10759 )
10760 }
10761 LogicalType::UInteger => {
10762 let values =
10763 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10764 Data::UInt32(
10765 values
10766 .chunks_exact(4)
10767 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
10768 .collect::<Vec<_>>()
10769 .into(),
10770 )
10771 }
10772 LogicalType::UBigInt => {
10773 let values =
10774 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10775 Data::UInt64(
10776 values
10777 .chunks_exact(8)
10778 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
10779 .collect::<Vec<_>>()
10780 .into(),
10781 )
10782 }
10783 LogicalType::Integer | LogicalType::Date => {
10784 let values =
10785 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10786 Data::Int32(
10787 values
10788 .chunks_exact(4)
10789 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10790 .collect::<Vec<_>>()
10791 .into(),
10792 )
10793 }
10794 LogicalType::BigInt
10795 | LogicalType::Timestamp
10796 | LogicalType::Time
10797 | LogicalType::TimeTz
10798 | LogicalType::TimestampTz
10799 | LogicalType::TimestampS
10800 | LogicalType::TimestampMs
10801 | LogicalType::TimestampNs => {
10802 let values =
10803 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10804 Data::Int64(
10805 values
10806 .chunks_exact(8)
10807 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10808 .collect::<Vec<_>>()
10809 .into(),
10810 )
10811 }
10812 LogicalType::HugeInt | LogicalType::Uuid => {
10813 let values =
10814 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10815 Data::Int128(
10816 values
10817 .chunks_exact(16)
10818 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10819 .collect::<Vec<_>>()
10820 .into(),
10821 )
10822 }
10823 LogicalType::UHugeInt => {
10824 let values =
10825 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10826 Data::UInt128(
10827 values
10828 .chunks_exact(16)
10829 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10830 .collect::<Vec<_>>()
10831 .into(),
10832 )
10833 }
10834 LogicalType::Float => {
10835 let values =
10836 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10837 Data::Float32(
10838 values
10839 .chunks_exact(4)
10840 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10841 .collect::<Vec<_>>()
10842 .into(),
10843 )
10844 }
10845 LogicalType::Double => {
10846 let values =
10847 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10848 Data::Float64(
10849 values
10850 .chunks_exact(8)
10851 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
10852 .collect::<Vec<_>>()
10853 .into(),
10854 )
10855 }
10856 LogicalType::Interval => {
10857 let values =
10858 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10859 Data::Interval(
10860 values
10861 .chunks_exact(16)
10862 .map(|item| {
10863 (
10864 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
10865 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
10866 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
10867 )
10868 })
10869 .collect::<Vec<_>>()
10870 .into(),
10871 )
10872 }
10873 LogicalType::Boolean => {
10874 let values = cur.take(rows)?;
10875 if values.iter().any(|value| *value > 1) {
10876 return Err(invalid("boolean page has another value"));
10877 }
10878 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
10879 }
10880 LogicalType::Decimal { .. } => match ty.physical() {
10883 PhysicalType::Int16 => {
10884 let values =
10885 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10886 Data::Int16(
10887 values
10888 .chunks_exact(2)
10889 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10890 .collect::<Vec<_>>()
10891 .into(),
10892 )
10893 }
10894 PhysicalType::Int32 => {
10895 let values =
10896 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10897 Data::Int32(
10898 values
10899 .chunks_exact(4)
10900 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10901 .collect::<Vec<_>>()
10902 .into(),
10903 )
10904 }
10905 PhysicalType::Int64 => {
10906 let values =
10907 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10908 Data::Int64(
10909 values
10910 .chunks_exact(8)
10911 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10912 .collect::<Vec<_>>()
10913 .into(),
10914 )
10915 }
10916 _ => {
10917 let values =
10918 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10919 Data::Int128(
10920 values
10921 .chunks_exact(16)
10922 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10923 .collect::<Vec<_>>()
10924 .into(),
10925 )
10926 }
10927 },
10928 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
10929 let offset_bytes = cur
10930 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
10931 let offsets = offset_bytes
10932 .chunks_exact(4)
10933 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10934 .collect::<Vec<_>>();
10935 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
10936 if offsets.first() != Some(&0)
10937 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10938 || offsets.windows(2).any(|pair| pair[0] > pair[1])
10939 {
10940 return Err(invalid("string offsets do not bound the payload"));
10941 }
10942 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10950 let text = ty == &LogicalType::Varchar;
10951 for pair in offsets.windows(2) {
10952 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
10953 if text {
10954 values.push_in_place(at, len)?;
10955 } else {
10956 values.push_bytes_in_place(at, len)?;
10957 }
10958 }
10959 Data::Varlen(values)
10960 }
10961 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
10962 };
10963 if cur.at != bytes.len() {
10964 return Err(invalid("page has trailing bytes"));
10965 }
10966 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
10967}
10968
10969#[cfg(test)]
10970mod tests {
10971 use std::fs;
10972 use std::io::{Seek, SeekFrom, Write};
10973 use std::path::PathBuf;
10974 use std::time::{SystemTime, UNIX_EPOCH};
10975
10976 use rudb_common::Stat;
10977 use rudb_common::Value;
10978 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
10979 use rudb_common::stat::Provenance;
10980
10981 use super::*;
10982
10983 #[test]
10984 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
10985 let bytes: Vec<u8> =
10986 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
10987 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
10988 let whole = content_name(&bytes[..length]);
10989 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
10990 let mut namer = ContentNamer::default();
10991 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
10992 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
10993 }
10994 }
10995 }
10996
10997 #[derive(Debug)]
11000 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
11001
11002 impl chooser::Chooser for TestsEverything<'_> {
11003 fn name(&self) -> &'static str {
11004 "tests everything"
11005 }
11006
11007 fn narrow_strings(
11008 &self,
11009 values: &[&[u8]],
11010 offered: &[string::Kind],
11011 depth: u8,
11012 ) -> Vec<string::Kind> {
11013 self.0.narrow_strings(values, offered, depth)
11014 }
11015
11016 fn narrow_integers(
11017 &self,
11018 values: &[i64],
11019 offered: &[integer::Kind],
11020 depth: u8,
11021 ) -> Vec<integer::Kind> {
11022 self.0.narrow_integers(values, offered, depth)
11023 }
11024 }
11025
11026 #[test]
11027 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
11028 let columns: Vec<Vec<i64>> = vec![
11029 vec![],
11030 vec![5; 1000],
11031 (0..1000).collect(),
11032 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
11033 (0..1000).map(|row| row / 50).collect(),
11034 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
11035 (0..1000).map(|row| (row * 7919) % 13).collect(),
11036 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
11037 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
11038 (0..1000).map(|row| i64::MIN + row % 3).collect(),
11039 ];
11040 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
11041 for column in &columns {
11042 for chooser in choosers {
11043 let quick = integer::encode_with(column, chooser).unwrap();
11044 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
11045 assert_eq!(
11046 quick,
11047 full,
11048 "{} on {:?}",
11049 chooser.name(),
11050 &column[..column.len().min(8)]
11051 );
11052 }
11053 }
11054 }
11055
11056 #[test]
11059 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
11060 let mut settling = Settling::default();
11061 for part in 0..STRIPE_PARTS as i64 {
11062 let values: Vec<i64> = (0..2048)
11063 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
11064 .collect();
11065 let searched = integer::encode_with(&values, &Fixed).unwrap();
11066 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
11067 }
11068 }
11069
11070 #[test]
11074 fn a_column_that_changes_under_the_shape_is_searched_again() {
11075 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11076 let mut noise = move || {
11077 state ^= state << 13;
11078 state ^= state >> 7;
11079 state ^= state << 17;
11080 (state % 1_000_000) as i64
11081 };
11082 let mut settling = Settling::default();
11083 for part in 0..STRIPE_PARTS as i64 {
11084 let values: Vec<i64> = match part / 16 {
11085 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
11086 1 => (0..2048).map(|_| noise()).collect(),
11087 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
11088 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
11089 };
11090 let settled = settling.encode(&values).unwrap();
11091 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
11092 let searched = integer::encode_with(&values, &Fixed).unwrap();
11093 assert!(
11094 settled.len() * 4 <= searched.len() * 5,
11095 "part {part}: {} settled against {} searched, {} against {}",
11096 settled.len(),
11097 searched.len(),
11098 integer::describe(&settled).unwrap(),
11099 integer::describe(&searched).unwrap(),
11100 );
11101 }
11102 }
11103
11104 #[test]
11105 fn checksum_matches_fixed_vectors() {
11106 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
11107 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
11108 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
11109 }
11110
11111 #[test]
11112 fn sorting_across_threads_matches_sorting_on_one() {
11113 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11114 let mut next = move || {
11115 state ^= state << 13;
11116 state ^= state >> 7;
11117 state ^= state << 17;
11118 state
11119 };
11120 let mut values = Vec::new();
11121 for at in 0..150_000_u64 {
11122 let value = match next() % 6 {
11123 0 => Vec::new(),
11124 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
11125 2 => format!("https://example.com/path/{at}").into_bytes(),
11126 3 => b"same".to_vec(),
11127 4 => vec![0xff; (next() % 12) as usize],
11128 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
11129 };
11130 values.push(value);
11131 }
11132 let value = |code: u32| values[code as usize].as_slice();
11133 for workers in [1, 2, 3, 8, 32] {
11134 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
11135 let mut across = one.clone();
11136 sort_by_value(&mut one, value);
11137 sort_by_value_across(&mut across, value, workers);
11138 assert_eq!(one, across, "{workers} workers");
11139 }
11140 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
11141 sort_by_value_across(&mut sorted, value, 8);
11142 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
11143 }
11144
11145 fn path(label: &str) -> PathBuf {
11146 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
11147 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
11148 }
11149
11150 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
11155 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
11156 (0..dictionary.values())
11157 .map(|code| {
11158 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
11159 flat[from..to].to_vec()
11160 })
11161 .collect()
11162 }
11163
11164 fn attached(table: &Table) -> Vec<&Section> {
11171 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
11172 }
11173
11174 #[test]
11176 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
11177 const SPANS: usize = 64;
11178 const SPAN: usize = 512;
11179 let path = path("positional");
11180 let content: Vec<u8> =
11181 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
11182 fs::write(&path, &content).expect("the file is written");
11183 let file = Arc::new(File::open(&path).expect("the file opens"));
11184 std::thread::scope(|scope| {
11185 for _ in 0..8 {
11186 let file = Arc::clone(&file);
11187 scope.spawn(move || {
11188 for _ in 0..64 {
11189 for span in 0..SPANS {
11190 let mut bytes = [0_u8; SPAN];
11191 read_at(&file, (span * SPAN) as u64, &mut bytes)
11192 .expect("the span reads");
11193 assert!(
11194 bytes.iter().all(|byte| *byte == span as u8),
11195 "span {span} came back as {}",
11196 bytes[0],
11197 );
11198 }
11199 }
11200 });
11201 }
11202 });
11203 let mut past = [0_u8; SPAN];
11204 let end = (SPANS * SPAN) as u64;
11205 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
11206 assert!(error.message().contains("ends before its declared length"), "{error}");
11207 drop(file);
11208 let _ = fs::remove_file(&path);
11209 }
11210
11211 #[test]
11217 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
11218 let path = path("cursor");
11219 let mut writer = Writer::create(
11220 &path,
11221 "items",
11222 vec![
11223 Field::required("id", LogicalType::Integer),
11224 Field::new("text", LogicalType::Varchar),
11225 ],
11226 )
11227 .expect("new file");
11228 writer.append(&sample()).expect("first part");
11229 writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
11230 writer.append(&sample()).expect("second part");
11231 writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
11232 writer.finish().expect("commit");
11233 let reader = Reader::open(&path).expect("reopen from disk");
11234 assert_eq!(reader.table().rows(), 6);
11235 let ids = reader.read(0, &[0]).expect("the integer page reads back");
11236 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
11237 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
11238 let text = reader.read(1, &[1]).expect("the text page reads back");
11239 assert_eq!(text.value_at(1, 0), Value::Null);
11240 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11241 let end = reader.table().stripes().iter().flat_map(|stripe| {
11244 stripe
11245 .pages
11246 .iter()
11247 .map(|page| page.offset + u64::from(page.length))
11248 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
11249 });
11250 let last = end.fold(HEADER, u64::max);
11251 let directory = fs::metadata(&path).expect("the file is there").len();
11252 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
11253 fs::remove_file(path).expect("remove scratch file");
11254 }
11255
11256 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
11262 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
11263 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
11264 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11265 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
11266 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
11267 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
11268 DICTIONARY_HEADER as u64
11269 + offset_bytes(count as usize, bits) as u64
11270 + blocks * payload_words * 8
11271 + rank_blocks * 16
11272 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
11273 }
11274
11275 fn sample() -> Chunk {
11276 Chunk::new(vec![
11277 Vector::from_values(
11278 LogicalType::Integer,
11279 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
11280 )
11281 .expect("integers"),
11282 Vector::from_values(
11283 LogicalType::Varchar,
11284 &[
11285 Value::Varchar("alpha".into()),
11286 Value::Null,
11287 Value::Varchar("long text after a slash".into()),
11288 ],
11289 )
11290 .expect("strings"),
11291 ])
11292 .expect("matching rows")
11293 }
11294
11295 fn sample_ids() -> Chunk {
11296 Chunk::new(vec![
11297 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
11298 .expect("integers"),
11299 ])
11300 .expect("one column")
11301 }
11302
11303 #[test]
11304 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
11305 let path = path("nulls_for_the_planner");
11308 let mut writer =
11309 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
11310 .expect("new file");
11311 let rows = Chunk::new(vec![
11312 Vector::from_values(
11313 LogicalType::Integer,
11314 &[
11315 Value::Integer(4),
11316 Value::Null,
11317 Value::Integer(9),
11318 Value::Null,
11319 Value::Integer(1),
11320 Value::Integer(2),
11321 ],
11322 )
11323 .expect("integers"),
11324 ])
11325 .expect("one column");
11326 writer.append(&rows).expect("the only part");
11327 writer.finish().expect("commit");
11328 let reader = Reader::open(&path).expect("reopen from disk");
11329 let stripes = Stripes::new(reader);
11330 let column = stripes.column("a").expect("the file has that column");
11331 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
11332 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
11335 fs::remove_file(&path).expect("clean up");
11336 }
11337
11338 #[test]
11339 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
11340 let path = path("frequencies_for_the_planner");
11345 let mut writer =
11346 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11347 .expect("new file");
11348 let rows = Chunk::new(vec![
11349 Vector::from_values(
11350 LogicalType::Integer,
11351 &[
11352 Value::Integer(4),
11353 Value::Integer(4),
11354 Value::Integer(4),
11355 Value::Integer(9),
11356 Value::Integer(9),
11357 Value::Integer(1),
11358 ],
11359 )
11360 .expect("integers"),
11361 ])
11362 .expect("one column");
11363 writer.append(&rows).expect("the only part");
11364 writer.finish().expect("commit");
11365 let reader = Reader::open(&path).expect("reopen from disk");
11366 let common = Common::new(reader);
11367 assert_eq!(common.rows(), 6);
11368 let column = common.column("id").expect("the file has that column");
11369 assert_eq!(common.column("nothing"), None);
11370 assert_eq!(
11371 common.rows_with(column, &Bound::Int(4)),
11372 Stat::exact(3, Provenance::FrequencySynopsis)
11373 );
11374 assert_eq!(
11376 common.rows_with(column, &Bound::Int(7)),
11377 Stat::exact(0, Provenance::FrequencySynopsis)
11378 );
11379 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
11382 assert_eq!(common.remainder(column), None);
11385 fs::remove_file(&path).expect("clean up");
11386 }
11387
11388 #[test]
11389 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
11390 let path = path("string_frequencies_for_the_planner");
11391 let mut writer =
11392 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
11393 .expect("new file");
11394 let rows = Chunk::new(vec![
11395 Vector::from_values(
11396 LogicalType::Varchar,
11397 &[
11398 Value::Varchar(String::new()),
11399 Value::Varchar("alpha".into()),
11400 Value::Varchar(String::new()),
11401 Value::Varchar("beta".into()),
11402 Value::Varchar(String::new()),
11403 ],
11404 )
11405 .expect("strings"),
11406 ])
11407 .expect("one column");
11408 writer.append(&rows).expect("the only part");
11409 writer.finish().expect("commit");
11410
11411 let reader = Reader::open(&path).expect("reopen from disk");
11412 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
11413 let common = Common::new(reader.clone());
11414 let column = common.column("text").expect("the file has that column");
11415 assert_eq!(
11416 common.rows_with(column, &Bound::Bytes(Vec::new())),
11417 Stat::exact(3, Provenance::FrequencySynopsis)
11418 );
11419 assert_eq!(
11420 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
11421 Stat::exact(0, Provenance::FrequencySynopsis)
11422 );
11423 assert_eq!(
11424 reader.reads().dictionaries,
11425 0,
11426 "the bounded spellings answer without opening the dictionary index"
11427 );
11428 fs::remove_file(&path).expect("clean up");
11429 }
11430
11431 #[test]
11432 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
11433 let path = path("certified_host_groups");
11434 let mut writer =
11435 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
11436 .expect("new file");
11437 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
11438 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
11439 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
11440 values.push(Value::Varchar(String::new()));
11441 for part in values.chunks(512) {
11442 writer
11443 .append(
11444 &Chunk::new(vec![
11445 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
11446 ])
11447 .expect("one column"),
11448 )
11449 .expect("part written");
11450 }
11451 writer.finish().expect("commit");
11452 let reader = Reader::open(&path).expect("reopen");
11453 let summary = reader.table.host_groups.as_ref().expect("bounded host metadata");
11454 assert!(summary.omitted_max < 220);
11455 assert!(reader.host_groups(0, summary.omitted_max).expect("valid column").is_none());
11456 let groups = reader.host_groups(0, 220).expect("valid column").expect("certified");
11457 let example = groups.iter().find(|entry| entry.host == "example.com").expect("leader");
11458 assert_eq!(example.count, 220);
11459 assert_eq!(example.bytes_sum, 150 * 24 + 70 * 21);
11460 assert_eq!(example.minimum, "http://www.example.com/a");
11461 assert_eq!(reader.reads().dictionaries, 0, "the directory settles the question");
11462 fs::remove_file(&path).expect("clean up");
11463 }
11464
11465 fn bare_table(sections: Vec<Section>) -> Table {
11470 Table {
11471 name: "linked".to_owned(),
11472 fields: vec![Field::required("id", LogicalType::Integer)],
11473 stripes: Vec::new(),
11474 rows: 0,
11475 dictionaries: vec![None],
11476 dictionary_payloads: Vec::new(),
11477 distincts: vec![None],
11478 frequencies: vec![None],
11479 pair_frequencies: Vec::new(),
11480 frequency_texts: Vec::new(),
11481 host_groups: None,
11482 clustering: None,
11483 generation: 1,
11484 sections,
11485 }
11486 }
11487
11488 fn a_key_map_section() -> Section {
11489 Section {
11490 kind: *section::KEY_MAP,
11491 id: 1,
11492 generation: 3,
11493 extents: 1,
11494 extent_page: HEADER,
11495 extent_bytes: section::EXTENT_BYTES as u32,
11496 hash: 0x1234_5678_9abc_def0,
11497 flags: 0,
11498 header_bytes: 24,
11499 }
11500 }
11501
11502 #[test]
11503 fn a_section_table_round_trips_through_a_directory() {
11504 let mut later = a_key_map_section();
11505 later.kind = *b"RUDBZZ9\0";
11506 later.id = 2;
11507 let table = bare_table(vec![a_key_map_section(), later]);
11508 let directory = encode_directory(&table).expect("directory");
11509 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11510 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
11511 assert!(decoded.sections()[0].known());
11515 assert!(!decoded.sections()[1].known());
11516 }
11517
11518 #[test]
11519 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
11520 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11524 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
11525 let older = &directory[..directory.len() - block];
11526 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
11527 assert!(decoded.sections().is_empty());
11528 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
11529 assert_eq!(decoded.name(), "linked");
11530 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
11531 }
11532
11533 #[test]
11534 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
11535 let path = path("format_twenty_two");
11542 let mut writer =
11543 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11544 .expect("new file");
11545 let rows = Chunk::new(vec![
11546 Vector::from_values(
11547 LogicalType::Integer,
11548 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
11549 )
11550 .expect("integers"),
11551 ])
11552 .expect("one column");
11553 writer.append(&rows).expect("the only part");
11554 writer.finish().expect("commit");
11555
11556 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11557 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11558 drop(file);
11559
11560 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
11561 assert_eq!(reader.table().rows(), 3);
11562 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
11567
11568 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11571 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
11572 drop(file);
11573 let error = Reader::open(&path).expect_err("format 21 is not readable");
11574 assert!(error.to_string().contains("format 21"), "{error}");
11575
11576 fs::remove_file(&path).expect("clean up");
11577 }
11578
11579 #[test]
11580 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
11581 let mut past = a_key_map_section();
11586 past.extent_page = 1 << 30;
11587 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
11588 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
11589 assert!(error.to_string().contains("outside the file"), "{error}");
11590
11591 let mut inside_the_header = a_key_map_section();
11592 inside_the_header.extent_page = 8;
11593 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
11594 assert!(
11595 decode_directory(&directory, 1 << 20).is_err(),
11596 "a section may not overlap a header"
11597 );
11598 }
11599
11600 #[test]
11601 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
11602 let not_built = Section {
11606 kind: *section::FORWARD_LINK,
11607 id: 9,
11608 generation: 3,
11609 extents: 0,
11610 extent_page: 0,
11611 extent_bytes: 0,
11612 hash: 0,
11613 flags: 0,
11614 header_bytes: 0,
11615 };
11616 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
11617 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11618 assert_eq!(decoded.sections(), &[not_built]);
11619
11620 let mut incoherent = not_built;
11623 incoherent.extent_bytes = 28;
11624 incoherent.extent_page = HEADER;
11625 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
11626 assert!(decode_directory(&directory, 1 << 20).is_err());
11627 }
11628
11629 #[test]
11630 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
11631 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11632 let mut torn = directory.clone();
11633 let count_at = torn.len() - size_of::<u16>();
11634 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
11635 assert!(decode_directory(&torn, 1 << 20).is_err());
11638 }
11639
11640 fn linked_file(label: &str, rows: i32) -> PathBuf {
11642 let path = path(label);
11643 let mut writer =
11644 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11645 .expect("new file");
11646 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
11647 let chunk =
11648 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
11649 .expect("one column");
11650 writer.append(&chunk).expect("the only part");
11651 writer.finish().expect("commit");
11652 path
11653 }
11654
11655 fn a_key_map_payload() -> Vec<u8> {
11656 (0..512_u32).flat_map(u32::to_le_bytes).collect()
11659 }
11660
11661 #[test]
11662 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
11663 let path = linked_file("attach", 64);
11664 let payload = a_key_map_payload();
11665 let table = attach(
11666 &path,
11667 "items",
11668 &[section::Attachment {
11669 kind: *section::KEY_MAP,
11670 id: 0,
11671 flags: 2,
11672 header_bytes: 40,
11673 bytes: &payload,
11674 }],
11675 )
11676 .expect("attach a key map");
11677 assert_eq!(attached(&table).len(), 1);
11678
11679 let reader = Reader::open(&path).expect("reopen after the attach");
11680 let held = attached(reader.table());
11681 assert_eq!(held.len(), 1);
11682 assert_eq!(held[0].kind, *section::KEY_MAP);
11683 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
11684 assert_eq!(held[0].header_bytes, 40);
11685 assert_eq!(held[0].generation, 1);
11689 assert!(held[0].usable(reader.table().generation()));
11690 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
11691 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
11692
11693 fs::remove_file(&path).expect("clean up");
11694 }
11695
11696 #[test]
11697 fn attaching_a_section_answers_every_row_exactly_as_before() {
11698 let path = linked_file("attach_changes_nothing", 300);
11703 let before = Reader::open(&path).expect("open before");
11704 let rows = before.table().rows();
11705 let first = before.read(0, &[0]).expect("read before");
11706 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
11707 let layout = before.layout().columns_total();
11708 drop(before);
11709
11710 let payload = a_key_map_payload();
11711 attach(
11712 &path,
11713 "items",
11714 &[section::Attachment {
11715 kind: *section::KEY_MAP,
11716 id: 0,
11717 flags: 0,
11718 header_bytes: 0,
11719 bytes: &payload,
11720 }],
11721 )
11722 .expect("attach");
11723
11724 let after = Reader::open(&path).expect("open after");
11725 assert_eq!(after.table().rows(), rows);
11726 let read = after.read(0, &[0]).expect("read after");
11727 for (at, value) in values.iter().enumerate() {
11728 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
11729 }
11730 assert_eq!(
11731 after.layout().columns_total(),
11732 layout,
11733 "an attach appends and does not rewrite a column page"
11734 );
11735
11736 fs::remove_file(&path).expect("clean up");
11737 }
11738
11739 #[test]
11740 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
11741 let path = linked_file("attach_twice", 32);
11745 let one = a_key_map_payload();
11746 let two = vec![7_u8; 1024];
11747 let entry = |bytes| section::Attachment {
11748 kind: *section::KEY_MAP,
11749 id: 4,
11750 flags: 1,
11751 header_bytes: 0,
11752 bytes,
11753 };
11754 attach(&path, "items", &[entry(&one)]).expect("first build");
11755 attach(&path, "items", &[entry(&two)]).expect("rebuild");
11756
11757 let reader = Reader::open(&path).expect("reopen");
11758 let held = attached(reader.table());
11759 assert_eq!(held.len(), 1, "one map per column and not one per build");
11760 assert_eq!(reader.payload(held[0]).expect("payload"), two);
11761
11762 fs::remove_file(&path).expect("clean up");
11763 }
11764
11765 #[test]
11766 fn an_attach_carries_through_a_kind_it_does_not_know() {
11767 let path = linked_file("attach_unknown", 16);
11771 let payload = vec![3_u8; 96];
11772 attach(
11773 &path,
11774 "items",
11775 &[section::Attachment {
11776 kind: *b"RUDBZZ9\0",
11777 id: 1,
11778 flags: 0,
11779 header_bytes: 0,
11780 bytes: &payload,
11781 }],
11782 )
11783 .expect("a kind this build does not know still writes");
11784 let key_map = a_key_map_payload();
11785 attach(
11786 &path,
11787 "items",
11788 &[section::Attachment {
11789 kind: *section::KEY_MAP,
11790 id: 0,
11791 flags: 0,
11792 header_bytes: 0,
11793 bytes: &key_map,
11794 }],
11795 )
11796 .expect("attach beside it");
11797
11798 let reader = Reader::open(&path).expect("reopen");
11799 let held = attached(reader.table());
11800 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11801 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11802 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11803
11804 fs::remove_file(&path).expect("clean up");
11805 }
11806
11807 #[test]
11808 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11809 let path = linked_file("attach_not_built", 8);
11810 attach(
11811 &path,
11812 "items",
11813 &[section::Attachment {
11814 kind: *section::FORWARD_LINK,
11815 id: 2,
11816 flags: 0,
11817 header_bytes: 0,
11818 bytes: &[],
11819 }],
11820 )
11821 .expect("record a link that did not fit the budget");
11822
11823 let reader = Reader::open(&path).expect("reopen");
11824 let held = attached(reader.table());
11825 assert_eq!(held.len(), 1);
11826 assert_eq!(held[0].extents, 0);
11827 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11828 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11829 assert!(reader.payload(held[0]).expect("no payload").is_empty());
11830
11831 fs::remove_file(&path).expect("clean up");
11832 }
11833
11834 #[test]
11835 fn a_payload_past_one_extent_is_split_and_joined_back() {
11836 let path = linked_file("attach_two_extents", 8);
11840 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11841 attach(
11842 &path,
11843 "items",
11844 &[section::Attachment {
11845 kind: *section::KEY_MAP,
11846 id: 0,
11847 flags: 0,
11848 header_bytes: 0,
11849 bytes: &payload,
11850 }],
11851 )
11852 .expect("attach a payload past the bound");
11853
11854 let reader = Reader::open(&path).expect("reopen");
11855 let held = attached(reader.table());
11856 let extents = reader.extents(held[0]).expect("extent table");
11857 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
11858 assert_eq!(extents[0].length, section::MAX_EXTENT);
11859 assert_eq!(extents[1].length, 1);
11860 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
11861 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
11863 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
11864
11865 fs::remove_file(&path).expect("clean up");
11866 }
11867
11868 #[test]
11869 fn a_torn_extent_is_refused_rather_than_decoded() {
11870 let path = linked_file("attach_torn", 8);
11871 let payload = a_key_map_payload();
11872 attach(
11873 &path,
11874 "items",
11875 &[section::Attachment {
11876 kind: *section::KEY_MAP,
11877 id: 0,
11878 flags: 0,
11879 header_bytes: 0,
11880 bytes: &payload,
11881 }],
11882 )
11883 .expect("attach");
11884
11885 let reader = Reader::open(&path).expect("reopen");
11886 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
11887 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
11888 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
11889 drop(file);
11890
11891 let reader = Reader::open(&path).expect("the table still opens");
11892 let error = reader
11893 .payload(&reader.table().sections()[0])
11894 .expect_err("a corrupt payload is not handed out");
11895 assert!(error.to_string().contains("checksum"), "{error}");
11896 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
11899
11900 fs::remove_file(&path).expect("clean up");
11901 }
11902
11903 #[test]
11904 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
11905 let path = linked_file("attach_old_format", 8);
11908 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11909 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11910 drop(file);
11911
11912 let payload = a_key_map_payload();
11913 let error = attach(
11914 &path,
11915 "items",
11916 &[section::Attachment {
11917 kind: *section::KEY_MAP,
11918 id: 0,
11919 flags: 0,
11920 header_bytes: 0,
11921 bytes: &payload,
11922 }],
11923 )
11924 .expect_err("format 22 cannot gain a section");
11925 assert!(error.to_string().contains("format 22"), "{error}");
11926 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
11927
11928 fs::remove_file(&path).expect("clean up");
11929 }
11930
11931 #[test]
11932 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
11933 let path = linked_file("attach_bad_header", 8);
11934 let error = attach(
11935 &path,
11936 "items",
11937 &[section::Attachment {
11938 kind: *section::KEY_MAP,
11939 id: 0,
11940 flags: 0,
11941 header_bytes: 40,
11942 bytes: &[1, 2, 3],
11943 }],
11944 )
11945 .expect_err("a writer's bug stops at the write");
11946 assert!(error.to_string().contains("header is longer"), "{error}");
11947
11948 fs::remove_file(&path).expect("clean up");
11949 }
11950
11951 #[test]
11952 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
11953 let path = linked_file("attach_wrong_name", 8);
11954 let error = attach(&path, "orders", &[]).expect_err("no such table");
11955 assert!(error.to_string().contains("orders"), "{error}");
11956 fs::remove_file(&path).expect("clean up");
11957 }
11958
11959 #[test]
11960 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
11961 let path = path("frequency_prefix_for_the_planner");
11968 let mut writer =
11969 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11970 .expect("new file");
11971 let mut values = vec![Value::Integer(1); 10_000];
11972 for _ in 0..10 {
11973 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
11974 }
11975 for part in values.chunks(8_000) {
11978 let rows = Chunk::new(vec![
11979 Vector::from_values(LogicalType::Integer, part).expect("integers"),
11980 ])
11981 .expect("one column");
11982 writer.append(&rows).expect("a part");
11983 }
11984 writer.finish().expect("commit");
11985 let reader = Reader::open(&path).expect("reopen from disk");
11986 let prefix =
11987 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
11988 assert_eq!(prefix.entries.len(), 512);
11991 assert_eq!(prefix.omitted_max, 10);
11992 let common = Common::new(reader);
11993 assert_eq!(common.rows(), 16_000);
11994 let column = common.column("id").expect("the file has that column");
11995 assert_eq!(
11996 common.rows_with(column, &Bound::Int(1)),
11997 Stat::exact(10_000, Provenance::FrequencySynopsis)
11998 );
11999 assert_eq!(
12001 common.rows_with(column, &Bound::Int(1_100)),
12002 Stat::exact(10, Provenance::FrequencySynopsis)
12003 );
12004 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
12007 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
12010 let remainder = common.remainder(column).expect("the list is a prefix");
12014 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
12015 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
12016 fs::remove_file(&path).expect("clean up");
12017 }
12018
12019 #[test]
12021 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
12022 let path = path("empty");
12023 Writer::empty(&path, &[]).expect("a file with nothing in it");
12024 let catalog = Catalog::open(&path).expect("the empty file opens");
12025 assert_eq!(catalog.len(), 0);
12026 assert!(catalog.is_empty());
12027 assert_eq!(catalog.names().count(), 0);
12028 let mut writer =
12031 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12032 .expect("a table goes into the empty file");
12033 writer.append(&sample_ids()).expect("rows");
12034 writer.finish().expect("commit");
12035 let catalog = Catalog::open(&path).expect("the file opens again");
12036 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12037 fs::remove_file(&path).expect("clean up");
12038 }
12039
12040 #[test]
12050 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
12051 let path = path("empty-name");
12052 let field = || vec![Field::required("id", LogicalType::Integer)];
12053 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
12054 let catalog = Catalog::open(&path).expect("the file opens");
12055 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
12056
12057 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
12058 writer.append(&sample_ids()).expect("rows");
12059 writer.finish().expect("commit");
12060 let catalog = Catalog::open(&path).expect("the file opens again");
12061 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12063 let held = catalog.rows().collect::<Vec<_>>();
12064 assert_eq!(held.len(), 1);
12065 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
12066
12067 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
12069 assert!(error.to_string().contains("same name"), "{error}");
12070 fs::remove_file(&path).expect("clean up");
12071 }
12072
12073 fn sample_view(name: &str) -> ViewEntry {
12075 ViewEntry {
12076 name: name.to_string(),
12077 sql: "SELECT id FROM items WHERE id > 0".to_string(),
12078 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
12079 aliases: vec!["n".to_string()],
12080 columns: vec![Field::new("n", LogicalType::Integer)],
12081 }
12082 }
12083
12084 #[test]
12085 fn a_view_written_into_the_catalog_comes_back_whole() {
12086 let path = path("views");
12087 let mut writer =
12088 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12089 .expect("new file");
12090 writer.append(&sample_ids()).expect("rows");
12091 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12092 let catalog = Catalog::open(&path).expect("reopen");
12093 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
12094 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12097 fs::remove_file(&path).expect("clean up");
12098 }
12099
12100 #[test]
12102 fn appending_a_table_carries_the_views_forward() {
12103 let path = path("viewscarry");
12104 let mut writer =
12105 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12106 .expect("new file");
12107 writer.append(&sample_ids()).expect("rows");
12108 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12109 let mut writer =
12110 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
12111 .expect("a second table");
12112 writer.append(&sample_ids()).expect("rows");
12113 writer.finish().expect("commit");
12114 let catalog = Catalog::open(&path).expect("reopen");
12115 assert_eq!(catalog.views().count(), 1);
12116 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
12117 fs::remove_file(&path).expect("clean up");
12118 }
12119
12120 #[test]
12122 fn restating_the_views_leaves_every_table_where_it_was() {
12123 let path = path("restate");
12124 let mut writer =
12125 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12126 .expect("new file");
12127 writer.append(&sample_ids()).expect("rows");
12128 writer.finish().expect("commit");
12129 let before = fs::metadata(&path).expect("the file is there").len();
12130 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
12131 let catalog = Catalog::open(&path).expect("reopen");
12132 assert_eq!(catalog.views().count(), 2);
12133 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12134 let after = fs::metadata(&path).expect("the file is there").len();
12137 assert!(after > before, "a generation was written");
12138 assert!(after - before < before, "the table was not written again");
12139 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
12142 assert_eq!(reader.table().rows, 3);
12143 Writer::restate(&path, &[]).expect("no views at all");
12146 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
12147 fs::remove_file(&path).expect("clean up");
12148 }
12149
12150 #[test]
12152 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
12153 let bytes = encode_catalog(
12154 &[Entry {
12155 name: "items".to_string(),
12156 fields: vec![Field::required("id", LogicalType::Integer)],
12157 rows: 1,
12158 directory: Page { offset: HEADER, length: 8, hash: 0 },
12159 nonzero: vec![None],
12160 aggregates: vec![None],
12161 distincts: vec![None],
12162 extremes: vec![None],
12163 frequencies: vec![None],
12164 }],
12165 &[sample_view("items")],
12166 )
12167 .expect("it encodes, because encoding does not look");
12168 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
12169 assert!(error.to_string().contains("same name"), "{error}");
12170 }
12171
12172 #[test]
12173 fn committed_file_reopens_and_reads_only_requested_columns() {
12174 let path = path("reopen");
12175 let mut writer = Writer::create(
12176 &path,
12177 "items",
12178 vec![
12179 Field::required("id", LogicalType::Integer),
12180 Field::new("text", LogicalType::Varchar),
12181 ],
12182 )
12183 .expect("new file");
12184 writer.append(&sample()).expect("first part");
12185 writer.append(&sample()).expect("second part");
12186 writer.finish().expect("commit");
12187 let reader = Reader::open(&path).expect("reopen from disk");
12188 assert_eq!(reader.table().rows(), 6);
12189 assert_eq!(reader.table().stripes().len(), 1);
12192 assert_eq!(reader.parts(), 2);
12193 assert_eq!(reader.part_rows(0), 3);
12194 assert_eq!(reader.part_rows(1), 3);
12195 let text = reader.read(1, &[1]).expect("only text page");
12196 assert_eq!(text.width(), 1);
12197 assert_eq!(text.value_at(1, 0), Value::Null);
12198 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12199 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
12200 assert_eq!(sparse.width(), 1);
12201 assert_eq!(sparse.value_at(1, 0), Value::Null);
12202 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12203 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
12204 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
12205 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
12206 let count = reader.read(0, &[]).expect("no page is needed for count");
12207 assert_eq!(count.len(), 3);
12208 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
12209 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
12210 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
12211 assert_eq!(
12212 integers,
12213 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
12214 );
12215 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
12216 assert_eq!(strings.len(), 3);
12217 assert!(strings.contains(&(Value::Null, 2)));
12218 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
12219 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
12220 fs::remove_file(path).expect("remove scratch file");
12221 }
12222
12223 #[test]
12231 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
12232 let path = path("interleaved-runs");
12233 let mut writer =
12234 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
12235 .expect("new file");
12236 for morsel in [2_u64, 0, 3, 1] {
12237 let parts = (0..4_u64)
12238 .map(|chunk| {
12239 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
12240 let values =
12241 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
12242 let column =
12243 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
12244 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
12245 })
12246 .collect::<Vec<_>>();
12247 writer.append_stripe(parts).expect("a stripe");
12248 }
12249 writer.finish().expect("commit");
12250
12251 let reader = Reader::open(&path).expect("valid directory");
12252 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
12253 assert_eq!(reader.table().rows(), 128);
12254 for part in 0..16_usize {
12255 let read = reader.read(part, &[0]).expect("a part back");
12256 for row in 0..8_usize {
12257 let want = i64::try_from(part * 8 + row).expect("small");
12258 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
12259 }
12260 }
12261 fs::remove_file(path).expect("remove scratch file");
12262 }
12263
12264 #[test]
12267 fn runs_that_overlap_each_other_are_refused_at_commit() {
12268 let path = path("overlapping-runs");
12269 let mut writer =
12270 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
12271 .expect("new file");
12272 let one = |order: (u64, u64)| {
12273 let column =
12274 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
12275 (order, Chunk::new(vec![column]).expect("one column"))
12276 };
12277 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
12280 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
12281 let error = writer.finish().expect_err("the runs overlap");
12282 assert!(error.message().contains("source order"), "{error}");
12283 fs::remove_file(path).expect("remove scratch file");
12284 }
12285
12286 #[test]
12289 fn a_run_longer_than_a_stripe_is_refused() {
12290 let path = path("overlong-run");
12291 let mut writer =
12292 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
12293 .expect("new file");
12294 let parts = (0..=STRIPE_PARTS)
12295 .map(|at| {
12296 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
12297 .expect("a column");
12298 let chunk = Chunk::new(vec![column]).expect("one column");
12299 ((0, u64::try_from(at).expect("small")), chunk)
12300 })
12301 .collect::<Vec<_>>();
12302 let error = writer.append_stripe(parts).expect_err("one part too many");
12303 assert!(error.message().contains("more parts than it holds"), "{error}");
12304 fs::remove_file(path).expect("remove scratch file");
12305 }
12306
12307 #[test]
12313 fn parts_past_the_stripe_bound_start_a_new_stripe() {
12314 let path = path("stripe-bound");
12315 let mut writer = Writer::create(
12316 &path,
12317 "items",
12318 vec![
12319 Field::required("id", LogicalType::Integer),
12320 Field::new("text", LogicalType::Varchar),
12321 ],
12322 )
12323 .expect("new file");
12324 let parts = STRIPE_PARTS * 2 + 3;
12325 for part in 0..parts {
12326 let id = part as i32;
12327 let chunk = Chunk::new(vec![
12328 Vector::from_values(
12329 LogicalType::Integer,
12330 &[Value::Integer(id), Value::Integer(-id)],
12331 )
12332 .expect("integers"),
12333 Vector::from_values(
12334 LogicalType::Varchar,
12335 &[Value::Varchar(format!("value {part}")), Value::Null],
12336 )
12337 .expect("strings"),
12338 ])
12339 .expect("matching rows");
12340 writer.append(&chunk).expect("one part");
12341 }
12342 writer.finish().expect("commit");
12343
12344 let reader = Reader::open(&path).expect("reopen from disk");
12345 assert_eq!(reader.parts(), parts);
12346 assert_eq!(reader.table().rows(), parts * 2);
12347 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
12348 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
12349 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
12350 assert_eq!(reader.table().stripes()[2].parts(), 3);
12351 for part in (0..parts).rev() {
12354 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
12355 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
12356 for chunk in [&dense, &sparse] {
12357 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
12358 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12359 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12360 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
12361 assert_eq!(chunk.value_at(1, 1), Value::Null);
12362 }
12363 }
12364 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
12367 assert!(reader.skips(0, &above), "the first stripe stops at 63");
12368 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
12369 fs::remove_file(path).expect("remove scratch file");
12370 }
12371
12372 fn scattered(n: i64) -> i64 {
12374 n.wrapping_mul(-7_046_029_254_386_353_131)
12375 }
12376
12377 #[test]
12383 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
12384 let path = path("sieve-skip");
12385 let mut writer =
12386 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12387 .expect("new file");
12388 let parts = STRIPE_PARTS + 3;
12389 let per_part = 128;
12393 for part in 0..parts {
12394 let held: Vec<Value> = (0..per_part)
12395 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
12396 .collect();
12397 let chunk =
12398 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12399 .expect("one column");
12400 writer.append(&chunk).expect("one part");
12401 }
12402 writer.finish().expect("commit");
12403
12404 let reader = Reader::open(&path).expect("reopen from disk");
12405 let probe = |value: i64| Probe {
12406 column: 0,
12407 op: Op::Equal,
12408 value: Bound::Int(i128::from(scattered(value))),
12409 };
12410 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
12411 let tests = [probe(wanted)];
12412 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
12413 let home = wanted as usize / per_part;
12414 assert!(kept.contains(&home), "the part holding {wanted} is read");
12415 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
12419 }
12420 let absent = [probe((parts * per_part) as i64 + 1)];
12421 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
12422 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
12423 let tests = [probe(0)];
12426 assert!(
12427 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
12428 "the bounds rule out no stripe at all"
12429 );
12430 fs::remove_file(path).expect("remove scratch file");
12431 }
12432
12433 #[test]
12439 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
12440 let path = path("part-range-skip");
12441 let mut writer =
12442 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12443 .expect("new file");
12444 let parts = STRIPE_PARTS + 3;
12445 let per_part = 128;
12446 for part in 0..parts {
12447 let held: Vec<Value> = (0..per_part)
12451 .map(|row| {
12452 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12453 })
12454 .collect();
12455 let chunk =
12456 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12457 .expect("one column");
12458 writer.append(&chunk).expect("one part");
12459 }
12460 writer.finish().expect("commit");
12461
12462 let reader = Reader::open(&path).expect("reopen from disk");
12463 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12464 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
12465 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
12466 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
12468 fs::remove_file(path).expect("remove scratch file");
12469 }
12470
12471 #[test]
12475 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
12476 let path = path("part-range-certain");
12477 let mut writer =
12478 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12479 .expect("new file");
12480 let parts = STRIPE_PARTS + 3;
12481 let per_part = 128;
12482 for part in 0..parts {
12483 let held: Vec<Value> = (0..per_part)
12484 .map(|row| {
12485 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12486 })
12487 .collect();
12488 let chunk =
12489 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12490 .expect("one column");
12491 writer.append(&chunk).expect("one part");
12492 }
12493 writer.finish().expect("commit");
12494
12495 let reader = Reader::open(&path).expect("reopen from disk");
12496 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12497 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
12498 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
12499 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
12502 fs::remove_file(path).expect("remove scratch file");
12503 }
12504
12505 #[test]
12508 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
12509 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
12510 let path = path("part-range-page");
12511 let mut writer =
12512 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12513 .expect("new file");
12514 for part in 0..parts {
12515 let held: Vec<Value> = (0..128)
12516 .map(|row| {
12517 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
12518 })
12519 .collect();
12520 let chunk = Chunk::new(vec![
12521 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
12522 ])
12523 .expect("one column");
12524 writer.append(&chunk).expect("one part");
12525 }
12526 writer.finish().expect("commit");
12527 let reader = Reader::open(&path).expect("reopen from disk");
12528 let bytes = reader.layout().columns[0].part_ranges;
12529 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
12530 fs::remove_file(path).expect("remove scratch file");
12531 }
12532 }
12533
12534 #[test]
12537 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
12538 let long = vec![b'a'; PART_BOUND_BYTES * 2];
12539 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
12540 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
12541 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
12542 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
12543 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
12544 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
12545 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
12546 }
12547
12548 #[test]
12551 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
12552 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
12553 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
12554 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
12555 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
12556 }
12557
12558 #[test]
12570 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
12571 let parts = 4;
12572 let per_part = 1024;
12573 let rows = parts * per_part;
12574 let written = |name: &str, keys: &[i64]| {
12575 let path = path(name);
12576 let fields = vec![Field::required("key", LogicalType::BigInt)];
12577 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
12578 for part in 0..parts {
12579 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
12580 .iter()
12581 .map(|key| Value::BigInt(*key))
12582 .collect();
12583 let chunk = Chunk::new(vec![
12584 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
12585 ])
12586 .expect("one column");
12587 writer.append(&chunk).expect("one part");
12588 }
12589 writer.finish().expect("commit");
12590 path
12591 };
12592 let climbing = |step: &dyn Fn(usize) -> i64| {
12595 let mut key = 0;
12596 (0..rows)
12597 .map(|row| {
12598 key += step(row);
12599 key
12600 })
12601 .collect::<Vec<i64>>()
12602 };
12603 let ascending = climbing(&|row| (row % 3) as i64);
12604 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
12608 let near_path = written("stored-near", &ascending);
12609 let far_path = written("stored-far", &sparse);
12610
12611 let one = Reader::open(&near_path).expect("reopen from disk");
12612 let other = Reader::open(&far_path).expect("reopen from disk");
12613 let near = one.stored(0).expect("the column is stored");
12614 let far = other.stored(0).expect("the column is stored");
12615 assert_eq!(near.len(), parts, "one row per part");
12616 assert_eq!(far.len(), parts);
12617 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
12620 assert_eq!(total(&near), one.layout().columns[0].pages);
12621 assert_eq!(total(&far), other.layout().columns[0].pages);
12622 assert!(
12623 total(&near) * 2 < total(&far),
12624 "the sparse keys cost more, {} against {}",
12625 total(&far),
12626 total(&near)
12627 );
12628 for (at, part) in near.iter().enumerate() {
12630 assert_eq!(part.part, at);
12631 assert_eq!(part.row, at * per_part);
12632 assert_eq!(part.rows, per_part);
12633 let held = &ascending[at * per_part..(at + 1) * per_part];
12634 assert_eq!(part.low, Some(Value::BigInt(held[0])));
12635 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
12636 assert_eq!(part.nulls, Some(0));
12637 }
12638 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
12641 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
12642 assert_ne!(near[0].encoding, far[0].encoding);
12643 fs::remove_file(near_path).expect("remove scratch file");
12644 fs::remove_file(far_path).expect("remove scratch file");
12645 }
12646
12647 #[test]
12657 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
12658 let path = path("sieve-pays");
12659 let fields = vec![
12660 Field::required("spread", LogicalType::BigInt),
12661 Field::required("repeated", LogicalType::BigInt),
12662 ];
12663 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
12664 let parts = 3;
12665 let per_part = 1024;
12666 for part in 0..parts {
12667 let base = (part * per_part) as i64;
12668 let spread: Vec<Value> =
12669 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
12670 let repeated: Vec<Value> =
12671 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
12672 let chunk = Chunk::new(vec![
12673 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
12674 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
12675 ])
12676 .expect("two columns");
12677 writer.append(&chunk).expect("one part");
12678 }
12679 writer.finish().expect("commit");
12680
12681 let reader = Reader::open(&path).expect("reopen from disk");
12682 let layout = reader.layout();
12683 let spread = &layout.columns[0];
12684 let repeated = &layout.columns[1];
12685 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
12686 assert_eq!(
12687 repeated.sieves, 0,
12688 "a column whose filter costs more than its parts keeps none"
12689 );
12690 for column in &layout.columns {
12693 assert!(
12694 column.sieves < column.pages,
12695 "{} spends {} on sieves over {} of data",
12696 column.name,
12697 column.sieves,
12698 column.pages
12699 );
12700 }
12701 let absent = [Probe {
12703 column: 0,
12704 op: Op::Equal,
12705 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
12706 }];
12707 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
12708 fs::remove_file(path).expect("remove scratch file");
12709 }
12710
12711 #[test]
12717 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
12718 let path = path("sieve-damaged");
12719 let mut writer =
12720 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12721 .expect("new file");
12722 let rows = 128;
12723 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
12724 let chunk =
12725 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12726 .expect("one column");
12727 writer.append(&chunk).expect("one part");
12728 writer.finish().expect("commit");
12729
12730 let page = Reader::open(&path).expect("reopen").table.stripes[0]
12731 .sieves
12732 .get(0)
12733 .expect("a sieve page");
12734 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
12735 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
12736 file.write_all(&[0xff]).expect("damage one byte");
12737 drop(file);
12738
12739 let reader = Reader::open(&path).expect("reopen the damaged file");
12740 let absent =
12741 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
12742 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
12743 assert_eq!(
12744 reader.read(0, &[0]).expect("the rows are untouched").len(),
12745 usize::try_from(rows).expect("a small count")
12746 );
12747 fs::remove_file(path).expect("remove scratch file");
12748 }
12749
12750 #[test]
12761 fn workers_that_want_the_same_stripe_read_it_once() {
12762 let path = path("single-flight");
12763 let mut writer =
12764 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12765 .expect("new file");
12766 for part in 0..STRIPE_PARTS {
12767 let id = part as i32;
12768 let chunk = Chunk::new(vec![
12769 Vector::from_values(
12770 LogicalType::Integer,
12771 &[Value::Integer(id), Value::Integer(-id)],
12772 )
12773 .expect("integers"),
12774 ])
12775 .expect("matching rows");
12776 writer.append(&chunk).expect("one part");
12777 }
12778 writer.finish().expect("commit");
12779
12780 let reader = Reader::open(&path).expect("reopen from disk");
12781 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
12782 let barrier = std::sync::Barrier::new(8);
12783 std::thread::scope(|scope| {
12784 for worker in 0..8 {
12785 let reader = &reader;
12786 let barrier = &barrier;
12787 scope.spawn(move || {
12788 barrier.wait();
12789 for part in (worker..STRIPE_PARTS).step_by(8) {
12790 let chunk = reader.read(part, &[0]).expect("a whole page read");
12791 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12792 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12793 }
12794 });
12795 }
12796 });
12797 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
12798 fs::remove_file(path).expect("remove scratch file");
12799 }
12800
12801 #[test]
12814 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
12815 let opened = |label: &str, rows_per_part: i32| {
12816 let path = path(label);
12817 let mut writer =
12818 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12819 .expect("new file");
12820 for part in 0..STRIPE_PARTS * 3 {
12821 let values = (0..rows_per_part)
12825 .map(|row| {
12826 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
12827 })
12828 .collect::<Vec<_>>();
12829 let chunk = Chunk::new(vec![
12830 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
12831 ])
12832 .expect("matching rows");
12833 writer.append(&chunk).expect("one part");
12834 }
12835 writer.finish().expect("commit");
12836 let reader = Reader::open(&path).expect("reopen from disk");
12837 let size = fs::metadata(&path).expect("the file is there").len();
12838 let out = (reader.reads(), reader.table().stripes().len(), size);
12839 fs::remove_file(path).expect("remove scratch file");
12840 out
12841 };
12842
12843 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
12844 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
12845 assert_eq!(
12846 thin_stripes, fat_stripes,
12847 "the same stripe count is what makes this a fair ask"
12848 );
12849 assert!(
12850 fat_size > thin_size * 50,
12851 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
12852 );
12853
12854 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
12855 assert_eq!(thin.pages, 0, "opening read a page");
12856 assert_eq!(fat.pages, 0, "opening read a page");
12857 assert_eq!(thin.indexes, 0, "opening read an index");
12858 assert_eq!(fat.indexes, 0, "opening read an index");
12859 assert!(
12862 fat.opening.bytes < thin.opening.bytes * 2,
12863 "opening the thin file read {} bytes and the fat one read {}",
12864 thin.opening.bytes,
12865 fat.opening.bytes
12866 );
12867 }
12868
12869 #[test]
12877 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
12878 let path = path("open-twice");
12879 let mut writer =
12880 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12881 .expect("new file");
12882 for part in 0..STRIPE_PARTS * 3 {
12883 let chunk = Chunk::new(vec![
12884 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12885 .expect("integers"),
12886 ])
12887 .expect("matching rows");
12888 writer.append(&chunk).expect("one part");
12889 }
12890 writer.finish().expect("commit");
12891
12892 let first = Reader::open(&path).expect("open");
12893 for part in 0..first.parts() {
12896 first.read(part, &[0]).expect("a part");
12897 }
12898 assert!(first.reads().pages > 0, "the scan has to have read something");
12899 let second = Reader::open(&path).expect("open again");
12900
12901 assert_eq!(first.reads().opening, second.reads().opening);
12902 assert_eq!(
12903 second.reads().pages,
12904 0,
12905 "the second open read a page off the back of the first"
12906 );
12907 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
12908 fs::remove_file(path).expect("remove scratch file");
12909 }
12910
12911 #[test]
12919 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
12920 let path = path("index-cache");
12921 let mut writer =
12922 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12923 .expect("new file");
12924 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12925 for part in 0..parts {
12926 let id = part as i32;
12927 let chunk = Chunk::new(vec![
12928 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
12929 ])
12930 .expect("matching rows");
12931 writer.append(&chunk).expect("one part");
12932 }
12933 writer.finish().expect("commit");
12934
12935 let reader = Reader::open(&path).expect("reopen from disk");
12936 let stripes = reader.table().stripes().len();
12937 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
12938 for _ in 0..2 {
12940 for part in 0..parts {
12941 let chunk = reader.read(part, &[0]).expect("a part");
12942 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12943 }
12944 }
12945 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
12946 assert!(
12947 reader.pages.load(Atomic::Relaxed) > stripes,
12948 "the pages are the ones that get read again, which is what makes the index count mean \
12949 something"
12950 );
12951 fs::remove_file(path).expect("remove scratch file");
12952 }
12953
12954 #[test]
12961 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
12962 let path = path("page-pool");
12963 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
12964 let fields = || vec![Field::required("id", LogicalType::Integer)];
12965 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
12966 for table in ["a", "b"] {
12967 if table == "b" {
12968 writer = writer.next("b".to_string(), fields()).expect("a second table");
12969 }
12970 for part in 0..parts {
12971 let chunk = Chunk::new(vec![
12972 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
12973 .expect("integers"),
12974 ])
12975 .expect("matching rows");
12976 writer.append(&chunk).expect("one part");
12977 }
12978 }
12979 writer.finish().expect("commit");
12980
12981 let pool = PagePool::new(usize::MAX);
12982 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
12983 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
12984 let stripes = a.table().stripes().len();
12985 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
12986 let scan = |reader: &Reader| {
12987 for part in 0..parts {
12988 let chunk = reader.read(part, &[0]).expect("a part");
12989 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12990 }
12991 };
12992 scan(&a);
12993 scan(&a);
12994 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
12995 let one = pool.bytes();
12996 assert!(one > 0, "the pool counts what the reader holds");
12997
12998 pool.budget.store(one, Atomic::Relaxed);
13000 scan(&b);
13001 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
13002 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
13003 let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
13004 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
13005
13006 drop((a, b, catalog));
13008 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
13009 scan(&c);
13010 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
13011 fs::remove_file(path).expect("remove scratch file");
13012 }
13013
13014 #[test]
13023 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
13024 let workers = CACHED_STRIPES_PER_COLUMN + 4;
13025 let path = path("stripe-per-worker");
13026 let mut writer =
13027 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13028 .expect("new file");
13029 for part in 0..STRIPE_PARTS * workers {
13030 let chunk = Chunk::new(vec![
13031 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13032 .expect("integers"),
13033 ])
13034 .expect("matching rows");
13035 writer.append(&chunk).expect("one part");
13036 }
13037 writer.finish().expect("commit");
13038
13039 let read = |told: bool| {
13040 let reader = Reader::open(&path).expect("reopen from disk");
13041 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
13042 if told {
13043 reader.keep_stripes(workers);
13044 }
13045 let barrier = std::sync::Barrier::new(workers);
13046 std::thread::scope(|scope| {
13047 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
13048 let reader = &reader;
13049 let barrier = &barrier;
13050 scope.spawn(move || {
13051 for part in run {
13052 barrier.wait();
13053 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
13054 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13055 }
13056 assert!(worker < workers);
13057 });
13058 }
13059 });
13060 reader.pages.load(Atomic::Relaxed)
13061 };
13062
13063 assert_eq!(read(true), workers, "one page read per stripe and no more");
13064 assert!(read(false) > workers, "a cache that small is read again on every part");
13065 fs::remove_file(path).expect("remove scratch file");
13066 }
13067
13068 #[test]
13073 fn a_damaged_index_page_is_an_error() {
13074 let path = path("damaged-index");
13075 let mut writer =
13076 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13077 .expect("new file");
13078 writer.append(&sample_ids()).expect("first part");
13079 writer.append(&sample_ids()).expect("second part");
13080 writer.finish().expect("commit");
13081
13082 let reader = Reader::open(&path).expect("valid directory");
13083 let index = reader.table.stripes[0].index;
13084 let mut byte = [0; 1];
13085 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
13086 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
13087 file.seek(SeekFrom::Start(index.offset)).expect("index start");
13088 file.write_all(&[!byte[0]]).expect("damage the first part length");
13089 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
13090 assert!(error.message().contains("index page section checksum differs"), "{error}");
13091 fs::remove_file(path).expect("remove scratch file");
13092 }
13093
13094 #[test]
13101 fn every_integer_width_round_trips_through_a_page() {
13102 let path = path("integer-widths");
13103 let columns = [
13104 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
13105 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
13106 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
13107 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
13108 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
13109 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
13110 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
13111 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
13112 ];
13113 let fields = columns
13114 .iter()
13115 .enumerate()
13116 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13117 .collect::<Vec<_>>();
13118 let vectors = columns
13119 .iter()
13120 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13121 .collect::<Vec<_>>();
13122 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
13123 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13124 writer.finish().expect("commit");
13125
13126 let reader = Reader::open(&path).expect("reopen from disk");
13127 let wanted = (0..columns.len()).collect::<Vec<_>>();
13128 let read = reader.read(0, &wanted).expect("every column");
13129 assert_eq!(read.len(), 2);
13130 for (at, (ty, values)) in columns.iter().enumerate() {
13132 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13133 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13134 }
13135 fs::remove_file(path).expect("remove scratch file");
13136 }
13137
13138 #[test]
13149 fn every_other_type_the_format_knows_round_trips_through_a_page() {
13150 let path = path("other-types");
13151 let columns = [
13152 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
13153 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
13154 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
13155 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
13156 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
13157 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
13158 (
13159 LogicalType::TimestampTz,
13160 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
13161 ),
13162 (
13163 LogicalType::Interval,
13164 vec![
13165 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
13166 Value::Interval { months: 13, days: -1, micros: 1 },
13167 ],
13168 ),
13169 (
13170 LogicalType::Blob,
13171 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
13172 ),
13173 ];
13174 let fields = columns
13175 .iter()
13176 .enumerate()
13177 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13178 .collect::<Vec<_>>();
13179 let vectors = columns
13180 .iter()
13181 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13182 .collect::<Vec<_>>();
13183 let mut writer = Writer::create(&path, "others", fields).expect("new file");
13184 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13185 writer.finish().expect("commit");
13186
13187 let reader = Reader::open(&path).expect("reopen from disk");
13188 let wanted = (0..columns.len()).collect::<Vec<_>>();
13189 let read = reader.read(0, &wanted).expect("every column");
13190 assert_eq!(read.len(), 2);
13191 for (at, (ty, values)) in columns.iter().enumerate() {
13192 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13193 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13194 }
13195 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
13198 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
13199
13200 fs::remove_file(path).expect("remove scratch file");
13201 }
13202
13203 #[test]
13209 fn a_nan_survives_being_written_down() {
13210 let path = path("nan");
13211 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
13212 .expect("a NaN vector");
13213 let mut writer =
13214 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
13215 .expect("new file");
13216 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
13217 writer.finish().expect("commit");
13218 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
13219 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
13220 assert!(back.is_nan(), "a NaN came back as {back}");
13221 fs::remove_file(path).expect("remove scratch file");
13222 }
13223
13224 #[test]
13231 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
13232 let path = path("uuid-and-bit");
13233 let uuids = vec![0_i128, i128::MIN, -1];
13234 let mut bits = StringColumn::new();
13235 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
13236 bits.push_bytes(value);
13237 }
13238 let expected = bits.clone();
13239 let fields =
13240 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
13241 let vectors = vec![
13242 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
13243 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
13244 ];
13245 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
13246 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13247 writer.finish().expect("commit");
13248
13249 let reader = Reader::open(&path).expect("reopen from disk");
13250 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
13251 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
13252 panic!("a uuid column is the 128 bit lane")
13253 };
13254 assert_eq!(back.as_slice(), uuids.as_slice());
13255 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
13256 panic!("a bit column is bytes")
13257 };
13258 for row in 0..expected.len() {
13259 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
13260 }
13261 fs::remove_file(path).expect("remove scratch file");
13262 }
13263
13264 #[test]
13267 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
13268 let mut rows: Vec<Option<u64>> = Vec::new();
13269 let mut state = 0x2545_f491_4f6c_dd1d_u64;
13270 for index in 0..400_000_u64 {
13271 state ^= state << 13;
13272 state ^= state >> 7;
13273 state ^= state << 17;
13274 let times = 1 + (state % 7) as usize;
13275 let bits = match state % 11 {
13276 0 => None,
13277 1..=3 => Some(state % 16),
13278 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
13279 };
13280 rows.extend(std::iter::repeat_n(bits, times));
13281 }
13282 let mut by_row = Candidates::default();
13283 for &bits in &rows {
13284 by_row.add(bits, 1);
13285 }
13286 let mut by_run = Candidates::default();
13287 let mut run = Run::default();
13288 let mut runs = 0_usize;
13289 for &bits in &rows {
13290 if let Some((bits, times)) = run.push(bits) {
13291 by_run.add(bits, times);
13292 runs += 1;
13293 }
13294 }
13295 if let Some((bits, times)) = run.take() {
13296 by_run.add(bits, times);
13297 }
13298 assert!(runs < rows.len() / 2, "the rows came in runs");
13299 assert!(by_row.decrements > 0, "the table filled and turned values away");
13300 assert_eq!(by_run.counts, by_row.counts);
13301 assert_eq!(by_run.nulls, by_row.nulls);
13302 assert_eq!(by_run.decrements, by_row.decrements);
13303 }
13304
13305 #[test]
13306 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
13307 let path = path("frequency-ordinals");
13308 let mut writer =
13309 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
13310 .expect("new file");
13311 let mut values = Vec::new();
13312 for leader in 0..10_i64 {
13313 values.extend(std::iter::repeat_n(leader, 100));
13314 }
13315 values.extend(1_000_i64..41_000);
13316 for part in values.chunks(1_024) {
13317 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
13318 .expect("big integers");
13319 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
13320 }
13321 writer.finish().expect("commit");
13322
13323 let reader = Reader::open(&path).expect("reopen from disk");
13324 let occurrences =
13325 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
13326 assert!(occurrences.omitted_max < 100);
13327 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
13328 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
13329 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
13330 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
13331 assert_eq!(
13332 &occurrences.anchor_indices[..1_000]
13333 .iter()
13334 .map(|&entry| occurrences.anchors[entry as usize].clone())
13335 .collect::<Vec<_>>(),
13336 &(0_i64..10)
13337 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
13338 .collect::<Vec<_>>()
13339 );
13340 fs::remove_file(path).expect("remove scratch file");
13341 }
13342
13343 #[test]
13344 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
13345 let path = path("frequency-bits");
13350 let mut writer = Writer::create(
13351 &path,
13352 "items",
13353 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
13354 )
13355 .expect("new file");
13356 let mut rows = Vec::new();
13357 let mut leaders = Vec::new();
13358 for leader in 0..10_u64 {
13359 let count = 300 - leader * 10;
13360 let (unsigned, signed) = if leader == 0 {
13361 (Value::Null, Value::Null)
13362 } else {
13363 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
13364 };
13365 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
13366 leaders.push(((unsigned, count), (signed, count)));
13367 }
13368 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
13369 for part in rows.chunks(1_024) {
13370 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
13371 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
13372 let chunk = Chunk::new(vec![
13373 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
13374 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
13375 ])
13376 .expect("matching columns");
13377 writer.append(&chunk).expect("rows");
13378 }
13379 writer.finish().expect("commit");
13380
13381 let reader = Reader::open(&path).expect("reopen from disk");
13382 for column in 0..2 {
13383 let prefix =
13384 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
13385 let wanted = leaders
13386 .iter()
13387 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
13388 .cloned()
13389 .collect::<Vec<_>>();
13390 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
13391 assert!(prefix.omitted_max < 210, "column {column}");
13392 assert_eq!(
13393 reader.distinct_values(column).expect("valid metadata"),
13394 Some(9 + 40_000),
13395 "column {column}"
13396 );
13397 }
13398 fs::remove_file(path).expect("remove scratch file");
13399 }
13400
13401 #[test]
13402 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
13403 let path = path("quick-nonzero");
13404 let mut writer = Writer::create(
13405 &path,
13406 "items",
13407 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
13408 )
13409 .expect("create");
13410 for ids in [
13411 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
13412 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
13413 ] {
13414 let labels = vec![Value::Varchar("same".into()); ids.len()];
13415 writer
13416 .append(
13417 &Chunk::new(vec![
13418 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
13419 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
13420 ])
13421 .expect("chunk"),
13422 )
13423 .expect("append");
13424 }
13425 writer.finish().expect("finish");
13426 let catalog = Catalog::open(&path).expect("catalog");
13427 assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
13428 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
13429 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
13430 let frequencies =
13431 catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
13432 assert_eq!(frequencies.len(), 4);
13433 for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
13434 assert!(frequencies.contains(&pair), "missing {pair:?}");
13435 }
13436 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
13437 assert_eq!(
13438 catalog.integer_extremes("items", 1).expect("extremes"),
13439 Some(IntegerExtremes::Values { low: 0, high: 7 })
13440 );
13441 assert_eq!(
13442 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
13443 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13444 );
13445 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
13446 assert_eq!(
13447 reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
13448 vec![None, Some(2)]
13449 );
13450 Writer::certify_counts(&path).expect("recertify");
13451 assert_eq!(
13452 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
13453 Some(2)
13454 );
13455 assert_eq!(
13456 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
13457 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13458 );
13459 assert_eq!(
13460 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
13461 Some(3)
13462 );
13463 assert_eq!(
13464 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
13465 Some(IntegerExtremes::Values { low: 0, high: 7 })
13466 );
13467 assert_eq!(
13468 Catalog::open(&path)
13469 .expect("reopen")
13470 .exact_numeric_frequencies("items", 1)
13471 .expect("frequencies"),
13472 Some(frequencies)
13473 );
13474 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
13475 fs::remove_file(path).expect("remove scratch file");
13476 }
13477
13478 #[test]
13479 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
13480 let path = path("pair-frequencies");
13481 let mut pairs = Vec::new();
13482 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
13483 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
13484 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
13485 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
13486 let mut writer = Writer::create(
13487 &path,
13488 "items",
13489 vec![
13490 Field::required("id", LogicalType::BigInt),
13491 Field::required("phrase", LogicalType::Varchar),
13492 ],
13493 )
13494 .expect("new file");
13495 for part in pairs.chunks(1_024) {
13496 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
13497 let phrases =
13498 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
13499 writer
13500 .append(
13501 &Chunk::new(vec![
13502 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
13503 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
13504 ])
13505 .expect("matching columns"),
13506 )
13507 .expect("rows");
13508 }
13509 writer.finish().expect("commit");
13510
13511 let reader = Reader::open(&path).expect("reopen from disk");
13512 let leaders = reader
13513 .top_pair_frequencies(0, 1, 2)
13514 .expect("valid pair metadata")
13515 .expect("the top two beat the omitted tail");
13516 assert!(
13517 leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("alpha".to_string())], 100,))
13518 );
13519 assert!(
13520 leaders.contains(&(vec![Value::BigInt(1), Value::Varchar("beta".to_string())], 50,))
13521 );
13522 fs::remove_file(path).expect("remove scratch file");
13523 }
13524
13525 #[test]
13531 fn a_file_from_another_format_says_which_format_it_is() {
13532 let older = path("older-format");
13533 let mut writer =
13534 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
13535 .expect("new file");
13536 let chunk = Chunk::new(vec![
13537 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13538 .expect("integers"),
13539 ])
13540 .expect("chunk");
13541 writer.append(&chunk).expect("page written");
13542 writer.finish().expect("commit");
13543
13544 let unreadable =
13548 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
13549 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13550 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
13551 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
13552 drop(file);
13553 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
13554 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
13555 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
13556
13557 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13558 file.seek(SeekFrom::Start(0)).expect("the magic is first");
13559 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
13560 drop(file);
13561 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
13562 assert!(complaint.contains("magic"), "{complaint}");
13563 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
13564 fs::remove_file(older).expect("remove scratch file");
13565 }
13566
13567 #[test]
13568 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
13569 let unfinished = path("unfinished");
13570 let mut writer =
13571 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
13572 .expect("new file");
13573 let chunk = Chunk::new(vec![
13574 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13575 .expect("integers"),
13576 ])
13577 .expect("chunk");
13578 writer.append(&chunk).expect("page written");
13579 drop(writer);
13580 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
13581 fs::remove_file(unfinished).expect("remove scratch file");
13582
13583 let damaged = path("damaged");
13584 let mut writer =
13585 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
13586 .expect("new file");
13587 writer.append(&chunk).expect("page written");
13588 writer.finish().expect("commit");
13589 let reader = Reader::open(&damaged).expect("valid directory");
13590 let mut file =
13591 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
13592 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
13593 file.write_all(&[255]).expect("damage one byte");
13594 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
13595 fs::remove_file(damaged).expect("remove scratch file");
13596 }
13597
13598 #[test]
13599 fn damaged_lazy_dictionary_payload_is_an_error() {
13600 let path = path("damaged-dictionary");
13601 let mut writer = Writer::create(
13602 &path,
13603 "items",
13604 vec![
13605 Field::required("id", LogicalType::Integer),
13606 Field::new("text", LogicalType::Varchar),
13607 ],
13608 )
13609 .expect("new file");
13610 writer.append(&sample()).expect("stripe written");
13611 writer.finish().expect("commit");
13612
13613 let reader = Reader::open(&path).expect("valid directory");
13614 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
13615 let mut header = [0; DICTIONARY_HEADER];
13618 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13619 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13622 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13623 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
13624 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
13625 let mut start = [0; 8];
13626 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
13627 read_at(&reader.file, at, &mut start).expect("the first block's start");
13628 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13629 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
13630 file.write_all(&[255]).expect("damage dictionary payload");
13631
13632 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
13633 let error =
13634 chunk.validate_external().expect_err("payload corruption must reach the caller");
13635 assert!(error.message().contains("payload checksum differs"), "{error}");
13636 fs::remove_file(path).expect("remove scratch file");
13637 }
13638
13639 #[test]
13649 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
13650 let path = path("dictionary-decide");
13651 let rows = 20_000;
13652 let unique =
13654 |row: usize| format!("{row:09} a value that appears exactly once in the table");
13655 let repeated = |row: usize| unique(row / 40);
13657 let mut writer = Writer::create(
13658 &path,
13659 "items",
13660 vec![
13661 Field::required("unique", LogicalType::Varchar),
13662 Field::required("repeated", LogicalType::Varchar),
13663 ],
13664 )
13665 .expect("new file");
13666 for part in (0..rows).step_by(1_000) {
13667 let span = part..(part + 1_000).min(rows);
13668 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
13669 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
13670 writer
13671 .append(
13672 &Chunk::new(vec![
13673 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
13674 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
13675 ])
13676 .expect("two columns"),
13677 )
13678 .expect("a part");
13679 }
13680 writer.finish().expect("commit");
13681
13682 let reader = Reader::open(&path).expect("reopen from disk");
13683 assert!(
13684 reader.table.dictionaries[0].is_none(),
13685 "a column with no repeats has nothing to say twice"
13686 );
13687 assert!(
13688 reader.table.dictionaries[1].is_some(),
13689 "a column whose values come round again keeps its dictionary"
13690 );
13691 let mut first = 0;
13692 for part in 0..reader.parts() {
13693 let chunk = reader.read(part, &[0, 1]).expect("a part");
13694 for row in 0..chunk.len() {
13695 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
13696 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
13697 }
13698 first += chunk.len();
13699 }
13700 assert_eq!(first, rows, "every row was read back");
13701 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
13702 let size = fs::metadata(&path).expect("the file is there").len() as usize;
13703 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
13704 fs::remove_file(path).expect("remove scratch file");
13705 }
13706
13707 #[test]
13720 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
13721 let path = path("dictionary-blocks");
13722 let value = |row: usize| {
13723 let row = row.saturating_sub(8_000);
13724 format!("{row:07} a value long enough to be worth a payload block")
13725 };
13726 let parts = 40;
13727 let per_part = 1000;
13728 let mut writer =
13729 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13730 .expect("new file");
13731 for part in 0..parts {
13732 let values = (0..per_part)
13733 .map(|row| Value::Varchar(value(part * per_part + row)))
13734 .collect::<Vec<_>>();
13735 let chunk = Chunk::new(vec![
13736 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13737 ])
13738 .expect("matching rows");
13739 writer.append(&chunk).expect("a part");
13740 }
13741 writer.finish().expect("commit");
13742
13743 let reader = Reader::open(&path).expect("reopen from disk");
13744 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
13745 assert!(
13746 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
13747 "the dictionary has to be several blocks for this to be testing anything"
13748 );
13749 for part in [0, parts - 1] {
13750 let chunk = reader.read(part, &[0]).expect("a part");
13751 chunk.validate_external().expect("every payload block checks out");
13752 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
13753 }
13754
13755 let mut header = [0; DICTIONARY_HEADER];
13757 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13758 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13759 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
13760 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13761 let bits = (width & !(DICTIONARY_SCATTERED | DICTIONARY_GRAMS)) as usize;
13762 let mut place = [0; 16];
13763 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
13764 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
13765 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
13766 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
13767 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13768 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
13769 file.write_all(&[255]).expect("damage the last payload block");
13770 let reader = Reader::open(&path).expect("the directory and the index are untouched");
13771 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
13772 let error = chunk.validate_external().expect_err("the damage must reach the caller");
13773 assert!(error.message().contains("payload checksum differs"), "{error}");
13774 fs::remove_file(path).expect("remove scratch file");
13775 }
13776
13777 #[test]
13791 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
13792 let path = path("dictionary-offsets");
13793 let value = |row: usize| {
13794 let row = row % 5_000;
13795 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
13796 };
13797 let rows = 6_000;
13798 let mut writer =
13799 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13800 .expect("new file");
13801 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
13802 for part in values.chunks(1_000) {
13803 let chunk =
13804 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
13805 .expect("matching rows");
13806 writer.append(&chunk).expect("a part");
13807 }
13808 writer.finish().expect("commit");
13809
13810 let reader = Reader::open(&path).expect("reopen from disk");
13811 assert!(
13812 rows > TEXT_PAYLOAD_VALUES * 4,
13813 "the dictionary has to be several blocks for this to be testing anything"
13814 );
13815 for part in 0..rows / 1_000 {
13816 let chunk = reader.read(part, &[0]).expect("a part");
13817 for row in 0..1_000 {
13818 let row = part * 1_000 + row;
13819 assert_eq!(
13820 chunk.value_at(row % 1_000, 0),
13821 Value::Varchar(value(row)),
13822 "value {row}"
13823 );
13824 }
13825 }
13826 for _ in 0..2 {
13829 for part in 0..rows / 1_000 {
13830 let chunk = reader.read(part, &[0]).expect("a part");
13831 let mut lens = vec![0_i64; 1_000];
13832 let column = chunk.column(0).expect("one column");
13833 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
13834 for (row, &len) in lens.iter().enumerate() {
13835 let row = part * 1_000 + row;
13836 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
13837 }
13838 }
13839 }
13840 fs::remove_file(path).expect("remove scratch file");
13841 }
13842
13843 #[test]
13845 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
13846 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
13847 ends.extend([3, 3, 10]);
13848 let lens = lengths_of(&ends).expect("ordered ends");
13849 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
13850 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
13851 ends.push(9);
13852 assert_eq!(lengths_of(&ends), None);
13853 }
13854
13855 #[test]
13867 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
13868 let path = path("dictionary-once");
13869 let parts = 8;
13870 let per_part = 500;
13871 let value =
13872 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
13873 let mut writer =
13874 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13875 .expect("new file");
13876 for part in 0..parts {
13877 let values = (0..per_part)
13878 .map(|row| Value::Varchar(value(part * per_part + row)))
13879 .collect::<Vec<_>>();
13880 let chunk = Chunk::new(vec![
13881 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13882 ])
13883 .expect("matching rows");
13884 writer.append(&chunk).expect("a part");
13885 }
13886 writer.finish().expect("commit");
13887
13888 let reader = Reader::open(&path).expect("reopen from disk");
13889 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
13890 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
13891
13892 let workers = 16;
13893 let gate = std::sync::Barrier::new(workers);
13894 std::thread::scope(|scope| {
13895 for worker in 0..workers {
13896 let reader = reader.clone();
13897 let gate = &gate;
13898 scope.spawn(move || {
13899 gate.wait();
13900 let chunk = reader.read(worker % parts, &[0]).expect("a part");
13901 assert_eq!(
13902 chunk.value_at(0, 0),
13903 Value::Varchar(value((worker % parts) * per_part))
13904 );
13905 });
13906 }
13907 });
13908
13909 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
13910 fs::remove_file(path).expect("remove scratch file");
13911 }
13912
13913 #[test]
13918 fn a_damaged_sorted_order_is_an_error() {
13919 let path = path("damaged-order");
13920 let mut writer = Writer::create(
13921 &path,
13922 "items",
13923 vec![
13924 Field::required("id", LogicalType::Integer),
13925 Field::new("text", LogicalType::Varchar),
13926 ],
13927 )
13928 .expect("new file");
13929 writer.append(&sample()).expect("stripe written");
13930 writer.finish().expect("commit");
13931
13932 let reader = Reader::open(&path).expect("valid directory");
13933 let page = reader.table.dictionaries[1].expect("string dictionary page");
13934 let mut header = [0; DICTIONARY_HEADER];
13935 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
13936 let index_len = dictionary_index_len(&header);
13937 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13938 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
13939 file.write_all(&[255]).expect("damage the order");
13940
13941 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
13942 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
13943 assert!(error.message().contains("rank checksum differs"), "{error}");
13944 fs::remove_file(path).expect("remove scratch file");
13945 }
13946
13947 #[test]
13951 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
13952 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
13955 let path = path("dictionary-order");
13956 let mut writer =
13957 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
13958 .expect("new file");
13959 writer
13960 .append(
13961 &Chunk::new(vec![
13962 Vector::from_values(
13963 LogicalType::Varchar,
13964 &spellings.map(|text| Value::Varchar(text.into())),
13965 )
13966 .expect("strings"),
13967 ])
13968 .expect("one column"),
13969 )
13970 .expect("stripe written");
13971 writer.finish().expect("commit");
13972
13973 let reader = Reader::open(&path).expect("valid directory");
13974 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
13975 let count = dictionary.ranks().expect("a v10 file stores one");
13976 assert_eq!(count, spellings.len(), "every distinct value has a rank");
13977 let order = (0..count)
13978 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
13979 .collect::<Vec<_>>();
13980 let mut seen = order.clone();
13981 seen.sort_unstable();
13982 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
13983
13984 let ranked = order
13985 .iter()
13986 .map(|&code| {
13987 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
13988 })
13989 .collect::<Vec<_>>();
13990 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
13991 expected.sort();
13992 assert_eq!(ranked, expected, "rank order is value order");
13993
13994 for (rank, value) in expected.iter().enumerate() {
13997 assert_eq!(
13998 dictionary.compare_rank(rank, value).expect("compare"),
13999 Ordering::Equal,
14000 "rank {rank} is its own value"
14001 );
14002 if rank > 0 {
14003 assert_eq!(
14004 dictionary.compare_rank(rank - 1, value).expect("compare"),
14005 Ordering::Less,
14006 "rank {rank} follows the one before it"
14007 );
14008 }
14009 }
14010 fs::remove_file(path).expect("remove scratch file");
14011 }
14012
14013 #[test]
14020 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
14021 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
14022 let path = path("dictionaries-at-once");
14023 let fields = (0..sizes.len())
14024 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
14025 .collect::<Vec<_>>();
14026 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14027 let rows = 10_000_usize;
14028 for start in (0..rows).step_by(1_024) {
14029 let columns = sizes
14030 .iter()
14031 .enumerate()
14032 .map(|(column, &size)| {
14033 let values = (start..(start + 1_024).min(rows))
14034 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
14035 .collect::<Vec<_>>();
14036 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
14037 })
14038 .collect::<Vec<_>>();
14039 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
14040 }
14041 writer.finish().expect("commit");
14042
14043 let reader = Reader::open(&path).expect("valid directory");
14044 for (column, &size) in sizes.iter().enumerate() {
14045 let dictionary =
14046 reader.dictionary(column).expect("read").expect("a string column has one");
14047 let count = dictionary.ranks().expect("a v10 file stores one");
14048 assert_eq!(count, size, "column {column} has its own distinct count");
14049 let ranked = (0..count)
14050 .map(|rank| {
14051 let code = dictionary.code_at_rank(rank).expect("a code");
14052 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14053 })
14054 .collect::<Vec<_>>();
14055 let expected = (0..size)
14056 .map(|value| format!("c{column}-{value:05}").into_bytes())
14057 .collect::<Vec<_>>();
14058 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
14059 }
14060 fs::remove_file(path).expect("remove scratch file");
14061 }
14062
14063 #[test]
14071 fn a_large_dictionary_ranks_in_value_order() {
14072 let path = path("dictionary-large-rank");
14073 let value = |row: u64| {
14074 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
14075 match row % 3 {
14076 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
14077 1 => format!("{mixed}"),
14078 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
14079 }
14080 };
14081 let distinct = 70_000;
14082 let parts = 4 * distinct / 1000;
14083 let per_part = 1000;
14084 let mut writer =
14085 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14086 .expect("new file");
14087 for part in 0..parts {
14088 let values = (0..per_part)
14089 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
14090 .collect::<Vec<_>>();
14091 let chunk = Chunk::new(vec![
14092 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14093 ])
14094 .expect("matching rows");
14095 writer.append(&chunk).expect("a part");
14096 }
14097 writer.finish().expect("commit");
14098
14099 let reader = Reader::open(&path).expect("reopen from disk");
14100 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14101 let count = dictionary.ranks().expect("a ranked dictionary");
14102 assert_eq!(count, distinct as usize, "every distinct value has a rank");
14103 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
14104 let ranked = (0..count)
14105 .map(|rank| {
14106 let code = dictionary.code_at_rank(rank).expect("a code");
14107 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14108 })
14109 .collect::<Vec<_>>();
14110 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
14111 expected.sort();
14112 assert_eq!(ranked, expected, "rank order is value order");
14113 fs::remove_file(path).expect("remove scratch file");
14114 }
14115
14116 #[test]
14129 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
14130 let path = path("windowed-directory");
14131 let fields = vec![
14132 Field::required("id", LogicalType::BigInt),
14133 Field::required("word", LogicalType::Varchar),
14134 Field::new("score", LogicalType::Double),
14135 ];
14136 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14137 for part in 0..70_i64 {
14138 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
14139 let words = (0..100)
14140 .map(|row| Value::Varchar(format!("word {}", row % 13)))
14141 .collect::<Vec<_>>();
14142 let scores = (0..100)
14143 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
14144 .collect::<Vec<_>>();
14145 let chunk = Chunk::new(vec![
14146 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
14147 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
14148 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
14149 ])
14150 .expect("three columns");
14151 writer.append(&chunk).expect("a part");
14152 }
14153 writer.finish().expect("commit");
14154
14155 let catalog = Catalog::open(&path).expect("reopen");
14156 let entry = catalog.entries.first().expect("one table").directory;
14157 let (offset, length) = (entry.offset, entry.length as usize);
14158 let mut bytes = vec![0; length];
14159 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
14160 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
14161 let whole = decode_directory(&bytes, catalog.size).expect("whole");
14162 assert!(whole.stripes.len() > 1, "the table should span stripes");
14163 for size in [1, 7, 33, 4_096] {
14164 let mut cursor = Cursor::over(&catalog.file, offset, length);
14165 cursor.window.as_mut().expect("a window").size = size;
14166 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
14167 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
14168 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
14169 let mut stored = 0;
14170 for (column, (left, held)) in
14171 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
14172 {
14173 match (left, held) {
14174 (None, None) => {}
14175 (
14176 Some(super::Frequencies::Stored { span, values }),
14177 Some(super::Frequencies::Held(summary)),
14178 ) => {
14179 let mut one = vec![0; span.length as usize];
14180 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
14181 let read = decode_summary(
14182 &mut Cursor::new(&one),
14183 &whole.fields[column],
14184 whole.rows,
14185 *values,
14186 )
14187 .expect("a valid synopsis")
14188 .expect("one is there");
14189 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
14190 stored += 1;
14191 }
14192 other => panic!("column {column} came back as {other:?}"),
14193 }
14194 }
14195 assert!(stored >= 2, "only {stored} synopses were left in the file");
14196 }
14197 let reader = catalog.table("items").expect("the table");
14198 assert!(reader.frequency_summaries[1].get().is_none());
14199 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
14200 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
14201 let clone = reader.clone();
14202 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
14203 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
14204 fs::remove_file(path).expect("remove scratch file");
14205 }
14206
14207 #[test]
14208 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
14209 let path = path("file-checksum");
14210 let bytes = (0..200_000_u32)
14211 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
14212 .collect::<Vec<_>>();
14213 fs::write(&path, &bytes).expect("scratch file");
14214 let file = File::open(&path).expect("open");
14215 for (offset, length) in [
14216 (0, 0),
14217 (3, 1),
14218 (5, 31),
14219 (0, 32),
14220 (9, 33),
14221 (1, 65_536),
14222 (7, 65_567),
14223 (0, 200_000),
14224 (11, 131_101),
14225 ] {
14226 let whole = checksum(&bytes[offset..offset + length]);
14227 assert_eq!(
14228 file_checksum(&file, offset as u64, length).expect("read"),
14229 whole,
14230 "{offset} {length}"
14231 );
14232 }
14233 fs::remove_file(path).expect("remove scratch file");
14234 }
14235
14236 #[test]
14237 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
14238 let path = path("synopsis-keeps-no-block");
14239 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
14240 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
14241 for _ in 0..3 {
14242 values.extend((0..3_000).step_by(5).map(spelled));
14243 }
14244 let mut writer =
14245 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14246 .expect("new file");
14247 for part in values.chunks(1_024) {
14248 writer
14249 .append(
14250 &Chunk::new(vec![
14251 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14252 ])
14253 .expect("one column"),
14254 )
14255 .expect("a part");
14256 }
14257 writer.finish().expect("commit");
14258
14259 let reader = Reader::open(&path).expect("reopen from disk");
14260 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14261 let resting = dictionary.footprint();
14262 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14263 assert_eq!(prefix.entries.len(), 512);
14264 for (value, count) in &prefix.entries {
14265 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
14266 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
14267 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
14268 }
14269 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
14270 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14271 assert_eq!(again.entries, prefix.entries);
14272 fs::remove_file(path).expect("remove scratch file");
14273 }
14274
14275 #[test]
14285 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
14286 let path = path("dictionary-sweep");
14287 let spellings = (0..2_500)
14290 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14291 .collect::<Vec<_>>();
14292 let mut writer =
14293 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14294 .expect("new file");
14295 for part in spellings.chunks(1_024) {
14298 writer
14299 .append(
14300 &Chunk::new(vec![
14301 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14302 ])
14303 .expect("one column"),
14304 )
14305 .expect("stripe written");
14306 }
14307 writer.finish().expect("commit");
14308
14309 let reader = Reader::open(&path).expect("valid directory");
14310 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14311 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14312 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
14313 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
14314 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
14315 }
14316
14317 let resting = dictionary.footprint();
14318 let sweep = || {
14319 let mut swept: Vec<Vec<u8>> = Vec::new();
14320 let mut at = 0;
14321 let mut calls = 0;
14322 while at < dictionary.len() {
14323 let stopped = dictionary
14324 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14325 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14326 swept.push(text.to_vec());
14327 Ok(())
14328 })
14329 .expect("a sweep reads");
14330 assert!(stopped > at, "a sweep moves");
14331 at = stopped;
14332 calls += 1;
14333 }
14334 assert_eq!(calls, 3, "a sweep hands over one block at a time");
14335 swept
14336 };
14337 let swept = sweep();
14338 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
14339 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
14340 let after = dictionary.footprint();
14341 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
14342
14343 let read = (0..dictionary.len())
14344 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14345 .collect::<Vec<_>>();
14346 assert_eq!(swept, read, "a sweep answers what a point read answers");
14347 let grown = dictionary.footprint() - after;
14351 assert!(
14352 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
14353 "a point read of a kept block decodes nothing, and {grown} bytes grew"
14354 );
14355 fs::remove_file(path).expect("remove scratch file");
14356 }
14357
14358 #[test]
14359 fn a_damaged_substring_signature_is_checked_only_when_used() {
14360 let path = path("damaged-substring-signature");
14361 let mut writer =
14362 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14363 .expect("new file");
14364 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
14365 writer
14366 .append(
14367 &Chunk::new(vec![
14368 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
14369 ])
14370 .expect("one column"),
14371 )
14372 .expect("stripe written");
14373 writer.finish().expect("commit");
14374
14375 let reader = Reader::open(&path).expect("valid directory");
14376 let page = reader.table.dictionaries[0].expect("string dictionary page");
14377 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14378 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
14379 .expect("last signature byte");
14380 file.write_all(&[255]).expect("damage signature");
14381 let reader = Reader::open(&path).expect("the directory is still valid");
14382 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
14383 let error = dictionary
14384 .text_block_might_contain(0, b"goog")
14385 .expect_err("a used signature checks its own checksum");
14386 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
14387 fs::remove_file(path).expect("remove scratch file");
14388 }
14389
14390 #[test]
14401 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
14402 let path = path("dictionary-sweep-short-run");
14403 let spellings = (0..2_800)
14404 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14405 .collect::<Vec<_>>();
14406 let mut writer =
14407 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14408 .expect("new file");
14409 for part in spellings.chunks(1_024) {
14410 writer
14411 .append(
14412 &Chunk::new(vec![
14413 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14414 ])
14415 .expect("one column"),
14416 )
14417 .expect("stripe written");
14418 }
14419 writer.finish().expect("commit");
14420
14421 let reader = Reader::open(&path).expect("valid directory");
14422 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14423 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14424 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
14425 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
14426 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
14427
14428 let mut swept: Vec<Vec<u8>> = Vec::new();
14429 let mut at = 0;
14430 while at < dictionary.len() {
14431 let stopped = dictionary
14432 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14433 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14434 swept.push(text.to_vec());
14435 Ok(())
14436 })
14437 .expect("a sweep reads");
14438 assert!(stopped > at, "a sweep moves");
14439 at = stopped;
14440 }
14441 let read = (0..dictionary.len())
14442 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14443 .collect::<Vec<_>>();
14444 assert_eq!(swept, read, "a sweep answers what a point read answers");
14445 fs::remove_file(path).expect("remove scratch file");
14446 }
14447
14448 #[test]
14457 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
14458 let path = path("dictionary-unpacked-ends");
14459 let spellings = (0..2_800)
14460 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14461 .collect::<Vec<_>>();
14462 let mut writer =
14463 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14464 .expect("new file");
14465 for part in spellings.chunks(1_024) {
14466 writer
14467 .append(
14468 &Chunk::new(vec![
14469 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14470 ])
14471 .expect("one column"),
14472 )
14473 .expect("stripe written");
14474 }
14475 writer.finish().expect("commit");
14476
14477 let reader = Reader::open(&path).expect("valid directory");
14478 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14479 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14480 let wanted = (0..spellings.len())
14481 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
14482 .collect::<Vec<_>>();
14483
14484 let pass = |what: &str| {
14485 for (index, value) in wanted.iter().enumerate() {
14486 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
14487 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
14488 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
14489 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
14490 }
14491 };
14492 pass("the first pass");
14493 pass("the second pass");
14494
14495 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
14499 let mut whole = vec![0i64; wanted.len()];
14500 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
14501 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
14502 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
14503 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
14504 let mut through = vec![0i64; codes.len()];
14505 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
14506 for (row, &code) in codes.iter().enumerate() {
14507 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
14508 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
14509 assert_eq!(through[row], one as i64, "row {row} a row at a time");
14510 }
14511
14512 let fresh = Reader::open(&path).expect("valid directory");
14515 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
14516 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
14517 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
14518 let mut short = vec![0i64; few.len()];
14519 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
14520 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
14521 assert_eq!(short, expected, "the packed ends answer what the table answers");
14522 fs::remove_file(path).expect("remove scratch file");
14523 }
14524
14525 #[test]
14535 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
14536 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
14537 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
14538 fit::<i8>(&[128]).expect_err("one past the top does not fit");
14539 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
14540 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
14541 fit::<u8>(&[256]).expect_err("one past the top does not fit");
14542 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
14543 assert_eq!(
14544 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
14545 vec![-32_768_i16, 0, 32_767]
14546 );
14547 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
14548 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
14549 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
14550 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
14551 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
14552 assert_eq!(
14553 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
14554 vec![i32::MIN, 0, i32::MAX]
14555 );
14556 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
14557 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
14558 assert_eq!(
14559 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
14560 vec![0_u32, 4_294_967_295]
14561 );
14562 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
14563 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
14564
14565 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
14568 }
14569
14570 #[test]
14577 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
14578 for value in -70_000_i64..70_000 {
14579 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
14580 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
14581 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
14582 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
14583 }
14584 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
14585 for edge in wide {
14586 for step in -2_i64..=2 {
14587 let value = edge.saturating_add(step);
14588 assert_eq!(
14589 fit::<i32>(&[value]).is_ok(),
14590 i32::try_from(value).is_ok(),
14591 "{value} as i32"
14592 );
14593 assert_eq!(
14594 fit::<u32>(&[value]).is_ok(),
14595 u32::try_from(value).is_ok(),
14596 "{value} as u32"
14597 );
14598 }
14599 }
14600 }
14601
14602 #[test]
14617 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
14618 let spellings = (0..3_000)
14619 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
14620 .collect::<Vec<_>>();
14621 let mut read = Vec::new();
14622 for layout in ["outside", "inside", "behind"] {
14623 let mut dictionary = GlobalDictionary::new();
14624 for text in &spellings {
14625 dictionary.code(text).expect("a code for every spelling");
14626 }
14627 dictionary.finish_blocks().expect("the last block encodes");
14628 let order = dictionary.ranked(None).expect("a sorted order");
14629 let laid = |from: u64| {
14631 let mut at = from;
14632 dictionary
14633 .blocks
14634 .iter()
14635 .map(|block| {
14636 let place =
14637 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
14638 at += block.len() as u64;
14639 place
14640 })
14641 .collect::<Vec<_>>()
14642 };
14643 let payload = dictionary.blocks.concat();
14644 let scattered = layout != "behind";
14645 let (bytes, encoded, offset, length) = if layout == "outside" {
14646 let mut bytes = vec![0; HEADER as usize];
14647 bytes.extend_from_slice(&payload);
14648 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
14649 .expect("an encoding");
14650 let offset = bytes.len() as u64;
14651 bytes.extend_from_slice(&encoded.index);
14652 bytes.extend_from_slice(&encoded.ranks);
14653 bytes.extend_from_slice(&encoded.grams);
14654 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
14655 (bytes, encoded, offset, length)
14656 } else {
14657 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
14660 .expect("an encoding");
14661 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
14662 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
14663 .expect("an encoding");
14664 let mut bytes = encoded.index.clone();
14665 bytes.extend_from_slice(&encoded.ranks);
14666 bytes.extend_from_slice(&encoded.grams);
14667 bytes.extend_from_slice(&payload);
14668 let length = bytes.len();
14669 (bytes, encoded, 0, length)
14670 };
14671 let path = path(&format!("blocks-{layout}"));
14672 fs::write(&path, &bytes).expect("the dictionary is written on its own");
14673 let file = Arc::new(File::open(&path).expect("it opens again"));
14674 let page = Page {
14675 offset,
14676 length: u32::try_from(length).expect("a test dictionary is small"),
14677 hash: checksum(&encoded.index),
14678 };
14679 let opened =
14680 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
14681 .expect("a dictionary laid out either way opens");
14682 let mut swept: Vec<Vec<u8>> = Vec::new();
14683 let mut at = 0;
14684 while at < opened.len() {
14685 at = opened
14686 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
14687 swept.push(text.to_vec());
14688 Ok(())
14689 })
14690 .expect("a sweep reads");
14691 }
14692 fs::remove_file(&path).expect("clean up");
14693 read.push(swept);
14694 }
14695 let wanted =
14696 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
14697 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
14698 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
14699 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
14700 }
14701
14702 #[test]
14710 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
14711 let path = path("dictionary-budget");
14712 let spellings = (0..2_500)
14713 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
14714 .collect::<Vec<_>>();
14715 let mut writer =
14716 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14717 .expect("new file");
14718 for part in spellings.chunks(1_024) {
14719 writer
14720 .append(
14721 &Chunk::new(vec![
14722 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14723 ])
14724 .expect("one column"),
14725 )
14726 .expect("stripe written");
14727 }
14728 writer.finish().expect("commit");
14729
14730 let reader = Reader::open(&path).expect("valid directory");
14731 let page = reader.table.dictionaries[0].expect("a string column has one");
14732 let file = Arc::clone(&reader.file);
14733 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
14734 .expect("a dictionary opens whatever it may keep");
14735
14736 let resting = starved.footprint();
14737 let mut swept: Vec<Vec<u8>> = Vec::new();
14738 let mut at = 0;
14739 while at < starved.len() {
14740 at = starved
14741 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
14742 swept.push(text.to_vec());
14743 Ok(())
14744 })
14745 .expect("a sweep reads");
14746 }
14747 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
14748 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
14749
14750 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
14751 let read = (0..generous.len())
14752 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
14753 .collect::<Vec<_>>();
14754 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
14755 fs::remove_file(path).expect("remove scratch file");
14756 }
14757
14758 #[test]
14759 fn damaged_membership_cannot_skip_a_string_page() {
14760 let path = path("damaged-membership");
14761 let mut writer = Writer::create(
14762 &path,
14763 "items",
14764 vec![
14765 Field::required("id", LogicalType::Integer),
14766 Field::new("text", LogicalType::Varchar),
14767 ],
14768 )
14769 .expect("new file");
14770 writer.append(&sample()).expect("stripe written");
14771 writer.finish().expect("commit");
14772
14773 let reader = Reader::open(&path).expect("valid directory");
14774 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
14775 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
14776 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
14777 file.write_all(&[255]).expect("damage membership");
14778 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
14779 assert!(error.message().contains("membership page checksum differs"), "{error}");
14780 fs::remove_file(path).expect("remove scratch file");
14781 }
14782
14783 #[test]
14784 fn membership_delta_stream_is_sorted_exact_and_bounded() {
14785 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
14786 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
14787 let encoded = encode_membership(&unique);
14788 assert_eq!(
14789 decode_membership(&encoded).expect("valid membership"),
14790 [4, 9, 72, 900, u32::MAX]
14791 );
14792 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
14795 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
14796 assert_eq!(
14797 decode_membership(&encode_membership(&merged)).expect("valid membership"),
14798 unique
14799 );
14800 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
14801 assert!(
14802 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
14803 "a value past u32 is invalid"
14804 );
14805 }
14806
14807 #[test]
14808 fn a_global_dictionary_may_be_larger_than_one_column_page() {
14809 let dictionary = Page {
14810 offset: HEADER,
14811 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
14812 hash: 0,
14813 };
14814 let table = Table {
14815 name: "items".to_owned(),
14816 fields: vec![Field::new("text", LogicalType::Varchar)],
14817 stripes: Vec::new(),
14818 rows: 0,
14819 dictionaries: vec![Some(dictionary)],
14820 dictionary_payloads: Vec::new(),
14821 distincts: vec![None],
14822 frequencies: vec![None],
14823 pair_frequencies: Vec::new(),
14824 frequency_texts: Vec::new(),
14825 host_groups: None,
14826 clustering: None,
14827 generation: 1,
14828 sections: Vec::new(),
14829 };
14830 let directory = encode_directory(&table).expect("directory");
14831 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
14832
14833 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
14834 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
14835 }
14836
14837 #[test]
14838 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
14839 let path = path("constant-codes");
14840 let mut writer =
14841 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14842 .expect("new file");
14843 let empty = vec![Value::Varchar(String::new()); 1024];
14844 for _ in 0..4 {
14845 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
14846 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
14847 }
14848 writer.finish().expect("commit");
14849
14850 let reader = Reader::open(&path).expect("valid directory");
14851 let pages = reader.layout().columns.first().expect("one column").pages;
14852 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
14856 let read = reader.read(3, &[0]).expect("the last part back");
14857 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
14858 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
14859 fs::remove_file(path).expect("remove scratch file");
14860 }
14861
14862 #[test]
14863 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
14864 let over = vec![i64::from(i32::MAX) + 1];
14867 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
14868 assert!(format!("{error}").contains("not of its type"), "{error}");
14869 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
14870 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
14871 }
14872
14873 #[test]
14874 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
14875 let mut state: u32 = 0x9e37_79b9;
14879 let spread: Vec<u32> = (0..1024)
14880 .map(|_| {
14881 state ^= state << 13;
14882 state ^= state >> 17;
14883 state ^= state << 5;
14884 state
14885 })
14886 .collect();
14887 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
14888 let near: Vec<u32> = (0..1024).collect();
14889 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
14890 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
14891 }
14892
14893 #[test]
14899 fn two_writes_of_the_same_rows_give_the_same_bytes() {
14900 fn written(path: &PathBuf) {
14901 let fields = (0..40)
14902 .map(|column| {
14903 let ty =
14904 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
14905 Field::new(format!("c{column}"), ty)
14906 })
14907 .collect::<Vec<_>>();
14908 let mut writer = Writer::create(path, "wide", fields).expect("new file");
14909 for part in 0..70_u64 {
14910 let columns = (0..40)
14911 .map(|column| {
14912 let values = (0..64_u64)
14913 .map(|row| {
14914 let seed = part.wrapping_mul(31).wrapping_add(row);
14915 if column % 4 == 0 {
14916 Value::Varchar(format!("v{}", seed % 17))
14917 } else {
14918 Value::BigInt(i64::try_from(seed % 97).expect("small"))
14919 }
14920 })
14921 .collect::<Vec<_>>();
14922 let ty = if column % 4 == 0 {
14923 LogicalType::Varchar
14924 } else {
14925 LogicalType::BigInt
14926 };
14927 Vector::from_values(ty, &values).expect("a column")
14928 })
14929 .collect::<Vec<_>>();
14930 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
14931 }
14932 writer.finish().expect("commit");
14933 }
14934
14935 let first = path("repeatable-one");
14936 let second = path("repeatable-two");
14937 written(&first);
14938 written(&second);
14939 let left = fs::read(&first).expect("the first file");
14940 let right = fs::read(&second).expect("the second file");
14941 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
14942 assert!(left == right, "two writes of the same rows differ in their bytes");
14943
14944 let reader = Reader::open(&first).expect("valid directory");
14947 assert_eq!(reader.table().rows(), 70 * 64);
14948 let read = reader.read(0, &[0, 1]).expect("the first part back");
14949 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
14950 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
14951 fs::remove_file(first).expect("remove scratch file");
14952 fs::remove_file(second).expect("remove scratch file");
14953 }
14954
14955 fn three_tables(path: &PathBuf) {
14957 let writer = Writer::create(
14958 path,
14959 "region",
14960 vec![
14961 Field::new("r_key", LogicalType::Integer),
14962 Field::new("r_name", LogicalType::Varchar),
14963 ],
14964 )
14965 .expect("new file");
14966 let mut writer = writer;
14967 writer
14968 .append(
14969 &Chunk::new(vec![
14970 Vector::from_values(
14971 LogicalType::Integer,
14972 &[Value::Integer(0), Value::Integer(1)],
14973 )
14974 .expect("keys"),
14975 Vector::from_values(
14976 LogicalType::Varchar,
14977 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
14978 )
14979 .expect("names"),
14980 ])
14981 .expect("two columns"),
14982 )
14983 .expect("a part");
14984 let mut writer = writer
14985 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
14986 .expect("a second table");
14987 writer
14988 .append(
14989 &Chunk::new(vec![
14990 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
14991 ])
14992 .expect("one column"),
14993 )
14994 .expect("a part");
14995 let mut writer =
14996 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
14997 for part in 0..70_i64 {
14998 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
14999 writer
15000 .append(
15001 &Chunk::new(vec![
15002 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
15003 ])
15004 .expect("one column"),
15005 )
15006 .expect("a part");
15007 }
15008 writer.finish().expect("commit");
15009 }
15010
15011 #[test]
15012 fn three_tables_in_one_file_read_back_by_name() {
15013 let file = path("three-tables");
15014 three_tables(&file);
15015 let catalog = Catalog::open(&file).expect("a committed catalog");
15016 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
15017
15018 let region = catalog.table("region").expect("the first table");
15019 assert_eq!(region.table().rows(), 2);
15020 assert_eq!(
15021 region.read(0, &[1]).expect("names").value_at(1, 0),
15022 Value::Varchar("ASIA".to_owned())
15023 );
15024
15025 let wide = catalog.table("wide").expect("the third table");
15026 assert_eq!(wide.table().rows(), 70 * 64);
15027 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
15028
15029 let empty = catalog.table("empty").expect("the second table");
15032 assert_eq!(empty.table().rows(), 1);
15033 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
15034
15035 fs::remove_file(file).expect("remove scratch file");
15036 }
15037
15038 #[test]
15039 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
15040 let file = path("three-tables-missing");
15041 three_tables(&file);
15042 let catalog = Catalog::open(&file).expect("a committed catalog");
15043 let error = catalog.table("nation").expect_err("no such table");
15044 assert!(error.message().contains("nation"), "{}", error.message());
15045 fs::remove_file(file).expect("remove scratch file");
15046 }
15047
15048 #[test]
15049 fn a_file_of_three_tables_will_not_open_as_one() {
15050 let file = path("three-tables-unnamed");
15051 three_tables(&file);
15052 let error = Reader::open(&file).expect_err("more than one table");
15053 assert!(error.message().contains("more than one table"), "{}", error.message());
15054 fs::remove_file(file).expect("remove scratch file");
15055 }
15056
15057 #[test]
15059 fn decimals_of_every_storage_width_round_trip() {
15060 let file = path("decimals");
15061 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
15062 let fields = widths
15063 .iter()
15064 .enumerate()
15065 .map(|(index, (width, scale))| {
15066 Field::new(
15067 format!("d{index}"),
15068 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15069 )
15070 })
15071 .collect::<Vec<_>>();
15072 let mut writer = Writer::create(&file, "money", fields).expect("new file");
15073 let rows: [i128; 3] = [-1234, 0, 999];
15074 let columns = widths
15075 .iter()
15076 .map(|(width, scale)| {
15077 let values = rows
15078 .iter()
15079 .map(|unscaled| Value::Decimal {
15080 unscaled: *unscaled,
15081 width: *width,
15082 scale: *scale,
15083 })
15084 .collect::<Vec<_>>();
15085 Vector::from_values(
15086 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15087 &values,
15088 )
15089 .expect("a decimal column")
15090 })
15091 .collect::<Vec<_>>();
15092 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
15093 writer.finish().expect("commit");
15094
15095 let reader = Reader::open(&file).expect("a committed file");
15096 for (index, (width, scale)) in widths.iter().enumerate() {
15097 assert_eq!(
15098 reader.table().fields()[index].ty,
15099 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15100 "column {index} came back as another type"
15101 );
15102 let column = reader.read(0, &[index]).expect("the column");
15103 for (row, unscaled) in rows.iter().enumerate() {
15104 assert_eq!(
15105 column.value_at(row, 0),
15106 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
15107 "column {index} row {row}"
15108 );
15109 }
15110 }
15111 fs::remove_file(file).expect("remove scratch file");
15112 }
15113
15114 #[test]
15115 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
15116 let file = path("two-of-a-name");
15117 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
15118 .expect("new file");
15119 let error = writer
15120 .next("t", vec![Field::new("a", LogicalType::BigInt)])
15121 .expect_err("the same name twice");
15122 assert!(error.message().contains("same name"), "{}", error.message());
15123 fs::remove_file(file).expect("remove scratch file");
15124 }
15125
15126 #[test]
15127 fn opening_the_catalog_reads_no_table_directory() {
15128 let file = path("catalog-only");
15129 three_tables(&file);
15130 let catalog = Catalog::open(&file).expect("a committed catalog");
15131 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
15134 assert_eq!(catalog.names().len(), 3);
15135 fs::remove_file(file).expect("remove scratch file");
15136 }
15137
15138 #[test]
15149 fn the_checksum_answers_what_it_has_always_answered() {
15150 let bytes: Vec<u8> =
15151 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
15152 for (length, expected) in [
15153 (0, 0xef46_db37_51d8_e999),
15154 (1, 0xa96c_7f0c_e858_bbb7),
15155 (3, 0x56e6_9576_32a4_87f9),
15156 (4, 0xc60d_15b1_e3ff_8f04),
15157 (5, 0x8088_1585_8624_dd4e),
15158 (7, 0xafbe_fc3d_6c6f_9a8e),
15159 (8, 0x3da5_c7aa_2696_83e0),
15160 (9, 0x465e_c429_b13c_3892),
15161 (15, 0xdee8_9d8a_065a_6233),
15162 (16, 0x1330_489a_7767_9c80),
15163 (31, 0x3391_303d_485e_846e),
15164 (32, 0x40b7_aff7_5d45_bbc8),
15165 (33, 0x4997_cae4_951c_17a5),
15166 (39, 0x5807_28fd_5c14_5739),
15167 (40, 0xf95c_f6f5_c08a_3d3b),
15168 (63, 0x2944_b4da_fc69_b206),
15169 (64, 0xbb76_f6ef_19bd_5a1b),
15170 (65, 0x814e_0c65_4a9f_d640),
15171 (127, 0x00de_aab1_31cf_f89b),
15172 (1000, 0x9e33_00c1_cde3_c58d),
15173 ] {
15174 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
15175 }
15176 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
15177 }
15178 #[test]
15185 fn a_declared_order_comes_back_out_of_the_file() {
15186 let path = path("clustered");
15187 let shipped = vec![
15188 Field::new("key", LogicalType::BigInt),
15189 Field::new("line", LogicalType::Integer),
15190 Field::new("shipdate", LogicalType::Date),
15191 ];
15192 let plain = vec![Field::new("a", LogicalType::Integer)];
15193 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
15194
15195 let mut writer = Writer::create(&path, "lineitem", shipped)
15196 .expect("new file")
15197 .declare(stage_zero.clone())
15198 .expect("the columns are the table's");
15199 let column = |ty: LogicalType, values: &[Value]| {
15200 Vector::from_values(ty, values).expect("the values match the type")
15201 };
15202 writer
15203 .append(
15204 &Chunk::new(vec![
15205 column(
15206 LogicalType::BigInt,
15207 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
15208 ),
15209 column(
15210 LogicalType::Integer,
15211 &[
15212 Value::Integer(1),
15213 Value::Integer(1),
15214 Value::Integer(1),
15215 Value::Integer(1),
15216 ],
15217 ),
15218 column(
15219 LogicalType::Date,
15220 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
15221 ),
15222 ])
15223 .expect("three columns"),
15224 )
15225 .expect("four rows");
15226 let mut writer = writer.next("nation", plain).expect("a second table");
15227 writer
15228 .append(
15229 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
15230 .expect("one column"),
15231 )
15232 .expect("one row");
15233 writer.finish().expect("commit");
15234
15235 let catalog = Catalog::open(&path).expect("reopen");
15236 let lineitem = catalog.table("lineitem").expect("the clustered table");
15237 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
15238 let nation = catalog.table("nation").expect("the plain table");
15239 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
15240
15241 assert_eq!(lineitem.table().rows(), 4);
15244 assert_eq!(nation.table().rows(), 1);
15245 fs::remove_file(&path).ok();
15246 }
15247
15248 #[test]
15250 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
15251 let path = path("clustered-bad");
15252 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
15253 .expect("new file");
15254 let four =
15255 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
15256 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
15257 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
15258 fs::remove_file(&path).ok();
15259 }
15260
15261 #[test]
15267 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
15268 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
15269 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
15270 .collect::<Vec<_>>();
15271 let filled = || {
15272 let mut dictionary = GlobalDictionary::new();
15273 for value in &values {
15274 dictionary.code(value).expect("a code for every value");
15275 }
15276 dictionary.settle().expect("a shape");
15277 dictionary
15278 };
15279 let mut in_place = filled();
15280 in_place.finish_blocks().expect("every block encodes");
15281
15282 let mut handed = filled();
15283 let out = handed.hand_out(3);
15284 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
15285 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
15286 for job in out.iter().rev() {
15287 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
15288 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
15289 }
15290 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
15291 handed.finish_blocks().expect("the last block encodes");
15292
15293 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
15294 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
15295 }
15296
15297 #[test]
15299 fn a_block_given_back_twice_is_refused() {
15300 let mut dictionary = GlobalDictionary::new();
15301 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
15302 dictionary.code(&format!("value {at}")).expect("a code");
15303 }
15304 dictionary.settle().expect("a shape");
15305 let out = dictionary.hand_out(0);
15306 let last = out.last().expect("blocks went out");
15307 let at = last.place().1;
15308 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
15309 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
15310 }
15311
15312 #[test]
15318 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
15319 let mut values = vec![String::new(), "http://".to_owned()];
15320 for host in 0..7 {
15321 for path in 0..30 {
15322 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
15323 values.push(format!("http://example{host}.test/page/{path:04}"));
15324 }
15325 }
15326 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
15327
15328 let mut dictionary = GlobalDictionary::new();
15329 for value in &values {
15330 dictionary.code(value).expect("a code for every value");
15331 }
15332 dictionary.finish_blocks().expect("the last block encodes");
15333 let ranked = dictionary.ranked(None).expect("a sorted order");
15334 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
15335
15336 let spellings = dictionary_values(&dictionary);
15337 let seen = ranked
15338 .iter()
15339 .map(|&(_, code)| {
15340 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
15341 })
15342 .collect::<Vec<_>>();
15343 let mut wanted = values.clone();
15344 wanted.sort_unstable();
15345 assert_eq!(seen, wanted, "the order is the order the bytes give");
15346
15347 for &(carried, code) in &ranked {
15348 let value = &spellings[code as usize];
15349 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
15350 }
15351 }
15352
15353 #[test]
15358 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
15359 let entry =
15360 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
15361 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
15362 .map(|code| entry(code, u64::from(code % 7) + 1))
15363 .collect::<Vec<_>>();
15364 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
15365
15366 let mut sorted = all.clone();
15367 sorted.sort_unstable_by(|left, right| {
15368 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
15369 });
15370 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
15371 sorted.truncate(FREQUENCY_ENTRIES);
15372
15373 let mut picked = all.clone();
15374 let omitted = keep_most_frequent(&mut picked);
15375 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
15376 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
15377 assert!(
15378 picked
15379 .iter()
15380 .zip(&sorted)
15381 .all(|(one, two)| one.value == two.value && one.count == two.count),
15382 "the same entries in the same order"
15383 );
15384
15385 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
15386 let omitted = keep_most_frequent(&mut short);
15387 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
15388 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
15389 }
15390
15391 #[test]
15393 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
15394 let empty = GlobalDictionary::new();
15395 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
15396
15397 let mut dictionary = GlobalDictionary::new();
15398 for value in ["pear", "apple", "", "apples", "app"] {
15399 dictionary.code(value).expect("a code for every value");
15400 }
15401 dictionary.finish_blocks().expect("the one block encodes");
15402 let spellings = dictionary_values(&dictionary);
15403 let seen = dictionary
15404 .ranked(None)
15405 .expect("a sorted order")
15406 .iter()
15407 .map(|&(_, code)| spellings[code as usize].clone())
15408 .collect::<Vec<_>>();
15409 let wanted: Vec<Vec<u8>> =
15410 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
15411 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
15412 }
15413}