1#![forbid(unsafe_code)]
34
35use std::borrow::Cow;
36use std::cmp::Ordering;
37use std::collections::{BTreeMap, HashMap, VecDeque};
38use std::fs::{File, OpenOptions};
39use std::io::{Read, Seek, SeekFrom};
40use std::mem::{size_of, size_of_val};
41use std::path::Path;
42use std::slice;
43use std::sync::atomic::{AtomicBool, AtomicUsize, Ordering as Atomic};
44use std::sync::{Arc, Condvar, Mutex, OnceLock, PoisonError, Weak};
45
46use rudb_common::bounds::{self, Bound, Op, scaled_as};
47use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
48use rudb_encoding::{bitpack, chooser, integer, string};
49use rudb_metrics::{LoadProfile, Stage};
50use rudb_storage::sieve::Sieve;
51use rudb_storage::{Probe, Range, Zone};
52use rudb_vector::string::StringColumn;
53use rudb_vector::validity::Validity;
54use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
55
56mod distinct;
57pub mod graph;
58pub mod host;
59mod prepare;
60use prepare::Lent;
61pub mod section;
62pub mod stats;
63mod zones;
64
65pub use prepare::{Merged, Merger, Paged, Prepared, Preparer};
66pub use section::Section;
67pub use zones::{Common, Stripes, ascending, distincts};
68
69const MAGIC: &[u8; 8] = b"RUDBNV10";
70const DIRECTORY: &[u8; 8] = b"RUDBDI10";
71const CATALOG: &[u8; 8] = b"RUDBCA10";
72const NONZERO_COUNTS: &[u8; 8] = b"RUDBNZ10";
73const AGGREGATE_SUMS: &[u8; 8] = b"RUDBAG10";
74const DISTINCT_COUNTS: &[u8; 8] = b"RUDBDC10";
75const INTEGER_EXTREMES: &[u8; 8] = b"RUDBEX10";
76const COMPLETE_FREQUENCIES: &[u8; 8] = b"RUDBFQ10";
77const MAX_CATALOG_FREQUENCIES: usize = 64;
78const FORMAT: u32 = 29;
79
80const READABLE: &[u32] = &[22, 23, 24, 25, 26, 27, 28, FORMAT];
111
112const HEADER: u64 = 80;
113const SLOT_BYTES: usize = 28;
114const MAX_PAGE: usize = 256 * 1024 * 1024;
115const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
116const FREQUENCIES_V2: &[u8; 8] = b"RUDBFQ2\0";
117const FREQUENCIES: &[u8; 8] = b"RUDBFQ3\0";
118const FREQUENCY_TEXTS: &[u8; 8] = b"RUDBFT1\0";
126const HOST_GROUPS: &[u8; 8] = b"RUDBHG1\0";
128const PAIR_FREQUENCIES: &[u8; 8] = b"RUDBPF1\0";
134const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
149const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
157const DICTIONARY_PAYLOADS: &[u8; 8] = b"RUDBDP1\0";
165
166const MAX_SECTIONS: usize = 4096;
173const FREQUENCY_CANDIDATES: usize = 32_768;
174const FREQUENCY_ENTRIES: usize = 512;
175const FREQUENCY_BUILD_RANK: usize = 10;
176const FREQUENCY_ORDINALS: usize = 131_072;
177const MAX_PAIR_FREQUENCIES: usize = 1024;
178const FREQUENCY_TEXT_BUDGET: usize = 1024 * 1024;
183const MAX_FREQUENCY_WORKERS: usize = 32;
190
191fn close_workers() -> usize {
193 std::thread::available_parallelism().map_or(1, usize::from).min(MAX_FREQUENCY_WORKERS)
194}
195
196const CLOSE_DICTIONARY_BYTES: usize = 1 << 30;
206
207const MAX_ENCODE_WORKERS: usize = 32;
214
215const WRITEBACK_STRETCH: u64 = 32 << 20;
223
224const SIEVE_BUDGET: usize = 8 * 1024;
232
233const PART_BOUND_BYTES: usize = 24;
242
243fn io(error: std::io::Error) -> Error {
244 Error::io(error.to_string())
245}
246
247fn invalid(message: &str) -> Error {
248 Error::invalid_input(format!("invalid rudb native file: {message}"))
249}
250
251fn sum(counts: impl Iterator<Item = u64>) -> u64 {
253 counts.fold(0, u64::saturating_add)
254}
255
256fn span_bytes(spans: &[Span], at: usize) -> u64 {
258 spans.get(at).map_or(0, |span| u64::from(span.length))
259}
260
261fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
263 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
264}
265
266fn dictionary_bytes(table: &Table, at: usize) -> u64 {
268 page_bytes(&table.dictionaries, at)
269 .saturating_add(table.dictionary_payloads.get(at).copied().unwrap_or(0))
270}
271
272fn checksum(bytes: &[u8]) -> u64 {
282 seeded_checksum(bytes, 0)
283}
284
285#[must_use]
292pub fn content_name(bytes: &[u8]) -> u128 {
293 let seed = u64::from(FORMAT);
294 u128::from(seeded_checksum(bytes, seed)) << 64 | u128::from(seeded_checksum(bytes, !seed))
295}
296
297#[derive(Debug, Clone)]
303pub struct ContentNamer {
304 seeds: [u64; 2],
305 lanes: [[u64; 4]; 2],
306 held: [u8; 32],
307 filled: usize,
308 length: u64,
309}
310
311impl Default for ContentNamer {
312 fn default() -> Self {
313 let seed = u64::from(FORMAT);
314 let seeds = [seed, !seed];
315 let lanes = seeds.map(|seed| {
316 [
317 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
318 seed.wrapping_add(XXH_P2),
319 seed,
320 seed.wrapping_sub(XXH_P1),
321 ]
322 });
323 Self { seeds, lanes, held: [0; 32], filled: 0, length: 0 }
324 }
325}
326
327impl ContentNamer {
328 pub fn update(&mut self, mut bytes: &[u8]) {
330 self.length += bytes.len() as u64;
331 if self.filled > 0 {
332 let take = (32 - self.filled).min(bytes.len());
333 self.held[self.filled..self.filled + take].copy_from_slice(&bytes[..take]);
334 self.filled += take;
335 bytes = &bytes[take..];
336 if self.filled < 32 {
337 return;
338 }
339 let block = self.held;
340 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, &block));
341 self.filled = 0;
342 }
343 let mut blocks = bytes.chunks_exact(32);
344 for block in blocks.by_ref() {
345 self.lanes.iter_mut().for_each(|lanes| checksum_block(lanes, block));
346 }
347 let rest = blocks.remainder();
348 self.held[..rest.len()].copy_from_slice(rest);
349 self.filled = rest.len();
350 }
351
352 #[must_use]
354 pub fn finish(&self) -> u128 {
355 let rest = &self.held[..self.filled];
356 let [first, second] = [0, 1].map(|at| {
357 if self.length < 32 {
358 checksum_tail(self.seeds[at].wrapping_add(XXH_P5).wrapping_add(self.length), rest)
359 } else {
360 finish_checksum(self.lanes[at], rest, self.length)
361 }
362 });
363 u128::from(first) << 64 | u128::from(second)
364 }
365}
366
367fn seeded_checksum(bytes: &[u8], seed: u64) -> u64 {
376 let mut blocks = bytes.chunks_exact(32);
379 let rest = blocks.remainder();
380 if bytes.len() < 32 {
381 return checksum_tail(seed.wrapping_add(XXH_P5).wrapping_add(bytes.len() as u64), rest);
382 }
383 let mut lanes = [
384 seed.wrapping_add(XXH_P1).wrapping_add(XXH_P2),
385 seed.wrapping_add(XXH_P2),
386 seed,
387 seed.wrapping_sub(XXH_P1),
388 ];
389 for block in blocks.by_ref() {
390 checksum_block(&mut lanes, block);
391 }
392 finish_checksum(lanes, rest, bytes.len() as u64)
393}
394
395const XXH_P1: u64 = 11_400_714_785_074_694_791;
396const XXH_P2: u64 = 14_029_467_366_897_019_727;
397const XXH_P3: u64 = 1_609_587_929_392_839_161;
398const XXH_P4: u64 = 9_650_029_242_287_828_579;
399const XXH_P5: u64 = 2_870_177_450_012_600_261;
400
401fn checksum_round(state: u64, word: u64) -> u64 {
402 state.wrapping_add(word.wrapping_mul(XXH_P2)).rotate_left(31).wrapping_mul(XXH_P1)
403}
404
405fn checksum_word(chunk: &[u8]) -> u64 {
406 u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"))
407}
408
409fn checksum_block(lanes: &mut [u64; 4], block: &[u8]) {
411 for (lane, chunk) in lanes.iter_mut().zip(block.chunks_exact(8)) {
412 *lane = checksum_round(*lane, checksum_word(chunk));
413 }
414}
415
416fn finish_checksum(lanes: [u64; 4], rest: &[u8], length: u64) -> u64 {
418 let merge = |state: u64, lane: u64| {
419 (state ^ checksum_round(0, lane)).wrapping_mul(XXH_P1).wrapping_add(XXH_P4)
420 };
421 let [one, two, three, four] = lanes;
422 let combined = one
423 .rotate_left(1)
424 .wrapping_add(two.rotate_left(7))
425 .wrapping_add(three.rotate_left(12))
426 .wrapping_add(four.rotate_left(18));
427 let hash = merge(merge(merge(merge(combined, one), two), three), four);
428 checksum_tail(hash.wrapping_add(length), rest)
429}
430
431fn checksum_tail(mut hash: u64, mut rest: &[u8]) -> u64 {
433 let mut words = rest.chunks_exact(8);
434 for chunk in words.by_ref() {
435 hash ^= checksum_round(0, checksum_word(chunk));
436 hash = hash.rotate_left(27).wrapping_mul(XXH_P1).wrapping_add(XXH_P4);
437 }
438 rest = words.remainder();
439 if rest.len() >= 4 {
440 let (head, tail) = rest.split_at(4);
441 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
442 hash ^= u64::from(quarter).wrapping_mul(XXH_P1);
443 hash = hash.rotate_left(23).wrapping_mul(XXH_P2).wrapping_add(XXH_P3);
444 rest = tail;
445 }
446 for &byte in rest {
447 hash ^= u64::from(byte).wrapping_mul(XXH_P5);
448 hash = hash.rotate_left(11).wrapping_mul(XXH_P1);
449 }
450 hash ^= hash >> 33;
451 hash = hash.wrapping_mul(XXH_P2);
452 hash ^= hash >> 29;
453 hash = hash.wrapping_mul(XXH_P3);
454 hash ^ (hash >> 32)
455}
456
457fn file_checksum(file: &File, offset: u64, length: usize) -> Result<u64> {
463 walk_checksummed(file, offset, length, DIRECTORY_WINDOW, |_| Ok(()))
464}
465
466fn walk_checksummed(
472 file: &File,
473 offset: u64,
474 length: usize,
475 window: usize,
476 mut each: impl FnMut(&[u8]) -> Result<()>,
477) -> Result<u64> {
478 debug_assert!(window % 32 == 0 && window > 0, "a window is whole blocks of the hash");
479 if length < 32 {
480 let mut bytes = vec![0; length];
481 read_at(file, offset, &mut bytes)?;
482 each(&bytes)?;
483 return Ok(checksum(&bytes));
484 }
485 let mut lanes = [XXH_P1.wrapping_add(XXH_P2), XXH_P2, 0, 0_u64.wrapping_sub(XXH_P1)];
486 let mut buffer = vec![0; window.min(length)];
487 let mut read = 0;
488 let (mut whole, mut filled) = (0, 0);
489 while read < length {
490 filled = buffer.len().min(length - read);
491 read_at(file, offset + read as u64, &mut buffer[..filled])?;
492 read += filled;
493 each(&buffer[..filled])?;
494 whole = filled / 32 * 32;
495 for block in buffer[..whole].chunks_exact(32) {
496 checksum_block(&mut lanes, block);
497 }
498 }
499 Ok(finish_checksum(lanes, &buffer[whole..filled], length as u64))
500}
501
502#[derive(Debug, Clone, Copy)]
503struct Slot {
504 offset: u64,
505 length: u32,
506 generation: u64,
507 hash: u64,
508}
509
510impl Slot {
511 fn bytes(self) -> [u8; SLOT_BYTES] {
512 let mut result = [0; SLOT_BYTES];
513 result[..8].copy_from_slice(&self.offset.to_le_bytes());
514 result[8..12].copy_from_slice(&self.length.to_le_bytes());
515 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
516 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
517 result
518 }
519
520 fn read(bytes: &[u8]) -> Self {
521 Self {
522 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
523 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
524 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
525 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
526 }
527 }
528}
529
530#[derive(Debug, Clone, Copy)]
531struct Page {
532 offset: u64,
533 length: u32,
534 hash: u64,
535}
536
537impl Page {
538 fn bytes(&self) -> u64 {
540 u64::from(self.length)
541 }
542}
543
544#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
545enum FrequencyValue {
546 Null,
547 Integer(i128),
548 Code(u32),
549}
550
551type FrequencyMap<V> = HashMap<u64, V, Spread>;
557
558#[derive(Debug, Default)]
561struct Candidates {
562 counts: FrequencyMap<u32>,
563 nulls: u32,
564 decrements: u64,
565}
566
567impl Candidates {
568 fn add(&mut self, bits: Option<u64>, mut times: u32) {
575 while times > 0 {
576 let held = match bits {
577 Some(bits) => self.counts.get_mut(&bits),
578 None if self.nulls != 0 => Some(&mut self.nulls),
579 None => None,
580 };
581 if let Some(count) = held {
582 *count = count.saturating_add(times);
583 return;
584 }
585 if self.counts.len() + usize::from(self.nulls != 0) < FREQUENCY_CANDIDATES {
586 match bits {
587 Some(bits) => {
588 self.counts.insert(bits, times);
589 }
590 None => self.nulls = times,
591 }
592 return;
593 }
594 self.counts.retain(|_, count| {
595 *count -= 1;
596 *count != 0
597 });
598 self.nulls = self.nulls.saturating_sub(1);
599 self.decrements = self.decrements.saturating_add(1);
600 times -= 1;
601 }
602 }
603}
604
605#[derive(Debug, Default)]
607struct Run {
608 bits: Option<u64>,
609 times: u32,
610}
611
612impl Run {
613 fn push(&mut self, bits: Option<u64>) -> Option<(Option<u64>, u32)> {
615 if self.times != 0 && self.bits == bits && self.times < u32::MAX {
616 self.times += 1;
617 return None;
618 }
619 let ended = self.take();
620 self.bits = bits;
621 self.times = 1;
622 ended
623 }
624
625 fn take(&mut self) -> Option<(Option<u64>, u32)> {
627 let times = std::mem::take(&mut self.times);
628 (times != 0).then_some((self.bits, times))
629 }
630}
631
632#[derive(Debug, Default, Clone, Copy)]
634struct Spread;
635
636impl std::hash::BuildHasher for Spread {
637 type Hasher = SpreadHasher;
638
639 fn build_hasher(&self) -> SpreadHasher {
640 SpreadHasher(0)
641 }
642}
643
644#[derive(Debug)]
651struct SpreadHasher(u64);
652
653impl SpreadHasher {
654 fn mix(&mut self, word: u64) {
655 let product = u128::from(self.0 ^ word) * 0x9E37_79B9_7F4A_7C15_u128;
656 self.0 = (product as u64) ^ ((product >> 64) as u64);
657 }
658}
659
660impl std::hash::Hasher for SpreadHasher {
661 fn write(&mut self, bytes: &[u8]) {
662 for part in bytes.chunks(8) {
663 let mut word = [0; 8];
664 word[..part.len()].copy_from_slice(part);
665 self.mix(u64::from_le_bytes(word));
666 }
667 }
668
669 fn write_u32(&mut self, value: u32) {
670 self.mix(u64::from(value));
671 }
672
673 fn write_u64(&mut self, value: u64) {
674 self.mix(value);
675 }
676
677 fn write_i128(&mut self, value: i128) {
678 self.mix(value as u64);
679 self.mix((value >> 64) as u64);
680 }
681
682 fn write_isize(&mut self, value: isize) {
683 self.mix(value as u64);
684 }
685
686 fn finish(&self) -> u64 {
687 self.0
688 }
689}
690
691#[derive(Debug, Clone)]
692struct FrequencyEntry {
693 value: FrequencyValue,
694 count: u64,
695}
696
697#[derive(Debug, Clone)]
702struct FrequencySummary {
703 entries: Vec<FrequencyEntry>,
704 omitted_max: u64,
705 ordinals: Vec<u64>,
706 ordinal_entries: Vec<u16>,
707}
708
709#[derive(Debug, Clone)]
710struct PairFrequencyEntry {
711 first_entry: u16,
712 second: Option<u32>,
713 count: u64,
714}
715
716#[derive(Debug, Clone)]
722struct PairFrequencySummary {
723 first: u16,
724 second: u16,
725 entries: Vec<PairFrequencyEntry>,
726 omitted_max: u64,
727}
728
729#[derive(Debug, Clone)]
737enum Frequencies {
738 Held(FrequencySummary),
739 Stored {
742 span: Span,
743 values: bool,
744 },
745}
746
747#[derive(Debug, Clone)]
752pub struct FrequencyPrefix {
753 pub entries: Vec<(Value, u64)>,
755 pub omitted_max: u64,
757}
758
759#[derive(Debug, Clone, PartialEq)]
761pub struct FrequencyOccurrences {
762 pub omitted_max: u64,
764 pub ordinals: Vec<u64>,
766 pub anchors: Vec<Value>,
768 pub anchor_indices: Vec<u16>,
770}
771
772pub type PairFrequencyCounts = Vec<(Vec<Value>, u64)>;
774
775#[derive(Debug, Clone, Copy, Default)]
782struct Span {
783 offset: u64,
784 length: u32,
785}
786
787#[derive(Debug, Clone, Default)]
795struct Pages {
796 columns: usize,
797 held: Box<[StripePage]>,
798}
799
800#[derive(Debug, Clone, Copy)]
802struct StripePage {
803 offset: u64,
804 hash: u64,
805 length: u32,
806 column: u32,
807}
808
809impl Pages {
810 fn from_slots(slots: Vec<Option<Page>>) -> Result<Self> {
812 let mut held = Vec::with_capacity(slots.iter().flatten().count());
813 for (column, page) in slots.iter().enumerate() {
814 if let Some(page) = page {
815 let column =
816 u32::try_from(column).map_err(|_| invalid("too many columns for a page"))?;
817 held.push(StripePage {
818 offset: page.offset,
819 hash: page.hash,
820 length: page.length,
821 column,
822 });
823 }
824 }
825 Ok(Self { columns: slots.len(), held: held.into_boxed_slice() })
826 }
827
828 fn get(&self, column: usize) -> Option<Page> {
830 let at = self.held.binary_search_by_key(&column, |placed| placed.column as usize).ok()?;
831 let placed = self.held[at];
832 Some(Page { offset: placed.offset, length: placed.length, hash: placed.hash })
833 }
834
835 fn slots(&self) -> impl Iterator<Item = Option<Page>> + '_ {
837 (0..self.columns).map(|column| self.get(column))
838 }
839
840 fn bytes(&self, column: usize) -> u64 {
842 self.get(column).map_or(0, |page| page.bytes())
843 }
844}
845
846#[derive(Debug, Clone)]
848pub struct Stripe {
849 rows: usize,
850 parts: Vec<u32>,
853 index: Span,
857 pages: Vec<Span>,
858 memberships: Pages,
859 sieves: Pages,
862 part_ranges: Pages,
873 zone: Zone,
874}
875
876impl Stripe {
877 #[must_use]
879 pub fn rows(&self) -> usize {
880 self.rows
881 }
882
883 #[must_use]
885 pub fn parts(&self) -> usize {
886 self.parts.len()
887 }
888
889 #[must_use]
895 pub fn zone(&self) -> &Zone {
896 &self.zone
897 }
898}
899
900#[derive(Debug, Clone)]
902pub struct Table {
903 name: String,
904 fields: Vec<Field>,
905 stripes: Vec<Stripe>,
906 rows: usize,
907 dictionaries: Vec<Option<Page>>,
908 dictionary_payloads: Vec<u64>,
914 frequencies: Vec<Option<Frequencies>>,
915 pair_frequencies: Vec<PairFrequencySummary>,
916 frequency_texts: Vec<Vec<Option<Vec<u8>>>>,
921 host_groups: Option<host::HostSummary>,
923 distincts: Vec<Option<u64>>,
933 clustering: Option<Clustering>,
941 generation: u64,
955 sections: Vec<Section>,
962}
963
964impl Table {
965 #[must_use]
967 pub fn name(&self) -> &str {
968 &self.name
969 }
970
971 #[must_use]
973 pub fn fields(&self) -> &[Field] {
974 &self.fields
975 }
976
977 #[must_use]
979 pub fn rows(&self) -> usize {
980 self.rows
981 }
982
983 #[must_use]
985 pub fn stripes(&self) -> &[Stripe] {
986 &self.stripes
987 }
988
989 #[must_use]
991 pub fn clustering(&self) -> Option<&Clustering> {
992 self.clustering.as_ref()
993 }
994
995 #[must_use]
1000 pub fn generation(&self) -> u64 {
1001 self.generation
1002 }
1003
1004 #[must_use]
1011 pub fn sections(&self) -> &[Section] {
1012 &self.sections
1013 }
1014}
1015
1016#[derive(Debug, Clone)]
1028struct Entry {
1029 name: String,
1030 fields: Vec<Field>,
1031 rows: usize,
1032 directory: Page,
1034 nonzero: Vec<Option<u64>>,
1036 aggregates: Vec<Option<(i128, u64)>>,
1038 distincts: Vec<Option<u64>>,
1040 extremes: Vec<StoredIntegerExtremes>,
1042 frequencies: Vec<StoredNumericFrequencies>,
1044}
1045
1046type StoredIntegerExtremes = Option<Option<(i128, i128)>>;
1047type StoredNumericFrequencies = Option<NumericFrequencies>;
1048
1049#[derive(Debug, Clone, PartialEq, Eq)]
1062pub struct ViewEntry {
1063 pub name: String,
1065 pub sql: String,
1067 pub statement: String,
1069 pub aliases: Vec<String>,
1071 pub columns: Vec<Field>,
1073}
1074
1075#[derive(Debug, Clone)]
1077pub struct ColumnLayout {
1078 pub name: String,
1080 pub kind: String,
1082 pub pages: u64,
1084 pub memberships: u64,
1086 pub sieves: u64,
1088 pub part_ranges: u64,
1090 pub dictionary: u64,
1092}
1093
1094impl ColumnLayout {
1095 #[must_use]
1097 pub fn total(&self) -> u64 {
1098 self.pages
1099 .saturating_add(self.memberships)
1100 .saturating_add(self.sieves)
1101 .saturating_add(self.part_ranges)
1102 .saturating_add(self.dictionary)
1103 }
1104}
1105
1106#[derive(Debug, Clone)]
1117pub struct Layout {
1118 pub file: u64,
1120 pub rows: usize,
1122 pub stripes: usize,
1124 pub parts: usize,
1126 pub columns: Vec<ColumnLayout>,
1128 pub indexes: u64,
1131 pub directory: u64,
1133 pub header: u64,
1135}
1136
1137impl Layout {
1138 #[must_use]
1140 pub fn columns_total(&self) -> u64 {
1141 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
1142 }
1143
1144 #[must_use]
1150 pub fn unaccounted(&self) -> u64 {
1151 self.file
1152 .saturating_sub(self.columns_total())
1153 .saturating_sub(self.indexes)
1154 .saturating_sub(self.directory)
1155 .saturating_sub(self.header)
1156 }
1157}
1158
1159#[derive(Debug, Clone)]
1170pub struct StoredPart {
1171 pub stripe: usize,
1173 pub part: usize,
1175 pub row: usize,
1177 pub rows: usize,
1179 pub encoding: String,
1181 pub bytes: u64,
1183 pub page: u64,
1185 pub offset: u64,
1187 pub low: Option<Value>,
1189 pub high: Option<Value>,
1191 pub nulls: Option<usize>,
1193}
1194
1195const DICTIONARY_CHECK_SEED: u64 = 11_400_714_819_323_198_485;
1202
1203#[derive(Debug)]
1228struct GlobalDictionary {
1229 primary: HashMap<u64, u32>,
1230 collisions: HashMap<u64, Vec<u32>>,
1231 checks: Vec<u64>,
1233 ends: Vec<u32>,
1235 counts: Vec<u64>,
1236 nulls: u64,
1237 filling: Vec<u8>,
1239 grams: Vec<[u8; TEXT_GRAM_BYTES]>,
1245 waiting: Vec<(usize, Vec<u8>)>,
1250 sample: Vec<(usize, Vec<u8>)>,
1256 stride: usize,
1258 shape: Option<chooser::Settled>,
1260 settled: usize,
1262 blocks: Vec<Vec<u8>>,
1267 early: BTreeMap<usize, EncodedBlock>,
1273 placed: Vec<Placed>,
1275}
1276
1277#[derive(Debug, Clone, Copy)]
1279struct Placed {
1280 start: u64,
1281 length: u64,
1282 hash: u64,
1283}
1284
1285type RankedDictionary = (Vec<(u64, u32)>, Vec<u8>, Vec<u64>);
1287
1288impl GlobalDictionary {
1289 fn new() -> Self {
1290 Self {
1291 primary: HashMap::new(),
1292 collisions: HashMap::new(),
1293 checks: Vec::new(),
1294 ends: Vec::new(),
1295 counts: Vec::new(),
1296 nulls: 0,
1297 filling: Vec::new(),
1298 grams: Vec::new(),
1299 waiting: Vec::new(),
1300 sample: Vec::new(),
1301 stride: 1,
1302 shape: None,
1303 settled: 0,
1304 blocks: Vec::new(),
1305 early: BTreeMap::new(),
1306 placed: Vec::new(),
1307 }
1308 }
1309
1310 fn values(&self) -> usize {
1312 self.ends.len()
1313 }
1314
1315 fn closing_bytes(&self) -> usize {
1318 let values = self.values();
1319 let decoded = (0..values.div_ceil(TEXT_PAYLOAD_VALUES))
1320 .map(|block| self.ends[((block + 1) * TEXT_PAYLOAD_VALUES).min(values) - 1] as usize)
1321 .sum::<usize>();
1322 decoded.saturating_add(values.saturating_mul(size_of::<(u64, u32)>() + size_of::<u32>()))
1323 }
1324
1325 fn encoded(&self) -> usize {
1327 self.placed.len() + self.blocks.len()
1328 }
1329
1330 #[cfg(test)]
1331 fn code(&mut self, text: &str) -> Result<u32> {
1332 let bytes = text.as_bytes();
1333 self.code_hashed(bytes, checksum(bytes), seeded_checksum(bytes, DICTIONARY_CHECK_SEED))
1334 }
1335
1336 fn code_hashed(&mut self, text: &[u8], hash: u64, check: u64) -> Result<u32> {
1342 if let Some(&code) = self.primary.get(&hash) {
1343 if self.checks.get(code as usize) == Some(&check) {
1344 return Ok(code);
1345 }
1346 if let Some(codes) = self.collisions.get(&hash) {
1347 if let Some(code) =
1348 codes.iter().copied().find(|&code| self.checks[code as usize] == check)
1349 {
1350 return Ok(code);
1351 }
1352 }
1353 let code = self.insert(text, check)?;
1354 self.collisions.entry(hash).or_default().push(code);
1355 return Ok(code);
1356 }
1357 let code = self.insert(text, check)?;
1358 self.primary.insert(hash, code);
1359 Ok(code)
1360 }
1361
1362 fn insert(&mut self, text: &[u8], check: u64) -> Result<u32> {
1363 let code = u32::try_from(self.ends.len())
1364 .map_err(|_| invalid("global dictionary has too many values"))?;
1365 self.filling.extend_from_slice(text);
1366 self.ends.push(
1367 u32::try_from(self.filling.len())
1368 .map_err(|_| invalid("a global dictionary value exceeds 4 GiB"))?,
1369 );
1370 self.checks.push(check);
1371 self.counts.push(0);
1372 if self.ends.len() % TEXT_PAYLOAD_VALUES == 0 {
1373 self.seal();
1374 }
1375 Ok(code)
1376 }
1377
1378 fn seal(&mut self) {
1384 let at = self.ends.len().div_ceil(TEXT_PAYLOAD_VALUES) - 1;
1385 let bytes = std::mem::take(&mut self.filling);
1386 if at % self.stride == 0 {
1387 self.sample.push((at, bytes.clone()));
1388 if self.sample.len() > PAYLOAD_SAMPLE_BLOCKS {
1389 self.stride *= 2;
1390 let stride = self.stride;
1391 self.sample.retain(|(at, _)| at % stride == 0);
1392 }
1393 }
1394 self.waiting.push((at, bytes));
1395 }
1396
1397 fn slices<'a>(&self, at: usize, bytes: &'a [u8]) -> Vec<&'a [u8]> {
1399 block_values(self.block_ends(at), bytes)
1400 }
1401
1402 fn block_ends(&self, at: usize) -> &[u32] {
1404 let first = (at * TEXT_PAYLOAD_VALUES).min(self.ends.len());
1405 let last = (first + TEXT_PAYLOAD_VALUES).min(self.ends.len());
1406 &self.ends[first..last]
1407 }
1408
1409 fn hand_out(&mut self, column: usize) -> Vec<Unencoded> {
1416 let Some(shape) = &self.shape else { return Vec::new() };
1417 let waiting = std::mem::take(&mut self.waiting);
1418 waiting
1419 .into_iter()
1420 .map(|(at, bytes)| Unencoded {
1421 column,
1422 at,
1423 ends: self.block_ends(at).to_vec(),
1424 bytes,
1425 shape: shape.clone(),
1426 })
1427 .collect()
1428 }
1429
1430 fn take_back(&mut self, at: usize, block: EncodedBlock) -> Result<()> {
1433 if at < self.encoded() || self.early.insert(at, block).is_some() {
1434 return Err(Error::internal("a dictionary block came back twice"));
1435 }
1436 while let Some(block) = self.early.remove(&self.encoded()) {
1437 self.push_block(block);
1438 }
1439 Ok(())
1440 }
1441
1442 fn push_block(&mut self, (bytes, grams): EncodedBlock) {
1444 self.blocks.push(bytes);
1445 self.grams.push(*grams);
1446 }
1447
1448 fn settle(&mut self) -> Result<()> {
1456 if self.sample.len() < PAYLOAD_SAMPLE_BLOCKS {
1457 return Ok(());
1458 }
1459 let complete = self.ends.len() / TEXT_PAYLOAD_VALUES;
1460 if self.shape.is_some() && complete < self.settled.saturating_mul(4) {
1461 return Ok(());
1462 }
1463 let sample =
1464 self.sample.iter().map(|(at, bytes)| self.slices(*at, bytes)).collect::<Vec<_>>();
1465 self.shape = Some(string::with_symbols(settle_shape(&sample)?, &sample));
1466 self.settled = complete;
1467 Ok(())
1468 }
1469
1470 fn seal_rest(&mut self) {
1472 if self.ends.len() % TEXT_PAYLOAD_VALUES != 0 {
1475 self.seal();
1476 }
1477 }
1478
1479 fn encode_waiting(&self, at: usize) -> Result<EncodedBlock> {
1482 let (block, bytes) = &self.waiting[at];
1483 let values = self.slices(*block, bytes);
1484 let encoded = match &self.shape {
1485 Some(shape) => string::encode_with(&values, shape)?,
1486 None => string::encode(&values)?,
1487 };
1488 Ok((encoded, block_grams(&values)))
1489 }
1490
1491 #[cfg(test)]
1493 fn finish_blocks(&mut self) -> Result<()> {
1494 self.seal_rest();
1495 let made = (0..self.waiting.len())
1496 .map(|at| self.encode_waiting(at))
1497 .collect::<Result<Vec<_>>>()?;
1498 for ((at, _), block) in std::mem::take(&mut self.waiting).into_iter().zip(made) {
1499 if self.encoded() != at {
1500 return Err(Error::internal("a dictionary block was encoded out of order"));
1501 }
1502 self.push_block(block);
1503 }
1504 Ok(())
1505 }
1506
1507 fn decoded(&self, file: Option<&File>) -> Result<(Vec<u8>, Vec<u64>)> {
1525 let count = self.placed.len() + self.blocks.len();
1526 if count != self.values().div_ceil(TEXT_PAYLOAD_VALUES) {
1527 return Err(invalid("global dictionary blocks do not cover its values"));
1528 }
1529 let mut bases = Vec::with_capacity(count);
1530 let mut total = 0_usize;
1531 for block in 0..count {
1532 bases.push(total as u64);
1533 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(self.values()) - 1;
1534 total = total
1535 .checked_add(self.ends[last] as usize)
1536 .ok_or_else(|| invalid("global dictionary does not fit in memory"))?;
1537 }
1538 let mut flat = vec![0_u8; total];
1539 let mut outs = Vec::with_capacity(count);
1540 let mut rest = flat.as_mut_slice();
1541 for block in 0..count {
1542 let end = bases.get(block + 1).map_or(total, |&base| base as usize);
1543 let (out, after) = rest.split_at_mut(end - bases[block] as usize);
1544 outs.push((block, out));
1545 rest = after;
1546 }
1547 let one = |run: &mut [(usize, &mut [u8])]| -> Result<()> {
1548 let mut stored = Vec::new();
1549 for (block, out) in run {
1550 let encoded = match self.placed.get(*block) {
1551 Some(place) => {
1552 let file = file.ok_or_else(|| {
1553 Error::internal("a written dictionary block has no file")
1554 })?;
1555 let length = usize::try_from(place.length).map_err(|_| {
1556 invalid("global dictionary block does not fit in memory")
1557 })?;
1558 stored.resize(length, 0);
1559 read_at(file, place.start, &mut stored)?;
1560 if checksum(&stored) != place.hash {
1561 return Err(invalid(
1562 "a global dictionary block did not read back as written",
1563 ));
1564 }
1565 stored.as_slice()
1566 }
1567 None => &self.blocks[*block - self.placed.len()],
1568 };
1569 let decoded = string::decode_flat(encoded)?;
1570 if decoded.bytes().len() != out.len() {
1571 return Err(invalid(
1572 "a global dictionary block is not the length its ends say",
1573 ));
1574 }
1575 out.copy_from_slice(decoded.bytes());
1576 }
1577 Ok(())
1578 };
1579 let workers = close_workers().min(count / 16).max(1);
1582 if workers <= 1 {
1583 one(&mut outs)?;
1584 } else {
1585 let per = count.div_ceil(workers);
1586 std::thread::scope(|scope| {
1587 outs.chunks_mut(per)
1588 .map(|run| scope.spawn(|| one(run)))
1589 .collect::<Vec<_>>()
1590 .into_iter()
1591 .try_for_each(|handle| {
1592 handle.join().map_err(|_| {
1593 Error::internal("a global dictionary decode worker panicked")
1594 })?
1595 })
1596 })?;
1597 }
1598 drop(outs);
1599 Ok((flat, bases))
1600 }
1601
1602 fn value_span(ends: &[u32], bases: &[u64], code: usize) -> (usize, usize) {
1607 let Some(&base) = bases.get(code / TEXT_PAYLOAD_VALUES) else { return (0, 0) };
1608 let Some(&end) = ends.get(code) else { return (0, 0) };
1609 let base = base as usize;
1610 let from = if code % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[code - 1] as usize };
1611 (base + from, base + end as usize)
1612 }
1613
1614 fn ranked_with_values(&self, file: Option<&File>) -> Result<RankedDictionary> {
1634 let (flat, bases) = self.decoded(file)?;
1635 let value = |code: u32| {
1636 let (from, to) = Self::value_span(&self.ends, &bases, code as usize);
1637 flat.get(from..to).unwrap_or_default()
1638 };
1639 let mut codes = (0..self.values() as u32).collect::<Vec<_>>();
1640 sort_by_value_across(&mut codes, value, close_workers());
1641 let order = codes.into_iter().map(|code| (head(value(code)), code)).collect();
1642 Ok((order, flat, bases))
1643 }
1644
1645 #[cfg(test)]
1646 fn ranked(&self, file: Option<&File>) -> Result<Vec<(u64, u32)>> {
1647 self.ranked_with_values(file).map(|(order, _, _)| order)
1648 }
1649}
1650
1651#[derive(Debug)]
1659pub struct Writer {
1660 file: File,
1661 at: u64,
1669 written_back: u64,
1671 table: Table,
1672 generation: u64,
1673 order: Vec<((u64, u64), (u64, u64))>,
1676 next_order: u64,
1677 dictionaries: Vec<Option<GlobalDictionary>>,
1678 coded: Arc<[AtomicBool]>,
1681 gathers: Vec<Option<stats::Gather>>,
1687 lent: Option<Arc<Lent>>,
1690 pending: Vec<PendingChunk>,
1691 closed: Vec<Entry>,
1693 views: Vec<ViewEntry>,
1698 profile: Option<Arc<LoadProfile>>,
1704}
1705
1706#[derive(Debug)]
1714struct PendingChunk {
1715 order: (u64, u64),
1716 chunk: Chunk,
1717}
1718
1719#[derive(Debug, Clone, Copy)]
1725struct Part {
1726 order: (u64, u64),
1727 rows: usize,
1728 footprint: usize,
1729}
1730
1731impl Part {
1732 fn of(pending: &PendingChunk) -> Self {
1733 Self {
1734 order: pending.order,
1735 rows: pending.chunk.len(),
1736 footprint: pending.chunk.footprint(),
1737 }
1738 }
1739}
1740
1741#[derive(Debug)]
1747struct ColumnStripe {
1748 pages: Vec<Vec<u8>>,
1749 codes: Vec<Option<Vec<u32>>>,
1750 sieves: Vec<Option<Sieve>>,
1751 ranges: Vec<Range>,
1752}
1753
1754fn weight(ty: &LogicalType) -> usize {
1762 match ty {
1763 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
1764 LogicalType::HugeInt
1765 | LogicalType::UHugeInt
1766 | LogicalType::Uuid
1767 | LogicalType::Interval => 16,
1768 LogicalType::BigInt
1769 | LogicalType::UBigInt
1770 | LogicalType::Timestamp
1771 | LogicalType::Time
1772 | LogicalType::TimeTz
1773 | LogicalType::TimestampTz
1774 | LogicalType::TimestampS
1775 | LogicalType::TimestampMs
1776 | LogicalType::TimestampNs
1777 | LogicalType::Double
1778 | LogicalType::Decimal { .. } => 8,
1779 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
1780 LogicalType::SmallInt | LogicalType::USmallInt => 2,
1781 _ => 1,
1782 }
1783}
1784
1785pub const STRIPE_PARTS: usize = 64;
1792
1793const DICTIONARY_DECIDE_ROWS: usize = 4_096;
1801
1802const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
1818
1819const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
1821
1822fn index_section(parts: usize) -> Result<usize> {
1824 parts
1825 .checked_mul(INDEX_ENTRY)
1826 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
1827 .ok_or_else(|| invalid("index page length overflow"))
1828}
1829
1830impl Writer {
1831 pub fn open(
1850 path: impl AsRef<Path>,
1851 name: impl Into<String>,
1852 fields: Vec<Field>,
1853 ) -> Result<Self> {
1854 for field in &fields {
1855 type_tag(&field.ty)?;
1856 }
1857 let name = name.into();
1858 let path = path.as_ref();
1859 let (_, size, slot, bytes, _) = slot_bytes(path)?;
1860 let (mut closed, views) = decode_catalog(&bytes, size)?;
1861 if let Some(at) = closed.iter().position(|held| held.name == name) {
1872 if closed[at].rows > 0 {
1873 return Err(invalid("two tables in one native file have the same name"));
1874 }
1875 closed.remove(at);
1876 }
1877 let generation = slot
1882 .generation
1883 .checked_add(1)
1884 .ok_or_else(|| invalid("native file generation overflow"))?;
1885 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1886 Ok(Self {
1887 file,
1888 at: size,
1891 written_back: size,
1892 dictionaries: fields
1893 .iter()
1894 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1895 .collect(),
1896 coded: fields
1897 .iter()
1898 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1899 .collect(),
1900 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
1901 lent: None,
1902 table: Table {
1903 name,
1904 dictionaries: vec![None; fields.len()],
1905 dictionary_payloads: Vec::new(),
1906 distincts: vec![None; fields.len()],
1907 fields,
1908 stripes: Vec::new(),
1909 rows: 0,
1910 frequencies: Vec::new(),
1911 pair_frequencies: Vec::new(),
1912 frequency_texts: Vec::new(),
1913 host_groups: None,
1914 clustering: None,
1915 generation,
1916 sections: Vec::new(),
1917 },
1918 generation,
1919 order: Vec::new(),
1920 next_order: 0,
1921 pending: Vec::with_capacity(STRIPE_PARTS),
1922 closed,
1923 views,
1924 profile: None,
1925 })
1926 }
1927
1928 pub fn create(
1934 path: impl AsRef<Path>,
1935 name: impl Into<String>,
1936 fields: Vec<Field>,
1937 ) -> Result<Self> {
1938 for field in &fields {
1939 type_tag(&field.ty)?;
1940 }
1941 let file =
1942 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1943 let mut header = [0; HEADER as usize];
1944 header[..8].copy_from_slice(MAGIC);
1945 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1946 write_at(&file, 0, &header)?;
1947 Ok(Self {
1948 file,
1949 at: HEADER,
1950 written_back: HEADER,
1951 dictionaries: fields
1952 .iter()
1953 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1954 .collect(),
1955 coded: fields
1956 .iter()
1957 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
1958 .collect(),
1959 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, 1)).collect(),
1960 lent: None,
1961 table: Table {
1962 name: name.into(),
1963 dictionaries: vec![None; fields.len()],
1964 dictionary_payloads: Vec::new(),
1965 distincts: vec![None; fields.len()],
1966 fields,
1967 stripes: Vec::new(),
1968 rows: 0,
1969 frequencies: Vec::new(),
1970 pair_frequencies: Vec::new(),
1971 frequency_texts: Vec::new(),
1972 host_groups: None,
1973 clustering: None,
1974 generation: 1,
1975 sections: Vec::new(),
1976 },
1977 generation: 1,
1978 order: Vec::new(),
1979 next_order: 0,
1980 pending: Vec::with_capacity(STRIPE_PARTS),
1981 closed: Vec::new(),
1982 views: Vec::new(),
1983 profile: None,
1984 })
1985 }
1986
1987 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
2009 let file =
2010 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
2011 let mut header = [0; HEADER as usize];
2012 header[..8].copy_from_slice(MAGIC);
2013 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
2014 write_at(&file, 0, &header)?;
2015 let catalog = encode_catalog(&[], views)?;
2016 write_at(&file, HEADER, &catalog)?;
2017 file.sync_all().map_err(io)?;
2021 let slot = Slot {
2022 offset: HEADER,
2023 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2024 generation: 1,
2025 hash: checksum(&catalog),
2026 };
2027 write_at(&file, slot_offset(1), &slot.bytes())?;
2028 file.sync_all().map_err(io)?;
2029 Ok(())
2030 }
2031
2032 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
2043 for field in &fields {
2044 type_tag(&field.ty)?;
2045 }
2046 let name = name.into();
2047 let entry = self.close()?;
2048 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
2049 return Err(invalid("two tables in one native file have the same name"));
2050 }
2051 let Self { file, at, generation, mut closed, views, .. } = self;
2052 closed.push(entry);
2053 Ok(Self {
2054 file,
2055 written_back: at,
2056 at,
2057 generation,
2058 closed,
2059 views,
2060 profile: None,
2061 dictionaries: fields
2062 .iter()
2063 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
2064 .collect(),
2065 coded: fields
2066 .iter()
2067 .map(|field| AtomicBool::new(field.ty == LogicalType::Varchar))
2068 .collect(),
2069 gathers: fields.iter().map(|field| stats::Gather::new(&field.ty, generation)).collect(),
2070 lent: None,
2071 table: Table {
2072 name,
2073 dictionaries: vec![None; fields.len()],
2074 dictionary_payloads: Vec::new(),
2075 distincts: vec![None; fields.len()],
2076 fields,
2077 stripes: Vec::new(),
2078 rows: 0,
2079 frequencies: Vec::new(),
2080 pair_frequencies: Vec::new(),
2081 frequency_texts: Vec::new(),
2082 host_groups: None,
2083 clustering: None,
2084 generation,
2085 sections: Vec::new(),
2086 },
2087 order: Vec::new(),
2088 next_order: 0,
2089 pending: Vec::with_capacity(STRIPE_PARTS),
2090 })
2091 }
2092
2093 #[must_use]
2103 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
2104 self.views = views;
2105 self
2106 }
2107
2108 #[must_use]
2114 pub fn with_profile(mut self, profile: Arc<LoadProfile>) -> Self {
2115 self.profile = Some(profile);
2116 self
2117 }
2118
2119 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
2134 self.table.clustering = Some(Clustering::new(
2137 clustering.columns().to_vec(),
2138 clustering.width(),
2139 &self.table.fields,
2140 )?);
2141 Ok(self)
2142 }
2143
2144 fn put(&mut self, bytes: &[u8]) -> Result<()> {
2149 write_at(&self.file, self.at, bytes)?;
2150 self.at = self
2151 .at
2152 .checked_add(bytes.len() as u64)
2153 .ok_or_else(|| invalid("native file length overflow"))?;
2154 if self.at - self.written_back >= WRITEBACK_STRETCH {
2155 rudb_io::start_writeback(&self.file, self.written_back, self.at - self.written_back);
2156 self.written_back = self.at;
2157 }
2158 Ok(())
2159 }
2160
2161 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
2167 let order = (self.next_order, 0);
2168 self.next_order = self.next_order.saturating_add(1);
2169 self.append_at(order, chunk)
2170 }
2171
2172 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
2183 if chunk.is_empty() {
2184 return Ok(());
2185 }
2186 self.admit(chunk)?;
2187 if self.pending.last().is_some_and(|last| last.order > order) {
2188 self.flush_pending()?;
2189 }
2190 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
2195 if self.pending.len() == STRIPE_PARTS {
2196 self.flush_pending()?;
2197 }
2198 Ok(())
2199 }
2200
2201 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
2217 if parts.len() > STRIPE_PARTS {
2218 return Err(invalid("a stripe was handed more parts than it holds"));
2219 }
2220 self.flush_pending()?;
2223 for (order, chunk) in parts {
2224 if chunk.is_empty() {
2225 continue;
2226 }
2227 self.admit(&chunk)?;
2228 self.pending.push(PendingChunk { order, chunk });
2229 }
2230 self.flush_pending()
2231 }
2232
2233 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
2235 if chunk.width() != self.table.fields.len() {
2236 return Err(invalid("chunk width differs from table schema"));
2237 }
2238 for (index, field) in self.table.fields.iter().enumerate() {
2239 if chunk.column(index)?.logical_type() != &field.ty {
2240 return Err(invalid("chunk type differs from table schema"));
2241 }
2242 }
2243 self.table.rows = self
2244 .table
2245 .rows
2246 .checked_add(chunk.len())
2247 .ok_or_else(|| invalid("row count overflow"))?;
2248 Ok(())
2249 }
2250
2251 fn encode_pages(columns: &[&Vector]) -> Result<ColumnStripe> {
2253 let mut stripe = ColumnStripe {
2254 pages: Vec::with_capacity(columns.len()),
2255 codes: Vec::with_capacity(columns.len()),
2256 sieves: Vec::with_capacity(columns.len()),
2257 ranges: Vec::with_capacity(columns.len()),
2258 };
2259 let mut settling = Settling::default();
2260 for &column in columns {
2261 let bytes = encode(column, &mut settling)?;
2262 if bytes.len() > MAX_PAGE {
2263 return Err(invalid("column page exceeds the configured bound"));
2264 }
2265 let range = Range::of(column);
2268 let sieve =
2279 Sieve::of(column, &range, SIEVE_BUDGET).filter(|sieve| sieve.len() < bytes.len());
2280 stripe.pages.push(bytes);
2281 stripe.codes.push(None);
2282 stripe.sieves.push(sieve);
2283 stripe.ranges.push(range);
2284 }
2285 Ok(stripe)
2286 }
2287
2288 fn place_blocks(&mut self) -> Result<()> {
2293 if let Some(lent) = self.lent.clone() {
2294 return self.place_lent_blocks(&lent);
2295 }
2296 let mut dictionaries = std::mem::take(&mut self.dictionaries);
2297 let placed = dictionaries.iter_mut().flatten().try_for_each(|dictionary| {
2298 for block in std::mem::take(&mut dictionary.blocks) {
2299 let start = self.at;
2300 self.put(&block)?;
2301 dictionary.placed.push(Placed {
2302 start,
2303 length: block.len() as u64,
2304 hash: checksum(&block),
2305 });
2306 }
2307 Ok(())
2308 });
2309 self.dictionaries = dictionaries;
2310 placed
2311 }
2312
2313 fn place_lent_blocks(&mut self, lent: &Lent) -> Result<()> {
2319 for column in lent.columns() {
2320 let Ok(mut held) = column.try_lock() else { continue };
2321 let Some(dictionary) = held.dictionary.as_mut() else { continue };
2322 for block in std::mem::take(&mut dictionary.blocks) {
2323 let start = self.at;
2324 self.put(&block)?;
2325 dictionary.placed.push(Placed {
2326 start,
2327 length: block.len() as u64,
2328 hash: checksum(&block),
2329 });
2330 }
2331 }
2332 Ok(())
2333 }
2334
2335 fn reclaim(&mut self) -> Result<()> {
2339 let Some(lent) = self.lent.take() else { return Ok(()) };
2340 let (dictionaries, gathers) = lent.reclaim()?;
2341 self.dictionaries = dictionaries;
2342 self.gathers = gathers;
2343 Ok(())
2344 }
2345
2346 fn flush_pending(&mut self) -> Result<()> {
2351 if self.pending.is_empty() {
2352 return Ok(());
2353 }
2354 let held = std::mem::take(&mut self.pending);
2355 let prepared = self.preparer().prepare_held(held)?;
2356 let merged = self.merge_held(prepared)?;
2357 let paged = merged.pages()?;
2358 self.write_paged(paged)
2359 }
2360
2361 fn write_stripe(&mut self, held: &[Part], encoded: Vec<ColumnStripe>) -> Result<()> {
2363 let width = self.table.fields.len();
2364 let parts = held.len();
2365 if encoded.len() != width {
2366 return Err(Error::internal("a stripe came to the writer with the wrong columns"));
2367 }
2368 let profile = self.profile.clone();
2369 if let Some(profile) = &profile {
2370 let rows = held.iter().map(|part| part.rows as u64).sum();
2371 let raw = held.iter().map(|part| part.footprint as u64).sum();
2372 let pages =
2373 encoded.iter().flat_map(|stripe| &stripe.pages).map(|page| page.len() as u64).sum();
2374 profile.moved(Stage::Pages, raw, pages, rows);
2375 }
2376 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2379 let before = self.at;
2380 self.place_blocks()?;
2381 drop(timing);
2382 if let Some(profile) = &profile {
2383 profile.moved(Stage::Dictionary, 0, self.at - before, 0);
2384 }
2385 let timing = profile.as_deref().map(|profile| profile.span(Stage::Write));
2386 let before = self.at;
2387 let mut pages = Vec::with_capacity(width);
2388 let mut memberships = vec![None; width];
2389 let mut ranges = Vec::with_capacity(width);
2390 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
2391 for stripe in &encoded {
2392 let offset = self.at;
2393 let section = index.len();
2394 let mut length = 0_usize;
2395 for bytes in &stripe.pages {
2396 write_at(&self.file, self.at + length as u64, bytes)?;
2397 put_u32(
2398 &mut index,
2399 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
2400 );
2401 put_u64(&mut index, checksum(bytes));
2402 length = length
2403 .checked_add(bytes.len())
2404 .ok_or_else(|| invalid("column page length overflow"))?;
2405 }
2406 let hash = checksum(&index[section..]);
2407 put_u64(&mut index, hash);
2408 if length > MAX_PAGE {
2409 return Err(invalid("column page exceeds the configured bound"));
2410 }
2411 self.at = self
2412 .at
2413 .checked_add(length as u64)
2414 .ok_or_else(|| invalid("native file length overflow"))?;
2415 pages.push(Span {
2416 offset,
2417 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
2418 });
2419 ranges.push(merged_range(stripe.ranges.iter().cloned()));
2420 }
2421 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
2422 if stripe.codes.iter().all(Option::is_none) {
2423 continue;
2424 }
2425 let lists = stripe
2426 .codes
2427 .iter()
2428 .map(|codes| codes.clone().unwrap_or_default())
2429 .collect::<Vec<_>>();
2430 let bytes = encode_membership(&merged_codes(lists));
2431 let offset = self.at;
2432 self.put(&bytes)?;
2433 *membership = Some(Page {
2434 offset,
2435 length: u32::try_from(bytes.len())
2436 .map_err(|_| invalid("membership page length overflow"))?,
2437 hash: checksum(&bytes),
2438 });
2439 }
2440 let mut sieves = vec![None; width];
2441 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
2442 if stripe.sieves.iter().all(Option::is_none) {
2443 continue;
2444 }
2445 let bytes = encode_sieves(stripe.sieves.iter())?;
2446 let offset = self.at;
2447 self.put(&bytes)?;
2448 *page = Some(Page {
2449 offset,
2450 length: u32::try_from(bytes.len())
2451 .map_err(|_| invalid("sieve page length overflow"))?,
2452 hash: checksum(&bytes),
2453 });
2454 }
2455 let mut part_ranges = vec![None; width];
2461 if parts > 1 {
2462 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
2463 let bytes = encode_part_ranges(&stripe.ranges)?;
2464 if bytes.len() >= span.length as usize {
2465 continue;
2466 }
2467 let offset = self.at;
2468 self.put(&bytes)?;
2469 *page = Some(Page {
2470 offset,
2471 length: u32::try_from(bytes.len())
2472 .map_err(|_| invalid("part range page length overflow"))?,
2473 hash: checksum(&bytes),
2474 });
2475 }
2476 }
2477 let offset = self.at;
2478 self.put(&index)?;
2479 let index = Span {
2480 offset,
2481 length: u32::try_from(index.len())
2482 .map_err(|_| invalid("index page length overflow"))?,
2483 };
2484 let mut rows = 0_usize;
2485 let mut lengths = Vec::with_capacity(parts);
2486 let mut span = None;
2487 for part in held {
2488 rows = rows.checked_add(part.rows).ok_or_else(|| invalid("row count overflow"))?;
2489 lengths.push(u32::try_from(part.rows).map_err(|_| invalid("part row count overflow"))?);
2490 span = Some(span.map_or((part.order, part.order), |(first, _)| (first, part.order)));
2491 }
2492 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
2493 self.table.stripes.push(Stripe {
2494 rows,
2495 parts: lengths,
2496 index,
2497 pages,
2498 memberships: Pages::from_slots(memberships)?,
2499 sieves: Pages::from_slots(sieves)?,
2500 part_ranges: Pages::from_slots(part_ranges)?,
2501 zone: Zone::from_ranges(ranges),
2502 });
2503 drop(timing);
2504 if let Some(profile) = &profile {
2505 profile.moved(Stage::Write, 0, self.at - before, rows as u64);
2506 }
2507 Ok(())
2508 }
2509
2510 fn numeric_frequency(&self, column: usize) -> Result<(Option<FrequencySummary>, Option<u64>)> {
2526 let signed = match self.table.fields[column].ty {
2527 LogicalType::TinyInt
2528 | LogicalType::SmallInt
2529 | LogicalType::Integer
2530 | LogicalType::BigInt
2531 | LogicalType::Date
2532 | LogicalType::Timestamp => true,
2533 LogicalType::UTinyInt
2534 | LogicalType::USmallInt
2535 | LogicalType::UInteger
2536 | LogicalType::UBigInt => false,
2537 _ => return Ok((None, None)),
2538 };
2539 let value_of = |bits: Option<u64>| match bits {
2540 None => FrequencyValue::Null,
2541 Some(bits) if signed => FrequencyValue::Integer(i128::from(bits as i64)),
2542 Some(bits) => FrequencyValue::Integer(i128::from(bits)),
2543 };
2544 let mut first = Candidates::default();
2547 let mut distinct = distinct::ExactDistinct::new();
2548 let mut run = Run::default();
2549 self.visit_numeric(column, signed, |_, bits| {
2550 if let Some((bits, times)) = run.push(bits) {
2551 first.add(bits, times);
2552 }
2553 if run.times == 1 {
2554 if let Some(bits) = bits {
2555 distinct.insert(bits);
2556 }
2557 }
2558 })?;
2559 if let Some((bits, times)) = run.take() {
2560 first.add(bits, times);
2561 }
2562 let Candidates { counts: candidates, nulls, decrements } = first;
2563 let (exact, null_count) = if decrements == 0 {
2564 let exact = candidates
2565 .into_iter()
2566 .map(|(bits, count)| (bits, u64::from(count)))
2567 .collect::<FrequencyMap<_>>();
2568 (exact, (nulls != 0).then_some(u64::from(nulls)))
2569 } else {
2570 let mut lower = candidates.values().copied().collect::<Vec<_>>();
2571 if nulls != 0 {
2572 lower.push(nulls);
2573 }
2574 lower.sort_unstable_by(|left, right| right.cmp(left));
2575 if lower.len() < FREQUENCY_BUILD_RANK
2576 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
2577 {
2578 return Ok((None, distinct.count()));
2579 }
2580 let mut exact =
2581 candidates.into_keys().map(|bits| (bits, 0_u64)).collect::<FrequencyMap<_>>();
2582 let mut null_count = (nulls != 0).then_some(0_u64);
2583 let mut recount = |bits: Option<u64>, times: u32| {
2584 let held = match bits {
2585 Some(bits) => exact.get_mut(&bits),
2586 None => null_count.as_mut(),
2587 };
2588 if let Some(count) = held {
2589 *count = count.saturating_add(u64::from(times));
2590 }
2591 };
2592 let mut run = Run::default();
2593 self.visit_numeric(column, signed, |_, bits| {
2594 if let Some((bits, times)) = run.push(bits) {
2595 recount(bits, times);
2596 }
2597 })?;
2598 if let Some((bits, times)) = run.take() {
2599 recount(bits, times);
2600 }
2601 (exact, null_count)
2602 };
2603 let mut entries = exact
2604 .into_iter()
2605 .map(|(bits, count)| FrequencyEntry { value: value_of(Some(bits)), count })
2606 .chain(null_count.map(|count| FrequencyEntry { value: FrequencyValue::Null, count }))
2607 .collect::<Vec<_>>();
2608 let omitted_max = keep_most_frequent(&mut entries).max(decrements);
2609 let kept_rows = entries.iter().try_fold(0_u64, |total, entry| {
2610 total.checked_add(entry.count).filter(|&total| total <= FREQUENCY_ORDINALS as u64)
2611 });
2612 let mut ordinals = Vec::new();
2613 let mut ordinal_entries = Vec::new();
2614 if let Some(kept_rows) = kept_rows {
2615 let mut kept = FrequencyMap::default();
2616 let mut null_kept = None;
2617 for (at, entry) in entries.iter().enumerate() {
2618 let at = u16::try_from(at)
2619 .map_err(|_| invalid("too many retained frequency entries"))?;
2620 match entry.value {
2621 FrequencyValue::Integer(value) => {
2622 kept.insert(value as u64, at);
2623 }
2624 FrequencyValue::Null => null_kept = Some(at),
2625 FrequencyValue::Code(_) => {}
2626 }
2627 }
2628 ordinals.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2629 ordinal_entries.reserve(usize::try_from(kept_rows).unwrap_or(FREQUENCY_ORDINALS));
2630 self.visit_numeric(column, signed, |ordinal, bits| {
2631 let held = match bits {
2632 Some(bits) => kept.get(&bits).copied(),
2633 None => null_kept,
2634 };
2635 if let Some(entry) = held {
2636 ordinals.push(ordinal);
2637 ordinal_entries.push(entry);
2638 }
2639 })?;
2640 }
2641 Ok((
2642 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries }),
2643 distinct.count(),
2644 ))
2645 }
2646
2647 fn visit_numeric(
2654 &self,
2655 column: usize,
2656 signed: bool,
2657 mut visit: impl FnMut(u64, Option<u64>),
2658 ) -> Result<()> {
2659 let ty = &self.table.fields[column].ty;
2660 let mut start = 0_u64;
2661 let mut block = Vec::new();
2662 for stripe in &self.table.stripes {
2663 let spans = read_index(&self.file, stripe, column)?;
2664 let page = stripe.pages[column];
2665 let mut bytes = vec![0; page.length as usize];
2666 read_at(&self.file, page.offset, &mut bytes)?;
2667 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2668 let part = part_bytes(&bytes, *span)?;
2669 if checksum(part) != span.hash {
2670 return Err(invalid("column page checksum differs while building frequencies"));
2671 }
2672 let rows = rows as usize;
2673 let vector = decode(ty, rows, part, None)?;
2674 if signed && vector.signed_block(&mut block) && block.len() == rows {
2678 if vector.none_null() {
2679 for (row, &value) in block.iter().enumerate() {
2680 visit(start.saturating_add(row as u64), Some(value as u64));
2681 }
2682 } else {
2683 for (row, &value) in block.iter().enumerate() {
2684 let bits = (!vector.is_null_at(row)).then_some(value as u64);
2685 visit(start.saturating_add(row as u64), bits);
2686 }
2687 }
2688 start = start.saturating_add(rows as u64);
2689 continue;
2690 }
2691 for row in 0..rows {
2693 let bits = if vector.is_null_at(row) {
2694 None
2695 } else {
2696 let widened = match vector.signed_at(row) {
2700 Some(value) => Some(value as u64),
2701 None => match vector.value_at(row) {
2702 Value::UTinyInt(value) => Some(u64::from(value)),
2703 Value::USmallInt(value) => Some(u64::from(value)),
2704 Value::UInteger(value) => Some(u64::from(value)),
2705 Value::UBigInt(value) => Some(value),
2706 _ => None,
2707 },
2708 };
2709 Some(widened.ok_or_else(|| {
2710 invalid("numeric frequency page did not contain an integer value")
2711 })?)
2712 };
2713 visit(start.saturating_add(row as u64), bits);
2714 }
2715 start = start.saturating_add(rows as u64);
2716 }
2717 }
2718 Ok(())
2719 }
2720
2721 fn numeric_frequencies(&self) -> Result<Vec<(Option<FrequencySummary>, Option<u64>)>> {
2729 let mut columns = self
2730 .table
2731 .fields
2732 .iter()
2733 .enumerate()
2734 .filter_map(|(column, field)| {
2735 matches!(
2736 field.ty,
2737 LogicalType::TinyInt
2738 | LogicalType::SmallInt
2739 | LogicalType::Integer
2740 | LogicalType::BigInt
2741 | LogicalType::UTinyInt
2742 | LogicalType::USmallInt
2743 | LogicalType::UInteger
2744 | LogicalType::UBigInt
2745 | LogicalType::Date
2746 | LogicalType::Timestamp
2747 )
2748 .then_some(column)
2749 })
2750 .collect::<Vec<_>>();
2751 let workers = std::thread::available_parallelism()
2752 .map_or(1, usize::from)
2753 .min(MAX_FREQUENCY_WORKERS)
2754 .min(columns.len());
2755 let profile = self.profile.as_deref();
2756 if workers <= 1 {
2757 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2758 let mut frequencies = vec![(None, None); self.table.fields.len()];
2759 for column in columns {
2760 frequencies[column] = self.numeric_frequency(column)?;
2761 }
2762 return Ok(frequencies);
2763 }
2764 columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
2767 let queue = Mutex::new(columns);
2768 let pieces = std::thread::scope(|scope| {
2769 (0..workers)
2770 .map(|_| {
2771 scope.spawn(|| {
2772 let _timing = profile.map(|profile| profile.span(Stage::Publish));
2773 let mut mine = Vec::new();
2774 loop {
2775 let taken = queue
2776 .lock()
2777 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2778 .pop();
2779 let Some(column) = taken else { break };
2780 mine.push((column, self.numeric_frequency(column)?));
2781 }
2782 Ok(mine)
2783 })
2784 })
2785 .collect::<Vec<_>>()
2786 .into_iter()
2787 .map(|handle| {
2788 handle
2789 .join()
2790 .map_err(|_| Error::internal("a native frequency worker panicked"))?
2791 })
2792 .collect::<Result<Vec<_>>>()
2793 })?;
2794 let mut frequencies = vec![(None, None); self.table.fields.len()];
2795 for piece in pieces {
2796 for (column, summary) in piece {
2797 frequencies[column] = summary;
2798 }
2799 }
2800 Ok(frequencies)
2801 }
2802
2803 #[allow(dead_code)]
2805 fn stable_codes_at(&self, column: usize, ordinals: &[u64]) -> Result<Option<Vec<Option<u32>>>> {
2806 if self.dictionaries.get(column).and_then(Option::as_ref).is_none() {
2807 return Ok(None);
2808 }
2809 if ordinals.windows(2).any(|pair| pair[0] >= pair[1]) {
2810 return Err(invalid("frequency ordinals are not sorted and unique"));
2811 }
2812 let mut out = Vec::with_capacity(ordinals.len());
2813 let mut wanted = 0;
2814 let mut stripe_start = 0_u64;
2815 for stripe in &self.table.stripes {
2816 let stripe_end = stripe_start.saturating_add(stripe.rows as u64);
2817 if wanted == ordinals.len() || ordinals[wanted] >= stripe_end {
2818 stripe_start = stripe_end;
2819 continue;
2820 }
2821 let spans = read_index(&self.file, stripe, column)?;
2822 let page = stripe.pages[column];
2823 let mut bytes = vec![0; page.length as usize];
2824 read_at(&self.file, page.offset, &mut bytes)?;
2825 let mut part_start = stripe_start;
2826 for (span, &rows) in spans.iter().zip(&stripe.parts) {
2827 let part_end = part_start.saturating_add(u64::from(rows));
2828 if wanted < ordinals.len() && ordinals[wanted] < part_end {
2829 let part = part_bytes(&bytes, *span)?;
2830 if checksum(part) != span.hash {
2831 return Err(invalid(
2832 "column page checksum differs while building pair frequencies",
2833 ));
2834 }
2835 let upto = ordinals.partition_point(|&ordinal| ordinal < part_end);
2836 let positions = ordinals[wanted..upto]
2837 .iter()
2838 .map(|&ordinal| {
2839 usize::try_from(ordinal.saturating_sub(part_start))
2840 .map_err(|_| invalid("frequency row offset does not fit in memory"))
2841 })
2842 .collect::<Result<Vec<_>>>()?;
2843 if !decode_selected_stable_codes(rows as usize, part, &positions, &mut out)? {
2844 return Ok(None);
2845 }
2846 wanted = upto;
2847 }
2848 part_start = part_end;
2849 }
2850 stripe_start = stripe_end;
2851 }
2852 if wanted != ordinals.len() {
2853 return Err(invalid("frequency ordinal is outside the table"));
2854 }
2855 Ok(Some(out))
2856 }
2857
2858 #[allow(dead_code)]
2860 fn pair_frequencies(
2861 &self,
2862 frequencies: &[Option<Frequencies>],
2863 ) -> Result<Vec<PairFrequencySummary>> {
2864 let anchors = frequencies
2865 .iter()
2866 .enumerate()
2867 .filter_map(|(column, summary)| {
2868 match summary {
2870 Some(Frequencies::Held(summary)) => Some(summary),
2871 _ => None,
2872 }
2873 .filter(|summary| {
2874 !summary.ordinals.is_empty()
2875 && summary.ordinal_entries.len() == summary.ordinals.len()
2876 })
2877 .cloned()
2878 .map(|summary| (column, summary))
2879 })
2880 .collect::<Vec<_>>();
2881 let strings = self
2882 .dictionaries
2883 .iter()
2884 .enumerate()
2885 .filter_map(|(column, dictionary)| dictionary.as_ref().map(|_| column))
2886 .collect::<Vec<_>>();
2887 let mut summaries = Vec::new();
2888 for (first, anchors) in anchors {
2889 for &second in &strings {
2890 if summaries.len() == MAX_PAIR_FREQUENCIES {
2891 return Ok(summaries);
2892 }
2893 let Some(codes) = self.stable_codes_at(second, &anchors.ordinals)? else {
2894 continue;
2895 };
2896 if codes.len() != anchors.ordinal_entries.len() {
2897 return Err(invalid("pair frequency columns have different lengths"));
2898 }
2899 let mut counts = HashMap::<(u16, Option<u32>), u64>::new();
2900 for (&anchor, code) in anchors.ordinal_entries.iter().zip(codes) {
2901 *counts.entry((anchor, code)).or_default() += 1;
2902 }
2903 let mut entries = counts
2904 .into_iter()
2905 .map(|((first_entry, second), count)| PairFrequencyEntry {
2906 first_entry,
2907 second,
2908 count,
2909 })
2910 .collect::<Vec<_>>();
2911 entries.sort_unstable_by(|left, right| {
2912 right
2913 .count
2914 .cmp(&left.count)
2915 .then_with(|| left.first_entry.cmp(&right.first_entry))
2916 .then_with(|| left.second.cmp(&right.second))
2917 });
2918 let pair_omitted = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
2919 entries.truncate(FREQUENCY_ENTRIES);
2920 summaries.push(PairFrequencySummary {
2921 first: u16::try_from(first)
2922 .map_err(|_| invalid("pair frequency column index overflows"))?,
2923 second: u16::try_from(second)
2924 .map_err(|_| invalid("pair frequency column index overflows"))?,
2925 entries,
2926 omitted_max: anchors.omitted_max.max(pair_omitted),
2927 });
2928 }
2929 }
2930 Ok(summaries)
2931 }
2932
2933 fn close(&mut self) -> Result<Entry> {
2944 self.reclaim()?;
2945 self.flush_pending()?;
2946 let profile = self.profile.clone();
2950 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
2951 let before = self.at;
2952 let mut stripes = std::mem::take(&mut self.order)
2953 .into_iter()
2954 .zip(std::mem::take(&mut self.table.stripes))
2955 .collect::<Vec<_>>();
2956 stripes.sort_by_key(|(order, _)| order.0);
2957 let mut previous: Option<(u64, u64)> = None;
2958 for ((first, last), _) in &stripes {
2959 if previous.is_some_and(|previous| previous >= *first) {
2960 return Err(invalid("chunks did not arrive in source order"));
2961 }
2962 previous = Some(*last);
2963 }
2964 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
2965 drop(timing);
2966 let timing = profile.as_deref().map(|profile| profile.span(Stage::Dictionary));
2967 let placing = self.at;
2968 finish_dictionaries(&mut self.dictionaries)?;
2969 self.place_blocks()?;
2970 let this = &*self;
2977 let (numeric, closed) = std::thread::scope(|scope| {
2978 let numeric = scope.spawn(|| {
2979 let (frequencies, distincts): (Vec<Option<FrequencySummary>>, Vec<_>) =
2980 this.numeric_frequencies()?.into_iter().unzip();
2981 let frequencies = frequencies
2982 .into_iter()
2983 .map(|held| held.map(Frequencies::Held))
2984 .collect::<Vec<_>>();
2985 let pairs = Vec::new();
2987 Ok::<_, Error>((frequencies, distincts, pairs))
2988 });
2989 let closed = this.close_dictionaries();
2990 let numeric =
2991 numeric.join().map_err(|_| Error::internal("the native frequency thread panicked"));
2992 (numeric, closed)
2993 });
2994 let (frequencies, distincts, pairs) = numeric??;
2995 let closed = closed?;
2996 self.table.frequencies = frequencies;
2997 self.table.distincts = distincts;
2998 self.table.pair_frequencies = pairs;
2999 self.dictionaries = Vec::new();
3000 self.table.dictionary_payloads = vec![0; self.table.fields.len()];
3001 self.table.frequency_texts = vec![Vec::new(); self.table.fields.len()];
3002 self.table.host_groups = None;
3003 for (index, closed) in closed.into_iter().enumerate() {
3004 let Some(closed) = closed else { continue };
3005 let ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload } = closed;
3006 self.table.distincts[index] = Some(distinct);
3007 self.table.frequencies[index] = Some(Frequencies::Held(frequencies));
3008 self.table.frequency_texts[index] = texts;
3009 if hosts.is_some() {
3010 self.table.host_groups = hosts;
3011 }
3012 let offset = self.at;
3013 self.put(&encoded.index)?;
3014 self.put(&encoded.ranks)?;
3015 self.put(&encoded.grams)?;
3016 self.table.dictionary_payloads[index] = payload;
3017 let length = encoded
3018 .index
3019 .len()
3020 .checked_add(encoded.ranks.len())
3021 .and_then(|len| len.checked_add(encoded.grams.len()))
3022 .ok_or_else(|| invalid("dictionary page length overflow"))?;
3023 self.table.dictionaries[index] = Some(Page {
3024 offset,
3025 length: u32::try_from(length)
3026 .map_err(|_| invalid("dictionary page length overflow"))?,
3027 hash: checksum(&encoded.index),
3028 });
3029 }
3030 drop(timing);
3031 let timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3032 let placed = self.at - placing;
3033 self.write_stats()?;
3034 let directory = encode_directory(&self.table)?;
3035 if directory.len() > MAX_DIRECTORY {
3036 return Err(invalid("directory exceeds the configured bound"));
3037 }
3038 let offset = self.at;
3039 self.put(&directory)?;
3040 drop(timing);
3041 if let Some(profile) = &profile {
3042 profile.moved(Stage::Dictionary, 0, placed, 0);
3043 profile.moved(Stage::Publish, 0, self.at - before - placed, 0);
3044 }
3045 Ok(Entry {
3046 name: self.table.name.clone(),
3047 fields: self.table.fields.clone(),
3048 rows: self.table.rows,
3049 nonzero: table_nonzero_counts(&self.table),
3050 aggregates: table_aggregate_sums(&self.table),
3051 distincts: self.table.distincts.clone(),
3052 extremes: table_integer_extremes(&self.table),
3053 frequencies: table_complete_numeric_frequencies(&self.table),
3054 directory: Page {
3055 offset,
3056 length: u32::try_from(directory.len())
3057 .map_err(|_| invalid("directory length overflow"))?,
3058 hash: checksum(&directory),
3059 },
3060 })
3061 }
3062
3063 fn close_dictionaries(&self) -> Result<Vec<Option<ClosedDictionary>>> {
3070 let mut jobs = self
3071 .dictionaries
3072 .iter()
3073 .enumerate()
3074 .filter_map(|(index, dictionary)| {
3075 dictionary
3076 .as_ref()
3077 .map(|dictionary| (index, dictionary, dictionary.closing_bytes()))
3078 })
3079 .collect::<Vec<_>>();
3080 jobs.sort_by_key(|&(_, _, bytes)| bytes);
3081 let mut closed = (0..self.dictionaries.len()).map(|_| None).collect::<Vec<_>>();
3082 let workers = close_workers().min(jobs.len());
3083 if workers <= 1 {
3084 for (index, dictionary, _) in jobs {
3085 closed[index] = Some(self.close_dictionary(index, dictionary)?);
3086 }
3087 return Ok(closed);
3088 }
3089 let state = Mutex::new((jobs, 0_usize));
3091 let finished = Condvar::new();
3092 let profile = self.profile.as_deref();
3093 let pieces = std::thread::scope(|scope| {
3094 (0..workers)
3095 .map(|_| {
3096 scope.spawn(|| {
3097 let _timing = profile.map(|profile| profile.span(Stage::Dictionary));
3098 let mut mine = Vec::new();
3099 loop {
3100 let mut held = state.lock().map_err(|_| {
3101 Error::internal("a native dictionary worker panicked")
3102 })?;
3103 let (index, dictionary, bytes) = loop {
3104 let (jobs, busy) = &mut *held;
3105 if jobs.is_empty() {
3106 return Ok(mine);
3107 }
3108 let fits = jobs.iter().rposition(|&(_, _, bytes)| {
3109 *busy == 0
3110 || busy.saturating_add(bytes) <= CLOSE_DICTIONARY_BYTES
3111 });
3112 if let Some(at) = fits {
3113 let job = jobs.remove(at);
3114 *busy += job.2;
3115 break job;
3116 }
3117 held = finished.wait(held).map_err(|_| {
3118 Error::internal("a native dictionary worker panicked")
3119 })?;
3120 };
3121 drop(held);
3122 let _room = Room { state: &state, finished: &finished, bytes };
3125 mine.push((index, self.close_dictionary(index, dictionary)?));
3126 }
3127 })
3128 })
3129 .collect::<Vec<_>>()
3130 .into_iter()
3131 .map(|handle| {
3132 handle
3133 .join()
3134 .map_err(|_| Error::internal("a native dictionary worker panicked"))?
3135 })
3136 .collect::<Result<Vec<_>>>()
3137 })?;
3138 for (index, one) in pieces.into_iter().flatten() {
3139 closed[index] = Some(one);
3140 }
3141 Ok(closed)
3142 }
3143
3144 fn close_dictionary(
3151 &self,
3152 _index: usize,
3153 dictionary: &GlobalDictionary,
3154 ) -> Result<ClosedDictionary> {
3155 let (order, flat, bases) = dictionary.ranked_with_values(Some(&self.file))?;
3156 let distinct = dictionary.counts.iter().filter(|count| **count != 0).count() as u64;
3160 let (frequencies, texts) = code_frequency(dictionary, &flat, &bases)?;
3161 let hosts = None;
3163 drop(flat);
3164 drop(bases);
3165 let encoded = encode_global_dictionary(dictionary, &order, &dictionary.placed, true)?;
3166 let payload = dictionary
3167 .placed
3168 .iter()
3169 .try_fold(0_u64, |sum, place| sum.checked_add(place.length))
3170 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
3171 Ok(ClosedDictionary { distinct, frequencies, texts, hosts, encoded, payload })
3172 }
3173
3174 fn write_stats(&mut self) -> Result<()> {
3186 let gathers = std::mem::take(&mut self.gathers);
3187 let rows = self.table.rows as u64;
3188 let mut payloads = Vec::new();
3189 for (column, gather) in gathers.into_iter().enumerate() {
3190 let Some(gather) = gather else { continue };
3191 if gather.rows() != rows {
3197 continue;
3198 }
3199 let Some(stats) = gather.finish() else { continue };
3200 let mut summary = Vec::new();
3201 stats.summary.encode(&mut summary)?;
3202 let mut sketches = Vec::new();
3203 stats.sketches.encode(&mut sketches)?;
3204 payloads.push((column, summary, sketches));
3205 }
3206 if payloads.is_empty() {
3207 return Ok(());
3208 }
3209 let costs = payloads
3210 .iter()
3211 .map(|(_, summary, sketches)| summary.len() + sketches.len())
3212 .collect::<Vec<_>>();
3213 let allowance = stats::allowance(stats::column_bytes(&self.table), stats::BUDGET_SHARE);
3214 let keep = stats::within(&costs, allowance, 0);
3217 for ((column, summary, sketches), _) in
3218 payloads.iter().zip(&keep).filter(|&(_, &keep)| keep)
3219 {
3220 let id = u64::try_from(*column).map_err(|_| invalid("column index overflow"))?;
3221 for (kind, bytes, header_bytes) in [
3222 (*section::SUMMARY, summary, summary.len() as u32),
3225 (*section::SKETCHES, sketches, rudb_stats::sketches::HEADER_BYTES),
3226 ] {
3227 let written = write_section(
3228 &self.file,
3229 &mut self.at,
3230 §ion::Attachment { kind, id, flags: 0, header_bytes, bytes },
3231 self.generation,
3232 )?;
3233 self.table.sections.push(written);
3234 }
3235 }
3236 if self.table.sections.len() > MAX_SECTIONS {
3237 return Err(invalid("the table would name more sections than the bound allows"));
3238 }
3239 Ok(())
3240 }
3241
3242 pub fn finish(mut self) -> Result<Table> {
3252 let entry = self.close()?;
3253 let profile = self.profile.take();
3254 let _timing = profile.as_deref().map(|profile| profile.span(Stage::Publish));
3255 let mut tables = std::mem::take(&mut self.closed);
3256 tables.push(entry);
3257 let catalog = encode_catalog(&tables, &self.views)?;
3258 if catalog.len() > MAX_DIRECTORY {
3259 return Err(invalid("catalog exceeds the configured bound"));
3260 }
3261 let offset = self.at;
3262 self.put(&catalog)?;
3263 if let Some(profile) = &profile {
3264 profile.moved(Stage::Publish, 0, catalog.len() as u64, 0);
3265 }
3266 synced(&self.file, profile.as_deref())?;
3270 let slot = Slot {
3271 offset,
3272 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3273 generation: self.generation,
3274 hash: checksum(&catalog),
3275 };
3276 write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
3281 synced(&self.file, profile.as_deref())?;
3282 Ok(self.table)
3283 }
3284
3285 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
3302 let path = path.as_ref();
3303 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3304 let (closed, _) = decode_catalog(&bytes, size)?;
3305 let generation = slot
3306 .generation
3307 .checked_add(1)
3308 .ok_or_else(|| invalid("native file generation overflow"))?;
3309 let catalog = encode_catalog(&closed, views)?;
3310 if catalog.len() > MAX_DIRECTORY {
3311 return Err(invalid("catalog exceeds the configured bound"));
3312 }
3313 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3314 write_at(&file, size, &catalog)?;
3315 file.sync_all().map_err(io)?;
3316 let slot = Slot {
3317 offset: size,
3318 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3319 generation,
3320 hash: checksum(&catalog),
3321 };
3322 write_at(&file, slot_offset(generation), &slot.bytes())?;
3323 file.sync_all().map_err(io)?;
3324 Ok(())
3325 }
3326
3327 pub fn certify_summaries(path: impl AsRef<Path>) -> Result<()> {
3330 let path = path.as_ref();
3331 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3332 let (mut entries, views) = decode_catalog(&bytes, size)?;
3333 let native = Catalog::open(path)?;
3334 for entry in &mut entries {
3335 let reader = native.table(&entry.name)?;
3336 entry.nonzero = reader_nonzero_counts(&reader)?;
3337 entry.aggregates = reader_aggregate_sums(&reader)?;
3338 entry.distincts = (0..entry.fields.len())
3339 .map(|column| reader.distinct_values(column))
3340 .collect::<Result<Vec<_>>>()?;
3341 entry.extremes = reader_integer_extremes(&reader)?;
3342 entry.frequencies = reader_complete_numeric_frequencies(&reader)?;
3343 }
3344 let generation = slot
3345 .generation
3346 .checked_add(1)
3347 .ok_or_else(|| invalid("native file generation overflow"))?;
3348 let catalog = encode_catalog(&entries, &views)?;
3349 if catalog.len() > MAX_DIRECTORY {
3350 return Err(invalid("catalog exceeds the configured bound"));
3351 }
3352 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3353 write_at(&file, size, &catalog)?;
3354 file.sync_all().map_err(io)?;
3355 let slot = Slot {
3356 offset: size,
3357 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3358 generation,
3359 hash: checksum(&catalog),
3360 };
3361 write_at(&file, slot_offset(generation), &slot.bytes())?;
3362 file.sync_all().map_err(io)?;
3363 Ok(())
3364 }
3365
3366 pub fn certify_counts(path: impl AsRef<Path>) -> Result<()> {
3368 Self::certify_summaries(path)
3369 }
3370}
3371
3372fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
3378 let offset = *at;
3379 write_at(file, offset, bytes)?;
3380 *at =
3381 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
3382 Ok(offset)
3383}
3384
3385fn write_section(
3391 file: &File,
3392 at: &mut u64,
3393 one: §ion::Attachment<'_>,
3394 generation: u64,
3395) -> Result<Section> {
3396 if !one.bytes.is_empty() && one.header_bytes as usize > one.bytes.len() {
3400 return Err(invalid("a section's header is longer than its payload"));
3401 }
3402 let mut extents = Vec::new();
3403 let mut first = 0_u64;
3404 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
3405 let offset = append(file, at, chunk)?;
3406 extents.push(section::Extent {
3407 offset,
3408 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
3409 hash: checksum(chunk),
3410 first,
3411 });
3412 first += chunk.len() as u64;
3413 }
3414 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
3415 section::encode_extents(&extents, &mut table)?;
3416 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
3420 Ok(Section {
3421 kind: one.kind,
3422 id: one.id,
3423 generation,
3424 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
3425 extent_page,
3426 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
3427 hash: checksum(&table),
3428 flags: one.flags,
3429 header_bytes: one.header_bytes,
3430 })
3431}
3432
3433pub fn attach(
3457 path: impl AsRef<Path>,
3458 table: &str,
3459 attachments: &[section::Attachment<'_>],
3460) -> Result<Table> {
3461 let path = path.as_ref();
3462 let (_, size, slot, bytes, _) = slot_bytes(path)?;
3463 let (mut entries, views) = decode_catalog(&bytes, size)?;
3464 let at = entries
3465 .iter()
3466 .position(|entry| entry.name == table)
3467 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
3468 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
3469 let mut version = [0; 4];
3470 read_at(&file, 8, &mut version)?;
3471 let version = u32::from_le_bytes(version);
3472 if version != FORMAT {
3478 return Err(invalid(&format!(
3479 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
3480 to be written again"
3481 )));
3482 }
3483 let mut directory = vec![0; entries[at].directory.length as usize];
3484 read_at(&file, entries[at].directory.offset, &mut directory)?;
3485 if checksum(&directory) != entries[at].directory.hash {
3486 return Err(invalid(&format!("the directory of table {table} does not checksum")));
3487 }
3488 let mut held = decode_directory(&directory, size)?;
3489 let mut cursor = size;
3490 for one in attachments {
3491 let written = write_section(&file, &mut cursor, one, held.generation)?;
3492 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
3493 held.sections.push(written);
3494 }
3495 if held.sections.len() > MAX_SECTIONS {
3496 return Err(invalid("the table would name more sections than the bound allows"));
3497 }
3498 let encoded = encode_directory(&held)?;
3499 if encoded.len() > MAX_DIRECTORY {
3500 return Err(invalid("directory exceeds the configured bound"));
3501 }
3502 let offset = append(&file, &mut cursor, &encoded)?;
3503 entries[at].directory = Page {
3504 offset,
3505 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
3506 hash: checksum(&encoded),
3507 };
3508 let catalog = encode_catalog(&entries, &views)?;
3511 if catalog.len() > MAX_DIRECTORY {
3512 return Err(invalid("catalog exceeds the configured bound"));
3513 }
3514 let offset = append(&file, &mut cursor, &catalog)?;
3515 file.sync_all().map_err(io)?;
3516 let generation =
3517 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
3518 let committed = Slot {
3519 offset,
3520 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
3521 generation,
3522 hash: checksum(&catalog),
3523 };
3524 write_at(&file, slot_offset(generation), &committed.bytes())?;
3525 file.sync_all().map_err(io)?;
3526 Ok(held)
3527}
3528
3529type Synopsis = Arc<Vec<(Value, u64)>>;
3532
3533#[derive(Debug, Clone)]
3535pub struct Reader {
3536 file: Arc<File>,
3537 table: Arc<Table>,
3538 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
3539 loading: Arc<Vec<Mutex<()>>>,
3548 frequency_values: Arc<Vec<OnceLock<Synopsis>>>,
3551 frequency_summaries: Arc<Vec<OnceLock<Arc<FrequencySummary>>>>,
3555 opened: Arc<AtomicUsize>,
3559 sieves: Arc<Vec<Vec<SieveSlot>>>,
3563 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
3566 places: Arc<Vec<Place>>,
3568 cache: Arc<Shelf>,
3569 pool: PagePool,
3571 pages: Arc<AtomicUsize>,
3574 indexes: Arc<AtomicUsize>,
3577 size: u64,
3579 directory: u64,
3581 opening: Opening,
3583}
3584
3585#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3597pub struct Opening {
3598 pub reads: u32,
3601 pub bytes: u64,
3603}
3604
3605#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
3607pub struct Reads {
3608 pub opening: Opening,
3610 pub pages: usize,
3612 pub indexes: usize,
3614 pub dictionaries: usize,
3617}
3618
3619#[derive(Debug, Clone, Copy)]
3621struct Place {
3622 stripe: u32,
3623 part: u32,
3624 rows: u32,
3625}
3626
3627#[derive(Debug, Clone, Copy)]
3629struct PartSpan {
3630 start: usize,
3631 length: usize,
3632 hash: u64,
3633}
3634
3635#[derive(Debug, Clone)]
3641struct CachedColumn {
3642 stripe: usize,
3643 index: Arc<Vec<PartSpan>>,
3644 page: Option<Arc<Vec<u8>>>,
3645}
3646
3647#[derive(Debug, Default)]
3667struct Cached {
3668 pages: Vec<Option<Resident>>,
3669 loading: Vec<usize>,
3670 index: Vec<Option<Arc<Vec<PartSpan>>>>,
3671}
3672
3673#[derive(Debug, Clone)]
3675struct Resident {
3676 page: Arc<Vec<u8>>,
3677 used: Arc<AtomicBool>,
3678}
3679
3680#[derive(Debug)]
3682struct Shelf {
3683 columns: Vec<Mutex<Cached>>,
3684 held: Vec<AtomicUsize>,
3687 kept: AtomicUsize,
3690}
3691
3692#[derive(Debug, Clone, Default)]
3711pub struct PagePool {
3712 ring: Arc<Mutex<Ring>>,
3713 budget: Arc<AtomicUsize>,
3714}
3715
3716#[derive(Debug, Default)]
3717struct Ring {
3718 held: VecDeque<Held>,
3719 bytes: usize,
3720}
3721
3722#[derive(Debug)]
3727struct Held {
3728 shelf: Weak<Shelf>,
3729 column: usize,
3730 stripe: usize,
3731 bytes: usize,
3732 used: Arc<AtomicBool>,
3733}
3734
3735impl PagePool {
3736 #[must_use]
3738 pub fn new(budget: usize) -> Self {
3739 let pool = Self::default();
3740 pool.budget.store(budget, Atomic::Relaxed);
3741 pool
3742 }
3743
3744 #[must_use]
3750 pub fn bytes(&self) -> usize {
3751 self.ring.lock().map_or(0, |ring| ring.bytes)
3752 }
3753
3754 fn admit(&self, held: Held) {
3760 let budget = self.budget.load(Atomic::Relaxed);
3761 let mut gone = Vec::new();
3762 {
3763 let Ok(mut ring) = self.ring.lock() else { return };
3764 ring.bytes += held.bytes;
3765 ring.held.push_back(held);
3766 let mut looked = 0;
3769 let limit = ring.held.len();
3770 while ring.bytes > budget && looked < limit {
3771 looked += 1;
3772 let Some(entry) = ring.held.pop_front() else { break };
3773 let Some(shelf) = entry.shelf.upgrade() else {
3774 ring.bytes -= entry.bytes;
3775 continue;
3776 };
3777 if entry.used.swap(false, Atomic::Relaxed) {
3778 ring.held.push_back(entry);
3779 continue;
3780 }
3781 let count = &shelf.held[entry.column];
3782 if count.load(Atomic::Relaxed) <= shelf.kept.load(Atomic::Relaxed).max(1) {
3783 ring.held.push_back(entry);
3784 continue;
3785 }
3786 count.fetch_sub(1, Atomic::Relaxed);
3787 ring.bytes -= entry.bytes;
3788 gone.push((shelf, entry));
3789 }
3790 while ring.held.front().is_some_and(|entry| entry.shelf.strong_count() == 0) {
3793 if let Some(entry) = ring.held.pop_front() {
3794 ring.bytes -= entry.bytes;
3795 }
3796 }
3797 }
3798 for (shelf, entry) in gone {
3799 let Ok(mut cached) = shelf.columns[entry.column].lock() else { continue };
3800 if let Some(slot) = cached.pages.get_mut(entry.stripe) {
3801 if slot.as_ref().is_some_and(|slot| Arc::ptr_eq(&slot.used, &entry.used)) {
3802 *slot = None;
3803 }
3804 }
3805 }
3806 }
3807}
3808
3809const CACHED_STRIPES_PER_COLUMN: usize = 4;
3821
3822type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
3824
3825type RangeSlot = OnceLock<Arc<Vec<Range>>>;
3826
3827#[derive(Debug)]
3828struct NativeText {
3829 file: Arc<File>,
3830 values: usize,
3832 offsets: Vec<u8>,
3841 offset_bits: usize,
3844 value_ends: OnceLock<Option<Vec<u32>>>,
3857 value_lens: OnceLock<Option<Vec<u32>>>,
3867 ends_asked: AtomicUsize,
3873 ranks: usize,
3875 rank_at: u64,
3879 rank_ends: Vec<u64>,
3883 rank_hashes: Vec<u64>,
3884 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3885 code_bits: usize,
3888 code_ranks: OnceLock<Option<Vec<u32>>>,
3895 starts: Vec<u64>,
3902 lengths: Vec<u64>,
3903 hashes: Vec<u64>,
3904 grams: Option<NativeGrams>,
3906 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
3908 keep_budget: usize,
3911 payload_kept: AtomicUsize,
3919 swept: Vec<AtomicBool>,
3927 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
3944}
3945
3946#[derive(Debug)]
3947struct NativeGrams {
3948 start: u64,
3949 length: usize,
3950 width: usize,
3952 hash: u64,
3953 verdicts: Mutex<Vec<Verdict>>,
3960}
3961
3962type Verdict = (Vec<u8>, Arc<[bool]>);
3964
3965const GRAM_VERDICTS: usize = 8;
3967
3968impl NativeGrams {
3969 fn verdicts(&self, file: &File, literal: &[u8]) -> Result<Arc<[bool]>> {
3974 let mut held = self.verdicts.lock().map_err(|_| invalid("a poisoned signature verdict"))?;
3975 if let Some((_, verdict)) = held.iter().find(|(asked, _)| asked == literal) {
3976 return Ok(Arc::clone(verdict));
3977 }
3978 let wanted = literal.windows(4).map(|gram| gram_bits(gram, self.width)).collect::<Vec<_>>();
3979 let mut verdict = Vec::with_capacity(self.length / self.width);
3980 let window = GRAM_WINDOW / self.width * self.width;
3981 let hash = walk_checksummed(file, self.start, self.length, window, |bytes| {
3982 verdict.extend(bytes.chunks(self.width).map(|bits| {
3983 wanted
3984 .iter()
3985 .flatten()
3986 .all(|&bit| bits.get(bit / 8).is_some_and(|byte| byte & (1 << (bit % 8)) != 0))
3987 }));
3988 Ok(())
3989 })?;
3990 if hash != self.hash {
3991 return Err(invalid("global dictionary substring signatures checksum differs"));
3992 }
3993 let verdict: Arc<[bool]> = verdict.into();
3994 if held.len() >= GRAM_VERDICTS {
3995 held.remove(0);
3996 }
3997 held.push((literal.to_vec(), Arc::clone(&verdict)));
3998 Ok(verdict)
3999 }
4000
4001 fn footprint(&self) -> usize {
4002 self.verdicts.lock().map_or(0, |held| {
4003 held.iter().map(|(asked, verdict)| asked.capacity() + verdict.len()).sum()
4004 })
4005 }
4006}
4007
4008const TEXT_SEARCH_MEMO: usize = 64;
4013
4014const TEXT_PAYLOAD_VALUES: usize = 1024;
4030
4031const TEXT_GRAM_BYTES: usize = 8192;
4042
4043const NARROW_GRAM_BYTES: usize = 2048;
4045
4046const GRAM_WINDOW: usize = 256 << 10;
4048
4049fn gram_bits(bytes: &[u8], width: usize) -> [usize; 2] {
4052 let original = u32::from_le_bytes(bytes.try_into().expect("a four-byte gram"));
4053 let mut first = original ^ (original >> 16);
4054 first = first.wrapping_mul(0x7feb_352d);
4055 first ^= first >> 15;
4056 let mut second = original ^ (original >> 17);
4057 second = second.wrapping_mul(0x846c_a68b);
4058 second ^= second >> 16;
4059 let mask = width * 8 - 1;
4060 [(first as usize) & mask, (second as usize) & mask]
4061}
4062
4063const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
4084
4085fn lengths_of(ends: &[u32]) -> Option<Vec<u32>> {
4091 let mut lens = Vec::with_capacity(ends.len());
4092 for block in ends.chunks(TEXT_PAYLOAD_VALUES) {
4093 let mut start = 0;
4094 for &end in block {
4095 lens.push(end.checked_sub(start)?);
4096 start = end;
4097 }
4098 }
4099 Some(lens)
4100}
4101
4102const TEXT_OFFSET_RUN: usize = 512;
4109
4110const DICTIONARY_HEADER: usize = 16;
4113
4114const DICTIONARY_SCATTERED: u32 = 1 << 31;
4128const DICTIONARY_GRAMS: u32 = 1 << 30;
4130const DICTIONARY_WIDE_GRAMS: u32 = 1 << 29;
4133const DICTIONARY_FLAGS: u32 = DICTIONARY_SCATTERED | DICTIONARY_GRAMS | DICTIONARY_WIDE_GRAMS;
4135
4136const TEXT_RANK_BLOCK: usize = 512;
4147
4148const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
4162
4163impl NativeText {
4164 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
4171 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
4172 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
4173 Ok(Some(bytes.as_slice()))
4174 }
4175
4176 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
4181 let len = self.lengths[block];
4182 let mut stored = vec![
4183 0;
4184 usize::try_from(len).map_err(|_| invalid(
4185 "global dictionary block does not fit in memory"
4186 ))?
4187 ];
4188 read_at(&self.file, self.starts[block], &mut stored)?;
4189 if checksum(&stored) != self.hashes[block] {
4190 return Err(invalid("global dictionary payload checksum differs"));
4191 }
4192 let first = block * TEXT_PAYLOAD_VALUES;
4193 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
4194 let want = self.end_within(last - 1)? as usize;
4195 let values = string::decode_flat(&stored)?;
4196 if values.len() != last - first {
4197 return Err(invalid("global dictionary block holds the wrong value count"));
4198 }
4199 let bytes = values.into_bytes();
4200 if bytes.len() != want {
4201 return Err(invalid("global dictionary block decodes to the wrong length"));
4202 }
4203 Ok(bytes)
4204 }
4205
4206 fn ends_worth_unpacking(&self) -> usize {
4223 self.values.max(TEXT_PAYLOAD_VALUES)
4224 }
4225
4226 fn value_ends(&self) -> Option<&[u32]> {
4228 if let Some(built) = self.value_ends.get() {
4229 return built.as_deref();
4230 }
4231 if self.ends_asked.fetch_add(1, Atomic::Relaxed) < self.ends_worth_unpacking() {
4232 return None;
4233 }
4234 self.value_ends.get_or_init(|| self.unpack_ends()).as_deref()
4235 }
4236
4237 fn unpack_ends(&self) -> Option<Vec<u32>> {
4243 let mut ends = vec![0u32; self.values];
4244 for (run, into) in ends.chunks_mut(TEXT_OFFSET_RUN).enumerate() {
4245 let bytes = self.offsets.get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)?;
4246 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| {
4247 u32::try_from(bits).unwrap_or(u32::MAX)
4248 })
4249 .ok()?;
4250 }
4251 if ends.contains(&u32::MAX) { None } else { Some(ends) }
4254 }
4255
4256 fn end_within(&self, index: usize) -> Result<u32> {
4258 if let Some(ends) = self.value_ends() {
4259 return ends
4260 .get(index)
4261 .copied()
4262 .ok_or_else(|| invalid("global dictionary offsets are short"));
4263 }
4264 let run = index / TEXT_OFFSET_RUN;
4265 let bytes = self
4266 .offsets
4267 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4268 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4269 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
4270 .map_err(|_| invalid("global dictionary offsets are short"))?;
4271 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
4272 }
4273
4274 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
4292 let mut ends = vec![0u64; last.saturating_sub(first)];
4293 let mut scratch = Vec::new();
4294 let mut at = first;
4295 while at < last {
4296 let run = at / TEXT_OFFSET_RUN;
4297 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
4298 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
4299 let bytes = self
4300 .offsets
4301 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4302 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4303 let from = at % TEXT_OFFSET_RUN;
4304 let upto = stop - run * TEXT_OFFSET_RUN;
4305 if upto > held || bytes.len() < bitpack::tail_len(held, self.offset_bits) {
4306 return Err(invalid("global dictionary offsets are short"));
4307 }
4308 let into = &mut ends[at - first..stop - first];
4309 if from == 0 {
4310 bitpack::unpack_tail_into(bytes, self.offset_bits, into, |bits| bits)
4311 .map_err(|_| invalid("global dictionary offsets are short"))?;
4312 } else {
4313 scratch.resize(held, 0);
4314 bitpack::unpack_tail_into(bytes, self.offset_bits, &mut scratch, |bits| bits)
4315 .map_err(|_| invalid("global dictionary offsets are short"))?;
4316 into.copy_from_slice(&scratch[from..upto]);
4317 }
4318 at = stop;
4319 }
4320 Ok(ends)
4321 }
4322
4323 fn start_within(&self, index: usize) -> Result<u32> {
4326 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
4327 }
4328
4329 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
4337 if let Some(ends) = self.value_ends() {
4338 let end =
4339 *ends.get(index).ok_or_else(|| invalid("global dictionary offsets are short"))?;
4340 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4343 if start > end {
4344 return Err(invalid("global dictionary value ends before it starts"));
4345 }
4346 return Ok((start, end));
4347 }
4348 let within = index % TEXT_OFFSET_RUN;
4349 let (start, end) = if within == 0 {
4350 (self.start_within(index)?, self.end_within(index)?)
4351 } else {
4352 let run = index / TEXT_OFFSET_RUN;
4353 let bytes = self
4354 .offsets
4355 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
4356 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
4357 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
4358 .map_err(|_| invalid("global dictionary offsets are short"))?;
4359 let ends = u32::try_from(end)
4360 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4361 let starts = u32::try_from(start)
4362 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
4363 (starts, ends)
4364 };
4365 if start > end {
4366 return Err(invalid("global dictionary value ends before it starts"));
4367 }
4368 Ok((start, end))
4369 }
4370
4371 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
4378 let slot = self
4379 .rank_blocks
4380 .get(rank / TEXT_RANK_BLOCK)
4381 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
4382 let block = slot
4383 .get_or_init(|| {
4384 let which = rank / TEXT_RANK_BLOCK;
4385 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
4386 let end = self.rank_ends[which];
4387 let mut bytes = vec![0; (end - start) as usize];
4388 read_at(&self.file, self.rank_at + start, &mut bytes)?;
4389 if checksum(&bytes)
4390 != *self
4391 .rank_hashes
4392 .get(rank / TEXT_RANK_BLOCK)
4393 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
4394 {
4395 return Err(invalid("global dictionary rank checksum differs"));
4396 }
4397 Ok(bytes)
4398 })
4399 .as_ref()
4400 .map_err(Clone::clone)?;
4401 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
4402 }
4403
4404 fn head_at(&self, rank: usize) -> Result<u64> {
4406 let (block, within) = self.rank_parts(rank)?;
4407 let (base, width, packed) = rank_heads(block)?;
4408 let above = bitpack::tail_at(packed, width, within)
4409 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
4410 Ok(base.wrapping_add(above))
4411 }
4412
4413 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
4415 let (_, width, packed) = rank_heads(block)?;
4416 packed
4417 .get(bitpack::tail_len(count, width)..)
4418 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
4419 }
4420
4421 fn rank_block_len(&self, rank: usize) -> usize {
4423 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
4424 TEXT_RANK_BLOCK.min(self.ranks - first)
4425 }
4426}
4427
4428fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
4430 let header = block
4431 .get(..RANK_BLOCK_HEADER)
4432 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
4433 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
4434 let width = header[8] as usize;
4435 if width > 64 {
4436 return Err(invalid("global dictionary rank block packs heads past a word"));
4437 }
4438 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
4439}
4440
4441fn offset_width(ends: &[u32]) -> usize {
4448 let span = ends.iter().copied().max().unwrap_or(0);
4452 (u32::BITS - span.leading_zeros()) as usize
4453}
4454
4455fn offset_bytes(values: usize, bits: usize) -> usize {
4458 let full = values / TEXT_OFFSET_RUN;
4459 let rest = values % TEXT_OFFSET_RUN;
4460 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
4461}
4462
4463fn encode_offsets(ends: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
4467 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
4468 for chunk in ends.chunks(TEXT_OFFSET_RUN) {
4469 run.clear();
4470 run.extend(chunk.iter().map(|&end| u64::from(end)));
4471 bitpack::pack_tail(&run, bits, out)
4472 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
4473 }
4474 Ok(())
4475}
4476
4477fn code_width(values: usize) -> usize {
4479 match u64::try_from(values).unwrap_or(u64::MAX) {
4480 0 | 1 => 0,
4481 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
4482 }
4483}
4484
4485impl TextSource for NativeText {
4486 fn len(&self) -> usize {
4487 self.values
4488 }
4489
4490 fn might_contain(&self, first: usize, literal: &[u8]) -> Result<bool> {
4491 let Some(grams) = &self.grams else { return Ok(true) };
4492 if literal.len() < 4 || first >= self.values {
4493 return Ok(true);
4494 }
4495 let verdict = grams.verdicts(&self.file, literal)?;
4496 Ok(verdict.get(first / TEXT_PAYLOAD_VALUES).copied().unwrap_or(true))
4497 }
4498
4499 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
4500 if index >= self.values {
4501 return Ok(None);
4502 }
4503 let (start, end) = self.span_within(index)?;
4504 if start == end {
4505 return Ok(Some(&[]));
4506 }
4507 let block = index / TEXT_PAYLOAD_VALUES;
4510 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
4511 Ok(bytes.get(start as usize..end as usize))
4512 }
4513
4514 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
4515 if index >= self.values {
4516 return Ok(None);
4517 }
4518 let (start, end) = self.span_within(index)?;
4519 Ok(Some((end - start) as usize))
4520 }
4521
4522 fn bytes_lens_at(&self, indices: &[u32], into: &mut [i64]) -> Result<()> {
4529 self.ends_asked.fetch_add(indices.len(), Atomic::Relaxed);
4530 let Some(ends) = self.value_ends() else {
4531 for (slot, &index) in into.iter_mut().zip(indices) {
4532 *slot = self
4533 .bytes_len_at(index as usize)?
4534 .map_or(0, |len| i64::try_from(len).unwrap_or(i64::MAX));
4535 }
4536 return Ok(());
4537 };
4538 if let Some(lens) = self.value_lens.get_or_init(|| lengths_of(ends)) {
4539 for (slot, &index) in into.iter_mut().zip(indices) {
4540 *slot = lens.get(index as usize).map_or(0, |&len| i64::from(len));
4543 }
4544 return Ok(());
4545 }
4546 for (slot, &index) in into.iter_mut().zip(indices) {
4547 let index = index as usize;
4548 let Some(&end) = ends.get(index) else {
4550 *slot = 0;
4551 continue;
4552 };
4553 let start = if index % TEXT_PAYLOAD_VALUES == 0 { 0 } else { ends[index - 1] };
4554 if start > end {
4555 return Err(invalid("global dictionary value ends before it starts"));
4556 }
4557 *slot = i64::from(end - start);
4558 }
4559 Ok(())
4560 }
4561
4562 fn sweep(
4575 &self,
4576 first: usize,
4577 limit: usize,
4578 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4579 ) -> Result<usize> {
4580 let limit = limit.min(self.values);
4581 if first >= limit {
4582 return Ok(first);
4583 }
4584 let block = first / TEXT_PAYLOAD_VALUES;
4585 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
4586 let decoded;
4587 let kept = self.blocks.get(block).and_then(OnceLock::get);
4588 let again = kept.is_none()
4589 && self.swept.get(block).is_some_and(|swept| swept.swap(true, Atomic::Relaxed));
4590 let bytes: &[u8] = match kept {
4591 Some(Ok(kept)) => kept,
4592 _ if again && self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
4593 let kept = self
4594 .payload_block(block)?
4595 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
4596 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
4597 kept
4598 }
4599 _ => {
4600 decoded = self.decode_block(block)?;
4601 &decoded
4602 }
4603 };
4604 let ends = self.ends_within(first, last)?;
4605 if ends.len() != last - first {
4606 return Err(invalid("global dictionary offsets are short"));
4607 }
4608 let mut start = u64::from(self.start_within(first)?);
4609 for (index, &end) in (first..last).zip(&ends) {
4612 let value = usize::try_from(start)
4613 .ok()
4614 .zip(usize::try_from(end).ok())
4615 .and_then(|(from, to)| bytes.get(from..to))
4616 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4617 body(index, value)?;
4618 start = end;
4619 }
4620 Ok(last)
4621 }
4622
4623 fn visit(
4629 &self,
4630 indices: &[usize],
4631 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
4632 ) -> Result<()> {
4633 let mut at = 0;
4634 while at < indices.len() {
4635 let block = indices[at] / TEXT_PAYLOAD_VALUES;
4636 let upto =
4637 at + indices[at..].partition_point(|&index| index / TEXT_PAYLOAD_VALUES == block);
4638 let wanted = &indices[at..upto];
4639 if wanted.iter().any(|&index| index >= self.values) {
4640 return Err(invalid("a visited value is past the global dictionary"));
4641 }
4642 let decoded;
4643 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
4644 Some(Ok(kept)) => kept,
4645 _ => {
4646 decoded = self.decode_block(block)?;
4647 &decoded
4648 }
4649 };
4650 for (offset, &index) in wanted.iter().enumerate() {
4651 let (start, end) = self.span_within(index)?;
4652 let value = bytes
4653 .get(start as usize..end as usize)
4654 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
4655 body(at + offset, value)?;
4656 }
4657 at = upto;
4658 }
4659 Ok(())
4660 }
4661
4662 fn ranks(&self) -> Option<usize> {
4663 (self.ranks > 0).then_some(self.ranks)
4664 }
4665
4666 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
4674 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
4675 if let Some(&answer) = memo.get(wanted) {
4676 return Ok(answer);
4677 }
4678 let answer = search_below(self, ranks, wanted)?;
4679 if memo.len() >= TEXT_SEARCH_MEMO {
4680 memo.clear();
4681 }
4682 memo.insert(wanted.to_vec(), answer);
4683 Ok(answer)
4684 }
4685
4686 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
4687 let settled = self.head_at(rank)?.cmp(&head(wanted));
4691 if settled != Ordering::Equal {
4692 return Ok(settled);
4693 }
4694 let code = self.code_at_rank(rank)?;
4695 let bytes = self
4696 .bytes_at(code as usize)?
4697 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4698 Ok(bytes.cmp(wanted))
4699 }
4700
4701 fn code_at_rank(&self, rank: usize) -> Result<u32> {
4702 let (block, within) = self.rank_parts(rank)?;
4703 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
4704 let code = bitpack::tail_at(codes, self.code_bits, within)
4705 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
4706 let code = u32::try_from(code)
4707 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
4708 if code as usize >= self.len() {
4709 return Err(invalid("global dictionary order names a code it does not have"));
4710 }
4711 Ok(code)
4712 }
4713
4714 fn code_ranks(&self) -> Option<&[u32]> {
4715 if self.ranks == 0 || self.ranks != self.len() {
4719 return None;
4720 }
4721 self.code_ranks
4722 .get_or_init(|| {
4723 let mut ranks = vec![u32::MAX; self.ranks];
4724 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
4727 let (block, _) = self.rank_parts(first).ok()?;
4728 let count = self.rank_block_len(first);
4729 let codes = self.rank_codes(block, count).ok()?;
4730 for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
4731 .ok()?
4732 .into_iter()
4733 .enumerate()
4734 {
4735 let code = usize::try_from(code).ok()?;
4736 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
4737 }
4738 }
4739 if ranks.contains(&u32::MAX) {
4740 return None;
4741 }
4742 Some(ranks)
4743 })
4744 .as_deref()
4745 }
4746
4747 fn footprint(&self) -> usize {
4748 self.offsets.capacity()
4749 + self
4750 .value_ends
4751 .get()
4752 .and_then(Option::as_ref)
4753 .map_or(0, |ends| ends.capacity() * size_of::<u32>())
4754 + self
4755 .value_lens
4756 .get()
4757 .and_then(Option::as_ref)
4758 .map_or(0, |lens| lens.capacity() * size_of::<u32>())
4759 + self
4760 .code_ranks
4761 .get()
4762 .and_then(Option::as_ref)
4763 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
4764 + self.rank_hashes.capacity() * size_of::<u64>()
4765 + self.rank_ends.capacity() * size_of::<u64>()
4766 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4767 + self
4768 .rank_blocks
4769 .iter()
4770 .filter_map(OnceLock::get)
4771 .filter_map(|result| result.as_ref().ok())
4772 .map(Vec::capacity)
4773 .sum::<usize>()
4774 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
4775 + self.hashes.capacity() * size_of::<u64>()
4776 + self.starts.capacity() * size_of::<u64>()
4777 + self.lengths.capacity() * size_of::<u64>()
4778 + self.grams.as_ref().map_or(0, NativeGrams::footprint)
4779 + self
4780 .blocks
4781 .iter()
4782 .filter_map(OnceLock::get)
4783 .filter_map(|result| result.as_ref().ok())
4784 .map(Vec::capacity)
4785 .sum::<usize>()
4786 }
4787}
4788
4789fn places(table: &Table) -> Result<Vec<Place>> {
4791 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
4792 for (at, stripe) in table.stripes.iter().enumerate() {
4793 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
4794 for (part, &rows) in stripe.parts.iter().enumerate() {
4795 places.push(Place {
4796 stripe: index,
4797 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
4798 rows,
4799 });
4800 }
4801 }
4802 Ok(places)
4803}
4804
4805fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
4810 let parts = stripe.parts.len();
4811 let section = index_section(parts)?;
4812 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
4813 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
4814 if end > stripe.index.length as usize {
4815 return Err(invalid("index page is shorter than its columns"));
4816 }
4817 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
4818 let mut bytes = vec![0; section];
4819 let offset = stripe
4820 .index
4821 .offset
4822 .checked_add(at as u64)
4823 .ok_or_else(|| invalid("index page offset overflow"))?;
4824 read_at(file, offset, &mut bytes)?;
4825 let entries = section - size_of::<u64>();
4826 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
4827 if checksum(&bytes[..entries]) != stored {
4828 return Err(invalid(&format!(
4831 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
4832 wanted {stored:016x} and got {:016x}",
4833 checksum(&bytes[..entries]),
4834 )));
4835 }
4836 let mut spans = Vec::with_capacity(parts);
4837 let mut start = 0_usize;
4838 for part in 0..parts {
4839 let at = part * INDEX_ENTRY;
4840 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
4841 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
4842 spans.push(PartSpan { start, length, hash });
4843 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
4844 }
4845 if start != page.length as usize {
4846 return Err(invalid("column page length differs from its index"));
4847 }
4848 Ok(spans)
4849}
4850
4851fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
4853 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
4854 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
4855}
4856
4857fn remember(cached: &mut Cached, held: &CachedColumn) -> Option<(usize, Arc<AtomicBool>)> {
4863 if let Some(slot) = cached.index.get_mut(held.stripe) {
4864 if slot.is_none() {
4865 *slot = Some(Arc::clone(&held.index));
4866 }
4867 }
4868 let page = held.page.clone()?;
4869 let slot = cached.pages.get_mut(held.stripe)?;
4870 if slot.is_some() {
4871 return None;
4872 }
4873 let bytes = page.len();
4874 let used = Arc::new(AtomicBool::new(true));
4877 *slot = Some(Resident { page, used: Arc::clone(&used) });
4878 Some((bytes, used))
4879}
4880
4881#[derive(Debug, Clone)]
4890pub struct Catalog {
4891 file: Arc<File>,
4892 size: u64,
4893 entries: Arc<Vec<Entry>>,
4894 views: Arc<Vec<ViewEntry>>,
4896 opening: Opening,
4897 pool: PagePool,
4899}
4900
4901#[derive(Debug, Clone, PartialEq, Eq)]
4903pub struct CertifiedSums {
4904 pub columns: Vec<(i128, u64)>,
4905 pub rows: u64,
4906}
4907
4908#[derive(Debug, Clone, Copy, PartialEq, Eq)]
4910pub enum IntegerExtremes {
4911 Null,
4912 Values { low: i128, high: i128 },
4913}
4914
4915pub type NumericFrequencies = Vec<(Option<i128>, u64)>;
4917
4918impl Catalog {
4919 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
4928 Self::open_in(path, &PagePool::default())
4929 }
4930
4931 pub fn open_in(path: impl AsRef<Path>, pool: &PagePool) -> Result<Self> {
4937 let (file, size, _, bytes, opening) = slot_bytes(path)?;
4938 let (entries, views) = decode_catalog(&bytes, size)?;
4939 Ok(Self {
4940 file: Arc::new(file),
4941 size,
4942 entries: Arc::new(entries),
4943 views: Arc::new(views),
4944 opening,
4945 pool: pool.clone(),
4946 })
4947 }
4948
4949 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
4951 self.entries.iter().map(|entry| entry.name.as_str())
4952 }
4953
4954 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
4961 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
4962 }
4963
4964 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
4970 self.views.iter()
4971 }
4972
4973 #[must_use]
4975 pub fn len(&self) -> usize {
4976 self.entries.len()
4977 }
4978
4979 #[must_use]
4982 pub fn is_empty(&self) -> bool {
4983 self.entries.is_empty()
4984 }
4985
4986 pub fn table(&self, name: &str) -> Result<Reader> {
4992 let entry = self
4993 .entries
4994 .iter()
4995 .find(|entry| entry.name == name)
4996 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
4997 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5001 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5002 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5003 }
5004 let mut opening = self.opening;
5005 opening.reads += 1;
5006 opening.bytes += u64::from(entry.directory.length);
5007 Reader::build(
5008 Arc::clone(&self.file),
5009 self.size,
5010 read_directory(Cursor::over(&self.file, offset, length), self.size, Some(offset))?,
5011 u64::from(entry.directory.length),
5012 opening,
5013 self.pool.clone(),
5014 )
5015 }
5016
5017 pub fn nonzero_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5021 let entry = self
5022 .entries
5023 .iter()
5024 .find(|entry| entry.name == name)
5025 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5026 let Some(field) = entry.fields.get(column) else {
5027 return Err(invalid("frequency column index out of range"));
5028 };
5029 if !matches!(
5030 field.ty,
5031 LogicalType::TinyInt
5032 | LogicalType::SmallInt
5033 | LogicalType::Integer
5034 | LogicalType::BigInt
5035 | LogicalType::UTinyInt
5036 | LogicalType::USmallInt
5037 | LogicalType::UInteger
5038 | LogicalType::UBigInt
5039 ) {
5040 return Ok(None);
5041 }
5042 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5043 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5044 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5045 }
5046 if let Some(count) = entry.nonzero.get(column).copied().flatten() {
5047 return Ok(Some(count));
5048 }
5049 quick_nonzero(
5050 Cursor::over(&self.file, offset, length),
5051 &entry.name,
5052 &entry.fields,
5053 entry.rows,
5054 column,
5055 )
5056 }
5057
5058 pub fn aggregate_sums(&self, name: &str, columns: &[usize]) -> Result<Option<CertifiedSums>> {
5061 let entry = self
5062 .entries
5063 .iter()
5064 .find(|entry| entry.name == name)
5065 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5066 let mut sums = Vec::with_capacity(columns.len());
5067 for &column in columns {
5068 let Some(field) = entry.fields.get(column) else {
5069 return Err(invalid("aggregate column index out of range"));
5070 };
5071 if !signed_integer(&field.ty) {
5072 return Ok(None);
5073 }
5074 let Some(sum) = entry.aggregates[column] else {
5075 return Ok(None);
5076 };
5077 sums.push(sum);
5078 }
5079 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5080 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5081 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5082 }
5083 Ok(Some(CertifiedSums { columns: sums, rows: entry.rows as u64 }))
5084 }
5085
5086 pub fn distinct_count(&self, name: &str, column: usize) -> Result<Option<u64>> {
5088 let entry = self
5089 .entries
5090 .iter()
5091 .find(|entry| entry.name == name)
5092 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5093 let Some(count) = entry.distincts.get(column).copied() else {
5094 return Err(invalid("distinct column index out of range"));
5095 };
5096 let Some(count) = count else { return Ok(None) };
5097 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5098 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5099 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5100 }
5101 Ok(Some(count))
5102 }
5103
5104 pub fn integer_extremes(&self, name: &str, column: usize) -> Result<Option<IntegerExtremes>> {
5106 let entry = self
5107 .entries
5108 .iter()
5109 .find(|entry| entry.name == name)
5110 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5111 let Some(extremes) = entry.extremes.get(column).copied() else {
5112 return Err(invalid("extremes column index out of range"));
5113 };
5114 let Some(extremes) = extremes else { return Ok(None) };
5115 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5116 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5117 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5118 }
5119 Ok(Some(match extremes {
5120 None => IntegerExtremes::Null,
5121 Some((low, high)) => IntegerExtremes::Values { low, high },
5122 }))
5123 }
5124
5125 pub fn exact_numeric_frequencies(
5127 &self,
5128 name: &str,
5129 column: usize,
5130 ) -> Result<Option<NumericFrequencies>> {
5131 let entry = self
5132 .entries
5133 .iter()
5134 .find(|entry| entry.name == name)
5135 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
5136 let Some(frequencies) = entry.frequencies.get(column).cloned() else {
5137 return Err(invalid("numeric frequency column index out of range"));
5138 };
5139 let Some(frequencies) = frequencies else { return Ok(None) };
5140 let (offset, length) = (entry.directory.offset, entry.directory.length as usize);
5141 if file_checksum(&self.file, offset, length)? != entry.directory.hash {
5142 return Err(invalid(&format!("the directory of table {name} does not checksum")));
5143 }
5144 Ok(Some(frequencies))
5145 }
5146
5147 pub fn table_fields(&self, name: &str) -> Option<&[Field]> {
5149 self.entries.iter().find(|entry| entry.name == name).map(|entry| entry.fields.as_slice())
5150 }
5151}
5152
5153fn slot_offset(generation: u64) -> u64 {
5158 16 + (generation - 1) % 2 * SLOT_BYTES as u64
5159}
5160
5161fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
5166 let mut file = File::open(path).map_err(io)?;
5167 let size = file.metadata().map_err(io)?.len();
5168 if size < HEADER {
5169 return Err(invalid("file is shorter than its header"));
5170 }
5171 let mut header = [0; HEADER as usize];
5172 file.read_exact(&mut header).map_err(io)?;
5173 let mut opening = Opening { reads: 1, bytes: HEADER };
5174 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
5175 if &header[..8] != MAGIC {
5180 return Err(invalid("the header does not begin with a rudb native magic"));
5181 }
5182 if !READABLE.contains(&version) {
5183 return Err(invalid(&format!(
5184 "the file is format {version} and this build reads format {FORMAT}, so it has to \
5185 be written again"
5186 )));
5187 }
5188 let mut selected = None;
5189 for start in [16, 16 + SLOT_BYTES] {
5190 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
5191 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
5192 continue;
5193 }
5194 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
5195 if slot.offset < HEADER || end > size {
5196 continue;
5197 }
5198 let mut bytes = vec![0; slot.length as usize];
5199 file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
5200 file.read_exact(&mut bytes).map_err(io)?;
5201 opening.reads += 1;
5202 opening.bytes += u64::from(slot.length);
5203 if checksum(&bytes) == slot.hash
5204 && selected
5205 .as_ref()
5206 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
5207 {
5208 selected = Some((slot, bytes));
5209 }
5210 }
5211 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
5212 Ok((file, size, slot, bytes, opening))
5213}
5214
5215impl Reader {
5216 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
5223 let catalog = Catalog::open(path)?;
5224 let mut names = catalog.names();
5225 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
5226 if names.next().is_some() {
5227 return Err(invalid(
5228 "the file holds more than one table, so it has to be opened by name",
5229 ));
5230 }
5231 catalog.table(&name)
5232 }
5233
5234 fn build(
5236 file: Arc<File>,
5237 size: u64,
5238 table: Table,
5239 directory: u64,
5240 opening: Opening,
5241 pool: PagePool,
5242 ) -> Result<Self> {
5243 let places = places(&table)?;
5244 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
5245 let table_fields = table.fields.len();
5246 let stripes = table.stripes.len();
5247 let columns = (0..table.fields.len())
5248 .map(|_| {
5249 Mutex::new(Cached {
5250 pages: (0..stripes).map(|_| None).collect(),
5251 index: (0..stripes).map(|_| None).collect(),
5252 ..Cached::default()
5253 })
5254 })
5255 .collect::<Vec<_>>();
5256 let cache = Shelf {
5257 columns,
5258 held: (0..table_fields).map(|_| AtomicUsize::new(0)).collect(),
5259 kept: AtomicUsize::new(CACHED_STRIPES_PER_COLUMN),
5260 };
5261 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
5262 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5263 .collect();
5264 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
5265 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
5266 .collect();
5267 Ok(Self {
5268 file,
5269 table: Arc::new(table),
5270 dictionaries: Arc::new(dictionaries),
5271 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
5272 frequency_values: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5273 frequency_summaries: Arc::new((0..table_fields).map(|_| OnceLock::new()).collect()),
5274 opened: Arc::new(AtomicUsize::new(0)),
5275 sieves: Arc::new(sieves),
5276 part_ranges: Arc::new(part_ranges),
5277 places: Arc::new(places),
5278 cache: Arc::new(cache),
5279 pool,
5280 pages: Arc::new(AtomicUsize::new(0)),
5281 indexes: Arc::new(AtomicUsize::new(0)),
5282 size,
5283 directory,
5284 opening,
5285 })
5286 }
5287
5288 #[must_use]
5295 pub fn reads(&self) -> Reads {
5296 Reads {
5297 opening: self.opening,
5298 pages: self.pages.load(Atomic::Relaxed),
5299 indexes: self.indexes.load(Atomic::Relaxed),
5300 dictionaries: self.opened.load(Atomic::Relaxed),
5301 }
5302 }
5303
5304 #[must_use]
5309 pub fn layout(&self) -> Layout {
5310 let table = &self.table;
5311 let stripes = table.stripes.as_slice();
5312 let columns = table
5313 .fields
5314 .iter()
5315 .enumerate()
5316 .map(|(at, field)| ColumnLayout {
5317 name: field.name.clone(),
5318 kind: field.ty.to_string(),
5319 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
5320 memberships: sum(stripes.iter().map(|stripe| stripe.memberships.bytes(at))),
5321 sieves: sum(stripes.iter().map(|stripe| stripe.sieves.bytes(at))),
5322 part_ranges: sum(stripes.iter().map(|stripe| stripe.part_ranges.bytes(at))),
5323 dictionary: dictionary_bytes(table, at),
5324 })
5325 .collect();
5326 Layout {
5327 file: self.size,
5328 rows: table.rows,
5329 stripes: stripes.len(),
5330 parts: self.places.len(),
5331 columns,
5332 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
5333 directory: self.directory,
5334 header: HEADER,
5335 }
5336 }
5337
5338 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
5355 let field = self
5356 .table
5357 .fields
5358 .get(column)
5359 .ok_or_else(|| invalid("stored column index out of range"))?;
5360 let mut stored = Vec::with_capacity(self.places.len());
5361 let mut row = 0;
5362 for (at, stripe) in self.table.stripes.iter().enumerate() {
5363 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
5364 let index = read_index(&self.file, stripe, column)?;
5365 let mut bytes = vec![0; page.length as usize];
5366 read_at(&self.file, page.offset, &mut bytes)?;
5367 let ranges = self.stripe_part_ranges(at, column);
5368 for (part, &rows) in stripe.parts.iter().enumerate() {
5369 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
5370 let held = part_bytes(&bytes, span)?;
5371 let range = ranges.and_then(|held| held.get(part));
5372 stored.push(StoredPart {
5373 stripe: at,
5374 part,
5375 row,
5376 rows: rows as usize,
5377 encoding: page_encoding(&field.ty, rows as usize, held),
5378 bytes: span.length as u64,
5379 page: page.offset,
5380 offset: span.start as u64,
5381 low: range
5382 .and_then(|range| range.low.clone())
5383 .and_then(|bound| bound.into_value(&field.ty)),
5384 high: range
5385 .and_then(|range| range.high.clone())
5386 .and_then(|bound| bound.into_value(&field.ty)),
5387 nulls: range.map(|range| range.nulls),
5388 });
5389 row += rows as usize;
5390 }
5391 }
5392 Ok(stored)
5393 }
5394
5395 #[must_use]
5397 pub fn parts(&self) -> usize {
5398 self.places.len()
5399 }
5400
5401 #[must_use]
5408 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
5409 let mut runs = Vec::with_capacity(self.table.stripes.len());
5410 let mut start = 0;
5411 for stripe in &self.table.stripes {
5412 let end = start + stripe.parts.len();
5413 runs.push(start..end);
5414 start = end;
5415 }
5416 runs
5417 }
5418
5419 #[must_use]
5424 pub fn stripe_rows(&self, stripe: usize) -> usize {
5425 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
5426 }
5427
5428 pub fn keep_stripes(&self, stripes: usize) {
5435 self.cache.kept.fetch_max(stripes, Atomic::Relaxed);
5436 }
5437
5438 #[must_use]
5440 pub fn part_rows(&self, at: usize) -> usize {
5441 self.places.get(at).map_or(0, |place| place.rows as usize)
5442 }
5443
5444 #[must_use]
5446 pub fn table(&self) -> &Table {
5447 &self.table
5448 }
5449
5450 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
5459 let field = self
5460 .table
5461 .fields
5462 .get(column)
5463 .ok_or_else(|| invalid("frequency column index out of range"))?;
5464 let Some(summary) = self.frequency_summary(column)? else {
5465 return Ok(None);
5466 };
5467 if top == 0 || summary.entries.len() < top {
5468 return Ok(None);
5469 }
5470 let boundary = summary.entries[top - 1].count;
5471 if boundary <= summary.omitted_max {
5472 return Ok(None);
5473 }
5474 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
5475 }
5476
5477 pub fn top_pair_frequencies(
5488 &self,
5489 first: usize,
5490 second: usize,
5491 top: usize,
5492 ) -> Result<Option<PairFrequencyCounts>> {
5493 if first >= self.table.fields.len() || second >= self.table.fields.len() {
5494 return Err(invalid("pair frequency column index out of range"));
5495 }
5496 let Some(summary) =
5497 self.table.pair_frequencies.iter().find(|summary| {
5498 summary.first as usize == first && summary.second as usize == second
5499 })
5500 else {
5501 return Ok(None);
5502 };
5503 if top == 0 || summary.entries.len() < top {
5504 return Ok(None);
5505 }
5506 let boundary = summary.entries[top - 1].count;
5507 if boundary <= summary.omitted_max {
5508 return Ok(None);
5509 }
5510 let first_summary = self
5511 .frequency_summary(first)?
5512 .ok_or_else(|| invalid("pair frequency first column has no synopsis"))?;
5513 let anchors = self
5514 .decode_frequencies(first, &self.table.fields[first].ty, &first_summary.entries)?
5515 .into_iter()
5516 .map(|(value, _)| value)
5517 .collect::<Vec<_>>();
5518 let dictionary = self
5519 .dictionary(second)?
5520 .ok_or_else(|| invalid("pair frequency second column has no dictionary"))?;
5521 let mut codes = summary.entries.iter().filter_map(|entry| entry.second).collect::<Vec<_>>();
5522 codes.sort_unstable();
5523 codes.dedup();
5524 let texts = dictionary
5525 .try_values_visited(&codes.iter().map(|&code| code as usize).collect::<Vec<_>>())?;
5526 let mut out = Vec::with_capacity(summary.entries.len());
5527 for entry in &summary.entries {
5528 if entry.count < boundary {
5529 break;
5530 }
5531 let first = anchors
5532 .get(entry.first_entry as usize)
5533 .cloned()
5534 .ok_or_else(|| invalid("pair frequency anchor is outside its values"))?;
5535 let second = match entry.second {
5536 None => Value::Null,
5537 Some(code) => {
5538 let at = codes
5539 .binary_search(&code)
5540 .map_err(|_| invalid("pair frequency code was not among the codes read"))?;
5541 texts[at].clone()
5542 }
5543 };
5544 out.push((vec![first, second], entry.count));
5545 }
5546 Ok(Some(out))
5547 }
5548
5549 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
5569 let Some(prefix) = self.frequency_prefix(column)? else {
5570 return Ok(None);
5571 };
5572 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
5573 }
5574
5575 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
5598 let field = self
5599 .table
5600 .fields
5601 .get(column)
5602 .ok_or_else(|| invalid("frequency column index out of range"))?;
5603 let Some(summary) = self.frequency_summary(column)? else {
5604 return Ok(None);
5605 };
5606 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5607 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
5608 }
5609
5610 fn frequency_summary(&self, column: usize) -> Result<Option<Cow<'_, FrequencySummary>>> {
5612 Ok(match self.table.frequencies.get(column) {
5613 None | Some(None) => None,
5614 Some(Some(Frequencies::Held(summary))) => Some(Cow::Borrowed(summary)),
5615 Some(Some(Frequencies::Stored { span, values })) => {
5616 let slot = self
5617 .frequency_summaries
5618 .get(column)
5619 .ok_or_else(|| invalid("frequency column index out of range"))?;
5620 if let Some(summary) = slot.get() {
5621 return Ok(Some(Cow::Borrowed(summary.as_ref())));
5622 }
5623 let field = self
5624 .table
5625 .fields
5626 .get(column)
5627 .ok_or_else(|| invalid("frequency column index out of range"))?;
5628 let mut bytes = vec![0; span.length as usize];
5629 read_at(&self.file, span.offset, &mut bytes)?;
5630 let summary =
5631 decode_summary(&mut Cursor::new(&bytes), field, self.table.rows, *values)?;
5632 let summary = summary.ok_or_else(|| invalid("a stored synopsis is missing"))?;
5633 let _ = slot.set(Arc::new(summary));
5634 Some(Cow::Borrowed(slot.get().expect("the decoded summary was stored").as_ref()))
5635 }
5636 })
5637 }
5638
5639 fn decode_frequencies(
5647 &self,
5648 column: usize,
5649 ty: &LogicalType,
5650 entries: &[FrequencyEntry],
5651 ) -> Result<Vec<(Value, u64)>> {
5652 if let Some(values) = self.frequency_values.get(column).and_then(OnceLock::get) {
5653 return Ok(values.as_ref().clone());
5654 }
5655 let values = self.decode_frequencies_once(column, ty, entries)?;
5656 if let Some(slot) = self.frequency_values.get(column) {
5657 let _ = slot.set(Arc::new(values.clone()));
5658 }
5659 Ok(values)
5660 }
5661
5662 fn decode_frequencies_once(
5663 &self,
5664 column: usize,
5665 ty: &LogicalType,
5666 entries: &[FrequencyEntry],
5667 ) -> Result<Vec<(Value, u64)>> {
5668 let stored_texts = self.table.frequency_texts.get(column).filter(|texts| !texts.is_empty());
5669 if stored_texts.is_some_and(|texts| texts.len() != entries.len()) {
5670 return Err(invalid("frequency text count differs from its synopsis"));
5671 }
5672 let dictionary = if *ty == LogicalType::Varchar && stored_texts.is_none() {
5673 self.dictionary(column)?
5674 } else {
5675 None
5676 };
5677 let mut codes = entries
5678 .iter()
5679 .filter_map(|entry| match entry.value {
5680 FrequencyValue::Code(code) => Some(code as usize),
5681 _ => None,
5682 })
5683 .collect::<Vec<_>>();
5684 codes.sort_unstable();
5685 codes.dedup();
5686 let texts = match &dictionary {
5687 Some(dictionary) if !codes.is_empty() => dictionary.try_values_visited(&codes)?,
5688 _ => Vec::new(),
5689 };
5690 let mut out = Vec::with_capacity(entries.len());
5691 for (entry_at, entry) in entries.iter().enumerate() {
5692 let value = match entry.value {
5693 FrequencyValue::Null => {
5694 if stored_texts.and_then(|texts| texts[entry_at].as_ref()).is_some() {
5695 return Err(invalid("a null frequency entry has text"));
5696 }
5697 Value::Null
5698 }
5699 FrequencyValue::Integer(value) => match *ty {
5700 LogicalType::TinyInt => Value::TinyInt(
5701 i8::try_from(value)
5702 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
5703 ),
5704 LogicalType::UTinyInt => Value::UTinyInt(
5705 u8::try_from(value)
5706 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
5707 ),
5708 LogicalType::USmallInt => Value::USmallInt(
5709 u16::try_from(value)
5710 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
5711 ),
5712 LogicalType::UInteger => Value::UInteger(
5713 u32::try_from(value)
5714 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
5715 ),
5716 LogicalType::UBigInt => Value::UBigInt(
5717 u64::try_from(value)
5718 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
5719 ),
5720 LogicalType::SmallInt => Value::SmallInt(
5721 i16::try_from(value)
5722 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
5723 ),
5724 LogicalType::Integer => Value::Integer(
5725 i32::try_from(value)
5726 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
5727 ),
5728 LogicalType::BigInt => Value::BigInt(
5729 i64::try_from(value)
5730 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
5731 ),
5732 LogicalType::Date => Value::Date(
5733 i32::try_from(value)
5734 .map_err(|_| invalid("frequency DATE is out of range"))?,
5735 ),
5736 LogicalType::Timestamp => Value::Timestamp(
5737 i64::try_from(value)
5738 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
5739 ),
5740 _ => return Err(invalid("integer frequency belongs to another type")),
5741 },
5742 FrequencyValue::Code(code) => {
5743 if let Some(text) = stored_texts.and_then(|texts| texts[entry_at].as_ref()) {
5744 Value::Varchar(
5745 String::from_utf8(text.clone())
5746 .map_err(|_| invalid("frequency text is not UTF-8"))?,
5747 )
5748 } else {
5749 if dictionary.is_none() {
5750 return Err(invalid("frequency code has no dictionary or stored text"));
5751 }
5752 let at = codes
5753 .binary_search(&(code as usize))
5754 .map_err(|_| invalid("frequency code was not among the codes read"))?;
5755 texts[at].clone()
5756 }
5757 }
5758 };
5759 out.push((value, entry.count));
5760 }
5761 Ok(out)
5762 }
5763
5764 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
5774 let field = self
5775 .table
5776 .fields
5777 .get(column)
5778 .ok_or_else(|| invalid("frequency column index out of range"))?;
5779 let Some(summary) = self.frequency_summary(column)? else {
5780 return Ok(None);
5781 };
5782 if summary.ordinals.is_empty() {
5783 return Ok(None);
5784 }
5785 let (anchors, anchor_indices) = if summary.ordinal_entries.len() == summary.ordinals.len() {
5786 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
5787 (entries.into_iter().map(|(value, _)| value).collect(), summary.ordinal_entries.clone())
5788 } else {
5789 (Vec::new(), Vec::new())
5790 };
5791 Ok(Some(FrequencyOccurrences {
5792 omitted_max: summary.omitted_max,
5793 ordinals: summary.ordinals.clone(),
5794 anchors,
5795 anchor_indices,
5796 }))
5797 }
5798
5799 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
5825 self.table
5826 .distincts
5827 .get(column)
5828 .copied()
5829 .ok_or_else(|| invalid("distinct column index out of range"))
5830 }
5831
5832 pub fn null_count(&self, column: usize) -> Result<u64> {
5843 if column >= self.table.fields.len() {
5844 return Err(invalid("null count column index out of range"));
5845 }
5846 let mut nulls = 0_u64;
5847 for stripe in &self.table.stripes {
5848 let range = stripe
5849 .zone
5850 .column(column)
5851 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5852 nulls = nulls
5853 .checked_add(range.nulls as u64)
5854 .ok_or_else(|| invalid("null count overflow"))?;
5855 }
5856 Ok(nulls)
5857 }
5858
5859 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
5874 if self.null_count(column)? > 0 {
5875 return Ok(None);
5876 }
5877 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
5878 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
5879 if ranks == 0 {
5880 return Ok(None);
5881 }
5882 let low = text_at_rank(&dictionary, 0)?;
5883 let high = text_at_rank(&dictionary, ranks - 1)?;
5884 Ok(Some((low, high)))
5885 }
5886
5887 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
5910 if column >= self.table.fields.len() {
5911 return Err(invalid("extremes column index out of range"));
5912 }
5913 let mut low: Option<Bound> = None;
5914 let mut high: Option<Bound> = None;
5915 for stripe in &self.table.stripes {
5916 let range = stripe
5917 .zone
5918 .column(column)
5919 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5920 if !range.exact {
5921 return Ok(None);
5922 }
5923 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
5928 if stripe.rows > range.nulls {
5929 return Ok(None);
5930 }
5931 continue;
5932 };
5933 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
5934 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
5935 }
5936 Ok(low.zip(high))
5937 }
5938
5939 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
5952 if column >= self.table.fields.len() {
5953 return Err(invalid("sum column index out of range"));
5954 }
5955 let mut total = 0_i128;
5956 let mut rows = 0_u64;
5957 for stripe in &self.table.stripes {
5958 let range = stripe
5959 .zone
5960 .column(column)
5961 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
5962 let Some(part) = range.sum else { return Ok(None) };
5963 let Some(sum) = total.checked_add(part) else { return Ok(None) };
5964 total = sum;
5965 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
5966 }
5967 Ok(Some((total, rows)))
5968 }
5969
5970 pub fn host_groups(
5973 &self,
5974 column: usize,
5975 minimum_count: u64,
5976 ) -> Result<Option<Vec<host::HostEntry>>> {
5977 if column >= self.table.fields.len() {
5978 return Err(invalid("host group column index out of range"));
5979 }
5980 let Some(summary) = &self.table.host_groups else { return Ok(None) };
5981 if summary.column != column || minimum_count <= summary.omitted_max {
5982 return Ok(None);
5983 }
5984 Ok(Some(summary.entries.clone()))
5985 }
5986
5987 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
5996 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
5997 if let Some(dictionary) = self.dictionaries[column].get() {
5998 return Ok(Some(Arc::clone(dictionary)));
5999 }
6000 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
6001 if let Some(dictionary) = self.dictionaries[column].get() {
6002 return Ok(Some(Arc::clone(dictionary)));
6003 }
6004 self.opened.fetch_add(1, Atomic::Relaxed);
6005 let dictionary = Arc::new(open_global_dictionary(
6006 Arc::clone(&self.file),
6007 page,
6008 &self.table.fields[column].ty,
6009 TEXT_KEEP_BUDGET,
6010 )?);
6011 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
6012 Ok(Some(dictionary))
6013 }
6014
6015 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
6022 if of.extent_bytes == 0 {
6023 return Ok(Vec::new());
6024 }
6025 let mut bytes = vec![0; of.extent_bytes as usize];
6026 read_at(&self.file, of.extent_page, &mut bytes)?;
6027 if checksum(&bytes) != of.hash {
6028 return Err(invalid("a section's extent table does not checksum"));
6029 }
6030 let extents = section::decode_extents(&bytes)?;
6031 if extents.len() != of.extents as usize {
6032 return Err(invalid("a section's extent table is not the length the entry says"));
6033 }
6034 Ok(extents)
6035 }
6036
6037 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
6047 let end = of
6048 .offset
6049 .checked_add(u64::from(of.length))
6050 .ok_or_else(|| invalid("an extent overflows the file"))?;
6051 if of.offset < HEADER || end > self.size {
6052 return Err(invalid("an extent is outside the file"));
6053 }
6054 let mut bytes = vec![0; of.length as usize];
6055 read_at(&self.file, of.offset, &mut bytes)?;
6056 if checksum(&bytes) != of.hash {
6057 return Err(invalid("an extent does not checksum"));
6058 }
6059 Ok(bytes)
6060 }
6061
6062 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
6071 let extents = self.extents(of)?;
6072 let mut bytes =
6073 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
6074 for one in &extents {
6075 if one.first != bytes.len() as u64 {
6076 return Err(invalid("a section's extents do not join up"));
6077 }
6078 bytes.extend_from_slice(&self.extent(one)?);
6079 }
6080 if !bytes.is_empty() && of.header_bytes as usize > bytes.len() {
6083 return Err(invalid("a section's header is longer than its payload"));
6084 }
6085 Ok(bytes)
6086 }
6087
6088 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6097 self.read_impl(part, columns, true, None)
6098 }
6099
6100 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
6110 self.read_impl(part, columns, false, None)
6111 }
6112
6113 pub fn read_rows(
6126 &self,
6127 part: usize,
6128 columns: &[usize],
6129 positions: &[u32],
6130 whole: bool,
6131 ) -> Result<Chunk> {
6132 self.read_impl(part, columns, whole, Some(positions))
6133 }
6134
6135 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
6142 if candidates.is_empty() {
6143 return Ok(true);
6144 }
6145 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
6146 return Err(Error::internal("native code candidates are not sorted and unique"));
6147 }
6148 let stripe = self.stripe_of(part)?;
6149 let Some(page) = stripe.memberships.get(column) else {
6150 return Ok(false);
6151 };
6152 let mut bytes = vec![0; page.length as usize];
6153 read_at(&self.file, page.offset, &mut bytes)?;
6154 if checksum(&bytes) != page.hash {
6155 return Err(invalid("membership page checksum differs"));
6156 }
6157 let codes = decode_membership(&bytes)?;
6158 let mut left = 0;
6159 let mut right = 0;
6160 while left < codes.len() && right < candidates.len() {
6161 match codes[left].cmp(&candidates[right]) {
6162 Ordering::Less => left += 1,
6163 Ordering::Greater => right += 1,
6164 Ordering::Equal => return Ok(false),
6165 }
6166 }
6167 Ok(true)
6168 }
6169
6170 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
6171 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
6172 self.table
6173 .stripes
6174 .get(place.stripe as usize)
6175 .ok_or_else(|| invalid("stripe index out of range"))
6176 }
6177
6178 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
6195 let cache =
6196 self.cache.columns.get(column).ok_or_else(|| invalid("column index out of range"))?;
6197 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6198 let known = cached.index.get(at).and_then(Clone::clone);
6199 let page = cached.pages.get(at).and_then(Option::as_ref).map(|slot| {
6200 slot.used.store(true, Atomic::Relaxed);
6201 Arc::clone(&slot.page)
6202 });
6203 if let Some(index) = known.clone() {
6204 if !whole || page.is_some() {
6205 return Ok(CachedColumn { stripe: at, index, page });
6206 }
6207 }
6208 if cached.loading.contains(&at) {
6209 drop(cached);
6210 if let Some(index) = known {
6214 return Ok(CachedColumn { stripe: at, index, page: None });
6215 }
6216 let held = self.page_of(stripe, column, at, false, None)?;
6217 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6218 remember(&mut cached, &held);
6219 return Ok(held);
6220 }
6221 cached.loading.push(at);
6222 drop(cached);
6223
6224 let read = self.page_of(stripe, column, at, whole, known);
6225
6226 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
6230 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
6231 cached.loading.remove(position);
6232 }
6233 let held = read?;
6234 let taken = remember(&mut cached, &held);
6235 drop(cached);
6236 if let Some((bytes, used)) = taken {
6237 self.cache.held[column].fetch_add(1, Atomic::Relaxed);
6238 self.pool.admit(Held {
6239 shelf: Arc::downgrade(&self.cache),
6240 column,
6241 stripe: at,
6242 bytes,
6243 used,
6244 });
6245 }
6246 Ok(held)
6247 }
6248
6249 fn page_of(
6255 &self,
6256 stripe: &Stripe,
6257 column: usize,
6258 at: usize,
6259 whole: bool,
6260 known: Option<Arc<Vec<PartSpan>>>,
6261 ) -> Result<CachedColumn> {
6262 let index = match known {
6263 Some(index) => index,
6264 None => {
6265 self.indexes.fetch_add(1, Atomic::Relaxed);
6266 Arc::new(read_index(&self.file, stripe, column)?)
6267 }
6268 };
6269 let page = if whole {
6270 self.pages.fetch_add(1, Atomic::Relaxed);
6271 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6272 let mut bytes = vec![0; span.length as usize];
6273 read_at(&self.file, span.offset, &mut bytes)?;
6274 Some(Arc::new(bytes))
6275 } else {
6276 None
6277 };
6278 Ok(CachedColumn { stripe: at, index, page })
6279 }
6280
6281 fn read_impl(
6282 &self,
6283 at: usize,
6284 columns: &[usize],
6285 whole: bool,
6286 positions: Option<&[u32]>,
6287 ) -> Result<Chunk> {
6288 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
6289 let index = place.stripe as usize;
6290 let stripe =
6291 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
6292 let rows = place.rows as usize;
6293 let mut picked = Vec::with_capacity(columns.len());
6294 for &column in columns {
6295 let field = self
6296 .table
6297 .fields
6298 .get(column)
6299 .ok_or_else(|| invalid("column index out of range"))?;
6300 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
6301 let held = self.held(index, stripe, column, whole)?;
6302 let span = *held
6303 .index
6304 .get(place.part as usize)
6305 .ok_or_else(|| invalid("part index out of range"))?;
6306 let owned;
6307 let bytes = match &held.page {
6308 Some(held) => part_bytes(held, span)?,
6309 None => {
6310 let offset = page
6311 .offset
6312 .checked_add(span.start as u64)
6313 .ok_or_else(|| invalid("part range overflow"))?;
6314 let mut bytes = vec![0; span.length];
6315 read_at(&self.file, offset, &mut bytes)?;
6316 owned = bytes;
6317 &owned
6318 }
6319 };
6320 if checksum(bytes) != span.hash {
6321 return Err(invalid(&format!(
6322 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
6323 wanted {:016x} and got {:016x}",
6324 place.part,
6325 page.offset,
6326 span.start,
6327 span.length,
6328 span.hash,
6329 checksum(bytes),
6330 )));
6331 }
6332 let dictionary = self.dictionary(column)?;
6333 let vector = match positions {
6339 None => decode(&field.ty, rows, bytes, dictionary)?,
6340 Some(positions) => decode_at(&field.ty, rows, bytes, dictionary, positions)?,
6341 };
6342 picked.push(vector.into_pages());
6343 }
6344 Chunk::with_rows(picked, positions.map_or(rows, <[u32]>::len))
6345 }
6346
6347 #[must_use]
6363 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
6364 let Some(place) = self.places.get(part).copied() else { return false };
6365 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6366 if stripe.zone.skips(probes) {
6367 return true;
6368 }
6369 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
6370 }
6371
6372 fn outside(&self, place: Place, probe: &Probe) -> bool {
6378 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6379 Some(ranges) => ranges
6380 .get(place.part as usize)
6381 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
6382 None => false,
6383 }
6384 }
6385
6386 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
6392 let slot = self.part_ranges.get(column)?.get(stripe)?;
6393 if let Some(held) = slot.get() {
6394 return Some(held);
6395 }
6396 let page = self.table.stripes.get(stripe)?.part_ranges.get(column)?;
6397 let mut bytes = vec![0; page.length as usize];
6398 read_at(&self.file, page.offset, &mut bytes).ok()?;
6399 if checksum(&bytes) != page.hash {
6400 return None;
6401 }
6402 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
6403 let _ = slot.set(ranges);
6404 slot.get().map(|held| held.as_slice())
6405 }
6406
6407 #[must_use]
6424 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
6425 let Some(place) = self.places.get(part).copied() else { return false };
6426 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
6427 if stripe.zone.certain(probes) {
6428 return true;
6429 }
6430 probes
6431 .iter()
6432 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
6433 }
6434
6435 fn inside(&self, place: Place, probe: &Probe) -> bool {
6441 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
6442 Some(ranges) => ranges
6443 .get(place.part as usize)
6444 .is_some_and(|range| range.certain(probe.op, &probe.value)),
6445 None => false,
6446 }
6447 }
6448
6449 #[must_use]
6460 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
6461 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
6462 }
6463
6464 fn sifted(&self, place: Place, probe: &Probe) -> bool {
6470 if probe.op != Op::Equal {
6471 return false;
6472 }
6473 match self.stripe_sieves(place.stripe as usize, probe.column) {
6474 Some(sieves) => sieves
6475 .get(place.part as usize)
6476 .and_then(Option::as_ref)
6477 .is_some_and(|sieve| sieve.excludes(&probe.value)),
6478 None => false,
6479 }
6480 }
6481
6482 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
6489 let slot = self.sieves.get(column)?.get(stripe)?;
6490 if let Some(held) = slot.get() {
6491 return Some(held);
6492 }
6493 let page = self.table.stripes.get(stripe)?.sieves.get(column)?;
6494 let mut bytes = vec![0; page.length as usize];
6495 read_at(&self.file, page.offset, &mut bytes).ok()?;
6496 if checksum(&bytes) != page.hash {
6497 return None;
6498 }
6499 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
6500 let _ = slot.set(sieves);
6501 slot.get().map(|held| held.as_slice())
6502 }
6503}
6504
6505fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
6507 let code = dictionary.code_at_rank(rank)? as usize;
6508 let text = dictionary
6509 .try_text_at(code)?
6510 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
6511 Ok(Value::Varchar(text.into()))
6512}
6513
6514#[cfg(unix)]
6519fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6520 use std::os::unix::fs::FileExt;
6521 while !bytes.is_empty() {
6522 let written = file.write_at(bytes, offset).map_err(io)?;
6523 if written == 0 {
6524 return Err(invalid("a write to the native file wrote nothing"));
6525 }
6526 offset += written as u64;
6527 bytes = &bytes[written..];
6528 }
6529 Ok(())
6530}
6531
6532#[cfg(windows)]
6534fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
6535 use std::os::windows::fs::FileExt;
6536 while !bytes.is_empty() {
6537 let written = file.seek_write(bytes, offset).map_err(io)?;
6538 if written == 0 {
6539 return Err(invalid("a write to the native file wrote nothing"));
6540 }
6541 offset += written as u64;
6542 bytes = &bytes[written..];
6543 }
6544 Ok(())
6545}
6546
6547#[cfg(not(any(unix, windows)))]
6549fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
6550 use std::io::Write;
6551 let mut file = file.try_clone().map_err(io)?;
6552 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6553 file.write_all(bytes).map_err(io)
6554}
6555
6556#[cfg(unix)]
6566fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6567 use std::os::unix::fs::FileExt;
6568 while !bytes.is_empty() {
6569 let read = file.read_at(bytes, offset).map_err(io)?;
6570 if read == 0 {
6571 return Err(invalid("column page ends before its declared length"));
6572 }
6573 offset += read as u64;
6574 bytes = &mut bytes[read..];
6575 }
6576 Ok(())
6577}
6578
6579#[cfg(windows)]
6585fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
6586 use std::os::windows::fs::FileExt;
6587 while !bytes.is_empty() {
6588 let read = file.seek_read(bytes, offset).map_err(io)?;
6589 if read == 0 {
6590 return Err(invalid("column page ends before its declared length"));
6591 }
6592 offset += read as u64;
6593 bytes = &mut bytes[read..];
6594 }
6595 Ok(())
6596}
6597
6598#[cfg(not(any(unix, windows)))]
6603fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
6604 let mut file = file.try_clone().map_err(io)?;
6605 file.seek(SeekFrom::Start(offset)).map_err(io)?;
6606 file.read_exact(bytes).map_err(io)
6607}
6608
6609fn type_tag(ty: &LogicalType) -> Result<u8> {
6616 match ty {
6617 LogicalType::SmallInt => Ok(1),
6618 LogicalType::Integer => Ok(2),
6619 LogicalType::BigInt => Ok(3),
6620 LogicalType::Varchar => Ok(4),
6621 LogicalType::Date => Ok(5),
6622 LogicalType::Timestamp => Ok(6),
6623 LogicalType::Boolean => Ok(7),
6624 LogicalType::TinyInt => Ok(8),
6625 LogicalType::UTinyInt => Ok(9),
6626 LogicalType::USmallInt => Ok(10),
6627 LogicalType::UInteger => Ok(11),
6628 LogicalType::UBigInt => Ok(12),
6629 LogicalType::Decimal { .. } => Ok(13),
6630 LogicalType::Float => Ok(14),
6631 LogicalType::Double => Ok(15),
6632 LogicalType::HugeInt => Ok(16),
6633 LogicalType::UHugeInt => Ok(17),
6634 LogicalType::Time => Ok(18),
6635 LogicalType::TimeTz => Ok(19),
6636 LogicalType::TimestampTz => Ok(20),
6637 LogicalType::Interval => Ok(21),
6638 LogicalType::Uuid => Ok(22),
6639 LogicalType::Blob => Ok(23),
6640 LogicalType::Bit => Ok(24),
6641 LogicalType::TimestampS => Ok(25),
6642 LogicalType::TimestampMs => Ok(26),
6643 LogicalType::TimestampNs => Ok(27),
6644 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
6645 }
6646}
6647
6648fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
6654 out.push(type_tag(ty)?);
6655 if let LogicalType::Decimal { width, scale } = ty {
6656 out.push(*width);
6657 out.push(*scale);
6658 }
6659 Ok(())
6660}
6661
6662fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
6664 let tag = cur.u8()?;
6665 if tag == 13 {
6666 let width = cur.u8()?;
6667 let scale = cur.u8()?;
6668 return LogicalType::decimal(width, scale)
6669 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
6670 }
6671 tag_type(tag)
6672}
6673
6674fn tag_type(tag: u8) -> Result<LogicalType> {
6675 match tag {
6676 1 => Ok(LogicalType::SmallInt),
6677 2 => Ok(LogicalType::Integer),
6678 3 => Ok(LogicalType::BigInt),
6679 4 => Ok(LogicalType::Varchar),
6680 5 => Ok(LogicalType::Date),
6681 6 => Ok(LogicalType::Timestamp),
6682 7 => Ok(LogicalType::Boolean),
6683 8 => Ok(LogicalType::TinyInt),
6684 9 => Ok(LogicalType::UTinyInt),
6685 10 => Ok(LogicalType::USmallInt),
6686 11 => Ok(LogicalType::UInteger),
6687 12 => Ok(LogicalType::UBigInt),
6688 14 => Ok(LogicalType::Float),
6689 15 => Ok(LogicalType::Double),
6690 16 => Ok(LogicalType::HugeInt),
6691 17 => Ok(LogicalType::UHugeInt),
6692 18 => Ok(LogicalType::Time),
6693 19 => Ok(LogicalType::TimeTz),
6694 20 => Ok(LogicalType::TimestampTz),
6695 21 => Ok(LogicalType::Interval),
6696 22 => Ok(LogicalType::Uuid),
6697 23 => Ok(LogicalType::Blob),
6698 24 => Ok(LogicalType::Bit),
6699 25 => Ok(LogicalType::TimestampS),
6700 26 => Ok(LogicalType::TimestampMs),
6701 27 => Ok(LogicalType::TimestampNs),
6702 _ => Err(invalid("column type tag is unknown")),
6703 }
6704}
6705
6706fn put_u16(out: &mut Vec<u8>, value: u16) {
6707 out.extend_from_slice(&value.to_le_bytes());
6708}
6709fn put_u32(out: &mut Vec<u8>, value: u32) {
6710 out.extend_from_slice(&value.to_le_bytes());
6711}
6712fn put_u64(out: &mut Vec<u8>, value: u64) {
6713 out.extend_from_slice(&value.to_le_bytes());
6714}
6715fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
6716 while value >= 0x80 {
6717 out.push((value as u8 & 0x7f) | 0x80);
6718 value >>= 7;
6719 }
6720 out.push(value as u8);
6721}
6722
6723fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
6724 match (left, right) {
6725 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
6726 (FrequencyValue::Null, _) => Ordering::Less,
6727 (_, FrequencyValue::Null) => Ordering::Greater,
6728 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
6729 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
6730 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
6731 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
6732 }
6733}
6734
6735fn keep_most_frequent(entries: &mut Vec<FrequencyEntry>) -> u64 {
6748 let order = |left: &FrequencyEntry, right: &FrequencyEntry| {
6749 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
6750 };
6751 let omitted_max = if entries.len() > FREQUENCY_ENTRIES {
6752 let (_, next, _) = entries.select_nth_unstable_by(FREQUENCY_ENTRIES, order);
6753 let omitted_max = next.count;
6754 entries.truncate(FREQUENCY_ENTRIES);
6755 omitted_max
6756 } else {
6757 0
6758 };
6759 entries.sort_unstable_by(order);
6760 omitted_max
6761}
6762
6763fn code_frequency(
6764 dictionary: &GlobalDictionary,
6765 flat: &[u8],
6766 bases: &[u64],
6767) -> Result<(FrequencySummary, Vec<Option<Vec<u8>>>)> {
6768 let mut entries = dictionary
6769 .counts
6770 .iter()
6771 .enumerate()
6772 .filter(|(_, count)| **count != 0)
6773 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
6774 .collect::<Vec<_>>();
6775 if dictionary.nulls != 0 {
6776 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
6777 }
6778 let omitted_max = keep_most_frequent(&mut entries);
6779 let mut spans = Vec::with_capacity(entries.len());
6780 let mut text_bytes = 0_usize;
6781 for entry in &entries {
6782 let span = match entry.value {
6783 FrequencyValue::Code(code) => {
6784 let span = GlobalDictionary::value_span(&dictionary.ends, bases, code as usize);
6785 let bytes = flat
6786 .get(span.0..span.1)
6787 .ok_or_else(|| invalid("a frequency code is outside its dictionary"))?;
6788 text_bytes = text_bytes.saturating_add(bytes.len());
6789 Some(span)
6790 }
6791 FrequencyValue::Null | FrequencyValue::Integer(_) => None,
6792 };
6793 spans.push(span);
6794 }
6795 let texts = if text_bytes > FREQUENCY_TEXT_BUDGET {
6796 Vec::new()
6797 } else {
6798 spans.into_iter().map(|span| span.map(|(from, to)| flat[from..to].to_vec())).collect()
6799 };
6800 Ok((
6801 FrequencySummary {
6802 entries,
6803 omitted_max,
6804 ordinals: Vec::new(),
6805 ordinal_entries: Vec::new(),
6806 },
6807 texts,
6808 ))
6809}
6810
6811fn encode_directory(table: &Table) -> Result<Vec<u8>> {
6812 let mut out = DIRECTORY.to_vec();
6813 let name = table.name.as_bytes();
6814 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
6815 out.extend_from_slice(name);
6816 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
6817 for field in &table.fields {
6818 let name = field.name.as_bytes();
6819 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
6820 out.extend_from_slice(name);
6821 put_type(&mut out, &field.ty)?;
6822 out.push(u8::from(field.not_null));
6823 }
6824 for dictionary in &table.dictionaries {
6825 match dictionary {
6826 None => out.push(0),
6827 Some(page) => {
6828 out.push(1);
6829 put_u64(&mut out, page.offset);
6830 put_u32(&mut out, page.length);
6831 put_u64(&mut out, page.hash);
6832 }
6833 }
6834 }
6835 for distinct in &table.distincts {
6836 match distinct {
6837 None => out.push(0),
6838 Some(count) => {
6839 out.push(1);
6840 put_u64(&mut out, *count);
6841 }
6842 }
6843 }
6844 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
6845 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
6846 for stripe in &table.stripes {
6847 put_u32(
6848 &mut out,
6849 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6850 );
6851 for &rows in &stripe.parts {
6852 put_u32(&mut out, rows);
6853 }
6854 put_u64(&mut out, stripe.index.offset);
6855 put_u32(&mut out, stripe.index.length);
6856 for page in &stripe.pages {
6857 put_u64(&mut out, page.offset);
6858 put_u32(&mut out, page.length);
6859 }
6860 for ((field, dictionary), membership) in
6865 table.fields.iter().zip(&table.dictionaries).zip(stripe.memberships.slots())
6866 {
6867 if field.ty != LogicalType::Varchar || dictionary.is_none() {
6868 continue;
6869 }
6870 let page =
6871 membership.ok_or_else(|| invalid("string page has no code membership index"))?;
6872 put_u64(&mut out, page.offset);
6873 put_u32(&mut out, page.length);
6874 put_u64(&mut out, page.hash);
6875 }
6876 for sieve in stripe.sieves.slots() {
6877 match sieve {
6878 None => out.push(0),
6879 Some(page) => {
6880 out.push(1);
6881 put_u64(&mut out, page.offset);
6882 put_u32(&mut out, page.length);
6883 put_u64(&mut out, page.hash);
6884 }
6885 }
6886 }
6887 for held in stripe.part_ranges.slots() {
6888 match held {
6889 None => out.push(0),
6890 Some(page) => {
6891 out.push(1);
6892 put_u64(&mut out, page.offset);
6893 put_u32(&mut out, page.length);
6894 put_u64(&mut out, page.hash);
6895 }
6896 }
6897 }
6898 for range in stripe.zone.columns() {
6899 put_bound(&mut out, range.low.as_ref())?;
6900 put_bound(&mut out, range.high.as_ref())?;
6901 put_u32(
6902 &mut out,
6903 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
6904 );
6905 out.push(u8::from(range.exact));
6906 match range.sum {
6907 None => out.push(0),
6908 Some(total) => {
6909 out.push(1);
6910 out.extend_from_slice(&total.to_le_bytes());
6911 }
6912 }
6913 }
6914 }
6915 out.extend_from_slice(FREQUENCIES);
6916 put_u16(
6917 &mut out,
6918 u16::try_from(table.frequencies.len())
6919 .map_err(|_| invalid("too many frequency columns"))?,
6920 );
6921 for summary in &table.frequencies {
6922 let summary = match summary {
6923 None => {
6924 out.push(0);
6925 continue;
6926 }
6927 Some(Frequencies::Held(summary)) => summary,
6928 Some(Frequencies::Stored { .. }) => {
6930 return Err(invalid("a synopsis left in the file cannot be written back"));
6931 }
6932 };
6933 out.push(1);
6934 put_u64(&mut out, summary.omitted_max);
6935 put_u32(
6936 &mut out,
6937 u32::try_from(summary.entries.len())
6938 .map_err(|_| invalid("too many frequency entries"))?,
6939 );
6940 for entry in &summary.entries {
6941 match entry.value {
6942 FrequencyValue::Null => out.push(0),
6943 FrequencyValue::Integer(value) => {
6944 out.push(1);
6945 out.extend_from_slice(&value.to_le_bytes());
6946 }
6947 FrequencyValue::Code(value) => {
6948 out.push(2);
6949 put_u32(&mut out, value);
6950 }
6951 }
6952 put_u64(&mut out, entry.count);
6953 }
6954 put_u32(
6955 &mut out,
6956 u32::try_from(summary.ordinals.len())
6957 .map_err(|_| invalid("too many frequency ordinals"))?,
6958 );
6959 let mut previous = 0_u64;
6960 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
6961 let delta = if at == 0 {
6962 ordinal
6963 } else {
6964 ordinal
6965 .checked_sub(previous)
6966 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
6967 };
6968 if at != 0 && delta == 0 {
6969 return Err(invalid("frequency ordinals are not unique"));
6970 }
6971 put_var_u64(&mut out, delta);
6972 previous = ordinal;
6973 }
6974 if summary.ordinal_entries.len() != summary.ordinals.len() {
6975 return Err(invalid("frequency ordinal values have a different length"));
6976 }
6977 for &entry in &summary.ordinal_entries {
6978 if entry as usize >= summary.entries.len() {
6979 return Err(invalid("frequency ordinal value is outside its entries"));
6980 }
6981 put_u16(&mut out, entry);
6982 }
6983 }
6984 if !table.pair_frequencies.is_empty() {
6985 out.extend_from_slice(PAIR_FREQUENCIES);
6986 put_u16(
6987 &mut out,
6988 u16::try_from(table.pair_frequencies.len())
6989 .map_err(|_| invalid("too many pair frequency summaries"))?,
6990 );
6991 for summary in &table.pair_frequencies {
6992 put_u16(&mut out, summary.first);
6993 put_u16(&mut out, summary.second);
6994 put_u64(&mut out, summary.omitted_max);
6995 put_u16(
6996 &mut out,
6997 u16::try_from(summary.entries.len())
6998 .map_err(|_| invalid("too many pair frequency entries"))?,
6999 );
7000 for entry in &summary.entries {
7001 put_u16(&mut out, entry.first_entry);
7002 match entry.second {
7003 None => out.push(0),
7004 Some(code) => {
7005 out.push(1);
7006 put_u32(&mut out, code);
7007 }
7008 }
7009 put_u64(&mut out, entry.count);
7010 }
7011 }
7012 }
7013 let text_columns = table.frequency_texts.iter().filter(|texts| !texts.is_empty()).count();
7014 if text_columns != 0 {
7015 out.extend_from_slice(FREQUENCY_TEXTS);
7016 put_u16(
7017 &mut out,
7018 u16::try_from(text_columns)
7019 .map_err(|_| invalid("too many string frequency columns"))?,
7020 );
7021 for (column, texts) in table.frequency_texts.iter().enumerate() {
7022 if texts.is_empty() {
7023 continue;
7024 }
7025 put_u16(
7026 &mut out,
7027 u16::try_from(column).map_err(|_| invalid("frequency text column overflows"))?,
7028 );
7029 put_u16(
7030 &mut out,
7031 u16::try_from(texts.len())
7032 .map_err(|_| invalid("too many frequency text entries"))?,
7033 );
7034 for text in texts {
7035 match text {
7036 None => out.push(0),
7037 Some(text) => {
7038 out.push(1);
7039 put_u32(
7040 &mut out,
7041 u32::try_from(text.len())
7042 .map_err(|_| invalid("frequency text is too long"))?,
7043 );
7044 out.extend_from_slice(text);
7045 }
7046 }
7047 }
7048 }
7049 }
7050 if let Some(summary) = &table.host_groups {
7051 out.extend_from_slice(HOST_GROUPS);
7052 put_u16(
7053 &mut out,
7054 u16::try_from(summary.column).map_err(|_| invalid("host column overflows"))?,
7055 );
7056 put_u64(&mut out, summary.omitted_max);
7057 put_u16(
7058 &mut out,
7059 u16::try_from(summary.entries.len()).map_err(|_| invalid("too many host groups"))?,
7060 );
7061 for entry in &summary.entries {
7062 put_u32(
7063 &mut out,
7064 u32::try_from(entry.host.len()).map_err(|_| invalid("host name is too long"))?,
7065 );
7066 out.extend_from_slice(entry.host.as_bytes());
7067 put_u64(&mut out, entry.count);
7068 out.extend_from_slice(&entry.bytes_sum.to_le_bytes());
7069 put_u32(
7070 &mut out,
7071 u32::try_from(entry.minimum.len())
7072 .map_err(|_| invalid("host minimum is too long"))?,
7073 );
7074 out.extend_from_slice(entry.minimum.as_bytes());
7075 }
7076 }
7077 if let Some(clustering) = &table.clustering {
7080 out.extend_from_slice(CLUSTERING);
7081 out.push(clustering.width().tag());
7082 put_u16(
7083 &mut out,
7084 u16::try_from(clustering.columns().len())
7085 .map_err(|_| invalid("too many clustering columns"))?,
7086 );
7087 for &column in clustering.columns() {
7088 put_u16(
7089 &mut out,
7090 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
7091 );
7092 }
7093 }
7094 out.extend_from_slice(SECTIONS);
7100 put_u64(&mut out, table.generation);
7101 put_u16(
7102 &mut out,
7103 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
7104 );
7105 for held in &table.sections {
7106 held.encode(&mut out)?;
7107 }
7108 if table.dictionary_payloads.iter().any(|&bytes| bytes != 0) {
7109 out.extend_from_slice(DICTIONARY_PAYLOADS);
7110 put_u16(
7111 &mut out,
7112 u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?,
7113 );
7114 for at in 0..table.fields.len() {
7115 put_u64(&mut out, table.dictionary_payloads.get(at).copied().unwrap_or(0));
7116 }
7117 }
7118 Ok(out)
7119}
7120
7121fn table_nonzero_counts(table: &Table) -> Vec<Option<u64>> {
7130 table
7131 .fields
7132 .iter()
7133 .enumerate()
7134 .map(|(column, field)| {
7135 if !matches!(
7136 field.ty,
7137 LogicalType::TinyInt
7138 | LogicalType::SmallInt
7139 | LogicalType::Integer
7140 | LogicalType::BigInt
7141 | LogicalType::UTinyInt
7142 | LogicalType::USmallInt
7143 | LogicalType::UInteger
7144 | LogicalType::UBigInt
7145 ) {
7146 return None;
7147 }
7148 let Some(Frequencies::Held(summary)) = &table.frequencies[column] else {
7149 return None;
7150 };
7151 let zero = summary
7152 .entries
7153 .iter()
7154 .find(|entry| entry.value == FrequencyValue::Integer(0))
7155 .map(|entry| entry.count)
7156 .or_else(|| (summary.omitted_max == 0).then_some(0))?;
7157 let nulls = table.stripes.iter().try_fold(0_u64, |count, stripe| {
7158 count.checked_add(stripe.zone.column(column)?.nulls as u64)
7159 })?;
7160 (table.rows as u64).checked_sub(nulls)?.checked_sub(zero)
7161 })
7162 .collect()
7163}
7164
7165fn signed_integer(ty: &LogicalType) -> bool {
7166 matches!(
7167 ty,
7168 LogicalType::TinyInt | LogicalType::SmallInt | LogicalType::Integer | LogicalType::BigInt
7169 )
7170}
7171
7172fn integer_or_date(ty: &LogicalType) -> bool {
7173 matches!(
7174 ty,
7175 LogicalType::TinyInt
7176 | LogicalType::SmallInt
7177 | LogicalType::Integer
7178 | LogicalType::BigInt
7179 | LogicalType::UTinyInt
7180 | LogicalType::USmallInt
7181 | LogicalType::UInteger
7182 | LogicalType::UBigInt
7183 | LogicalType::Date
7184 )
7185}
7186
7187fn table_integer_extremes(table: &Table) -> Vec<StoredIntegerExtremes> {
7188 table
7189 .fields
7190 .iter()
7191 .enumerate()
7192 .map(|(column, field)| {
7193 if !integer_or_date(&field.ty) {
7194 return None;
7195 }
7196 let mut low: Option<i128> = None;
7197 let mut high: Option<i128> = None;
7198 for stripe in &table.stripes {
7199 let range = stripe.zone.column(column)?;
7200 if !range.exact {
7201 return None;
7202 }
7203 match (range.low.as_ref(), range.high.as_ref()) {
7204 (Some(Bound::Int(small)), Some(Bound::Int(large))) => {
7205 low = Some(low.map_or(*small, |held| held.min(*small)));
7206 high = Some(high.map_or(*large, |held| held.max(*large)));
7207 }
7208 (None, None) if stripe.rows == range.nulls => {}
7209 _ => return None,
7210 }
7211 }
7212 Some(low.zip(high))
7213 })
7214 .collect()
7215}
7216
7217fn reader_integer_extremes(reader: &Reader) -> Result<Vec<StoredIntegerExtremes>> {
7218 reader
7219 .table
7220 .fields
7221 .iter()
7222 .enumerate()
7223 .map(|(column, field)| {
7224 if !integer_or_date(&field.ty) {
7225 return Ok(None);
7226 }
7227 match reader.exact_extremes(column)? {
7228 Some((Bound::Int(low), Bound::Int(high))) => Ok(Some(Some((low, high)))),
7229 None if reader.null_count(column)? == reader.table.rows as u64 => Ok(Some(None)),
7230 _ => Ok(None),
7231 }
7232 })
7233 .collect()
7234}
7235
7236fn table_complete_numeric_frequencies(table: &Table) -> Vec<StoredNumericFrequencies> {
7237 table
7238 .fields
7239 .iter()
7240 .enumerate()
7241 .map(|(column, field)| {
7242 if !integer_or_date(&field.ty) {
7243 return None;
7244 }
7245 let Some(Frequencies::Held(summary)) = table.frequencies.get(column)?.as_ref() else {
7246 return None;
7247 };
7248 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7249 return None;
7250 }
7251 let entries = summary
7252 .entries
7253 .iter()
7254 .map(|entry| {
7255 let value = match entry.value {
7256 FrequencyValue::Null => None,
7257 FrequencyValue::Integer(value) => Some(value),
7258 FrequencyValue::Code(_) => return None,
7259 };
7260 Some((value, entry.count))
7261 })
7262 .collect::<Option<Vec<_>>>()?;
7263 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count))?;
7264 (rows == table.rows as u64).then_some(entries)
7265 })
7266 .collect()
7267}
7268
7269fn numeric_frequency_value(value: &Value) -> Option<Option<i128>> {
7270 Some(match value {
7271 Value::Null => None,
7272 Value::TinyInt(value) => Some(i128::from(*value)),
7273 Value::SmallInt(value) => Some(i128::from(*value)),
7274 Value::Integer(value) | Value::Date(value) => Some(i128::from(*value)),
7275 Value::BigInt(value) => Some(i128::from(*value)),
7276 Value::UTinyInt(value) => Some(i128::from(*value)),
7277 Value::USmallInt(value) => Some(i128::from(*value)),
7278 Value::UInteger(value) => Some(i128::from(*value)),
7279 Value::UBigInt(value) => Some(i128::from(*value)),
7280 _ => return None,
7281 })
7282}
7283
7284fn reader_complete_numeric_frequencies(reader: &Reader) -> Result<Vec<StoredNumericFrequencies>> {
7285 reader
7286 .table
7287 .fields
7288 .iter()
7289 .enumerate()
7290 .map(|(column, field)| {
7291 if !integer_or_date(&field.ty) {
7292 return Ok(None);
7293 }
7294 let Some(summary) = reader.frequency_summary(column)? else { return Ok(None) };
7295 if summary.omitted_max != 0 || summary.entries.len() > MAX_CATALOG_FREQUENCIES {
7296 return Ok(None);
7297 }
7298 let entries = reader.decode_frequencies(column, &field.ty, &summary.entries)?;
7299 let Some(entries) = entries
7300 .iter()
7301 .map(|(value, count)| Some((numeric_frequency_value(value)?, *count)))
7302 .collect::<Option<Vec<_>>>()
7303 else {
7304 return Ok(None);
7305 };
7306 let rows = entries.iter().try_fold(0_u64, |sum, (_, count)| sum.checked_add(*count));
7307 Ok((rows == Some(reader.table.rows as u64)).then_some(entries))
7308 })
7309 .collect()
7310}
7311
7312fn table_exact_sum(table: &Table, column: usize) -> Option<(i128, u64)> {
7313 table.stripes.iter().try_fold((0_i128, 0_u64), |(sum, count), stripe| {
7314 let range = stripe.zone.column(column)?;
7315 let sum = sum.checked_add(range.sum?)?;
7316 let nonnull = (stripe.rows as u64).checked_sub(range.nulls as u64)?;
7317 Some((sum, count.checked_add(nonnull)?))
7318 })
7319}
7320
7321fn table_aggregate_sums(table: &Table) -> Vec<Option<(i128, u64)>> {
7322 table
7323 .fields
7324 .iter()
7325 .enumerate()
7326 .map(|(column, field)| {
7327 signed_integer(&field.ty).then(|| table_exact_sum(table, column)).flatten()
7328 })
7329 .collect()
7330}
7331
7332fn reader_nonzero_counts(reader: &Reader) -> Result<Vec<Option<u64>>> {
7333 reader
7334 .table
7335 .fields
7336 .iter()
7337 .enumerate()
7338 .map(|(column, field)| {
7339 if !matches!(
7340 field.ty,
7341 LogicalType::TinyInt
7342 | LogicalType::SmallInt
7343 | LogicalType::Integer
7344 | LogicalType::BigInt
7345 | LogicalType::UTinyInt
7346 | LogicalType::USmallInt
7347 | LogicalType::UInteger
7348 | LogicalType::UBigInt
7349 ) {
7350 return Ok(None);
7351 }
7352 let Some(summary) = reader.frequency_summary(column)? else {
7353 return Ok(None);
7354 };
7355 let zero = summary
7356 .entries
7357 .iter()
7358 .find(|entry| entry.value == FrequencyValue::Integer(0))
7359 .map(|entry| entry.count)
7360 .or_else(|| (summary.omitted_max == 0).then_some(0));
7361 let Some(zero) = zero else { return Ok(None) };
7362 let nulls = reader.null_count(column)?;
7363 Ok((reader.table.rows as u64)
7364 .checked_sub(nulls)
7365 .and_then(|count| count.checked_sub(zero)))
7366 })
7367 .collect()
7368}
7369
7370fn reader_aggregate_sums(reader: &Reader) -> Result<Vec<Option<(i128, u64)>>> {
7371 reader
7372 .table
7373 .fields
7374 .iter()
7375 .enumerate()
7376 .map(
7377 |(column, field)| {
7378 if signed_integer(&field.ty) { reader.exact_sum(column) } else { Ok(None) }
7379 },
7380 )
7381 .collect()
7382}
7383
7384fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
7385 let mut out = CATALOG.to_vec();
7386 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
7387 for entry in entries {
7388 let name = entry.name.as_bytes();
7389 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
7390 out.extend_from_slice(name);
7391 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
7392 put_u16(
7393 &mut out,
7394 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
7395 );
7396 for field in &entry.fields {
7397 let name = field.name.as_bytes();
7398 put_u16(
7399 &mut out,
7400 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7401 );
7402 out.extend_from_slice(name);
7403 put_type(&mut out, &field.ty)?;
7404 out.push(u8::from(field.not_null));
7405 }
7406 put_u64(&mut out, entry.directory.offset);
7407 put_u32(&mut out, entry.directory.length);
7408 put_u64(&mut out, entry.directory.hash);
7409 }
7410 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
7411 for view in views {
7412 let name = view.name.as_bytes();
7413 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
7414 out.extend_from_slice(name);
7415 put_long_text(&mut out, &view.sql, "view body")?;
7416 put_long_text(&mut out, &view.statement, "view statement")?;
7417 put_u16(
7418 &mut out,
7419 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
7420 );
7421 for alias in &view.aliases {
7422 let alias = alias.as_bytes();
7423 put_u16(
7424 &mut out,
7425 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
7426 );
7427 out.extend_from_slice(alias);
7428 }
7429 put_u16(
7430 &mut out,
7431 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
7432 );
7433 for field in &view.columns {
7434 let name = field.name.as_bytes();
7435 put_u16(
7436 &mut out,
7437 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
7438 );
7439 out.extend_from_slice(name);
7440 put_type(&mut out, &field.ty)?;
7441 out.push(u8::from(field.not_null));
7442 }
7443 }
7444 out.extend_from_slice(NONZERO_COUNTS);
7445 for entry in entries {
7446 if entry.nonzero.len() != entry.fields.len() {
7447 return Err(invalid("nonzero count width differs from schema"));
7448 }
7449 for count in &entry.nonzero {
7450 match count {
7451 None => out.push(0),
7452 Some(count) => {
7453 out.push(1);
7454 put_u64(&mut out, *count);
7455 }
7456 }
7457 }
7458 }
7459 out.extend_from_slice(AGGREGATE_SUMS);
7460 for entry in entries {
7461 if entry.aggregates.len() != entry.fields.len() {
7462 return Err(invalid("aggregate sum width differs from schema"));
7463 }
7464 for summary in &entry.aggregates {
7465 match summary {
7466 None => out.push(0),
7467 Some((sum, count)) => {
7468 out.push(1);
7469 out.extend_from_slice(&sum.to_le_bytes());
7470 put_u64(&mut out, *count);
7471 }
7472 }
7473 }
7474 }
7475 out.extend_from_slice(DISTINCT_COUNTS);
7476 for entry in entries {
7477 if entry.distincts.len() != entry.fields.len() {
7478 return Err(invalid("distinct count width differs from schema"));
7479 }
7480 for count in &entry.distincts {
7481 match count {
7482 None => out.push(0),
7483 Some(count) => {
7484 if *count > entry.rows as u64 {
7485 return Err(invalid("distinct count exceeds table rows"));
7486 }
7487 out.push(1);
7488 put_u64(&mut out, *count);
7489 }
7490 }
7491 }
7492 }
7493 out.extend_from_slice(INTEGER_EXTREMES);
7494 for entry in entries {
7495 if entry.extremes.len() != entry.fields.len() {
7496 return Err(invalid("integer extremes width differs from schema"));
7497 }
7498 for (field, extremes) in entry.fields.iter().zip(&entry.extremes) {
7499 match extremes {
7500 None => out.push(0),
7501 Some(None) if integer_or_date(&field.ty) => out.push(1),
7502 Some(Some((low, high))) if integer_or_date(&field.ty) && low <= high => {
7503 out.push(2);
7504 out.extend_from_slice(&low.to_le_bytes());
7505 out.extend_from_slice(&high.to_le_bytes());
7506 }
7507 _ => return Err(invalid("integer extremes type or range differs")),
7508 }
7509 }
7510 }
7511 out.extend_from_slice(COMPLETE_FREQUENCIES);
7512 for entry in entries {
7513 if entry.frequencies.len() != entry.fields.len() {
7514 return Err(invalid("numeric frequency width differs from schema"));
7515 }
7516 for (field, frequencies) in entry.fields.iter().zip(&entry.frequencies) {
7517 match frequencies {
7518 None => out.push(0),
7519 Some(entries)
7520 if integer_or_date(&field.ty) && entries.len() <= MAX_CATALOG_FREQUENCIES =>
7521 {
7522 let mut total = 0_u64;
7523 for (at, (value, count)) in entries.iter().enumerate() {
7524 if entries[..at].iter().any(|(held, _)| held == value) {
7525 return Err(invalid("numeric frequency value repeats"));
7526 }
7527 total = total
7528 .checked_add(*count)
7529 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7530 }
7531 if total != entry.rows as u64 {
7532 return Err(invalid("numeric frequencies do not cover table rows"));
7533 }
7534 out.push(1);
7535 out.push(entries.len() as u8);
7536 for (value, count) in entries {
7537 match value {
7538 None => out.push(0),
7539 Some(value) => {
7540 out.push(1);
7541 out.extend_from_slice(&value.to_le_bytes());
7542 }
7543 }
7544 put_u64(&mut out, *count);
7545 }
7546 }
7547 _ => return Err(invalid("numeric frequency type or width differs")),
7548 }
7549 }
7550 }
7551 Ok(out)
7552}
7553
7554fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
7556 let bytes = text.as_bytes();
7557 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
7558 out.extend_from_slice(bytes);
7559 Ok(())
7560}
7561
7562fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
7565 let mut cur = Cursor::new(bytes);
7566 if cur.take(8)? != CATALOG {
7567 return Err(invalid("catalog magic differs"));
7568 }
7569 let count = cur.u32()? as usize;
7570 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
7571 for _ in 0..count {
7572 let name = cur.text()?;
7573 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
7574 let width = cur.u16()? as usize;
7575 let mut fields = Vec::with_capacity(width);
7576 for _ in 0..width {
7577 let name = cur.text()?;
7578 let ty = read_type(&mut cur)?;
7579 let not_null = match cur.u8()? {
7580 0 => false,
7581 1 => true,
7582 _ => return Err(invalid("nullability flag differs")),
7583 };
7584 fields.push(Field { name, ty, not_null });
7585 }
7586 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
7587 let end = directory
7588 .offset
7589 .checked_add(u64::from(directory.length))
7590 .ok_or_else(|| invalid("table directory offset overflow"))?;
7591 if directory.offset < HEADER
7592 || end > size
7593 || directory.length as usize > MAX_DIRECTORY
7594 || directory.length == 0
7595 {
7596 return Err(invalid("table directory range is outside the file"));
7597 }
7598 if entries.iter().any(|held| held.name == name) {
7599 return Err(invalid("two tables in the catalog have the same name"));
7600 }
7601 let nonzero = vec![None; fields.len()];
7602 let aggregates = vec![None; fields.len()];
7603 let distincts = vec![None; fields.len()];
7604 let extremes = vec![None; fields.len()];
7605 let frequencies = vec![None; fields.len()];
7606 entries.push(Entry {
7607 name,
7608 fields,
7609 rows,
7610 directory,
7611 nonzero,
7612 aggregates,
7613 distincts,
7614 extremes,
7615 frequencies,
7616 });
7617 }
7618 let count = if cur.done() { 0 } else { cur.u32()? as usize };
7623 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
7624 for _ in 0..count {
7625 let name = cur.text()?;
7626 let sql = cur.long_text()?;
7627 let statement = cur.long_text()?;
7628 let width = cur.u16()? as usize;
7629 let mut aliases = Vec::with_capacity(width);
7630 for _ in 0..width {
7631 aliases.push(cur.text()?);
7632 }
7633 let width = cur.u16()? as usize;
7634 let mut columns = Vec::with_capacity(width);
7635 for _ in 0..width {
7636 let name = cur.text()?;
7637 let ty = read_type(&mut cur)?;
7638 let not_null = match cur.u8()? {
7639 0 => false,
7640 1 => true,
7641 _ => return Err(invalid("nullability flag differs")),
7642 };
7643 columns.push(Field { name, ty, not_null });
7644 }
7645 if views.iter().any(|held| held.name == name) {
7649 return Err(invalid("two views in the catalog have the same name"));
7650 }
7651 if entries.iter().any(|held| held.name == name) {
7652 return Err(invalid("a table and a view in the catalog have the same name"));
7653 }
7654 views.push(ViewEntry { name, sql, statement, aliases, columns });
7655 }
7656 if !cur.done() {
7657 if cur.take(8)? != NONZERO_COUNTS {
7658 return Err(invalid("catalog extension magic differs"));
7659 }
7660 for entry in &mut entries {
7661 for (field, count) in entry.fields.iter().zip(&mut entry.nonzero) {
7662 *count = match cur.u8()? {
7663 0 => None,
7664 1 if matches!(
7665 field.ty,
7666 LogicalType::TinyInt
7667 | LogicalType::SmallInt
7668 | LogicalType::Integer
7669 | LogicalType::BigInt
7670 | LogicalType::UTinyInt
7671 | LogicalType::USmallInt
7672 | LogicalType::UInteger
7673 | LogicalType::UBigInt
7674 ) =>
7675 {
7676 let value = cur.u64()?;
7677 if value > entry.rows as u64 {
7678 return Err(invalid("nonzero count exceeds rows"));
7679 }
7680 Some(value)
7681 }
7682 _ => return Err(invalid("nonzero count tag or column type differs")),
7683 };
7684 }
7685 }
7686 }
7687 if !cur.done() {
7688 if cur.take(8)? != AGGREGATE_SUMS {
7689 return Err(invalid("aggregate catalog extension magic differs"));
7690 }
7691 for entry in &mut entries {
7692 for (field, summary) in entry.fields.iter().zip(&mut entry.aggregates) {
7693 *summary = match cur.u8()? {
7694 0 => None,
7695 1 if signed_integer(&field.ty) => {
7696 let sum = i128::from_le_bytes(
7697 cur.take(16)?
7698 .try_into()
7699 .map_err(|_| invalid("aggregate sum is truncated"))?,
7700 );
7701 let count = cur.u64()?;
7702 if count > entry.rows as u64 {
7703 return Err(invalid("aggregate count exceeds table rows"));
7704 }
7705 Some((sum, count))
7706 }
7707 _ => return Err(invalid("aggregate sum tag or column type differs")),
7708 };
7709 }
7710 }
7711 }
7712 if !cur.done() {
7713 if cur.take(8)? != DISTINCT_COUNTS {
7714 return Err(invalid("distinct catalog extension magic differs"));
7715 }
7716 for entry in &mut entries {
7717 for count in &mut entry.distincts {
7718 *count = match cur.u8()? {
7719 0 => None,
7720 1 => {
7721 let value = cur.u64()?;
7722 if value > entry.rows as u64 {
7723 return Err(invalid("distinct count exceeds table rows"));
7724 }
7725 Some(value)
7726 }
7727 _ => return Err(invalid("distinct count tag differs")),
7728 };
7729 }
7730 }
7731 }
7732 if !cur.done() {
7733 if cur.take(8)? != INTEGER_EXTREMES {
7734 return Err(invalid("integer extremes catalog extension magic differs"));
7735 }
7736 for entry in &mut entries {
7737 for (field, extremes) in entry.fields.iter().zip(&mut entry.extremes) {
7738 *extremes = match cur.u8()? {
7739 0 => None,
7740 1 if integer_or_date(&field.ty) => Some(None),
7741 2 if integer_or_date(&field.ty) => {
7742 let low = i128::from_le_bytes(
7743 cur.take(16)?
7744 .try_into()
7745 .map_err(|_| invalid("minimum is truncated"))?,
7746 );
7747 let high = i128::from_le_bytes(
7748 cur.take(16)?
7749 .try_into()
7750 .map_err(|_| invalid("maximum is truncated"))?,
7751 );
7752 if low > high {
7753 return Err(invalid("integer extremes are reversed"));
7754 }
7755 Some(Some((low, high)))
7756 }
7757 _ => return Err(invalid("integer extremes tag or type differs")),
7758 };
7759 }
7760 }
7761 }
7762 if !cur.done() {
7763 if cur.take(8)? != COMPLETE_FREQUENCIES {
7764 return Err(invalid("numeric frequency catalog extension magic differs"));
7765 }
7766 for entry in &mut entries {
7767 for (field, frequencies) in entry.fields.iter().zip(&mut entry.frequencies) {
7768 *frequencies = match cur.u8()? {
7769 0 => None,
7770 1 if integer_or_date(&field.ty) => {
7771 let len = cur.u8()? as usize;
7772 if len > MAX_CATALOG_FREQUENCIES {
7773 return Err(invalid("too many catalog numeric frequencies"));
7774 }
7775 let mut values = Vec::with_capacity(len);
7776 let mut total = 0_u64;
7777 for _ in 0..len {
7778 let value = match cur.u8()? {
7779 0 => None,
7780 1 => Some(i128::from_le_bytes(cur.take(16)?.try_into().map_err(
7781 |_| invalid("numeric frequency value is truncated"),
7782 )?)),
7783 _ => return Err(invalid("numeric frequency value tag differs")),
7784 };
7785 if values.iter().any(|(held, _)| *held == value) {
7786 return Err(invalid("numeric frequency value repeats"));
7787 }
7788 let count = cur.u64()?;
7789 total = total
7790 .checked_add(count)
7791 .ok_or_else(|| invalid("numeric frequency count overflows"))?;
7792 values.push((value, count));
7793 }
7794 if total != entry.rows as u64 {
7795 return Err(invalid("numeric frequencies do not cover table rows"));
7796 }
7797 Some(values)
7798 }
7799 _ => return Err(invalid("numeric frequency tag or type differs")),
7800 };
7801 }
7802 }
7803 }
7804 if !cur.done() {
7805 return Err(invalid("catalog has trailing bytes"));
7806 }
7807 Ok((entries, views))
7808}
7809
7810struct Cursor<'a> {
7818 bytes: &'a [u8],
7819 at: usize,
7820 window: Option<Window<'a>>,
7821}
7822
7823struct Window<'a> {
7825 file: &'a File,
7826 offset: u64,
7827 length: usize,
7828 start: usize,
7830 held: Vec<u8>,
7831 size: usize,
7833}
7834
7835const DIRECTORY_WINDOW: usize = 64 << 10;
7837
7838impl<'a> Cursor<'a> {
7839 fn new(bytes: &'a [u8]) -> Self {
7840 Self { bytes, at: 0, window: None }
7841 }
7842
7843 fn over(file: &'a File, offset: u64, length: usize) -> Self {
7845 let window =
7846 Window { file, offset, length, start: 0, held: Vec::new(), size: DIRECTORY_WINDOW };
7847 Self { bytes: &[], at: 0, window: Some(window) }
7848 }
7849
7850 fn len(&self) -> usize {
7852 self.window.as_ref().map_or(self.bytes.len(), |window| window.length)
7853 }
7854
7855 fn ensure(&mut self, len: usize) -> Result<()> {
7857 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7858 if end > self.len() {
7859 return Err(invalid("directory is truncated"));
7860 }
7861 let Some(window) = &mut self.window else { return Ok(()) };
7862 if self.at < window.start || end > window.start + window.held.len() {
7863 let want = len.max(window.size).min(window.length - self.at);
7864 window.start = self.at;
7865 window.held.resize(want, 0);
7866 read_at(window.file, window.offset + self.at as u64, &mut window.held)?;
7867 }
7868 Ok(())
7869 }
7870
7871 fn held(&self, at: usize, len: usize) -> &[u8] {
7873 match &self.window {
7874 Some(window) => &window.held[at - window.start..at - window.start + len],
7875 None => &self.bytes[at..at + len],
7876 }
7877 }
7878
7879 #[inline]
7881 fn peek(&mut self, len: usize) -> Result<&[u8]> {
7882 if self.window.is_none() {
7883 let bytes = self.bytes;
7884 return Ok(&bytes[self.at..self.end(len)?]);
7885 }
7886 self.ensure(len)?;
7887 Ok(self.held(self.at, len))
7888 }
7889
7890 #[inline]
7896 fn take(&mut self, len: usize) -> Result<&[u8]> {
7897 if self.window.is_none() {
7898 let bytes = self.bytes;
7899 let (at, end) = (self.at, self.end(len)?);
7900 self.at = end;
7901 return Ok(&bytes[at..end]);
7902 }
7903 self.take_windowed(len)
7904 }
7905
7906 fn skip(&mut self, len: usize) -> Result<()> {
7908 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7909 if end > self.len() {
7910 return Err(invalid("directory is truncated"));
7911 }
7912 self.at = end;
7913 Ok(())
7914 }
7915
7916 fn skip_bound(&mut self) -> Result<()> {
7917 match self.u8()? {
7918 0 => Ok(()),
7919 1 => self.skip(16),
7920 2 => self.skip(8),
7921 3 => {
7922 let length = self.u32()? as usize;
7923 self.skip(length)
7924 }
7925 4 => self.skip(17),
7926 _ => Err(invalid("a stored bound has an unknown tag")),
7927 }
7928 }
7929
7930 #[inline]
7932 fn end(&self, len: usize) -> Result<usize> {
7933 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
7934 if end > self.bytes.len() {
7935 return Err(invalid("directory is truncated"));
7936 }
7937 Ok(end)
7938 }
7939
7940 #[inline(never)]
7942 fn take_windowed(&mut self, len: usize) -> Result<&[u8]> {
7943 self.ensure(len)?;
7944 self.at += len;
7945 Ok(self.held(self.at - len, len))
7946 }
7947 #[inline]
7948 fn u8(&mut self) -> Result<u8> {
7949 Ok(self.take(1)?[0])
7950 }
7951 #[inline]
7952 fn u16(&mut self) -> Result<u16> {
7953 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
7954 }
7955 #[inline]
7956 fn u32(&mut self) -> Result<u32> {
7957 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
7958 }
7959 #[inline]
7960 fn u64(&mut self) -> Result<u64> {
7961 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
7962 }
7963 fn var_u64(&mut self) -> Result<u64> {
7964 let mut value = 0_u64;
7965 for shift in (0..=63).step_by(7) {
7966 let byte = self.u8()?;
7967 let part = u64::from(byte & 0x7f);
7968 if shift == 63 && part > 1 {
7969 return Err(invalid("frequency ordinal varint overflows"));
7970 }
7971 value |= part << shift;
7972 if byte & 0x80 == 0 {
7973 return Ok(value);
7974 }
7975 }
7976 Err(invalid("frequency ordinal varint is too long"))
7977 }
7978 fn bound(&mut self) -> Result<Option<Bound>> {
7987 let rest = self.len().saturating_sub(self.at);
7988 let mut want = 32;
7989 loop {
7990 let offered = self.peek(want.min(rest))?;
7991 let mut used = 0;
7992 match bounds::get(offered, &mut used) {
7993 Ok(bound) => {
7994 self.at += used;
7995 return Ok(bound);
7996 }
7997 Err(_) if want < rest => want *= 2,
7998 Err(error) => return Err(error),
7999 }
8000 }
8001 }
8002 fn text(&mut self) -> Result<String> {
8003 let len = self.u16()? as usize;
8004 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
8005 }
8006 fn done(&self) -> bool {
8009 self.at >= self.len()
8010 }
8011 fn long_text(&mut self) -> Result<String> {
8018 let len = self.u32()? as usize;
8019 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
8020 }
8021}
8022
8023fn decode_summary(
8025 cur: &mut Cursor<'_>,
8026 field: &Field,
8027 rows: usize,
8028 values: bool,
8029) -> Result<Option<FrequencySummary>> {
8030 Ok(match cur.u8()? {
8031 0 => None,
8032 1 => {
8033 let omitted_max = cur.u64()?;
8034 let count = cur.u32()? as usize;
8035 if count > FREQUENCY_ENTRIES {
8036 return Err(invalid("frequency entry count exceeds its bound"));
8037 }
8038 let mut entries = Vec::with_capacity(count);
8039 for _ in 0..count {
8041 let value = match cur.u8()? {
8042 0 => FrequencyValue::Null,
8043 1 => FrequencyValue::Integer(i128::from_le_bytes(
8044 cur.take(16)?.try_into().expect("sixteen bytes"),
8045 )),
8046 2 => FrequencyValue::Code(cur.u32()?),
8047 _ => return Err(invalid("frequency value tag differs")),
8048 };
8049 let valid = matches!(
8050 (&field.ty, value),
8051 (_, FrequencyValue::Null)
8052 | (LogicalType::Varchar, FrequencyValue::Code(_))
8053 | (
8054 LogicalType::TinyInt
8055 | LogicalType::SmallInt
8056 | LogicalType::Integer
8057 | LogicalType::BigInt
8058 | LogicalType::UTinyInt
8059 | LogicalType::USmallInt
8060 | LogicalType::UInteger
8061 | LogicalType::UBigInt
8062 | LogicalType::Date
8063 | LogicalType::Timestamp,
8064 FrequencyValue::Integer(_),
8065 )
8066 );
8067 if !valid {
8068 return Err(invalid("frequency value does not match its column"));
8069 }
8070 let count = cur.u64()?;
8071 if count == 0 || count > rows as u64 {
8072 return Err(invalid("frequency count is outside the table"));
8073 }
8074 entries.push(FrequencyEntry { value, count });
8075 }
8076 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8077 return Err(invalid("frequency entries are not descending"));
8078 }
8079 let ordinals = {
8080 let ordinal_count = cur.u32()? as usize;
8081 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
8082 return Err(invalid("frequency ordinal count exceeds its bound"));
8083 }
8084 let mut ordinals = Vec::with_capacity(ordinal_count);
8085 let mut previous = 0_u64;
8086 for at in 0..ordinal_count {
8087 let delta = cur.var_u64()?;
8088 if at != 0 && delta == 0 {
8089 return Err(invalid("frequency ordinals are not increasing"));
8090 }
8091 let ordinal = if at == 0 {
8092 delta
8093 } else {
8094 previous
8095 .checked_add(delta)
8096 .ok_or_else(|| invalid("frequency ordinal overflows"))?
8097 };
8098 if ordinal >= rows as u64 {
8099 return Err(invalid("frequency ordinal is outside the table"));
8100 }
8101 ordinals.push(ordinal);
8102 previous = ordinal;
8103 }
8104 ordinals
8105 };
8106 let ordinal_entries = if values {
8107 let mut ordinal_entries = Vec::with_capacity(ordinals.len());
8108 for _ in 0..ordinals.len() {
8109 let entry = cur.u16()?;
8110 if entry as usize >= entries.len() {
8111 return Err(invalid("frequency ordinal value is outside its entries"));
8112 }
8113 ordinal_entries.push(entry);
8114 }
8115 ordinal_entries
8116 } else {
8117 Vec::new()
8118 };
8119 Some(FrequencySummary { entries, omitted_max, ordinals, ordinal_entries })
8120 }
8121 _ => return Err(invalid("frequency summary tag differs")),
8122 })
8123}
8124
8125fn skip_summary(cur: &mut Cursor<'_>, values: bool, rows: usize) -> Result<()> {
8128 match cur.u8()? {
8129 0 => Ok(()),
8130 1 => {
8131 cur.skip(8)?;
8132 let entries = cur.u32()? as usize;
8133 if entries > FREQUENCY_ENTRIES {
8134 return Err(invalid("frequency entry count exceeds its bound"));
8135 }
8136 for _ in 0..entries {
8137 match cur.u8()? {
8138 0 => {}
8139 1 => cur.skip(16)?,
8140 2 => cur.skip(4)?,
8141 _ => return Err(invalid("frequency value tag differs")),
8142 }
8143 cur.skip(8)?;
8144 }
8145 let ordinals = cur.u32()? as usize;
8146 if ordinals > FREQUENCY_ORDINALS || ordinals > rows {
8147 return Err(invalid("frequency ordinal count exceeds its bound"));
8148 }
8149 for _ in 0..ordinals {
8150 cur.var_u64()?;
8151 }
8152 if values {
8153 cur.skip(ordinals * 2)?;
8154 }
8155 Ok(())
8156 }
8157 _ => Err(invalid("frequency summary tag differs")),
8158 }
8159}
8160
8161fn quick_nonzero(
8165 mut cur: Cursor<'_>,
8166 name: &str,
8167 fields: &[Field],
8168 rows: usize,
8169 wanted: usize,
8170) -> Result<Option<u64>> {
8171 if cur.take(8)? != DIRECTORY || cur.text()? != name {
8172 return Err(invalid("table directory differs from the catalog"));
8173 }
8174 let width = cur.u16()? as usize;
8175 if width != fields.len() {
8176 return Err(invalid("table directory width differs from the catalog"));
8177 }
8178 for field in fields {
8179 let stored =
8180 Field { name: cur.text()?, ty: read_type(&mut cur)?, not_null: cur.u8()? != 0 };
8181 if &stored != field {
8182 return Err(invalid("table directory schema differs from the catalog"));
8183 }
8184 }
8185 let mut dictionaries = Vec::with_capacity(width);
8186 for _ in 0..width {
8187 let held = match cur.u8()? {
8188 0 => false,
8189 1 => {
8190 cur.skip(20)?;
8191 true
8192 }
8193 _ => return Err(invalid("dictionary page tag differs")),
8194 };
8195 dictionaries.push(held);
8196 }
8197 for _ in 0..width {
8198 match cur.u8()? {
8199 0 => {}
8200 1 => cur.skip(8)?,
8201 _ => return Err(invalid("distinct count tag differs")),
8202 }
8203 }
8204 if cur.u64()? != rows as u64 {
8205 return Err(invalid("table row count differs from the catalog"));
8206 }
8207 let stripes = cur.u32()? as usize;
8208 let mut total = 0_usize;
8209 let mut nulls = 0_u64;
8210 for _ in 0..stripes {
8211 let parts = cur.u32()? as usize;
8212 if parts == 0 || parts > STRIPE_PARTS {
8213 return Err(invalid("stripe part count is outside its bound"));
8214 }
8215 let mut stripe_rows = 0_usize;
8216 for _ in 0..parts {
8217 stripe_rows = stripe_rows
8218 .checked_add(cur.u32()? as usize)
8219 .ok_or_else(|| invalid("stripe row count overflow"))?;
8220 }
8221 total =
8222 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8223 cur.skip(12 + width * 12)?;
8224 for (field, held) in fields.iter().zip(&dictionaries) {
8225 if field.ty == LogicalType::Varchar && *held {
8226 cur.skip(20)?;
8227 }
8228 }
8229 for _ in 0..width * 2 {
8230 match cur.u8()? {
8231 0 => {}
8232 1 => cur.skip(20)?,
8233 _ => return Err(invalid("stripe page tag differs")),
8234 }
8235 }
8236 for column in 0..width {
8237 cur.skip_bound()?;
8238 cur.skip_bound()?;
8239 let count = cur.u32()? as u64;
8240 if count > stripe_rows as u64 {
8241 return Err(invalid("null count exceeds stripe rows"));
8242 }
8243 if column == wanted {
8244 nulls = nulls.checked_add(count).ok_or_else(|| invalid("null count overflow"))?;
8245 }
8246 cur.skip(1)?;
8247 match cur.u8()? {
8248 0 => {}
8249 1 => cur.skip(16)?,
8250 _ => return Err(invalid("a stripe sum has an unknown tag")),
8251 }
8252 }
8253 }
8254 if total != rows {
8255 return Err(invalid("table row count differs from stripes"));
8256 }
8257 if cur.done() {
8258 return Ok(None);
8259 }
8260 let magic = cur.take(8)?;
8261 let values = magic == FREQUENCIES;
8262 if !values && magic != FREQUENCIES_V2 {
8263 return Err(invalid("directory extension magic differs"));
8264 }
8265 if cur.u16()? as usize != width {
8266 return Err(invalid("frequency column count differs"));
8267 }
8268 for _ in 0..wanted {
8269 skip_summary(&mut cur, values, rows)?;
8270 }
8271 let Some(summary) = decode_summary(&mut cur, &fields[wanted], rows, values)? else {
8272 return Ok(None);
8273 };
8274 let zero = summary
8275 .entries
8276 .iter()
8277 .find(|entry| entry.value == FrequencyValue::Integer(0))
8278 .map(|entry| entry.count)
8279 .or_else(|| (summary.omitted_max == 0).then_some(0));
8280 Ok(zero.and_then(|zero| (rows as u64).checked_sub(nulls)?.checked_sub(zero)))
8281}
8282
8283fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
8284 read_directory(Cursor::new(bytes), size, None)
8285}
8286
8287fn read_directory(mut cur: Cursor<'_>, size: u64, stored_at: Option<u64>) -> Result<Table> {
8292 if cur.take(8)? != DIRECTORY {
8293 return Err(invalid("directory magic differs"));
8294 }
8295 let name = cur.text()?;
8296 let width = cur.u16()? as usize;
8297 let mut fields = Vec::with_capacity(width);
8298 for _ in 0..width {
8299 let name = cur.text()?;
8300 let ty = read_type(&mut cur)?;
8301 let not_null = match cur.u8()? {
8302 0 => false,
8303 1 => true,
8304 _ => return Err(invalid("nullability flag differs")),
8305 };
8306 fields.push(Field { name, ty, not_null });
8307 }
8308 let mut dictionaries = Vec::with_capacity(width);
8309 for _ in 0..width {
8310 dictionaries.push(match cur.u8()? {
8311 0 => None,
8312 1 => {
8313 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8314 let end = page
8315 .offset
8316 .checked_add(u64::from(page.length))
8317 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
8318 if page.offset < HEADER || end > size {
8323 return Err(invalid("dictionary page range is outside the file"));
8324 }
8325 Some(page)
8326 }
8327 _ => return Err(invalid("dictionary page tag differs")),
8328 });
8329 }
8330 let mut distincts = Vec::with_capacity(width);
8331 for _ in 0..width {
8332 distincts.push(match cur.u8()? {
8333 0 => None,
8334 1 => Some(cur.u64()?),
8335 _ => return Err(invalid("distinct count tag differs")),
8336 });
8337 }
8338 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
8339 let count = cur.u32()? as usize;
8340 let mut stripes = Vec::with_capacity(count);
8341 let mut total = 0_usize;
8342 for _ in 0..count {
8343 let count = cur.u32()? as usize;
8344 if count == 0 || count > STRIPE_PARTS {
8345 return Err(invalid("stripe part count is outside its bound"));
8346 }
8347 let mut parts = Vec::with_capacity(count);
8348 let mut stripe_rows = 0_usize;
8349 for _ in 0..count {
8350 let rows = cur.u32()?;
8351 if rows == 0 {
8352 return Err(invalid("empty part"));
8353 }
8354 parts.push(rows);
8355 stripe_rows = stripe_rows
8356 .checked_add(rows as usize)
8357 .ok_or_else(|| invalid("stripe row count overflow"))?;
8358 }
8359 total =
8360 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
8361 let index = Span { offset: cur.u64()?, length: cur.u32()? };
8362 let section = index_section(count)?;
8363 let wanted = section
8364 .checked_mul(width)
8365 .and_then(|bytes| u32::try_from(bytes).ok())
8366 .ok_or_else(|| invalid("index page length overflow"))?;
8367 let end = index
8368 .offset
8369 .checked_add(u64::from(index.length))
8370 .ok_or_else(|| invalid("index page offset overflow"))?;
8371 if index.offset < HEADER || end > size || index.length != wanted {
8372 return Err(invalid("index page range is outside the file"));
8373 }
8374 let mut pages = Vec::with_capacity(width);
8375 for _ in 0..width {
8376 let offset = cur.u64()?;
8377 let length = cur.u32()?;
8378 let end = offset
8379 .checked_add(u64::from(length))
8380 .ok_or_else(|| invalid("page offset overflow"))?;
8381 if offset < HEADER || end > size || length as usize > MAX_PAGE {
8382 return Err(invalid("page range is outside the file"));
8383 }
8384 pages.push(Span { offset, length });
8385 }
8386 let mut memberships = vec![None; width];
8387 for (column, field) in fields.iter().enumerate() {
8388 if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
8389 continue;
8390 }
8391 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8392 let end = page
8393 .offset
8394 .checked_add(u64::from(page.length))
8395 .ok_or_else(|| invalid("membership page offset overflow"))?;
8396 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8397 return Err(invalid("membership page range is outside the file"));
8398 }
8399 memberships[column] = Some(page);
8400 }
8401 let mut sieves = vec![None; width];
8402 for sieve in sieves.iter_mut().take(width) {
8403 match cur.u8()? {
8404 0 => continue,
8405 1 => {}
8406 _ => return Err(invalid("a sieve page has an unknown tag")),
8407 }
8408 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8409 let end = page
8410 .offset
8411 .checked_add(u64::from(page.length))
8412 .ok_or_else(|| invalid("sieve page offset overflow"))?;
8413 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8414 return Err(invalid("sieve page range is outside the file"));
8415 }
8416 *sieve = Some(page);
8417 }
8418 let mut part_ranges = vec![None; width];
8419 for held in part_ranges.iter_mut().take(width) {
8420 match cur.u8()? {
8421 0 => continue,
8422 1 => {}
8423 _ => return Err(invalid("a part range page has an unknown tag")),
8424 }
8425 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
8426 let end = page
8427 .offset
8428 .checked_add(u64::from(page.length))
8429 .ok_or_else(|| invalid("part range page offset overflow"))?;
8430 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
8431 return Err(invalid("part range page range is outside the file"));
8432 }
8433 *held = Some(page);
8434 }
8435 let mut ranges = Vec::with_capacity(width);
8436 for column in 0..width {
8437 let low = cur.bound()?;
8438 let high = cur.bound()?;
8439 let nulls = cur.u32()? as usize;
8440 if nulls > stripe_rows {
8441 return Err(invalid("null count exceeds stripe rows"));
8442 }
8443 let exact = cur.u8()? != 0;
8444 let sum = match cur.u8()? {
8445 0 => None,
8446 1 => Some(i128::from_le_bytes(
8447 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
8448 )),
8449 _ => return Err(invalid("a stripe sum has an unknown tag")),
8450 };
8451 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
8457 let low = low.map(|bound| scaled_as(bound, ty));
8458 let high = high.map(|bound| scaled_as(bound, ty));
8459 ranges.push(Range { low, high, nulls, exact, sum });
8460 }
8461 stripes.push(Stripe {
8462 rows: stripe_rows,
8463 parts,
8464 index,
8465 pages,
8466 memberships: Pages::from_slots(memberships)?,
8467 sieves: Pages::from_slots(sieves)?,
8468 part_ranges: Pages::from_slots(part_ranges)?,
8469 zone: Zone::from_ranges(ranges),
8470 });
8471 }
8472 if total != rows {
8473 return Err(invalid("table row count differs from stripes"));
8474 }
8475 let mut entry_counts = vec![0; width];
8478 let frequencies = if cur.done() {
8479 vec![None; width]
8480 } else {
8481 let frequency_magic = cur.take(8)?;
8482 let frequency_values = frequency_magic == FREQUENCIES;
8483 if !frequency_values && frequency_magic != FREQUENCIES_V2 {
8484 return Err(invalid("directory extension magic differs"));
8485 }
8486 if cur.u16()? as usize != width {
8487 return Err(invalid("frequency column count differs"));
8488 }
8489 let mut frequencies = Vec::with_capacity(width);
8490 for (field, entry_count) in fields.iter().zip(&mut entry_counts) {
8491 let start = cur.at;
8492 let summary = decode_summary(&mut cur, field, rows, frequency_values)?;
8493 *entry_count = summary.as_ref().map_or(0, |summary| summary.entries.len());
8494 frequencies.push(match (summary, stored_at) {
8495 (None, _) => None,
8496 (Some(summary), None) => Some(Frequencies::Held(summary)),
8497 (Some(_), Some(offset)) => Some(Frequencies::Stored {
8498 span: Span {
8499 offset: offset + start as u64,
8500 length: u32::try_from(cur.at - start)
8501 .map_err(|_| invalid("a frequency synopsis is too long"))?,
8502 },
8503 values: frequency_values,
8504 }),
8505 });
8506 }
8507 frequencies
8508 };
8509 let mut clustering = None;
8519 let mut sections = Vec::new();
8520 let mut pair_frequencies = Vec::new();
8521 let mut seen_pair_frequencies = false;
8522 let mut frequency_texts = vec![Vec::new(); width];
8523 let mut seen_frequency_texts = false;
8524 let mut host_groups = None;
8525 let mut seen_sections = false;
8526 let mut dictionary_payloads = Vec::new();
8527 let mut seen_payloads = false;
8528 let mut generation = 0;
8531 while !cur.done() {
8532 let mut tag = [0u8; 8];
8533 tag.copy_from_slice(cur.take(8)?);
8534 if &tag == PAIR_FREQUENCIES {
8535 if seen_pair_frequencies {
8536 return Err(invalid("directory names two pair frequency blocks"));
8537 }
8538 seen_pair_frequencies = true;
8539 let count = cur.u16()? as usize;
8540 if count > MAX_PAIR_FREQUENCIES {
8541 return Err(invalid("pair frequency count exceeds its bound"));
8542 }
8543 pair_frequencies = Vec::with_capacity(count);
8544 for _ in 0..count {
8545 let first = cur.u16()?;
8546 let second = cur.u16()?;
8547 let first_at = first as usize;
8548 let second_at = second as usize;
8549 if frequencies.get(first_at).and_then(Option::as_ref).is_none() {
8550 return Err(invalid("pair frequency first column has no synopsis"));
8551 }
8552 let first_entries = entry_counts[first_at];
8553 if !matches!(fields.get(second_at), Some(field) if field.ty == LogicalType::Varchar)
8554 || dictionaries.get(second_at).copied().flatten().is_none()
8555 {
8556 return Err(invalid("pair frequency second column has no stable dictionary"));
8557 }
8558 if pair_frequencies
8559 .iter()
8560 .any(|held: &PairFrequencySummary| held.first == first && held.second == second)
8561 {
8562 return Err(invalid("directory repeats a pair frequency summary"));
8563 }
8564 let omitted_max = cur.u64()?;
8565 if omitted_max > rows as u64 {
8566 return Err(invalid("pair frequency omitted count exceeds the table"));
8567 }
8568 let entries_count = cur.u16()? as usize;
8569 if entries_count > FREQUENCY_ENTRIES {
8570 return Err(invalid("pair frequency entry count exceeds its bound"));
8571 }
8572 let mut entries = Vec::with_capacity(entries_count);
8573 for _ in 0..entries_count {
8574 let first_entry = cur.u16()?;
8575 if first_entry as usize >= first_entries {
8576 return Err(invalid("pair frequency anchor is outside its synopsis"));
8577 }
8578 let second = match cur.u8()? {
8579 0 => None,
8580 1 => Some(cur.u32()?),
8581 _ => return Err(invalid("pair frequency string tag differs")),
8582 };
8583 let count = cur.u64()?;
8584 if count == 0 || count > rows as u64 {
8585 return Err(invalid("pair frequency count is outside the table"));
8586 }
8587 entries.push(PairFrequencyEntry { first_entry, second, count });
8588 }
8589 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
8590 return Err(invalid("pair frequency entries are not descending"));
8591 }
8592 pair_frequencies.push(PairFrequencySummary { first, second, entries, omitted_max });
8593 }
8594 } else if &tag == FREQUENCY_TEXTS {
8595 if seen_frequency_texts {
8596 return Err(invalid("directory names two frequency text blocks"));
8597 }
8598 seen_frequency_texts = true;
8599 let columns = cur.u16()? as usize;
8600 if columns > width {
8601 return Err(invalid("frequency text column count exceeds the schema"));
8602 }
8603 for _ in 0..columns {
8604 let column = cur.u16()? as usize;
8605 if !frequency_texts.get(column).is_some_and(Vec::is_empty) {
8606 return Err(invalid("frequency text column is repeated or out of range"));
8607 }
8608 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8609 || dictionaries.get(column).copied().flatten().is_none()
8610 || frequencies.get(column).and_then(Option::as_ref).is_none()
8611 {
8612 return Err(invalid("frequency texts belong to a non-string synopsis"));
8613 }
8614 let count = cur.u16()? as usize;
8615 if count == 0 || count != entry_counts[column] {
8616 return Err(invalid("frequency text count differs from its synopsis"));
8617 }
8618 let mut texts = Vec::with_capacity(count);
8619 for _ in 0..count {
8620 texts.push(match cur.u8()? {
8621 0 => None,
8622 1 => {
8623 let length = cur.u32()? as usize;
8624 let bytes = cur.take(length)?.to_vec();
8625 std::str::from_utf8(&bytes)
8626 .map_err(|_| invalid("frequency text is not UTF-8"))?;
8627 Some(bytes)
8628 }
8629 _ => return Err(invalid("frequency text tag differs")),
8630 });
8631 }
8632 frequency_texts[column] = texts;
8633 }
8634 } else if &tag == HOST_GROUPS {
8635 if host_groups.is_some() {
8636 return Err(invalid("directory names two host group blocks"));
8637 }
8638 let column = cur.u16()? as usize;
8639 if !matches!(fields.get(column), Some(field) if field.ty == LogicalType::Varchar)
8640 || dictionaries.get(column).copied().flatten().is_none()
8641 {
8642 return Err(invalid("host groups belong to a non-string dictionary"));
8643 }
8644 let omitted_max = cur.u64()?;
8645 if omitted_max > rows as u64 {
8646 return Err(invalid("host group bound exceeds the table"));
8647 }
8648 let count = cur.u16()? as usize;
8649 if count > host::CAPACITY {
8650 return Err(invalid("host group count exceeds its bound"));
8651 }
8652 let mut entries = Vec::with_capacity(count);
8653 let mut bytes = 0_usize;
8654 for _ in 0..count {
8655 let host_len = cur.u32()? as usize;
8656 bytes =
8657 bytes.checked_add(host_len).ok_or_else(|| invalid("host bytes overflow"))?;
8658 if bytes > host::BYTE_BUDGET {
8659 return Err(invalid("host groups exceed their byte budget"));
8660 }
8661 let host = std::str::from_utf8(cur.take(host_len)?)
8662 .map_err(|_| invalid("host is not UTF-8"))?
8663 .to_owned();
8664 let count = cur.u64()?;
8665 if count == 0 || count > rows as u64 {
8666 return Err(invalid("host group count exceeds the table"));
8667 }
8668 let bytes_sum = i128::from_le_bytes(
8669 cur.take(16)?
8670 .try_into()
8671 .map_err(|_| invalid("host length sum is truncated"))?,
8672 );
8673 if bytes_sum < 0 {
8674 return Err(invalid("host length sum is negative"));
8675 }
8676 let minimum_len = cur.u32()? as usize;
8677 bytes =
8678 bytes.checked_add(minimum_len).ok_or_else(|| invalid("host bytes overflow"))?;
8679 if bytes > host::BYTE_BUDGET {
8680 return Err(invalid("host groups exceed their byte budget"));
8681 }
8682 let minimum = std::str::from_utf8(cur.take(minimum_len)?)
8683 .map_err(|_| invalid("host minimum is not UTF-8"))?
8684 .to_owned();
8685 entries.push(host::HostEntry { host, count, bytes_sum, minimum });
8686 }
8687 if entries.windows(2).any(|pair| pair[0].count < pair[1].count)
8688 || entries.iter().any(|entry| entry.host.is_empty() || entry.minimum.is_empty())
8689 {
8690 return Err(invalid("host groups are not in certified order"));
8691 }
8692 host_groups = Some(host::HostSummary { column, omitted_max, entries });
8693 } else if &tag == CLUSTERING {
8694 if clustering.is_some() {
8695 return Err(invalid("directory names two clustering declarations"));
8696 }
8697 let bucket = Width::from_tag(cur.u8()?)
8698 .ok_or_else(|| invalid("clustering width tag differs"))?;
8699 let count = cur.u16()? as usize;
8700 let mut columns = Vec::with_capacity(count.min(fields.len()));
8701 for _ in 0..count {
8702 columns.push(u32::from(cur.u16()?));
8703 }
8704 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
8707 invalid("stored clustering declaration does not match the table it is on")
8708 })?);
8709 } else if &tag == SECTIONS {
8710 if seen_sections {
8711 return Err(invalid("directory names two section tables"));
8712 }
8713 seen_sections = true;
8714 generation = cur.u64()?;
8715 let count = cur.u16()? as usize;
8716 if count > MAX_SECTIONS {
8717 return Err(invalid("section count exceeds its bound"));
8718 }
8719 sections = Vec::with_capacity(count);
8720 for _ in 0..count {
8723 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
8724 }
8725 for held in §ions {
8726 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
8727 return Err(invalid("a section's extent table overflows the file"));
8728 };
8729 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
8733 return Err(invalid("a section's extent table is outside the file"));
8734 }
8735 if held.extents == 0 && held.extent_bytes != 0 {
8736 return Err(invalid("a section with no extents names an extent table"));
8737 }
8738 }
8739 } else if &tag == DICTIONARY_PAYLOADS {
8740 if seen_payloads {
8741 return Err(invalid("directory names two dictionary payload blocks"));
8742 }
8743 seen_payloads = true;
8744 let count = cur.u16()? as usize;
8745 if count != fields.len() {
8746 return Err(invalid("dictionary payload block does not match the table's columns"));
8747 }
8748 dictionary_payloads = Vec::with_capacity(count);
8749 for _ in 0..count {
8750 let bytes = cur.u64()?;
8751 if bytes > size {
8752 return Err(invalid("a dictionary payload is larger than the file"));
8753 }
8754 dictionary_payloads.push(bytes);
8755 }
8756 } else {
8757 return Err(invalid("directory extension magic differs"));
8758 }
8759 }
8760 if !cur.done() {
8761 return Err(invalid("directory has trailing bytes"));
8762 }
8763 Ok(Table {
8764 name,
8765 fields,
8766 stripes,
8767 rows,
8768 dictionaries,
8769 dictionary_payloads,
8770 distincts,
8771 frequencies,
8772 pair_frequencies,
8773 frequency_texts,
8774 host_groups,
8775 clustering,
8776 generation,
8777 sections,
8778 })
8779}
8780
8781fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
8783 bounds::put(out, bound)
8784}
8785
8786#[derive(Debug)]
8803struct Codes;
8804
8805impl chooser::Chooser for Codes {
8806 fn name(&self) -> &'static str {
8807 "codes"
8808 }
8809
8810 fn narrow_strings(
8811 &self,
8812 _values: &[&[u8]],
8813 offered: &[string::Kind],
8814 _depth: u8,
8815 ) -> Vec<string::Kind> {
8816 offered.to_vec()
8819 }
8820
8821 fn narrow_integers(
8822 &self,
8823 _values: &[i64],
8824 offered: &[integer::Kind],
8825 depth: u8,
8826 ) -> Vec<integer::Kind> {
8827 narrowed_to(Codes::keep(depth), offered)
8830 }
8831
8832 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8833 Codes::keep(depth).contains(&kind)
8834 }
8835}
8836
8837impl Codes {
8838 fn keep(depth: u8) -> &'static [integer::Kind] {
8839 if depth == 0 {
8840 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
8841 } else {
8842 &[integer::Kind::Constant, integer::Kind::Packed]
8843 }
8844 }
8845}
8846
8847fn narrowed_to(keep: &[integer::Kind], offered: &[integer::Kind]) -> Vec<integer::Kind> {
8855 let narrowed: Vec<integer::Kind> =
8856 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
8857 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
8858}
8859
8860#[derive(Debug)]
8872struct Fixed;
8873
8874impl chooser::Chooser for Fixed {
8875 fn name(&self) -> &'static str {
8876 "fixed"
8877 }
8878
8879 fn narrow_strings(
8880 &self,
8881 _values: &[&[u8]],
8882 offered: &[string::Kind],
8883 _depth: u8,
8884 ) -> Vec<string::Kind> {
8885 offered.to_vec()
8886 }
8887
8888 fn narrow_integers(
8889 &self,
8890 _values: &[i64],
8891 offered: &[integer::Kind],
8892 depth: u8,
8893 ) -> Vec<integer::Kind> {
8894 narrowed_to(Fixed::keep(depth), offered)
8895 }
8896
8897 fn considers_integer(&self, kind: integer::Kind, depth: u8) -> bool {
8898 Fixed::keep(depth).contains(&kind)
8899 }
8900}
8901
8902impl Fixed {
8903 fn keep(depth: u8) -> &'static [integer::Kind] {
8904 if depth == 0 {
8905 &[
8906 integer::Kind::Constant,
8907 integer::Kind::Packed,
8908 integer::Kind::Delta,
8909 integer::Kind::Rle,
8910 integer::Kind::Sparse,
8911 integer::Kind::Strided,
8912 ]
8913 } else {
8914 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
8915 }
8916 }
8917}
8918
8919fn widened(data: &Data) -> Option<Vec<i64>> {
8926 match data {
8927 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8928 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8929 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8930 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8931 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8932 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
8933 Data::Int64(values) => Some(values.to_vec()),
8934 _ => None,
8935 }
8936}
8937
8938trait Narrow: Copy {
8945 const BIASED: (u32, u64);
8950
8951 fn narrow(value: i64) -> Self;
8953}
8954
8955#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
8972fn residue<T: Narrow>(value: i64) -> u64 {
8973 let (bits, bias) = T::BIASED;
8974 (value as u64).wrapping_add(bias) >> bits
8975}
8976
8977macro_rules! narrows {
8982 ($($ty:ty => $bias:expr),* $(,)?) => {$(
8983 impl Narrow for $ty {
8984 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
8985
8986 #[allow(
8987 clippy::cast_possible_truncation,
8988 clippy::cast_sign_loss,
8989 reason = "the caller has checked the bits this truncates away"
8990 )]
8991 fn narrow(value: i64) -> Self {
8992 value as Self
8993 }
8994 }
8995 )*};
8996}
8997
8998narrows! {
8999 i8 => 1 << 7,
9000 u8 => 0,
9001 i16 => 1 << 15,
9002 u16 => 0,
9003 i32 => 1 << 31,
9004 u32 => 0,
9005}
9006
9007fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
9020 let mut spilled = 0u64;
9021 for value in values {
9022 spilled |= residue::<T>(*value);
9023 }
9024 if spilled != 0 {
9025 return Err(invalid("page value is not of its type"));
9026 }
9027 Ok(values.iter().map(|value| T::narrow(*value)).collect())
9028}
9029
9030fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
9035 Ok(match ty {
9036 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
9037 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
9038 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
9039 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
9040 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
9041 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
9042 LogicalType::BigInt
9043 | LogicalType::Timestamp
9044 | LogicalType::Time
9045 | LogicalType::TimeTz
9046 | LogicalType::TimestampTz
9047 | LogicalType::TimestampS
9048 | LogicalType::TimestampMs
9049 | LogicalType::TimestampNs => Data::Int64(values.into()),
9050 LogicalType::Decimal { .. } => match ty.physical() {
9053 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
9054 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
9055 PhysicalType::Int64 => Data::Int64(values.into()),
9056 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
9057 },
9058 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
9059 })
9060}
9061
9062fn plain_width(ty: &LogicalType) -> Option<usize> {
9065 Some(match ty {
9066 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
9067 LogicalType::SmallInt | LogicalType::USmallInt => 2,
9068 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
9069 LogicalType::BigInt
9070 | LogicalType::Timestamp
9071 | LogicalType::Time
9072 | LogicalType::TimeTz
9073 | LogicalType::TimestampTz
9074 | LogicalType::TimestampS
9075 | LogicalType::TimestampMs
9076 | LogicalType::TimestampNs => 8,
9077 LogicalType::Decimal { .. } => match ty.physical() {
9078 PhysicalType::Int16 => 2,
9079 PhysicalType::Int32 => 4,
9080 PhysicalType::Int64 => 8,
9081 _ => return None,
9084 },
9085 _ => return None,
9086 })
9087}
9088
9089fn cascaded(
9095 flat: &Vector,
9096 ty: &LogicalType,
9097 packed: Option<&Packed<'_>>,
9098 settling: &mut Settling,
9099) -> Result<Option<Vec<u8>>> {
9100 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
9101 let Some(values) = widened(data) else { return Ok(None) };
9102 let plain = values.len().saturating_mul(width);
9103 let best = match packed {
9104 Some(packed) => plain.min(21 + size_of_val(packed.words())),
9106 None => plain,
9107 };
9108 let out = settling.encode(&values)?;
9109 Ok((out.len() < best).then_some(out))
9110}
9111
9112const SEARCH_EVERY: usize = 16;
9119
9120#[derive(Debug, Default)]
9125struct Settling {
9126 shape: Option<Shape>,
9129 since: usize,
9131}
9132
9133impl Settling {
9134 fn encode(&mut self, values: &[i64]) -> Result<Vec<u8>> {
9141 if let Some(shape) = self.shape.as_ref().filter(|_| self.since < SEARCH_EVERY) {
9142 let replay = chooser::Replay::new(&shape.kinds, &Fixed).expecting(&shape.offered);
9143 let out = integer::encode_with(values, &replay)?;
9144 if !replay.held() {
9145 self.settle(&out, values.len(), replay.first_offered())?;
9146 return Ok(out);
9147 }
9148 let grown = (out.len() as u128) * (shape.rows as u128) * 4;
9149 if grown <= (shape.len as u128) * (values.len() as u128) * 5 {
9150 self.since += 1;
9151 return Ok(out);
9152 }
9153 }
9154 let search = chooser::Replay::new(&[], &Fixed);
9156 let out = integer::encode_with(values, &search)?;
9157 self.settle(&out, values.len(), search.first_offered())?;
9158 Ok(out)
9159 }
9160
9161 fn settle(&mut self, out: &[u8], rows: usize, offered: Vec<integer::Kind>) -> Result<()> {
9162 let kinds = integer::shape(out)?;
9163 self.shape = Some(Shape { kinds, offered, len: out.len().max(1), rows: rows.max(1) });
9164 self.since = 0;
9165 Ok(())
9166 }
9167}
9168
9169#[derive(Debug)]
9171struct Shape {
9172 kinds: Vec<integer::Kind>,
9173 offered: Vec<integer::Kind>,
9174 len: usize,
9175 rows: usize,
9176}
9177
9178fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
9216 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
9217 let mut payload = 0_usize;
9218 for row in 0..flat.len() {
9219 let text = flat.bytes_at(row).unwrap_or(b"");
9222 payload = payload.saturating_add(text.len());
9223 values.push(text);
9224 }
9225 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
9227 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
9228 return Ok(None);
9229 };
9230 Ok((out.len() < plain).then_some(out))
9231}
9232
9233fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
9234 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
9235 let coded = integer::encode_with(&wide, &Codes)?;
9236 let plain = codes.len().saturating_mul(size_of::<u32>());
9237 Ok((coded.len() < plain).then_some(coded))
9238}
9239
9240fn push_validity(out: &mut Vec<u8>, flat: &Vector) {
9243 let flag = match flat.validity() {
9244 Validity::AllValid => 0,
9245 Validity::AllInvalid => 1,
9246 Validity::Mask(_) => 2,
9247 };
9248 out.push(flag);
9249 if flag == 2 {
9250 for group in (0..flat.len()).step_by(8) {
9251 let mut bits = 0_u8;
9252 for bit in 0..8 {
9253 if group + bit < flat.len() && !flat.is_null_at(group + bit) {
9254 bits |= 1 << bit;
9255 }
9256 }
9257 out.push(bits);
9258 }
9259 }
9260}
9261
9262fn coded_page(codes: &[u32], validity: &[u8]) -> Result<Vec<u8>> {
9269 let coded = encoded_codes(codes)?;
9270 let mut out = Vec::with_capacity(
9271 1 + validity.len() + coded.as_ref().map_or(size_of_val(codes), Vec::len),
9272 );
9273 out.push(if coded.is_some() { 4 } else { 3 });
9274 out.extend_from_slice(validity);
9275 match coded {
9276 Some(coded) => out.extend_from_slice(&coded),
9277 None => {
9278 for &code in codes {
9279 put_u32(&mut out, code);
9280 }
9281 }
9282 }
9283 Ok(out)
9284}
9285
9286fn encode(vector: &Vector, settling: &mut Settling) -> Result<Vec<u8>> {
9289 let ty = vector.logical_type();
9290 let flat = vector.flatten()?;
9292 let mut out = Vec::new();
9293 let dictionary = if ty == &LogicalType::Varchar { string_dictionary(&flat)? } else { None };
9294 let compressed_text = if dictionary.is_none() && ty == &LogicalType::Varchar {
9295 text_compressed(&flat)?
9296 } else {
9297 None
9298 };
9299 let packed_vector = if dictionary.is_none() { Some(flat.bit_packed()?) } else { None };
9300 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
9301 let cascade =
9305 if dictionary.is_none() { cascaded(&flat, ty, packed.as_ref(), settling)? } else { None };
9306 out.push(if cascade.is_some() {
9307 5
9308 } else if dictionary.is_some() {
9309 1
9310 } else if compressed_text.is_some() {
9311 6
9312 } else if packed.is_some() {
9313 2
9314 } else {
9315 0
9316 });
9317 push_validity(&mut out, &flat);
9318 if let Some(cascade) = cascade {
9319 out.extend_from_slice(&cascade);
9320 return Ok(out);
9321 }
9322 if let Some(dictionary) = dictionary {
9323 out.extend_from_slice(&dictionary);
9324 return Ok(out);
9325 }
9326 if let Some(compressed_text) = compressed_text {
9327 out.extend_from_slice(&compressed_text);
9328 return Ok(out);
9329 }
9330 if let Some(packed) = packed {
9331 if packed.offset() != 0 {
9332 return Err(invalid("writer received a sliced packed vector"));
9333 }
9334 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
9335 out.extend_from_slice(&packed.base().to_le_bytes());
9336 put_u32(
9337 &mut out,
9338 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
9339 );
9340 for word in packed.words() {
9341 put_u64(&mut out, *word);
9342 }
9343 return Ok(out);
9344 }
9345 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
9346 match (ty, data) {
9347 (LogicalType::TinyInt, Data::Int8(values)) => {
9348 for value in &**values {
9349 out.extend_from_slice(&value.to_le_bytes());
9350 }
9351 }
9352 (LogicalType::UTinyInt, Data::UInt8(values)) => {
9353 for value in &**values {
9354 out.extend_from_slice(&value.to_le_bytes());
9355 }
9356 }
9357 (LogicalType::SmallInt, Data::Int16(values)) => {
9358 for value in &**values {
9359 out.extend_from_slice(&value.to_le_bytes());
9360 }
9361 }
9362 (LogicalType::USmallInt, Data::UInt16(values)) => {
9363 for value in &**values {
9364 out.extend_from_slice(&value.to_le_bytes());
9365 }
9366 }
9367 (LogicalType::UInteger, Data::UInt32(values)) => {
9368 for value in &**values {
9369 out.extend_from_slice(&value.to_le_bytes());
9370 }
9371 }
9372 (LogicalType::UBigInt, Data::UInt64(values)) => {
9373 for value in &**values {
9374 out.extend_from_slice(&value.to_le_bytes());
9375 }
9376 }
9377 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
9378 for value in &**values {
9379 out.extend_from_slice(&value.to_le_bytes());
9380 }
9381 }
9382 (
9383 LogicalType::BigInt
9384 | LogicalType::Timestamp
9385 | LogicalType::Time
9386 | LogicalType::TimeTz
9387 | LogicalType::TimestampTz
9388 | LogicalType::TimestampS
9389 | LogicalType::TimestampMs
9390 | LogicalType::TimestampNs,
9391 Data::Int64(values),
9392 ) => {
9393 for value in &**values {
9394 out.extend_from_slice(&value.to_le_bytes());
9395 }
9396 }
9397 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
9400 for value in &**values {
9401 out.extend_from_slice(&value.to_le_bytes());
9402 }
9403 }
9404 (LogicalType::UHugeInt, Data::UInt128(values)) => {
9405 for value in &**values {
9406 out.extend_from_slice(&value.to_le_bytes());
9407 }
9408 }
9409 (LogicalType::Float, Data::Float32(values)) => {
9412 for value in &**values {
9413 out.extend_from_slice(&value.to_le_bytes());
9414 }
9415 }
9416 (LogicalType::Double, Data::Float64(values)) => {
9417 for value in &**values {
9418 out.extend_from_slice(&value.to_le_bytes());
9419 }
9420 }
9421 (LogicalType::Interval, Data::Interval(values)) => {
9425 for (months, days, micros) in &**values {
9426 out.extend_from_slice(&months.to_le_bytes());
9427 out.extend_from_slice(&days.to_le_bytes());
9428 out.extend_from_slice(µs.to_le_bytes());
9429 }
9430 }
9431 (LogicalType::Boolean, Data::Bool(values)) => {
9432 for value in &**values {
9433 out.push(u8::from(*value));
9434 }
9435 }
9436 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
9439 for value in &**values {
9440 out.extend_from_slice(&value.to_le_bytes());
9441 }
9442 }
9443 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
9444 for value in &**values {
9445 out.extend_from_slice(&value.to_le_bytes());
9446 }
9447 }
9448 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
9449 for value in &**values {
9450 out.extend_from_slice(&value.to_le_bytes());
9451 }
9452 }
9453 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
9454 for value in &**values {
9455 out.extend_from_slice(&value.to_le_bytes());
9456 }
9457 }
9458 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
9463 let mut bytes = Vec::new();
9464 put_u32(&mut out, 0);
9465 for row in 0..vector.len() {
9466 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
9467 bytes.extend_from_slice(value);
9468 put_u32(
9469 &mut out,
9470 u32::try_from(bytes.len())
9471 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
9472 );
9473 }
9474 out.extend_from_slice(&bytes);
9475 }
9476 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
9477 }
9478 Ok(out)
9479}
9480
9481fn put_varint(out: &mut Vec<u8>, mut value: u32) {
9482 while value >= 0x80 {
9483 out.push((value as u8 & 0x7f) | 0x80);
9484 value >>= 7;
9485 }
9486 out.push(value as u8);
9487}
9488
9489fn unique_codes(codes: &[u32]) -> Vec<u32> {
9491 let mut unique = codes.to_vec();
9492 unique.sort_unstable();
9493 unique.dedup();
9494 unique
9495}
9496
9497fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
9503 let mut lists = lists;
9504 while lists.len() > 1 {
9505 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
9506 for pair in lists.chunks(2) {
9507 match pair {
9508 [left, right] => next.push(merged_pair(left, right)),
9509 [only] => next.push(only.clone()),
9510 _ => {}
9511 }
9512 }
9513 lists = next;
9514 }
9515 lists.pop().unwrap_or_default()
9516}
9517
9518fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
9519 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
9520 let mut at = 0;
9521 let mut to = 0;
9522 while at < left.len() && to < right.len() {
9523 match left[at].cmp(&right[to]) {
9524 Ordering::Less => {
9525 out.push(left[at]);
9526 at += 1;
9527 }
9528 Ordering::Greater => {
9529 out.push(right[to]);
9530 to += 1;
9531 }
9532 Ordering::Equal => {
9533 out.push(left[at]);
9534 at += 1;
9535 to += 1;
9536 }
9537 }
9538 }
9539 out.extend_from_slice(&left[at..]);
9540 out.extend_from_slice(&right[to..]);
9541 out
9542}
9543
9544fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
9549 let mut merged = Range::default();
9550 let mut first = true;
9551 for range in ranges {
9552 merged.nulls = merged.nulls.saturating_add(range.nulls);
9553 merged.sum = match (merged.sum.take(), range.sum) {
9557 (Some(held), Some(next)) if !first => held.checked_add(next),
9558 (_, next) if first => next,
9559 _ => None,
9560 };
9561 merged.exact = if first { range.exact } else { merged.exact && range.exact };
9562 if first {
9563 merged.low = range.low;
9564 merged.high = range.high;
9565 first = false;
9566 continue;
9567 }
9568 merged.low = match (merged.low.take(), range.low) {
9569 (Some(held), Some(next)) => Some(held.smaller(next)),
9570 _ => None,
9571 };
9572 merged.high = match (merged.high.take(), range.high) {
9573 (Some(held), Some(next)) => Some(held.larger(next)),
9574 _ => None,
9575 };
9576 }
9577 merged
9578}
9579
9580fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
9593 match bound {
9594 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
9595 value.truncate(PART_BOUND_BYTES);
9596 if !high {
9597 return Some(Bound::Bytes(value));
9598 }
9599 while let Some(last) = value.pop() {
9600 if last < u8::MAX {
9601 value.push(last + 1);
9602 return Some(Bound::Bytes(value));
9603 }
9604 }
9605 None
9606 }
9607 other => other,
9608 }
9609}
9610
9611fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
9619 let mut out = Vec::new();
9620 put_u32(
9621 &mut out,
9622 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9623 );
9624 for range in ranges {
9625 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
9626 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
9627 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
9628 }
9629 Ok(out)
9630}
9631
9632fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
9634 let mut cur = Cursor::new(bytes);
9635 let parts = cur.u32()? as usize;
9636 let mut out = Vec::new();
9637 for _ in 0..parts {
9638 let low = cur.bound()?;
9639 let high = cur.bound()?;
9640 let nulls = cur.u32()? as usize;
9641 out.push(Range { low, high, nulls, exact: false, sum: None });
9642 }
9643 Ok(out)
9644}
9645
9646fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
9647 let held: Vec<&Option<Sieve>> = sieves.collect();
9648 let mut out = Vec::new();
9649 put_u32(
9650 &mut out,
9651 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
9652 );
9653 for sieve in &held {
9654 let length = sieve.as_ref().map_or(0, Sieve::len);
9655 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
9656 }
9657 for sieve in held.into_iter().flatten() {
9659 out.extend_from_slice(&sieve.to_bytes());
9660 }
9661 Ok(out)
9662}
9663
9664fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
9670 let parts = u32::from_le_bytes(
9671 bytes
9672 .get(..4)
9673 .ok_or_else(|| invalid("sieve page is truncated"))?
9674 .try_into()
9675 .map_err(|_| invalid("sieve page is truncated"))?,
9676 ) as usize;
9677 let mut lengths = Vec::with_capacity(parts);
9678 for part in 0..parts {
9679 let at = 4 + part * 4;
9680 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
9681 lengths.push(u32::from_le_bytes(
9682 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
9683 ) as usize);
9684 }
9685 let mut at = 4 + parts * 4;
9686 let mut out = Vec::with_capacity(parts);
9687 for length in lengths {
9688 if length == 0 {
9689 out.push(None);
9690 continue;
9691 }
9692 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
9693 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
9694 out.push(Sieve::from_bytes(field));
9695 at = end;
9696 }
9697 if at != bytes.len() {
9698 return Err(invalid("sieve page has trailing bytes"));
9699 }
9700 Ok(out)
9701}
9702
9703fn encode_membership(unique: &[u32]) -> Vec<u8> {
9709 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
9710 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
9711 let mut previous = 0;
9712 for (at, &code) in unique.iter().enumerate() {
9713 put_varint(&mut out, if at == 0 { code } else { code - previous });
9714 previous = code;
9715 }
9716 out
9717}
9718
9719fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
9720 let mut value = 0_u32;
9721 for shift in (0..35).step_by(7) {
9722 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
9723 *at += 1;
9724 let part = u32::from(byte & 0x7f);
9725 if shift == 28 && part > 0x0f {
9726 return Err(invalid("membership varint overflow"));
9727 }
9728 value = value
9729 .checked_add(
9730 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
9731 )
9732 .ok_or_else(|| invalid("membership varint overflow"))?;
9733 if byte & 0x80 == 0 {
9734 return Ok(value);
9735 }
9736 }
9737 Err(invalid("membership varint is too long"))
9738}
9739
9740fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
9741 let mut at = 0;
9742 let count = take_varint(bytes, &mut at)? as usize;
9743 let mut codes = Vec::with_capacity(count);
9744 let mut previous = 0_u32;
9745 for index in 0..count {
9746 let delta = take_varint(bytes, &mut at)?;
9747 let code = if index == 0 {
9748 delta
9749 } else {
9750 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
9751 };
9752 if index > 0 && code <= previous {
9753 return Err(invalid("membership codes are not increasing"));
9754 }
9755 codes.push(code);
9756 previous = code;
9757 }
9758 if at != bytes.len() {
9759 return Err(invalid("membership page has trailing bytes"));
9760 }
9761 Ok(codes)
9762}
9763
9764fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
9765 let mut by_text = HashMap::new();
9766 let mut values = Vec::new();
9767 let mut codes = Vec::with_capacity(vector.len());
9768 let mut plain_bytes = 0_usize;
9769 for row in 0..vector.len() {
9770 let text = vector.bytes_at(row).unwrap_or(b"");
9771 plain_bytes = plain_bytes.saturating_add(text.len());
9772 let code = match by_text.get(text) {
9773 Some(&code) => code,
9774 None => {
9775 let code = u32::try_from(values.len())
9776 .map_err(|_| invalid("too many dictionary values"))?;
9777 by_text.insert(text, code);
9778 values.push(text);
9779 code
9780 }
9781 };
9782 codes.push(code);
9783 }
9784 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
9785 let encoded = 8_usize
9786 .saturating_add((values.len() + 1).saturating_mul(4))
9787 .saturating_add(dictionary_bytes)
9788 .saturating_add(codes.len().saturating_mul(4));
9789 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
9790 if encoded >= plain {
9791 return Ok(None);
9792 }
9793 let mut out = Vec::with_capacity(encoded);
9794 put_u32(
9795 &mut out,
9796 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
9797 );
9798 put_u32(
9799 &mut out,
9800 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
9801 );
9802 let mut offset = 0_u32;
9803 put_u32(&mut out, offset);
9804 for value in &values {
9805 offset = offset
9806 .checked_add(
9807 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
9808 )
9809 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
9810 put_u32(&mut out, offset);
9811 }
9812 for value in values {
9813 out.extend_from_slice(value);
9814 }
9815 for code in codes {
9816 put_u32(&mut out, code);
9817 }
9818 Ok(Some(out))
9819}
9820
9821struct Room<'a, T> {
9823 state: &'a Mutex<(T, usize)>,
9824 finished: &'a Condvar,
9825 bytes: usize,
9826}
9827
9828impl<T> Drop for Room<'_, T> {
9829 fn drop(&mut self) {
9830 let mut held = self.state.lock().unwrap_or_else(PoisonError::into_inner);
9831 held.1 -= self.bytes;
9832 drop(held);
9833 self.finished.notify_all();
9834 }
9835}
9836
9837struct ClosedDictionary {
9839 distinct: u64,
9840 frequencies: FrequencySummary,
9841 texts: Vec<Option<Vec<u8>>>,
9842 hosts: Option<host::HostSummary>,
9843 encoded: EncodedDictionary,
9844 payload: u64,
9846}
9847
9848struct EncodedDictionary {
9849 index: Vec<u8>,
9850 ranks: Vec<u8>,
9851 grams: Vec<u8>,
9852}
9853
9854fn sort_by_value<'a>(codes: &mut [u32], values: impl Fn(u32) -> &'a [u8]) {
9895 let mut work = vec![(0, codes.len(), 0)];
9896 let mut keyed: Vec<(u64, u8, u32)> = Vec::new();
9897 while let Some((from, to, depth)) = work.pop() {
9898 let part = &mut codes[from..to];
9899 keyed.clear();
9900 keyed.extend(part.iter().map(|&code| {
9901 let value = values(code);
9902 let rest = value.get(depth..).unwrap_or_default();
9903 (head(rest), rest.len().min(8) as u8, code)
9904 }));
9905 keyed.sort_unstable();
9906 for (slot, entry) in part.iter_mut().zip(keyed.iter()) {
9907 *slot = entry.2;
9908 }
9909 let mut start = 0;
9910 while start < keyed.len() {
9911 let (key, taken, _) = keyed[start];
9912 let mut end = start + 1;
9913 while end < keyed.len() && keyed[end].0 == key && keyed[end].1 == taken {
9914 end += 1;
9915 }
9916 if taken == 8 && end - start > 1 {
9917 work.push((from + start, from + end, depth + 8));
9918 }
9919 start = end;
9920 }
9921 }
9922}
9923
9924const PARALLEL_SORT_MIN: usize = 1 << 16;
9926
9927const BUCKETS_PER_WORKER: usize = 4;
9930
9931const SAMPLES_PER_BUCKET: usize = 32;
9933
9934fn sort_by_value_across<'a>(
9952 codes: &mut [u32],
9953 values: impl Fn(u32) -> &'a [u8] + Sync,
9954 workers: usize,
9955) {
9956 if workers <= 1 || codes.len() < PARALLEL_SORT_MIN {
9957 sort_by_value(codes, values);
9958 return;
9959 }
9960 let buckets = workers * BUCKETS_PER_WORKER;
9961 let wanted = buckets * SAMPLES_PER_BUCKET;
9962 let mut sample = (0..wanted).map(|at| codes[at * codes.len() / wanted]).collect::<Vec<_>>();
9963 sort_by_value(&mut sample, &values);
9964 let splitters =
9965 (1..buckets).map(|cut| values(sample[cut * sample.len() / buckets])).collect::<Vec<_>>();
9966 let values = &values;
9967 let splitters = &splitters;
9968 let per = codes.len().div_ceil(workers);
9969 let places = std::thread::scope(|scope| {
9971 codes
9972 .chunks(per)
9973 .map(|run| {
9974 scope.spawn(move || {
9975 run.iter()
9976 .map(|&code| {
9977 let value = values(code);
9978 splitters.partition_point(|splitter| *splitter <= value) as u32
9979 })
9980 .collect::<Vec<_>>()
9981 })
9982 })
9983 .collect::<Vec<_>>()
9984 .into_iter()
9985 .flat_map(|handle| {
9986 handle.join().unwrap_or_else(|panic| std::panic::resume_unwind(panic))
9987 })
9988 .collect::<Vec<_>>()
9989 });
9990 let mut starts = vec![0_usize; buckets + 1];
9991 for &place in &places {
9992 starts[place as usize + 1] += 1;
9993 }
9994 for bucket in 0..buckets {
9995 starts[bucket + 1] += starts[bucket];
9996 }
9997 let mut laid = vec![0_u32; codes.len()];
9998 let mut next = starts.clone();
9999 for (&code, &place) in codes.iter().zip(&places) {
10000 laid[next[place as usize]] = code;
10001 next[place as usize] += 1;
10002 }
10003 drop(places);
10004 let mut runs = Vec::with_capacity(buckets);
10005 let mut rest = laid.as_mut_slice();
10006 for bucket in 0..buckets {
10007 let (run, after) = rest.split_at_mut(starts[bucket + 1] - starts[bucket]);
10008 runs.push(run);
10009 rest = after;
10010 }
10011 runs.sort_by_key(|run| run.len());
10013 let queue = Mutex::new(runs);
10014 std::thread::scope(|scope| {
10015 for _ in 0..workers {
10016 scope.spawn(|| {
10017 loop {
10018 let taken = queue.lock().unwrap_or_else(PoisonError::into_inner).pop();
10019 let Some(run) = taken else { break };
10020 sort_by_value(run, values);
10021 }
10022 });
10023 }
10024 });
10025 codes.copy_from_slice(&laid);
10026}
10027
10028fn head(bytes: &[u8]) -> u64 {
10030 let mut word = [0; 8];
10031 let take = bytes.len().min(8);
10032 word[..take].copy_from_slice(&bytes[..take]);
10033 u64::from_be_bytes(word)
10034}
10035
10036fn encode_global_dictionary(
10047 dictionary: &GlobalDictionary,
10048 order: &[(u64, u32)],
10049 places: &[Placed],
10050 scattered: bool,
10051) -> Result<EncodedDictionary> {
10052 let values = dictionary.values();
10053 if order.len() != values {
10054 return Err(invalid("global dictionary order does not cover its values"));
10055 }
10056 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
10057 if places.len() != blocks {
10058 return Err(invalid("global dictionary payload is not the blocks it says it is"));
10059 }
10060 if dictionary.grams.len() != blocks {
10061 return Err(invalid("global dictionary signatures do not cover its blocks"));
10062 }
10063 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
10064 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
10065 let offset_bits = offset_width(&dictionary.ends);
10066 let payload_words = if scattered { 3 } else { 2 };
10067 let index_len = DICTIONARY_HEADER
10068 .checked_add(offset_bytes(values, offset_bits))
10069 .and_then(|len| len.checked_add(blocks.checked_mul(payload_words * 8)?))
10070 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10071 .and_then(|len| len.checked_add(8))
10072 .ok_or_else(|| invalid("global dictionary index length overflow"))?;
10073 let mut index = Vec::with_capacity(index_len);
10074 put_u32(
10075 &mut index,
10076 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
10077 );
10078 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
10079 put_u32(
10080 &mut index,
10081 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
10082 );
10083 let flag = (if scattered { DICTIONARY_SCATTERED } else { 0 })
10084 | DICTIONARY_GRAMS
10085 | DICTIONARY_WIDE_GRAMS;
10086 put_u32(&mut index, offset_bits as u32 | flag);
10087 encode_offsets(&dictionary.ends, offset_bits, &mut index)?;
10088 let mut end = 0_u64;
10093 for place in places {
10094 if scattered {
10095 put_u64(&mut index, place.start);
10096 put_u64(&mut index, place.length);
10097 } else {
10098 end = end
10099 .checked_add(place.length)
10100 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
10101 put_u64(&mut index, end);
10102 }
10103 }
10104 for place in places {
10105 put_u64(&mut index, place.hash);
10106 }
10107 if rank_ends.len() != rank_blocks {
10110 return Err(invalid("global dictionary order is not the blocks it says it is"));
10111 }
10112 for end in &rank_ends {
10113 put_u64(&mut index, *end);
10114 }
10115 let mut at = 0_usize;
10116 for end in &rank_ends {
10117 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
10118 put_u64(&mut index, checksum(&ranks[at..end]));
10119 at = end;
10120 }
10121 let gram_len = blocks
10122 .checked_mul(TEXT_GRAM_BYTES)
10123 .ok_or_else(|| invalid("global dictionary signature count overflow"))?;
10124 let mut grams = Vec::with_capacity(gram_len);
10125 for block in &dictionary.grams {
10126 grams.extend_from_slice(block);
10127 }
10128 put_u64(&mut index, checksum(&grams));
10129 if index.len() != index_len {
10130 return Err(invalid("global dictionary index is not the length it was laid out for"));
10131 }
10132 Ok(EncodedDictionary { index, ranks, grams })
10133}
10134
10135const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
10142
10143fn payload_shapes() -> Vec<chooser::Settled> {
10169 let integers = vec![integer::Kind::Packed];
10170 [
10171 vec![string::Kind::Front, string::Kind::Lz],
10172 vec![string::Kind::Lz, string::Kind::Fsst],
10173 vec![string::Kind::Lz, string::Kind::Plain],
10174 vec![string::Kind::Fsst],
10175 vec![string::Kind::Plain],
10176 ]
10177 .into_iter()
10178 .map(|strings| chooser::Settled::new(strings, integers.clone()))
10179 .collect()
10180}
10181
10182fn synced(file: &File, profile: Option<&LoadProfile>) -> Result<()> {
10189 let started = profile.map(|_| std::time::Instant::now());
10190 file.sync_all().map_err(io)?;
10191 if let (Some(profile), Some(started)) = (profile, started) {
10192 profile.waited(
10193 Stage::Publish,
10194 u64::try_from(started.elapsed().as_nanos()).unwrap_or(u64::MAX),
10195 );
10196 }
10197 Ok(())
10198}
10199
10200#[derive(Debug)]
10205pub(crate) struct Unencoded {
10206 column: usize,
10207 at: usize,
10208 ends: Vec<u32>,
10209 bytes: Vec<u8>,
10210 shape: chooser::Settled,
10211}
10212
10213impl Unencoded {
10214 pub(crate) fn encode(&self) -> Result<EncodedBlock> {
10216 let values = block_values(&self.ends, &self.bytes);
10217 Ok((string::encode_with(&values, &self.shape)?, block_grams(&values)))
10218 }
10219
10220 pub(crate) fn place(&self) -> (usize, usize) {
10222 (self.column, self.at)
10223 }
10224}
10225
10226pub(crate) type EncodedBlock = (Vec<u8>, Box<[u8; TEXT_GRAM_BYTES]>);
10230
10231fn block_grams(values: &[&[u8]]) -> Box<[u8; TEXT_GRAM_BYTES]> {
10233 let mut grams = Box::new([0_u8; TEXT_GRAM_BYTES]);
10234 for value in values {
10235 for gram in value.windows(4) {
10236 for bit in gram_bits(gram, TEXT_GRAM_BYTES) {
10237 grams[bit / 8] |= 1 << (bit % 8);
10238 }
10239 }
10240 }
10241 grams
10242}
10243
10244fn block_values<'a>(ends: &[u32], bytes: &'a [u8]) -> Vec<&'a [u8]> {
10246 let mut out = Vec::with_capacity(ends.len());
10247 let mut from = 0;
10248 for &to in ends {
10249 out.push(&bytes[from..to as usize]);
10250 from = to as usize;
10251 }
10252 out
10253}
10254
10255fn finish_dictionaries(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10262 for dictionary in dictionaries.iter_mut().flatten() {
10263 if !dictionary.early.is_empty() {
10264 return Err(Error::internal("a dictionary block handed out never came back"));
10265 }
10266 dictionary.seal_rest();
10267 }
10268 encode_waiting(dictionaries)?;
10269 if dictionaries
10272 .iter()
10273 .flatten()
10274 .any(|dictionary| dictionary.encoded() != dictionary.values().div_ceil(TEXT_PAYLOAD_VALUES))
10275 {
10276 return Err(Error::internal("a dictionary block handed out never came back"));
10277 }
10278 Ok(())
10279}
10280
10281fn encode_waiting(dictionaries: &mut [Option<GlobalDictionary>]) -> Result<()> {
10284 let jobs = dictionaries
10285 .iter()
10286 .enumerate()
10287 .flat_map(|(column, held)| {
10288 (0..held.as_ref().map_or(0, |held| held.waiting.len())).map(move |at| (column, at))
10289 })
10290 .collect::<Vec<_>>();
10291 if jobs.is_empty() {
10292 return Ok(());
10293 }
10294 let one = |column: usize, at: usize| -> Result<(usize, usize, EncodedBlock)> {
10295 let held = dictionaries[column].as_ref().ok_or_else(|| Error::internal("no dictionary"))?;
10296 Ok((column, at, held.encode_waiting(at)?))
10297 };
10298 let workers = std::thread::available_parallelism()
10299 .map_or(1, usize::from)
10300 .min(MAX_FREQUENCY_WORKERS)
10301 .min(jobs.len());
10302 let made = if workers <= 1 {
10303 jobs.iter().map(|&(column, at)| one(column, at)).collect::<Result<Vec<_>>>()?
10304 } else {
10305 let next = AtomicUsize::new(0);
10306 let jobs = &jobs;
10307 let pieces = std::thread::scope(|scope| {
10308 (0..workers)
10309 .map(|_| {
10310 scope.spawn(|| {
10311 let mut mine = Vec::new();
10312 loop {
10313 let job = next.fetch_add(1, Atomic::Relaxed);
10314 let Some(&(column, at)) = jobs.get(job) else { break };
10315 mine.push(one(column, at)?);
10316 }
10317 Ok(mine)
10318 })
10319 })
10320 .collect::<Vec<_>>()
10321 .into_iter()
10322 .map(|handle| {
10323 handle
10324 .join()
10325 .map_err(|_| Error::internal("a dictionary encode worker panicked"))?
10326 })
10327 .collect::<Result<Vec<_>>>()
10328 })?;
10329 pieces.into_iter().flatten().collect()
10330 };
10331 let mut done: Vec<Vec<(usize, EncodedBlock)>> =
10332 (0..dictionaries.len()).map(|_| Vec::new()).collect();
10333 for (column, at, bytes) in made {
10334 done[column].push((at, bytes));
10335 }
10336 for (column, mut made) in done.into_iter().enumerate() {
10337 if made.is_empty() {
10338 continue;
10339 }
10340 let Some(held) = dictionaries[column].as_mut() else { continue };
10341 made.sort_by_key(|(at, _)| *at);
10342 let waiting = std::mem::take(&mut held.waiting);
10343 for ((at, _), (_, block)) in waiting.into_iter().zip(made) {
10344 if held.encoded() != at {
10345 return Err(Error::internal("a dictionary block was encoded out of order"));
10346 }
10347 held.push_block(block);
10348 }
10349 }
10350 Ok(())
10351}
10352
10353fn settle_shape(sample: &[Vec<&[u8]>]) -> Result<chooser::Settled> {
10363 let mut best: Option<(chooser::Settled, usize)> = None;
10364 for shape in payload_shapes() {
10365 let mut size = 0;
10366 for block in sample {
10367 size += string::encode_with(block, &shape)?.len();
10368 }
10369 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
10370 best = Some((shape, size));
10371 }
10372 }
10373 best.map(|(shape, _)| shape)
10374 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
10375}
10376
10377fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
10384 let mut out = Vec::with_capacity(order.len() * 4);
10385 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
10386 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
10387 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
10388 for block in order.chunks(TEXT_RANK_BLOCK) {
10389 let base = block.first().map_or(0, |&(head, _)| head);
10392 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
10393 let width = (u64::BITS - span.leading_zeros()) as usize;
10394 heads.clear();
10395 codes.clear();
10396 for &(head, code) in block {
10397 heads.push(head.wrapping_sub(base));
10398 codes.push(u64::from(code));
10399 }
10400 put_u64(&mut out, base);
10401 out.push(width as u8);
10402 bitpack::pack_tail(&heads, width, &mut out)
10403 .map_err(|_| invalid("global dictionary heads do not pack"))?;
10404 bitpack::pack_tail(&codes, code_bits, &mut out)
10405 .map_err(|_| invalid("global dictionary codes do not pack"))?;
10406 ends.push(out.len() as u64);
10407 }
10408 Ok((out, ends))
10409}
10410
10411fn open_global_dictionary(
10418 file: Arc<File>,
10419 page: Page,
10420 ty: &LogicalType,
10421 keep_budget: usize,
10422) -> Result<Vector> {
10423 if ty != &LogicalType::Varchar {
10424 return Err(invalid("global dictionary belongs to a non-string column"));
10425 }
10426 let mut header = [0; DICTIONARY_HEADER];
10427 read_at(&file, page.offset, &mut header)?;
10428 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
10429 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
10430 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
10431 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
10432 let scattered = width & DICTIONARY_SCATTERED != 0;
10433 let has_grams = width & DICTIONARY_GRAMS != 0;
10434 let gram_width =
10435 if width & DICTIONARY_WIDE_GRAMS != 0 { TEXT_GRAM_BYTES } else { NARROW_GRAM_BYTES };
10436 let offset_bits = (width & !DICTIONARY_FLAGS) as usize;
10437 if per_block != TEXT_PAYLOAD_VALUES {
10438 return Err(invalid("global dictionary block width differs"));
10439 }
10440 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
10441 return Err(invalid("global dictionary block count differs from its value count"));
10442 }
10443 if offset_bits > u32::BITS as usize {
10444 return Err(invalid("global dictionary packs offsets past a payload"));
10445 }
10446 let offset_len = offset_bytes(count, offset_bits);
10447 let ranks = count;
10452 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
10453 let payload_words = if scattered { 3 } else { 2 };
10457 let hash_len = blocks
10458 .checked_mul(payload_words * 8)
10459 .and_then(|len| len.checked_add(rank_blocks.checked_mul(16)?))
10460 .and_then(|len| len.checked_add(usize::from(has_grams) * 8))
10461 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
10462 let gram_len = if has_grams {
10463 blocks
10464 .checked_mul(gram_width)
10465 .ok_or_else(|| invalid("global dictionary signature count overflow"))?
10466 } else {
10467 0
10468 };
10469 let index_len = DICTIONARY_HEADER
10470 .checked_add(offset_len)
10471 .and_then(|len| len.checked_add(hash_len))
10472 .ok_or_else(|| invalid("global dictionary header overflow"))?;
10473 if index_len > page.length as usize {
10474 return Err(invalid("global dictionary offset index exceeds its page"));
10475 }
10476 let mut index = vec![0; index_len];
10477 index[..DICTIONARY_HEADER].copy_from_slice(&header);
10478 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
10479 if checksum(&index) != page.hash {
10480 return Err(invalid("global dictionary index checksum differs"));
10481 }
10482 let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
10483 let word_end = index_len - usize::from(has_grams) * 8;
10484 let gram_hash = has_grams
10485 .then(|| u64::from_le_bytes(index[word_end..index_len].try_into().expect("eight bytes")));
10486 let mut words = index[DICTIONARY_HEADER + offset_len..word_end]
10487 .chunks_exact(8)
10488 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
10489 .collect::<Vec<_>>();
10490 let mut rest = words.split_off(blocks * payload_words);
10491 let rank_hashes = rest.split_off(rank_blocks);
10492 let rank_ends = rest;
10493 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
10496 return Err(invalid("global dictionary order blocks do not rise"));
10497 }
10498 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
10499 .map_err(|_| invalid("global dictionary rank overflow"))?;
10500 let body_len = index_len
10501 .checked_add(rank_len)
10502 .ok_or_else(|| invalid("global dictionary header overflow"))?;
10503 if body_len > page.length as usize {
10504 return Err(invalid("global dictionary order exceeds its page"));
10505 }
10506 let gram_end = body_len
10507 .checked_add(gram_len)
10508 .ok_or_else(|| invalid("global dictionary signature length overflow"))?;
10509 if gram_end > page.length as usize {
10510 return Err(invalid("global dictionary signatures exceed their page"));
10511 }
10512 let grams = gram_hash.map(|hash| NativeGrams {
10513 start: page.offset + body_len as u64,
10514 length: gram_len,
10515 width: gram_width,
10516 hash,
10517 verdicts: Mutex::new(Vec::new()),
10518 });
10519 let hashes = words.split_off(blocks * (payload_words - 1));
10520 let (starts, lengths) = if scattered {
10521 let mut starts = Vec::with_capacity(blocks);
10522 let mut lengths = Vec::with_capacity(blocks);
10523 for pair in words.chunks_exact(2) {
10524 starts.push(pair[0]);
10525 lengths.push(pair[1]);
10526 }
10527 (starts, lengths)
10528 } else {
10529 let base = page.offset + gram_end as u64;
10533 let mut starts = Vec::with_capacity(blocks);
10534 let mut lengths = Vec::with_capacity(blocks);
10535 let mut at = 0_u64;
10536 for &end in &words {
10537 let len = end
10538 .checked_sub(at)
10539 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
10540 starts.push(base + at);
10541 lengths.push(len);
10542 at = end;
10543 }
10544 (starts, lengths)
10545 };
10546 let stored_len = page.length as u64 - gram_end as u64;
10552 if scattered && stored_len == 0 {
10553 let size = file.metadata().map_err(io)?.len();
10554 let inside = starts.iter().zip(&lengths).all(|(&start, &len)| {
10555 start >= HEADER && start.checked_add(len).is_some_and(|end| end <= size)
10556 });
10557 if !inside {
10558 return Err(invalid("global dictionary block lies outside the file"));
10559 }
10560 } else if lengths.iter().try_fold(0_u64, |sum, len| sum.checked_add(*len)) != Some(stored_len) {
10561 return Err(invalid("global dictionary blocks do not bound the payload"));
10562 }
10563 Vector::external_text(
10564 LogicalType::Varchar,
10565 Arc::new(NativeText {
10566 file,
10567 values: count,
10568 offsets,
10569 offset_bits,
10570 value_ends: OnceLock::new(),
10571 value_lens: OnceLock::new(),
10572 ends_asked: AtomicUsize::new(0),
10573 ranks,
10574 rank_at: page.offset + index_len as u64,
10575 rank_ends,
10576 rank_hashes,
10577 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
10578 code_bits: code_width(count),
10579 code_ranks: OnceLock::new(),
10580 starts,
10581 lengths,
10582 hashes,
10583 grams,
10584 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
10585 keep_budget,
10586 payload_kept: AtomicUsize::new(0),
10587 swept: (0..blocks).map(|_| AtomicBool::new(false)).collect(),
10588 searched: Mutex::new(HashMap::new()),
10589 }),
10590 )
10591}
10592
10593fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
10606 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
10608 let mut cur = Cursor::new(bytes);
10609 let codec = cur.u8()?;
10610 if cur.u8()? == 2 {
10611 cur.take(rows.div_ceil(8))?;
10612 }
10613 Ok((codec, cur.at))
10614 }
10615 let Ok((codec, at)) = cascade_at(rows, bytes) else {
10616 return "UNREADABLE".to_string();
10617 };
10618 let tail = &bytes[at..];
10619 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
10620 match codec {
10621 0 => match ty {
10622 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
10623 _ => "FIXED".to_string(),
10624 },
10625 1 => "DICT(PLAIN)".to_string(),
10626 2 => "FOR+BITPACK".to_string(),
10627 3 => "TABLE DICT".to_string(),
10628 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
10629 5 => described(integer::describe(tail)),
10630 6 => described(string::describe(tail)),
10631 other => format!("CODEC {other}"),
10632 }
10633}
10634
10635fn decode_selected_stable_codes(
10640 rows: usize,
10641 bytes: &[u8],
10642 positions: &[usize],
10643 out: &mut Vec<Option<u32>>,
10644) -> Result<bool> {
10645 if positions.windows(2).any(|pair| pair[0] >= pair[1])
10646 || positions.last().is_some_and(|&position| position >= rows)
10647 {
10648 return Err(invalid("selected code positions are not sorted and in range"));
10649 }
10650 let mut cur = Cursor::new(bytes);
10651 let codec = cur.u8()?;
10652 if codec != 3 && codec != 4 {
10653 return Ok(false);
10654 }
10655 let flag = cur.u8()?;
10656 let mask = match flag {
10657 0 | 1 => None,
10658 2 => {
10659 let at = cur.at;
10660 let len = rows.div_ceil(8);
10661 cur.take(len)?;
10662 Some((at, len))
10663 }
10664 _ => return Err(invalid("page validity tag differs")),
10665 };
10666 let valid = |row: usize| match flag {
10667 0 => true,
10668 1 => false,
10669 2 => mask.is_some_and(|(at, _)| bytes[at + row / 8] >> (row % 8) & 1 == 1),
10670 _ => unreachable!("the validity tag was checked"),
10671 };
10672 if codec == 4 {
10673 let wide = integer::decode_selected(&bytes[cur.at..], positions)?;
10674 for (&row, code) in positions.iter().zip(wide) {
10675 let code = u32::try_from(code).map_err(|_| invalid("code is not a code"))?;
10676 out.push(valid(row).then_some(code));
10677 }
10678 return Ok(true);
10679 }
10680 let codes_at = cur.at;
10681 let codes_len = rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?;
10682 cur.take(codes_len)?;
10683 if cur.at != bytes.len() {
10684 return Err(invalid("global code page has trailing bytes"));
10685 }
10686 let codes = &bytes[codes_at..codes_at + codes_len];
10687 for &row in positions {
10688 let at = row.checked_mul(4).ok_or_else(|| invalid("dictionary code offset overflow"))?;
10689 let code = u32::from_le_bytes(
10690 codes[at..at + 4].try_into().map_err(|_| invalid("dictionary code is truncated"))?,
10691 );
10692 out.push(valid(row).then_some(code));
10693 }
10694 Ok(true)
10695}
10696
10697fn decode_at(
10703 ty: &LogicalType,
10704 rows: usize,
10705 bytes: &[u8],
10706 global: Option<Arc<Vector>>,
10707 positions: &[u32],
10708) -> Result<Vector> {
10709 if positions.last().is_some_and(|&last| last as usize >= rows) {
10710 return Err(invalid("a position is past the end of the part"));
10711 }
10712 if bytes.first() != Some(&6) {
10713 return decode(ty, rows, bytes, global)?.gather(positions);
10714 }
10715 if ty != &LogicalType::Varchar {
10716 return Err(invalid("compressed text codec belongs to a non-string page"));
10717 }
10718 let mut cur = Cursor::new(bytes);
10719 cur.u8()?;
10720 let validity = match cur.u8()? {
10721 0 => Validity::AllValid,
10722 1 => Validity::AllInvalid,
10723 2 => {
10724 let mask = cur.take(rows.div_ceil(8))?;
10725 Validity::from_iter(positions.len(), |at| {
10726 let row = positions[at] as usize;
10727 mask[row / 8] >> (row % 8) & 1 == 1
10728 })
10729 }
10730 _ => return Err(invalid("page validity tag differs")),
10731 };
10732 let (payload, ends) = string::decode_flat_at(&bytes[cur.at..], positions)?.into_parts();
10733 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10734 let mut start = 0;
10735 for end in ends {
10736 let len = end
10737 .checked_sub(start)
10738 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10739 values.push_in_place(start, len)?;
10740 start = end;
10741 }
10742 Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity))
10743}
10744
10745fn decode(
10746 ty: &LogicalType,
10747 rows: usize,
10748 bytes: &[u8],
10749 global: Option<Arc<Vector>>,
10750) -> Result<Vector> {
10751 let mut cur = Cursor::new(bytes);
10752 let codec = cur.u8()?;
10753 let flag = cur.u8()?;
10754 let validity = match flag {
10755 0 => Validity::AllValid,
10756 1 => Validity::AllInvalid,
10757 2 => {
10758 let mask = cur.take(rows.div_ceil(8))?;
10759 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
10760 }
10761 _ => return Err(invalid("page validity tag differs")),
10762 };
10763 if codec == 1 {
10764 if ty != &LogicalType::Varchar {
10765 return Err(invalid("dictionary codec belongs to a non-string page"));
10766 }
10767 let count = cur.u32()? as usize;
10768 let payload_len = cur.u32()? as usize;
10769 let offset_bytes = cur.take(
10770 (count + 1)
10771 .checked_mul(4)
10772 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
10773 )?;
10774 let offsets = offset_bytes
10775 .chunks_exact(4)
10776 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
10777 .collect::<Vec<_>>();
10778 let payload = cur.take(payload_len)?.to_vec();
10779 if offsets.first() != Some(&0)
10780 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
10781 || offsets.windows(2).any(|pair| pair[0] > pair[1])
10782 {
10783 return Err(invalid("dictionary offsets do not bound the payload"));
10784 }
10785 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
10788 for pair in offsets.windows(2) {
10789 strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
10790 }
10791 let mut codes = Vec::with_capacity(rows);
10792 for _ in 0..rows {
10793 codes.push(cur.u32()?);
10794 }
10795 if codes.iter().any(|code| *code as usize >= count) {
10796 return Err(invalid("dictionary code is out of range"));
10797 }
10798 if cur.at != bytes.len() {
10799 return Err(invalid("dictionary page has trailing bytes"));
10800 }
10801 let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
10802 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
10803 }
10804 if codec == 3 || codec == 4 {
10805 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
10806 let codes = if codec == 4 {
10807 let wide = integer::decode(&bytes[cur.at..])?;
10810 if wide.len() != rows {
10811 return Err(invalid("encoded code page holds the wrong number of rows"));
10812 }
10813 let seen = wide.iter().fold(0_i64, |seen, &code| seen | code);
10820 if seen < 0 || seen > i64::from(u32::MAX) {
10821 return Err(invalid("code is not a code"));
10822 }
10823 wide.iter().map(|&code| code as u32).collect()
10824 } else {
10825 let mut codes = Vec::with_capacity(rows);
10826 for _ in 0..rows {
10827 codes.push(cur.u32()?);
10828 }
10829 if cur.at != bytes.len() {
10830 return Err(invalid("global code page has trailing bytes"));
10831 }
10832 codes
10833 };
10834 let highest = codes.iter().copied().max();
10835 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
10836 .with_validity(validity));
10837 }
10838 if codec == 6 {
10839 if ty != &LogicalType::Varchar {
10840 return Err(invalid("compressed text codec belongs to a non-string page"));
10841 }
10842 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
10846 if ends.len() != rows {
10847 return Err(invalid("compressed text page holds the wrong number of rows"));
10848 }
10849 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
10852 let mut start = 0;
10853 for end in ends {
10854 let len = end
10855 .checked_sub(start)
10856 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
10857 values.push_in_place(start, len)?;
10858 start = end;
10859 }
10860 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
10861 }
10862 if codec == 5 {
10863 let values = integer::decode(&bytes[cur.at..])?;
10865 if values.len() != rows {
10866 return Err(invalid("cascade page holds the wrong number of rows"));
10867 }
10868 let data = narrowed(ty, values)?;
10869 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
10870 }
10871 if codec == 2 {
10872 let width = u32::from(cur.u8()?);
10873 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
10874 let count = cur.u32()? as usize;
10875 let length = count.checked_mul(8).ok_or_else(|| invalid("packed page is too long"))?;
10876 let words: Vec<u64> = cur
10877 .take(length)?
10878 .chunks_exact(8)
10879 .map(|word| u64::from_le_bytes(word.try_into().expect("eight bytes")))
10880 .collect();
10881 if cur.at != bytes.len() {
10882 return Err(invalid("packed page has trailing bytes"));
10883 }
10884 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
10885 }
10886 if codec != 0 {
10887 return Err(invalid("page codec is unknown"));
10888 }
10889 let data = match ty {
10890 LogicalType::TinyInt => {
10891 let values = cur.take(rows)?;
10892 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
10893 }
10894 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
10895 LogicalType::SmallInt => {
10896 let values =
10897 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10898 Data::Int16(
10899 values
10900 .chunks_exact(2)
10901 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
10902 .collect::<Vec<_>>()
10903 .into(),
10904 )
10905 }
10906 LogicalType::USmallInt => {
10907 let values =
10908 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
10909 Data::UInt16(
10910 values
10911 .chunks_exact(2)
10912 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
10913 .collect::<Vec<_>>()
10914 .into(),
10915 )
10916 }
10917 LogicalType::UInteger => {
10918 let values =
10919 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10920 Data::UInt32(
10921 values
10922 .chunks_exact(4)
10923 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
10924 .collect::<Vec<_>>()
10925 .into(),
10926 )
10927 }
10928 LogicalType::UBigInt => {
10929 let values =
10930 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10931 Data::UInt64(
10932 values
10933 .chunks_exact(8)
10934 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
10935 .collect::<Vec<_>>()
10936 .into(),
10937 )
10938 }
10939 LogicalType::Integer | LogicalType::Date => {
10940 let values =
10941 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10942 Data::Int32(
10943 values
10944 .chunks_exact(4)
10945 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
10946 .collect::<Vec<_>>()
10947 .into(),
10948 )
10949 }
10950 LogicalType::BigInt
10951 | LogicalType::Timestamp
10952 | LogicalType::Time
10953 | LogicalType::TimeTz
10954 | LogicalType::TimestampTz
10955 | LogicalType::TimestampS
10956 | LogicalType::TimestampMs
10957 | LogicalType::TimestampNs => {
10958 let values =
10959 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
10960 Data::Int64(
10961 values
10962 .chunks_exact(8)
10963 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
10964 .collect::<Vec<_>>()
10965 .into(),
10966 )
10967 }
10968 LogicalType::HugeInt | LogicalType::Uuid => {
10969 let values =
10970 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10971 Data::Int128(
10972 values
10973 .chunks_exact(16)
10974 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10975 .collect::<Vec<_>>()
10976 .into(),
10977 )
10978 }
10979 LogicalType::UHugeInt => {
10980 let values =
10981 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
10982 Data::UInt128(
10983 values
10984 .chunks_exact(16)
10985 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
10986 .collect::<Vec<_>>()
10987 .into(),
10988 )
10989 }
10990 LogicalType::Float => {
10991 let values =
10992 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
10993 Data::Float32(
10994 values
10995 .chunks_exact(4)
10996 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
10997 .collect::<Vec<_>>()
10998 .into(),
10999 )
11000 }
11001 LogicalType::Double => {
11002 let values =
11003 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11004 Data::Float64(
11005 values
11006 .chunks_exact(8)
11007 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
11008 .collect::<Vec<_>>()
11009 .into(),
11010 )
11011 }
11012 LogicalType::Interval => {
11013 let values =
11014 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
11015 Data::Interval(
11016 values
11017 .chunks_exact(16)
11018 .map(|item| {
11019 (
11020 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
11021 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
11022 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
11023 )
11024 })
11025 .collect::<Vec<_>>()
11026 .into(),
11027 )
11028 }
11029 LogicalType::Boolean => {
11030 let values = cur.take(rows)?;
11031 if values.iter().any(|value| *value > 1) {
11032 return Err(invalid("boolean page has another value"));
11033 }
11034 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
11035 }
11036 LogicalType::Decimal { .. } => match ty.physical() {
11039 PhysicalType::Int16 => {
11040 let values =
11041 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
11042 Data::Int16(
11043 values
11044 .chunks_exact(2)
11045 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
11046 .collect::<Vec<_>>()
11047 .into(),
11048 )
11049 }
11050 PhysicalType::Int32 => {
11051 let values =
11052 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
11053 Data::Int32(
11054 values
11055 .chunks_exact(4)
11056 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
11057 .collect::<Vec<_>>()
11058 .into(),
11059 )
11060 }
11061 PhysicalType::Int64 => {
11062 let values =
11063 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
11064 Data::Int64(
11065 values
11066 .chunks_exact(8)
11067 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
11068 .collect::<Vec<_>>()
11069 .into(),
11070 )
11071 }
11072 _ => {
11073 let values =
11074 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
11075 Data::Int128(
11076 values
11077 .chunks_exact(16)
11078 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
11079 .collect::<Vec<_>>()
11080 .into(),
11081 )
11082 }
11083 },
11084 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
11085 let offset_bytes = cur
11086 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
11087 let offsets = offset_bytes
11088 .chunks_exact(4)
11089 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
11090 .collect::<Vec<_>>();
11091 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
11092 if offsets.first() != Some(&0)
11093 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
11094 || offsets.windows(2).any(|pair| pair[0] > pair[1])
11095 {
11096 return Err(invalid("string offsets do not bound the payload"));
11097 }
11098 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
11106 let text = ty == &LogicalType::Varchar;
11107 for pair in offsets.windows(2) {
11108 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
11109 if text {
11110 values.push_in_place(at, len)?;
11111 } else {
11112 values.push_bytes_in_place(at, len)?;
11113 }
11114 }
11115 Data::Varlen(values)
11116 }
11117 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
11118 };
11119 if cur.at != bytes.len() {
11120 return Err(invalid("page has trailing bytes"));
11121 }
11122 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
11123}
11124
11125#[cfg(test)]
11126mod tests {
11127 use std::fs;
11128 use std::io::{Seek, SeekFrom, Write};
11129 use std::path::PathBuf;
11130 use std::time::{SystemTime, UNIX_EPOCH};
11131
11132 use rudb_common::Stat;
11133 use rudb_common::Value;
11134 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
11135 use rudb_common::stat::Provenance;
11136
11137 use super::*;
11138
11139 #[test]
11140 fn a_name_taken_in_pieces_is_the_name_of_the_pieces_joined() {
11141 let bytes: Vec<u8> =
11142 (0..300_u32).map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8).collect();
11143 for length in [0, 1, 7, 31, 32, 33, 63, 64, 65, 100, 300] {
11144 let whole = content_name(&bytes[..length]);
11145 for step in [1, 3, 8, 31, 32, 33, 64, 301] {
11146 let mut namer = ContentNamer::default();
11147 bytes[..length].chunks(step).for_each(|piece| namer.update(piece));
11148 assert_eq!(namer.finish(), whole, "{length} bytes in pieces of {step}");
11149 }
11150 }
11151 }
11152
11153 #[derive(Debug)]
11156 struct TestsEverything<'a>(&'a dyn chooser::Chooser);
11157
11158 impl chooser::Chooser for TestsEverything<'_> {
11159 fn name(&self) -> &'static str {
11160 "tests everything"
11161 }
11162
11163 fn narrow_strings(
11164 &self,
11165 values: &[&[u8]],
11166 offered: &[string::Kind],
11167 depth: u8,
11168 ) -> Vec<string::Kind> {
11169 self.0.narrow_strings(values, offered, depth)
11170 }
11171
11172 fn narrow_integers(
11173 &self,
11174 values: &[i64],
11175 offered: &[integer::Kind],
11176 depth: u8,
11177 ) -> Vec<integer::Kind> {
11178 self.0.narrow_integers(values, offered, depth)
11179 }
11180 }
11181
11182 #[test]
11183 fn ruling_kinds_out_before_testing_for_them_writes_the_same_bytes() {
11184 let columns: Vec<Vec<i64>> = vec![
11185 vec![],
11186 vec![5; 1000],
11187 (0..1000).collect(),
11188 (0..1000).map(|row| 1_600_000_000_000_000 + row * 1_000_000).collect(),
11189 (0..1000).map(|row| row / 50).collect(),
11190 (0..1000).map(|row| if row % 97 == 0 { row } else { 0 }).collect(),
11191 (0..1000).map(|row| (row * 7919) % 13).collect(),
11192 (0..1000).map(|row| (row * 2_654_435_761) % 1_000_003).collect(),
11193 (0..1000).map(|row| [3, 3, 3, 9, 9, 1][row as usize % 6]).collect(),
11194 (0..1000).map(|row| i64::MIN + row % 3).collect(),
11195 ];
11196 let choosers: [&dyn chooser::Chooser; 2] = [&Fixed, &Codes];
11197 for column in &columns {
11198 for chooser in choosers {
11199 let quick = integer::encode_with(column, chooser).unwrap();
11200 let full = integer::encode_with(column, &TestsEverything(chooser)).unwrap();
11201 assert_eq!(
11202 quick,
11203 full,
11204 "{} on {:?}",
11205 chooser.name(),
11206 &column[..column.len().min(8)]
11207 );
11208 }
11209 }
11210 }
11211
11212 #[test]
11215 fn parts_that_look_alike_replay_to_the_bytes_a_search_writes() {
11216 let mut settling = Settling::default();
11217 for part in 0..STRIPE_PARTS as i64 {
11218 let values: Vec<i64> = (0..2048)
11219 .map(|row| 1_600_000_000_000_000 + (part * 2048 + row) * 1_000_000 + row % 7)
11220 .collect();
11221 let searched = integer::encode_with(&values, &Fixed).unwrap();
11222 assert_eq!(settling.encode(&values).unwrap(), searched, "part {part}");
11223 }
11224 }
11225
11226 #[test]
11230 fn a_column_that_changes_under_the_shape_is_searched_again() {
11231 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11232 let mut noise = move || {
11233 state ^= state << 13;
11234 state ^= state >> 7;
11235 state ^= state << 17;
11236 (state % 1_000_000) as i64
11237 };
11238 let mut settling = Settling::default();
11239 for part in 0..STRIPE_PARTS as i64 {
11240 let values: Vec<i64> = match part / 16 {
11241 0 => (0..2048).map(|row| (part * 2048 + row) / 300).collect(),
11242 1 => (0..2048).map(|_| noise()).collect(),
11243 2 => (0..2048).map(|row| if row % 97 == 0 { row } else { 42 }).collect(),
11244 _ => (0..2048).map(|row| 5 + (part * 2048 + row) * 1_000_000).collect(),
11245 };
11246 let settled = settling.encode(&values).unwrap();
11247 assert_eq!(integer::decode(&settled).unwrap(), values, "part {part}");
11248 let searched = integer::encode_with(&values, &Fixed).unwrap();
11249 assert!(
11250 settled.len() * 4 <= searched.len() * 5,
11251 "part {part}: {} settled against {} searched, {} against {}",
11252 settled.len(),
11253 searched.len(),
11254 integer::describe(&settled).unwrap(),
11255 integer::describe(&searched).unwrap(),
11256 );
11257 }
11258 }
11259
11260 #[test]
11261 fn checksum_matches_fixed_vectors() {
11262 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
11263 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
11264 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
11265 }
11266
11267 #[test]
11268 fn sorting_across_threads_matches_sorting_on_one() {
11269 let mut state = 0x9e37_79b9_7f4a_7c15_u64;
11270 let mut next = move || {
11271 state ^= state << 13;
11272 state ^= state >> 7;
11273 state ^= state << 17;
11274 state
11275 };
11276 let mut values = Vec::new();
11277 for at in 0..150_000_u64 {
11278 let value = match next() % 6 {
11279 0 => Vec::new(),
11280 1 => format!("https://example.com/{}", next() % 5_000).into_bytes(),
11281 2 => format!("https://example.com/path/{at}").into_bytes(),
11282 3 => b"same".to_vec(),
11283 4 => vec![0xff; (next() % 12) as usize],
11284 _ => (0..next() % 20).map(|_| (next() % 3) as u8).collect(),
11285 };
11286 values.push(value);
11287 }
11288 let value = |code: u32| values[code as usize].as_slice();
11289 for workers in [1, 2, 3, 8, 32] {
11290 let mut one = (0..values.len() as u32).rev().collect::<Vec<_>>();
11291 let mut across = one.clone();
11292 sort_by_value(&mut one, value);
11293 sort_by_value_across(&mut across, value, workers);
11294 assert_eq!(one, across, "{workers} workers");
11295 }
11296 let mut sorted = (0..values.len() as u32).collect::<Vec<_>>();
11297 sort_by_value_across(&mut sorted, value, 8);
11298 assert!(sorted.windows(2).all(|pair| value(pair[0]) <= value(pair[1])));
11299 }
11300
11301 fn path(label: &str) -> PathBuf {
11302 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
11303 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
11304 }
11305
11306 fn dictionary_values(dictionary: &GlobalDictionary) -> Vec<Vec<u8>> {
11311 let (flat, bases) = dictionary.decoded(None).expect("the blocks decode");
11312 (0..dictionary.values())
11313 .map(|code| {
11314 let (from, to) = GlobalDictionary::value_span(&dictionary.ends, &bases, code);
11315 flat[from..to].to_vec()
11316 })
11317 .collect()
11318 }
11319
11320 fn attached(table: &Table) -> Vec<&Section> {
11327 table.sections().iter().filter(|held| !held.among(section::STATISTICS_KINDS)).collect()
11328 }
11329
11330 #[test]
11332 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
11333 const SPANS: usize = 64;
11334 const SPAN: usize = 512;
11335 let path = path("positional");
11336 let content: Vec<u8> =
11337 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
11338 fs::write(&path, &content).expect("the file is written");
11339 let file = Arc::new(File::open(&path).expect("the file opens"));
11340 std::thread::scope(|scope| {
11341 for _ in 0..8 {
11342 let file = Arc::clone(&file);
11343 scope.spawn(move || {
11344 for _ in 0..64 {
11345 for span in 0..SPANS {
11346 let mut bytes = [0_u8; SPAN];
11347 read_at(&file, (span * SPAN) as u64, &mut bytes)
11348 .expect("the span reads");
11349 assert!(
11350 bytes.iter().all(|byte| *byte == span as u8),
11351 "span {span} came back as {}",
11352 bytes[0],
11353 );
11354 }
11355 }
11356 });
11357 }
11358 });
11359 let mut past = [0_u8; SPAN];
11360 let end = (SPANS * SPAN) as u64;
11361 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
11362 assert!(error.message().contains("ends before its declared length"), "{error}");
11363 drop(file);
11364 let _ = fs::remove_file(&path);
11365 }
11366
11367 #[test]
11373 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
11374 let path = path("cursor");
11375 let mut writer = Writer::create(
11376 &path,
11377 "items",
11378 vec![
11379 Field::required("id", LogicalType::Integer),
11380 Field::new("text", LogicalType::Varchar),
11381 ],
11382 )
11383 .expect("new file");
11384 writer.append(&sample()).expect("first part");
11385 writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
11386 writer.append(&sample()).expect("second part");
11387 writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
11388 writer.finish().expect("commit");
11389 let reader = Reader::open(&path).expect("reopen from disk");
11390 assert_eq!(reader.table().rows(), 6);
11391 let ids = reader.read(0, &[0]).expect("the integer page reads back");
11392 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
11393 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
11394 let text = reader.read(1, &[1]).expect("the text page reads back");
11395 assert_eq!(text.value_at(1, 0), Value::Null);
11396 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
11397 let end = reader.table().stripes().iter().flat_map(|stripe| {
11400 stripe
11401 .pages
11402 .iter()
11403 .map(|page| page.offset + u64::from(page.length))
11404 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
11405 });
11406 let last = end.fold(HEADER, u64::max);
11407 let directory = fs::metadata(&path).expect("the file is there").len();
11408 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
11409 fs::remove_file(path).expect("remove scratch file");
11410 }
11411
11412 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
11418 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
11419 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
11420 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
11421 let bits = (width & !DICTIONARY_FLAGS) as usize;
11422 let payload_words = if width & DICTIONARY_SCATTERED == 0 { 2 } else { 3 };
11423 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
11424 DICTIONARY_HEADER as u64
11425 + offset_bytes(count as usize, bits) as u64
11426 + blocks * payload_words * 8
11427 + rank_blocks * 16
11428 + if width & DICTIONARY_GRAMS == 0 { 0 } else { 8 }
11429 }
11430
11431 fn sample() -> Chunk {
11432 Chunk::new(vec![
11433 Vector::from_values(
11434 LogicalType::Integer,
11435 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
11436 )
11437 .expect("integers"),
11438 Vector::from_values(
11439 LogicalType::Varchar,
11440 &[
11441 Value::Varchar("alpha".into()),
11442 Value::Null,
11443 Value::Varchar("long text after a slash".into()),
11444 ],
11445 )
11446 .expect("strings"),
11447 ])
11448 .expect("matching rows")
11449 }
11450
11451 fn sample_ids() -> Chunk {
11452 Chunk::new(vec![
11453 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
11454 .expect("integers"),
11455 ])
11456 .expect("one column")
11457 }
11458
11459 #[test]
11460 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
11461 let path = path("nulls_for_the_planner");
11464 let mut writer =
11465 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
11466 .expect("new file");
11467 let rows = Chunk::new(vec![
11468 Vector::from_values(
11469 LogicalType::Integer,
11470 &[
11471 Value::Integer(4),
11472 Value::Null,
11473 Value::Integer(9),
11474 Value::Null,
11475 Value::Integer(1),
11476 Value::Integer(2),
11477 ],
11478 )
11479 .expect("integers"),
11480 ])
11481 .expect("one column");
11482 writer.append(&rows).expect("the only part");
11483 writer.finish().expect("commit");
11484 let reader = Reader::open(&path).expect("reopen from disk");
11485 let stripes = Stripes::new(reader);
11486 let column = stripes.column("a").expect("the file has that column");
11487 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
11488 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
11491 fs::remove_file(&path).expect("clean up");
11492 }
11493
11494 #[test]
11495 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
11496 let path = path("frequencies_for_the_planner");
11501 let mut writer =
11502 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11503 .expect("new file");
11504 let rows = Chunk::new(vec![
11505 Vector::from_values(
11506 LogicalType::Integer,
11507 &[
11508 Value::Integer(4),
11509 Value::Integer(4),
11510 Value::Integer(4),
11511 Value::Integer(9),
11512 Value::Integer(9),
11513 Value::Integer(1),
11514 ],
11515 )
11516 .expect("integers"),
11517 ])
11518 .expect("one column");
11519 writer.append(&rows).expect("the only part");
11520 writer.finish().expect("commit");
11521 let reader = Reader::open(&path).expect("reopen from disk");
11522 let common = Common::new(reader);
11523 assert_eq!(common.rows(), 6);
11524 let column = common.column("id").expect("the file has that column");
11525 assert_eq!(common.column("nothing"), None);
11526 assert_eq!(
11527 common.rows_with(column, &Bound::Int(4)),
11528 Stat::exact(3, Provenance::FrequencySynopsis)
11529 );
11530 assert_eq!(
11532 common.rows_with(column, &Bound::Int(7)),
11533 Stat::exact(0, Provenance::FrequencySynopsis)
11534 );
11535 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
11538 assert_eq!(common.remainder(column), None);
11541 fs::remove_file(&path).expect("clean up");
11542 }
11543
11544 #[test]
11545 fn string_frequency_estimates_do_not_open_the_global_dictionary() {
11546 let path = path("string_frequencies_for_the_planner");
11547 let mut writer =
11548 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
11549 .expect("new file");
11550 let rows = Chunk::new(vec![
11551 Vector::from_values(
11552 LogicalType::Varchar,
11553 &[
11554 Value::Varchar(String::new()),
11555 Value::Varchar("alpha".into()),
11556 Value::Varchar(String::new()),
11557 Value::Varchar("beta".into()),
11558 Value::Varchar(String::new()),
11559 ],
11560 )
11561 .expect("strings"),
11562 ])
11563 .expect("one column");
11564 writer.append(&rows).expect("the only part");
11565 writer.finish().expect("commit");
11566
11567 let reader = Reader::open(&path).expect("reopen from disk");
11568 assert_eq!(reader.reads().dictionaries, 0, "open reads only the directory");
11569 let common = Common::new(reader.clone());
11570 let column = common.column("text").expect("the file has that column");
11571 assert_eq!(
11572 common.rows_with(column, &Bound::Bytes(Vec::new())),
11573 Stat::exact(3, Provenance::FrequencySynopsis)
11574 );
11575 assert_eq!(
11576 common.rows_with(column, &Bound::Bytes(b"missing".to_vec())),
11577 Stat::exact(0, Provenance::FrequencySynopsis)
11578 );
11579 assert_eq!(
11580 reader.reads().dictionaries,
11581 0,
11582 "the bounded spellings answer without opening the dictionary index"
11583 );
11584 fs::remove_file(&path).expect("clean up");
11585 }
11586
11587 #[test]
11588 fn host_groups_certify_omitted_hosts_and_keep_exact_aggregates() {
11589 let path = path("certified_host_groups");
11590 let mut writer =
11591 Writer::create(&path, "hits", vec![Field::required("Referer", LogicalType::Varchar)])
11592 .expect("new file");
11593 let mut values = vec![Value::Varchar("http://www.example.com/a".into()); 150];
11594 values.extend(vec![Value::Varchar("https://example.com/b".into()); 70]);
11595 values.extend((0..550).map(|at| Value::Varchar(format!("https://site{at}.test/x"))));
11596 values.push(Value::Varchar(String::new()));
11597 for part in values.chunks(512) {
11598 writer
11599 .append(
11600 &Chunk::new(vec![
11601 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
11602 ])
11603 .expect("one column"),
11604 )
11605 .expect("part written");
11606 }
11607 writer.finish().expect("commit");
11608 let reader = Reader::open(&path).expect("reopen");
11609 assert!(reader.table.host_groups.is_none(), "no query-specific host result is stored");
11610 fs::remove_file(&path).expect("clean up");
11611 }
11612
11613 fn bare_table(sections: Vec<Section>) -> Table {
11618 Table {
11619 name: "linked".to_owned(),
11620 fields: vec![Field::required("id", LogicalType::Integer)],
11621 stripes: Vec::new(),
11622 rows: 0,
11623 dictionaries: vec![None],
11624 dictionary_payloads: Vec::new(),
11625 distincts: vec![None],
11626 frequencies: vec![None],
11627 pair_frequencies: Vec::new(),
11628 frequency_texts: Vec::new(),
11629 host_groups: None,
11630 clustering: None,
11631 generation: 1,
11632 sections,
11633 }
11634 }
11635
11636 fn a_key_map_section() -> Section {
11637 Section {
11638 kind: *section::KEY_MAP,
11639 id: 1,
11640 generation: 3,
11641 extents: 1,
11642 extent_page: HEADER,
11643 extent_bytes: section::EXTENT_BYTES as u32,
11644 hash: 0x1234_5678_9abc_def0,
11645 flags: 0,
11646 header_bytes: 24,
11647 }
11648 }
11649
11650 #[test]
11651 fn a_section_table_round_trips_through_a_directory() {
11652 let mut later = a_key_map_section();
11653 later.kind = *b"RUDBZZ9\0";
11654 later.id = 2;
11655 let table = bare_table(vec![a_key_map_section(), later]);
11656 let directory = encode_directory(&table).expect("directory");
11657 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11658 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
11659 assert!(decoded.sections()[0].known());
11663 assert!(!decoded.sections()[1].known());
11664 }
11665
11666 #[test]
11667 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
11668 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11672 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
11673 let older = &directory[..directory.len() - block];
11674 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
11675 assert!(decoded.sections().is_empty());
11676 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
11677 assert_eq!(decoded.name(), "linked");
11678 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
11679 }
11680
11681 #[test]
11682 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
11683 let path = path("format_twenty_two");
11690 let mut writer =
11691 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11692 .expect("new file");
11693 let rows = Chunk::new(vec![
11694 Vector::from_values(
11695 LogicalType::Integer,
11696 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
11697 )
11698 .expect("integers"),
11699 ])
11700 .expect("one column");
11701 writer.append(&rows).expect("the only part");
11702 writer.finish().expect("commit");
11703
11704 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11705 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
11706 drop(file);
11707
11708 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
11709 assert_eq!(reader.table().rows(), 3);
11710 assert_eq!(reader.read(0, &[0]).expect("the part still reads").len(), 3);
11715
11716 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
11719 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
11720 drop(file);
11721 let error = Reader::open(&path).expect_err("format 21 is not readable");
11722 assert!(error.to_string().contains("format 21"), "{error}");
11723
11724 fs::remove_file(&path).expect("clean up");
11725 }
11726
11727 #[test]
11728 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
11729 let mut past = a_key_map_section();
11734 past.extent_page = 1 << 30;
11735 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
11736 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
11737 assert!(error.to_string().contains("outside the file"), "{error}");
11738
11739 let mut inside_the_header = a_key_map_section();
11740 inside_the_header.extent_page = 8;
11741 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
11742 assert!(
11743 decode_directory(&directory, 1 << 20).is_err(),
11744 "a section may not overlap a header"
11745 );
11746 }
11747
11748 #[test]
11749 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
11750 let not_built = Section {
11754 kind: *section::FORWARD_LINK,
11755 id: 9,
11756 generation: 3,
11757 extents: 0,
11758 extent_page: 0,
11759 extent_bytes: 0,
11760 hash: 0,
11761 flags: 0,
11762 header_bytes: 0,
11763 };
11764 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
11765 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
11766 assert_eq!(decoded.sections(), &[not_built]);
11767
11768 let mut incoherent = not_built;
11771 incoherent.extent_bytes = 28;
11772 incoherent.extent_page = HEADER;
11773 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
11774 assert!(decode_directory(&directory, 1 << 20).is_err());
11775 }
11776
11777 #[test]
11778 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
11779 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
11780 let mut torn = directory.clone();
11781 let count_at = torn.len() - size_of::<u16>();
11782 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
11783 assert!(decode_directory(&torn, 1 << 20).is_err());
11786 }
11787
11788 fn linked_file(label: &str, rows: i32) -> PathBuf {
11790 let path = path(label);
11791 let mut writer =
11792 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
11793 .expect("new file");
11794 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
11795 let chunk =
11796 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
11797 .expect("one column");
11798 writer.append(&chunk).expect("the only part");
11799 writer.finish().expect("commit");
11800 path
11801 }
11802
11803 fn a_key_map_payload() -> Vec<u8> {
11804 (0..512_u32).flat_map(u32::to_le_bytes).collect()
11807 }
11808
11809 #[test]
11810 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
11811 let path = linked_file("attach", 64);
11812 let payload = a_key_map_payload();
11813 let table = attach(
11814 &path,
11815 "items",
11816 &[section::Attachment {
11817 kind: *section::KEY_MAP,
11818 id: 0,
11819 flags: 2,
11820 header_bytes: 40,
11821 bytes: &payload,
11822 }],
11823 )
11824 .expect("attach a key map");
11825 assert_eq!(attached(&table).len(), 1);
11826
11827 let reader = Reader::open(&path).expect("reopen after the attach");
11828 let held = attached(reader.table());
11829 assert_eq!(held.len(), 1);
11830 assert_eq!(held[0].kind, *section::KEY_MAP);
11831 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
11832 assert_eq!(held[0].header_bytes, 40);
11833 assert_eq!(held[0].generation, 1);
11837 assert!(held[0].usable(reader.table().generation()));
11838 assert_eq!(reader.payload(held[0]).expect("read the payload"), payload);
11839 assert_eq!(reader.extents(held[0]).expect("extent table").len(), 1);
11840
11841 fs::remove_file(&path).expect("clean up");
11842 }
11843
11844 #[test]
11845 fn attaching_a_section_answers_every_row_exactly_as_before() {
11846 let path = linked_file("attach_changes_nothing", 300);
11851 let before = Reader::open(&path).expect("open before");
11852 let rows = before.table().rows();
11853 let first = before.read(0, &[0]).expect("read before");
11854 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
11855 let layout = before.layout().columns_total();
11856 drop(before);
11857
11858 let payload = a_key_map_payload();
11859 attach(
11860 &path,
11861 "items",
11862 &[section::Attachment {
11863 kind: *section::KEY_MAP,
11864 id: 0,
11865 flags: 0,
11866 header_bytes: 0,
11867 bytes: &payload,
11868 }],
11869 )
11870 .expect("attach");
11871
11872 let after = Reader::open(&path).expect("open after");
11873 assert_eq!(after.table().rows(), rows);
11874 let read = after.read(0, &[0]).expect("read after");
11875 for (at, value) in values.iter().enumerate() {
11876 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
11877 }
11878 assert_eq!(
11879 after.layout().columns_total(),
11880 layout,
11881 "an attach appends and does not rewrite a column page"
11882 );
11883
11884 fs::remove_file(&path).expect("clean up");
11885 }
11886
11887 #[test]
11888 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
11889 let path = linked_file("attach_twice", 32);
11893 let one = a_key_map_payload();
11894 let two = vec![7_u8; 1024];
11895 let entry = |bytes| section::Attachment {
11896 kind: *section::KEY_MAP,
11897 id: 4,
11898 flags: 1,
11899 header_bytes: 0,
11900 bytes,
11901 };
11902 attach(&path, "items", &[entry(&one)]).expect("first build");
11903 attach(&path, "items", &[entry(&two)]).expect("rebuild");
11904
11905 let reader = Reader::open(&path).expect("reopen");
11906 let held = attached(reader.table());
11907 assert_eq!(held.len(), 1, "one map per column and not one per build");
11908 assert_eq!(reader.payload(held[0]).expect("payload"), two);
11909
11910 fs::remove_file(&path).expect("clean up");
11911 }
11912
11913 #[test]
11914 fn an_attach_carries_through_a_kind_it_does_not_know() {
11915 let path = linked_file("attach_unknown", 16);
11919 let payload = vec![3_u8; 96];
11920 attach(
11921 &path,
11922 "items",
11923 &[section::Attachment {
11924 kind: *b"RUDBZZ9\0",
11925 id: 1,
11926 flags: 0,
11927 header_bytes: 0,
11928 bytes: &payload,
11929 }],
11930 )
11931 .expect("a kind this build does not know still writes");
11932 let key_map = a_key_map_payload();
11933 attach(
11934 &path,
11935 "items",
11936 &[section::Attachment {
11937 kind: *section::KEY_MAP,
11938 id: 0,
11939 flags: 0,
11940 header_bytes: 0,
11941 bytes: &key_map,
11942 }],
11943 )
11944 .expect("attach beside it");
11945
11946 let reader = Reader::open(&path).expect("reopen");
11947 let held = attached(reader.table());
11948 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
11949 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
11950 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
11951
11952 fs::remove_file(&path).expect("clean up");
11953 }
11954
11955 #[test]
11956 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
11957 let path = linked_file("attach_not_built", 8);
11958 attach(
11959 &path,
11960 "items",
11961 &[section::Attachment {
11962 kind: *section::FORWARD_LINK,
11963 id: 2,
11964 flags: 0,
11965 header_bytes: 0,
11966 bytes: &[],
11967 }],
11968 )
11969 .expect("record a link that did not fit the budget");
11970
11971 let reader = Reader::open(&path).expect("reopen");
11972 let held = attached(reader.table());
11973 assert_eq!(held.len(), 1);
11974 assert_eq!(held[0].extents, 0);
11975 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
11976 assert!(reader.extents(held[0]).expect("no extent table").is_empty());
11977 assert!(reader.payload(held[0]).expect("no payload").is_empty());
11978
11979 fs::remove_file(&path).expect("clean up");
11980 }
11981
11982 #[test]
11983 fn a_payload_past_one_extent_is_split_and_joined_back() {
11984 let path = linked_file("attach_two_extents", 8);
11988 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
11989 attach(
11990 &path,
11991 "items",
11992 &[section::Attachment {
11993 kind: *section::KEY_MAP,
11994 id: 0,
11995 flags: 0,
11996 header_bytes: 0,
11997 bytes: &payload,
11998 }],
11999 )
12000 .expect("attach a payload past the bound");
12001
12002 let reader = Reader::open(&path).expect("reopen");
12003 let held = attached(reader.table());
12004 let extents = reader.extents(held[0]).expect("extent table");
12005 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
12006 assert_eq!(extents[0].length, section::MAX_EXTENT);
12007 assert_eq!(extents[1].length, 1);
12008 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
12009 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
12011 assert_eq!(reader.payload(held[0]).expect("the whole payload").len(), payload.len());
12012
12013 fs::remove_file(&path).expect("clean up");
12014 }
12015
12016 #[test]
12017 fn a_torn_extent_is_refused_rather_than_decoded() {
12018 let path = linked_file("attach_torn", 8);
12019 let payload = a_key_map_payload();
12020 attach(
12021 &path,
12022 "items",
12023 &[section::Attachment {
12024 kind: *section::KEY_MAP,
12025 id: 0,
12026 flags: 0,
12027 header_bytes: 0,
12028 bytes: &payload,
12029 }],
12030 )
12031 .expect("attach");
12032
12033 let reader = Reader::open(&path).expect("reopen");
12034 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
12035 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
12036 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
12037 drop(file);
12038
12039 let reader = Reader::open(&path).expect("the table still opens");
12040 let error = reader
12041 .payload(&reader.table().sections()[0])
12042 .expect_err("a corrupt payload is not handed out");
12043 assert!(error.to_string().contains("checksum"), "{error}");
12044 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
12047
12048 fs::remove_file(&path).expect("clean up");
12049 }
12050
12051 #[test]
12052 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
12053 let path = linked_file("attach_old_format", 8);
12056 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
12057 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
12058 drop(file);
12059
12060 let payload = a_key_map_payload();
12061 let error = attach(
12062 &path,
12063 "items",
12064 &[section::Attachment {
12065 kind: *section::KEY_MAP,
12066 id: 0,
12067 flags: 0,
12068 header_bytes: 0,
12069 bytes: &payload,
12070 }],
12071 )
12072 .expect_err("format 22 cannot gain a section");
12073 assert!(error.to_string().contains("format 22"), "{error}");
12074 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
12075
12076 fs::remove_file(&path).expect("clean up");
12077 }
12078
12079 #[test]
12080 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
12081 let path = linked_file("attach_bad_header", 8);
12082 let error = attach(
12083 &path,
12084 "items",
12085 &[section::Attachment {
12086 kind: *section::KEY_MAP,
12087 id: 0,
12088 flags: 0,
12089 header_bytes: 40,
12090 bytes: &[1, 2, 3],
12091 }],
12092 )
12093 .expect_err("a writer's bug stops at the write");
12094 assert!(error.to_string().contains("header is longer"), "{error}");
12095
12096 fs::remove_file(&path).expect("clean up");
12097 }
12098
12099 #[test]
12100 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
12101 let path = linked_file("attach_wrong_name", 8);
12102 let error = attach(&path, "orders", &[]).expect_err("no such table");
12103 assert!(error.to_string().contains("orders"), "{error}");
12104 fs::remove_file(&path).expect("clean up");
12105 }
12106
12107 #[test]
12108 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
12109 let path = path("frequency_prefix_for_the_planner");
12116 let mut writer =
12117 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12118 .expect("new file");
12119 let mut values = vec![Value::Integer(1); 10_000];
12120 for _ in 0..10 {
12121 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
12122 }
12123 for part in values.chunks(8_000) {
12126 let rows = Chunk::new(vec![
12127 Vector::from_values(LogicalType::Integer, part).expect("integers"),
12128 ])
12129 .expect("one column");
12130 writer.append(&rows).expect("a part");
12131 }
12132 writer.finish().expect("commit");
12133 let reader = Reader::open(&path).expect("reopen from disk");
12134 let prefix =
12135 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
12136 assert_eq!(prefix.entries.len(), 512);
12139 assert_eq!(prefix.omitted_max, 10);
12140 let common = Common::new(reader);
12141 assert_eq!(common.rows(), 16_000);
12142 let column = common.column("id").expect("the file has that column");
12143 assert_eq!(
12144 common.rows_with(column, &Bound::Int(1)),
12145 Stat::exact(10_000, Provenance::FrequencySynopsis)
12146 );
12147 assert_eq!(
12149 common.rows_with(column, &Bound::Int(1_100)),
12150 Stat::exact(10, Provenance::FrequencySynopsis)
12151 );
12152 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
12155 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
12158 let remainder = common.remainder(column).expect("the list is a prefix");
12162 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
12163 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
12164 fs::remove_file(&path).expect("clean up");
12165 }
12166
12167 #[test]
12169 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
12170 let path = path("empty");
12171 Writer::empty(&path, &[]).expect("a file with nothing in it");
12172 let catalog = Catalog::open(&path).expect("the empty file opens");
12173 assert_eq!(catalog.len(), 0);
12174 assert!(catalog.is_empty());
12175 assert_eq!(catalog.names().count(), 0);
12176 let mut writer =
12179 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12180 .expect("a table goes into the empty file");
12181 writer.append(&sample_ids()).expect("rows");
12182 writer.finish().expect("commit");
12183 let catalog = Catalog::open(&path).expect("the file opens again");
12184 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12185 fs::remove_file(&path).expect("clean up");
12186 }
12187
12188 #[test]
12198 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
12199 let path = path("empty-name");
12200 let field = || vec![Field::required("id", LogicalType::Integer)];
12201 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
12202 let catalog = Catalog::open(&path).expect("the file opens");
12203 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
12204
12205 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
12206 writer.append(&sample_ids()).expect("rows");
12207 writer.finish().expect("commit");
12208 let catalog = Catalog::open(&path).expect("the file opens again");
12209 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12211 let held = catalog.rows().collect::<Vec<_>>();
12212 assert_eq!(held.len(), 1);
12213 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
12214
12215 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
12217 assert!(error.to_string().contains("same name"), "{error}");
12218 fs::remove_file(&path).expect("clean up");
12219 }
12220
12221 fn sample_view(name: &str) -> ViewEntry {
12223 ViewEntry {
12224 name: name.to_string(),
12225 sql: "SELECT id FROM items WHERE id > 0".to_string(),
12226 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
12227 aliases: vec!["n".to_string()],
12228 columns: vec![Field::new("n", LogicalType::Integer)],
12229 }
12230 }
12231
12232 #[test]
12233 fn a_view_written_into_the_catalog_comes_back_whole() {
12234 let path = path("views");
12235 let mut writer =
12236 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12237 .expect("new file");
12238 writer.append(&sample_ids()).expect("rows");
12239 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12240 let catalog = Catalog::open(&path).expect("reopen");
12241 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
12242 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12245 fs::remove_file(&path).expect("clean up");
12246 }
12247
12248 #[test]
12250 fn appending_a_table_carries_the_views_forward() {
12251 let path = path("viewscarry");
12252 let mut writer =
12253 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12254 .expect("new file");
12255 writer.append(&sample_ids()).expect("rows");
12256 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
12257 let mut writer =
12258 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
12259 .expect("a second table");
12260 writer.append(&sample_ids()).expect("rows");
12261 writer.finish().expect("commit");
12262 let catalog = Catalog::open(&path).expect("reopen");
12263 assert_eq!(catalog.views().count(), 1);
12264 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
12265 fs::remove_file(&path).expect("clean up");
12266 }
12267
12268 #[test]
12270 fn restating_the_views_leaves_every_table_where_it_was() {
12271 let path = path("restate");
12272 let mut writer =
12273 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12274 .expect("new file");
12275 writer.append(&sample_ids()).expect("rows");
12276 writer.finish().expect("commit");
12277 let before = fs::metadata(&path).expect("the file is there").len();
12278 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
12279 let catalog = Catalog::open(&path).expect("reopen");
12280 assert_eq!(catalog.views().count(), 2);
12281 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
12282 let after = fs::metadata(&path).expect("the file is there").len();
12285 assert!(after > before, "a generation was written");
12286 assert!(after - before < before, "the table was not written again");
12287 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
12290 assert_eq!(reader.table().rows, 3);
12291 Writer::restate(&path, &[]).expect("no views at all");
12294 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
12295 fs::remove_file(&path).expect("clean up");
12296 }
12297
12298 #[test]
12300 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
12301 let bytes = encode_catalog(
12302 &[Entry {
12303 name: "items".to_string(),
12304 fields: vec![Field::required("id", LogicalType::Integer)],
12305 rows: 1,
12306 directory: Page { offset: HEADER, length: 8, hash: 0 },
12307 nonzero: vec![None],
12308 aggregates: vec![None],
12309 distincts: vec![None],
12310 extremes: vec![None],
12311 frequencies: vec![None],
12312 }],
12313 &[sample_view("items")],
12314 )
12315 .expect("it encodes, because encoding does not look");
12316 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
12317 assert!(error.to_string().contains("same name"), "{error}");
12318 }
12319
12320 #[test]
12323 fn a_compressed_text_page_read_at_some_rows_is_those_rows_of_the_whole() {
12324 let rows: usize = 300;
12325 let text: Vec<String> =
12326 (0..rows).map(|row| format!("a street named after number {}", row * 7)).collect();
12327 let values: Vec<&[u8]> = text.iter().map(String::as_bytes).collect();
12328 let mut page = vec![6, 2];
12329 page.extend((0..rows.div_ceil(8)).map(|byte| {
12330 (0..8).filter(|bit| (byte * 8 + bit) % 5 != 3).fold(0_u8, |mask, bit| mask | 1 << bit)
12331 }));
12332 let compressed = string::encode_only(string::Kind::Fsst, &values)
12333 .expect("encoded")
12334 .expect("text this repetitive compresses");
12335 page.extend_from_slice(&compressed);
12336 let whole = decode(&LogicalType::Varchar, rows, &page, None).expect("the whole page");
12337 let positions = [0_u32, 3, 8, 13, 200, 299];
12338 let some =
12339 decode_at(&LogicalType::Varchar, rows, &page, None, &positions).expect("some rows");
12340 assert_eq!(some.len(), positions.len());
12341 for (at, &row) in positions.iter().enumerate() {
12342 assert_eq!(some.value_at(at), whole.value_at(row as usize), "row {row}");
12343 }
12344 assert_eq!(some.value_at(1), Value::Null, "row 3 is null");
12345 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[300]).is_err());
12346 assert!(decode_at(&LogicalType::Varchar, rows, &page, None, &[8, 3]).is_err());
12347 }
12348
12349 #[test]
12352 fn a_part_read_at_some_rows_is_the_part_read_whole_and_gathered() {
12353 let path = path("rows");
12354 let mut writer = Writer::create(
12355 &path,
12356 "items",
12357 vec![
12358 Field::required("id", LogicalType::Integer),
12359 Field::new("text", LogicalType::Varchar),
12360 ],
12361 )
12362 .expect("new file");
12363 let rows = 2_000;
12364 let chunk = Chunk::new(vec![
12365 Vector::from_values(
12366 LogicalType::Integer,
12367 &(0..rows).map(Value::Integer).collect::<Vec<_>>(),
12368 )
12369 .expect("integers"),
12370 Vector::from_values(
12371 LogicalType::Varchar,
12372 &(0..rows)
12373 .map(|row| {
12374 if row % 7 == 2 {
12375 Value::Null
12376 } else {
12377 Value::Varchar(format!("a comment about order {}", row * 13))
12378 }
12379 })
12380 .collect::<Vec<_>>(),
12381 )
12382 .expect("strings"),
12383 ])
12384 .expect("matching rows");
12385 writer.append(&chunk).expect("one part");
12386 writer.finish().expect("commit");
12387 let reader = Reader::open(&path).expect("reopen from disk");
12388 let positions = [1_u32, 2, 9, 1_000, 1_999];
12389 for whole in [true, false] {
12390 let some = reader.read_rows(0, &[0, 1], &positions, whole).expect("some rows");
12391 let all = reader.read(0, &[0, 1]).expect("the whole part");
12392 assert_eq!(some.len(), positions.len());
12393 for column in 0..2 {
12394 for (at, &row) in positions.iter().enumerate() {
12395 assert_eq!(some.value_at(at, column), all.value_at(row as usize, column));
12396 }
12397 }
12398 }
12399 assert!(reader.read_rows(0, &[1], &[2_000], true).is_err());
12400 }
12401
12402 #[test]
12403 fn committed_file_reopens_and_reads_only_requested_columns() {
12404 let path = path("reopen");
12405 let mut writer = Writer::create(
12406 &path,
12407 "items",
12408 vec![
12409 Field::required("id", LogicalType::Integer),
12410 Field::new("text", LogicalType::Varchar),
12411 ],
12412 )
12413 .expect("new file");
12414 writer.append(&sample()).expect("first part");
12415 writer.append(&sample()).expect("second part");
12416 writer.finish().expect("commit");
12417 let reader = Reader::open(&path).expect("reopen from disk");
12418 assert_eq!(reader.table().rows(), 6);
12419 assert_eq!(reader.table().stripes().len(), 1);
12422 assert_eq!(reader.parts(), 2);
12423 assert_eq!(reader.part_rows(0), 3);
12424 assert_eq!(reader.part_rows(1), 3);
12425 let text = reader.read(1, &[1]).expect("only text page");
12426 assert_eq!(text.width(), 1);
12427 assert_eq!(text.value_at(1, 0), Value::Null);
12428 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12429 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
12430 assert_eq!(sparse.width(), 1);
12431 assert_eq!(sparse.value_at(1, 0), Value::Null);
12432 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
12433 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
12434 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
12435 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
12436 let count = reader.read(0, &[]).expect("no page is needed for count");
12437 assert_eq!(count.len(), 3);
12438 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
12439 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
12440 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
12441 assert_eq!(
12442 integers,
12443 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
12444 );
12445 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
12446 assert_eq!(strings.len(), 3);
12447 assert!(strings.contains(&(Value::Null, 2)));
12448 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
12449 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
12450 fs::remove_file(path).expect("remove scratch file");
12451 }
12452
12453 #[test]
12461 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
12462 let path = path("interleaved-runs");
12463 let mut writer =
12464 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
12465 .expect("new file");
12466 for morsel in [2_u64, 0, 3, 1] {
12467 let parts = (0..4_u64)
12468 .map(|chunk| {
12469 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
12470 let values =
12471 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
12472 let column =
12473 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
12474 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
12475 })
12476 .collect::<Vec<_>>();
12477 writer.append_stripe(parts).expect("a stripe");
12478 }
12479 writer.finish().expect("commit");
12480
12481 let reader = Reader::open(&path).expect("valid directory");
12482 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
12483 assert_eq!(reader.table().rows(), 128);
12484 for part in 0..16_usize {
12485 let read = reader.read(part, &[0]).expect("a part back");
12486 for row in 0..8_usize {
12487 let want = i64::try_from(part * 8 + row).expect("small");
12488 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
12489 }
12490 }
12491 fs::remove_file(path).expect("remove scratch file");
12492 }
12493
12494 #[test]
12497 fn runs_that_overlap_each_other_are_refused_at_commit() {
12498 let path = path("overlapping-runs");
12499 let mut writer =
12500 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
12501 .expect("new file");
12502 let one = |order: (u64, u64)| {
12503 let column =
12504 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
12505 (order, Chunk::new(vec![column]).expect("one column"))
12506 };
12507 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
12510 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
12511 let error = writer.finish().expect_err("the runs overlap");
12512 assert!(error.message().contains("source order"), "{error}");
12513 fs::remove_file(path).expect("remove scratch file");
12514 }
12515
12516 #[test]
12519 fn a_run_longer_than_a_stripe_is_refused() {
12520 let path = path("overlong-run");
12521 let mut writer =
12522 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
12523 .expect("new file");
12524 let parts = (0..=STRIPE_PARTS)
12525 .map(|at| {
12526 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
12527 .expect("a column");
12528 let chunk = Chunk::new(vec![column]).expect("one column");
12529 ((0, u64::try_from(at).expect("small")), chunk)
12530 })
12531 .collect::<Vec<_>>();
12532 let error = writer.append_stripe(parts).expect_err("one part too many");
12533 assert!(error.message().contains("more parts than it holds"), "{error}");
12534 fs::remove_file(path).expect("remove scratch file");
12535 }
12536
12537 #[test]
12543 fn parts_past_the_stripe_bound_start_a_new_stripe() {
12544 let path = path("stripe-bound");
12545 let mut writer = Writer::create(
12546 &path,
12547 "items",
12548 vec![
12549 Field::required("id", LogicalType::Integer),
12550 Field::new("text", LogicalType::Varchar),
12551 ],
12552 )
12553 .expect("new file");
12554 let parts = STRIPE_PARTS * 2 + 3;
12555 for part in 0..parts {
12556 let id = part as i32;
12557 let chunk = Chunk::new(vec![
12558 Vector::from_values(
12559 LogicalType::Integer,
12560 &[Value::Integer(id), Value::Integer(-id)],
12561 )
12562 .expect("integers"),
12563 Vector::from_values(
12564 LogicalType::Varchar,
12565 &[Value::Varchar(format!("value {part}")), Value::Null],
12566 )
12567 .expect("strings"),
12568 ])
12569 .expect("matching rows");
12570 writer.append(&chunk).expect("one part");
12571 }
12572 writer.finish().expect("commit");
12573
12574 let reader = Reader::open(&path).expect("reopen from disk");
12575 assert_eq!(reader.parts(), parts);
12576 assert_eq!(reader.table().rows(), parts * 2);
12577 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
12578 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
12579 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
12580 assert_eq!(reader.table().stripes()[2].parts(), 3);
12581 for part in (0..parts).rev() {
12584 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
12585 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
12586 for chunk in [&dense, &sparse] {
12587 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
12588 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
12589 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
12590 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
12591 assert_eq!(chunk.value_at(1, 1), Value::Null);
12592 }
12593 }
12594 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
12597 assert!(reader.skips(0, &above), "the first stripe stops at 63");
12598 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
12599 fs::remove_file(path).expect("remove scratch file");
12600 }
12601
12602 fn scattered(n: i64) -> i64 {
12604 n.wrapping_mul(-7_046_029_254_386_353_131)
12605 }
12606
12607 #[test]
12613 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
12614 let path = path("sieve-skip");
12615 let mut writer =
12616 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12617 .expect("new file");
12618 let parts = STRIPE_PARTS + 3;
12619 let per_part = 128;
12623 for part in 0..parts {
12624 let held: Vec<Value> = (0..per_part)
12625 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
12626 .collect();
12627 let chunk =
12628 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12629 .expect("one column");
12630 writer.append(&chunk).expect("one part");
12631 }
12632 writer.finish().expect("commit");
12633
12634 let reader = Reader::open(&path).expect("reopen from disk");
12635 let probe = |value: i64| Probe {
12636 column: 0,
12637 op: Op::Equal,
12638 value: Bound::Int(i128::from(scattered(value))),
12639 };
12640 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
12641 let tests = [probe(wanted)];
12642 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
12643 let home = wanted as usize / per_part;
12644 assert!(kept.contains(&home), "the part holding {wanted} is read");
12645 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
12649 }
12650 let absent = [probe((parts * per_part) as i64 + 1)];
12651 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
12652 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
12653 let tests = [probe(0)];
12656 assert!(
12657 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
12658 "the bounds rule out no stripe at all"
12659 );
12660 fs::remove_file(path).expect("remove scratch file");
12661 }
12662
12663 #[test]
12669 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
12670 let path = path("part-range-skip");
12671 let mut writer =
12672 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12673 .expect("new file");
12674 let parts = STRIPE_PARTS + 3;
12675 let per_part = 128;
12676 for part in 0..parts {
12677 let held: Vec<Value> = (0..per_part)
12681 .map(|row| {
12682 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12683 })
12684 .collect();
12685 let chunk =
12686 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12687 .expect("one column");
12688 writer.append(&chunk).expect("one part");
12689 }
12690 writer.finish().expect("commit");
12691
12692 let reader = Reader::open(&path).expect("reopen from disk");
12693 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12694 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
12695 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
12696 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
12698 fs::remove_file(path).expect("remove scratch file");
12699 }
12700
12701 #[test]
12705 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
12706 let path = path("part-range-certain");
12707 let mut writer =
12708 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12709 .expect("new file");
12710 let parts = STRIPE_PARTS + 3;
12711 let per_part = 128;
12712 for part in 0..parts {
12713 let held: Vec<Value> = (0..per_part)
12714 .map(|row| {
12715 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
12716 })
12717 .collect();
12718 let chunk =
12719 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12720 .expect("one column");
12721 writer.append(&chunk).expect("one part");
12722 }
12723 writer.finish().expect("commit");
12724
12725 let reader = Reader::open(&path).expect("reopen from disk");
12726 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
12727 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
12728 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
12729 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
12732 fs::remove_file(path).expect("remove scratch file");
12733 }
12734
12735 #[test]
12738 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
12739 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
12740 let path = path("part-range-page");
12741 let mut writer =
12742 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
12743 .expect("new file");
12744 for part in 0..parts {
12745 let held: Vec<Value> = (0..128)
12746 .map(|row| {
12747 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
12748 })
12749 .collect();
12750 let chunk = Chunk::new(vec![
12751 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
12752 ])
12753 .expect("one column");
12754 writer.append(&chunk).expect("one part");
12755 }
12756 writer.finish().expect("commit");
12757 let reader = Reader::open(&path).expect("reopen from disk");
12758 let bytes = reader.layout().columns[0].part_ranges;
12759 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
12760 fs::remove_file(path).expect("remove scratch file");
12761 }
12762 }
12763
12764 #[test]
12767 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
12768 let long = vec![b'a'; PART_BOUND_BYTES * 2];
12769 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
12770 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
12771 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
12772 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
12773 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
12774 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
12775 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
12776 }
12777
12778 #[test]
12781 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
12782 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
12783 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
12784 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
12785 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
12786 }
12787
12788 #[test]
12800 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
12801 let parts = 4;
12802 let per_part = 1024;
12803 let rows = parts * per_part;
12804 let written = |name: &str, keys: &[i64]| {
12805 let path = path(name);
12806 let fields = vec![Field::required("key", LogicalType::BigInt)];
12807 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
12808 for part in 0..parts {
12809 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
12810 .iter()
12811 .map(|key| Value::BigInt(*key))
12812 .collect();
12813 let chunk = Chunk::new(vec![
12814 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
12815 ])
12816 .expect("one column");
12817 writer.append(&chunk).expect("one part");
12818 }
12819 writer.finish().expect("commit");
12820 path
12821 };
12822 let climbing = |step: &dyn Fn(usize) -> i64| {
12825 let mut key = 0;
12826 (0..rows)
12827 .map(|row| {
12828 key += step(row);
12829 key
12830 })
12831 .collect::<Vec<i64>>()
12832 };
12833 let ascending = climbing(&|row| (row % 3) as i64);
12834 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
12838 let near_path = written("stored-near", &ascending);
12839 let far_path = written("stored-far", &sparse);
12840
12841 let one = Reader::open(&near_path).expect("reopen from disk");
12842 let other = Reader::open(&far_path).expect("reopen from disk");
12843 let near = one.stored(0).expect("the column is stored");
12844 let far = other.stored(0).expect("the column is stored");
12845 assert_eq!(near.len(), parts, "one row per part");
12846 assert_eq!(far.len(), parts);
12847 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
12850 assert_eq!(total(&near), one.layout().columns[0].pages);
12851 assert_eq!(total(&far), other.layout().columns[0].pages);
12852 assert!(
12853 total(&near) * 2 < total(&far),
12854 "the sparse keys cost more, {} against {}",
12855 total(&far),
12856 total(&near)
12857 );
12858 for (at, part) in near.iter().enumerate() {
12860 assert_eq!(part.part, at);
12861 assert_eq!(part.row, at * per_part);
12862 assert_eq!(part.rows, per_part);
12863 let held = &ascending[at * per_part..(at + 1) * per_part];
12864 assert_eq!(part.low, Some(Value::BigInt(held[0])));
12865 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
12866 assert_eq!(part.nulls, Some(0));
12867 }
12868 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
12871 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
12872 assert_ne!(near[0].encoding, far[0].encoding);
12873 fs::remove_file(near_path).expect("remove scratch file");
12874 fs::remove_file(far_path).expect("remove scratch file");
12875 }
12876
12877 #[test]
12887 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
12888 let path = path("sieve-pays");
12889 let fields = vec![
12890 Field::required("spread", LogicalType::BigInt),
12891 Field::required("repeated", LogicalType::BigInt),
12892 ];
12893 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
12894 let parts = 3;
12895 let per_part = 1024;
12896 for part in 0..parts {
12897 let base = (part * per_part) as i64;
12898 let spread: Vec<Value> =
12899 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
12900 let repeated: Vec<Value> =
12901 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
12902 let chunk = Chunk::new(vec![
12903 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
12904 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
12905 ])
12906 .expect("two columns");
12907 writer.append(&chunk).expect("one part");
12908 }
12909 writer.finish().expect("commit");
12910
12911 let reader = Reader::open(&path).expect("reopen from disk");
12912 let layout = reader.layout();
12913 let spread = &layout.columns[0];
12914 let repeated = &layout.columns[1];
12915 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
12916 assert_eq!(
12917 repeated.sieves, 0,
12918 "a column whose filter costs more than its parts keeps none"
12919 );
12920 for column in &layout.columns {
12923 assert!(
12924 column.sieves < column.pages,
12925 "{} spends {} on sieves over {} of data",
12926 column.name,
12927 column.sieves,
12928 column.pages
12929 );
12930 }
12931 let absent = [Probe {
12933 column: 0,
12934 op: Op::Equal,
12935 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
12936 }];
12937 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
12938 fs::remove_file(path).expect("remove scratch file");
12939 }
12940
12941 #[test]
12947 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
12948 let path = path("sieve-damaged");
12949 let mut writer =
12950 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
12951 .expect("new file");
12952 let rows = 128;
12953 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
12954 let chunk =
12955 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
12956 .expect("one column");
12957 writer.append(&chunk).expect("one part");
12958 writer.finish().expect("commit");
12959
12960 let page = Reader::open(&path).expect("reopen").table.stripes[0]
12961 .sieves
12962 .get(0)
12963 .expect("a sieve page");
12964 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
12965 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
12966 file.write_all(&[0xff]).expect("damage one byte");
12967 drop(file);
12968
12969 let reader = Reader::open(&path).expect("reopen the damaged file");
12970 let absent =
12971 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
12972 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
12973 assert_eq!(
12974 reader.read(0, &[0]).expect("the rows are untouched").len(),
12975 usize::try_from(rows).expect("a small count")
12976 );
12977 fs::remove_file(path).expect("remove scratch file");
12978 }
12979
12980 #[test]
12991 fn workers_that_want_the_same_stripe_read_it_once() {
12992 let path = path("single-flight");
12993 let mut writer =
12994 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
12995 .expect("new file");
12996 for part in 0..STRIPE_PARTS {
12997 let id = part as i32;
12998 let chunk = Chunk::new(vec![
12999 Vector::from_values(
13000 LogicalType::Integer,
13001 &[Value::Integer(id), Value::Integer(-id)],
13002 )
13003 .expect("integers"),
13004 ])
13005 .expect("matching rows");
13006 writer.append(&chunk).expect("one part");
13007 }
13008 writer.finish().expect("commit");
13009
13010 let reader = Reader::open(&path).expect("reopen from disk");
13011 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
13012 let barrier = std::sync::Barrier::new(8);
13013 std::thread::scope(|scope| {
13014 for worker in 0..8 {
13015 let reader = &reader;
13016 let barrier = &barrier;
13017 scope.spawn(move || {
13018 barrier.wait();
13019 for part in (worker..STRIPE_PARTS).step_by(8) {
13020 let chunk = reader.read(part, &[0]).expect("a whole page read");
13021 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13022 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
13023 }
13024 });
13025 }
13026 });
13027 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
13028 fs::remove_file(path).expect("remove scratch file");
13029 }
13030
13031 #[test]
13044 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
13045 let opened = |label: &str, rows_per_part: i32| {
13046 let path = path(label);
13047 let mut writer =
13048 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13049 .expect("new file");
13050 for part in 0..STRIPE_PARTS * 3 {
13051 let values = (0..rows_per_part)
13055 .map(|row| {
13056 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
13057 })
13058 .collect::<Vec<_>>();
13059 let chunk = Chunk::new(vec![
13060 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
13061 ])
13062 .expect("matching rows");
13063 writer.append(&chunk).expect("one part");
13064 }
13065 writer.finish().expect("commit");
13066 let reader = Reader::open(&path).expect("reopen from disk");
13067 let size = fs::metadata(&path).expect("the file is there").len();
13068 let out = (reader.reads(), reader.table().stripes().len(), size);
13069 fs::remove_file(path).expect("remove scratch file");
13070 out
13071 };
13072
13073 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
13074 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
13075 assert_eq!(
13076 thin_stripes, fat_stripes,
13077 "the same stripe count is what makes this a fair ask"
13078 );
13079 assert!(
13080 fat_size > thin_size * 50,
13081 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
13082 );
13083
13084 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
13085 assert_eq!(thin.pages, 0, "opening read a page");
13086 assert_eq!(fat.pages, 0, "opening read a page");
13087 assert_eq!(thin.indexes, 0, "opening read an index");
13088 assert_eq!(fat.indexes, 0, "opening read an index");
13089 assert!(
13092 fat.opening.bytes < thin.opening.bytes * 2,
13093 "opening the thin file read {} bytes and the fat one read {}",
13094 thin.opening.bytes,
13095 fat.opening.bytes
13096 );
13097 }
13098
13099 #[test]
13107 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
13108 let path = path("open-twice");
13109 let mut writer =
13110 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13111 .expect("new file");
13112 for part in 0..STRIPE_PARTS * 3 {
13113 let chunk = Chunk::new(vec![
13114 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13115 .expect("integers"),
13116 ])
13117 .expect("matching rows");
13118 writer.append(&chunk).expect("one part");
13119 }
13120 writer.finish().expect("commit");
13121
13122 let first = Reader::open(&path).expect("open");
13123 for part in 0..first.parts() {
13126 first.read(part, &[0]).expect("a part");
13127 }
13128 assert!(first.reads().pages > 0, "the scan has to have read something");
13129 let second = Reader::open(&path).expect("open again");
13130
13131 assert_eq!(first.reads().opening, second.reads().opening);
13132 assert_eq!(
13133 second.reads().pages,
13134 0,
13135 "the second open read a page off the back of the first"
13136 );
13137 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
13138 fs::remove_file(path).expect("remove scratch file");
13139 }
13140
13141 #[test]
13149 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
13150 let path = path("index-cache");
13151 let mut writer =
13152 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13153 .expect("new file");
13154 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
13155 for part in 0..parts {
13156 let id = part as i32;
13157 let chunk = Chunk::new(vec![
13158 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
13159 ])
13160 .expect("matching rows");
13161 writer.append(&chunk).expect("one part");
13162 }
13163 writer.finish().expect("commit");
13164
13165 let reader = Reader::open(&path).expect("reopen from disk");
13166 let stripes = reader.table().stripes().len();
13167 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
13168 for _ in 0..2 {
13170 for part in 0..parts {
13171 let chunk = reader.read(part, &[0]).expect("a part");
13172 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13173 }
13174 }
13175 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
13176 assert!(
13177 reader.pages.load(Atomic::Relaxed) > stripes,
13178 "the pages are the ones that get read again, which is what makes the index count mean \
13179 something"
13180 );
13181 fs::remove_file(path).expect("remove scratch file");
13182 }
13183
13184 #[test]
13191 fn a_pool_keeps_pages_between_scans_and_gives_them_to_the_table_being_read() {
13192 let path = path("page-pool");
13193 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
13194 let fields = || vec![Field::required("id", LogicalType::Integer)];
13195 let mut writer = Writer::create(&path, "a", fields()).expect("new file");
13196 for table in ["a", "b"] {
13197 if table == "b" {
13198 writer = writer.next("b".to_string(), fields()).expect("a second table");
13199 }
13200 for part in 0..parts {
13201 let chunk = Chunk::new(vec![
13202 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13203 .expect("integers"),
13204 ])
13205 .expect("matching rows");
13206 writer.append(&chunk).expect("one part");
13207 }
13208 }
13209 writer.finish().expect("commit");
13210
13211 let pool = PagePool::new(usize::MAX);
13212 let catalog = Catalog::open_in(&path, &pool).expect("the file opens");
13213 let (a, b) = (catalog.table("a").expect("a"), catalog.table("b").expect("b"));
13214 let stripes = a.table().stripes().len();
13215 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the floor has to be smaller than a table");
13216 let scan = |reader: &Reader| {
13217 for part in 0..parts {
13218 let chunk = reader.read(part, &[0]).expect("a part");
13219 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13220 }
13221 };
13222 scan(&a);
13223 scan(&a);
13224 assert_eq!(a.pages.load(Atomic::Relaxed), stripes, "the second scan reads nothing");
13225 let one = pool.bytes();
13226 assert!(one > 0, "the pool counts what the reader holds");
13227
13228 pool.budget.store(one, Atomic::Relaxed);
13230 scan(&b);
13231 assert_eq!(b.pages.load(Atomic::Relaxed), stripes, "a page is never let go while in use");
13232 assert_eq!(a.cache.held[0].load(Atomic::Relaxed), CACHED_STRIPES_PER_COLUMN);
13233 let held = a.cache.columns[0].lock().expect("the column").pages.iter().flatten().count();
13234 assert_eq!(held, CACHED_STRIPES_PER_COLUMN, "the count and the slots agree");
13235
13236 drop((a, b, catalog));
13238 let c = Catalog::open_in(&path, &pool).expect("again").table("a").expect("a");
13239 scan(&c);
13240 assert!(pool.bytes() <= one, "only what the live reader holds is counted");
13241 fs::remove_file(path).expect("remove scratch file");
13242 }
13243
13244 #[test]
13253 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
13254 let workers = CACHED_STRIPES_PER_COLUMN + 4;
13255 let path = path("stripe-per-worker");
13256 let mut writer =
13257 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13258 .expect("new file");
13259 for part in 0..STRIPE_PARTS * workers {
13260 let chunk = Chunk::new(vec![
13261 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
13262 .expect("integers"),
13263 ])
13264 .expect("matching rows");
13265 writer.append(&chunk).expect("one part");
13266 }
13267 writer.finish().expect("commit");
13268
13269 let read = |told: bool| {
13270 let reader = Reader::open(&path).expect("reopen from disk");
13271 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
13272 if told {
13273 reader.keep_stripes(workers);
13274 }
13275 let barrier = std::sync::Barrier::new(workers);
13276 std::thread::scope(|scope| {
13277 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
13278 let reader = &reader;
13279 let barrier = &barrier;
13280 scope.spawn(move || {
13281 for part in run {
13282 barrier.wait();
13283 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
13284 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
13285 }
13286 assert!(worker < workers);
13287 });
13288 }
13289 });
13290 reader.pages.load(Atomic::Relaxed)
13291 };
13292
13293 assert_eq!(read(true), workers, "one page read per stripe and no more");
13294 assert!(read(false) > workers, "a cache that small is read again on every part");
13295 fs::remove_file(path).expect("remove scratch file");
13296 }
13297
13298 #[test]
13303 fn a_damaged_index_page_is_an_error() {
13304 let path = path("damaged-index");
13305 let mut writer =
13306 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
13307 .expect("new file");
13308 writer.append(&sample_ids()).expect("first part");
13309 writer.append(&sample_ids()).expect("second part");
13310 writer.finish().expect("commit");
13311
13312 let reader = Reader::open(&path).expect("valid directory");
13313 let index = reader.table.stripes[0].index;
13314 let mut byte = [0; 1];
13315 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
13316 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
13317 file.seek(SeekFrom::Start(index.offset)).expect("index start");
13318 file.write_all(&[!byte[0]]).expect("damage the first part length");
13319 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
13320 assert!(error.message().contains("index page section checksum differs"), "{error}");
13321 fs::remove_file(path).expect("remove scratch file");
13322 }
13323
13324 #[test]
13331 fn every_integer_width_round_trips_through_a_page() {
13332 let path = path("integer-widths");
13333 let columns = [
13334 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
13335 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
13336 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
13337 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
13338 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
13339 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
13340 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
13341 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
13342 ];
13343 let fields = columns
13344 .iter()
13345 .enumerate()
13346 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13347 .collect::<Vec<_>>();
13348 let vectors = columns
13349 .iter()
13350 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13351 .collect::<Vec<_>>();
13352 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
13353 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13354 writer.finish().expect("commit");
13355
13356 let reader = Reader::open(&path).expect("reopen from disk");
13357 let wanted = (0..columns.len()).collect::<Vec<_>>();
13358 let read = reader.read(0, &wanted).expect("every column");
13359 assert_eq!(read.len(), 2);
13360 for (at, (ty, values)) in columns.iter().enumerate() {
13362 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13363 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13364 }
13365 fs::remove_file(path).expect("remove scratch file");
13366 }
13367
13368 #[test]
13379 fn every_other_type_the_format_knows_round_trips_through_a_page() {
13380 let path = path("other-types");
13381 let columns = [
13382 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
13383 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
13384 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
13385 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
13386 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
13387 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
13388 (
13389 LogicalType::TimestampTz,
13390 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
13391 ),
13392 (
13393 LogicalType::Interval,
13394 vec![
13395 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
13396 Value::Interval { months: 13, days: -1, micros: 1 },
13397 ],
13398 ),
13399 (
13400 LogicalType::Blob,
13401 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
13402 ),
13403 ];
13404 let fields = columns
13405 .iter()
13406 .enumerate()
13407 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
13408 .collect::<Vec<_>>();
13409 let vectors = columns
13410 .iter()
13411 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
13412 .collect::<Vec<_>>();
13413 let mut writer = Writer::create(&path, "others", fields).expect("new file");
13414 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13415 writer.finish().expect("commit");
13416
13417 let reader = Reader::open(&path).expect("reopen from disk");
13418 let wanted = (0..columns.len()).collect::<Vec<_>>();
13419 let read = reader.read(0, &wanted).expect("every column");
13420 assert_eq!(read.len(), 2);
13421 for (at, (ty, values)) in columns.iter().enumerate() {
13422 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
13423 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
13424 }
13425 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
13428 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
13429
13430 fs::remove_file(path).expect("remove scratch file");
13431 }
13432
13433 #[test]
13439 fn a_nan_survives_being_written_down() {
13440 let path = path("nan");
13441 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
13442 .expect("a NaN vector");
13443 let mut writer =
13444 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
13445 .expect("new file");
13446 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
13447 writer.finish().expect("commit");
13448 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
13449 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
13450 assert!(back.is_nan(), "a NaN came back as {back}");
13451 fs::remove_file(path).expect("remove scratch file");
13452 }
13453
13454 #[test]
13461 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
13462 let path = path("uuid-and-bit");
13463 let uuids = vec![0_i128, i128::MIN, -1];
13464 let mut bits = StringColumn::new();
13465 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
13466 bits.push_bytes(value);
13467 }
13468 let expected = bits.clone();
13469 let fields =
13470 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
13471 let vectors = vec![
13472 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
13473 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
13474 ];
13475 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
13476 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
13477 writer.finish().expect("commit");
13478
13479 let reader = Reader::open(&path).expect("reopen from disk");
13480 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
13481 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
13482 panic!("a uuid column is the 128 bit lane")
13483 };
13484 assert_eq!(back.as_slice(), uuids.as_slice());
13485 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
13486 panic!("a bit column is bytes")
13487 };
13488 for row in 0..expected.len() {
13489 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
13490 }
13491 fs::remove_file(path).expect("remove scratch file");
13492 }
13493
13494 #[test]
13497 fn a_run_counted_at_once_leaves_the_candidates_a_row_at_a_time_would() {
13498 let mut rows: Vec<Option<u64>> = Vec::new();
13499 let mut state = 0x2545_f491_4f6c_dd1d_u64;
13500 for index in 0..400_000_u64 {
13501 state ^= state << 13;
13502 state ^= state >> 7;
13503 state ^= state << 17;
13504 let times = 1 + (state % 7) as usize;
13505 let bits = match state % 11 {
13506 0 => None,
13507 1..=3 => Some(state % 16),
13508 _ => Some(index.wrapping_mul(0x9e37_79b9_7f4a_7c15)),
13509 };
13510 rows.extend(std::iter::repeat_n(bits, times));
13511 }
13512 let mut by_row = Candidates::default();
13513 for &bits in &rows {
13514 by_row.add(bits, 1);
13515 }
13516 let mut by_run = Candidates::default();
13517 let mut run = Run::default();
13518 let mut runs = 0_usize;
13519 for &bits in &rows {
13520 if let Some((bits, times)) = run.push(bits) {
13521 by_run.add(bits, times);
13522 runs += 1;
13523 }
13524 }
13525 if let Some((bits, times)) = run.take() {
13526 by_run.add(bits, times);
13527 }
13528 assert!(runs < rows.len() / 2, "the rows came in runs");
13529 assert!(by_row.decrements > 0, "the table filled and turned values away");
13530 assert_eq!(by_run.counts, by_row.counts);
13531 assert_eq!(by_run.nulls, by_row.nulls);
13532 assert_eq!(by_run.decrements, by_row.decrements);
13533 }
13534
13535 #[test]
13536 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
13537 let path = path("frequency-ordinals");
13538 let mut writer =
13539 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
13540 .expect("new file");
13541 let mut values = Vec::new();
13542 for leader in 0..10_i64 {
13543 values.extend(std::iter::repeat_n(leader, 100));
13544 }
13545 values.extend(1_000_i64..41_000);
13546 for part in values.chunks(1_024) {
13547 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
13548 .expect("big integers");
13549 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
13550 }
13551 writer.finish().expect("commit");
13552
13553 let reader = Reader::open(&path).expect("reopen from disk");
13554 let occurrences =
13555 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
13556 assert!(occurrences.omitted_max < 100);
13557 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
13558 assert_eq!(occurrences.anchor_indices.len(), occurrences.ordinals.len());
13559 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
13560 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
13561 assert_eq!(
13562 &occurrences.anchor_indices[..1_000]
13563 .iter()
13564 .map(|&entry| occurrences.anchors[entry as usize].clone())
13565 .collect::<Vec<_>>(),
13566 &(0_i64..10)
13567 .flat_map(|leader| std::iter::repeat_n(Value::BigInt(leader), 100))
13568 .collect::<Vec<_>>()
13569 );
13570 fs::remove_file(path).expect("remove scratch file");
13571 }
13572
13573 #[test]
13574 fn numeric_frequencies_count_nulls_and_values_past_the_top_of_bigint() {
13575 let path = path("frequency-bits");
13580 let mut writer = Writer::create(
13581 &path,
13582 "items",
13583 vec![Field::new("u", LogicalType::UBigInt), Field::new("s", LogicalType::BigInt)],
13584 )
13585 .expect("new file");
13586 let mut rows = Vec::new();
13587 let mut leaders = Vec::new();
13588 for leader in 0..10_u64 {
13589 let count = 300 - leader * 10;
13590 let (unsigned, signed) = if leader == 0 {
13591 (Value::Null, Value::Null)
13592 } else {
13593 (Value::UBigInt(u64::MAX - leader), Value::BigInt(-(leader as i64)))
13594 };
13595 rows.extend(std::iter::repeat_n((unsigned.clone(), signed.clone()), count as usize));
13596 leaders.push(((unsigned, count), (signed, count)));
13597 }
13598 rows.extend((1_000..41_000_u64).map(|id| (Value::UBigInt(id), Value::BigInt(id as i64))));
13599 for part in rows.chunks(1_024) {
13600 let unsigned = part.iter().map(|(value, _)| value.clone()).collect::<Vec<_>>();
13601 let signed = part.iter().map(|(_, value)| value.clone()).collect::<Vec<_>>();
13602 let chunk = Chunk::new(vec![
13603 Vector::from_values(LogicalType::UBigInt, &unsigned).expect("unsigned"),
13604 Vector::from_values(LogicalType::BigInt, &signed).expect("signed"),
13605 ])
13606 .expect("matching columns");
13607 writer.append(&chunk).expect("rows");
13608 }
13609 writer.finish().expect("commit");
13610
13611 let reader = Reader::open(&path).expect("reopen from disk");
13612 for column in 0..2 {
13613 let prefix =
13614 reader.frequency_prefix(column).expect("valid metadata").expect("a synopsis");
13615 let wanted = leaders
13616 .iter()
13617 .map(|(unsigned, signed)| if column == 0 { unsigned } else { signed })
13618 .cloned()
13619 .collect::<Vec<_>>();
13620 assert_eq!(&prefix.entries[..10], &wanted[..], "column {column}");
13621 assert!(prefix.omitted_max < 210, "column {column}");
13622 assert_eq!(
13623 reader.distinct_values(column).expect("valid metadata"),
13624 Some(9 + 40_000),
13625 "column {column}"
13626 );
13627 }
13628 fs::remove_file(path).expect("remove scratch file");
13629 }
13630
13631 #[test]
13632 fn narrow_nonzero_count_matches_the_full_reader_across_stripes() {
13633 let path = path("quick-nonzero");
13634 let mut writer = Writer::create(
13635 &path,
13636 "items",
13637 vec![Field::new("label", LogicalType::Varchar), Field::new("id", LogicalType::Integer)],
13638 )
13639 .expect("create");
13640 for ids in [
13641 &[Value::Integer(0), Value::Null, Value::Integer(3)][..],
13642 &[Value::Integer(0), Value::Integer(7), Value::Null][..],
13643 ] {
13644 let labels = vec![Value::Varchar("same".into()); ids.len()];
13645 writer
13646 .append(
13647 &Chunk::new(vec![
13648 Vector::from_values(LogicalType::Varchar, &labels).expect("labels"),
13649 Vector::from_values(LogicalType::Integer, ids).expect("ids"),
13650 ])
13651 .expect("chunk"),
13652 )
13653 .expect("append");
13654 }
13655 writer.finish().expect("finish");
13656 let catalog = Catalog::open(&path).expect("catalog");
13657 assert_eq!(catalog.entries[0].nonzero, vec![None, Some(2)]);
13658 assert_eq!(catalog.entries[0].aggregates, vec![None, Some((10, 4))]);
13659 assert_eq!(catalog.entries[0].distincts, vec![Some(1), Some(3)]);
13660 let frequencies =
13661 catalog.exact_numeric_frequencies("items", 1).expect("frequencies").expect("complete");
13662 assert_eq!(frequencies.len(), 4);
13663 for pair in [(Some(0), 2), (Some(3), 1), (Some(7), 1), (None, 2)] {
13664 assert!(frequencies.contains(&pair), "missing {pair:?}");
13665 }
13666 assert_eq!(catalog.distinct_count("items", 1).expect("distinct count"), Some(3));
13667 assert_eq!(
13668 catalog.integer_extremes("items", 1).expect("extremes"),
13669 Some(IntegerExtremes::Values { low: 0, high: 7 })
13670 );
13671 assert_eq!(
13672 catalog.aggregate_sums("items", &[1]).expect("catalog sums"),
13673 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13674 );
13675 assert_eq!(catalog.nonzero_count("items", 1).expect("quick count"), Some(2));
13676 assert_eq!(
13677 reader_nonzero_counts(&catalog.table("items").expect("reader")).expect("counts"),
13678 vec![None, Some(2)]
13679 );
13680 Writer::certify_counts(&path).expect("recertify");
13681 assert_eq!(
13682 Catalog::open(&path).expect("reopen").nonzero_count("items", 1).expect("count"),
13683 Some(2)
13684 );
13685 assert_eq!(
13686 Catalog::open(&path).expect("reopen").aggregate_sums("items", &[1]).expect("sums"),
13687 Some(CertifiedSums { columns: vec![(10, 4)], rows: 6 })
13688 );
13689 assert_eq!(
13690 Catalog::open(&path).expect("reopen").distinct_count("items", 1).expect("distinct"),
13691 Some(3)
13692 );
13693 assert_eq!(
13694 Catalog::open(&path).expect("reopen").integer_extremes("items", 1).expect("ends"),
13695 Some(IntegerExtremes::Values { low: 0, high: 7 })
13696 );
13697 assert_eq!(
13698 Catalog::open(&path)
13699 .expect("reopen")
13700 .exact_numeric_frequencies("items", 1)
13701 .expect("frequencies"),
13702 Some(frequencies)
13703 );
13704 assert_eq!(catalog.table("items").expect("reader").null_count(1).expect("nulls"), 2);
13705 fs::remove_file(path).expect("remove scratch file");
13706 }
13707
13708 #[test]
13709 fn numeric_string_pair_leaders_are_certified_in_the_directory() {
13710 let path = path("pair-frequencies");
13711 let mut pairs = Vec::new();
13712 pairs.extend(std::iter::repeat_n((1_i64, "alpha".to_string()), 100));
13713 pairs.extend(std::iter::repeat_n((1_i64, "beta".to_string()), 50));
13714 pairs.extend(std::iter::repeat_n((2_i64, "gamma".to_string()), 40));
13715 pairs.extend((1_000_i64..1_600).map(|id| (id, format!("tail {id}"))));
13716 let mut writer = Writer::create(
13717 &path,
13718 "items",
13719 vec![
13720 Field::required("id", LogicalType::BigInt),
13721 Field::required("phrase", LogicalType::Varchar),
13722 ],
13723 )
13724 .expect("new file");
13725 for part in pairs.chunks(1_024) {
13726 let ids = part.iter().map(|(id, _)| Value::BigInt(*id)).collect::<Vec<_>>();
13727 let phrases =
13728 part.iter().map(|(_, phrase)| Value::Varchar(phrase.clone())).collect::<Vec<_>>();
13729 writer
13730 .append(
13731 &Chunk::new(vec![
13732 Vector::from_values(LogicalType::BigInt, &ids).expect("ids"),
13733 Vector::from_values(LogicalType::Varchar, &phrases).expect("phrases"),
13734 ])
13735 .expect("matching columns"),
13736 )
13737 .expect("rows");
13738 }
13739 writer.finish().expect("commit");
13740
13741 let reader = Reader::open(&path).expect("reopen from disk");
13742 assert!(
13743 reader.table.pair_frequencies.is_empty(),
13744 "no query-specific pair result is stored"
13745 );
13746 fs::remove_file(path).expect("remove scratch file");
13747 }
13748
13749 #[test]
13755 fn a_file_from_another_format_says_which_format_it_is() {
13756 let older = path("older-format");
13757 let mut writer =
13758 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
13759 .expect("new file");
13760 let chunk = Chunk::new(vec![
13761 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13762 .expect("integers"),
13763 ])
13764 .expect("chunk");
13765 writer.append(&chunk).expect("page written");
13766 writer.finish().expect("commit");
13767
13768 let unreadable =
13772 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
13773 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13774 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
13775 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
13776 drop(file);
13777 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
13778 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
13779 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
13780
13781 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
13782 file.seek(SeekFrom::Start(0)).expect("the magic is first");
13783 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
13784 drop(file);
13785 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
13786 assert!(complaint.contains("magic"), "{complaint}");
13787 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
13788 fs::remove_file(older).expect("remove scratch file");
13789 }
13790
13791 #[test]
13792 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
13793 let unfinished = path("unfinished");
13794 let mut writer =
13795 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
13796 .expect("new file");
13797 let chunk = Chunk::new(vec![
13798 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
13799 .expect("integers"),
13800 ])
13801 .expect("chunk");
13802 writer.append(&chunk).expect("page written");
13803 drop(writer);
13804 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
13805 fs::remove_file(unfinished).expect("remove scratch file");
13806
13807 let damaged = path("damaged");
13808 let mut writer =
13809 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
13810 .expect("new file");
13811 writer.append(&chunk).expect("page written");
13812 writer.finish().expect("commit");
13813 let reader = Reader::open(&damaged).expect("valid directory");
13814 let mut file =
13815 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
13816 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
13817 file.write_all(&[255]).expect("damage one byte");
13818 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
13819 fs::remove_file(damaged).expect("remove scratch file");
13820 }
13821
13822 #[test]
13823 fn damaged_lazy_dictionary_payload_is_an_error() {
13824 let path = path("damaged-dictionary");
13825 let mut writer = Writer::create(
13826 &path,
13827 "items",
13828 vec![
13829 Field::required("id", LogicalType::Integer),
13830 Field::new("text", LogicalType::Varchar),
13831 ],
13832 )
13833 .expect("new file");
13834 writer.append(&sample()).expect("stripe written");
13835 writer.finish().expect("commit");
13836
13837 let reader = Reader::open(&path).expect("valid directory");
13838 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
13839 let mut header = [0; DICTIONARY_HEADER];
13842 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13843 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13846 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13847 assert_ne!(width & DICTIONARY_SCATTERED, 0, "the blocks say where they are");
13848 let bits = (width & !DICTIONARY_FLAGS) as usize;
13849 let mut start = [0; 8];
13850 let at = dictionary.offset + (DICTIONARY_HEADER + offset_bytes(count, bits)) as u64;
13851 read_at(&reader.file, at, &mut start).expect("the first block's start");
13852 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13853 file.seek(SeekFrom::Start(u64::from_le_bytes(start))).expect("inside dictionary payload");
13854 file.write_all(&[255]).expect("damage dictionary payload");
13855
13856 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
13857 let error =
13858 chunk.validate_external().expect_err("payload corruption must reach the caller");
13859 assert!(error.message().contains("payload checksum differs"), "{error}");
13860 fs::remove_file(path).expect("remove scratch file");
13861 }
13862
13863 #[test]
13873 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
13874 let path = path("dictionary-decide");
13875 let rows = 20_000;
13876 let unique =
13878 |row: usize| format!("{row:09} a value that appears exactly once in the table");
13879 let repeated = |row: usize| unique(row / 40);
13881 let mut writer = Writer::create(
13882 &path,
13883 "items",
13884 vec![
13885 Field::required("unique", LogicalType::Varchar),
13886 Field::required("repeated", LogicalType::Varchar),
13887 ],
13888 )
13889 .expect("new file");
13890 for part in (0..rows).step_by(1_000) {
13891 let span = part..(part + 1_000).min(rows);
13892 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
13893 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
13894 writer
13895 .append(
13896 &Chunk::new(vec![
13897 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
13898 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
13899 ])
13900 .expect("two columns"),
13901 )
13902 .expect("a part");
13903 }
13904 writer.finish().expect("commit");
13905
13906 let reader = Reader::open(&path).expect("reopen from disk");
13907 assert!(
13908 reader.table.dictionaries[0].is_none(),
13909 "a column with no repeats has nothing to say twice"
13910 );
13911 assert!(
13912 reader.table.dictionaries[1].is_some(),
13913 "a column whose values come round again keeps its dictionary"
13914 );
13915 let mut first = 0;
13916 for part in 0..reader.parts() {
13917 let chunk = reader.read(part, &[0, 1]).expect("a part");
13918 for row in 0..chunk.len() {
13919 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
13920 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
13921 }
13922 first += chunk.len();
13923 }
13924 assert_eq!(first, rows, "every row was read back");
13925 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
13926 let size = fs::metadata(&path).expect("the file is there").len() as usize;
13927 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
13928 fs::remove_file(path).expect("remove scratch file");
13929 }
13930
13931 #[test]
13944 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
13945 let path = path("dictionary-blocks");
13946 let value = |row: usize| {
13947 let row = row.saturating_sub(8_000);
13948 format!("{row:07} a value long enough to be worth a payload block")
13949 };
13950 let parts = 40;
13951 let per_part = 1000;
13952 let mut writer =
13953 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
13954 .expect("new file");
13955 for part in 0..parts {
13956 let values = (0..per_part)
13957 .map(|row| Value::Varchar(value(part * per_part + row)))
13958 .collect::<Vec<_>>();
13959 let chunk = Chunk::new(vec![
13960 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
13961 ])
13962 .expect("matching rows");
13963 writer.append(&chunk).expect("a part");
13964 }
13965 writer.finish().expect("commit");
13966
13967 let reader = Reader::open(&path).expect("reopen from disk");
13968 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
13969 assert!(
13970 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
13971 "the dictionary has to be several blocks for this to be testing anything"
13972 );
13973 for part in [0, parts - 1] {
13974 let chunk = reader.read(part, &[0]).expect("a part");
13975 chunk.validate_external().expect("every payload block checks out");
13976 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
13977 }
13978
13979 let mut header = [0; DICTIONARY_HEADER];
13981 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
13982 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
13983 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
13984 let width = u32::from_le_bytes(header[12..16].try_into().expect("four bytes"));
13985 let bits = (width & !DICTIONARY_FLAGS) as usize;
13986 let mut place = [0; 16];
13987 let at = DICTIONARY_HEADER + offset_bytes(count, bits) + (blocks - 1) * 16;
13988 read_at(&reader.file, dictionary.offset + at as u64, &mut place).expect("its place");
13989 let start = u64::from_le_bytes(place[..8].try_into().expect("eight bytes"));
13990 let length = u64::from_le_bytes(place[8..].try_into().expect("eight bytes"));
13991 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
13992 file.seek(SeekFrom::Start(start + length - 4)).expect("the last bytes of the last block");
13993 file.write_all(&[255]).expect("damage the last payload block");
13994 let reader = Reader::open(&path).expect("the directory and the index are untouched");
13995 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
13996 let error = chunk.validate_external().expect_err("the damage must reach the caller");
13997 assert!(error.message().contains("payload checksum differs"), "{error}");
13998 fs::remove_file(path).expect("remove scratch file");
13999 }
14000
14001 #[test]
14015 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
14016 let path = path("dictionary-offsets");
14017 let value = |row: usize| {
14018 let row = row % 5_000;
14019 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
14020 };
14021 let rows = 6_000;
14022 let mut writer =
14023 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14024 .expect("new file");
14025 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
14026 for part in values.chunks(1_000) {
14027 let chunk =
14028 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
14029 .expect("matching rows");
14030 writer.append(&chunk).expect("a part");
14031 }
14032 writer.finish().expect("commit");
14033
14034 let reader = Reader::open(&path).expect("reopen from disk");
14035 assert!(
14036 rows > TEXT_PAYLOAD_VALUES * 4,
14037 "the dictionary has to be several blocks for this to be testing anything"
14038 );
14039 for part in 0..rows / 1_000 {
14040 let chunk = reader.read(part, &[0]).expect("a part");
14041 for row in 0..1_000 {
14042 let row = part * 1_000 + row;
14043 assert_eq!(
14044 chunk.value_at(row % 1_000, 0),
14045 Value::Varchar(value(row)),
14046 "value {row}"
14047 );
14048 }
14049 }
14050 for _ in 0..2 {
14053 for part in 0..rows / 1_000 {
14054 let chunk = reader.read(part, &[0]).expect("a part");
14055 let mut lens = vec![0_i64; 1_000];
14056 let column = chunk.column(0).expect("one column");
14057 assert!(column.try_bytes_lens(&mut lens).expect("lengths"), "a stored column");
14058 for (row, &len) in lens.iter().enumerate() {
14059 let row = part * 1_000 + row;
14060 assert_eq!(len as usize, value(row).len(), "the length of value {row}");
14061 }
14062 }
14063 }
14064 fs::remove_file(path).expect("remove scratch file");
14065 }
14066
14067 #[test]
14069 fn lengths_restart_at_each_block_and_refuse_ends_that_go_backwards() {
14070 let mut ends: Vec<u32> = (1..=TEXT_PAYLOAD_VALUES as u32).map(|at| at * 2).collect();
14071 ends.extend([3, 3, 10]);
14072 let lens = lengths_of(&ends).expect("ordered ends");
14073 assert!(lens[..TEXT_PAYLOAD_VALUES].iter().all(|&len| len == 2));
14074 assert_eq!(&lens[TEXT_PAYLOAD_VALUES..], &[3, 0, 7]);
14075 ends.push(9);
14076 assert_eq!(lengths_of(&ends), None);
14077 }
14078
14079 #[test]
14091 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
14092 let path = path("dictionary-once");
14093 let parts = 8;
14094 let per_part = 500;
14095 let value =
14096 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
14097 let mut writer =
14098 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14099 .expect("new file");
14100 for part in 0..parts {
14101 let values = (0..per_part)
14102 .map(|row| Value::Varchar(value(part * per_part + row)))
14103 .collect::<Vec<_>>();
14104 let chunk = Chunk::new(vec![
14105 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14106 ])
14107 .expect("matching rows");
14108 writer.append(&chunk).expect("a part");
14109 }
14110 writer.finish().expect("commit");
14111
14112 let reader = Reader::open(&path).expect("reopen from disk");
14113 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
14114 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
14115
14116 let workers = 16;
14117 let gate = std::sync::Barrier::new(workers);
14118 std::thread::scope(|scope| {
14119 for worker in 0..workers {
14120 let reader = reader.clone();
14121 let gate = &gate;
14122 scope.spawn(move || {
14123 gate.wait();
14124 let chunk = reader.read(worker % parts, &[0]).expect("a part");
14125 assert_eq!(
14126 chunk.value_at(0, 0),
14127 Value::Varchar(value((worker % parts) * per_part))
14128 );
14129 });
14130 }
14131 });
14132
14133 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
14134 fs::remove_file(path).expect("remove scratch file");
14135 }
14136
14137 #[test]
14142 fn a_damaged_sorted_order_is_an_error() {
14143 let path = path("damaged-order");
14144 let mut writer = Writer::create(
14145 &path,
14146 "items",
14147 vec![
14148 Field::required("id", LogicalType::Integer),
14149 Field::new("text", LogicalType::Varchar),
14150 ],
14151 )
14152 .expect("new file");
14153 writer.append(&sample()).expect("stripe written");
14154 writer.finish().expect("commit");
14155
14156 let reader = Reader::open(&path).expect("valid directory");
14157 let page = reader.table.dictionaries[1].expect("string dictionary page");
14158 let mut header = [0; DICTIONARY_HEADER];
14159 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
14160 let index_len = dictionary_index_len(&header);
14161 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14162 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
14163 file.write_all(&[255]).expect("damage the order");
14164
14165 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
14166 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
14167 assert!(error.message().contains("rank checksum differs"), "{error}");
14168 fs::remove_file(path).expect("remove scratch file");
14169 }
14170
14171 #[test]
14175 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
14176 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
14179 let path = path("dictionary-order");
14180 let mut writer =
14181 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14182 .expect("new file");
14183 writer
14184 .append(
14185 &Chunk::new(vec![
14186 Vector::from_values(
14187 LogicalType::Varchar,
14188 &spellings.map(|text| Value::Varchar(text.into())),
14189 )
14190 .expect("strings"),
14191 ])
14192 .expect("one column"),
14193 )
14194 .expect("stripe written");
14195 writer.finish().expect("commit");
14196
14197 let reader = Reader::open(&path).expect("valid directory");
14198 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14199 let count = dictionary.ranks().expect("a v10 file stores one");
14200 assert_eq!(count, spellings.len(), "every distinct value has a rank");
14201 let order = (0..count)
14202 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
14203 .collect::<Vec<_>>();
14204 let mut seen = order.clone();
14205 seen.sort_unstable();
14206 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
14207
14208 let ranked = order
14209 .iter()
14210 .map(|&code| {
14211 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14212 })
14213 .collect::<Vec<_>>();
14214 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
14215 expected.sort();
14216 assert_eq!(ranked, expected, "rank order is value order");
14217
14218 for (rank, value) in expected.iter().enumerate() {
14221 assert_eq!(
14222 dictionary.compare_rank(rank, value).expect("compare"),
14223 Ordering::Equal,
14224 "rank {rank} is its own value"
14225 );
14226 if rank > 0 {
14227 assert_eq!(
14228 dictionary.compare_rank(rank - 1, value).expect("compare"),
14229 Ordering::Less,
14230 "rank {rank} follows the one before it"
14231 );
14232 }
14233 }
14234 fs::remove_file(path).expect("remove scratch file");
14235 }
14236
14237 #[test]
14244 fn text_columns_closed_at_once_each_keep_their_own_dictionary() {
14245 let sizes = [300_usize, 5_000, 40, 2_000, 1_200];
14246 let path = path("dictionaries-at-once");
14247 let fields = (0..sizes.len())
14248 .map(|column| Field::new(format!("text{column}"), LogicalType::Varchar))
14249 .collect::<Vec<_>>();
14250 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14251 let rows = 10_000_usize;
14252 for start in (0..rows).step_by(1_024) {
14253 let columns = sizes
14254 .iter()
14255 .enumerate()
14256 .map(|(column, &size)| {
14257 let values = (start..(start + 1_024).min(rows))
14258 .map(|row| Value::Varchar(format!("c{column}-{:05}", (row * 7919) % size)))
14259 .collect::<Vec<_>>();
14260 Vector::from_values(LogicalType::Varchar, &values).expect("strings")
14261 })
14262 .collect::<Vec<_>>();
14263 writer.append(&Chunk::new(columns).expect("five columns")).expect("stripe written");
14264 }
14265 writer.finish().expect("commit");
14266
14267 let reader = Reader::open(&path).expect("valid directory");
14268 for (column, &size) in sizes.iter().enumerate() {
14269 let dictionary =
14270 reader.dictionary(column).expect("read").expect("a string column has one");
14271 let count = dictionary.ranks().expect("a v10 file stores one");
14272 assert_eq!(count, size, "column {column} has its own distinct count");
14273 let ranked = (0..count)
14274 .map(|rank| {
14275 let code = dictionary.code_at_rank(rank).expect("a code");
14276 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14277 })
14278 .collect::<Vec<_>>();
14279 let expected = (0..size)
14280 .map(|value| format!("c{column}-{value:05}").into_bytes())
14281 .collect::<Vec<_>>();
14282 assert_eq!(ranked, expected, "column {column} ranks its own values in order");
14283 }
14284 fs::remove_file(path).expect("remove scratch file");
14285 }
14286
14287 #[test]
14295 fn a_large_dictionary_ranks_in_value_order() {
14296 let path = path("dictionary-large-rank");
14297 let value = |row: u64| {
14298 let mixed = row.wrapping_mul(0x9e37_79b9_7f4a_7c15) >> 40;
14299 match row % 3 {
14300 0 => format!("https://example.com/a/long/shared/path/{mixed:08}"),
14301 1 => format!("{mixed}"),
14302 _ => format!("x{}", row % 1000).repeat(1 + (row % 4) as usize) + &row.to_string(),
14303 }
14304 };
14305 let distinct = 70_000;
14306 let parts = 4 * distinct / 1000;
14307 let per_part = 1000;
14308 let mut writer =
14309 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14310 .expect("new file");
14311 for part in 0..parts {
14312 let values = (0..per_part)
14313 .map(|row| Value::Varchar(value((part * per_part + row) / 4)))
14314 .collect::<Vec<_>>();
14315 let chunk = Chunk::new(vec![
14316 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
14317 ])
14318 .expect("matching rows");
14319 writer.append(&chunk).expect("a part");
14320 }
14321 writer.finish().expect("commit");
14322
14323 let reader = Reader::open(&path).expect("reopen from disk");
14324 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14325 let count = dictionary.ranks().expect("a ranked dictionary");
14326 assert_eq!(count, distinct as usize, "every distinct value has a rank");
14327 assert!(count >= PARALLEL_SORT_MIN, "too few values to be sorted on more than one thread");
14328 let ranked = (0..count)
14329 .map(|rank| {
14330 let code = dictionary.code_at_rank(rank).expect("a code");
14331 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
14332 })
14333 .collect::<Vec<_>>();
14334 let mut expected = (0..distinct).map(|row| value(row).into_bytes()).collect::<Vec<_>>();
14335 expected.sort();
14336 assert_eq!(ranked, expected, "rank order is value order");
14337 fs::remove_file(path).expect("remove scratch file");
14338 }
14339
14340 #[test]
14353 fn a_directory_read_a_window_at_a_time_is_the_directory_read_whole() {
14354 let path = path("windowed-directory");
14355 let fields = vec![
14356 Field::required("id", LogicalType::BigInt),
14357 Field::required("word", LogicalType::Varchar),
14358 Field::new("score", LogicalType::Double),
14359 ];
14360 let mut writer = Writer::create(&path, "items", fields).expect("new file");
14361 for part in 0..70_i64 {
14362 let ids = (0..100).map(|row| Value::BigInt(part * 100 + row % 7)).collect::<Vec<_>>();
14363 let words = (0..100)
14364 .map(|row| Value::Varchar(format!("word {}", row % 13)))
14365 .collect::<Vec<_>>();
14366 let scores = (0..100)
14367 .map(|row| if row % 4 == 0 { Value::Null } else { Value::Double(row as f64) })
14368 .collect::<Vec<_>>();
14369 let chunk = Chunk::new(vec![
14370 Vector::from_values(LogicalType::BigInt, &ids).expect("integers"),
14371 Vector::from_values(LogicalType::Varchar, &words).expect("strings"),
14372 Vector::from_values(LogicalType::Double, &scores).expect("doubles"),
14373 ])
14374 .expect("three columns");
14375 writer.append(&chunk).expect("a part");
14376 }
14377 writer.finish().expect("commit");
14378
14379 let catalog = Catalog::open(&path).expect("reopen");
14380 let entry = catalog.entries.first().expect("one table").directory;
14381 let (offset, length) = (entry.offset, entry.length as usize);
14382 let mut bytes = vec![0; length];
14383 read_at(&catalog.file, offset, &mut bytes).expect("the directory");
14384 assert_eq!(file_checksum(&catalog.file, offset, length).expect("checksum"), entry.hash);
14385 let whole = decode_directory(&bytes, catalog.size).expect("whole");
14386 assert!(whole.stripes.len() > 1, "the table should span stripes");
14387 for size in [1, 7, 33, 4_096] {
14388 let mut cursor = Cursor::over(&catalog.file, offset, length);
14389 cursor.window.as_mut().expect("a window").size = size;
14390 let windowed = read_directory(cursor, catalog.size, Some(offset)).expect("windowed");
14391 assert_eq!(format!("{:?}", windowed.stripes), format!("{:?}", whole.stripes));
14392 assert_eq!(format!("{:?}", windowed.fields), format!("{:?}", whole.fields));
14393 let mut stored = 0;
14394 for (column, (left, held)) in
14395 windowed.frequencies.iter().zip(&whole.frequencies).enumerate()
14396 {
14397 match (left, held) {
14398 (None, None) => {}
14399 (
14400 Some(super::Frequencies::Stored { span, values }),
14401 Some(super::Frequencies::Held(summary)),
14402 ) => {
14403 let mut one = vec![0; span.length as usize];
14404 read_at(&catalog.file, span.offset, &mut one).expect("a synopsis");
14405 let read = decode_summary(
14406 &mut Cursor::new(&one),
14407 &whole.fields[column],
14408 whole.rows,
14409 *values,
14410 )
14411 .expect("a valid synopsis")
14412 .expect("one is there");
14413 assert_eq!(format!("{read:?}"), format!("{summary:?}"));
14414 stored += 1;
14415 }
14416 other => panic!("column {column} came back as {other:?}"),
14417 }
14418 }
14419 assert!(stored >= 2, "only {stored} synopses were left in the file");
14420 }
14421 let reader = catalog.table("items").expect("the table");
14422 assert!(reader.frequency_summaries[1].get().is_none());
14423 assert!(reader.top_frequencies(1, 1).expect("a readable synopsis").is_some());
14424 let first = reader.frequency_summaries[1].get().expect("decoded synopsis");
14425 let clone = reader.clone();
14426 assert!(clone.top_frequencies(1, 1).expect("cached synopsis").is_some());
14427 assert!(Arc::ptr_eq(first, clone.frequency_summaries[1].get().expect("same synopsis")));
14428 fs::remove_file(path).expect("remove scratch file");
14429 }
14430
14431 #[test]
14432 fn a_checksum_carried_across_reads_is_the_checksum_of_the_whole() {
14433 let path = path("file-checksum");
14434 let bytes = (0..200_000_u32)
14435 .map(|at| (at.wrapping_mul(2_654_435_761) >> 13) as u8)
14436 .collect::<Vec<_>>();
14437 fs::write(&path, &bytes).expect("scratch file");
14438 let file = File::open(&path).expect("open");
14439 for (offset, length) in [
14440 (0, 0),
14441 (3, 1),
14442 (5, 31),
14443 (0, 32),
14444 (9, 33),
14445 (1, 65_536),
14446 (7, 65_567),
14447 (0, 200_000),
14448 (11, 131_101),
14449 ] {
14450 let whole = checksum(&bytes[offset..offset + length]);
14451 assert_eq!(
14452 file_checksum(&file, offset as u64, length).expect("read"),
14453 whole,
14454 "{offset} {length}"
14455 );
14456 }
14457 fs::remove_file(path).expect("remove scratch file");
14458 }
14459
14460 #[test]
14461 fn a_string_synopsis_is_read_without_keeping_the_dictionary_blocks() {
14462 let path = path("synopsis-keeps-no-block");
14463 let spelled = |index: usize| Value::Varchar(format!("phrase {index:05}"));
14464 let mut values = (0..3_000).map(spelled).collect::<Vec<_>>();
14465 for _ in 0..3 {
14466 values.extend((0..3_000).step_by(5).map(spelled));
14467 }
14468 let mut writer =
14469 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
14470 .expect("new file");
14471 for part in values.chunks(1_024) {
14472 writer
14473 .append(
14474 &Chunk::new(vec![
14475 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14476 ])
14477 .expect("one column"),
14478 )
14479 .expect("a part");
14480 }
14481 writer.finish().expect("commit");
14482
14483 let reader = Reader::open(&path).expect("reopen from disk");
14484 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14485 let resting = dictionary.footprint();
14486 let prefix = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14487 assert_eq!(prefix.entries.len(), 512);
14488 for (value, count) in &prefix.entries {
14489 let Value::Varchar(text) = value else { panic!("a string column gave {value:?}") };
14490 let index = text["phrase ".len()..].parse::<usize>().expect("a spelled number");
14491 assert_eq!((index % 5, *count), (0, 4), "{text} came back with {count}");
14492 }
14493 assert_eq!(dictionary.footprint(), resting, "reading the synopsis kept a decoded block");
14494 let again = reader.frequency_prefix(0).expect("a readable synopsis").expect("one");
14495 assert_eq!(again.entries, prefix.entries);
14496 fs::remove_file(path).expect("remove scratch file");
14497 }
14498
14499 #[test]
14509 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
14510 let path = path("dictionary-sweep");
14511 let spellings = (0..2_500)
14514 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14515 .collect::<Vec<_>>();
14516 let mut writer =
14517 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14518 .expect("new file");
14519 for part in spellings.chunks(1_024) {
14522 writer
14523 .append(
14524 &Chunk::new(vec![
14525 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14526 ])
14527 .expect("one column"),
14528 )
14529 .expect("stripe written");
14530 }
14531 writer.finish().expect("commit");
14532
14533 let reader = Reader::open(&path).expect("valid directory");
14534 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14535 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14536 for first in [0, TEXT_PAYLOAD_VALUES, TEXT_PAYLOAD_VALUES * 2] {
14537 assert!(dictionary.text_block_might_contain(first, b"value").expect("signature"));
14538 assert!(!dictionary.text_block_might_contain(first, b"google").expect("signature"));
14539 }
14540
14541 let resting = dictionary.footprint();
14542 let sweep = || {
14543 let mut swept: Vec<Vec<u8>> = Vec::new();
14544 let mut at = 0;
14545 let mut calls = 0;
14546 while at < dictionary.len() {
14547 let stopped = dictionary
14548 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14549 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14550 swept.push(text.to_vec());
14551 Ok(())
14552 })
14553 .expect("a sweep reads");
14554 assert!(stopped > at, "a sweep moves");
14555 at = stopped;
14556 calls += 1;
14557 }
14558 assert_eq!(calls, 3, "a sweep hands over one block at a time");
14559 swept
14560 };
14561 let swept = sweep();
14562 assert_eq!(dictionary.footprint(), resting, "a first sweep keeps nothing it decoded");
14563 assert_eq!(sweep(), swept, "a second sweep reads what the first did");
14564 let after = dictionary.footprint();
14565 assert!(after > resting, "a second sweep under the budget keeps what it decoded");
14566
14567 let read = (0..dictionary.len())
14568 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14569 .collect::<Vec<_>>();
14570 assert_eq!(swept, read, "a sweep answers what a point read answers");
14571 let grown = dictionary.footprint() - after;
14575 assert!(
14576 grown == 0 || grown == dictionary.len() * size_of::<u32>(),
14577 "a point read of a kept block decodes nothing, and {grown} bytes grew"
14578 );
14579 fs::remove_file(path).expect("remove scratch file");
14580 }
14581
14582 #[test]
14583 fn a_narrow_signature_of_an_older_file_answers_by_its_own_width() {
14584 let path = path("narrow-substring-signature");
14585 let blocks = [&b"https://google.com/"[..], b"https://example.org/", b"mail.google.com"];
14586 let mut grams = Vec::new();
14587 for text in blocks {
14588 let mut bits = vec![0_u8; NARROW_GRAM_BYTES];
14589 for gram in text.windows(4) {
14590 for bit in gram_bits(gram, NARROW_GRAM_BYTES) {
14591 bits[bit / 8] |= 1 << (bit % 8);
14592 }
14593 }
14594 grams.extend(bits);
14595 }
14596 fs::write(&path, &grams).expect("scratch file");
14597 let file = File::open(&path).expect("open scratch file");
14598 let signatures = NativeGrams {
14599 start: 0,
14600 length: grams.len(),
14601 width: NARROW_GRAM_BYTES,
14602 hash: checksum(&grams),
14603 verdicts: Mutex::new(Vec::new()),
14604 };
14605 let verdict = signatures.verdicts(&file, b"google").expect("signatures read");
14606 assert_eq!(&verdict[..], &[true, false, true], "one verdict a block, at the narrow width");
14607 assert!(signatures.footprint() > 0, "a verdict is remembered");
14608 let again = signatures.verdicts(&file, b"google").expect("remembered");
14609 assert!(Arc::ptr_eq(&verdict, &again), "a second question about a literal reads nothing");
14610
14611 let damaged = NativeGrams {
14612 hash: signatures.hash ^ 1,
14613 verdicts: Mutex::new(Vec::new()),
14614 ..signatures
14615 };
14616 let error = damaged.verdicts(&file, b"google").expect_err("a damaged region is refused");
14617 assert!(error.to_string().contains("substring signatures checksum differs"), "{error}");
14618 fs::remove_file(path).expect("remove scratch file");
14619 }
14620
14621 #[test]
14622 fn a_damaged_substring_signature_is_checked_only_when_used() {
14623 let path = path("damaged-substring-signature");
14624 let mut writer =
14625 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14626 .expect("new file");
14627 let rows = [Value::Varchar("google".into()), Value::Varchar("example".into())];
14628 writer
14629 .append(
14630 &Chunk::new(vec![
14631 Vector::from_values(LogicalType::Varchar, &rows).expect("strings"),
14632 ])
14633 .expect("one column"),
14634 )
14635 .expect("stripe written");
14636 writer.finish().expect("commit");
14637
14638 let reader = Reader::open(&path).expect("valid directory");
14639 let page = reader.table.dictionaries[0].expect("string dictionary page");
14640 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
14641 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1))
14642 .expect("last signature byte");
14643 file.write_all(&[255]).expect("damage signature");
14644 let reader = Reader::open(&path).expect("the directory is still valid");
14645 let dictionary = reader.dictionary(0).expect("index is still valid").expect("dictionary");
14646 let error = dictionary
14647 .text_block_might_contain(0, b"goog")
14648 .expect_err("a used signature checks its own checksum");
14649 assert!(error.message().contains("substring signatures checksum differs"), "{error}");
14650 fs::remove_file(path).expect("remove scratch file");
14651 }
14652
14653 #[test]
14664 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
14665 let path = path("dictionary-sweep-short-run");
14666 let spellings = (0..2_800)
14667 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14668 .collect::<Vec<_>>();
14669 let mut writer =
14670 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14671 .expect("new file");
14672 for part in spellings.chunks(1_024) {
14673 writer
14674 .append(
14675 &Chunk::new(vec![
14676 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14677 ])
14678 .expect("one column"),
14679 )
14680 .expect("stripe written");
14681 }
14682 writer.finish().expect("commit");
14683
14684 let reader = Reader::open(&path).expect("valid directory");
14685 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14686 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14687 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
14688 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
14689 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
14690
14691 let mut swept: Vec<Vec<u8>> = Vec::new();
14692 let mut at = 0;
14693 while at < dictionary.len() {
14694 let stopped = dictionary
14695 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
14696 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
14697 swept.push(text.to_vec());
14698 Ok(())
14699 })
14700 .expect("a sweep reads");
14701 assert!(stopped > at, "a sweep moves");
14702 at = stopped;
14703 }
14704 let read = (0..dictionary.len())
14705 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
14706 .collect::<Vec<_>>();
14707 assert_eq!(swept, read, "a sweep answers what a point read answers");
14708 fs::remove_file(path).expect("remove scratch file");
14709 }
14710
14711 #[test]
14720 fn the_unpacked_ends_answer_what_the_packed_ends_answer() {
14721 let path = path("dictionary-unpacked-ends");
14722 let spellings = (0..2_800)
14723 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
14724 .collect::<Vec<_>>();
14725 let mut writer =
14726 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14727 .expect("new file");
14728 for part in spellings.chunks(1_024) {
14729 writer
14730 .append(
14731 &Chunk::new(vec![
14732 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14733 ])
14734 .expect("one column"),
14735 )
14736 .expect("stripe written");
14737 }
14738 writer.finish().expect("commit");
14739
14740 let reader = Reader::open(&path).expect("valid directory");
14741 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
14742 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
14743 let wanted = (0..spellings.len())
14744 .map(|index| format!("value {index:08} {}", "x".repeat(index % 40)).into_bytes())
14745 .collect::<Vec<_>>();
14746
14747 let pass = |what: &str| {
14748 for (index, value) in wanted.iter().enumerate() {
14749 let len = dictionary.try_bytes_len_at(index).expect("read").expect("a value");
14750 assert_eq!(len, value.len(), "{what} has the wrong length at {index}");
14751 let bytes = dictionary.try_bytes_at(index).expect("read").expect("a value");
14752 assert_eq!(bytes, value.as_slice(), "{what} has the wrong value at {index}");
14753 }
14754 };
14755 pass("the first pass");
14756 pass("the second pass");
14757
14758 let lens = wanted.iter().map(|value| value.len() as i64).collect::<Vec<_>>();
14762 let mut whole = vec![0i64; wanted.len()];
14763 assert!(dictionary.try_bytes_lens(&mut whole).expect("read"), "the text answers whole");
14764 assert_eq!(whole, lens, "a vector of lengths answers what a length at a time answers");
14765 let codes = (0..4_000_u32).map(|row| (7 * (4_000 - row)) % 2_800).collect::<Vec<_>>();
14766 let coded = Vector::dictionary_over(codes.clone(), dictionary).expect("codes in range");
14767 let mut through = vec![0i64; codes.len()];
14768 assert!(coded.try_bytes_lens(&mut through).expect("read"), "the codes answer whole");
14769 for (row, &code) in codes.iter().enumerate() {
14770 assert_eq!(through[row], lens[code as usize], "row {row} reads code {code}");
14771 let one = coded.try_bytes_len_at(row).expect("read").expect("a value");
14772 assert_eq!(through[row], one as i64, "row {row} a row at a time");
14773 }
14774
14775 let fresh = Reader::open(&path).expect("valid directory");
14778 let untouched = fresh.dictionary(0).expect("read").expect("a string column has one");
14779 let few = vec![2_799_u32, 0, 1_024, 1_023, 511, 512];
14780 let coded = Vector::dictionary_over(few.clone(), untouched).expect("in range");
14781 let mut short = vec![0i64; few.len()];
14782 assert!(coded.try_bytes_lens(&mut short).expect("read"), "the codes answer whole");
14783 let expected = few.iter().map(|&code| lens[code as usize]).collect::<Vec<_>>();
14784 assert_eq!(short, expected, "the packed ends answer what the table answers");
14785 fs::remove_file(path).expect("remove scratch file");
14786 }
14787
14788 #[test]
14798 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
14799 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
14800 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
14801 fit::<i8>(&[128]).expect_err("one past the top does not fit");
14802 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
14803 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
14804 fit::<u8>(&[256]).expect_err("one past the top does not fit");
14805 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
14806 assert_eq!(
14807 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
14808 vec![-32_768_i16, 0, 32_767]
14809 );
14810 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
14811 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
14812 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
14813 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
14814 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
14815 assert_eq!(
14816 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
14817 vec![i32::MIN, 0, i32::MAX]
14818 );
14819 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
14820 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
14821 assert_eq!(
14822 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
14823 vec![0_u32, 4_294_967_295]
14824 );
14825 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
14826 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
14827
14828 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
14831 }
14832
14833 #[test]
14840 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
14841 for value in -70_000_i64..70_000 {
14842 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
14843 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
14844 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
14845 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
14846 }
14847 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
14848 for edge in wide {
14849 for step in -2_i64..=2 {
14850 let value = edge.saturating_add(step);
14851 assert_eq!(
14852 fit::<i32>(&[value]).is_ok(),
14853 i32::try_from(value).is_ok(),
14854 "{value} as i32"
14855 );
14856 assert_eq!(
14857 fit::<u32>(&[value]).is_ok(),
14858 u32::try_from(value).is_ok(),
14859 "{value} as u32"
14860 );
14861 }
14862 }
14863 }
14864
14865 #[test]
14880 fn a_dictionary_reads_the_same_whether_its_blocks_say_where_they_are() {
14881 let spellings = (0..3_000)
14882 .map(|index| format!("value {index:08} {}", "y".repeat(index % 40)))
14883 .collect::<Vec<_>>();
14884 let mut read = Vec::new();
14885 for layout in ["outside", "inside", "behind"] {
14886 let mut dictionary = GlobalDictionary::new();
14887 for text in &spellings {
14888 dictionary.code(text).expect("a code for every spelling");
14889 }
14890 dictionary.finish_blocks().expect("the last block encodes");
14891 let order = dictionary.ranked(None).expect("a sorted order");
14892 let laid = |from: u64| {
14894 let mut at = from;
14895 dictionary
14896 .blocks
14897 .iter()
14898 .map(|block| {
14899 let place =
14900 Placed { start: at, length: block.len() as u64, hash: checksum(block) };
14901 at += block.len() as u64;
14902 place
14903 })
14904 .collect::<Vec<_>>()
14905 };
14906 let payload = dictionary.blocks.concat();
14907 let scattered = layout != "behind";
14908 let (bytes, encoded, offset, length) = if layout == "outside" {
14909 let mut bytes = vec![0; HEADER as usize];
14910 bytes.extend_from_slice(&payload);
14911 let encoded = encode_global_dictionary(&dictionary, &order, &laid(HEADER), true)
14912 .expect("an encoding");
14913 let offset = bytes.len() as u64;
14914 bytes.extend_from_slice(&encoded.index);
14915 bytes.extend_from_slice(&encoded.ranks);
14916 bytes.extend_from_slice(&encoded.grams);
14917 let length = encoded.index.len() + encoded.ranks.len() + encoded.grams.len();
14918 (bytes, encoded, offset, length)
14919 } else {
14920 let first = encode_global_dictionary(&dictionary, &order, &laid(0), scattered)
14923 .expect("an encoding");
14924 let body = (first.index.len() + first.ranks.len() + first.grams.len()) as u64;
14925 let encoded = encode_global_dictionary(&dictionary, &order, &laid(body), scattered)
14926 .expect("an encoding");
14927 let mut bytes = encoded.index.clone();
14928 bytes.extend_from_slice(&encoded.ranks);
14929 bytes.extend_from_slice(&encoded.grams);
14930 bytes.extend_from_slice(&payload);
14931 let length = bytes.len();
14932 (bytes, encoded, 0, length)
14933 };
14934 let path = path(&format!("blocks-{layout}"));
14935 fs::write(&path, &bytes).expect("the dictionary is written on its own");
14936 let file = Arc::new(File::open(&path).expect("it opens again"));
14937 let page = Page {
14938 offset,
14939 length: u32::try_from(length).expect("a test dictionary is small"),
14940 hash: checksum(&encoded.index),
14941 };
14942 let opened =
14943 open_global_dictionary(file, page, &LogicalType::Varchar, TEXT_KEEP_BUDGET)
14944 .expect("a dictionary laid out either way opens");
14945 let mut swept: Vec<Vec<u8>> = Vec::new();
14946 let mut at = 0;
14947 while at < opened.len() {
14948 at = opened
14949 .sweep_text(at, opened.len(), &mut |_index: usize, text: &[u8]| {
14950 swept.push(text.to_vec());
14951 Ok(())
14952 })
14953 .expect("a sweep reads");
14954 }
14955 fs::remove_file(&path).expect("clean up");
14956 read.push(swept);
14957 }
14958 let wanted =
14959 spellings.iter().map(|text| text.as_bytes().to_vec()).collect::<Vec<Vec<u8>>>();
14960 assert_eq!(read[0], wanted, "the blocks outside the page hold the values");
14961 assert_eq!(read[1], read[0], "the blocks inside the page hold the same values");
14962 assert_eq!(read[2], read[0], "the blocks behind one another hold the same values");
14963 }
14964
14965 #[test]
14973 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
14974 let path = path("dictionary-budget");
14975 let spellings = (0..2_500)
14976 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
14977 .collect::<Vec<_>>();
14978 let mut writer =
14979 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
14980 .expect("new file");
14981 for part in spellings.chunks(1_024) {
14982 writer
14983 .append(
14984 &Chunk::new(vec![
14985 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
14986 ])
14987 .expect("one column"),
14988 )
14989 .expect("stripe written");
14990 }
14991 writer.finish().expect("commit");
14992
14993 let reader = Reader::open(&path).expect("valid directory");
14994 let page = reader.table.dictionaries[0].expect("a string column has one");
14995 let file = Arc::clone(&reader.file);
14996 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
14997 .expect("a dictionary opens whatever it may keep");
14998
14999 let resting = starved.footprint();
15000 let mut swept: Vec<Vec<u8>> = Vec::new();
15001 let mut at = 0;
15002 while at < starved.len() {
15003 at = starved
15004 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
15005 swept.push(text.to_vec());
15006 Ok(())
15007 })
15008 .expect("a sweep reads");
15009 }
15010 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
15011 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
15012
15013 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
15014 let read = (0..generous.len())
15015 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
15016 .collect::<Vec<_>>();
15017 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
15018 fs::remove_file(path).expect("remove scratch file");
15019 }
15020
15021 #[test]
15022 fn damaged_membership_cannot_skip_a_string_page() {
15023 let path = path("damaged-membership");
15024 let mut writer = Writer::create(
15025 &path,
15026 "items",
15027 vec![
15028 Field::required("id", LogicalType::Integer),
15029 Field::new("text", LogicalType::Varchar),
15030 ],
15031 )
15032 .expect("new file");
15033 writer.append(&sample()).expect("stripe written");
15034 writer.finish().expect("commit");
15035
15036 let reader = Reader::open(&path).expect("valid directory");
15037 let membership = reader.table.stripes[0].memberships.get(1).expect("string membership");
15038 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
15039 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
15040 file.write_all(&[255]).expect("damage membership");
15041 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
15042 assert!(error.message().contains("membership page checksum differs"), "{error}");
15043 fs::remove_file(path).expect("remove scratch file");
15044 }
15045
15046 #[test]
15047 fn membership_delta_stream_is_sorted_exact_and_bounded() {
15048 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
15049 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
15050 let encoded = encode_membership(&unique);
15051 assert_eq!(
15052 decode_membership(&encoded).expect("valid membership"),
15053 [4, 9, 72, 900, u32::MAX]
15054 );
15055 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
15058 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
15059 assert_eq!(
15060 decode_membership(&encode_membership(&merged)).expect("valid membership"),
15061 unique
15062 );
15063 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
15064 assert!(
15065 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
15066 "a value past u32 is invalid"
15067 );
15068 }
15069
15070 #[test]
15071 fn a_global_dictionary_may_be_larger_than_one_column_page() {
15072 let dictionary = Page {
15073 offset: HEADER,
15074 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
15075 hash: 0,
15076 };
15077 let table = Table {
15078 name: "items".to_owned(),
15079 fields: vec![Field::new("text", LogicalType::Varchar)],
15080 stripes: Vec::new(),
15081 rows: 0,
15082 dictionaries: vec![Some(dictionary)],
15083 dictionary_payloads: Vec::new(),
15084 distincts: vec![None],
15085 frequencies: vec![None],
15086 pair_frequencies: Vec::new(),
15087 frequency_texts: Vec::new(),
15088 host_groups: None,
15089 clustering: None,
15090 generation: 1,
15091 sections: Vec::new(),
15092 };
15093 let directory = encode_directory(&table).expect("directory");
15094 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
15095
15096 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
15097 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
15098 }
15099
15100 #[test]
15101 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
15102 let path = path("constant-codes");
15103 let mut writer =
15104 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
15105 .expect("new file");
15106 let empty = vec![Value::Varchar(String::new()); 1024];
15107 for _ in 0..4 {
15108 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
15109 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
15110 }
15111 writer.finish().expect("commit");
15112
15113 let reader = Reader::open(&path).expect("valid directory");
15114 let pages = reader.layout().columns.first().expect("one column").pages;
15115 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
15119 let read = reader.read(3, &[0]).expect("the last part back");
15120 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
15121 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
15122 fs::remove_file(path).expect("remove scratch file");
15123 }
15124
15125 #[test]
15126 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
15127 let over = vec![i64::from(i32::MAX) + 1];
15130 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
15131 assert!(format!("{error}").contains("not of its type"), "{error}");
15132 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
15133 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
15134 }
15135
15136 #[test]
15137 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
15138 let mut state: u32 = 0x9e37_79b9;
15142 let spread: Vec<u32> = (0..1024)
15143 .map(|_| {
15144 state ^= state << 13;
15145 state ^= state >> 17;
15146 state ^= state << 5;
15147 state
15148 })
15149 .collect();
15150 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
15151 let near: Vec<u32> = (0..1024).collect();
15152 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
15153 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
15154 }
15155
15156 #[test]
15162 fn two_writes_of_the_same_rows_give_the_same_bytes() {
15163 fn written(path: &PathBuf) {
15164 let fields = (0..40)
15165 .map(|column| {
15166 let ty =
15167 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
15168 Field::new(format!("c{column}"), ty)
15169 })
15170 .collect::<Vec<_>>();
15171 let mut writer = Writer::create(path, "wide", fields).expect("new file");
15172 for part in 0..70_u64 {
15173 let columns = (0..40)
15174 .map(|column| {
15175 let values = (0..64_u64)
15176 .map(|row| {
15177 let seed = part.wrapping_mul(31).wrapping_add(row);
15178 if column % 4 == 0 {
15179 Value::Varchar(format!("v{}", seed % 17))
15180 } else {
15181 Value::BigInt(i64::try_from(seed % 97).expect("small"))
15182 }
15183 })
15184 .collect::<Vec<_>>();
15185 let ty = if column % 4 == 0 {
15186 LogicalType::Varchar
15187 } else {
15188 LogicalType::BigInt
15189 };
15190 Vector::from_values(ty, &values).expect("a column")
15191 })
15192 .collect::<Vec<_>>();
15193 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
15194 }
15195 writer.finish().expect("commit");
15196 }
15197
15198 let first = path("repeatable-one");
15199 let second = path("repeatable-two");
15200 written(&first);
15201 written(&second);
15202 let left = fs::read(&first).expect("the first file");
15203 let right = fs::read(&second).expect("the second file");
15204 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
15205 assert!(left == right, "two writes of the same rows differ in their bytes");
15206
15207 let reader = Reader::open(&first).expect("valid directory");
15210 assert_eq!(reader.table().rows(), 70 * 64);
15211 let read = reader.read(0, &[0, 1]).expect("the first part back");
15212 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
15213 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
15214 fs::remove_file(first).expect("remove scratch file");
15215 fs::remove_file(second).expect("remove scratch file");
15216 }
15217
15218 fn three_tables(path: &PathBuf) {
15220 let writer = Writer::create(
15221 path,
15222 "region",
15223 vec![
15224 Field::new("r_key", LogicalType::Integer),
15225 Field::new("r_name", LogicalType::Varchar),
15226 ],
15227 )
15228 .expect("new file");
15229 let mut writer = writer;
15230 writer
15231 .append(
15232 &Chunk::new(vec![
15233 Vector::from_values(
15234 LogicalType::Integer,
15235 &[Value::Integer(0), Value::Integer(1)],
15236 )
15237 .expect("keys"),
15238 Vector::from_values(
15239 LogicalType::Varchar,
15240 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
15241 )
15242 .expect("names"),
15243 ])
15244 .expect("two columns"),
15245 )
15246 .expect("a part");
15247 let mut writer = writer
15248 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
15249 .expect("a second table");
15250 writer
15251 .append(
15252 &Chunk::new(vec![
15253 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
15254 ])
15255 .expect("one column"),
15256 )
15257 .expect("a part");
15258 let mut writer =
15259 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
15260 for part in 0..70_i64 {
15261 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
15262 writer
15263 .append(
15264 &Chunk::new(vec![
15265 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
15266 ])
15267 .expect("one column"),
15268 )
15269 .expect("a part");
15270 }
15271 writer.finish().expect("commit");
15272 }
15273
15274 #[test]
15275 fn three_tables_in_one_file_read_back_by_name() {
15276 let file = path("three-tables");
15277 three_tables(&file);
15278 let catalog = Catalog::open(&file).expect("a committed catalog");
15279 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
15280
15281 let region = catalog.table("region").expect("the first table");
15282 assert_eq!(region.table().rows(), 2);
15283 assert_eq!(
15284 region.read(0, &[1]).expect("names").value_at(1, 0),
15285 Value::Varchar("ASIA".to_owned())
15286 );
15287
15288 let wide = catalog.table("wide").expect("the third table");
15289 assert_eq!(wide.table().rows(), 70 * 64);
15290 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
15291
15292 let empty = catalog.table("empty").expect("the second table");
15295 assert_eq!(empty.table().rows(), 1);
15296 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
15297
15298 fs::remove_file(file).expect("remove scratch file");
15299 }
15300
15301 #[test]
15302 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
15303 let file = path("three-tables-missing");
15304 three_tables(&file);
15305 let catalog = Catalog::open(&file).expect("a committed catalog");
15306 let error = catalog.table("nation").expect_err("no such table");
15307 assert!(error.message().contains("nation"), "{}", error.message());
15308 fs::remove_file(file).expect("remove scratch file");
15309 }
15310
15311 #[test]
15312 fn a_file_of_three_tables_will_not_open_as_one() {
15313 let file = path("three-tables-unnamed");
15314 three_tables(&file);
15315 let error = Reader::open(&file).expect_err("more than one table");
15316 assert!(error.message().contains("more than one table"), "{}", error.message());
15317 fs::remove_file(file).expect("remove scratch file");
15318 }
15319
15320 #[test]
15322 fn decimals_of_every_storage_width_round_trip() {
15323 let file = path("decimals");
15324 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
15325 let fields = widths
15326 .iter()
15327 .enumerate()
15328 .map(|(index, (width, scale))| {
15329 Field::new(
15330 format!("d{index}"),
15331 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15332 )
15333 })
15334 .collect::<Vec<_>>();
15335 let mut writer = Writer::create(&file, "money", fields).expect("new file");
15336 let rows: [i128; 3] = [-1234, 0, 999];
15337 let columns = widths
15338 .iter()
15339 .map(|(width, scale)| {
15340 let values = rows
15341 .iter()
15342 .map(|unscaled| Value::Decimal {
15343 unscaled: *unscaled,
15344 width: *width,
15345 scale: *scale,
15346 })
15347 .collect::<Vec<_>>();
15348 Vector::from_values(
15349 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15350 &values,
15351 )
15352 .expect("a decimal column")
15353 })
15354 .collect::<Vec<_>>();
15355 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
15356 writer.finish().expect("commit");
15357
15358 let reader = Reader::open(&file).expect("a committed file");
15359 for (index, (width, scale)) in widths.iter().enumerate() {
15360 assert_eq!(
15361 reader.table().fields()[index].ty,
15362 LogicalType::decimal(*width, *scale).expect("a decimal type"),
15363 "column {index} came back as another type"
15364 );
15365 let column = reader.read(0, &[index]).expect("the column");
15366 for (row, unscaled) in rows.iter().enumerate() {
15367 assert_eq!(
15368 column.value_at(row, 0),
15369 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
15370 "column {index} row {row}"
15371 );
15372 }
15373 }
15374 fs::remove_file(file).expect("remove scratch file");
15375 }
15376
15377 #[test]
15378 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
15379 let file = path("two-of-a-name");
15380 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
15381 .expect("new file");
15382 let error = writer
15383 .next("t", vec![Field::new("a", LogicalType::BigInt)])
15384 .expect_err("the same name twice");
15385 assert!(error.message().contains("same name"), "{}", error.message());
15386 fs::remove_file(file).expect("remove scratch file");
15387 }
15388
15389 #[test]
15390 fn opening_the_catalog_reads_no_table_directory() {
15391 let file = path("catalog-only");
15392 three_tables(&file);
15393 let catalog = Catalog::open(&file).expect("a committed catalog");
15394 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
15397 assert_eq!(catalog.names().len(), 3);
15398 fs::remove_file(file).expect("remove scratch file");
15399 }
15400
15401 #[test]
15412 fn the_checksum_answers_what_it_has_always_answered() {
15413 let bytes: Vec<u8> =
15414 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
15415 for (length, expected) in [
15416 (0, 0xef46_db37_51d8_e999),
15417 (1, 0xa96c_7f0c_e858_bbb7),
15418 (3, 0x56e6_9576_32a4_87f9),
15419 (4, 0xc60d_15b1_e3ff_8f04),
15420 (5, 0x8088_1585_8624_dd4e),
15421 (7, 0xafbe_fc3d_6c6f_9a8e),
15422 (8, 0x3da5_c7aa_2696_83e0),
15423 (9, 0x465e_c429_b13c_3892),
15424 (15, 0xdee8_9d8a_065a_6233),
15425 (16, 0x1330_489a_7767_9c80),
15426 (31, 0x3391_303d_485e_846e),
15427 (32, 0x40b7_aff7_5d45_bbc8),
15428 (33, 0x4997_cae4_951c_17a5),
15429 (39, 0x5807_28fd_5c14_5739),
15430 (40, 0xf95c_f6f5_c08a_3d3b),
15431 (63, 0x2944_b4da_fc69_b206),
15432 (64, 0xbb76_f6ef_19bd_5a1b),
15433 (65, 0x814e_0c65_4a9f_d640),
15434 (127, 0x00de_aab1_31cf_f89b),
15435 (1000, 0x9e33_00c1_cde3_c58d),
15436 ] {
15437 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
15438 }
15439 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
15440 }
15441 #[test]
15448 fn a_declared_order_comes_back_out_of_the_file() {
15449 let path = path("clustered");
15450 let shipped = vec![
15451 Field::new("key", LogicalType::BigInt),
15452 Field::new("line", LogicalType::Integer),
15453 Field::new("shipdate", LogicalType::Date),
15454 ];
15455 let plain = vec![Field::new("a", LogicalType::Integer)];
15456 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
15457
15458 let mut writer = Writer::create(&path, "lineitem", shipped)
15459 .expect("new file")
15460 .declare(stage_zero.clone())
15461 .expect("the columns are the table's");
15462 let column = |ty: LogicalType, values: &[Value]| {
15463 Vector::from_values(ty, values).expect("the values match the type")
15464 };
15465 writer
15466 .append(
15467 &Chunk::new(vec![
15468 column(
15469 LogicalType::BigInt,
15470 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
15471 ),
15472 column(
15473 LogicalType::Integer,
15474 &[
15475 Value::Integer(1),
15476 Value::Integer(1),
15477 Value::Integer(1),
15478 Value::Integer(1),
15479 ],
15480 ),
15481 column(
15482 LogicalType::Date,
15483 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
15484 ),
15485 ])
15486 .expect("three columns"),
15487 )
15488 .expect("four rows");
15489 let mut writer = writer.next("nation", plain).expect("a second table");
15490 writer
15491 .append(
15492 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
15493 .expect("one column"),
15494 )
15495 .expect("one row");
15496 writer.finish().expect("commit");
15497
15498 let catalog = Catalog::open(&path).expect("reopen");
15499 let lineitem = catalog.table("lineitem").expect("the clustered table");
15500 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
15501 let nation = catalog.table("nation").expect("the plain table");
15502 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
15503
15504 assert_eq!(lineitem.table().rows(), 4);
15507 assert_eq!(nation.table().rows(), 1);
15508 fs::remove_file(&path).ok();
15509 }
15510
15511 #[test]
15513 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
15514 let path = path("clustered-bad");
15515 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
15516 .expect("new file");
15517 let four =
15518 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
15519 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
15520 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
15521 fs::remove_file(&path).ok();
15522 }
15523
15524 #[test]
15530 fn blocks_handed_out_and_given_back_out_of_order_are_the_blocks_encoded_in_place() {
15531 let values = (0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES * 2 + 100)
15532 .map(|at| format!("http://example{}.test/page/{at:06}", at % 7))
15533 .collect::<Vec<_>>();
15534 let filled = || {
15535 let mut dictionary = GlobalDictionary::new();
15536 for value in &values {
15537 dictionary.code(value).expect("a code for every value");
15538 }
15539 dictionary.settle().expect("a shape");
15540 dictionary
15541 };
15542 let mut in_place = filled();
15543 in_place.finish_blocks().expect("every block encodes");
15544
15545 let mut handed = filled();
15546 let out = handed.hand_out(3);
15547 assert_eq!(out.len(), PAYLOAD_SAMPLE_BLOCKS * 2, "every sealed block goes out");
15548 assert!(handed.waiting.is_empty(), "and none is left to be encoded under the lock");
15549 for job in out.iter().rev() {
15550 assert_eq!(job.place().0, 3, "a block goes back to the column it came from");
15551 handed.take_back(job.place().1, job.encode().expect("encodes")).expect("taken back");
15552 }
15553 assert!(handed.early.is_empty(), "nothing is waiting on a gap");
15554 handed.finish_blocks().expect("the last block encodes");
15555
15556 assert_eq!(handed.blocks, in_place.blocks, "the same blocks in the same order");
15557 assert_eq!(handed.grams, in_place.grams, "with the same signatures");
15558 }
15559
15560 #[test]
15562 fn a_block_given_back_twice_is_refused() {
15563 let mut dictionary = GlobalDictionary::new();
15564 for at in 0..PAYLOAD_SAMPLE_BLOCKS * TEXT_PAYLOAD_VALUES {
15565 dictionary.code(&format!("value {at}")).expect("a code");
15566 }
15567 dictionary.settle().expect("a shape");
15568 let out = dictionary.hand_out(0);
15569 let last = out.last().expect("blocks went out");
15570 let at = last.place().1;
15571 dictionary.take_back(at, last.encode().expect("encodes")).expect("taken back once");
15572 assert!(dictionary.take_back(at, last.encode().expect("encodes")).is_err());
15573 }
15574
15575 #[test]
15581 fn the_dictionary_order_is_the_byte_order_however_deep_the_values_agree() {
15582 let mut values = vec![String::new(), "http://".to_owned()];
15583 for host in 0..7 {
15584 for path in 0..30 {
15585 values.push(format!("http://example{host}.test/page/{path:04}/index.html"));
15586 values.push(format!("http://example{host}.test/page/{path:04}"));
15587 }
15588 }
15589 values.push("http://example0.test/page/0000/index.htmlx".to_owned());
15590
15591 let mut dictionary = GlobalDictionary::new();
15592 for value in &values {
15593 dictionary.code(value).expect("a code for every value");
15594 }
15595 dictionary.finish_blocks().expect("the last block encodes");
15596 let ranked = dictionary.ranked(None).expect("a sorted order");
15597 assert_eq!(ranked.len(), values.len(), "one entry a distinct value");
15598
15599 let spellings = dictionary_values(&dictionary);
15600 let seen = ranked
15601 .iter()
15602 .map(|&(_, code)| {
15603 String::from_utf8(spellings[code as usize].clone()).expect("text in, text out")
15604 })
15605 .collect::<Vec<_>>();
15606 let mut wanted = values.clone();
15607 wanted.sort_unstable();
15608 assert_eq!(seen, wanted, "the order is the order the bytes give");
15609
15610 for &(carried, code) in &ranked {
15611 let value = &spellings[code as usize];
15612 assert_eq!(carried, head(value), "the head belongs to the value it is filed with");
15613 }
15614 }
15615
15616 #[test]
15621 fn the_commonest_entries_are_the_ones_a_full_sort_would_have_kept() {
15622 let entry =
15623 |value: u32, count: u64| FrequencyEntry { value: FrequencyValue::Code(value), count };
15624 let mut all = (0..FREQUENCY_ENTRIES as u32 * 3)
15625 .map(|code| entry(code, u64::from(code % 7) + 1))
15626 .collect::<Vec<_>>();
15627 all.push(FrequencyEntry { value: FrequencyValue::Null, count: 4 });
15628
15629 let mut sorted = all.clone();
15630 sorted.sort_unstable_by(|left, right| {
15631 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
15632 });
15633 let wanted_omitted = sorted[FREQUENCY_ENTRIES].count;
15634 sorted.truncate(FREQUENCY_ENTRIES);
15635
15636 let mut picked = all.clone();
15637 let omitted = keep_most_frequent(&mut picked);
15638 assert_eq!(omitted, wanted_omitted, "the largest count that did not make the cut");
15639 assert_eq!(picked.len(), FREQUENCY_ENTRIES, "the cut is where it says it is");
15640 assert!(
15641 picked
15642 .iter()
15643 .zip(&sorted)
15644 .all(|(one, two)| one.value == two.value && one.count == two.count),
15645 "the same entries in the same order"
15646 );
15647
15648 let mut short = all[..FREQUENCY_ENTRIES - 1].to_vec();
15649 let omitted = keep_most_frequent(&mut short);
15650 assert_eq!(omitted, 0, "nothing is omitted when everything fits");
15651 assert!(short.windows(2).all(|pair| pair[0].count >= pair[1].count), "still in order");
15652 }
15653
15654 #[test]
15656 fn a_short_dictionary_sorts_without_a_bucketing_pass() {
15657 let empty = GlobalDictionary::new();
15658 assert!(empty.ranked(None).expect("an empty order").is_empty(), "nothing in, nothing out");
15659
15660 let mut dictionary = GlobalDictionary::new();
15661 for value in ["pear", "apple", "", "apples", "app"] {
15662 dictionary.code(value).expect("a code for every value");
15663 }
15664 dictionary.finish_blocks().expect("the one block encodes");
15665 let spellings = dictionary_values(&dictionary);
15666 let seen = dictionary
15667 .ranked(None)
15668 .expect("a sorted order")
15669 .iter()
15670 .map(|&(_, code)| spellings[code as usize].clone())
15671 .collect::<Vec<_>>();
15672 let wanted: Vec<Vec<u8>> =
15673 [&b""[..], b"app", b"apple", b"apples", b"pear"].iter().map(|v| v.to_vec()).collect();
15674 assert_eq!(seen, wanted, "shorter first where one runs out inside another");
15675 }
15676}