1#![forbid(unsafe_code)]
34
35use std::cmp::Ordering;
36use std::collections::{HashMap, VecDeque};
37use std::fs::{File, OpenOptions};
38use std::io::{Read, Seek, SeekFrom};
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Mutex, OnceLock};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_storage::sieve::Sieve;
49use rudb_storage::{Probe, Range, Zone};
50use rudb_vector::string::StringColumn;
51use rudb_vector::validity::Validity;
52use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
53
54pub mod graph;
55pub mod section;
56pub mod stats;
57mod zones;
58
59pub use section::Section;
60pub use zones::{Common, Stripes, distincts};
61
62const MAGIC: &[u8; 8] = b"RUDBNV10";
63const DIRECTORY: &[u8; 8] = b"RUDBDI10";
64const CATALOG: &[u8; 8] = b"RUDBCA10";
65const FORMAT: u32 = 25;
66
67const READABLE: &[u32] = &[22, 23, 24, FORMAT];
86
87const HEADER: u64 = 80;
88const SLOT_BYTES: usize = 28;
89const MAX_PAGE: usize = 256 * 1024 * 1024;
90const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
91const FREQUENCIES: &[u8; 8] = b"RUDBFQ2\0";
92const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
107const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
115
116const MAX_SECTIONS: usize = 4096;
123const FREQUENCY_CANDIDATES: usize = 32_768;
124const FREQUENCY_ENTRIES: usize = 512;
125const FREQUENCY_BUILD_RANK: usize = 10;
126const FREQUENCY_ORDINALS: usize = 65_536;
127const MAX_FREQUENCY_WORKERS: usize = 32;
134
135const MAX_ENCODE_WORKERS: usize = 32;
142
143const SIEVE_BUDGET: usize = 8 * 1024;
151
152const PART_BOUND_BYTES: usize = 24;
161
162fn io(error: std::io::Error) -> Error {
163 Error::io(error.to_string())
164}
165
166fn invalid(message: &str) -> Error {
167 Error::invalid_input(format!("invalid rudb native file: {message}"))
168}
169
170fn sum(counts: impl Iterator<Item = u64>) -> u64 {
172 counts.fold(0, u64::saturating_add)
173}
174
175fn span_bytes(spans: &[Span], at: usize) -> u64 {
177 spans.get(at).map_or(0, |span| u64::from(span.length))
178}
179
180fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
182 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
183}
184
185fn checksum(bytes: &[u8]) -> u64 {
195 const P1: u64 = 11_400_714_785_074_694_791;
196 const P2: u64 = 14_029_467_366_897_019_727;
197 const P3: u64 = 1_609_587_929_392_839_161;
198 const P4: u64 = 9_650_029_242_287_828_579;
199 const P5: u64 = 2_870_177_450_012_600_261;
200 let round = |state: u64, word: u64| {
201 state.wrapping_add(word.wrapping_mul(P2)).rotate_left(31).wrapping_mul(P1)
202 };
203 let merge = |state: u64, lane: u64| (state ^ round(0, lane)).wrapping_mul(P1).wrapping_add(P4);
204 let word = |chunk: &[u8]| u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"));
205
206 let mut blocks = bytes.chunks_exact(32);
209 let mut rest = blocks.remainder();
210 let mut hash = if bytes.len() >= 32 {
211 let mut one = P1.wrapping_add(P2);
212 let mut two = P2;
213 let mut three = 0;
214 let mut four = 0_u64.wrapping_sub(P1);
215 for block in blocks.by_ref() {
216 one = round(one, word(&block[..8]));
217 two = round(two, word(&block[8..16]));
218 three = round(three, word(&block[16..24]));
219 four = round(four, word(&block[24..]));
220 }
221 let combined = one
222 .rotate_left(1)
223 .wrapping_add(two.rotate_left(7))
224 .wrapping_add(three.rotate_left(12))
225 .wrapping_add(four.rotate_left(18));
226 merge(merge(merge(merge(combined, one), two), three), four)
227 } else {
228 P5
229 };
230 hash = hash.wrapping_add(bytes.len() as u64);
231 let mut words = rest.chunks_exact(8);
232 for chunk in words.by_ref() {
233 hash ^= round(0, word(chunk));
234 hash = hash.rotate_left(27).wrapping_mul(P1).wrapping_add(P4);
235 }
236 rest = words.remainder();
237 if rest.len() >= 4 {
238 let (head, tail) = rest.split_at(4);
239 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
240 hash ^= u64::from(quarter).wrapping_mul(P1);
241 hash = hash.rotate_left(23).wrapping_mul(P2).wrapping_add(P3);
242 rest = tail;
243 }
244 for &byte in rest {
245 hash ^= u64::from(byte).wrapping_mul(P5);
246 hash = hash.rotate_left(11).wrapping_mul(P1);
247 }
248 hash ^= hash >> 33;
249 hash = hash.wrapping_mul(P2);
250 hash ^= hash >> 29;
251 hash = hash.wrapping_mul(P3);
252 hash ^ (hash >> 32)
253}
254
255#[derive(Debug, Clone, Copy)]
256struct Slot {
257 offset: u64,
258 length: u32,
259 generation: u64,
260 hash: u64,
261}
262
263impl Slot {
264 fn bytes(self) -> [u8; SLOT_BYTES] {
265 let mut result = [0; SLOT_BYTES];
266 result[..8].copy_from_slice(&self.offset.to_le_bytes());
267 result[8..12].copy_from_slice(&self.length.to_le_bytes());
268 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
269 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
270 result
271 }
272
273 fn read(bytes: &[u8]) -> Self {
274 Self {
275 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
276 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
277 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
278 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
279 }
280 }
281}
282
283#[derive(Debug, Clone, Copy)]
284struct Page {
285 offset: u64,
286 length: u32,
287 hash: u64,
288}
289
290impl Page {
291 fn bytes(&self) -> u64 {
293 u64::from(self.length)
294 }
295}
296
297#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
298enum FrequencyValue {
299 Null,
300 Integer(i128),
301 Code(u32),
302}
303
304#[derive(Debug, Clone)]
305struct FrequencyEntry {
306 value: FrequencyValue,
307 count: u64,
308}
309
310#[derive(Debug, Clone)]
315struct FrequencySummary {
316 entries: Vec<FrequencyEntry>,
317 omitted_max: u64,
318 ordinals: Vec<u64>,
319}
320
321#[derive(Debug, Clone)]
326pub struct FrequencyPrefix {
327 pub entries: Vec<(Value, u64)>,
329 pub omitted_max: u64,
331}
332
333#[derive(Debug, Clone, PartialEq, Eq)]
335pub struct FrequencyOccurrences {
336 pub omitted_max: u64,
338 pub ordinals: Vec<u64>,
340}
341
342#[derive(Debug, Clone, Copy, Default)]
349struct Span {
350 offset: u64,
351 length: u32,
352}
353
354#[derive(Debug, Clone)]
356pub struct Stripe {
357 rows: usize,
358 parts: Vec<u32>,
361 index: Span,
365 pages: Vec<Span>,
366 memberships: Vec<Option<Page>>,
367 sieves: Vec<Option<Page>>,
370 part_ranges: Vec<Option<Page>>,
381 zone: Zone,
382}
383
384impl Stripe {
385 #[must_use]
387 pub fn rows(&self) -> usize {
388 self.rows
389 }
390
391 #[must_use]
393 pub fn parts(&self) -> usize {
394 self.parts.len()
395 }
396
397 #[must_use]
403 pub fn zone(&self) -> &Zone {
404 &self.zone
405 }
406}
407
408#[derive(Debug, Clone)]
410pub struct Table {
411 name: String,
412 fields: Vec<Field>,
413 stripes: Vec<Stripe>,
414 rows: usize,
415 dictionaries: Vec<Option<Page>>,
416 frequencies: Vec<Option<FrequencySummary>>,
417 distincts: Vec<Option<u64>>,
427 clustering: Option<Clustering>,
435 generation: u64,
449 sections: Vec<Section>,
456}
457
458impl Table {
459 #[must_use]
461 pub fn name(&self) -> &str {
462 &self.name
463 }
464
465 #[must_use]
467 pub fn fields(&self) -> &[Field] {
468 &self.fields
469 }
470
471 #[must_use]
473 pub fn rows(&self) -> usize {
474 self.rows
475 }
476
477 #[must_use]
479 pub fn stripes(&self) -> &[Stripe] {
480 &self.stripes
481 }
482
483 #[must_use]
485 pub fn clustering(&self) -> Option<&Clustering> {
486 self.clustering.as_ref()
487 }
488
489 #[must_use]
494 pub fn generation(&self) -> u64 {
495 self.generation
496 }
497
498 #[must_use]
505 pub fn sections(&self) -> &[Section] {
506 &self.sections
507 }
508}
509
510#[derive(Debug, Clone)]
522struct Entry {
523 name: String,
524 fields: Vec<Field>,
525 rows: usize,
526 directory: Page,
528}
529
530#[derive(Debug, Clone, PartialEq, Eq)]
543pub struct ViewEntry {
544 pub name: String,
546 pub sql: String,
548 pub statement: String,
550 pub aliases: Vec<String>,
552 pub columns: Vec<Field>,
554}
555
556#[derive(Debug, Clone)]
558pub struct ColumnLayout {
559 pub name: String,
561 pub kind: String,
563 pub pages: u64,
565 pub memberships: u64,
567 pub sieves: u64,
569 pub part_ranges: u64,
571 pub dictionary: u64,
573}
574
575impl ColumnLayout {
576 #[must_use]
578 pub fn total(&self) -> u64 {
579 self.pages
580 .saturating_add(self.memberships)
581 .saturating_add(self.sieves)
582 .saturating_add(self.part_ranges)
583 .saturating_add(self.dictionary)
584 }
585}
586
587#[derive(Debug, Clone)]
598pub struct Layout {
599 pub file: u64,
601 pub rows: usize,
603 pub stripes: usize,
605 pub parts: usize,
607 pub columns: Vec<ColumnLayout>,
609 pub indexes: u64,
612 pub directory: u64,
614 pub header: u64,
616}
617
618impl Layout {
619 #[must_use]
621 pub fn columns_total(&self) -> u64 {
622 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
623 }
624
625 #[must_use]
631 pub fn unaccounted(&self) -> u64 {
632 self.file
633 .saturating_sub(self.columns_total())
634 .saturating_sub(self.indexes)
635 .saturating_sub(self.directory)
636 .saturating_sub(self.header)
637 }
638}
639
640#[derive(Debug, Clone)]
651pub struct StoredPart {
652 pub stripe: usize,
654 pub part: usize,
656 pub row: usize,
658 pub rows: usize,
660 pub encoding: String,
662 pub bytes: u64,
664 pub page: u64,
666 pub offset: u64,
668 pub low: Option<Value>,
670 pub high: Option<Value>,
672 pub nulls: Option<usize>,
674}
675
676#[derive(Debug)]
678struct GlobalDictionary {
679 primary: HashMap<u64, u32>,
680 collisions: HashMap<u64, Vec<u32>>,
681 offsets: Vec<u32>,
682 payload: Vec<u8>,
683 counts: Vec<u64>,
684 nulls: u64,
685}
686
687impl GlobalDictionary {
688 fn new() -> Self {
689 Self {
690 primary: HashMap::new(),
691 collisions: HashMap::new(),
692 offsets: vec![0],
693 payload: Vec::new(),
694 counts: Vec::new(),
695 nulls: 0,
696 }
697 }
698
699 fn bytes(&self, code: u32) -> Option<&[u8]> {
700 let start = *self.offsets.get(code as usize)? as usize;
701 let end = *self.offsets.get(code as usize + 1)? as usize;
702 self.payload.get(start..end)
703 }
704
705 fn code(&mut self, text: &str) -> Result<u32> {
706 let hash = checksum(text.as_bytes());
707 if let Some(&code) = self.primary.get(&hash) {
708 if self.bytes(code) == Some(text.as_bytes()) {
709 return Ok(code);
710 }
711 if let Some(codes) = self.collisions.get(&hash) {
712 if let Some(code) =
713 codes.iter().copied().find(|&code| self.bytes(code) == Some(text.as_bytes()))
714 {
715 return Ok(code);
716 }
717 }
718 let code = self.insert(text)?;
719 self.collisions.entry(hash).or_default().push(code);
720 return Ok(code);
721 }
722 let code = self.insert(text)?;
723 self.primary.insert(hash, code);
724 Ok(code)
725 }
726
727 fn insert(&mut self, text: &str) -> Result<u32> {
728 let code = u32::try_from(self.offsets.len() - 1)
729 .map_err(|_| invalid("global dictionary has too many values"))?;
730 self.payload.extend_from_slice(text.as_bytes());
731 self.offsets.push(
732 u32::try_from(self.payload.len())
733 .map_err(|_| invalid("global dictionary payload exceeds 4 GiB"))?,
734 );
735 self.counts.push(0);
736 Ok(code)
737 }
738
739 fn ranked(&self) -> Vec<(u64, u32)> {
759 let count = self.offsets.len() - 1;
760 let mut ranked = (0..count)
761 .map(|code| {
762 let code = code as u32;
763 (head(self.bytes(code).unwrap_or_default()), code)
764 })
765 .collect::<Vec<_>>();
766 ranked.sort_unstable_by(|left, right| {
767 left.0.cmp(&right.0).then_with(|| self.bytes(left.1).cmp(&self.bytes(right.1)))
768 });
769 ranked
770 }
771
772 fn observe(&mut self, code: u32, null: bool) -> Result<()> {
773 if null {
774 self.nulls = self.nulls.saturating_add(1);
775 return Ok(());
776 }
777 let count = self
778 .counts
779 .get_mut(code as usize)
780 .ok_or_else(|| invalid("global dictionary count code is out of range"))?;
781 *count = count.saturating_add(1);
782 Ok(())
783 }
784}
785
786#[derive(Debug)]
794pub struct Writer {
795 file: File,
796 at: u64,
804 table: Table,
805 generation: u64,
806 order: Vec<((u64, u64), (u64, u64))>,
809 next_order: u64,
810 dictionaries: Vec<Option<GlobalDictionary>>,
811 pending: Vec<PendingChunk>,
812 closed: Vec<Entry>,
814 views: Vec<ViewEntry>,
819}
820
821#[derive(Debug)]
829struct PendingChunk {
830 order: (u64, u64),
831 chunk: Chunk,
832}
833
834#[derive(Debug)]
840struct ColumnStripe {
841 pages: Vec<Vec<u8>>,
842 codes: Vec<Option<Vec<u32>>>,
843 sieves: Vec<Option<Sieve>>,
844 ranges: Vec<Range>,
845}
846
847fn weight(ty: &LogicalType) -> usize {
855 match ty {
856 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
857 LogicalType::HugeInt
858 | LogicalType::UHugeInt
859 | LogicalType::Uuid
860 | LogicalType::Interval => 16,
861 LogicalType::BigInt
862 | LogicalType::UBigInt
863 | LogicalType::Timestamp
864 | LogicalType::Time
865 | LogicalType::TimeTz
866 | LogicalType::TimestampTz
867 | LogicalType::TimestampS
868 | LogicalType::TimestampMs
869 | LogicalType::TimestampNs
870 | LogicalType::Double
871 | LogicalType::Decimal { .. } => 8,
872 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
873 LogicalType::SmallInt | LogicalType::USmallInt => 2,
874 _ => 1,
875 }
876}
877
878pub const STRIPE_PARTS: usize = 64;
885
886const DICTIONARY_DECIDE_ROWS: usize = 4_096;
894
895const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
911
912const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
914
915fn index_section(parts: usize) -> Result<usize> {
917 parts
918 .checked_mul(INDEX_ENTRY)
919 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
920 .ok_or_else(|| invalid("index page length overflow"))
921}
922
923impl Writer {
924 pub fn open(
943 path: impl AsRef<Path>,
944 name: impl Into<String>,
945 fields: Vec<Field>,
946 ) -> Result<Self> {
947 for field in &fields {
948 type_tag(&field.ty)?;
949 }
950 let name = name.into();
951 let path = path.as_ref();
952 let (_, size, slot, bytes, _) = slot_bytes(path)?;
953 let (mut closed, views) = decode_catalog(&bytes, size)?;
954 if let Some(at) = closed.iter().position(|held| held.name == name) {
965 if closed[at].rows > 0 {
966 return Err(invalid("two tables in one native file have the same name"));
967 }
968 closed.remove(at);
969 }
970 let generation = slot
975 .generation
976 .checked_add(1)
977 .ok_or_else(|| invalid("native file generation overflow"))?;
978 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
979 Ok(Self {
980 file,
981 at: size,
984 dictionaries: fields
985 .iter()
986 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
987 .collect(),
988 table: Table {
989 name,
990 dictionaries: vec![None; fields.len()],
991 distincts: vec![None; fields.len()],
992 fields,
993 stripes: Vec::new(),
994 rows: 0,
995 frequencies: Vec::new(),
996 clustering: None,
997 generation,
998 sections: Vec::new(),
999 },
1000 generation,
1001 order: Vec::new(),
1002 next_order: 0,
1003 pending: Vec::with_capacity(STRIPE_PARTS),
1004 closed,
1005 views,
1006 })
1007 }
1008
1009 pub fn create(
1015 path: impl AsRef<Path>,
1016 name: impl Into<String>,
1017 fields: Vec<Field>,
1018 ) -> Result<Self> {
1019 for field in &fields {
1020 type_tag(&field.ty)?;
1021 }
1022 let file =
1023 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1024 let mut header = [0; HEADER as usize];
1025 header[..8].copy_from_slice(MAGIC);
1026 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1027 write_at(&file, 0, &header)?;
1028 Ok(Self {
1029 file,
1030 at: HEADER,
1031 dictionaries: fields
1032 .iter()
1033 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1034 .collect(),
1035 table: Table {
1036 name: name.into(),
1037 dictionaries: vec![None; fields.len()],
1038 distincts: vec![None; fields.len()],
1039 fields,
1040 stripes: Vec::new(),
1041 rows: 0,
1042 frequencies: Vec::new(),
1043 clustering: None,
1044 generation: 1,
1045 sections: Vec::new(),
1046 },
1047 generation: 1,
1048 order: Vec::new(),
1049 next_order: 0,
1050 pending: Vec::with_capacity(STRIPE_PARTS),
1051 closed: Vec::new(),
1052 views: Vec::new(),
1053 })
1054 }
1055
1056 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1078 let file =
1079 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1080 let mut header = [0; HEADER as usize];
1081 header[..8].copy_from_slice(MAGIC);
1082 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1083 write_at(&file, 0, &header)?;
1084 let catalog = encode_catalog(&[], views)?;
1085 write_at(&file, HEADER, &catalog)?;
1086 file.sync_all().map_err(io)?;
1090 let slot = Slot {
1091 offset: HEADER,
1092 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1093 generation: 1,
1094 hash: checksum(&catalog),
1095 };
1096 write_at(&file, slot_offset(1), &slot.bytes())?;
1097 file.sync_all().map_err(io)?;
1098 Ok(())
1099 }
1100
1101 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1112 for field in &fields {
1113 type_tag(&field.ty)?;
1114 }
1115 let name = name.into();
1116 let entry = self.close()?;
1117 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1118 return Err(invalid("two tables in one native file have the same name"));
1119 }
1120 let Self { file, at, generation, mut closed, views, .. } = self;
1121 closed.push(entry);
1122 Ok(Self {
1123 file,
1124 at,
1125 generation,
1126 closed,
1127 views,
1128 dictionaries: fields
1129 .iter()
1130 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1131 .collect(),
1132 table: Table {
1133 name,
1134 dictionaries: vec![None; fields.len()],
1135 distincts: vec![None; fields.len()],
1136 fields,
1137 stripes: Vec::new(),
1138 rows: 0,
1139 frequencies: Vec::new(),
1140 clustering: None,
1141 generation,
1142 sections: Vec::new(),
1143 },
1144 order: Vec::new(),
1145 next_order: 0,
1146 pending: Vec::with_capacity(STRIPE_PARTS),
1147 })
1148 }
1149
1150 #[must_use]
1160 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1161 self.views = views;
1162 self
1163 }
1164
1165 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1180 self.table.clustering = Some(Clustering::new(
1183 clustering.columns().to_vec(),
1184 clustering.width(),
1185 &self.table.fields,
1186 )?);
1187 Ok(self)
1188 }
1189
1190 fn put(&mut self, bytes: &[u8]) -> Result<()> {
1195 write_at(&self.file, self.at, bytes)?;
1196 self.at = self
1197 .at
1198 .checked_add(bytes.len() as u64)
1199 .ok_or_else(|| invalid("native file length overflow"))?;
1200 Ok(())
1201 }
1202
1203 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1209 let order = (self.next_order, 0);
1210 self.next_order = self.next_order.saturating_add(1);
1211 self.append_at(order, chunk)
1212 }
1213
1214 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1225 if chunk.is_empty() {
1226 return Ok(());
1227 }
1228 self.admit(chunk)?;
1229 if self.pending.last().is_some_and(|last| last.order > order) {
1230 self.flush_pending()?;
1231 }
1232 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
1237 if self.pending.len() == STRIPE_PARTS {
1238 self.flush_pending()?;
1239 }
1240 Ok(())
1241 }
1242
1243 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
1259 if parts.len() > STRIPE_PARTS {
1260 return Err(invalid("a stripe was handed more parts than it holds"));
1261 }
1262 self.flush_pending()?;
1265 for (order, chunk) in parts {
1266 if chunk.is_empty() {
1267 continue;
1268 }
1269 self.admit(&chunk)?;
1270 self.pending.push(PendingChunk { order, chunk });
1271 }
1272 self.flush_pending()
1273 }
1274
1275 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
1277 if chunk.width() != self.table.fields.len() {
1278 return Err(invalid("chunk width differs from table schema"));
1279 }
1280 for (index, field) in self.table.fields.iter().enumerate() {
1281 if chunk.column(index)?.logical_type() != &field.ty {
1282 return Err(invalid("chunk type differs from table schema"));
1283 }
1284 }
1285 self.table.rows = self
1286 .table
1287 .rows
1288 .checked_add(chunk.len())
1289 .ok_or_else(|| invalid("row count overflow"))?;
1290 Ok(())
1291 }
1292
1293 fn encode_column(
1324 index: usize,
1325 held: &[PendingChunk],
1326 dictionary: &mut Option<GlobalDictionary>,
1327 ) -> Result<ColumnStripe> {
1328 let deciding = dictionary.as_ref().is_some_and(|held| held.offsets.len() == 1);
1331 let stripe = Self::encode_pages(index, held, dictionary.as_mut())?;
1332 if !deciding {
1333 return Ok(stripe);
1334 }
1335 let rows: usize = held.iter().map(|pending| pending.chunk.len()).sum();
1336 let distinct = dictionary.as_ref().map_or(0, |held| held.offsets.len() - 1);
1337 if rows < DICTIONARY_DECIDE_ROWS
1338 || distinct.saturating_mul(10) <= rows.saturating_mul(DICTIONARY_DISTINCT_IN_TEN)
1339 {
1340 return Ok(stripe);
1341 }
1342 *dictionary = None;
1343 Self::encode_pages(index, held, None)
1344 }
1345
1346 fn encode_pages(
1348 index: usize,
1349 held: &[PendingChunk],
1350 mut dictionary: Option<&mut GlobalDictionary>,
1351 ) -> Result<ColumnStripe> {
1352 let mut stripe = ColumnStripe {
1353 pages: Vec::with_capacity(held.len()),
1354 codes: Vec::with_capacity(held.len()),
1355 sieves: Vec::with_capacity(held.len()),
1356 ranges: Vec::with_capacity(held.len()),
1357 };
1358 for pending in held {
1359 let column = pending.chunk.column(index)?;
1360 let (bytes, unique) = encode(column, dictionary.as_deref_mut())?;
1361 if bytes.len() > MAX_PAGE {
1362 return Err(invalid("column page exceeds the configured bound"));
1363 }
1364 let range = Range::of(column);
1367 let sieve = match dictionary {
1380 Some(_) => None,
1381 None => Sieve::of(column, &range, SIEVE_BUDGET)
1382 .filter(|sieve| sieve.len() < bytes.len()),
1383 };
1384 stripe.pages.push(bytes);
1385 stripe.codes.push(unique);
1386 stripe.sieves.push(sieve);
1387 stripe.ranges.push(range);
1388 }
1389 Ok(stripe)
1390 }
1391
1392 fn encode_columns(&mut self, held: &[PendingChunk]) -> Result<Vec<ColumnStripe>> {
1401 let width = self.table.fields.len();
1402 let workers = std::thread::available_parallelism()
1403 .map_or(1, usize::from)
1404 .min(MAX_ENCODE_WORKERS)
1405 .min(width);
1406 if workers <= 1 || held.len() <= 1 {
1407 return self
1408 .dictionaries
1409 .iter_mut()
1410 .enumerate()
1411 .map(|(index, dictionary)| Self::encode_column(index, held, dictionary))
1412 .collect();
1413 }
1414 let mut jobs: Vec<(usize, Option<GlobalDictionary>)> =
1417 std::mem::take(&mut self.dictionaries).into_iter().enumerate().collect();
1418 jobs.sort_by_key(|(index, _)| weight(&self.table.fields[*index].ty));
1420 let queue = Mutex::new(jobs);
1421 let pieces = std::thread::scope(|scope| {
1422 (0..workers)
1423 .map(|_| {
1424 scope.spawn(|| {
1425 let mut mine = Vec::new();
1426 loop {
1427 let taken = queue
1428 .lock()
1429 .map_err(|_| Error::internal("a native encode worker panicked"))?
1430 .pop();
1431 let Some((index, mut dictionary)) = taken else { break };
1432 let encoded = Self::encode_column(index, held, &mut dictionary)?;
1433 mine.push((index, dictionary, encoded));
1434 }
1435 Ok(mine)
1436 })
1437 })
1438 .collect::<Vec<_>>()
1439 .into_iter()
1440 .map(|handle| {
1441 handle.join().map_err(|_| Error::internal("a native encode worker panicked"))?
1442 })
1443 .collect::<Result<Vec<_>>>()
1444 })?;
1445 let mut dictionaries: Vec<Option<GlobalDictionary>> = (0..width).map(|_| None).collect();
1446 let mut encoded: Vec<Option<ColumnStripe>> = (0..width).map(|_| None).collect();
1447 for piece in pieces {
1448 for (index, dictionary, stripe) in piece {
1449 dictionaries[index] = dictionary;
1450 encoded[index] = Some(stripe);
1451 }
1452 }
1453 self.dictionaries = dictionaries;
1454 encoded
1455 .into_iter()
1456 .map(|stripe| stripe.ok_or_else(|| Error::internal("a column was never encoded")))
1457 .collect()
1458 }
1459
1460 fn flush_pending(&mut self) -> Result<()> {
1462 if self.pending.is_empty() {
1463 return Ok(());
1464 }
1465 let width = self.table.fields.len();
1466 let mut held = std::mem::take(&mut self.pending);
1469 let parts = held.len();
1470 let encoded = self.encode_columns(&held)?;
1471 let mut pages = Vec::with_capacity(width);
1472 let mut memberships = vec![None; width];
1473 let mut ranges = Vec::with_capacity(width);
1474 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
1475 for stripe in &encoded {
1476 let offset = self.at;
1477 let section = index.len();
1478 let mut length = 0_usize;
1479 for bytes in &stripe.pages {
1480 write_at(&self.file, self.at + length as u64, bytes)?;
1481 put_u32(
1482 &mut index,
1483 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
1484 );
1485 put_u64(&mut index, checksum(bytes));
1486 length = length
1487 .checked_add(bytes.len())
1488 .ok_or_else(|| invalid("column page length overflow"))?;
1489 }
1490 let hash = checksum(&index[section..]);
1491 put_u64(&mut index, hash);
1492 if length > MAX_PAGE {
1493 return Err(invalid("column page exceeds the configured bound"));
1494 }
1495 self.at = self
1496 .at
1497 .checked_add(length as u64)
1498 .ok_or_else(|| invalid("native file length overflow"))?;
1499 pages.push(Span {
1500 offset,
1501 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
1502 });
1503 ranges.push(merged_range(stripe.ranges.iter().cloned()));
1504 }
1505 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
1506 if stripe.codes.iter().all(Option::is_none) {
1507 continue;
1508 }
1509 let lists = stripe
1510 .codes
1511 .iter()
1512 .map(|codes| codes.clone().unwrap_or_default())
1513 .collect::<Vec<_>>();
1514 let bytes = encode_membership(&merged_codes(lists));
1515 let offset = self.at;
1516 self.put(&bytes)?;
1517 *membership = Some(Page {
1518 offset,
1519 length: u32::try_from(bytes.len())
1520 .map_err(|_| invalid("membership page length overflow"))?,
1521 hash: checksum(&bytes),
1522 });
1523 }
1524 let mut sieves = vec![None; width];
1525 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
1526 if stripe.sieves.iter().all(Option::is_none) {
1527 continue;
1528 }
1529 let bytes = encode_sieves(stripe.sieves.iter())?;
1530 let offset = self.at;
1531 self.put(&bytes)?;
1532 *page = Some(Page {
1533 offset,
1534 length: u32::try_from(bytes.len())
1535 .map_err(|_| invalid("sieve page length overflow"))?,
1536 hash: checksum(&bytes),
1537 });
1538 }
1539 let mut part_ranges = vec![None; width];
1545 if parts > 1 {
1546 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
1547 let bytes = encode_part_ranges(&stripe.ranges)?;
1548 if bytes.len() >= span.length as usize {
1549 continue;
1550 }
1551 let offset = self.at;
1552 self.put(&bytes)?;
1553 *page = Some(Page {
1554 offset,
1555 length: u32::try_from(bytes.len())
1556 .map_err(|_| invalid("part range page length overflow"))?,
1557 hash: checksum(&bytes),
1558 });
1559 }
1560 }
1561 let offset = self.at;
1562 self.put(&index)?;
1563 let index = Span {
1564 offset,
1565 length: u32::try_from(index.len())
1566 .map_err(|_| invalid("index page length overflow"))?,
1567 };
1568 let mut rows = 0_usize;
1569 let mut lengths = Vec::with_capacity(parts);
1570 let mut span = None;
1571 for pending in held.drain(..) {
1572 let part = pending.chunk.len();
1573 rows = rows.checked_add(part).ok_or_else(|| invalid("row count overflow"))?;
1574 lengths.push(u32::try_from(part).map_err(|_| invalid("part row count overflow"))?);
1575 span = Some(
1576 span.map_or((pending.order, pending.order), |(first, _)| (first, pending.order)),
1577 );
1578 }
1579 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
1580 self.table.stripes.push(Stripe {
1581 rows,
1582 parts: lengths,
1583 index,
1584 pages,
1585 memberships,
1586 sieves,
1587 part_ranges,
1588 zone: Zone::from_ranges(ranges),
1589 });
1590 self.pending = held;
1592 Ok(())
1593 }
1594
1595 fn numeric_frequency(&self, column: usize) -> Result<Option<FrequencySummary>> {
1599 let ty = &self.table.fields[column].ty;
1600 if !matches!(
1601 ty,
1602 LogicalType::TinyInt
1603 | LogicalType::SmallInt
1604 | LogicalType::Integer
1605 | LogicalType::BigInt
1606 | LogicalType::UTinyInt
1607 | LogicalType::USmallInt
1608 | LogicalType::UInteger
1609 | LogicalType::UBigInt
1610 | LogicalType::Date
1611 | LogicalType::Timestamp
1612 ) {
1613 return Ok(None);
1614 }
1615 let mut candidates: HashMap<FrequencyValue, u32> = HashMap::new();
1616 let mut decrements = 0_u64;
1617 self.visit_numeric(column, |_, value| {
1618 if let Some(count) = candidates.get_mut(&value) {
1619 *count = count.saturating_add(1);
1620 } else if candidates.len() < FREQUENCY_CANDIDATES {
1621 candidates.insert(value, 1);
1622 } else {
1623 candidates.retain(|_, count| {
1624 *count -= 1;
1625 *count != 0
1626 });
1627 decrements = decrements.saturating_add(1);
1628 }
1629 })?;
1630 let (exact, ordinals) = if decrements == 0 {
1631 (
1632 candidates
1633 .into_iter()
1634 .map(|(value, count)| (value, u64::from(count)))
1635 .collect::<HashMap<_, _>>(),
1636 Vec::new(),
1637 )
1638 } else {
1639 let mut lower = candidates.values().copied().collect::<Vec<_>>();
1640 lower.sort_unstable_by(|left, right| right.cmp(left));
1641 if lower.len() < FREQUENCY_BUILD_RANK
1642 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
1643 {
1644 return Ok(None);
1645 }
1646 let mut exact =
1647 candidates.into_keys().map(|value| (value, 0_u64)).collect::<HashMap<_, _>>();
1648 let mut ordinals = Vec::new();
1649 let mut exceeded = false;
1650 self.visit_numeric(column, |ordinal, value| {
1651 if let Some(count) = exact.get_mut(&value) {
1652 *count = count.saturating_add(1);
1653 if !exceeded {
1654 if ordinals.len() < FREQUENCY_ORDINALS {
1655 ordinals.push(ordinal);
1656 } else {
1657 ordinals.clear();
1658 exceeded = true;
1659 }
1660 }
1661 }
1662 })?;
1663 (exact, ordinals)
1664 };
1665 let mut entries = exact
1666 .into_iter()
1667 .map(|(value, count)| FrequencyEntry { value, count })
1668 .collect::<Vec<_>>();
1669 entries.sort_unstable_by(|left, right| {
1670 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
1671 });
1672 let omitted_max =
1673 entries.get(FREQUENCY_ENTRIES).map_or(decrements, |entry| decrements.max(entry.count));
1674 entries.truncate(FREQUENCY_ENTRIES);
1675 Ok(Some(FrequencySummary { entries, omitted_max, ordinals }))
1676 }
1677
1678 fn visit_numeric(
1679 &self,
1680 column: usize,
1681 mut visit: impl FnMut(u64, FrequencyValue),
1682 ) -> Result<()> {
1683 let ty = &self.table.fields[column].ty;
1684 let mut start = 0_u64;
1685 for stripe in &self.table.stripes {
1686 let spans = read_index(&self.file, stripe, column)?;
1687 let page = stripe.pages[column];
1688 let mut bytes = vec![0; page.length as usize];
1689 read_at(&self.file, page.offset, &mut bytes)?;
1690 for (span, &rows) in spans.iter().zip(&stripe.parts) {
1691 let part = part_bytes(&bytes, *span)?;
1692 if checksum(part) != span.hash {
1693 return Err(invalid("column page checksum differs while building frequencies"));
1694 }
1695 let rows = rows as usize;
1696 let vector = decode(ty, rows, part, None)?;
1697 for row in 0..rows {
1699 let value = if vector.is_null_at(row) {
1700 FrequencyValue::Null
1701 } else {
1702 let widened = match vector.signed_at(row) {
1706 Some(value) => Some(value),
1707 None => match vector.value_at(row) {
1708 Value::UTinyInt(value) => Some(i128::from(value)),
1709 Value::USmallInt(value) => Some(i128::from(value)),
1710 Value::UInteger(value) => Some(i128::from(value)),
1711 Value::UBigInt(value) => Some(i128::from(value)),
1712 _ => None,
1713 },
1714 };
1715 FrequencyValue::Integer(widened.ok_or_else(|| {
1716 invalid("numeric frequency page did not contain an integer value")
1717 })?)
1718 };
1719 visit(start.saturating_add(row as u64), value);
1720 }
1721 start = start.saturating_add(rows as u64);
1722 }
1723 }
1724 Ok(())
1725 }
1726
1727 fn numeric_frequencies(&self) -> Result<Vec<Option<FrequencySummary>>> {
1735 let mut columns = self
1736 .table
1737 .fields
1738 .iter()
1739 .enumerate()
1740 .filter_map(|(column, field)| {
1741 matches!(
1742 field.ty,
1743 LogicalType::TinyInt
1744 | LogicalType::SmallInt
1745 | LogicalType::Integer
1746 | LogicalType::BigInt
1747 | LogicalType::UTinyInt
1748 | LogicalType::USmallInt
1749 | LogicalType::UInteger
1750 | LogicalType::UBigInt
1751 | LogicalType::Date
1752 | LogicalType::Timestamp
1753 )
1754 .then_some(column)
1755 })
1756 .collect::<Vec<_>>();
1757 let workers = std::thread::available_parallelism()
1758 .map_or(1, usize::from)
1759 .min(MAX_FREQUENCY_WORKERS)
1760 .min(columns.len());
1761 if workers <= 1 {
1762 let mut frequencies = vec![None; self.table.fields.len()];
1763 for column in columns {
1764 frequencies[column] = self.numeric_frequency(column)?;
1765 }
1766 return Ok(frequencies);
1767 }
1768 columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
1771 let queue = Mutex::new(columns);
1772 let pieces = std::thread::scope(|scope| {
1773 (0..workers)
1774 .map(|_| {
1775 scope.spawn(|| {
1776 let mut mine = Vec::new();
1777 loop {
1778 let taken = queue
1779 .lock()
1780 .map_err(|_| Error::internal("a native frequency worker panicked"))?
1781 .pop();
1782 let Some(column) = taken else { break };
1783 mine.push((column, self.numeric_frequency(column)?));
1784 }
1785 Ok(mine)
1786 })
1787 })
1788 .collect::<Vec<_>>()
1789 .into_iter()
1790 .map(|handle| {
1791 handle
1792 .join()
1793 .map_err(|_| Error::internal("a native frequency worker panicked"))?
1794 })
1795 .collect::<Result<Vec<_>>>()
1796 })?;
1797 let mut frequencies = vec![None; self.table.fields.len()];
1798 for piece in pieces {
1799 for (column, summary) in piece {
1800 frequencies[column] = summary;
1801 }
1802 }
1803 Ok(frequencies)
1804 }
1805
1806 fn close(&mut self) -> Result<Entry> {
1817 self.flush_pending()?;
1818 let mut stripes = std::mem::take(&mut self.order)
1819 .into_iter()
1820 .zip(std::mem::take(&mut self.table.stripes))
1821 .collect::<Vec<_>>();
1822 stripes.sort_by_key(|(order, _)| order.0);
1823 let mut previous: Option<(u64, u64)> = None;
1824 for ((first, last), _) in &stripes {
1825 if previous.is_some_and(|previous| previous >= *first) {
1826 return Err(invalid("chunks did not arrive in source order"));
1827 }
1828 previous = Some(*last);
1829 }
1830 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
1831 self.table.frequencies = self.numeric_frequencies()?;
1832 let dictionaries = std::mem::take(&mut self.dictionaries);
1833 let orders = rankings(&dictionaries)?;
1834 for (index, (dictionary, order)) in dictionaries.into_iter().zip(orders).enumerate() {
1835 let Some(dictionary) = dictionary else { continue };
1836 self.table.distincts[index] =
1840 Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
1841 self.table.frequencies[index] = Some(code_frequency(&dictionary));
1842 let encoded = encode_global_dictionary(dictionary, &order)?;
1843 let offset = self.at;
1844 self.put(&encoded.index)?;
1845 self.put(&encoded.ranks)?;
1846 for block in &encoded.payload {
1847 self.put(block)?;
1848 }
1849 let payload_len =
1850 encoded.payload.iter().try_fold(0_usize, |len, block| len.checked_add(block.len()));
1851 let length = payload_len
1852 .and_then(|len| len.checked_add(encoded.index.len()))
1853 .and_then(|len| len.checked_add(encoded.ranks.len()))
1854 .ok_or_else(|| invalid("dictionary page length overflow"))?;
1855 self.table.dictionaries[index] = Some(Page {
1856 offset,
1857 length: u32::try_from(length)
1858 .map_err(|_| invalid("dictionary page length overflow"))?,
1859 hash: checksum(&encoded.index),
1860 });
1861 }
1862 let directory = encode_directory(&self.table)?;
1863 if directory.len() > MAX_DIRECTORY {
1864 return Err(invalid("directory exceeds the configured bound"));
1865 }
1866 let offset = self.at;
1867 self.put(&directory)?;
1868 Ok(Entry {
1869 name: self.table.name.clone(),
1870 fields: self.table.fields.clone(),
1871 rows: self.table.rows,
1872 directory: Page {
1873 offset,
1874 length: u32::try_from(directory.len())
1875 .map_err(|_| invalid("directory length overflow"))?,
1876 hash: checksum(&directory),
1877 },
1878 })
1879 }
1880
1881 pub fn finish(mut self) -> Result<Table> {
1891 let entry = self.close()?;
1892 let mut tables = std::mem::take(&mut self.closed);
1893 tables.push(entry);
1894 let catalog = encode_catalog(&tables, &self.views)?;
1895 if catalog.len() > MAX_DIRECTORY {
1896 return Err(invalid("catalog exceeds the configured bound"));
1897 }
1898 let offset = self.at;
1899 self.put(&catalog)?;
1900 self.file.sync_all().map_err(io)?;
1904 let slot = Slot {
1905 offset,
1906 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1907 generation: self.generation,
1908 hash: checksum(&catalog),
1909 };
1910 write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
1915 self.file.sync_all().map_err(io)?;
1916 Ok(self.table)
1917 }
1918
1919 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1936 let path = path.as_ref();
1937 let (_, size, slot, bytes, _) = slot_bytes(path)?;
1938 let (closed, _) = decode_catalog(&bytes, size)?;
1939 let generation = slot
1940 .generation
1941 .checked_add(1)
1942 .ok_or_else(|| invalid("native file generation overflow"))?;
1943 let catalog = encode_catalog(&closed, views)?;
1944 if catalog.len() > MAX_DIRECTORY {
1945 return Err(invalid("catalog exceeds the configured bound"));
1946 }
1947 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1948 write_at(&file, size, &catalog)?;
1949 file.sync_all().map_err(io)?;
1950 let slot = Slot {
1951 offset: size,
1952 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1953 generation,
1954 hash: checksum(&catalog),
1955 };
1956 write_at(&file, slot_offset(generation), &slot.bytes())?;
1957 file.sync_all().map_err(io)?;
1958 Ok(())
1959 }
1960}
1961
1962fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
1968 let offset = *at;
1969 write_at(file, offset, bytes)?;
1970 *at =
1971 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
1972 Ok(offset)
1973}
1974
1975fn write_section(
1981 file: &File,
1982 at: &mut u64,
1983 one: §ion::Attachment<'_>,
1984 generation: u64,
1985) -> Result<Section> {
1986 if one.header_bytes as usize > one.bytes.len() {
1987 return Err(invalid("a section's header is longer than its payload"));
1988 }
1989 let mut extents = Vec::new();
1990 let mut first = 0_u64;
1991 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
1992 let offset = append(file, at, chunk)?;
1993 extents.push(section::Extent {
1994 offset,
1995 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
1996 hash: checksum(chunk),
1997 first,
1998 });
1999 first += chunk.len() as u64;
2000 }
2001 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
2002 section::encode_extents(&extents, &mut table)?;
2003 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
2007 Ok(Section {
2008 kind: one.kind,
2009 id: one.id,
2010 generation,
2011 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
2012 extent_page,
2013 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
2014 hash: checksum(&table),
2015 flags: one.flags,
2016 header_bytes: one.header_bytes,
2017 })
2018}
2019
2020pub fn attach(
2044 path: impl AsRef<Path>,
2045 table: &str,
2046 attachments: &[section::Attachment<'_>],
2047) -> Result<Table> {
2048 let path = path.as_ref();
2049 let (_, size, slot, bytes, _) = slot_bytes(path)?;
2050 let (mut entries, views) = decode_catalog(&bytes, size)?;
2051 let at = entries
2052 .iter()
2053 .position(|entry| entry.name == table)
2054 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
2055 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2056 let mut version = [0; 4];
2057 read_at(&file, 8, &mut version)?;
2058 let version = u32::from_le_bytes(version);
2059 if version != FORMAT {
2065 return Err(invalid(&format!(
2066 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
2067 to be written again"
2068 )));
2069 }
2070 let mut directory = vec![0; entries[at].directory.length as usize];
2071 read_at(&file, entries[at].directory.offset, &mut directory)?;
2072 if checksum(&directory) != entries[at].directory.hash {
2073 return Err(invalid(&format!("the directory of table {table} does not checksum")));
2074 }
2075 let mut held = decode_directory(&directory, size)?;
2076 let mut cursor = size;
2077 for one in attachments {
2078 let written = write_section(&file, &mut cursor, one, held.generation)?;
2079 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
2080 held.sections.push(written);
2081 }
2082 if held.sections.len() > MAX_SECTIONS {
2083 return Err(invalid("the table would name more sections than the bound allows"));
2084 }
2085 let encoded = encode_directory(&held)?;
2086 if encoded.len() > MAX_DIRECTORY {
2087 return Err(invalid("directory exceeds the configured bound"));
2088 }
2089 let offset = append(&file, &mut cursor, &encoded)?;
2090 entries[at].directory = Page {
2091 offset,
2092 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
2093 hash: checksum(&encoded),
2094 };
2095 let catalog = encode_catalog(&entries, &views)?;
2098 if catalog.len() > MAX_DIRECTORY {
2099 return Err(invalid("catalog exceeds the configured bound"));
2100 }
2101 let offset = append(&file, &mut cursor, &catalog)?;
2102 file.sync_all().map_err(io)?;
2103 let generation =
2104 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
2105 let committed = Slot {
2106 offset,
2107 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2108 generation,
2109 hash: checksum(&catalog),
2110 };
2111 write_at(&file, slot_offset(generation), &committed.bytes())?;
2112 file.sync_all().map_err(io)?;
2113 Ok(held)
2114}
2115
2116#[derive(Debug, Clone)]
2118pub struct Reader {
2119 file: Arc<File>,
2120 table: Arc<Table>,
2121 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
2122 loading: Arc<Vec<Mutex<()>>>,
2131 opened: Arc<AtomicUsize>,
2135 sieves: Arc<Vec<Vec<SieveSlot>>>,
2139 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
2142 places: Arc<Vec<Place>>,
2144 cache: Arc<Vec<Mutex<Cached>>>,
2145 pages: Arc<AtomicUsize>,
2148 indexes: Arc<AtomicUsize>,
2151 kept: Arc<AtomicUsize>,
2154 size: u64,
2156 directory: u64,
2158 opening: Opening,
2160}
2161
2162#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2174pub struct Opening {
2175 pub reads: u32,
2178 pub bytes: u64,
2180}
2181
2182#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2184pub struct Reads {
2185 pub opening: Opening,
2187 pub pages: usize,
2189 pub indexes: usize,
2191 pub dictionaries: usize,
2194}
2195
2196#[derive(Debug, Clone, Copy)]
2198struct Place {
2199 stripe: u32,
2200 part: u32,
2201 rows: u32,
2202}
2203
2204#[derive(Debug, Clone, Copy)]
2206struct PartSpan {
2207 start: usize,
2208 length: usize,
2209 hash: u64,
2210}
2211
2212#[derive(Debug, Clone)]
2218struct CachedColumn {
2219 stripe: usize,
2220 index: Arc<Vec<PartSpan>>,
2221 page: Option<Arc<Vec<u8>>>,
2222}
2223
2224#[derive(Debug, Default)]
2244struct Cached {
2245 pages: Vec<Option<Arc<Vec<u8>>>>,
2246 order: VecDeque<usize>,
2247 loading: Vec<usize>,
2248 index: Vec<Option<Arc<Vec<PartSpan>>>>,
2249}
2250
2251const CACHED_STRIPES_PER_COLUMN: usize = 4;
2263
2264type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
2266
2267type RangeSlot = OnceLock<Arc<Vec<Range>>>;
2268
2269#[derive(Debug)]
2270struct NativeText {
2271 file: Arc<File>,
2272 values: usize,
2274 offsets: Vec<u8>,
2283 offset_bits: usize,
2286 ranks: usize,
2288 rank_at: u64,
2292 rank_ends: Vec<u64>,
2296 rank_hashes: Vec<u64>,
2297 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2298 code_bits: usize,
2301 code_ranks: OnceLock<Option<Vec<u32>>>,
2308 payload: u64,
2309 ends: Vec<u64>,
2312 hashes: Vec<u64>,
2313 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2315 keep_budget: usize,
2318 payload_kept: AtomicUsize,
2326 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
2343}
2344
2345const TEXT_SEARCH_MEMO: usize = 64;
2350
2351const TEXT_PAYLOAD_VALUES: usize = 1024;
2367
2368const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
2389
2390const TEXT_OFFSET_RUN: usize = 512;
2397
2398const DICTIONARY_HEADER: usize = 16;
2401
2402const TEXT_RANK_BLOCK: usize = 512;
2413
2414const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
2428
2429impl NativeText {
2430 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
2437 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
2438 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
2439 Ok(Some(bytes.as_slice()))
2440 }
2441
2442 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
2447 let start = if block == 0 { 0 } else { self.ends[block - 1] };
2448 let end = self.ends[block];
2449 let len = end
2450 .checked_sub(start)
2451 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
2452 let mut stored = vec![
2453 0;
2454 usize::try_from(len).map_err(|_| invalid(
2455 "global dictionary block does not fit in memory"
2456 ))?
2457 ];
2458 read_at(&self.file, self.payload + start, &mut stored)?;
2459 if checksum(&stored) != self.hashes[block] {
2460 return Err(invalid("global dictionary payload checksum differs"));
2461 }
2462 let first = block * TEXT_PAYLOAD_VALUES;
2463 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
2464 let want = self.end_within(last - 1)? as usize;
2465 let values = string::decode_flat(&stored)?;
2466 if values.len() != last - first {
2467 return Err(invalid("global dictionary block holds the wrong value count"));
2468 }
2469 let bytes = values.into_bytes();
2470 if bytes.len() != want {
2471 return Err(invalid("global dictionary block decodes to the wrong length"));
2472 }
2473 Ok(bytes)
2474 }
2475
2476 fn end_within(&self, index: usize) -> Result<u32> {
2478 let run = index / TEXT_OFFSET_RUN;
2479 let bytes = self
2480 .offsets
2481 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2482 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2483 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
2484 .map_err(|_| invalid("global dictionary offsets are short"))?;
2485 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
2486 }
2487
2488 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
2501 let mut ends = Vec::with_capacity(last.saturating_sub(first));
2502 let mut at = first;
2503 while at < last {
2504 let run = at / TEXT_OFFSET_RUN;
2505 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
2506 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
2507 let bytes = self
2508 .offsets
2509 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2510 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2511 let run_ends = bitpack::unpack_tail(bytes, self.offset_bits, held)
2512 .map_err(|_| invalid("global dictionary offsets are short"))?;
2513 let within = run_ends
2514 .get(at % TEXT_OFFSET_RUN..stop - run * TEXT_OFFSET_RUN)
2515 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2516 ends.extend_from_slice(within);
2517 at = stop;
2518 }
2519 Ok(ends)
2520 }
2521
2522 fn start_within(&self, index: usize) -> Result<u32> {
2525 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
2526 }
2527
2528 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
2536 let within = index % TEXT_OFFSET_RUN;
2537 let (start, end) = if within == 0 {
2538 (self.start_within(index)?, self.end_within(index)?)
2539 } else {
2540 let run = index / TEXT_OFFSET_RUN;
2541 let bytes = self
2542 .offsets
2543 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2544 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2545 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
2546 .map_err(|_| invalid("global dictionary offsets are short"))?;
2547 let ends = u32::try_from(end)
2548 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2549 let starts = u32::try_from(start)
2550 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2551 (starts, ends)
2552 };
2553 if start > end {
2554 return Err(invalid("global dictionary value ends before it starts"));
2555 }
2556 Ok((start, end))
2557 }
2558
2559 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
2566 let slot = self
2567 .rank_blocks
2568 .get(rank / TEXT_RANK_BLOCK)
2569 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
2570 let block = slot
2571 .get_or_init(|| {
2572 let which = rank / TEXT_RANK_BLOCK;
2573 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
2574 let end = self.rank_ends[which];
2575 let mut bytes = vec![0; (end - start) as usize];
2576 read_at(&self.file, self.rank_at + start, &mut bytes)?;
2577 if checksum(&bytes)
2578 != *self
2579 .rank_hashes
2580 .get(rank / TEXT_RANK_BLOCK)
2581 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
2582 {
2583 return Err(invalid("global dictionary rank checksum differs"));
2584 }
2585 Ok(bytes)
2586 })
2587 .as_ref()
2588 .map_err(Clone::clone)?;
2589 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
2590 }
2591
2592 fn head_at(&self, rank: usize) -> Result<u64> {
2594 let (block, within) = self.rank_parts(rank)?;
2595 let (base, width, packed) = rank_heads(block)?;
2596 let above = bitpack::tail_at(packed, width, within)
2597 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
2598 Ok(base.wrapping_add(above))
2599 }
2600
2601 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
2603 let (_, width, packed) = rank_heads(block)?;
2604 packed
2605 .get(bitpack::tail_len(count, width)..)
2606 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
2607 }
2608
2609 fn rank_block_len(&self, rank: usize) -> usize {
2611 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
2612 TEXT_RANK_BLOCK.min(self.ranks - first)
2613 }
2614}
2615
2616fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
2618 let header = block
2619 .get(..RANK_BLOCK_HEADER)
2620 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
2621 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
2622 let width = header[8] as usize;
2623 if width > 64 {
2624 return Err(invalid("global dictionary rank block packs heads past a word"));
2625 }
2626 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
2627}
2628
2629fn offset_width(offsets: &[u32]) -> usize {
2636 let values = offsets.len() - 1;
2637 let mut span = 0;
2638 for first in (0..values).step_by(TEXT_PAYLOAD_VALUES) {
2639 let last = (first + TEXT_PAYLOAD_VALUES).min(values);
2640 span = span.max(offsets[last] - offsets[first]);
2641 }
2642 (u32::BITS - span.leading_zeros()) as usize
2643}
2644
2645fn offset_bytes(values: usize, bits: usize) -> usize {
2648 let full = values / TEXT_OFFSET_RUN;
2649 let rest = values % TEXT_OFFSET_RUN;
2650 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
2651}
2652
2653fn encode_offsets(offsets: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
2655 let values = offsets.len() - 1;
2656 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
2657 for first in (0..values).step_by(TEXT_OFFSET_RUN) {
2658 let last = (first + TEXT_OFFSET_RUN).min(values);
2659 let base = offsets[first / TEXT_PAYLOAD_VALUES * TEXT_PAYLOAD_VALUES];
2660 run.clear();
2661 run.extend((first..last).map(|value| u64::from(offsets[value + 1] - base)));
2662 bitpack::pack_tail(&run, bits, out)
2663 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
2664 }
2665 Ok(())
2666}
2667
2668fn code_width(values: usize) -> usize {
2670 match u64::try_from(values).unwrap_or(u64::MAX) {
2671 0 | 1 => 0,
2672 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
2673 }
2674}
2675
2676impl TextSource for NativeText {
2677 fn len(&self) -> usize {
2678 self.values
2679 }
2680
2681 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
2682 if index >= self.values {
2683 return Ok(None);
2684 }
2685 let (start, end) = self.span_within(index)?;
2686 if start == end {
2687 return Ok(Some(&[]));
2688 }
2689 let block = index / TEXT_PAYLOAD_VALUES;
2692 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
2693 Ok(bytes.get(start as usize..end as usize))
2694 }
2695
2696 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
2697 if index >= self.values {
2698 return Ok(None);
2699 }
2700 let (start, end) = self.span_within(index)?;
2701 Ok(Some((end - start) as usize))
2702 }
2703
2704 fn sweep(
2717 &self,
2718 first: usize,
2719 limit: usize,
2720 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
2721 ) -> Result<usize> {
2722 let limit = limit.min(self.values);
2723 if first >= limit {
2724 return Ok(first);
2725 }
2726 let block = first / TEXT_PAYLOAD_VALUES;
2727 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
2728 let decoded;
2729 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
2730 Some(Ok(kept)) => kept,
2731 _ if self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
2732 let kept = self
2733 .payload_block(block)?
2734 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
2735 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
2736 kept
2737 }
2738 _ => {
2739 decoded = self.decode_block(block)?;
2740 &decoded
2741 }
2742 };
2743 let ends = self.ends_within(first, last)?;
2744 if ends.len() != last - first {
2745 return Err(invalid("global dictionary offsets are short"));
2746 }
2747 let mut start = u64::from(self.start_within(first)?);
2748 for (index, &end) in (first..last).zip(&ends) {
2751 let value = usize::try_from(start)
2752 .ok()
2753 .zip(usize::try_from(end).ok())
2754 .and_then(|(from, to)| bytes.get(from..to))
2755 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
2756 body(index, value)?;
2757 start = end;
2758 }
2759 Ok(last)
2760 }
2761
2762 fn ranks(&self) -> Option<usize> {
2763 (self.ranks > 0).then_some(self.ranks)
2764 }
2765
2766 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
2774 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
2775 if let Some(&answer) = memo.get(wanted) {
2776 return Ok(answer);
2777 }
2778 let answer = search_below(self, ranks, wanted)?;
2779 if memo.len() >= TEXT_SEARCH_MEMO {
2780 memo.clear();
2781 }
2782 memo.insert(wanted.to_vec(), answer);
2783 Ok(answer)
2784 }
2785
2786 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
2787 let settled = self.head_at(rank)?.cmp(&head(wanted));
2791 if settled != Ordering::Equal {
2792 return Ok(settled);
2793 }
2794 let code = self.code_at_rank(rank)?;
2795 let bytes = self
2796 .bytes_at(code as usize)?
2797 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
2798 Ok(bytes.cmp(wanted))
2799 }
2800
2801 fn code_at_rank(&self, rank: usize) -> Result<u32> {
2802 let (block, within) = self.rank_parts(rank)?;
2803 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
2804 let code = bitpack::tail_at(codes, self.code_bits, within)
2805 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
2806 let code = u32::try_from(code)
2807 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
2808 if code as usize >= self.len() {
2809 return Err(invalid("global dictionary order names a code it does not have"));
2810 }
2811 Ok(code)
2812 }
2813
2814 fn code_ranks(&self) -> Option<&[u32]> {
2815 if self.ranks == 0 || self.ranks != self.len() {
2819 return None;
2820 }
2821 self.code_ranks
2822 .get_or_init(|| {
2823 let mut ranks = vec![u32::MAX; self.ranks];
2824 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
2827 let (block, _) = self.rank_parts(first).ok()?;
2828 let count = self.rank_block_len(first);
2829 let codes = self.rank_codes(block, count).ok()?;
2830 for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
2831 .ok()?
2832 .into_iter()
2833 .enumerate()
2834 {
2835 let code = usize::try_from(code).ok()?;
2836 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
2837 }
2838 }
2839 if ranks.contains(&u32::MAX) {
2840 return None;
2841 }
2842 Some(ranks)
2843 })
2844 .as_deref()
2845 }
2846
2847 fn footprint(&self) -> usize {
2848 self.offsets.capacity()
2849 + self
2850 .code_ranks
2851 .get()
2852 .and_then(Option::as_ref)
2853 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
2854 + self.rank_hashes.capacity() * size_of::<u64>()
2855 + self.rank_ends.capacity() * size_of::<u64>()
2856 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2857 + self
2858 .rank_blocks
2859 .iter()
2860 .filter_map(OnceLock::get)
2861 .filter_map(|result| result.as_ref().ok())
2862 .map(Vec::capacity)
2863 .sum::<usize>()
2864 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2865 + self.hashes.capacity() * size_of::<u64>()
2866 + self.ends.capacity() * size_of::<u64>()
2867 + self
2868 .blocks
2869 .iter()
2870 .filter_map(OnceLock::get)
2871 .filter_map(|result| result.as_ref().ok())
2872 .map(Vec::capacity)
2873 .sum::<usize>()
2874 }
2875}
2876
2877fn places(table: &Table) -> Result<Vec<Place>> {
2879 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
2880 for (at, stripe) in table.stripes.iter().enumerate() {
2881 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
2882 for (part, &rows) in stripe.parts.iter().enumerate() {
2883 places.push(Place {
2884 stripe: index,
2885 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
2886 rows,
2887 });
2888 }
2889 }
2890 Ok(places)
2891}
2892
2893fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
2898 let parts = stripe.parts.len();
2899 let section = index_section(parts)?;
2900 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
2901 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
2902 if end > stripe.index.length as usize {
2903 return Err(invalid("index page is shorter than its columns"));
2904 }
2905 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
2906 let mut bytes = vec![0; section];
2907 let offset = stripe
2908 .index
2909 .offset
2910 .checked_add(at as u64)
2911 .ok_or_else(|| invalid("index page offset overflow"))?;
2912 read_at(file, offset, &mut bytes)?;
2913 let entries = section - size_of::<u64>();
2914 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
2915 if checksum(&bytes[..entries]) != stored {
2916 return Err(invalid(&format!(
2919 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
2920 wanted {stored:016x} and got {:016x}",
2921 checksum(&bytes[..entries]),
2922 )));
2923 }
2924 let mut spans = Vec::with_capacity(parts);
2925 let mut start = 0_usize;
2926 for part in 0..parts {
2927 let at = part * INDEX_ENTRY;
2928 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
2929 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
2930 spans.push(PartSpan { start, length, hash });
2931 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
2932 }
2933 if start != page.length as usize {
2934 return Err(invalid("column page length differs from its index"));
2935 }
2936 Ok(spans)
2937}
2938
2939fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
2941 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
2942 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
2943}
2944
2945fn remember(cached: &mut Cached, held: &CachedColumn, kept: usize) {
2950 if let Some(slot) = cached.index.get_mut(held.stripe) {
2951 if slot.is_none() {
2952 *slot = Some(Arc::clone(&held.index));
2953 }
2954 }
2955 let Some(page) = held.page.clone() else { return };
2956 let Some(slot) = cached.pages.get_mut(held.stripe) else { return };
2957 if slot.is_none() {
2958 cached.order.push_back(held.stripe);
2959 }
2960 *slot = Some(page);
2961 while cached.order.len() > kept.max(1) {
2962 let Some(oldest) = cached.order.pop_front() else { break };
2963 if let Some(slot) = cached.pages.get_mut(oldest) {
2964 *slot = None;
2965 }
2966 }
2967}
2968
2969#[derive(Debug, Clone)]
2978pub struct Catalog {
2979 file: Arc<File>,
2980 size: u64,
2981 entries: Arc<Vec<Entry>>,
2982 views: Arc<Vec<ViewEntry>>,
2984 opening: Opening,
2985}
2986
2987impl Catalog {
2988 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
2994 let (file, size, _, bytes, opening) = slot_bytes(path)?;
2995 let (entries, views) = decode_catalog(&bytes, size)?;
2996 Ok(Self {
2997 file: Arc::new(file),
2998 size,
2999 entries: Arc::new(entries),
3000 views: Arc::new(views),
3001 opening,
3002 })
3003 }
3004
3005 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
3007 self.entries.iter().map(|entry| entry.name.as_str())
3008 }
3009
3010 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
3017 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
3018 }
3019
3020 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
3026 self.views.iter()
3027 }
3028
3029 #[must_use]
3031 pub fn len(&self) -> usize {
3032 self.entries.len()
3033 }
3034
3035 #[must_use]
3038 pub fn is_empty(&self) -> bool {
3039 self.entries.is_empty()
3040 }
3041
3042 pub fn table(&self, name: &str) -> Result<Reader> {
3048 let entry = self
3049 .entries
3050 .iter()
3051 .find(|entry| entry.name == name)
3052 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
3053 let mut bytes = vec![0; entry.directory.length as usize];
3054 read_at(&self.file, entry.directory.offset, &mut bytes)?;
3055 if checksum(&bytes) != entry.directory.hash {
3056 return Err(invalid(&format!("the directory of table {name} does not checksum")));
3057 }
3058 let mut opening = self.opening;
3059 opening.reads += 1;
3060 opening.bytes += u64::from(entry.directory.length);
3061 Reader::build(
3062 Arc::clone(&self.file),
3063 self.size,
3064 decode_directory(&bytes, self.size)?,
3065 u64::from(entry.directory.length),
3066 opening,
3067 )
3068 }
3069}
3070
3071fn slot_offset(generation: u64) -> u64 {
3076 16 + (generation - 1) % 2 * SLOT_BYTES as u64
3077}
3078
3079fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
3084 let mut file = File::open(path).map_err(io)?;
3085 let size = file.metadata().map_err(io)?.len();
3086 if size < HEADER {
3087 return Err(invalid("file is shorter than its header"));
3088 }
3089 let mut header = [0; HEADER as usize];
3090 file.read_exact(&mut header).map_err(io)?;
3091 let mut opening = Opening { reads: 1, bytes: HEADER };
3092 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
3093 if &header[..8] != MAGIC {
3098 return Err(invalid("the header does not begin with a rudb native magic"));
3099 }
3100 if !READABLE.contains(&version) {
3101 return Err(invalid(&format!(
3102 "the file is format {version} and this build reads format {FORMAT}, so it has to \
3103 be written again"
3104 )));
3105 }
3106 let mut selected = None;
3107 for start in [16, 16 + SLOT_BYTES] {
3108 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
3109 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
3110 continue;
3111 }
3112 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
3113 if slot.offset < HEADER || end > size {
3114 continue;
3115 }
3116 let mut bytes = vec![0; slot.length as usize];
3117 file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
3118 file.read_exact(&mut bytes).map_err(io)?;
3119 opening.reads += 1;
3120 opening.bytes += u64::from(slot.length);
3121 if checksum(&bytes) == slot.hash
3122 && selected
3123 .as_ref()
3124 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
3125 {
3126 selected = Some((slot, bytes));
3127 }
3128 }
3129 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
3130 Ok((file, size, slot, bytes, opening))
3131}
3132
3133impl Reader {
3134 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
3141 let catalog = Catalog::open(path)?;
3142 let mut names = catalog.names();
3143 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
3144 if names.next().is_some() {
3145 return Err(invalid(
3146 "the file holds more than one table, so it has to be opened by name",
3147 ));
3148 }
3149 catalog.table(&name)
3150 }
3151
3152 fn build(
3154 file: Arc<File>,
3155 size: u64,
3156 table: Table,
3157 directory: u64,
3158 opening: Opening,
3159 ) -> Result<Self> {
3160 let places = places(&table)?;
3161 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
3162 let table_fields = table.fields.len();
3163 let stripes = table.stripes.len();
3164 let cache = (0..table.fields.len())
3165 .map(|_| {
3166 Mutex::new(Cached {
3167 pages: (0..stripes).map(|_| None).collect(),
3168 index: (0..stripes).map(|_| None).collect(),
3169 ..Cached::default()
3170 })
3171 })
3172 .collect::<Vec<_>>();
3173 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
3174 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3175 .collect();
3176 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
3177 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3178 .collect();
3179 Ok(Self {
3180 file,
3181 table: Arc::new(table),
3182 dictionaries: Arc::new(dictionaries),
3183 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
3184 opened: Arc::new(AtomicUsize::new(0)),
3185 sieves: Arc::new(sieves),
3186 part_ranges: Arc::new(part_ranges),
3187 places: Arc::new(places),
3188 cache: Arc::new(cache),
3189 pages: Arc::new(AtomicUsize::new(0)),
3190 indexes: Arc::new(AtomicUsize::new(0)),
3191 kept: Arc::new(AtomicUsize::new(CACHED_STRIPES_PER_COLUMN)),
3192 size,
3193 directory,
3194 opening,
3195 })
3196 }
3197
3198 #[must_use]
3205 pub fn reads(&self) -> Reads {
3206 Reads {
3207 opening: self.opening,
3208 pages: self.pages.load(Atomic::Relaxed),
3209 indexes: self.indexes.load(Atomic::Relaxed),
3210 dictionaries: self.opened.load(Atomic::Relaxed),
3211 }
3212 }
3213
3214 #[must_use]
3219 pub fn layout(&self) -> Layout {
3220 let table = &self.table;
3221 let stripes = table.stripes.as_slice();
3222 let columns = table
3223 .fields
3224 .iter()
3225 .enumerate()
3226 .map(|(at, field)| ColumnLayout {
3227 name: field.name.clone(),
3228 kind: field.ty.to_string(),
3229 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
3230 memberships: sum(stripes.iter().map(|stripe| page_bytes(&stripe.memberships, at))),
3231 sieves: sum(stripes.iter().map(|stripe| page_bytes(&stripe.sieves, at))),
3232 part_ranges: sum(stripes.iter().map(|stripe| page_bytes(&stripe.part_ranges, at))),
3233 dictionary: page_bytes(&table.dictionaries, at),
3234 })
3235 .collect();
3236 Layout {
3237 file: self.size,
3238 rows: table.rows,
3239 stripes: stripes.len(),
3240 parts: self.places.len(),
3241 columns,
3242 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
3243 directory: self.directory,
3244 header: HEADER,
3245 }
3246 }
3247
3248 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
3265 let field = self
3266 .table
3267 .fields
3268 .get(column)
3269 .ok_or_else(|| invalid("stored column index out of range"))?;
3270 let mut stored = Vec::with_capacity(self.places.len());
3271 let mut row = 0;
3272 for (at, stripe) in self.table.stripes.iter().enumerate() {
3273 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3274 let index = read_index(&self.file, stripe, column)?;
3275 let mut bytes = vec![0; page.length as usize];
3276 read_at(&self.file, page.offset, &mut bytes)?;
3277 let ranges = self.stripe_part_ranges(at, column);
3278 for (part, &rows) in stripe.parts.iter().enumerate() {
3279 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
3280 let held = part_bytes(&bytes, span)?;
3281 let range = ranges.and_then(|held| held.get(part));
3282 stored.push(StoredPart {
3283 stripe: at,
3284 part,
3285 row,
3286 rows: rows as usize,
3287 encoding: page_encoding(&field.ty, rows as usize, held),
3288 bytes: span.length as u64,
3289 page: page.offset,
3290 offset: span.start as u64,
3291 low: range
3292 .and_then(|range| range.low.clone())
3293 .and_then(|bound| bound.into_value(&field.ty)),
3294 high: range
3295 .and_then(|range| range.high.clone())
3296 .and_then(|bound| bound.into_value(&field.ty)),
3297 nulls: range.map(|range| range.nulls),
3298 });
3299 row += rows as usize;
3300 }
3301 }
3302 Ok(stored)
3303 }
3304
3305 #[must_use]
3307 pub fn parts(&self) -> usize {
3308 self.places.len()
3309 }
3310
3311 #[must_use]
3318 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
3319 let mut runs = Vec::with_capacity(self.table.stripes.len());
3320 let mut start = 0;
3321 for stripe in &self.table.stripes {
3322 let end = start + stripe.parts.len();
3323 runs.push(start..end);
3324 start = end;
3325 }
3326 runs
3327 }
3328
3329 #[must_use]
3334 pub fn stripe_rows(&self, stripe: usize) -> usize {
3335 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
3336 }
3337
3338 pub fn keep_stripes(&self, stripes: usize) {
3345 self.kept.fetch_max(stripes, Atomic::Relaxed);
3346 }
3347
3348 #[must_use]
3350 pub fn part_rows(&self, at: usize) -> usize {
3351 self.places.get(at).map_or(0, |place| place.rows as usize)
3352 }
3353
3354 #[must_use]
3356 pub fn table(&self) -> &Table {
3357 &self.table
3358 }
3359
3360 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
3369 let field = self
3370 .table
3371 .fields
3372 .get(column)
3373 .ok_or_else(|| invalid("frequency column index out of range"))?;
3374 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3375 return Ok(None);
3376 };
3377 if top == 0 || summary.entries.len() < top {
3378 return Ok(None);
3379 }
3380 let boundary = summary.entries[top - 1].count;
3381 if boundary <= summary.omitted_max {
3382 return Ok(None);
3383 }
3384 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
3385 }
3386
3387 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
3407 let Some(prefix) = self.frequency_prefix(column)? else {
3408 return Ok(None);
3409 };
3410 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
3411 }
3412
3413 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
3436 let field = self
3437 .table
3438 .fields
3439 .get(column)
3440 .ok_or_else(|| invalid("frequency column index out of range"))?;
3441 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3442 return Ok(None);
3443 };
3444 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
3445 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
3446 }
3447
3448 fn decode_frequencies(
3450 &self,
3451 column: usize,
3452 ty: &LogicalType,
3453 entries: &[FrequencyEntry],
3454 ) -> Result<Vec<(Value, u64)>> {
3455 let dictionary = if *ty == LogicalType::Varchar { self.dictionary(column)? } else { None };
3456 let mut out = Vec::with_capacity(entries.len());
3457 for entry in entries {
3458 let value = match entry.value {
3459 FrequencyValue::Null => Value::Null,
3460 FrequencyValue::Integer(value) => match *ty {
3461 LogicalType::TinyInt => Value::TinyInt(
3462 i8::try_from(value)
3463 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
3464 ),
3465 LogicalType::UTinyInt => Value::UTinyInt(
3466 u8::try_from(value)
3467 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
3468 ),
3469 LogicalType::USmallInt => Value::USmallInt(
3470 u16::try_from(value)
3471 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
3472 ),
3473 LogicalType::UInteger => Value::UInteger(
3474 u32::try_from(value)
3475 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
3476 ),
3477 LogicalType::UBigInt => Value::UBigInt(
3478 u64::try_from(value)
3479 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
3480 ),
3481 LogicalType::SmallInt => Value::SmallInt(
3482 i16::try_from(value)
3483 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
3484 ),
3485 LogicalType::Integer => Value::Integer(
3486 i32::try_from(value)
3487 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
3488 ),
3489 LogicalType::BigInt => Value::BigInt(
3490 i64::try_from(value)
3491 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
3492 ),
3493 LogicalType::Date => Value::Date(
3494 i32::try_from(value)
3495 .map_err(|_| invalid("frequency DATE is out of range"))?,
3496 ),
3497 LogicalType::Timestamp => Value::Timestamp(
3498 i64::try_from(value)
3499 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
3500 ),
3501 _ => return Err(invalid("integer frequency belongs to another type")),
3502 },
3503 FrequencyValue::Code(code) => dictionary
3504 .as_ref()
3505 .ok_or_else(|| invalid("frequency code has no dictionary"))?
3506 .try_value_at(code as usize)?,
3507 };
3508 out.push((value, entry.count));
3509 }
3510 Ok(out)
3511 }
3512
3513 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
3523 self.table
3524 .fields
3525 .get(column)
3526 .ok_or_else(|| invalid("frequency column index out of range"))?;
3527 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3528 return Ok(None);
3529 };
3530 if summary.ordinals.is_empty() {
3531 return Ok(None);
3532 }
3533 Ok(Some(FrequencyOccurrences {
3534 omitted_max: summary.omitted_max,
3535 ordinals: summary.ordinals.clone(),
3536 }))
3537 }
3538
3539 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
3563 self.table
3564 .distincts
3565 .get(column)
3566 .copied()
3567 .ok_or_else(|| invalid("distinct column index out of range"))
3568 }
3569
3570 pub fn null_count(&self, column: usize) -> Result<u64> {
3581 if column >= self.table.fields.len() {
3582 return Err(invalid("null count column index out of range"));
3583 }
3584 let mut nulls = 0_u64;
3585 for stripe in &self.table.stripes {
3586 let range = stripe
3587 .zone
3588 .column(column)
3589 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3590 nulls = nulls
3591 .checked_add(range.nulls as u64)
3592 .ok_or_else(|| invalid("null count overflow"))?;
3593 }
3594 Ok(nulls)
3595 }
3596
3597 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
3612 if self.null_count(column)? > 0 {
3613 return Ok(None);
3614 }
3615 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
3616 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
3617 if ranks == 0 {
3618 return Ok(None);
3619 }
3620 let low = text_at_rank(&dictionary, 0)?;
3621 let high = text_at_rank(&dictionary, ranks - 1)?;
3622 Ok(Some((low, high)))
3623 }
3624
3625 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
3648 if column >= self.table.fields.len() {
3649 return Err(invalid("extremes column index out of range"));
3650 }
3651 let mut low: Option<Bound> = None;
3652 let mut high: Option<Bound> = None;
3653 for stripe in &self.table.stripes {
3654 let range = stripe
3655 .zone
3656 .column(column)
3657 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3658 if !range.exact {
3659 return Ok(None);
3660 }
3661 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
3666 if stripe.rows > range.nulls {
3667 return Ok(None);
3668 }
3669 continue;
3670 };
3671 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
3672 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
3673 }
3674 Ok(low.zip(high))
3675 }
3676
3677 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
3690 if column >= self.table.fields.len() {
3691 return Err(invalid("sum column index out of range"));
3692 }
3693 let mut total = 0_i128;
3694 let mut rows = 0_u64;
3695 for stripe in &self.table.stripes {
3696 let range = stripe
3697 .zone
3698 .column(column)
3699 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3700 let Some(part) = range.sum else { return Ok(None) };
3701 let Some(sum) = total.checked_add(part) else { return Ok(None) };
3702 total = sum;
3703 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
3704 }
3705 Ok(Some((total, rows)))
3706 }
3707
3708 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
3717 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
3718 if let Some(dictionary) = self.dictionaries[column].get() {
3719 return Ok(Some(Arc::clone(dictionary)));
3720 }
3721 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
3722 if let Some(dictionary) = self.dictionaries[column].get() {
3723 return Ok(Some(Arc::clone(dictionary)));
3724 }
3725 self.opened.fetch_add(1, Atomic::Relaxed);
3726 let dictionary = Arc::new(open_global_dictionary(
3727 Arc::clone(&self.file),
3728 page,
3729 &self.table.fields[column].ty,
3730 TEXT_KEEP_BUDGET,
3731 )?);
3732 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
3733 Ok(Some(dictionary))
3734 }
3735
3736 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
3743 if of.extent_bytes == 0 {
3744 return Ok(Vec::new());
3745 }
3746 let mut bytes = vec![0; of.extent_bytes as usize];
3747 read_at(&self.file, of.extent_page, &mut bytes)?;
3748 if checksum(&bytes) != of.hash {
3749 return Err(invalid("a section's extent table does not checksum"));
3750 }
3751 let extents = section::decode_extents(&bytes)?;
3752 if extents.len() != of.extents as usize {
3753 return Err(invalid("a section's extent table is not the length the entry says"));
3754 }
3755 Ok(extents)
3756 }
3757
3758 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
3768 let end = of
3769 .offset
3770 .checked_add(u64::from(of.length))
3771 .ok_or_else(|| invalid("an extent overflows the file"))?;
3772 if of.offset < HEADER || end > self.size {
3773 return Err(invalid("an extent is outside the file"));
3774 }
3775 let mut bytes = vec![0; of.length as usize];
3776 read_at(&self.file, of.offset, &mut bytes)?;
3777 if checksum(&bytes) != of.hash {
3778 return Err(invalid("an extent does not checksum"));
3779 }
3780 Ok(bytes)
3781 }
3782
3783 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
3792 let extents = self.extents(of)?;
3793 let mut bytes =
3794 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
3795 for one in &extents {
3796 if one.first != bytes.len() as u64 {
3797 return Err(invalid("a section's extents do not join up"));
3798 }
3799 bytes.extend_from_slice(&self.extent(one)?);
3800 }
3801 if of.header_bytes as usize > bytes.len() {
3802 return Err(invalid("a section's header is longer than its payload"));
3803 }
3804 Ok(bytes)
3805 }
3806
3807 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3816 self.read_impl(part, columns, true)
3817 }
3818
3819 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3829 self.read_impl(part, columns, false)
3830 }
3831
3832 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
3839 if candidates.is_empty() {
3840 return Ok(true);
3841 }
3842 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
3843 return Err(Error::internal("native code candidates are not sorted and unique"));
3844 }
3845 let stripe = self.stripe_of(part)?;
3846 let Some(page) = stripe.memberships.get(column).copied().flatten() else {
3847 return Ok(false);
3848 };
3849 let mut bytes = vec![0; page.length as usize];
3850 read_at(&self.file, page.offset, &mut bytes)?;
3851 if checksum(&bytes) != page.hash {
3852 return Err(invalid("membership page checksum differs"));
3853 }
3854 let codes = decode_membership(&bytes)?;
3855 let mut left = 0;
3856 let mut right = 0;
3857 while left < codes.len() && right < candidates.len() {
3858 match codes[left].cmp(&candidates[right]) {
3859 Ordering::Less => left += 1,
3860 Ordering::Greater => right += 1,
3861 Ordering::Equal => return Ok(false),
3862 }
3863 }
3864 Ok(true)
3865 }
3866
3867 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
3868 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
3869 self.table
3870 .stripes
3871 .get(place.stripe as usize)
3872 .ok_or_else(|| invalid("stripe index out of range"))
3873 }
3874
3875 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
3892 let cache = self.cache.get(column).ok_or_else(|| invalid("column index out of range"))?;
3893 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3894 let known = cached.index.get(at).and_then(Clone::clone);
3895 let page = cached.pages.get(at).and_then(Clone::clone);
3896 if let Some(index) = known.clone() {
3897 if !whole || page.is_some() {
3898 return Ok(CachedColumn { stripe: at, index, page });
3899 }
3900 }
3901 if cached.loading.contains(&at) {
3902 drop(cached);
3903 if let Some(index) = known {
3907 return Ok(CachedColumn { stripe: at, index, page: None });
3908 }
3909 let held = self.page_of(stripe, column, at, false, None)?;
3910 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3911 remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3912 return Ok(held);
3913 }
3914 cached.loading.push(at);
3915 drop(cached);
3916
3917 let read = self.page_of(stripe, column, at, whole, known);
3918
3919 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3923 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
3924 cached.loading.remove(position);
3925 }
3926 let held = read?;
3927 remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3928 Ok(held)
3929 }
3930
3931 fn page_of(
3937 &self,
3938 stripe: &Stripe,
3939 column: usize,
3940 at: usize,
3941 whole: bool,
3942 known: Option<Arc<Vec<PartSpan>>>,
3943 ) -> Result<CachedColumn> {
3944 let index = match known {
3945 Some(index) => index,
3946 None => {
3947 self.indexes.fetch_add(1, Atomic::Relaxed);
3948 Arc::new(read_index(&self.file, stripe, column)?)
3949 }
3950 };
3951 let page = if whole {
3952 self.pages.fetch_add(1, Atomic::Relaxed);
3953 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3954 let mut bytes = vec![0; span.length as usize];
3955 read_at(&self.file, span.offset, &mut bytes)?;
3956 Some(Arc::new(bytes))
3957 } else {
3958 None
3959 };
3960 Ok(CachedColumn { stripe: at, index, page })
3961 }
3962
3963 fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
3964 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
3965 let index = place.stripe as usize;
3966 let stripe =
3967 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
3968 let rows = place.rows as usize;
3969 let mut picked = Vec::with_capacity(columns.len());
3970 for &column in columns {
3971 let field = self
3972 .table
3973 .fields
3974 .get(column)
3975 .ok_or_else(|| invalid("column index out of range"))?;
3976 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3977 let held = self.held(index, stripe, column, whole)?;
3978 let span = *held
3979 .index
3980 .get(place.part as usize)
3981 .ok_or_else(|| invalid("part index out of range"))?;
3982 let owned;
3983 let bytes = match &held.page {
3984 Some(held) => part_bytes(held, span)?,
3985 None => {
3986 let offset = page
3987 .offset
3988 .checked_add(span.start as u64)
3989 .ok_or_else(|| invalid("part range overflow"))?;
3990 let mut bytes = vec![0; span.length];
3991 read_at(&self.file, offset, &mut bytes)?;
3992 owned = bytes;
3993 &owned
3994 }
3995 };
3996 if checksum(bytes) != span.hash {
3997 return Err(invalid(&format!(
3998 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
3999 wanted {:016x} and got {:016x}",
4000 place.part,
4001 page.offset,
4002 span.start,
4003 span.length,
4004 span.hash,
4005 checksum(bytes),
4006 )));
4007 }
4008 let dictionary = self.dictionary(column)?;
4009 picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
4015 }
4016 Chunk::with_rows(picked, rows)
4017 }
4018
4019 #[must_use]
4035 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
4036 let Some(place) = self.places.get(part).copied() else { return false };
4037 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4038 if stripe.zone.skips(probes) {
4039 return true;
4040 }
4041 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
4042 }
4043
4044 fn outside(&self, place: Place, probe: &Probe) -> bool {
4050 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4051 Some(ranges) => ranges
4052 .get(place.part as usize)
4053 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
4054 None => false,
4055 }
4056 }
4057
4058 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
4064 let slot = self.part_ranges.get(column)?.get(stripe)?;
4065 if let Some(held) = slot.get() {
4066 return Some(held);
4067 }
4068 let page = self.table.stripes.get(stripe)?.part_ranges.get(column).copied().flatten()?;
4069 let mut bytes = vec![0; page.length as usize];
4070 read_at(&self.file, page.offset, &mut bytes).ok()?;
4071 if checksum(&bytes) != page.hash {
4072 return None;
4073 }
4074 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
4075 let _ = slot.set(ranges);
4076 slot.get().map(|held| held.as_slice())
4077 }
4078
4079 #[must_use]
4096 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
4097 let Some(place) = self.places.get(part).copied() else { return false };
4098 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4099 if stripe.zone.certain(probes) {
4100 return true;
4101 }
4102 probes
4103 .iter()
4104 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
4105 }
4106
4107 fn inside(&self, place: Place, probe: &Probe) -> bool {
4113 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4114 Some(ranges) => ranges
4115 .get(place.part as usize)
4116 .is_some_and(|range| range.certain(probe.op, &probe.value)),
4117 None => false,
4118 }
4119 }
4120
4121 #[must_use]
4132 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
4133 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
4134 }
4135
4136 fn sifted(&self, place: Place, probe: &Probe) -> bool {
4142 if probe.op != Op::Equal {
4143 return false;
4144 }
4145 match self.stripe_sieves(place.stripe as usize, probe.column) {
4146 Some(sieves) => sieves
4147 .get(place.part as usize)
4148 .and_then(Option::as_ref)
4149 .is_some_and(|sieve| sieve.excludes(&probe.value)),
4150 None => false,
4151 }
4152 }
4153
4154 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
4161 let slot = self.sieves.get(column)?.get(stripe)?;
4162 if let Some(held) = slot.get() {
4163 return Some(held);
4164 }
4165 let page = self.table.stripes.get(stripe)?.sieves.get(column).copied().flatten()?;
4166 let mut bytes = vec![0; page.length as usize];
4167 read_at(&self.file, page.offset, &mut bytes).ok()?;
4168 if checksum(&bytes) != page.hash {
4169 return None;
4170 }
4171 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
4172 let _ = slot.set(sieves);
4173 slot.get().map(|held| held.as_slice())
4174 }
4175}
4176
4177fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
4179 let code = dictionary.code_at_rank(rank)? as usize;
4180 let text = dictionary
4181 .try_text_at(code)?
4182 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4183 Ok(Value::Varchar(text.into()))
4184}
4185
4186#[cfg(unix)]
4191fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4192 use std::os::unix::fs::FileExt;
4193 while !bytes.is_empty() {
4194 let written = file.write_at(bytes, offset).map_err(io)?;
4195 if written == 0 {
4196 return Err(invalid("a write to the native file wrote nothing"));
4197 }
4198 offset += written as u64;
4199 bytes = &bytes[written..];
4200 }
4201 Ok(())
4202}
4203
4204#[cfg(windows)]
4206fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4207 use std::os::windows::fs::FileExt;
4208 while !bytes.is_empty() {
4209 let written = file.seek_write(bytes, offset).map_err(io)?;
4210 if written == 0 {
4211 return Err(invalid("a write to the native file wrote nothing"));
4212 }
4213 offset += written as u64;
4214 bytes = &bytes[written..];
4215 }
4216 Ok(())
4217}
4218
4219#[cfg(not(any(unix, windows)))]
4221fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
4222 use std::io::Write;
4223 let mut file = file.try_clone().map_err(io)?;
4224 file.seek(SeekFrom::Start(offset)).map_err(io)?;
4225 file.write_all(bytes).map_err(io)
4226}
4227
4228#[cfg(unix)]
4238fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4239 use std::os::unix::fs::FileExt;
4240 while !bytes.is_empty() {
4241 let read = file.read_at(bytes, offset).map_err(io)?;
4242 if read == 0 {
4243 return Err(invalid("column page ends before its declared length"));
4244 }
4245 offset += read as u64;
4246 bytes = &mut bytes[read..];
4247 }
4248 Ok(())
4249}
4250
4251#[cfg(windows)]
4257fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4258 use std::os::windows::fs::FileExt;
4259 while !bytes.is_empty() {
4260 let read = file.seek_read(bytes, offset).map_err(io)?;
4261 if read == 0 {
4262 return Err(invalid("column page ends before its declared length"));
4263 }
4264 offset += read as u64;
4265 bytes = &mut bytes[read..];
4266 }
4267 Ok(())
4268}
4269
4270#[cfg(not(any(unix, windows)))]
4275fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
4276 let mut file = file.try_clone().map_err(io)?;
4277 file.seek(SeekFrom::Start(offset)).map_err(io)?;
4278 file.read_exact(bytes).map_err(io)
4279}
4280
4281fn type_tag(ty: &LogicalType) -> Result<u8> {
4288 match ty {
4289 LogicalType::SmallInt => Ok(1),
4290 LogicalType::Integer => Ok(2),
4291 LogicalType::BigInt => Ok(3),
4292 LogicalType::Varchar => Ok(4),
4293 LogicalType::Date => Ok(5),
4294 LogicalType::Timestamp => Ok(6),
4295 LogicalType::Boolean => Ok(7),
4296 LogicalType::TinyInt => Ok(8),
4297 LogicalType::UTinyInt => Ok(9),
4298 LogicalType::USmallInt => Ok(10),
4299 LogicalType::UInteger => Ok(11),
4300 LogicalType::UBigInt => Ok(12),
4301 LogicalType::Decimal { .. } => Ok(13),
4302 LogicalType::Float => Ok(14),
4303 LogicalType::Double => Ok(15),
4304 LogicalType::HugeInt => Ok(16),
4305 LogicalType::UHugeInt => Ok(17),
4306 LogicalType::Time => Ok(18),
4307 LogicalType::TimeTz => Ok(19),
4308 LogicalType::TimestampTz => Ok(20),
4309 LogicalType::Interval => Ok(21),
4310 LogicalType::Uuid => Ok(22),
4311 LogicalType::Blob => Ok(23),
4312 LogicalType::Bit => Ok(24),
4313 LogicalType::TimestampS => Ok(25),
4314 LogicalType::TimestampMs => Ok(26),
4315 LogicalType::TimestampNs => Ok(27),
4316 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
4317 }
4318}
4319
4320fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
4326 out.push(type_tag(ty)?);
4327 if let LogicalType::Decimal { width, scale } = ty {
4328 out.push(*width);
4329 out.push(*scale);
4330 }
4331 Ok(())
4332}
4333
4334fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
4336 let tag = cur.u8()?;
4337 if tag == 13 {
4338 let width = cur.u8()?;
4339 let scale = cur.u8()?;
4340 return LogicalType::decimal(width, scale)
4341 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
4342 }
4343 tag_type(tag)
4344}
4345
4346fn tag_type(tag: u8) -> Result<LogicalType> {
4347 match tag {
4348 1 => Ok(LogicalType::SmallInt),
4349 2 => Ok(LogicalType::Integer),
4350 3 => Ok(LogicalType::BigInt),
4351 4 => Ok(LogicalType::Varchar),
4352 5 => Ok(LogicalType::Date),
4353 6 => Ok(LogicalType::Timestamp),
4354 7 => Ok(LogicalType::Boolean),
4355 8 => Ok(LogicalType::TinyInt),
4356 9 => Ok(LogicalType::UTinyInt),
4357 10 => Ok(LogicalType::USmallInt),
4358 11 => Ok(LogicalType::UInteger),
4359 12 => Ok(LogicalType::UBigInt),
4360 14 => Ok(LogicalType::Float),
4361 15 => Ok(LogicalType::Double),
4362 16 => Ok(LogicalType::HugeInt),
4363 17 => Ok(LogicalType::UHugeInt),
4364 18 => Ok(LogicalType::Time),
4365 19 => Ok(LogicalType::TimeTz),
4366 20 => Ok(LogicalType::TimestampTz),
4367 21 => Ok(LogicalType::Interval),
4368 22 => Ok(LogicalType::Uuid),
4369 23 => Ok(LogicalType::Blob),
4370 24 => Ok(LogicalType::Bit),
4371 25 => Ok(LogicalType::TimestampS),
4372 26 => Ok(LogicalType::TimestampMs),
4373 27 => Ok(LogicalType::TimestampNs),
4374 _ => Err(invalid("column type tag is unknown")),
4375 }
4376}
4377
4378fn put_u16(out: &mut Vec<u8>, value: u16) {
4379 out.extend_from_slice(&value.to_le_bytes());
4380}
4381fn put_u32(out: &mut Vec<u8>, value: u32) {
4382 out.extend_from_slice(&value.to_le_bytes());
4383}
4384fn put_u64(out: &mut Vec<u8>, value: u64) {
4385 out.extend_from_slice(&value.to_le_bytes());
4386}
4387fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
4388 while value >= 0x80 {
4389 out.push((value as u8 & 0x7f) | 0x80);
4390 value >>= 7;
4391 }
4392 out.push(value as u8);
4393}
4394
4395fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
4396 match (left, right) {
4397 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
4398 (FrequencyValue::Null, _) => Ordering::Less,
4399 (_, FrequencyValue::Null) => Ordering::Greater,
4400 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
4401 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
4402 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
4403 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
4404 }
4405}
4406
4407fn code_frequency(dictionary: &GlobalDictionary) -> FrequencySummary {
4408 let mut entries = dictionary
4409 .counts
4410 .iter()
4411 .enumerate()
4412 .filter(|(_, count)| **count != 0)
4413 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
4414 .collect::<Vec<_>>();
4415 if dictionary.nulls != 0 {
4416 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
4417 }
4418 entries.sort_unstable_by(|left, right| {
4419 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
4420 });
4421 let omitted_max = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
4422 entries.truncate(FREQUENCY_ENTRIES);
4423 FrequencySummary { entries, omitted_max, ordinals: Vec::new() }
4424}
4425
4426fn encode_directory(table: &Table) -> Result<Vec<u8>> {
4427 let mut out = DIRECTORY.to_vec();
4428 let name = table.name.as_bytes();
4429 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4430 out.extend_from_slice(name);
4431 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
4432 for field in &table.fields {
4433 let name = field.name.as_bytes();
4434 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
4435 out.extend_from_slice(name);
4436 put_type(&mut out, &field.ty)?;
4437 out.push(u8::from(field.not_null));
4438 }
4439 for dictionary in &table.dictionaries {
4440 match dictionary {
4441 None => out.push(0),
4442 Some(page) => {
4443 out.push(1);
4444 put_u64(&mut out, page.offset);
4445 put_u32(&mut out, page.length);
4446 put_u64(&mut out, page.hash);
4447 }
4448 }
4449 }
4450 for distinct in &table.distincts {
4451 match distinct {
4452 None => out.push(0),
4453 Some(count) => {
4454 out.push(1);
4455 put_u64(&mut out, *count);
4456 }
4457 }
4458 }
4459 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
4460 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
4461 for stripe in &table.stripes {
4462 put_u32(
4463 &mut out,
4464 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
4465 );
4466 for &rows in &stripe.parts {
4467 put_u32(&mut out, rows);
4468 }
4469 put_u64(&mut out, stripe.index.offset);
4470 put_u32(&mut out, stripe.index.length);
4471 for page in &stripe.pages {
4472 put_u64(&mut out, page.offset);
4473 put_u32(&mut out, page.length);
4474 }
4475 for ((field, dictionary), membership) in
4480 table.fields.iter().zip(&table.dictionaries).zip(&stripe.memberships)
4481 {
4482 if field.ty != LogicalType::Varchar || dictionary.is_none() {
4483 continue;
4484 }
4485 let page =
4486 membership.ok_or_else(|| invalid("string page has no code membership index"))?;
4487 put_u64(&mut out, page.offset);
4488 put_u32(&mut out, page.length);
4489 put_u64(&mut out, page.hash);
4490 }
4491 for sieve in &stripe.sieves {
4492 match sieve {
4493 None => out.push(0),
4494 Some(page) => {
4495 out.push(1);
4496 put_u64(&mut out, page.offset);
4497 put_u32(&mut out, page.length);
4498 put_u64(&mut out, page.hash);
4499 }
4500 }
4501 }
4502 for held in &stripe.part_ranges {
4503 match held {
4504 None => out.push(0),
4505 Some(page) => {
4506 out.push(1);
4507 put_u64(&mut out, page.offset);
4508 put_u32(&mut out, page.length);
4509 put_u64(&mut out, page.hash);
4510 }
4511 }
4512 }
4513 for range in stripe.zone.columns() {
4514 put_bound(&mut out, range.low.as_ref())?;
4515 put_bound(&mut out, range.high.as_ref())?;
4516 put_u32(
4517 &mut out,
4518 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
4519 );
4520 out.push(u8::from(range.exact));
4521 match range.sum {
4522 None => out.push(0),
4523 Some(total) => {
4524 out.push(1);
4525 out.extend_from_slice(&total.to_le_bytes());
4526 }
4527 }
4528 }
4529 }
4530 out.extend_from_slice(FREQUENCIES);
4531 put_u16(
4532 &mut out,
4533 u16::try_from(table.frequencies.len())
4534 .map_err(|_| invalid("too many frequency columns"))?,
4535 );
4536 for summary in &table.frequencies {
4537 let Some(summary) = summary else {
4538 out.push(0);
4539 continue;
4540 };
4541 out.push(1);
4542 put_u64(&mut out, summary.omitted_max);
4543 put_u32(
4544 &mut out,
4545 u32::try_from(summary.entries.len())
4546 .map_err(|_| invalid("too many frequency entries"))?,
4547 );
4548 for entry in &summary.entries {
4549 match entry.value {
4550 FrequencyValue::Null => out.push(0),
4551 FrequencyValue::Integer(value) => {
4552 out.push(1);
4553 out.extend_from_slice(&value.to_le_bytes());
4554 }
4555 FrequencyValue::Code(value) => {
4556 out.push(2);
4557 put_u32(&mut out, value);
4558 }
4559 }
4560 put_u64(&mut out, entry.count);
4561 }
4562 put_u32(
4563 &mut out,
4564 u32::try_from(summary.ordinals.len())
4565 .map_err(|_| invalid("too many frequency ordinals"))?,
4566 );
4567 let mut previous = 0_u64;
4568 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
4569 let delta = if at == 0 {
4570 ordinal
4571 } else {
4572 ordinal
4573 .checked_sub(previous)
4574 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
4575 };
4576 if at != 0 && delta == 0 {
4577 return Err(invalid("frequency ordinals are not unique"));
4578 }
4579 put_var_u64(&mut out, delta);
4580 previous = ordinal;
4581 }
4582 }
4583 if let Some(clustering) = &table.clustering {
4586 out.extend_from_slice(CLUSTERING);
4587 out.push(clustering.width().tag());
4588 put_u16(
4589 &mut out,
4590 u16::try_from(clustering.columns().len())
4591 .map_err(|_| invalid("too many clustering columns"))?,
4592 );
4593 for &column in clustering.columns() {
4594 put_u16(
4595 &mut out,
4596 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
4597 );
4598 }
4599 }
4600 out.extend_from_slice(SECTIONS);
4606 put_u64(&mut out, table.generation);
4607 put_u16(
4608 &mut out,
4609 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
4610 );
4611 for held in &table.sections {
4612 held.encode(&mut out)?;
4613 }
4614 Ok(out)
4615}
4616
4617fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
4626 let mut out = CATALOG.to_vec();
4627 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
4628 for entry in entries {
4629 let name = entry.name.as_bytes();
4630 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4631 out.extend_from_slice(name);
4632 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
4633 put_u16(
4634 &mut out,
4635 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
4636 );
4637 for field in &entry.fields {
4638 let name = field.name.as_bytes();
4639 put_u16(
4640 &mut out,
4641 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4642 );
4643 out.extend_from_slice(name);
4644 put_type(&mut out, &field.ty)?;
4645 out.push(u8::from(field.not_null));
4646 }
4647 put_u64(&mut out, entry.directory.offset);
4648 put_u32(&mut out, entry.directory.length);
4649 put_u64(&mut out, entry.directory.hash);
4650 }
4651 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
4652 for view in views {
4653 let name = view.name.as_bytes();
4654 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
4655 out.extend_from_slice(name);
4656 put_long_text(&mut out, &view.sql, "view body")?;
4657 put_long_text(&mut out, &view.statement, "view statement")?;
4658 put_u16(
4659 &mut out,
4660 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
4661 );
4662 for alias in &view.aliases {
4663 let alias = alias.as_bytes();
4664 put_u16(
4665 &mut out,
4666 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
4667 );
4668 out.extend_from_slice(alias);
4669 }
4670 put_u16(
4671 &mut out,
4672 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
4673 );
4674 for field in &view.columns {
4675 let name = field.name.as_bytes();
4676 put_u16(
4677 &mut out,
4678 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4679 );
4680 out.extend_from_slice(name);
4681 put_type(&mut out, &field.ty)?;
4682 out.push(u8::from(field.not_null));
4683 }
4684 }
4685 Ok(out)
4686}
4687
4688fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
4690 let bytes = text.as_bytes();
4691 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
4692 out.extend_from_slice(bytes);
4693 Ok(())
4694}
4695
4696fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
4699 let mut cur = Cursor { bytes, at: 0 };
4700 if cur.take(8)? != CATALOG {
4701 return Err(invalid("catalog magic differs"));
4702 }
4703 let count = cur.u32()? as usize;
4704 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
4705 for _ in 0..count {
4706 let name = cur.text()?;
4707 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4708 let width = cur.u16()? as usize;
4709 let mut fields = Vec::with_capacity(width);
4710 for _ in 0..width {
4711 let name = cur.text()?;
4712 let ty = read_type(&mut cur)?;
4713 let not_null = match cur.u8()? {
4714 0 => false,
4715 1 => true,
4716 _ => return Err(invalid("nullability flag differs")),
4717 };
4718 fields.push(Field { name, ty, not_null });
4719 }
4720 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4721 let end = directory
4722 .offset
4723 .checked_add(u64::from(directory.length))
4724 .ok_or_else(|| invalid("table directory offset overflow"))?;
4725 if directory.offset < HEADER
4726 || end > size
4727 || directory.length as usize > MAX_DIRECTORY
4728 || directory.length == 0
4729 {
4730 return Err(invalid("table directory range is outside the file"));
4731 }
4732 if entries.iter().any(|held| held.name == name) {
4733 return Err(invalid("two tables in the catalog have the same name"));
4734 }
4735 entries.push(Entry { name, fields, rows, directory });
4736 }
4737 let count = if cur.done() { 0 } else { cur.u32()? as usize };
4742 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
4743 for _ in 0..count {
4744 let name = cur.text()?;
4745 let sql = cur.long_text()?;
4746 let statement = cur.long_text()?;
4747 let width = cur.u16()? as usize;
4748 let mut aliases = Vec::with_capacity(width);
4749 for _ in 0..width {
4750 aliases.push(cur.text()?);
4751 }
4752 let width = cur.u16()? as usize;
4753 let mut columns = Vec::with_capacity(width);
4754 for _ in 0..width {
4755 let name = cur.text()?;
4756 let ty = read_type(&mut cur)?;
4757 let not_null = match cur.u8()? {
4758 0 => false,
4759 1 => true,
4760 _ => return Err(invalid("nullability flag differs")),
4761 };
4762 columns.push(Field { name, ty, not_null });
4763 }
4764 if views.iter().any(|held| held.name == name) {
4768 return Err(invalid("two views in the catalog have the same name"));
4769 }
4770 if entries.iter().any(|held| held.name == name) {
4771 return Err(invalid("a table and a view in the catalog have the same name"));
4772 }
4773 views.push(ViewEntry { name, sql, statement, aliases, columns });
4774 }
4775 Ok((entries, views))
4776}
4777
4778struct Cursor<'a> {
4779 bytes: &'a [u8],
4780 at: usize,
4781}
4782impl<'a> Cursor<'a> {
4783 fn take(&mut self, len: usize) -> Result<&'a [u8]> {
4784 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
4785 let bytes =
4786 self.bytes.get(self.at..end).ok_or_else(|| invalid("directory is truncated"))?;
4787 self.at = end;
4788 Ok(bytes)
4789 }
4790 fn u8(&mut self) -> Result<u8> {
4791 Ok(self.take(1)?[0])
4792 }
4793 fn u16(&mut self) -> Result<u16> {
4794 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
4795 }
4796 fn u32(&mut self) -> Result<u32> {
4797 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
4798 }
4799 fn u64(&mut self) -> Result<u64> {
4800 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
4801 }
4802 fn var_u64(&mut self) -> Result<u64> {
4803 let mut value = 0_u64;
4804 for shift in (0..=63).step_by(7) {
4805 let byte = self.u8()?;
4806 let part = u64::from(byte & 0x7f);
4807 if shift == 63 && part > 1 {
4808 return Err(invalid("frequency ordinal varint overflows"));
4809 }
4810 value |= part << shift;
4811 if byte & 0x80 == 0 {
4812 return Ok(value);
4813 }
4814 }
4815 Err(invalid("frequency ordinal varint is too long"))
4816 }
4817 fn bound(&mut self) -> Result<Option<Bound>> {
4823 bounds::get(self.bytes, &mut self.at)
4824 }
4825 fn text(&mut self) -> Result<String> {
4826 let len = self.u16()? as usize;
4827 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
4828 }
4829 fn done(&self) -> bool {
4832 self.at >= self.bytes.len()
4833 }
4834 fn long_text(&mut self) -> Result<String> {
4841 let len = self.u32()? as usize;
4842 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
4843 }
4844}
4845
4846fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
4847 let mut cur = Cursor { bytes, at: 0 };
4848 if cur.take(8)? != DIRECTORY {
4849 return Err(invalid("directory magic differs"));
4850 }
4851 let name = cur.text()?;
4852 let width = cur.u16()? as usize;
4853 let mut fields = Vec::with_capacity(width);
4854 for _ in 0..width {
4855 let name = cur.text()?;
4856 let ty = read_type(&mut cur)?;
4857 let not_null = match cur.u8()? {
4858 0 => false,
4859 1 => true,
4860 _ => return Err(invalid("nullability flag differs")),
4861 };
4862 fields.push(Field { name, ty, not_null });
4863 }
4864 let mut dictionaries = Vec::with_capacity(width);
4865 for _ in 0..width {
4866 dictionaries.push(match cur.u8()? {
4867 0 => None,
4868 1 => {
4869 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4870 let end = page
4871 .offset
4872 .checked_add(u64::from(page.length))
4873 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
4874 if page.offset < HEADER || end > size {
4879 return Err(invalid("dictionary page range is outside the file"));
4880 }
4881 Some(page)
4882 }
4883 _ => return Err(invalid("dictionary page tag differs")),
4884 });
4885 }
4886 let mut distincts = Vec::with_capacity(width);
4887 for _ in 0..width {
4888 distincts.push(match cur.u8()? {
4889 0 => None,
4890 1 => Some(cur.u64()?),
4891 _ => return Err(invalid("distinct count tag differs")),
4892 });
4893 }
4894 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4895 let count = cur.u32()? as usize;
4896 let mut stripes = Vec::with_capacity(count);
4897 let mut total = 0_usize;
4898 for _ in 0..count {
4899 let count = cur.u32()? as usize;
4900 if count == 0 || count > STRIPE_PARTS {
4901 return Err(invalid("stripe part count is outside its bound"));
4902 }
4903 let mut parts = Vec::with_capacity(count);
4904 let mut stripe_rows = 0_usize;
4905 for _ in 0..count {
4906 let rows = cur.u32()?;
4907 if rows == 0 {
4908 return Err(invalid("empty part"));
4909 }
4910 parts.push(rows);
4911 stripe_rows = stripe_rows
4912 .checked_add(rows as usize)
4913 .ok_or_else(|| invalid("stripe row count overflow"))?;
4914 }
4915 total =
4916 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
4917 let index = Span { offset: cur.u64()?, length: cur.u32()? };
4918 let section = index_section(count)?;
4919 let wanted = section
4920 .checked_mul(width)
4921 .and_then(|bytes| u32::try_from(bytes).ok())
4922 .ok_or_else(|| invalid("index page length overflow"))?;
4923 let end = index
4924 .offset
4925 .checked_add(u64::from(index.length))
4926 .ok_or_else(|| invalid("index page offset overflow"))?;
4927 if index.offset < HEADER || end > size || index.length != wanted {
4928 return Err(invalid("index page range is outside the file"));
4929 }
4930 let mut pages = Vec::with_capacity(width);
4931 for _ in 0..width {
4932 let offset = cur.u64()?;
4933 let length = cur.u32()?;
4934 let end = offset
4935 .checked_add(u64::from(length))
4936 .ok_or_else(|| invalid("page offset overflow"))?;
4937 if offset < HEADER || end > size || length as usize > MAX_PAGE {
4938 return Err(invalid("page range is outside the file"));
4939 }
4940 pages.push(Span { offset, length });
4941 }
4942 let mut memberships = vec![None; width];
4943 for (column, field) in fields.iter().enumerate() {
4944 if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
4945 continue;
4946 }
4947 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4948 let end = page
4949 .offset
4950 .checked_add(u64::from(page.length))
4951 .ok_or_else(|| invalid("membership page offset overflow"))?;
4952 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4953 return Err(invalid("membership page range is outside the file"));
4954 }
4955 memberships[column] = Some(page);
4956 }
4957 let mut sieves = vec![None; width];
4958 for sieve in sieves.iter_mut().take(width) {
4959 match cur.u8()? {
4960 0 => continue,
4961 1 => {}
4962 _ => return Err(invalid("a sieve page has an unknown tag")),
4963 }
4964 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4965 let end = page
4966 .offset
4967 .checked_add(u64::from(page.length))
4968 .ok_or_else(|| invalid("sieve page offset overflow"))?;
4969 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4970 return Err(invalid("sieve page range is outside the file"));
4971 }
4972 *sieve = Some(page);
4973 }
4974 let mut part_ranges = vec![None; width];
4975 for held in part_ranges.iter_mut().take(width) {
4976 match cur.u8()? {
4977 0 => continue,
4978 1 => {}
4979 _ => return Err(invalid("a part range page has an unknown tag")),
4980 }
4981 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4982 let end = page
4983 .offset
4984 .checked_add(u64::from(page.length))
4985 .ok_or_else(|| invalid("part range page offset overflow"))?;
4986 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4987 return Err(invalid("part range page range is outside the file"));
4988 }
4989 *held = Some(page);
4990 }
4991 let mut ranges = Vec::with_capacity(width);
4992 for column in 0..width {
4993 let low = cur.bound()?;
4994 let high = cur.bound()?;
4995 let nulls = cur.u32()? as usize;
4996 if nulls > stripe_rows {
4997 return Err(invalid("null count exceeds stripe rows"));
4998 }
4999 let exact = cur.u8()? != 0;
5000 let sum = match cur.u8()? {
5001 0 => None,
5002 1 => Some(i128::from_le_bytes(
5003 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
5004 )),
5005 _ => return Err(invalid("a stripe sum has an unknown tag")),
5006 };
5007 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
5013 let low = low.map(|bound| scaled_as(bound, ty));
5014 let high = high.map(|bound| scaled_as(bound, ty));
5015 ranges.push(Range { low, high, nulls, exact, sum });
5016 }
5017 stripes.push(Stripe {
5018 rows: stripe_rows,
5019 parts,
5020 index,
5021 pages,
5022 memberships,
5023 sieves,
5024 part_ranges,
5025 zone: Zone::from_ranges(ranges),
5026 });
5027 }
5028 if total != rows {
5029 return Err(invalid("table row count differs from stripes"));
5030 }
5031 let frequencies = if cur.at == bytes.len() {
5032 vec![None; width]
5033 } else {
5034 if cur.take(8)? != FREQUENCIES {
5035 return Err(invalid("directory extension magic differs"));
5036 }
5037 if cur.u16()? as usize != width {
5038 return Err(invalid("frequency column count differs"));
5039 }
5040 let mut frequencies = Vec::with_capacity(width);
5041 for field in &fields {
5042 let summary = match cur.u8()? {
5043 0 => None,
5044 1 => {
5045 let omitted_max = cur.u64()?;
5046 let count = cur.u32()? as usize;
5047 if count > FREQUENCY_ENTRIES {
5048 return Err(invalid("frequency entry count exceeds its bound"));
5049 }
5050 let mut entries = Vec::with_capacity(count);
5051 for _ in 0..count {
5053 let value = match cur.u8()? {
5054 0 => FrequencyValue::Null,
5055 1 => FrequencyValue::Integer(i128::from_le_bytes(
5056 cur.take(16)?.try_into().expect("sixteen bytes"),
5057 )),
5058 2 => FrequencyValue::Code(cur.u32()?),
5059 _ => return Err(invalid("frequency value tag differs")),
5060 };
5061 let valid = matches!(
5062 (&field.ty, value),
5063 (_, FrequencyValue::Null)
5064 | (LogicalType::Varchar, FrequencyValue::Code(_))
5065 | (
5066 LogicalType::TinyInt
5067 | LogicalType::SmallInt
5068 | LogicalType::Integer
5069 | LogicalType::BigInt
5070 | LogicalType::UTinyInt
5071 | LogicalType::USmallInt
5072 | LogicalType::UInteger
5073 | LogicalType::UBigInt
5074 | LogicalType::Date
5075 | LogicalType::Timestamp,
5076 FrequencyValue::Integer(_),
5077 )
5078 );
5079 if !valid {
5080 return Err(invalid("frequency value does not match its column"));
5081 }
5082 let count = cur.u64()?;
5083 if count == 0 || count > rows as u64 {
5084 return Err(invalid("frequency count is outside the table"));
5085 }
5086 entries.push(FrequencyEntry { value, count });
5087 }
5088 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
5089 return Err(invalid("frequency entries are not descending"));
5090 }
5091 let ordinals = {
5092 let ordinal_count = cur.u32()? as usize;
5093 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
5094 return Err(invalid("frequency ordinal count exceeds its bound"));
5095 }
5096 let mut ordinals = Vec::with_capacity(ordinal_count);
5097 let mut previous = 0_u64;
5098 for at in 0..ordinal_count {
5099 let delta = cur.var_u64()?;
5100 if at != 0 && delta == 0 {
5101 return Err(invalid("frequency ordinals are not increasing"));
5102 }
5103 let ordinal = if at == 0 {
5104 delta
5105 } else {
5106 previous
5107 .checked_add(delta)
5108 .ok_or_else(|| invalid("frequency ordinal overflows"))?
5109 };
5110 if ordinal >= rows as u64 {
5111 return Err(invalid("frequency ordinal is outside the table"));
5112 }
5113 ordinals.push(ordinal);
5114 previous = ordinal;
5115 }
5116 ordinals
5117 };
5118 Some(FrequencySummary { entries, omitted_max, ordinals })
5119 }
5120 _ => return Err(invalid("frequency summary tag differs")),
5121 };
5122 frequencies.push(summary);
5123 }
5124 frequencies
5125 };
5126 let mut clustering = None;
5136 let mut sections = Vec::new();
5137 let mut seen_sections = false;
5138 let mut generation = 0;
5141 while cur.at != bytes.len() {
5142 let mut tag = [0u8; 8];
5143 tag.copy_from_slice(cur.take(8)?);
5144 if &tag == CLUSTERING {
5145 if clustering.is_some() {
5146 return Err(invalid("directory names two clustering declarations"));
5147 }
5148 let bucket = Width::from_tag(cur.u8()?)
5149 .ok_or_else(|| invalid("clustering width tag differs"))?;
5150 let count = cur.u16()? as usize;
5151 let mut columns = Vec::with_capacity(count.min(fields.len()));
5152 for _ in 0..count {
5153 columns.push(u32::from(cur.u16()?));
5154 }
5155 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
5158 invalid("stored clustering declaration does not match the table it is on")
5159 })?);
5160 } else if &tag == SECTIONS {
5161 if seen_sections {
5162 return Err(invalid("directory names two section tables"));
5163 }
5164 seen_sections = true;
5165 generation = cur.u64()?;
5166 let count = cur.u16()? as usize;
5167 if count > MAX_SECTIONS {
5168 return Err(invalid("section count exceeds its bound"));
5169 }
5170 sections = Vec::with_capacity(count);
5171 for _ in 0..count {
5174 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
5175 }
5176 for held in §ions {
5177 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
5178 return Err(invalid("a section's extent table overflows the file"));
5179 };
5180 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
5184 return Err(invalid("a section's extent table is outside the file"));
5185 }
5186 if held.extents == 0 && held.extent_bytes != 0 {
5187 return Err(invalid("a section with no extents names an extent table"));
5188 }
5189 }
5190 } else {
5191 return Err(invalid("directory extension magic differs"));
5192 }
5193 }
5194 if cur.at != bytes.len() {
5195 return Err(invalid("directory has trailing bytes"));
5196 }
5197 Ok(Table {
5198 name,
5199 fields,
5200 stripes,
5201 rows,
5202 dictionaries,
5203 distincts,
5204 frequencies,
5205 clustering,
5206 generation,
5207 sections,
5208 })
5209}
5210
5211fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
5213 bounds::put(out, bound)
5214}
5215
5216#[derive(Debug)]
5233struct Codes;
5234
5235impl chooser::Chooser for Codes {
5236 fn name(&self) -> &'static str {
5237 "codes"
5238 }
5239
5240 fn narrow_strings(
5241 &self,
5242 _values: &[&[u8]],
5243 offered: &[string::Kind],
5244 _depth: u8,
5245 ) -> Vec<string::Kind> {
5246 offered.to_vec()
5249 }
5250
5251 fn narrow_integers(
5252 &self,
5253 _values: &[i64],
5254 offered: &[integer::Kind],
5255 depth: u8,
5256 ) -> Vec<integer::Kind> {
5257 let keep: &[integer::Kind] = if depth == 0 {
5258 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
5259 } else {
5260 &[integer::Kind::Constant, integer::Kind::Packed]
5261 };
5262 let narrowed: Vec<integer::Kind> =
5263 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5264 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5267 }
5268}
5269
5270#[derive(Debug)]
5282struct Fixed;
5283
5284impl chooser::Chooser for Fixed {
5285 fn name(&self) -> &'static str {
5286 "fixed"
5287 }
5288
5289 fn narrow_strings(
5290 &self,
5291 _values: &[&[u8]],
5292 offered: &[string::Kind],
5293 _depth: u8,
5294 ) -> Vec<string::Kind> {
5295 offered.to_vec()
5296 }
5297
5298 fn narrow_integers(
5299 &self,
5300 _values: &[i64],
5301 offered: &[integer::Kind],
5302 depth: u8,
5303 ) -> Vec<integer::Kind> {
5304 let keep: &[integer::Kind] = if depth == 0 {
5305 &[
5306 integer::Kind::Constant,
5307 integer::Kind::Packed,
5308 integer::Kind::Delta,
5309 integer::Kind::Rle,
5310 integer::Kind::Sparse,
5311 integer::Kind::Strided,
5312 ]
5313 } else {
5314 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
5315 };
5316 let narrowed: Vec<integer::Kind> =
5317 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5318 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5319 }
5320}
5321
5322fn widened(data: &Data) -> Option<Vec<i64>> {
5329 match data {
5330 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5331 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5332 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5333 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5334 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5335 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5336 Data::Int64(values) => Some(values.to_vec()),
5337 _ => None,
5338 }
5339}
5340
5341trait Narrow: Copy {
5348 const BIASED: (u32, u64);
5353
5354 fn narrow(value: i64) -> Self;
5356}
5357
5358#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
5375fn residue<T: Narrow>(value: i64) -> u64 {
5376 let (bits, bias) = T::BIASED;
5377 (value as u64).wrapping_add(bias) >> bits
5378}
5379
5380macro_rules! narrows {
5385 ($($ty:ty => $bias:expr),* $(,)?) => {$(
5386 impl Narrow for $ty {
5387 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
5388
5389 #[allow(
5390 clippy::cast_possible_truncation,
5391 clippy::cast_sign_loss,
5392 reason = "the caller has checked the bits this truncates away"
5393 )]
5394 fn narrow(value: i64) -> Self {
5395 value as Self
5396 }
5397 }
5398 )*};
5399}
5400
5401narrows! {
5402 i8 => 1 << 7,
5403 u8 => 0,
5404 i16 => 1 << 15,
5405 u16 => 0,
5406 i32 => 1 << 31,
5407 u32 => 0,
5408}
5409
5410fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
5423 let mut spilled = 0u64;
5424 for value in values {
5425 spilled |= residue::<T>(*value);
5426 }
5427 if spilled != 0 {
5428 return Err(invalid("page value is not of its type"));
5429 }
5430 Ok(values.iter().map(|value| T::narrow(*value)).collect())
5431}
5432
5433fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
5438 Ok(match ty {
5439 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
5440 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
5441 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
5442 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
5443 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
5444 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
5445 LogicalType::BigInt
5446 | LogicalType::Timestamp
5447 | LogicalType::Time
5448 | LogicalType::TimeTz
5449 | LogicalType::TimestampTz
5450 | LogicalType::TimestampS
5451 | LogicalType::TimestampMs
5452 | LogicalType::TimestampNs => Data::Int64(values.into()),
5453 LogicalType::Decimal { .. } => match ty.physical() {
5456 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
5457 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
5458 PhysicalType::Int64 => Data::Int64(values.into()),
5459 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
5460 },
5461 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
5462 })
5463}
5464
5465fn plain_width(ty: &LogicalType) -> Option<usize> {
5468 Some(match ty {
5469 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
5470 LogicalType::SmallInt | LogicalType::USmallInt => 2,
5471 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
5472 LogicalType::BigInt
5473 | LogicalType::Timestamp
5474 | LogicalType::Time
5475 | LogicalType::TimeTz
5476 | LogicalType::TimestampTz
5477 | LogicalType::TimestampS
5478 | LogicalType::TimestampMs
5479 | LogicalType::TimestampNs => 8,
5480 LogicalType::Decimal { .. } => match ty.physical() {
5481 PhysicalType::Int16 => 2,
5482 PhysicalType::Int32 => 4,
5483 PhysicalType::Int64 => 8,
5484 _ => return None,
5487 },
5488 _ => return None,
5489 })
5490}
5491
5492fn cascaded(
5498 flat: &Vector,
5499 ty: &LogicalType,
5500 packed: Option<&Packed<'_>>,
5501) -> Result<Option<Vec<u8>>> {
5502 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
5503 let Some(values) = widened(data) else { return Ok(None) };
5504 let plain = values.len().saturating_mul(width);
5505 let best = match packed {
5506 Some(packed) => plain.min(21 + size_of_val(packed.words())),
5508 None => plain,
5509 };
5510 let out = integer::encode_with(&values, &Fixed)?;
5511 Ok((out.len() < best).then_some(out))
5512}
5513
5514fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
5552 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
5553 let mut payload = 0_usize;
5554 for row in 0..flat.len() {
5555 let text = flat.text_at(row).unwrap_or("").as_bytes();
5556 payload = payload.saturating_add(text.len());
5557 values.push(text);
5558 }
5559 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
5561 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
5562 return Ok(None);
5563 };
5564 Ok((out.len() < plain).then_some(out))
5565}
5566
5567fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
5568 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
5569 let coded = integer::encode_with(&wide, &Codes)?;
5570 let plain = codes.len().saturating_mul(size_of::<u32>());
5571 Ok((coded.len() < plain).then_some(coded))
5572}
5573
5574fn encode(
5575 vector: &Vector,
5576 global: Option<&mut GlobalDictionary>,
5577) -> Result<(Vec<u8>, Option<Vec<u32>>)> {
5578 let ty = vector.logical_type();
5579 let flat = vector.flatten()?;
5581 let mut out = Vec::new();
5582 let mut global_codes = None;
5583 if let Some(global) = global {
5584 let mut codes = Vec::with_capacity(flat.len());
5585 for row in 0..flat.len() {
5586 let text = flat.text_at(row).unwrap_or("");
5587 let code = global.code(text)?;
5588 global.observe(code, flat.is_null_at(row))?;
5589 codes.push(code);
5590 }
5591 global_codes = Some(codes);
5592 }
5593 let membership = global_codes.as_deref().map(unique_codes);
5594 let dictionary = if global_codes.is_none() && ty == &LogicalType::Varchar {
5595 string_dictionary(&flat)?
5596 } else {
5597 None
5598 };
5599 let compressed_text =
5600 if global_codes.is_none() && dictionary.is_none() && ty == &LogicalType::Varchar {
5601 text_compressed(&flat)?
5602 } else {
5603 None
5604 };
5605 let packed_vector = if dictionary.is_none() && global_codes.is_none() {
5606 Some(flat.bit_packed()?)
5607 } else {
5608 None
5609 };
5610 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
5611 let coded = match global_codes.as_deref() {
5612 Some(codes) => encoded_codes(codes)?,
5613 None => None,
5614 };
5615 let cascade = if dictionary.is_none() && global_codes.is_none() {
5619 cascaded(&flat, ty, packed.as_ref())?
5620 } else {
5621 None
5622 };
5623 out.push(if coded.is_some() {
5624 4
5625 } else if cascade.is_some() {
5626 5
5627 } else if global_codes.is_some() {
5628 3
5629 } else if dictionary.is_some() {
5630 1
5631 } else if compressed_text.is_some() {
5632 6
5633 } else if packed.is_some() {
5634 2
5635 } else {
5636 0
5637 });
5638 let nulls = flat.validity();
5639 let flag = match nulls {
5640 Validity::AllValid => 0,
5641 Validity::AllInvalid => 1,
5642 Validity::Mask(_) => 2,
5643 };
5644 out.push(flag);
5645 if flag == 2 {
5646 for group in (0..vector.len()).step_by(8) {
5647 let mut bits = 0_u8;
5648 for bit in 0..8 {
5649 if group + bit < vector.len() && !flat.is_null_at(group + bit) {
5650 bits |= 1 << bit;
5651 }
5652 }
5653 out.push(bits);
5654 }
5655 }
5656 if let Some(coded) = coded {
5657 out.extend_from_slice(&coded);
5658 return Ok((out, membership));
5659 }
5660 if let Some(cascade) = cascade {
5661 out.extend_from_slice(&cascade);
5662 return Ok((out, membership));
5663 }
5664 if let Some(codes) = global_codes {
5665 for code in codes {
5666 put_u32(&mut out, code);
5667 }
5668 return Ok((out, membership));
5669 }
5670 if let Some(dictionary) = dictionary {
5671 out.extend_from_slice(&dictionary);
5672 return Ok((out, membership));
5673 }
5674 if let Some(compressed_text) = compressed_text {
5675 out.extend_from_slice(&compressed_text);
5676 return Ok((out, membership));
5677 }
5678 if let Some(packed) = packed {
5679 if packed.offset() != 0 {
5680 return Err(invalid("writer received a sliced packed vector"));
5681 }
5682 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
5683 out.extend_from_slice(&packed.base().to_le_bytes());
5684 put_u32(
5685 &mut out,
5686 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
5687 );
5688 for word in packed.words() {
5689 put_u64(&mut out, *word);
5690 }
5691 return Ok((out, membership));
5692 }
5693 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
5694 match (ty, data) {
5695 (LogicalType::TinyInt, Data::Int8(values)) => {
5696 for value in &**values {
5697 out.extend_from_slice(&value.to_le_bytes());
5698 }
5699 }
5700 (LogicalType::UTinyInt, Data::UInt8(values)) => {
5701 for value in &**values {
5702 out.extend_from_slice(&value.to_le_bytes());
5703 }
5704 }
5705 (LogicalType::SmallInt, Data::Int16(values)) => {
5706 for value in &**values {
5707 out.extend_from_slice(&value.to_le_bytes());
5708 }
5709 }
5710 (LogicalType::USmallInt, Data::UInt16(values)) => {
5711 for value in &**values {
5712 out.extend_from_slice(&value.to_le_bytes());
5713 }
5714 }
5715 (LogicalType::UInteger, Data::UInt32(values)) => {
5716 for value in &**values {
5717 out.extend_from_slice(&value.to_le_bytes());
5718 }
5719 }
5720 (LogicalType::UBigInt, Data::UInt64(values)) => {
5721 for value in &**values {
5722 out.extend_from_slice(&value.to_le_bytes());
5723 }
5724 }
5725 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
5726 for value in &**values {
5727 out.extend_from_slice(&value.to_le_bytes());
5728 }
5729 }
5730 (
5731 LogicalType::BigInt
5732 | LogicalType::Timestamp
5733 | LogicalType::Time
5734 | LogicalType::TimeTz
5735 | LogicalType::TimestampTz
5736 | LogicalType::TimestampS
5737 | LogicalType::TimestampMs
5738 | LogicalType::TimestampNs,
5739 Data::Int64(values),
5740 ) => {
5741 for value in &**values {
5742 out.extend_from_slice(&value.to_le_bytes());
5743 }
5744 }
5745 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
5748 for value in &**values {
5749 out.extend_from_slice(&value.to_le_bytes());
5750 }
5751 }
5752 (LogicalType::UHugeInt, Data::UInt128(values)) => {
5753 for value in &**values {
5754 out.extend_from_slice(&value.to_le_bytes());
5755 }
5756 }
5757 (LogicalType::Float, Data::Float32(values)) => {
5760 for value in &**values {
5761 out.extend_from_slice(&value.to_le_bytes());
5762 }
5763 }
5764 (LogicalType::Double, Data::Float64(values)) => {
5765 for value in &**values {
5766 out.extend_from_slice(&value.to_le_bytes());
5767 }
5768 }
5769 (LogicalType::Interval, Data::Interval(values)) => {
5773 for (months, days, micros) in &**values {
5774 out.extend_from_slice(&months.to_le_bytes());
5775 out.extend_from_slice(&days.to_le_bytes());
5776 out.extend_from_slice(µs.to_le_bytes());
5777 }
5778 }
5779 (LogicalType::Boolean, Data::Bool(values)) => {
5780 for value in &**values {
5781 out.push(u8::from(*value));
5782 }
5783 }
5784 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
5787 for value in &**values {
5788 out.extend_from_slice(&value.to_le_bytes());
5789 }
5790 }
5791 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
5792 for value in &**values {
5793 out.extend_from_slice(&value.to_le_bytes());
5794 }
5795 }
5796 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
5797 for value in &**values {
5798 out.extend_from_slice(&value.to_le_bytes());
5799 }
5800 }
5801 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
5802 for value in &**values {
5803 out.extend_from_slice(&value.to_le_bytes());
5804 }
5805 }
5806 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
5811 let mut bytes = Vec::new();
5812 put_u32(&mut out, 0);
5813 for row in 0..vector.len() {
5814 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
5815 bytes.extend_from_slice(value);
5816 put_u32(
5817 &mut out,
5818 u32::try_from(bytes.len())
5819 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
5820 );
5821 }
5822 out.extend_from_slice(&bytes);
5823 }
5824 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
5825 }
5826 Ok((out, membership))
5827}
5828
5829fn put_varint(out: &mut Vec<u8>, mut value: u32) {
5830 while value >= 0x80 {
5831 out.push((value as u8 & 0x7f) | 0x80);
5832 value >>= 7;
5833 }
5834 out.push(value as u8);
5835}
5836
5837fn unique_codes(codes: &[u32]) -> Vec<u32> {
5839 let mut unique = codes.to_vec();
5840 unique.sort_unstable();
5841 unique.dedup();
5842 unique
5843}
5844
5845fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
5851 let mut lists = lists;
5852 while lists.len() > 1 {
5853 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
5854 for pair in lists.chunks(2) {
5855 match pair {
5856 [left, right] => next.push(merged_pair(left, right)),
5857 [only] => next.push(only.clone()),
5858 _ => {}
5859 }
5860 }
5861 lists = next;
5862 }
5863 lists.pop().unwrap_or_default()
5864}
5865
5866fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
5867 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
5868 let mut at = 0;
5869 let mut to = 0;
5870 while at < left.len() && to < right.len() {
5871 match left[at].cmp(&right[to]) {
5872 Ordering::Less => {
5873 out.push(left[at]);
5874 at += 1;
5875 }
5876 Ordering::Greater => {
5877 out.push(right[to]);
5878 to += 1;
5879 }
5880 Ordering::Equal => {
5881 out.push(left[at]);
5882 at += 1;
5883 to += 1;
5884 }
5885 }
5886 }
5887 out.extend_from_slice(&left[at..]);
5888 out.extend_from_slice(&right[to..]);
5889 out
5890}
5891
5892fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
5897 let mut merged = Range::default();
5898 let mut first = true;
5899 for range in ranges {
5900 merged.nulls = merged.nulls.saturating_add(range.nulls);
5901 merged.sum = match (merged.sum.take(), range.sum) {
5905 (Some(held), Some(next)) if !first => held.checked_add(next),
5906 (_, next) if first => next,
5907 _ => None,
5908 };
5909 merged.exact = if first { range.exact } else { merged.exact && range.exact };
5910 if first {
5911 merged.low = range.low;
5912 merged.high = range.high;
5913 first = false;
5914 continue;
5915 }
5916 merged.low = match (merged.low.take(), range.low) {
5917 (Some(held), Some(next)) => Some(held.smaller(next)),
5918 _ => None,
5919 };
5920 merged.high = match (merged.high.take(), range.high) {
5921 (Some(held), Some(next)) => Some(held.larger(next)),
5922 _ => None,
5923 };
5924 }
5925 merged
5926}
5927
5928fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
5941 match bound {
5942 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
5943 value.truncate(PART_BOUND_BYTES);
5944 if !high {
5945 return Some(Bound::Bytes(value));
5946 }
5947 while let Some(last) = value.pop() {
5948 if last < u8::MAX {
5949 value.push(last + 1);
5950 return Some(Bound::Bytes(value));
5951 }
5952 }
5953 None
5954 }
5955 other => other,
5956 }
5957}
5958
5959fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
5967 let mut out = Vec::new();
5968 put_u32(
5969 &mut out,
5970 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5971 );
5972 for range in ranges {
5973 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
5974 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
5975 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
5976 }
5977 Ok(out)
5978}
5979
5980fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
5982 let mut cur = Cursor { bytes, at: 0 };
5983 let parts = cur.u32()? as usize;
5984 let mut out = Vec::new();
5985 for _ in 0..parts {
5986 let low = cur.bound()?;
5987 let high = cur.bound()?;
5988 let nulls = cur.u32()? as usize;
5989 out.push(Range { low, high, nulls, exact: false, sum: None });
5990 }
5991 Ok(out)
5992}
5993
5994fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
5995 let held: Vec<&Option<Sieve>> = sieves.collect();
5996 let mut out = Vec::new();
5997 put_u32(
5998 &mut out,
5999 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
6000 );
6001 for sieve in &held {
6002 let length = sieve.as_ref().map_or(0, Sieve::len);
6003 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
6004 }
6005 for sieve in held.into_iter().flatten() {
6007 out.extend_from_slice(&sieve.to_bytes());
6008 }
6009 Ok(out)
6010}
6011
6012fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
6018 let parts = u32::from_le_bytes(
6019 bytes
6020 .get(..4)
6021 .ok_or_else(|| invalid("sieve page is truncated"))?
6022 .try_into()
6023 .map_err(|_| invalid("sieve page is truncated"))?,
6024 ) as usize;
6025 let mut lengths = Vec::with_capacity(parts);
6026 for part in 0..parts {
6027 let at = 4 + part * 4;
6028 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
6029 lengths.push(u32::from_le_bytes(
6030 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
6031 ) as usize);
6032 }
6033 let mut at = 4 + parts * 4;
6034 let mut out = Vec::with_capacity(parts);
6035 for length in lengths {
6036 if length == 0 {
6037 out.push(None);
6038 continue;
6039 }
6040 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
6041 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
6042 out.push(Sieve::from_bytes(field));
6043 at = end;
6044 }
6045 if at != bytes.len() {
6046 return Err(invalid("sieve page has trailing bytes"));
6047 }
6048 Ok(out)
6049}
6050
6051fn encode_membership(unique: &[u32]) -> Vec<u8> {
6057 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
6058 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
6059 let mut previous = 0;
6060 for (at, &code) in unique.iter().enumerate() {
6061 put_varint(&mut out, if at == 0 { code } else { code - previous });
6062 previous = code;
6063 }
6064 out
6065}
6066
6067fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
6068 let mut value = 0_u32;
6069 for shift in (0..35).step_by(7) {
6070 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
6071 *at += 1;
6072 let part = u32::from(byte & 0x7f);
6073 if shift == 28 && part > 0x0f {
6074 return Err(invalid("membership varint overflow"));
6075 }
6076 value = value
6077 .checked_add(
6078 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
6079 )
6080 .ok_or_else(|| invalid("membership varint overflow"))?;
6081 if byte & 0x80 == 0 {
6082 return Ok(value);
6083 }
6084 }
6085 Err(invalid("membership varint is too long"))
6086}
6087
6088fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
6089 let mut at = 0;
6090 let count = take_varint(bytes, &mut at)? as usize;
6091 let mut codes = Vec::with_capacity(count);
6092 let mut previous = 0_u32;
6093 for index in 0..count {
6094 let delta = take_varint(bytes, &mut at)?;
6095 let code = if index == 0 {
6096 delta
6097 } else {
6098 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
6099 };
6100 if index > 0 && code <= previous {
6101 return Err(invalid("membership codes are not increasing"));
6102 }
6103 codes.push(code);
6104 previous = code;
6105 }
6106 if at != bytes.len() {
6107 return Err(invalid("membership page has trailing bytes"));
6108 }
6109 Ok(codes)
6110}
6111
6112fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
6113 let mut by_text = HashMap::new();
6114 let mut values = Vec::new();
6115 let mut codes = Vec::with_capacity(vector.len());
6116 let mut plain_bytes = 0_usize;
6117 for row in 0..vector.len() {
6118 let text = vector.text_at(row).unwrap_or("");
6119 plain_bytes = plain_bytes.saturating_add(text.len());
6120 let code = match by_text.get(text) {
6121 Some(&code) => code,
6122 None => {
6123 let code = u32::try_from(values.len())
6124 .map_err(|_| invalid("too many dictionary values"))?;
6125 by_text.insert(text, code);
6126 values.push(text);
6127 code
6128 }
6129 };
6130 codes.push(code);
6131 }
6132 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
6133 let encoded = 8_usize
6134 .saturating_add((values.len() + 1).saturating_mul(4))
6135 .saturating_add(dictionary_bytes)
6136 .saturating_add(codes.len().saturating_mul(4));
6137 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
6138 if encoded >= plain {
6139 return Ok(None);
6140 }
6141 let mut out = Vec::with_capacity(encoded);
6142 put_u32(
6143 &mut out,
6144 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
6145 );
6146 put_u32(
6147 &mut out,
6148 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
6149 );
6150 let mut offset = 0_u32;
6151 put_u32(&mut out, offset);
6152 for value in &values {
6153 offset = offset
6154 .checked_add(
6155 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
6156 )
6157 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
6158 put_u32(&mut out, offset);
6159 }
6160 for value in values {
6161 out.extend_from_slice(value.as_bytes());
6162 }
6163 for code in codes {
6164 put_u32(&mut out, code);
6165 }
6166 Ok(Some(out))
6167}
6168
6169struct EncodedDictionary {
6170 index: Vec<u8>,
6171 ranks: Vec<u8>,
6172 payload: Vec<Vec<u8>>,
6175}
6176
6177fn head(bytes: &[u8]) -> u64 {
6179 let mut word = [0; 8];
6180 let take = bytes.len().min(8);
6181 word[..take].copy_from_slice(&bytes[..take]);
6182 u64::from_be_bytes(word)
6183}
6184
6185fn rankings(dictionaries: &[Option<GlobalDictionary>]) -> Result<Vec<Vec<(u64, u32)>>> {
6193 let present =
6194 dictionaries.iter().enumerate().filter(|(_, held)| held.is_some()).map(|(at, _)| at);
6195 let present = present.collect::<Vec<_>>();
6196 let mut orders = vec![Vec::new(); dictionaries.len()];
6197 let workers = std::thread::available_parallelism()
6198 .map_or(1, usize::from)
6199 .min(MAX_FREQUENCY_WORKERS)
6200 .min(present.len());
6201 if workers <= 1 {
6202 for at in present {
6203 if let Some(dictionary) = &dictionaries[at] {
6204 orders[at] = dictionary.ranked();
6205 }
6206 }
6207 return Ok(orders);
6208 }
6209 let width = present.len().div_ceil(workers);
6210 let pieces = std::thread::scope(|scope| {
6211 present
6212 .chunks(width)
6213 .map(|columns| {
6214 scope.spawn(|| {
6215 columns
6216 .iter()
6217 .filter_map(|&at| dictionaries[at].as_ref().map(|held| (at, held.ranked())))
6218 .collect::<Vec<_>>()
6219 })
6220 })
6221 .collect::<Vec<_>>()
6222 .into_iter()
6223 .map(|handle| {
6224 handle.join().map_err(|_| Error::internal("a dictionary sort worker panicked"))
6225 })
6226 .collect::<Result<Vec<_>>>()
6227 })?;
6228 for piece in pieces {
6229 for (at, order) in piece {
6230 orders[at] = order;
6231 }
6232 }
6233 Ok(orders)
6234}
6235
6236fn encode_global_dictionary(
6237 dictionary: GlobalDictionary,
6238 order: &[(u64, u32)],
6239) -> Result<EncodedDictionary> {
6240 let values = dictionary.offsets.len() - 1;
6241 if order.len() != values {
6242 return Err(invalid("global dictionary order does not cover its values"));
6243 }
6244 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6245 let payload = encode_payload(&dictionary)?;
6246 if payload.len() != blocks {
6247 return Err(invalid("global dictionary payload is not the blocks it says it is"));
6248 }
6249 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
6250 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
6251 let offset_bits = offset_width(&dictionary.offsets);
6252 let mut index = Vec::with_capacity(
6253 DICTIONARY_HEADER + offset_bytes(values, offset_bits) + (blocks + rank_blocks) * 16,
6254 );
6255 put_u32(
6256 &mut index,
6257 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
6258 );
6259 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
6260 put_u32(
6261 &mut index,
6262 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
6263 );
6264 put_u32(&mut index, offset_bits as u32);
6265 encode_offsets(&dictionary.offsets, offset_bits, &mut index)?;
6266 let mut at = 0_u64;
6270 for block in &payload {
6271 at = at
6272 .checked_add(block.len() as u64)
6273 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
6274 put_u64(&mut index, at);
6275 }
6276 for block in &payload {
6277 put_u64(&mut index, checksum(block));
6278 }
6279 if rank_ends.len() != rank_blocks {
6282 return Err(invalid("global dictionary order is not the blocks it says it is"));
6283 }
6284 for end in &rank_ends {
6285 put_u64(&mut index, *end);
6286 }
6287 let mut at = 0_usize;
6288 for end in &rank_ends {
6289 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
6290 put_u64(&mut index, checksum(&ranks[at..end]));
6291 at = end;
6292 }
6293 Ok(EncodedDictionary { index, ranks, payload })
6294}
6295
6296const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
6303
6304fn payload_shapes() -> Vec<chooser::Settled> {
6330 let integers = vec![integer::Kind::Packed];
6331 [
6332 vec![string::Kind::Front, string::Kind::Lz],
6333 vec![string::Kind::Lz, string::Kind::Fsst],
6334 vec![string::Kind::Lz, string::Kind::Plain],
6335 vec![string::Kind::Fsst],
6336 vec![string::Kind::Plain],
6337 ]
6338 .into_iter()
6339 .map(|strings| chooser::Settled::new(strings, integers.clone()))
6340 .collect()
6341}
6342
6343fn encode_payload(dictionary: &GlobalDictionary) -> Result<Vec<Vec<u8>>> {
6349 let values = dictionary.offsets.len() - 1;
6350 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6351 let run = |block: usize| {
6352 let first = block * TEXT_PAYLOAD_VALUES;
6353 let last = (first + TEXT_PAYLOAD_VALUES).min(values);
6354 (first..last)
6355 .map(|value| {
6356 let from = dictionary.offsets[value] as usize;
6357 let to = dictionary.offsets[value + 1] as usize;
6358 &dictionary.payload[from..to]
6359 })
6360 .collect::<Vec<_>>()
6361 };
6362 let shape = (blocks > PAYLOAD_SAMPLE_BLOCKS).then(|| settle_shape(&run, blocks)).transpose()?;
6365 let one = |block: usize| match &shape {
6366 Some(shape) => string::encode_with(&run(block), shape),
6367 None => string::encode(&run(block)),
6368 };
6369 let workers = std::thread::available_parallelism()
6370 .map_or(1, usize::from)
6371 .min(MAX_FREQUENCY_WORKERS)
6372 .min(blocks);
6373 if workers <= 1 {
6374 return (0..blocks).map(one).collect();
6375 }
6376 let next = AtomicUsize::new(0);
6377 let pieces = std::thread::scope(|scope| {
6378 (0..workers)
6379 .map(|_| {
6380 scope.spawn(|| {
6381 let mut mine = Vec::new();
6382 loop {
6383 let block = next.fetch_add(1, Atomic::Relaxed);
6384 if block >= blocks {
6385 break;
6386 }
6387 mine.push((block, one(block)?));
6388 }
6389 Ok(mine)
6390 })
6391 })
6392 .collect::<Vec<_>>()
6393 .into_iter()
6394 .map(|handle| {
6395 handle.join().map_err(|_| Error::internal("a dictionary encode worker panicked"))?
6396 })
6397 .collect::<Result<Vec<_>>>()
6398 })?;
6399 let mut payload = vec![Vec::new(); blocks];
6400 for piece in pieces {
6401 for (block, bytes) in piece {
6402 payload[block] = bytes;
6403 }
6404 }
6405 Ok(payload)
6406}
6407
6408fn settle_shape<'a>(
6416 run: &dyn Fn(usize) -> Vec<&'a [u8]>,
6417 blocks: usize,
6418) -> Result<chooser::Settled> {
6419 let last = blocks - 1;
6420 let sample = (0..PAYLOAD_SAMPLE_BLOCKS)
6421 .map(|region| run(region * last / (PAYLOAD_SAMPLE_BLOCKS - 1)))
6422 .collect::<Vec<_>>();
6423 let mut best: Option<(chooser::Settled, usize)> = None;
6424 for shape in payload_shapes() {
6425 let mut size = 0;
6426 for block in &sample {
6427 size += string::encode_with(block, &shape)?.len();
6428 }
6429 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
6430 best = Some((shape, size));
6431 }
6432 }
6433 best.map(|(shape, _)| shape)
6434 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
6435}
6436
6437fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
6444 let mut out = Vec::with_capacity(order.len() * 4);
6445 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
6446 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
6447 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
6448 for block in order.chunks(TEXT_RANK_BLOCK) {
6449 let base = block.first().map_or(0, |&(head, _)| head);
6452 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
6453 let width = (u64::BITS - span.leading_zeros()) as usize;
6454 heads.clear();
6455 codes.clear();
6456 for &(head, code) in block {
6457 heads.push(head.wrapping_sub(base));
6458 codes.push(u64::from(code));
6459 }
6460 put_u64(&mut out, base);
6461 out.push(width as u8);
6462 bitpack::pack_tail(&heads, width, &mut out)
6463 .map_err(|_| invalid("global dictionary heads do not pack"))?;
6464 bitpack::pack_tail(&codes, code_bits, &mut out)
6465 .map_err(|_| invalid("global dictionary codes do not pack"))?;
6466 ends.push(out.len() as u64);
6467 }
6468 Ok((out, ends))
6469}
6470
6471fn open_global_dictionary(
6478 file: Arc<File>,
6479 page: Page,
6480 ty: &LogicalType,
6481 keep_budget: usize,
6482) -> Result<Vector> {
6483 if ty != &LogicalType::Varchar {
6484 return Err(invalid("global dictionary belongs to a non-string column"));
6485 }
6486 let mut header = [0; DICTIONARY_HEADER];
6487 read_at(&file, page.offset, &mut header)?;
6488 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
6489 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
6490 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
6491 let offset_bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6492 if per_block != TEXT_PAYLOAD_VALUES {
6493 return Err(invalid("global dictionary block width differs"));
6494 }
6495 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
6496 return Err(invalid("global dictionary block count differs from its value count"));
6497 }
6498 if offset_bits > u32::BITS as usize {
6499 return Err(invalid("global dictionary packs offsets past a payload"));
6500 }
6501 let offset_len = offset_bytes(count, offset_bits);
6502 let ranks = count;
6507 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
6508 let hash_len = blocks
6511 .checked_add(rank_blocks)
6512 .and_then(|words| words.checked_mul(16))
6513 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
6514 let index_len = DICTIONARY_HEADER
6515 .checked_add(offset_len)
6516 .and_then(|len| len.checked_add(hash_len))
6517 .ok_or_else(|| invalid("global dictionary header overflow"))?;
6518 if index_len > page.length as usize {
6519 return Err(invalid("global dictionary offset index exceeds its page"));
6520 }
6521 let mut index = vec![0; index_len];
6522 index[..DICTIONARY_HEADER].copy_from_slice(&header);
6523 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
6524 if checksum(&index) != page.hash {
6525 return Err(invalid("global dictionary index checksum differs"));
6526 }
6527 let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
6528 let mut words = index[DICTIONARY_HEADER + offset_len..]
6529 .chunks_exact(8)
6530 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
6531 .collect::<Vec<_>>();
6532 let mut hashes = words.split_off(blocks);
6533 let mut rank_ends = hashes.split_off(blocks);
6534 let rank_hashes = rank_ends.split_off(rank_blocks);
6535 let ends = words;
6536 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
6539 return Err(invalid("global dictionary order blocks do not rise"));
6540 }
6541 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
6542 .map_err(|_| invalid("global dictionary rank overflow"))?;
6543 let body_len = index_len
6544 .checked_add(rank_len)
6545 .ok_or_else(|| invalid("global dictionary header overflow"))?;
6546 if body_len > page.length as usize {
6547 return Err(invalid("global dictionary order exceeds its page"));
6548 }
6549 let stored_len = page.length as usize - body_len;
6552 if ends.last().copied().unwrap_or_default() as usize != stored_len
6553 || ends.windows(2).any(|pair| pair[0] > pair[1])
6554 {
6555 return Err(invalid("global dictionary blocks do not bound the payload"));
6556 }
6557 Vector::external_text(
6558 LogicalType::Varchar,
6559 Arc::new(NativeText {
6560 file,
6561 values: count,
6562 offsets,
6563 offset_bits,
6564 ranks,
6565 rank_at: page.offset + index_len as u64,
6566 rank_ends,
6567 rank_hashes,
6568 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
6569 code_bits: code_width(count),
6570 code_ranks: OnceLock::new(),
6571 payload: page.offset + body_len as u64,
6572 ends,
6573 hashes,
6574 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
6575 keep_budget,
6576 payload_kept: AtomicUsize::new(0),
6577 searched: Mutex::new(HashMap::new()),
6578 }),
6579 )
6580}
6581
6582fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
6595 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
6597 let mut cur = Cursor { bytes, at: 0 };
6598 let codec = cur.u8()?;
6599 if cur.u8()? == 2 {
6600 cur.take(rows.div_ceil(8))?;
6601 }
6602 Ok((codec, cur.at))
6603 }
6604 let Ok((codec, at)) = cascade_at(rows, bytes) else {
6605 return "UNREADABLE".to_string();
6606 };
6607 let tail = &bytes[at..];
6608 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
6609 match codec {
6610 0 => match ty {
6611 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
6612 _ => "FIXED".to_string(),
6613 },
6614 1 => "DICT(PLAIN)".to_string(),
6615 2 => "FOR+BITPACK".to_string(),
6616 3 => "TABLE DICT".to_string(),
6617 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
6618 5 => described(integer::describe(tail)),
6619 6 => described(string::describe(tail)),
6620 other => format!("CODEC {other}"),
6621 }
6622}
6623
6624fn decode(
6625 ty: &LogicalType,
6626 rows: usize,
6627 bytes: &[u8],
6628 global: Option<Arc<Vector>>,
6629) -> Result<Vector> {
6630 let mut cur = Cursor { bytes, at: 0 };
6631 let codec = cur.u8()?;
6632 let flag = cur.u8()?;
6633 let validity = match flag {
6634 0 => Validity::AllValid,
6635 1 => Validity::AllInvalid,
6636 2 => {
6637 let mask = cur.take(rows.div_ceil(8))?;
6638 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
6639 }
6640 _ => return Err(invalid("page validity tag differs")),
6641 };
6642 if codec == 1 {
6643 if ty != &LogicalType::Varchar {
6644 return Err(invalid("dictionary codec belongs to a non-string page"));
6645 }
6646 let count = cur.u32()? as usize;
6647 let payload_len = cur.u32()? as usize;
6648 let offset_bytes = cur.take(
6649 (count + 1)
6650 .checked_mul(4)
6651 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
6652 )?;
6653 let offsets = offset_bytes
6654 .chunks_exact(4)
6655 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6656 .collect::<Vec<_>>();
6657 let payload = cur.take(payload_len)?.to_vec();
6658 if offsets.first() != Some(&0)
6659 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6660 || offsets.windows(2).any(|pair| pair[0] > pair[1])
6661 {
6662 return Err(invalid("dictionary offsets do not bound the payload"));
6663 }
6664 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
6667 for pair in offsets.windows(2) {
6668 strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
6669 }
6670 let mut codes = Vec::with_capacity(rows);
6671 for _ in 0..rows {
6672 codes.push(cur.u32()?);
6673 }
6674 if codes.iter().any(|code| *code as usize >= count) {
6675 return Err(invalid("dictionary code is out of range"));
6676 }
6677 if cur.at != bytes.len() {
6678 return Err(invalid("dictionary page has trailing bytes"));
6679 }
6680 let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
6681 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
6682 }
6683 if codec == 3 || codec == 4 {
6684 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
6685 let codes = if codec == 4 {
6686 let wide = integer::decode(&bytes[cur.at..])?;
6689 if wide.len() != rows {
6690 return Err(invalid("encoded code page holds the wrong number of rows"));
6691 }
6692 let mut codes = Vec::with_capacity(wide.len());
6699 let mut seen = 0_i64;
6700 for &code in &wide {
6701 seen |= code;
6702 codes.push(code as u32);
6703 }
6704 if seen < 0 || seen > i64::from(u32::MAX) {
6705 return Err(invalid("code is not a code"));
6706 }
6707 codes
6708 } else {
6709 let mut codes = Vec::with_capacity(rows);
6710 for _ in 0..rows {
6711 codes.push(cur.u32()?);
6712 }
6713 if cur.at != bytes.len() {
6714 return Err(invalid("global code page has trailing bytes"));
6715 }
6716 codes
6717 };
6718 let highest = codes.iter().copied().max();
6719 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
6720 .with_validity(validity));
6721 }
6722 if codec == 6 {
6723 if ty != &LogicalType::Varchar {
6724 return Err(invalid("compressed text codec belongs to a non-string page"));
6725 }
6726 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
6730 if ends.len() != rows {
6731 return Err(invalid("compressed text page holds the wrong number of rows"));
6732 }
6733 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6736 let mut start = 0;
6737 for end in ends {
6738 let len = end
6739 .checked_sub(start)
6740 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
6741 values.push_in_place(start, len)?;
6742 start = end;
6743 }
6744 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
6745 }
6746 if codec == 5 {
6747 let values = integer::decode(&bytes[cur.at..])?;
6749 if values.len() != rows {
6750 return Err(invalid("cascade page holds the wrong number of rows"));
6751 }
6752 let data = narrowed(ty, values)?;
6753 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
6754 }
6755 if codec == 2 {
6756 let width = u32::from(cur.u8()?);
6757 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
6758 let count = cur.u32()? as usize;
6759 let mut words = Vec::with_capacity(count);
6760 for _ in 0..count {
6761 words.push(cur.u64()?);
6762 }
6763 if cur.at != bytes.len() {
6764 return Err(invalid("packed page has trailing bytes"));
6765 }
6766 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
6767 }
6768 if codec != 0 {
6769 return Err(invalid("page codec is unknown"));
6770 }
6771 let data = match ty {
6772 LogicalType::TinyInt => {
6773 let values = cur.take(rows)?;
6774 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
6775 }
6776 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
6777 LogicalType::SmallInt => {
6778 let values =
6779 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6780 Data::Int16(
6781 values
6782 .chunks_exact(2)
6783 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6784 .collect::<Vec<_>>()
6785 .into(),
6786 )
6787 }
6788 LogicalType::USmallInt => {
6789 let values =
6790 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6791 Data::UInt16(
6792 values
6793 .chunks_exact(2)
6794 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
6795 .collect::<Vec<_>>()
6796 .into(),
6797 )
6798 }
6799 LogicalType::UInteger => {
6800 let values =
6801 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6802 Data::UInt32(
6803 values
6804 .chunks_exact(4)
6805 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
6806 .collect::<Vec<_>>()
6807 .into(),
6808 )
6809 }
6810 LogicalType::UBigInt => {
6811 let values =
6812 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6813 Data::UInt64(
6814 values
6815 .chunks_exact(8)
6816 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
6817 .collect::<Vec<_>>()
6818 .into(),
6819 )
6820 }
6821 LogicalType::Integer | LogicalType::Date => {
6822 let values =
6823 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6824 Data::Int32(
6825 values
6826 .chunks_exact(4)
6827 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6828 .collect::<Vec<_>>()
6829 .into(),
6830 )
6831 }
6832 LogicalType::BigInt
6833 | LogicalType::Timestamp
6834 | LogicalType::Time
6835 | LogicalType::TimeTz
6836 | LogicalType::TimestampTz
6837 | LogicalType::TimestampS
6838 | LogicalType::TimestampMs
6839 | LogicalType::TimestampNs => {
6840 let values =
6841 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6842 Data::Int64(
6843 values
6844 .chunks_exact(8)
6845 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6846 .collect::<Vec<_>>()
6847 .into(),
6848 )
6849 }
6850 LogicalType::HugeInt | LogicalType::Uuid => {
6851 let values =
6852 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6853 Data::Int128(
6854 values
6855 .chunks_exact(16)
6856 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6857 .collect::<Vec<_>>()
6858 .into(),
6859 )
6860 }
6861 LogicalType::UHugeInt => {
6862 let values =
6863 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6864 Data::UInt128(
6865 values
6866 .chunks_exact(16)
6867 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6868 .collect::<Vec<_>>()
6869 .into(),
6870 )
6871 }
6872 LogicalType::Float => {
6873 let values =
6874 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6875 Data::Float32(
6876 values
6877 .chunks_exact(4)
6878 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
6879 .collect::<Vec<_>>()
6880 .into(),
6881 )
6882 }
6883 LogicalType::Double => {
6884 let values =
6885 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6886 Data::Float64(
6887 values
6888 .chunks_exact(8)
6889 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
6890 .collect::<Vec<_>>()
6891 .into(),
6892 )
6893 }
6894 LogicalType::Interval => {
6895 let values =
6896 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6897 Data::Interval(
6898 values
6899 .chunks_exact(16)
6900 .map(|item| {
6901 (
6902 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
6903 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
6904 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
6905 )
6906 })
6907 .collect::<Vec<_>>()
6908 .into(),
6909 )
6910 }
6911 LogicalType::Boolean => {
6912 let values = cur.take(rows)?;
6913 if values.iter().any(|value| *value > 1) {
6914 return Err(invalid("boolean page has another value"));
6915 }
6916 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
6917 }
6918 LogicalType::Decimal { .. } => match ty.physical() {
6921 PhysicalType::Int16 => {
6922 let values =
6923 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6924 Data::Int16(
6925 values
6926 .chunks_exact(2)
6927 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6928 .collect::<Vec<_>>()
6929 .into(),
6930 )
6931 }
6932 PhysicalType::Int32 => {
6933 let values =
6934 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6935 Data::Int32(
6936 values
6937 .chunks_exact(4)
6938 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6939 .collect::<Vec<_>>()
6940 .into(),
6941 )
6942 }
6943 PhysicalType::Int64 => {
6944 let values =
6945 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6946 Data::Int64(
6947 values
6948 .chunks_exact(8)
6949 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6950 .collect::<Vec<_>>()
6951 .into(),
6952 )
6953 }
6954 _ => {
6955 let values =
6956 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6957 Data::Int128(
6958 values
6959 .chunks_exact(16)
6960 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6961 .collect::<Vec<_>>()
6962 .into(),
6963 )
6964 }
6965 },
6966 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
6967 let offset_bytes = cur
6968 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
6969 let offsets = offset_bytes
6970 .chunks_exact(4)
6971 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6972 .collect::<Vec<_>>();
6973 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
6974 if offsets.first() != Some(&0)
6975 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6976 || offsets.windows(2).any(|pair| pair[0] > pair[1])
6977 {
6978 return Err(invalid("string offsets do not bound the payload"));
6979 }
6980 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6988 let text = ty == &LogicalType::Varchar;
6989 for pair in offsets.windows(2) {
6990 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
6991 if text {
6992 values.push_in_place(at, len)?;
6993 } else {
6994 values.push_bytes_in_place(at, len)?;
6995 }
6996 }
6997 Data::Varlen(values)
6998 }
6999 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
7000 };
7001 if cur.at != bytes.len() {
7002 return Err(invalid("page has trailing bytes"));
7003 }
7004 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
7005}
7006
7007#[cfg(test)]
7008mod tests {
7009 use std::fs;
7010 use std::io::{Seek, SeekFrom, Write};
7011 use std::path::PathBuf;
7012 use std::time::{SystemTime, UNIX_EPOCH};
7013
7014 use rudb_common::Stat;
7015 use rudb_common::Value;
7016 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
7017 use rudb_common::stat::Provenance;
7018
7019 use super::*;
7020
7021 #[test]
7022 fn checksum_matches_fixed_vectors() {
7023 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
7024 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
7025 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
7026 }
7027
7028 fn path(label: &str) -> PathBuf {
7029 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
7030 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
7031 }
7032
7033 #[test]
7035 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
7036 const SPANS: usize = 64;
7037 const SPAN: usize = 512;
7038 let path = path("positional");
7039 let content: Vec<u8> =
7040 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
7041 fs::write(&path, &content).expect("the file is written");
7042 let file = Arc::new(File::open(&path).expect("the file opens"));
7043 std::thread::scope(|scope| {
7044 for _ in 0..8 {
7045 let file = Arc::clone(&file);
7046 scope.spawn(move || {
7047 for _ in 0..64 {
7048 for span in 0..SPANS {
7049 let mut bytes = [0_u8; SPAN];
7050 read_at(&file, (span * SPAN) as u64, &mut bytes)
7051 .expect("the span reads");
7052 assert!(
7053 bytes.iter().all(|byte| *byte == span as u8),
7054 "span {span} came back as {}",
7055 bytes[0],
7056 );
7057 }
7058 }
7059 });
7060 }
7061 });
7062 let mut past = [0_u8; SPAN];
7063 let end = (SPANS * SPAN) as u64;
7064 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
7065 assert!(error.message().contains("ends before its declared length"), "{error}");
7066 drop(file);
7067 let _ = fs::remove_file(&path);
7068 }
7069
7070 #[test]
7076 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
7077 let path = path("cursor");
7078 let mut writer = Writer::create(
7079 &path,
7080 "items",
7081 vec![
7082 Field::required("id", LogicalType::Integer),
7083 Field::new("text", LogicalType::Varchar),
7084 ],
7085 )
7086 .expect("new file");
7087 writer.append(&sample()).expect("first part");
7088 writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
7089 writer.append(&sample()).expect("second part");
7090 writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
7091 writer.finish().expect("commit");
7092 let reader = Reader::open(&path).expect("reopen from disk");
7093 assert_eq!(reader.table().rows(), 6);
7094 let ids = reader.read(0, &[0]).expect("the integer page reads back");
7095 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
7096 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
7097 let text = reader.read(1, &[1]).expect("the text page reads back");
7098 assert_eq!(text.value_at(1, 0), Value::Null);
7099 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7100 let end = reader.table().stripes().iter().flat_map(|stripe| {
7103 stripe
7104 .pages
7105 .iter()
7106 .map(|page| page.offset + u64::from(page.length))
7107 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
7108 });
7109 let last = end.fold(HEADER, u64::max);
7110 let directory = fs::metadata(&path).expect("the file is there").len();
7111 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
7112 fs::remove_file(path).expect("remove scratch file");
7113 }
7114
7115 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
7121 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7122 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7123 let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7124 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7125 DICTIONARY_HEADER as u64
7126 + offset_bytes(count as usize, bits) as u64
7127 + (blocks + rank_blocks) * 16
7128 }
7129
7130 fn last_rank_end(file: &File, offset: u64, header: &[u8; DICTIONARY_HEADER]) -> u64 {
7132 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7133 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7134 let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7135 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7136 let at = offset
7137 + DICTIONARY_HEADER as u64
7138 + offset_bytes(count as usize, bits) as u64
7139 + blocks * 16
7140 + (rank_blocks - 1) * 8;
7141 let mut end = [0; 8];
7142 read_at(file, at, &mut end).expect("the last rank block end");
7143 u64::from_le_bytes(end)
7144 }
7145
7146 fn sample() -> Chunk {
7147 Chunk::new(vec![
7148 Vector::from_values(
7149 LogicalType::Integer,
7150 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
7151 )
7152 .expect("integers"),
7153 Vector::from_values(
7154 LogicalType::Varchar,
7155 &[
7156 Value::Varchar("alpha".into()),
7157 Value::Null,
7158 Value::Varchar("long text after a slash".into()),
7159 ],
7160 )
7161 .expect("strings"),
7162 ])
7163 .expect("matching rows")
7164 }
7165
7166 fn sample_ids() -> Chunk {
7167 Chunk::new(vec![
7168 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
7169 .expect("integers"),
7170 ])
7171 .expect("one column")
7172 }
7173
7174 #[test]
7175 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
7176 let path = path("nulls_for_the_planner");
7179 let mut writer =
7180 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
7181 .expect("new file");
7182 let rows = Chunk::new(vec![
7183 Vector::from_values(
7184 LogicalType::Integer,
7185 &[
7186 Value::Integer(4),
7187 Value::Null,
7188 Value::Integer(9),
7189 Value::Null,
7190 Value::Integer(1),
7191 Value::Integer(2),
7192 ],
7193 )
7194 .expect("integers"),
7195 ])
7196 .expect("one column");
7197 writer.append(&rows).expect("the only part");
7198 writer.finish().expect("commit");
7199 let reader = Reader::open(&path).expect("reopen from disk");
7200 let stripes = Stripes::new(reader);
7201 let column = stripes.column("a").expect("the file has that column");
7202 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
7203 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
7206 fs::remove_file(&path).expect("clean up");
7207 }
7208
7209 #[test]
7210 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
7211 let path = path("frequencies_for_the_planner");
7216 let mut writer =
7217 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7218 .expect("new file");
7219 let rows = Chunk::new(vec![
7220 Vector::from_values(
7221 LogicalType::Integer,
7222 &[
7223 Value::Integer(4),
7224 Value::Integer(4),
7225 Value::Integer(4),
7226 Value::Integer(9),
7227 Value::Integer(9),
7228 Value::Integer(1),
7229 ],
7230 )
7231 .expect("integers"),
7232 ])
7233 .expect("one column");
7234 writer.append(&rows).expect("the only part");
7235 writer.finish().expect("commit");
7236 let reader = Reader::open(&path).expect("reopen from disk");
7237 let common = Common::new(reader);
7238 assert_eq!(common.rows(), 6);
7239 let column = common.column("id").expect("the file has that column");
7240 assert_eq!(common.column("nothing"), None);
7241 assert_eq!(
7242 common.rows_with(column, &Bound::Int(4)),
7243 Stat::exact(3, Provenance::FrequencySynopsis)
7244 );
7245 assert_eq!(
7247 common.rows_with(column, &Bound::Int(7)),
7248 Stat::exact(0, Provenance::FrequencySynopsis)
7249 );
7250 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
7253 assert_eq!(common.remainder(column), None);
7256 fs::remove_file(&path).expect("clean up");
7257 }
7258
7259 fn bare_table(sections: Vec<Section>) -> Table {
7264 Table {
7265 name: "linked".to_owned(),
7266 fields: vec![Field::required("id", LogicalType::Integer)],
7267 stripes: Vec::new(),
7268 rows: 0,
7269 dictionaries: vec![None],
7270 distincts: vec![None],
7271 frequencies: vec![None],
7272 clustering: None,
7273 generation: 1,
7274 sections,
7275 }
7276 }
7277
7278 fn a_key_map_section() -> Section {
7279 Section {
7280 kind: *section::KEY_MAP,
7281 id: 1,
7282 generation: 3,
7283 extents: 1,
7284 extent_page: HEADER,
7285 extent_bytes: section::EXTENT_BYTES as u32,
7286 hash: 0x1234_5678_9abc_def0,
7287 flags: 0,
7288 header_bytes: 24,
7289 }
7290 }
7291
7292 #[test]
7293 fn a_section_table_round_trips_through_a_directory() {
7294 let mut later = a_key_map_section();
7295 later.kind = *b"RUDBZZ9\0";
7296 later.id = 2;
7297 let table = bare_table(vec![a_key_map_section(), later]);
7298 let directory = encode_directory(&table).expect("directory");
7299 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7300 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
7301 assert!(decoded.sections()[0].known());
7305 assert!(!decoded.sections()[1].known());
7306 }
7307
7308 #[test]
7309 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
7310 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7314 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
7315 let older = &directory[..directory.len() - block];
7316 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
7317 assert!(decoded.sections().is_empty());
7318 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
7319 assert_eq!(decoded.name(), "linked");
7320 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
7321 }
7322
7323 #[test]
7324 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
7325 let path = path("format_twenty_two");
7332 let mut writer =
7333 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7334 .expect("new file");
7335 let rows = Chunk::new(vec![
7336 Vector::from_values(
7337 LogicalType::Integer,
7338 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
7339 )
7340 .expect("integers"),
7341 ])
7342 .expect("one column");
7343 writer.append(&rows).expect("the only part");
7344 writer.finish().expect("commit");
7345
7346 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7347 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7348 drop(file);
7349
7350 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
7351 assert_eq!(reader.table().rows(), 3);
7352 assert!(reader.table().sections().is_empty());
7353
7354 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7357 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
7358 drop(file);
7359 let error = Reader::open(&path).expect_err("format 21 is not readable");
7360 assert!(error.to_string().contains("format 21"), "{error}");
7361
7362 fs::remove_file(&path).expect("clean up");
7363 }
7364
7365 #[test]
7366 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
7367 let mut past = a_key_map_section();
7372 past.extent_page = 1 << 30;
7373 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
7374 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
7375 assert!(error.to_string().contains("outside the file"), "{error}");
7376
7377 let mut inside_the_header = a_key_map_section();
7378 inside_the_header.extent_page = 8;
7379 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
7380 assert!(
7381 decode_directory(&directory, 1 << 20).is_err(),
7382 "a section may not overlap a header"
7383 );
7384 }
7385
7386 #[test]
7387 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
7388 let not_built = Section {
7392 kind: *section::FORWARD_LINK,
7393 id: 9,
7394 generation: 3,
7395 extents: 0,
7396 extent_page: 0,
7397 extent_bytes: 0,
7398 hash: 0,
7399 flags: 0,
7400 header_bytes: 0,
7401 };
7402 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
7403 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7404 assert_eq!(decoded.sections(), &[not_built]);
7405
7406 let mut incoherent = not_built;
7409 incoherent.extent_bytes = 28;
7410 incoherent.extent_page = HEADER;
7411 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
7412 assert!(decode_directory(&directory, 1 << 20).is_err());
7413 }
7414
7415 #[test]
7416 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
7417 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7418 let mut torn = directory.clone();
7419 let count_at = torn.len() - size_of::<u16>();
7420 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
7421 assert!(decode_directory(&torn, 1 << 20).is_err());
7424 }
7425
7426 fn linked_file(label: &str, rows: i32) -> PathBuf {
7428 let path = path(label);
7429 let mut writer =
7430 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7431 .expect("new file");
7432 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
7433 let chunk =
7434 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
7435 .expect("one column");
7436 writer.append(&chunk).expect("the only part");
7437 writer.finish().expect("commit");
7438 path
7439 }
7440
7441 fn a_key_map_payload() -> Vec<u8> {
7442 (0..512_u32).flat_map(u32::to_le_bytes).collect()
7445 }
7446
7447 #[test]
7448 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
7449 let path = linked_file("attach", 64);
7450 let payload = a_key_map_payload();
7451 let table = attach(
7452 &path,
7453 "items",
7454 &[section::Attachment {
7455 kind: *section::KEY_MAP,
7456 id: 0,
7457 flags: 2,
7458 header_bytes: 40,
7459 bytes: &payload,
7460 }],
7461 )
7462 .expect("attach a key map");
7463 assert_eq!(table.sections().len(), 1);
7464
7465 let reader = Reader::open(&path).expect("reopen after the attach");
7466 let held = reader.table().sections();
7467 assert_eq!(held.len(), 1);
7468 assert_eq!(held[0].kind, *section::KEY_MAP);
7469 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
7470 assert_eq!(held[0].header_bytes, 40);
7471 assert_eq!(held[0].generation, 1);
7475 assert!(held[0].usable(reader.table().generation()));
7476 assert_eq!(reader.payload(&held[0]).expect("read the payload"), payload);
7477 assert_eq!(reader.extents(&held[0]).expect("extent table").len(), 1);
7478
7479 fs::remove_file(&path).expect("clean up");
7480 }
7481
7482 #[test]
7483 fn attaching_a_section_answers_every_row_exactly_as_before() {
7484 let path = linked_file("attach_changes_nothing", 300);
7489 let before = Reader::open(&path).expect("open before");
7490 let rows = before.table().rows();
7491 let first = before.read(0, &[0]).expect("read before");
7492 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
7493 let layout = before.layout().columns_total();
7494 drop(before);
7495
7496 let payload = a_key_map_payload();
7497 attach(
7498 &path,
7499 "items",
7500 &[section::Attachment {
7501 kind: *section::KEY_MAP,
7502 id: 0,
7503 flags: 0,
7504 header_bytes: 0,
7505 bytes: &payload,
7506 }],
7507 )
7508 .expect("attach");
7509
7510 let after = Reader::open(&path).expect("open after");
7511 assert_eq!(after.table().rows(), rows);
7512 let read = after.read(0, &[0]).expect("read after");
7513 for (at, value) in values.iter().enumerate() {
7514 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
7515 }
7516 assert_eq!(
7517 after.layout().columns_total(),
7518 layout,
7519 "an attach appends and does not rewrite a column page"
7520 );
7521
7522 fs::remove_file(&path).expect("clean up");
7523 }
7524
7525 #[test]
7526 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
7527 let path = linked_file("attach_twice", 32);
7531 let one = a_key_map_payload();
7532 let two = vec![7_u8; 1024];
7533 let entry = |bytes| section::Attachment {
7534 kind: *section::KEY_MAP,
7535 id: 4,
7536 flags: 1,
7537 header_bytes: 0,
7538 bytes,
7539 };
7540 attach(&path, "items", &[entry(&one)]).expect("first build");
7541 attach(&path, "items", &[entry(&two)]).expect("rebuild");
7542
7543 let reader = Reader::open(&path).expect("reopen");
7544 let held = reader.table().sections();
7545 assert_eq!(held.len(), 1, "one map per column and not one per build");
7546 assert_eq!(reader.payload(&held[0]).expect("payload"), two);
7547
7548 fs::remove_file(&path).expect("clean up");
7549 }
7550
7551 #[test]
7552 fn an_attach_carries_through_a_kind_it_does_not_know() {
7553 let path = linked_file("attach_unknown", 16);
7557 let payload = vec![3_u8; 96];
7558 attach(
7559 &path,
7560 "items",
7561 &[section::Attachment {
7562 kind: *b"RUDBZZ9\0",
7563 id: 1,
7564 flags: 0,
7565 header_bytes: 0,
7566 bytes: &payload,
7567 }],
7568 )
7569 .expect("a kind this build does not know still writes");
7570 let key_map = a_key_map_payload();
7571 attach(
7572 &path,
7573 "items",
7574 &[section::Attachment {
7575 kind: *section::KEY_MAP,
7576 id: 0,
7577 flags: 0,
7578 header_bytes: 0,
7579 bytes: &key_map,
7580 }],
7581 )
7582 .expect("attach beside it");
7583
7584 let reader = Reader::open(&path).expect("reopen");
7585 let held = reader.table().sections();
7586 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
7587 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
7588 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
7589
7590 fs::remove_file(&path).expect("clean up");
7591 }
7592
7593 #[test]
7594 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
7595 let path = linked_file("attach_not_built", 8);
7596 attach(
7597 &path,
7598 "items",
7599 &[section::Attachment {
7600 kind: *section::FORWARD_LINK,
7601 id: 2,
7602 flags: 0,
7603 header_bytes: 0,
7604 bytes: &[],
7605 }],
7606 )
7607 .expect("record a link that did not fit the budget");
7608
7609 let reader = Reader::open(&path).expect("reopen");
7610 let held = reader.table().sections();
7611 assert_eq!(held.len(), 1);
7612 assert_eq!(held[0].extents, 0);
7613 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
7614 assert!(reader.extents(&held[0]).expect("no extent table").is_empty());
7615 assert!(reader.payload(&held[0]).expect("no payload").is_empty());
7616
7617 fs::remove_file(&path).expect("clean up");
7618 }
7619
7620 #[test]
7621 fn a_payload_past_one_extent_is_split_and_joined_back() {
7622 let path = linked_file("attach_two_extents", 8);
7626 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
7627 attach(
7628 &path,
7629 "items",
7630 &[section::Attachment {
7631 kind: *section::KEY_MAP,
7632 id: 0,
7633 flags: 0,
7634 header_bytes: 0,
7635 bytes: &payload,
7636 }],
7637 )
7638 .expect("attach a payload past the bound");
7639
7640 let reader = Reader::open(&path).expect("reopen");
7641 let held = reader.table().sections();
7642 let extents = reader.extents(&held[0]).expect("extent table");
7643 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
7644 assert_eq!(extents[0].length, section::MAX_EXTENT);
7645 assert_eq!(extents[1].length, 1);
7646 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
7647 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
7649 assert_eq!(reader.payload(&held[0]).expect("the whole payload").len(), payload.len());
7650
7651 fs::remove_file(&path).expect("clean up");
7652 }
7653
7654 #[test]
7655 fn a_torn_extent_is_refused_rather_than_decoded() {
7656 let path = linked_file("attach_torn", 8);
7657 let payload = a_key_map_payload();
7658 attach(
7659 &path,
7660 "items",
7661 &[section::Attachment {
7662 kind: *section::KEY_MAP,
7663 id: 0,
7664 flags: 0,
7665 header_bytes: 0,
7666 bytes: &payload,
7667 }],
7668 )
7669 .expect("attach");
7670
7671 let reader = Reader::open(&path).expect("reopen");
7672 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
7673 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
7674 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
7675 drop(file);
7676
7677 let reader = Reader::open(&path).expect("the table still opens");
7678 let error = reader
7679 .payload(&reader.table().sections()[0])
7680 .expect_err("a corrupt payload is not handed out");
7681 assert!(error.to_string().contains("checksum"), "{error}");
7682 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
7685
7686 fs::remove_file(&path).expect("clean up");
7687 }
7688
7689 #[test]
7690 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
7691 let path = linked_file("attach_old_format", 8);
7694 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7695 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7696 drop(file);
7697
7698 let payload = a_key_map_payload();
7699 let error = attach(
7700 &path,
7701 "items",
7702 &[section::Attachment {
7703 kind: *section::KEY_MAP,
7704 id: 0,
7705 flags: 0,
7706 header_bytes: 0,
7707 bytes: &payload,
7708 }],
7709 )
7710 .expect_err("format 22 cannot gain a section");
7711 assert!(error.to_string().contains("format 22"), "{error}");
7712 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
7713
7714 fs::remove_file(&path).expect("clean up");
7715 }
7716
7717 #[test]
7718 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
7719 let path = linked_file("attach_bad_header", 8);
7720 let error = attach(
7721 &path,
7722 "items",
7723 &[section::Attachment {
7724 kind: *section::KEY_MAP,
7725 id: 0,
7726 flags: 0,
7727 header_bytes: 40,
7728 bytes: &[1, 2, 3],
7729 }],
7730 )
7731 .expect_err("a writer's bug stops at the write");
7732 assert!(error.to_string().contains("header is longer"), "{error}");
7733
7734 fs::remove_file(&path).expect("clean up");
7735 }
7736
7737 #[test]
7738 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
7739 let path = linked_file("attach_wrong_name", 8);
7740 let error = attach(&path, "orders", &[]).expect_err("no such table");
7741 assert!(error.to_string().contains("orders"), "{error}");
7742 fs::remove_file(&path).expect("clean up");
7743 }
7744
7745 #[test]
7746 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
7747 let path = path("frequency_prefix_for_the_planner");
7754 let mut writer =
7755 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7756 .expect("new file");
7757 let mut values = vec![Value::Integer(1); 10_000];
7758 for _ in 0..10 {
7759 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
7760 }
7761 for part in values.chunks(8_000) {
7764 let rows = Chunk::new(vec![
7765 Vector::from_values(LogicalType::Integer, part).expect("integers"),
7766 ])
7767 .expect("one column");
7768 writer.append(&rows).expect("a part");
7769 }
7770 writer.finish().expect("commit");
7771 let reader = Reader::open(&path).expect("reopen from disk");
7772 let prefix =
7773 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
7774 assert_eq!(prefix.entries.len(), 512);
7777 assert_eq!(prefix.omitted_max, 10);
7778 let common = Common::new(reader);
7779 assert_eq!(common.rows(), 16_000);
7780 let column = common.column("id").expect("the file has that column");
7781 assert_eq!(
7782 common.rows_with(column, &Bound::Int(1)),
7783 Stat::exact(10_000, Provenance::FrequencySynopsis)
7784 );
7785 assert_eq!(
7787 common.rows_with(column, &Bound::Int(1_100)),
7788 Stat::exact(10, Provenance::FrequencySynopsis)
7789 );
7790 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
7793 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
7796 let remainder = common.remainder(column).expect("the list is a prefix");
7800 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
7801 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
7802 fs::remove_file(&path).expect("clean up");
7803 }
7804
7805 #[test]
7807 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
7808 let path = path("empty");
7809 Writer::empty(&path, &[]).expect("a file with nothing in it");
7810 let catalog = Catalog::open(&path).expect("the empty file opens");
7811 assert_eq!(catalog.len(), 0);
7812 assert!(catalog.is_empty());
7813 assert_eq!(catalog.names().count(), 0);
7814 let mut writer =
7817 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7818 .expect("a table goes into the empty file");
7819 writer.append(&sample_ids()).expect("rows");
7820 writer.finish().expect("commit");
7821 let catalog = Catalog::open(&path).expect("the file opens again");
7822 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7823 fs::remove_file(&path).expect("clean up");
7824 }
7825
7826 #[test]
7836 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
7837 let path = path("empty-name");
7838 let field = || vec![Field::required("id", LogicalType::Integer)];
7839 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
7840 let catalog = Catalog::open(&path).expect("the file opens");
7841 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
7842
7843 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
7844 writer.append(&sample_ids()).expect("rows");
7845 writer.finish().expect("commit");
7846 let catalog = Catalog::open(&path).expect("the file opens again");
7847 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7849 let held = catalog.rows().collect::<Vec<_>>();
7850 assert_eq!(held.len(), 1);
7851 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
7852
7853 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
7855 assert!(error.to_string().contains("same name"), "{error}");
7856 fs::remove_file(&path).expect("clean up");
7857 }
7858
7859 fn sample_view(name: &str) -> ViewEntry {
7861 ViewEntry {
7862 name: name.to_string(),
7863 sql: "SELECT id FROM items WHERE id > 0".to_string(),
7864 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
7865 aliases: vec!["n".to_string()],
7866 columns: vec![Field::new("n", LogicalType::Integer)],
7867 }
7868 }
7869
7870 #[test]
7871 fn a_view_written_into_the_catalog_comes_back_whole() {
7872 let path = path("views");
7873 let mut writer =
7874 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7875 .expect("new file");
7876 writer.append(&sample_ids()).expect("rows");
7877 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7878 let catalog = Catalog::open(&path).expect("reopen");
7879 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
7880 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7883 fs::remove_file(&path).expect("clean up");
7884 }
7885
7886 #[test]
7888 fn appending_a_table_carries_the_views_forward() {
7889 let path = path("viewscarry");
7890 let mut writer =
7891 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7892 .expect("new file");
7893 writer.append(&sample_ids()).expect("rows");
7894 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7895 let mut writer =
7896 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
7897 .expect("a second table");
7898 writer.append(&sample_ids()).expect("rows");
7899 writer.finish().expect("commit");
7900 let catalog = Catalog::open(&path).expect("reopen");
7901 assert_eq!(catalog.views().count(), 1);
7902 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
7903 fs::remove_file(&path).expect("clean up");
7904 }
7905
7906 #[test]
7908 fn restating_the_views_leaves_every_table_where_it_was() {
7909 let path = path("restate");
7910 let mut writer =
7911 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7912 .expect("new file");
7913 writer.append(&sample_ids()).expect("rows");
7914 writer.finish().expect("commit");
7915 let before = fs::metadata(&path).expect("the file is there").len();
7916 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
7917 let catalog = Catalog::open(&path).expect("reopen");
7918 assert_eq!(catalog.views().count(), 2);
7919 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7920 let after = fs::metadata(&path).expect("the file is there").len();
7923 assert!(after > before, "a generation was written");
7924 assert!(after - before < before, "the table was not written again");
7925 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
7928 assert_eq!(reader.table().rows, 3);
7929 Writer::restate(&path, &[]).expect("no views at all");
7932 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
7933 fs::remove_file(&path).expect("clean up");
7934 }
7935
7936 #[test]
7938 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
7939 let bytes = encode_catalog(
7940 &[Entry {
7941 name: "items".to_string(),
7942 fields: vec![Field::required("id", LogicalType::Integer)],
7943 rows: 1,
7944 directory: Page { offset: HEADER, length: 8, hash: 0 },
7945 }],
7946 &[sample_view("items")],
7947 )
7948 .expect("it encodes, because encoding does not look");
7949 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
7950 assert!(error.to_string().contains("same name"), "{error}");
7951 }
7952
7953 #[test]
7954 fn committed_file_reopens_and_reads_only_requested_columns() {
7955 let path = path("reopen");
7956 let mut writer = Writer::create(
7957 &path,
7958 "items",
7959 vec![
7960 Field::required("id", LogicalType::Integer),
7961 Field::new("text", LogicalType::Varchar),
7962 ],
7963 )
7964 .expect("new file");
7965 writer.append(&sample()).expect("first part");
7966 writer.append(&sample()).expect("second part");
7967 writer.finish().expect("commit");
7968 let reader = Reader::open(&path).expect("reopen from disk");
7969 assert_eq!(reader.table().rows(), 6);
7970 assert_eq!(reader.table().stripes().len(), 1);
7973 assert_eq!(reader.parts(), 2);
7974 assert_eq!(reader.part_rows(0), 3);
7975 assert_eq!(reader.part_rows(1), 3);
7976 let text = reader.read(1, &[1]).expect("only text page");
7977 assert_eq!(text.width(), 1);
7978 assert_eq!(text.value_at(1, 0), Value::Null);
7979 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7980 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
7981 assert_eq!(sparse.width(), 1);
7982 assert_eq!(sparse.value_at(1, 0), Value::Null);
7983 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7984 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
7985 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
7986 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
7987 let count = reader.read(0, &[]).expect("no page is needed for count");
7988 assert_eq!(count.len(), 3);
7989 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
7990 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
7991 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
7992 assert_eq!(
7993 integers,
7994 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
7995 );
7996 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
7997 assert_eq!(strings.len(), 3);
7998 assert!(strings.contains(&(Value::Null, 2)));
7999 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
8000 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
8001 fs::remove_file(path).expect("remove scratch file");
8002 }
8003
8004 #[test]
8012 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
8013 let path = path("interleaved-runs");
8014 let mut writer =
8015 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
8016 .expect("new file");
8017 for morsel in [2_u64, 0, 3, 1] {
8018 let parts = (0..4_u64)
8019 .map(|chunk| {
8020 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
8021 let values =
8022 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
8023 let column =
8024 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
8025 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
8026 })
8027 .collect::<Vec<_>>();
8028 writer.append_stripe(parts).expect("a stripe");
8029 }
8030 writer.finish().expect("commit");
8031
8032 let reader = Reader::open(&path).expect("valid directory");
8033 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
8034 assert_eq!(reader.table().rows(), 128);
8035 for part in 0..16_usize {
8036 let read = reader.read(part, &[0]).expect("a part back");
8037 for row in 0..8_usize {
8038 let want = i64::try_from(part * 8 + row).expect("small");
8039 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
8040 }
8041 }
8042 fs::remove_file(path).expect("remove scratch file");
8043 }
8044
8045 #[test]
8048 fn runs_that_overlap_each_other_are_refused_at_commit() {
8049 let path = path("overlapping-runs");
8050 let mut writer =
8051 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
8052 .expect("new file");
8053 let one = |order: (u64, u64)| {
8054 let column =
8055 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
8056 (order, Chunk::new(vec![column]).expect("one column"))
8057 };
8058 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
8061 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
8062 let error = writer.finish().expect_err("the runs overlap");
8063 assert!(error.message().contains("source order"), "{error}");
8064 fs::remove_file(path).expect("remove scratch file");
8065 }
8066
8067 #[test]
8070 fn a_run_longer_than_a_stripe_is_refused() {
8071 let path = path("overlong-run");
8072 let mut writer =
8073 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
8074 .expect("new file");
8075 let parts = (0..=STRIPE_PARTS)
8076 .map(|at| {
8077 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
8078 .expect("a column");
8079 let chunk = Chunk::new(vec![column]).expect("one column");
8080 ((0, u64::try_from(at).expect("small")), chunk)
8081 })
8082 .collect::<Vec<_>>();
8083 let error = writer.append_stripe(parts).expect_err("one part too many");
8084 assert!(error.message().contains("more parts than it holds"), "{error}");
8085 fs::remove_file(path).expect("remove scratch file");
8086 }
8087
8088 #[test]
8094 fn parts_past_the_stripe_bound_start_a_new_stripe() {
8095 let path = path("stripe-bound");
8096 let mut writer = Writer::create(
8097 &path,
8098 "items",
8099 vec![
8100 Field::required("id", LogicalType::Integer),
8101 Field::new("text", LogicalType::Varchar),
8102 ],
8103 )
8104 .expect("new file");
8105 let parts = STRIPE_PARTS * 2 + 3;
8106 for part in 0..parts {
8107 let id = part as i32;
8108 let chunk = Chunk::new(vec![
8109 Vector::from_values(
8110 LogicalType::Integer,
8111 &[Value::Integer(id), Value::Integer(-id)],
8112 )
8113 .expect("integers"),
8114 Vector::from_values(
8115 LogicalType::Varchar,
8116 &[Value::Varchar(format!("value {part}")), Value::Null],
8117 )
8118 .expect("strings"),
8119 ])
8120 .expect("matching rows");
8121 writer.append(&chunk).expect("one part");
8122 }
8123 writer.finish().expect("commit");
8124
8125 let reader = Reader::open(&path).expect("reopen from disk");
8126 assert_eq!(reader.parts(), parts);
8127 assert_eq!(reader.table().rows(), parts * 2);
8128 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
8129 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
8130 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
8131 assert_eq!(reader.table().stripes()[2].parts(), 3);
8132 for part in (0..parts).rev() {
8135 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
8136 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
8137 for chunk in [&dense, &sparse] {
8138 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
8139 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8140 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8141 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
8142 assert_eq!(chunk.value_at(1, 1), Value::Null);
8143 }
8144 }
8145 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
8148 assert!(reader.skips(0, &above), "the first stripe stops at 63");
8149 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
8150 fs::remove_file(path).expect("remove scratch file");
8151 }
8152
8153 fn scattered(n: i64) -> i64 {
8155 n.wrapping_mul(-7_046_029_254_386_353_131)
8156 }
8157
8158 #[test]
8164 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
8165 let path = path("sieve-skip");
8166 let mut writer =
8167 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8168 .expect("new file");
8169 let parts = STRIPE_PARTS + 3;
8170 let per_part = 128;
8174 for part in 0..parts {
8175 let held: Vec<Value> = (0..per_part)
8176 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
8177 .collect();
8178 let chunk =
8179 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8180 .expect("one column");
8181 writer.append(&chunk).expect("one part");
8182 }
8183 writer.finish().expect("commit");
8184
8185 let reader = Reader::open(&path).expect("reopen from disk");
8186 let probe = |value: i64| Probe {
8187 column: 0,
8188 op: Op::Equal,
8189 value: Bound::Int(i128::from(scattered(value))),
8190 };
8191 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
8192 let tests = [probe(wanted)];
8193 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
8194 let home = wanted as usize / per_part;
8195 assert!(kept.contains(&home), "the part holding {wanted} is read");
8196 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
8200 }
8201 let absent = [probe((parts * per_part) as i64 + 1)];
8202 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
8203 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
8204 let tests = [probe(0)];
8207 assert!(
8208 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
8209 "the bounds rule out no stripe at all"
8210 );
8211 fs::remove_file(path).expect("remove scratch file");
8212 }
8213
8214 #[test]
8220 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
8221 let path = path("part-range-skip");
8222 let mut writer =
8223 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8224 .expect("new file");
8225 let parts = STRIPE_PARTS + 3;
8226 let per_part = 128;
8227 for part in 0..parts {
8228 let held: Vec<Value> = (0..per_part)
8232 .map(|row| {
8233 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8234 })
8235 .collect();
8236 let chunk =
8237 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8238 .expect("one column");
8239 writer.append(&chunk).expect("one part");
8240 }
8241 writer.finish().expect("commit");
8242
8243 let reader = Reader::open(&path).expect("reopen from disk");
8244 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8245 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
8246 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
8247 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
8249 fs::remove_file(path).expect("remove scratch file");
8250 }
8251
8252 #[test]
8256 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
8257 let path = path("part-range-certain");
8258 let mut writer =
8259 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8260 .expect("new file");
8261 let parts = STRIPE_PARTS + 3;
8262 let per_part = 128;
8263 for part in 0..parts {
8264 let held: Vec<Value> = (0..per_part)
8265 .map(|row| {
8266 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8267 })
8268 .collect();
8269 let chunk =
8270 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8271 .expect("one column");
8272 writer.append(&chunk).expect("one part");
8273 }
8274 writer.finish().expect("commit");
8275
8276 let reader = Reader::open(&path).expect("reopen from disk");
8277 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8278 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
8279 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
8280 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
8283 fs::remove_file(path).expect("remove scratch file");
8284 }
8285
8286 #[test]
8289 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
8290 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
8291 let path = path("part-range-page");
8292 let mut writer =
8293 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8294 .expect("new file");
8295 for part in 0..parts {
8296 let held: Vec<Value> = (0..128)
8297 .map(|row| {
8298 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
8299 })
8300 .collect();
8301 let chunk = Chunk::new(vec![
8302 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
8303 ])
8304 .expect("one column");
8305 writer.append(&chunk).expect("one part");
8306 }
8307 writer.finish().expect("commit");
8308 let reader = Reader::open(&path).expect("reopen from disk");
8309 let bytes = reader.layout().columns[0].part_ranges;
8310 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
8311 fs::remove_file(path).expect("remove scratch file");
8312 }
8313 }
8314
8315 #[test]
8318 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
8319 let long = vec![b'a'; PART_BOUND_BYTES * 2];
8320 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
8321 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
8322 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
8323 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
8324 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
8325 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
8326 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
8327 }
8328
8329 #[test]
8332 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
8333 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
8334 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
8335 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
8336 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
8337 }
8338
8339 #[test]
8351 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
8352 let parts = 4;
8353 let per_part = 1024;
8354 let rows = parts * per_part;
8355 let written = |name: &str, keys: &[i64]| {
8356 let path = path(name);
8357 let fields = vec![Field::required("key", LogicalType::BigInt)];
8358 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
8359 for part in 0..parts {
8360 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
8361 .iter()
8362 .map(|key| Value::BigInt(*key))
8363 .collect();
8364 let chunk = Chunk::new(vec![
8365 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
8366 ])
8367 .expect("one column");
8368 writer.append(&chunk).expect("one part");
8369 }
8370 writer.finish().expect("commit");
8371 path
8372 };
8373 let climbing = |step: &dyn Fn(usize) -> i64| {
8376 let mut key = 0;
8377 (0..rows)
8378 .map(|row| {
8379 key += step(row);
8380 key
8381 })
8382 .collect::<Vec<i64>>()
8383 };
8384 let ascending = climbing(&|row| (row % 3) as i64);
8385 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
8389 let near_path = written("stored-near", &ascending);
8390 let far_path = written("stored-far", &sparse);
8391
8392 let one = Reader::open(&near_path).expect("reopen from disk");
8393 let other = Reader::open(&far_path).expect("reopen from disk");
8394 let near = one.stored(0).expect("the column is stored");
8395 let far = other.stored(0).expect("the column is stored");
8396 assert_eq!(near.len(), parts, "one row per part");
8397 assert_eq!(far.len(), parts);
8398 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
8401 assert_eq!(total(&near), one.layout().columns[0].pages);
8402 assert_eq!(total(&far), other.layout().columns[0].pages);
8403 assert!(
8404 total(&near) * 2 < total(&far),
8405 "the sparse keys cost more, {} against {}",
8406 total(&far),
8407 total(&near)
8408 );
8409 for (at, part) in near.iter().enumerate() {
8411 assert_eq!(part.part, at);
8412 assert_eq!(part.row, at * per_part);
8413 assert_eq!(part.rows, per_part);
8414 let held = &ascending[at * per_part..(at + 1) * per_part];
8415 assert_eq!(part.low, Some(Value::BigInt(held[0])));
8416 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
8417 assert_eq!(part.nulls, Some(0));
8418 }
8419 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
8422 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
8423 assert_ne!(near[0].encoding, far[0].encoding);
8424 fs::remove_file(near_path).expect("remove scratch file");
8425 fs::remove_file(far_path).expect("remove scratch file");
8426 }
8427
8428 #[test]
8438 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
8439 let path = path("sieve-pays");
8440 let fields = vec![
8441 Field::required("spread", LogicalType::BigInt),
8442 Field::required("repeated", LogicalType::BigInt),
8443 ];
8444 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
8445 let parts = 3;
8446 let per_part = 1024;
8447 for part in 0..parts {
8448 let base = (part * per_part) as i64;
8449 let spread: Vec<Value> =
8450 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
8451 let repeated: Vec<Value> =
8452 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
8453 let chunk = Chunk::new(vec![
8454 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
8455 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
8456 ])
8457 .expect("two columns");
8458 writer.append(&chunk).expect("one part");
8459 }
8460 writer.finish().expect("commit");
8461
8462 let reader = Reader::open(&path).expect("reopen from disk");
8463 let layout = reader.layout();
8464 let spread = &layout.columns[0];
8465 let repeated = &layout.columns[1];
8466 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
8467 assert_eq!(
8468 repeated.sieves, 0,
8469 "a column whose filter costs more than its parts keeps none"
8470 );
8471 for column in &layout.columns {
8474 assert!(
8475 column.sieves < column.pages,
8476 "{} spends {} on sieves over {} of data",
8477 column.name,
8478 column.sieves,
8479 column.pages
8480 );
8481 }
8482 let absent = [Probe {
8484 column: 0,
8485 op: Op::Equal,
8486 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
8487 }];
8488 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
8489 fs::remove_file(path).expect("remove scratch file");
8490 }
8491
8492 #[test]
8498 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
8499 let path = path("sieve-damaged");
8500 let mut writer =
8501 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8502 .expect("new file");
8503 let rows = 128;
8504 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
8505 let chunk =
8506 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8507 .expect("one column");
8508 writer.append(&chunk).expect("one part");
8509 writer.finish().expect("commit");
8510
8511 let page =
8512 Reader::open(&path).expect("reopen").table.stripes[0].sieves[0].expect("a sieve page");
8513 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
8514 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
8515 file.write_all(&[0xff]).expect("damage one byte");
8516 drop(file);
8517
8518 let reader = Reader::open(&path).expect("reopen the damaged file");
8519 let absent =
8520 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
8521 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
8522 assert_eq!(
8523 reader.read(0, &[0]).expect("the rows are untouched").len(),
8524 usize::try_from(rows).expect("a small count")
8525 );
8526 fs::remove_file(path).expect("remove scratch file");
8527 }
8528
8529 #[test]
8540 fn workers_that_want_the_same_stripe_read_it_once() {
8541 let path = path("single-flight");
8542 let mut writer =
8543 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8544 .expect("new file");
8545 for part in 0..STRIPE_PARTS {
8546 let id = part as i32;
8547 let chunk = Chunk::new(vec![
8548 Vector::from_values(
8549 LogicalType::Integer,
8550 &[Value::Integer(id), Value::Integer(-id)],
8551 )
8552 .expect("integers"),
8553 ])
8554 .expect("matching rows");
8555 writer.append(&chunk).expect("one part");
8556 }
8557 writer.finish().expect("commit");
8558
8559 let reader = Reader::open(&path).expect("reopen from disk");
8560 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
8561 let barrier = std::sync::Barrier::new(8);
8562 std::thread::scope(|scope| {
8563 for worker in 0..8 {
8564 let reader = &reader;
8565 let barrier = &barrier;
8566 scope.spawn(move || {
8567 barrier.wait();
8568 for part in (worker..STRIPE_PARTS).step_by(8) {
8569 let chunk = reader.read(part, &[0]).expect("a whole page read");
8570 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8571 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8572 }
8573 });
8574 }
8575 });
8576 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
8577 fs::remove_file(path).expect("remove scratch file");
8578 }
8579
8580 #[test]
8593 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
8594 let opened = |label: &str, rows_per_part: i32| {
8595 let path = path(label);
8596 let mut writer =
8597 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8598 .expect("new file");
8599 for part in 0..STRIPE_PARTS * 3 {
8600 let values = (0..rows_per_part)
8604 .map(|row| {
8605 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
8606 })
8607 .collect::<Vec<_>>();
8608 let chunk = Chunk::new(vec![
8609 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
8610 ])
8611 .expect("matching rows");
8612 writer.append(&chunk).expect("one part");
8613 }
8614 writer.finish().expect("commit");
8615 let reader = Reader::open(&path).expect("reopen from disk");
8616 let size = fs::metadata(&path).expect("the file is there").len();
8617 let out = (reader.reads(), reader.table().stripes().len(), size);
8618 fs::remove_file(path).expect("remove scratch file");
8619 out
8620 };
8621
8622 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
8623 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
8624 assert_eq!(
8625 thin_stripes, fat_stripes,
8626 "the same stripe count is what makes this a fair ask"
8627 );
8628 assert!(
8629 fat_size > thin_size * 50,
8630 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
8631 );
8632
8633 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
8634 assert_eq!(thin.pages, 0, "opening read a page");
8635 assert_eq!(fat.pages, 0, "opening read a page");
8636 assert_eq!(thin.indexes, 0, "opening read an index");
8637 assert_eq!(fat.indexes, 0, "opening read an index");
8638 assert!(
8641 fat.opening.bytes < thin.opening.bytes * 2,
8642 "opening the thin file read {} bytes and the fat one read {}",
8643 thin.opening.bytes,
8644 fat.opening.bytes
8645 );
8646 }
8647
8648 #[test]
8656 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
8657 let path = path("open-twice");
8658 let mut writer =
8659 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8660 .expect("new file");
8661 for part in 0..STRIPE_PARTS * 3 {
8662 let chunk = Chunk::new(vec![
8663 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8664 .expect("integers"),
8665 ])
8666 .expect("matching rows");
8667 writer.append(&chunk).expect("one part");
8668 }
8669 writer.finish().expect("commit");
8670
8671 let first = Reader::open(&path).expect("open");
8672 for part in 0..first.parts() {
8675 first.read(part, &[0]).expect("a part");
8676 }
8677 assert!(first.reads().pages > 0, "the scan has to have read something");
8678 let second = Reader::open(&path).expect("open again");
8679
8680 assert_eq!(first.reads().opening, second.reads().opening);
8681 assert_eq!(
8682 second.reads().pages,
8683 0,
8684 "the second open read a page off the back of the first"
8685 );
8686 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
8687 fs::remove_file(path).expect("remove scratch file");
8688 }
8689
8690 #[test]
8698 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
8699 let path = path("index-cache");
8700 let mut writer =
8701 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8702 .expect("new file");
8703 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
8704 for part in 0..parts {
8705 let id = part as i32;
8706 let chunk = Chunk::new(vec![
8707 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
8708 ])
8709 .expect("matching rows");
8710 writer.append(&chunk).expect("one part");
8711 }
8712 writer.finish().expect("commit");
8713
8714 let reader = Reader::open(&path).expect("reopen from disk");
8715 let stripes = reader.table().stripes().len();
8716 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
8717 for _ in 0..2 {
8719 for part in 0..parts {
8720 let chunk = reader.read(part, &[0]).expect("a part");
8721 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8722 }
8723 }
8724 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
8725 assert!(
8726 reader.pages.load(Atomic::Relaxed) > stripes,
8727 "the pages are the ones that get read again, which is what makes the index count mean \
8728 something"
8729 );
8730 fs::remove_file(path).expect("remove scratch file");
8731 }
8732
8733 #[test]
8742 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
8743 let workers = CACHED_STRIPES_PER_COLUMN + 4;
8744 let path = path("stripe-per-worker");
8745 let mut writer =
8746 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8747 .expect("new file");
8748 for part in 0..STRIPE_PARTS * workers {
8749 let chunk = Chunk::new(vec![
8750 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8751 .expect("integers"),
8752 ])
8753 .expect("matching rows");
8754 writer.append(&chunk).expect("one part");
8755 }
8756 writer.finish().expect("commit");
8757
8758 let read = |told: bool| {
8759 let reader = Reader::open(&path).expect("reopen from disk");
8760 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
8761 if told {
8762 reader.keep_stripes(workers);
8763 }
8764 let barrier = std::sync::Barrier::new(workers);
8765 std::thread::scope(|scope| {
8766 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
8767 let reader = &reader;
8768 let barrier = &barrier;
8769 scope.spawn(move || {
8770 for part in run {
8771 barrier.wait();
8772 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
8773 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8774 }
8775 assert!(worker < workers);
8776 });
8777 }
8778 });
8779 reader.pages.load(Atomic::Relaxed)
8780 };
8781
8782 assert_eq!(read(true), workers, "one page read per stripe and no more");
8783 assert!(read(false) > workers, "a cache that small is read again on every part");
8784 fs::remove_file(path).expect("remove scratch file");
8785 }
8786
8787 #[test]
8792 fn a_damaged_index_page_is_an_error() {
8793 let path = path("damaged-index");
8794 let mut writer =
8795 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8796 .expect("new file");
8797 writer.append(&sample_ids()).expect("first part");
8798 writer.append(&sample_ids()).expect("second part");
8799 writer.finish().expect("commit");
8800
8801 let reader = Reader::open(&path).expect("valid directory");
8802 let index = reader.table.stripes[0].index;
8803 let mut byte = [0; 1];
8804 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
8805 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
8806 file.seek(SeekFrom::Start(index.offset)).expect("index start");
8807 file.write_all(&[!byte[0]]).expect("damage the first part length");
8808 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
8809 assert!(error.message().contains("index page section checksum differs"), "{error}");
8810 fs::remove_file(path).expect("remove scratch file");
8811 }
8812
8813 #[test]
8820 fn every_integer_width_round_trips_through_a_page() {
8821 let path = path("integer-widths");
8822 let columns = [
8823 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
8824 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
8825 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
8826 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
8827 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
8828 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
8829 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
8830 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
8831 ];
8832 let fields = columns
8833 .iter()
8834 .enumerate()
8835 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8836 .collect::<Vec<_>>();
8837 let vectors = columns
8838 .iter()
8839 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8840 .collect::<Vec<_>>();
8841 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
8842 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8843 writer.finish().expect("commit");
8844
8845 let reader = Reader::open(&path).expect("reopen from disk");
8846 let wanted = (0..columns.len()).collect::<Vec<_>>();
8847 let read = reader.read(0, &wanted).expect("every column");
8848 assert_eq!(read.len(), 2);
8849 for (at, (ty, values)) in columns.iter().enumerate() {
8851 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8852 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8853 }
8854 fs::remove_file(path).expect("remove scratch file");
8855 }
8856
8857 #[test]
8868 fn every_other_type_the_format_knows_round_trips_through_a_page() {
8869 let path = path("other-types");
8870 let columns = [
8871 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
8872 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
8873 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
8874 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
8875 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
8876 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
8877 (
8878 LogicalType::TimestampTz,
8879 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
8880 ),
8881 (
8882 LogicalType::Interval,
8883 vec![
8884 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
8885 Value::Interval { months: 13, days: -1, micros: 1 },
8886 ],
8887 ),
8888 (
8889 LogicalType::Blob,
8890 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
8891 ),
8892 ];
8893 let fields = columns
8894 .iter()
8895 .enumerate()
8896 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8897 .collect::<Vec<_>>();
8898 let vectors = columns
8899 .iter()
8900 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8901 .collect::<Vec<_>>();
8902 let mut writer = Writer::create(&path, "others", fields).expect("new file");
8903 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8904 writer.finish().expect("commit");
8905
8906 let reader = Reader::open(&path).expect("reopen from disk");
8907 let wanted = (0..columns.len()).collect::<Vec<_>>();
8908 let read = reader.read(0, &wanted).expect("every column");
8909 assert_eq!(read.len(), 2);
8910 for (at, (ty, values)) in columns.iter().enumerate() {
8911 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8912 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8913 }
8914 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
8917 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
8918
8919 fs::remove_file(path).expect("remove scratch file");
8920 }
8921
8922 #[test]
8928 fn a_nan_survives_being_written_down() {
8929 let path = path("nan");
8930 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
8931 .expect("a NaN vector");
8932 let mut writer =
8933 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
8934 .expect("new file");
8935 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
8936 writer.finish().expect("commit");
8937 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
8938 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
8939 assert!(back.is_nan(), "a NaN came back as {back}");
8940 fs::remove_file(path).expect("remove scratch file");
8941 }
8942
8943 #[test]
8950 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
8951 let path = path("uuid-and-bit");
8952 let uuids = vec![0_i128, i128::MIN, -1];
8953 let mut bits = StringColumn::new();
8954 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
8955 bits.push_bytes(value);
8956 }
8957 let expected = bits.clone();
8958 let fields =
8959 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
8960 let vectors = vec![
8961 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
8962 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
8963 ];
8964 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
8965 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8966 writer.finish().expect("commit");
8967
8968 let reader = Reader::open(&path).expect("reopen from disk");
8969 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
8970 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
8971 panic!("a uuid column is the 128 bit lane")
8972 };
8973 assert_eq!(back.as_slice(), uuids.as_slice());
8974 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
8975 panic!("a bit column is bytes")
8976 };
8977 for row in 0..expected.len() {
8978 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
8979 }
8980 fs::remove_file(path).expect("remove scratch file");
8981 }
8982
8983 #[test]
8984 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
8985 let path = path("frequency-ordinals");
8986 let mut writer =
8987 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
8988 .expect("new file");
8989 let mut values = Vec::new();
8990 for leader in 0..10_i64 {
8991 values.extend(std::iter::repeat_n(leader, 100));
8992 }
8993 values.extend(1_000_i64..41_000);
8994 for part in values.chunks(1_024) {
8995 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
8996 .expect("big integers");
8997 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
8998 }
8999 writer.finish().expect("commit");
9000
9001 let reader = Reader::open(&path).expect("reopen from disk");
9002 let occurrences =
9003 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
9004 assert!(occurrences.omitted_max < 100);
9005 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
9006 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
9007 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
9008 fs::remove_file(path).expect("remove scratch file");
9009 }
9010
9011 #[test]
9017 fn a_file_from_another_format_says_which_format_it_is() {
9018 let older = path("older-format");
9019 let mut writer =
9020 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
9021 .expect("new file");
9022 let chunk = Chunk::new(vec![
9023 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9024 .expect("integers"),
9025 ])
9026 .expect("chunk");
9027 writer.append(&chunk).expect("page written");
9028 writer.finish().expect("commit");
9029
9030 let unreadable =
9034 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
9035 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9036 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
9037 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
9038 drop(file);
9039 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
9040 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
9041 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
9042
9043 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9044 file.seek(SeekFrom::Start(0)).expect("the magic is first");
9045 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
9046 drop(file);
9047 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
9048 assert!(complaint.contains("magic"), "{complaint}");
9049 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
9050 fs::remove_file(older).expect("remove scratch file");
9051 }
9052
9053 #[test]
9054 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
9055 let unfinished = path("unfinished");
9056 let mut writer =
9057 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
9058 .expect("new file");
9059 let chunk = Chunk::new(vec![
9060 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9061 .expect("integers"),
9062 ])
9063 .expect("chunk");
9064 writer.append(&chunk).expect("page written");
9065 drop(writer);
9066 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
9067 fs::remove_file(unfinished).expect("remove scratch file");
9068
9069 let damaged = path("damaged");
9070 let mut writer =
9071 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
9072 .expect("new file");
9073 writer.append(&chunk).expect("page written");
9074 writer.finish().expect("commit");
9075 let reader = Reader::open(&damaged).expect("valid directory");
9076 let mut file =
9077 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
9078 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
9079 file.write_all(&[255]).expect("damage one byte");
9080 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
9081 fs::remove_file(damaged).expect("remove scratch file");
9082 }
9083
9084 #[test]
9085 fn damaged_lazy_dictionary_payload_is_an_error() {
9086 let path = path("damaged-dictionary");
9087 let mut writer = Writer::create(
9088 &path,
9089 "items",
9090 vec![
9091 Field::required("id", LogicalType::Integer),
9092 Field::new("text", LogicalType::Varchar),
9093 ],
9094 )
9095 .expect("new file");
9096 writer.append(&sample()).expect("stripe written");
9097 writer.finish().expect("commit");
9098
9099 let reader = Reader::open(&path).expect("valid directory");
9100 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
9101 let mut header = [0; DICTIONARY_HEADER];
9104 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
9105 let index_len = dictionary_index_len(&header);
9106 let rank_len = last_rank_end(&reader.file, dictionary.offset, &header);
9107 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9108 file.seek(SeekFrom::Start(dictionary.offset + index_len + rank_len))
9109 .expect("inside dictionary payload");
9110 file.write_all(&[255]).expect("damage dictionary payload");
9111
9112 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
9113 let error =
9114 chunk.validate_external().expect_err("payload corruption must reach the caller");
9115 assert!(error.message().contains("payload checksum differs"), "{error}");
9116 fs::remove_file(path).expect("remove scratch file");
9117 }
9118
9119 #[test]
9129 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
9130 let path = path("dictionary-decide");
9131 let rows = 20_000;
9132 let unique =
9134 |row: usize| format!("{row:09} a value that appears exactly once in the table");
9135 let repeated = |row: usize| unique(row / 40);
9137 let mut writer = Writer::create(
9138 &path,
9139 "items",
9140 vec![
9141 Field::required("unique", LogicalType::Varchar),
9142 Field::required("repeated", LogicalType::Varchar),
9143 ],
9144 )
9145 .expect("new file");
9146 for part in (0..rows).step_by(1_000) {
9147 let span = part..(part + 1_000).min(rows);
9148 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
9149 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
9150 writer
9151 .append(
9152 &Chunk::new(vec![
9153 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
9154 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
9155 ])
9156 .expect("two columns"),
9157 )
9158 .expect("a part");
9159 }
9160 writer.finish().expect("commit");
9161
9162 let reader = Reader::open(&path).expect("reopen from disk");
9163 assert!(
9164 reader.table.dictionaries[0].is_none(),
9165 "a column with no repeats has nothing to say twice"
9166 );
9167 assert!(
9168 reader.table.dictionaries[1].is_some(),
9169 "a column whose values come round again keeps its dictionary"
9170 );
9171 let mut first = 0;
9172 for part in 0..reader.parts() {
9173 let chunk = reader.read(part, &[0, 1]).expect("a part");
9174 for row in 0..chunk.len() {
9175 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
9176 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
9177 }
9178 first += chunk.len();
9179 }
9180 assert_eq!(first, rows, "every row was read back");
9181 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
9182 let size = fs::metadata(&path).expect("the file is there").len() as usize;
9183 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
9184 fs::remove_file(path).expect("remove scratch file");
9185 }
9186
9187 #[test]
9200 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
9201 let path = path("dictionary-blocks");
9202 let value = |row: usize| {
9203 let row = row.saturating_sub(8_000);
9204 format!("{row:07} a value long enough to be worth a payload block")
9205 };
9206 let parts = 40;
9207 let per_part = 1000;
9208 let mut writer =
9209 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9210 .expect("new file");
9211 for part in 0..parts {
9212 let values = (0..per_part)
9213 .map(|row| Value::Varchar(value(part * per_part + row)))
9214 .collect::<Vec<_>>();
9215 let chunk = Chunk::new(vec![
9216 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9217 ])
9218 .expect("matching rows");
9219 writer.append(&chunk).expect("a part");
9220 }
9221 writer.finish().expect("commit");
9222
9223 let reader = Reader::open(&path).expect("reopen from disk");
9224 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
9225 assert!(
9226 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
9227 "the dictionary has to be several blocks for this to be testing anything"
9228 );
9229 for part in [0, parts - 1] {
9230 let chunk = reader.read(part, &[0]).expect("a part");
9231 chunk.validate_external().expect("every payload block checks out");
9232 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
9233 }
9234
9235 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9236 file.seek(SeekFrom::Start(dictionary.offset + u64::from(dictionary.length) - 4))
9237 .expect("the last bytes of the page are payload");
9238 file.write_all(&[255]).expect("damage the last payload block");
9239 let reader = Reader::open(&path).expect("the directory and the index are untouched");
9240 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
9241 let error = chunk.validate_external().expect_err("the damage must reach the caller");
9242 assert!(error.message().contains("payload checksum differs"), "{error}");
9243 fs::remove_file(path).expect("remove scratch file");
9244 }
9245
9246 #[test]
9260 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
9261 let path = path("dictionary-offsets");
9262 let value = |row: usize| {
9263 let row = row % 5_000;
9264 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
9265 };
9266 let rows = 6_000;
9267 let mut writer =
9268 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9269 .expect("new file");
9270 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
9271 for part in values.chunks(1_000) {
9272 let chunk =
9273 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
9274 .expect("matching rows");
9275 writer.append(&chunk).expect("a part");
9276 }
9277 writer.finish().expect("commit");
9278
9279 let reader = Reader::open(&path).expect("reopen from disk");
9280 assert!(
9281 rows > TEXT_PAYLOAD_VALUES * 4,
9282 "the dictionary has to be several blocks for this to be testing anything"
9283 );
9284 for part in 0..rows / 1_000 {
9285 let chunk = reader.read(part, &[0]).expect("a part");
9286 for row in 0..1_000 {
9287 let row = part * 1_000 + row;
9288 assert_eq!(
9289 chunk.value_at(row % 1_000, 0),
9290 Value::Varchar(value(row)),
9291 "value {row}"
9292 );
9293 }
9294 }
9295 fs::remove_file(path).expect("remove scratch file");
9296 }
9297
9298 #[test]
9310 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
9311 let path = path("dictionary-once");
9312 let parts = 8;
9313 let per_part = 500;
9314 let value =
9315 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
9316 let mut writer =
9317 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9318 .expect("new file");
9319 for part in 0..parts {
9320 let values = (0..per_part)
9321 .map(|row| Value::Varchar(value(part * per_part + row)))
9322 .collect::<Vec<_>>();
9323 let chunk = Chunk::new(vec![
9324 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9325 ])
9326 .expect("matching rows");
9327 writer.append(&chunk).expect("a part");
9328 }
9329 writer.finish().expect("commit");
9330
9331 let reader = Reader::open(&path).expect("reopen from disk");
9332 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
9333 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
9334
9335 let workers = 16;
9336 let gate = std::sync::Barrier::new(workers);
9337 std::thread::scope(|scope| {
9338 for worker in 0..workers {
9339 let reader = reader.clone();
9340 let gate = &gate;
9341 scope.spawn(move || {
9342 gate.wait();
9343 let chunk = reader.read(worker % parts, &[0]).expect("a part");
9344 assert_eq!(
9345 chunk.value_at(0, 0),
9346 Value::Varchar(value((worker % parts) * per_part))
9347 );
9348 });
9349 }
9350 });
9351
9352 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
9353 fs::remove_file(path).expect("remove scratch file");
9354 }
9355
9356 #[test]
9361 fn a_damaged_sorted_order_is_an_error() {
9362 let path = path("damaged-order");
9363 let mut writer = Writer::create(
9364 &path,
9365 "items",
9366 vec![
9367 Field::required("id", LogicalType::Integer),
9368 Field::new("text", LogicalType::Varchar),
9369 ],
9370 )
9371 .expect("new file");
9372 writer.append(&sample()).expect("stripe written");
9373 writer.finish().expect("commit");
9374
9375 let reader = Reader::open(&path).expect("valid directory");
9376 let page = reader.table.dictionaries[1].expect("string dictionary page");
9377 let mut header = [0; DICTIONARY_HEADER];
9378 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
9379 let index_len = dictionary_index_len(&header);
9380 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9381 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
9382 file.write_all(&[255]).expect("damage the order");
9383
9384 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
9385 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
9386 assert!(error.message().contains("rank checksum differs"), "{error}");
9387 fs::remove_file(path).expect("remove scratch file");
9388 }
9389
9390 #[test]
9394 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
9395 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
9398 let path = path("dictionary-order");
9399 let mut writer =
9400 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9401 .expect("new file");
9402 writer
9403 .append(
9404 &Chunk::new(vec![
9405 Vector::from_values(
9406 LogicalType::Varchar,
9407 &spellings.map(|text| Value::Varchar(text.into())),
9408 )
9409 .expect("strings"),
9410 ])
9411 .expect("one column"),
9412 )
9413 .expect("stripe written");
9414 writer.finish().expect("commit");
9415
9416 let reader = Reader::open(&path).expect("valid directory");
9417 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9418 let count = dictionary.ranks().expect("a v10 file stores one");
9419 assert_eq!(count, spellings.len(), "every distinct value has a rank");
9420 let order = (0..count)
9421 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
9422 .collect::<Vec<_>>();
9423 let mut seen = order.clone();
9424 seen.sort_unstable();
9425 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
9426
9427 let ranked = order
9428 .iter()
9429 .map(|&code| {
9430 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
9431 })
9432 .collect::<Vec<_>>();
9433 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
9434 expected.sort();
9435 assert_eq!(ranked, expected, "rank order is value order");
9436
9437 for (rank, value) in expected.iter().enumerate() {
9440 assert_eq!(
9441 dictionary.compare_rank(rank, value).expect("compare"),
9442 Ordering::Equal,
9443 "rank {rank} is its own value"
9444 );
9445 if rank > 0 {
9446 assert_eq!(
9447 dictionary.compare_rank(rank - 1, value).expect("compare"),
9448 Ordering::Less,
9449 "rank {rank} follows the one before it"
9450 );
9451 }
9452 }
9453 fs::remove_file(path).expect("remove scratch file");
9454 }
9455
9456 #[test]
9464 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
9465 let path = path("dictionary-sweep");
9466 let spellings = (0..2_500)
9469 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9470 .collect::<Vec<_>>();
9471 let mut writer =
9472 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9473 .expect("new file");
9474 for part in spellings.chunks(1_024) {
9477 writer
9478 .append(
9479 &Chunk::new(vec![
9480 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9481 ])
9482 .expect("one column"),
9483 )
9484 .expect("stripe written");
9485 }
9486 writer.finish().expect("commit");
9487
9488 let reader = Reader::open(&path).expect("valid directory");
9489 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9490 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9491
9492 let resting = dictionary.footprint();
9493 let mut swept: Vec<Vec<u8>> = Vec::new();
9494 let mut at = 0;
9495 let mut calls = 0;
9496 while at < dictionary.len() {
9497 let stopped = dictionary
9498 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9499 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9500 swept.push(text.to_vec());
9501 Ok(())
9502 })
9503 .expect("a sweep reads");
9504 assert!(stopped > at, "a sweep moves");
9505 at = stopped;
9506 calls += 1;
9507 }
9508 assert_eq!(calls, 3, "a sweep hands over one block at a time");
9509 let after = dictionary.footprint();
9510 assert!(after > resting, "a sweep under the budget keeps what it decoded");
9511
9512 let read = (0..dictionary.len())
9513 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9514 .collect::<Vec<_>>();
9515 assert_eq!(swept, read, "a sweep answers what a point read answers");
9516 assert_eq!(dictionary.footprint(), after, "a point read of a kept block decodes nothing");
9517 fs::remove_file(path).expect("remove scratch file");
9518 }
9519
9520 #[test]
9531 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
9532 let path = path("dictionary-sweep-short-run");
9533 let spellings = (0..2_800)
9534 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9535 .collect::<Vec<_>>();
9536 let mut writer =
9537 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9538 .expect("new file");
9539 for part in spellings.chunks(1_024) {
9540 writer
9541 .append(
9542 &Chunk::new(vec![
9543 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9544 ])
9545 .expect("one column"),
9546 )
9547 .expect("stripe written");
9548 }
9549 writer.finish().expect("commit");
9550
9551 let reader = Reader::open(&path).expect("valid directory");
9552 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9553 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9554 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
9555 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
9556 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
9557
9558 let mut swept: Vec<Vec<u8>> = Vec::new();
9559 let mut at = 0;
9560 while at < dictionary.len() {
9561 let stopped = dictionary
9562 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9563 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9564 swept.push(text.to_vec());
9565 Ok(())
9566 })
9567 .expect("a sweep reads");
9568 assert!(stopped > at, "a sweep moves");
9569 at = stopped;
9570 }
9571 let read = (0..dictionary.len())
9572 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9573 .collect::<Vec<_>>();
9574 assert_eq!(swept, read, "a sweep answers what a point read answers");
9575 fs::remove_file(path).expect("remove scratch file");
9576 }
9577
9578 #[test]
9588 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
9589 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
9590 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
9591 fit::<i8>(&[128]).expect_err("one past the top does not fit");
9592 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
9593 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
9594 fit::<u8>(&[256]).expect_err("one past the top does not fit");
9595 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
9596 assert_eq!(
9597 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
9598 vec![-32_768_i16, 0, 32_767]
9599 );
9600 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
9601 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
9602 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
9603 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
9604 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
9605 assert_eq!(
9606 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
9607 vec![i32::MIN, 0, i32::MAX]
9608 );
9609 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
9610 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
9611 assert_eq!(
9612 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
9613 vec![0_u32, 4_294_967_295]
9614 );
9615 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
9616 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
9617
9618 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
9621 }
9622
9623 #[test]
9630 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
9631 for value in -70_000_i64..70_000 {
9632 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
9633 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
9634 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
9635 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
9636 }
9637 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
9638 for edge in wide {
9639 for step in -2_i64..=2 {
9640 let value = edge.saturating_add(step);
9641 assert_eq!(
9642 fit::<i32>(&[value]).is_ok(),
9643 i32::try_from(value).is_ok(),
9644 "{value} as i32"
9645 );
9646 assert_eq!(
9647 fit::<u32>(&[value]).is_ok(),
9648 u32::try_from(value).is_ok(),
9649 "{value} as u32"
9650 );
9651 }
9652 }
9653 }
9654
9655 #[test]
9663 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
9664 let path = path("dictionary-budget");
9665 let spellings = (0..2_500)
9666 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
9667 .collect::<Vec<_>>();
9668 let mut writer =
9669 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9670 .expect("new file");
9671 for part in spellings.chunks(1_024) {
9672 writer
9673 .append(
9674 &Chunk::new(vec![
9675 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9676 ])
9677 .expect("one column"),
9678 )
9679 .expect("stripe written");
9680 }
9681 writer.finish().expect("commit");
9682
9683 let reader = Reader::open(&path).expect("valid directory");
9684 let page = reader.table.dictionaries[0].expect("a string column has one");
9685 let file = Arc::clone(&reader.file);
9686 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
9687 .expect("a dictionary opens whatever it may keep");
9688
9689 let resting = starved.footprint();
9690 let mut swept: Vec<Vec<u8>> = Vec::new();
9691 let mut at = 0;
9692 while at < starved.len() {
9693 at = starved
9694 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
9695 swept.push(text.to_vec());
9696 Ok(())
9697 })
9698 .expect("a sweep reads");
9699 }
9700 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
9701 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
9702
9703 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
9704 let read = (0..generous.len())
9705 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
9706 .collect::<Vec<_>>();
9707 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
9708 fs::remove_file(path).expect("remove scratch file");
9709 }
9710
9711 #[test]
9712 fn damaged_membership_cannot_skip_a_string_page() {
9713 let path = path("damaged-membership");
9714 let mut writer = Writer::create(
9715 &path,
9716 "items",
9717 vec![
9718 Field::required("id", LogicalType::Integer),
9719 Field::new("text", LogicalType::Varchar),
9720 ],
9721 )
9722 .expect("new file");
9723 writer.append(&sample()).expect("stripe written");
9724 writer.finish().expect("commit");
9725
9726 let reader = Reader::open(&path).expect("valid directory");
9727 let membership = reader.table.stripes[0].memberships[1].expect("string membership");
9728 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
9729 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
9730 file.write_all(&[255]).expect("damage membership");
9731 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
9732 assert!(error.message().contains("membership page checksum differs"), "{error}");
9733 fs::remove_file(path).expect("remove scratch file");
9734 }
9735
9736 #[test]
9737 fn membership_delta_stream_is_sorted_exact_and_bounded() {
9738 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
9739 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
9740 let encoded = encode_membership(&unique);
9741 assert_eq!(
9742 decode_membership(&encoded).expect("valid membership"),
9743 [4, 9, 72, 900, u32::MAX]
9744 );
9745 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
9748 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
9749 assert_eq!(
9750 decode_membership(&encode_membership(&merged)).expect("valid membership"),
9751 unique
9752 );
9753 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
9754 assert!(
9755 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
9756 "a value past u32 is invalid"
9757 );
9758 }
9759
9760 #[test]
9761 fn a_global_dictionary_may_be_larger_than_one_column_page() {
9762 let dictionary = Page {
9763 offset: HEADER,
9764 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
9765 hash: 0,
9766 };
9767 let table = Table {
9768 name: "items".to_owned(),
9769 fields: vec![Field::new("text", LogicalType::Varchar)],
9770 stripes: Vec::new(),
9771 rows: 0,
9772 dictionaries: vec![Some(dictionary)],
9773 distincts: vec![None],
9774 frequencies: vec![None],
9775 clustering: None,
9776 generation: 1,
9777 sections: Vec::new(),
9778 };
9779 let directory = encode_directory(&table).expect("directory");
9780 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
9781
9782 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
9783 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
9784 }
9785
9786 #[test]
9787 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
9788 let path = path("constant-codes");
9789 let mut writer =
9790 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9791 .expect("new file");
9792 let empty = vec![Value::Varchar(String::new()); 1024];
9793 for _ in 0..4 {
9794 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
9795 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
9796 }
9797 writer.finish().expect("commit");
9798
9799 let reader = Reader::open(&path).expect("valid directory");
9800 let pages = reader.layout().columns.first().expect("one column").pages;
9801 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
9805 let read = reader.read(3, &[0]).expect("the last part back");
9806 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
9807 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
9808 fs::remove_file(path).expect("remove scratch file");
9809 }
9810
9811 #[test]
9812 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
9813 let over = vec![i64::from(i32::MAX) + 1];
9816 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
9817 assert!(format!("{error}").contains("not of its type"), "{error}");
9818 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
9819 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
9820 }
9821
9822 #[test]
9823 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
9824 let mut state: u32 = 0x9e37_79b9;
9828 let spread: Vec<u32> = (0..1024)
9829 .map(|_| {
9830 state ^= state << 13;
9831 state ^= state >> 17;
9832 state ^= state << 5;
9833 state
9834 })
9835 .collect();
9836 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
9837 let near: Vec<u32> = (0..1024).collect();
9838 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
9839 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
9840 }
9841
9842 #[test]
9848 fn two_writes_of_the_same_rows_give_the_same_bytes() {
9849 fn written(path: &PathBuf) {
9850 let fields = (0..40)
9851 .map(|column| {
9852 let ty =
9853 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
9854 Field::new(format!("c{column}"), ty)
9855 })
9856 .collect::<Vec<_>>();
9857 let mut writer = Writer::create(path, "wide", fields).expect("new file");
9858 for part in 0..70_u64 {
9859 let columns = (0..40)
9860 .map(|column| {
9861 let values = (0..64_u64)
9862 .map(|row| {
9863 let seed = part.wrapping_mul(31).wrapping_add(row);
9864 if column % 4 == 0 {
9865 Value::Varchar(format!("v{}", seed % 17))
9866 } else {
9867 Value::BigInt(i64::try_from(seed % 97).expect("small"))
9868 }
9869 })
9870 .collect::<Vec<_>>();
9871 let ty = if column % 4 == 0 {
9872 LogicalType::Varchar
9873 } else {
9874 LogicalType::BigInt
9875 };
9876 Vector::from_values(ty, &values).expect("a column")
9877 })
9878 .collect::<Vec<_>>();
9879 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
9880 }
9881 writer.finish().expect("commit");
9882 }
9883
9884 let first = path("repeatable-one");
9885 let second = path("repeatable-two");
9886 written(&first);
9887 written(&second);
9888 let left = fs::read(&first).expect("the first file");
9889 let right = fs::read(&second).expect("the second file");
9890 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
9891 assert!(left == right, "two writes of the same rows differ in their bytes");
9892
9893 let reader = Reader::open(&first).expect("valid directory");
9896 assert_eq!(reader.table().rows(), 70 * 64);
9897 let read = reader.read(0, &[0, 1]).expect("the first part back");
9898 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
9899 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
9900 fs::remove_file(first).expect("remove scratch file");
9901 fs::remove_file(second).expect("remove scratch file");
9902 }
9903
9904 fn three_tables(path: &PathBuf) {
9906 let writer = Writer::create(
9907 path,
9908 "region",
9909 vec![
9910 Field::new("r_key", LogicalType::Integer),
9911 Field::new("r_name", LogicalType::Varchar),
9912 ],
9913 )
9914 .expect("new file");
9915 let mut writer = writer;
9916 writer
9917 .append(
9918 &Chunk::new(vec![
9919 Vector::from_values(
9920 LogicalType::Integer,
9921 &[Value::Integer(0), Value::Integer(1)],
9922 )
9923 .expect("keys"),
9924 Vector::from_values(
9925 LogicalType::Varchar,
9926 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
9927 )
9928 .expect("names"),
9929 ])
9930 .expect("two columns"),
9931 )
9932 .expect("a part");
9933 let mut writer = writer
9934 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
9935 .expect("a second table");
9936 writer
9937 .append(
9938 &Chunk::new(vec![
9939 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
9940 ])
9941 .expect("one column"),
9942 )
9943 .expect("a part");
9944 let mut writer =
9945 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
9946 for part in 0..70_i64 {
9947 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
9948 writer
9949 .append(
9950 &Chunk::new(vec![
9951 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
9952 ])
9953 .expect("one column"),
9954 )
9955 .expect("a part");
9956 }
9957 writer.finish().expect("commit");
9958 }
9959
9960 #[test]
9961 fn three_tables_in_one_file_read_back_by_name() {
9962 let file = path("three-tables");
9963 three_tables(&file);
9964 let catalog = Catalog::open(&file).expect("a committed catalog");
9965 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
9966
9967 let region = catalog.table("region").expect("the first table");
9968 assert_eq!(region.table().rows(), 2);
9969 assert_eq!(
9970 region.read(0, &[1]).expect("names").value_at(1, 0),
9971 Value::Varchar("ASIA".to_owned())
9972 );
9973
9974 let wide = catalog.table("wide").expect("the third table");
9975 assert_eq!(wide.table().rows(), 70 * 64);
9976 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
9977
9978 let empty = catalog.table("empty").expect("the second table");
9981 assert_eq!(empty.table().rows(), 1);
9982 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
9983
9984 fs::remove_file(file).expect("remove scratch file");
9985 }
9986
9987 #[test]
9988 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
9989 let file = path("three-tables-missing");
9990 three_tables(&file);
9991 let catalog = Catalog::open(&file).expect("a committed catalog");
9992 let error = catalog.table("nation").expect_err("no such table");
9993 assert!(error.message().contains("nation"), "{}", error.message());
9994 fs::remove_file(file).expect("remove scratch file");
9995 }
9996
9997 #[test]
9998 fn a_file_of_three_tables_will_not_open_as_one() {
9999 let file = path("three-tables-unnamed");
10000 three_tables(&file);
10001 let error = Reader::open(&file).expect_err("more than one table");
10002 assert!(error.message().contains("more than one table"), "{}", error.message());
10003 fs::remove_file(file).expect("remove scratch file");
10004 }
10005
10006 #[test]
10008 fn decimals_of_every_storage_width_round_trip() {
10009 let file = path("decimals");
10010 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
10011 let fields = widths
10012 .iter()
10013 .enumerate()
10014 .map(|(index, (width, scale))| {
10015 Field::new(
10016 format!("d{index}"),
10017 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10018 )
10019 })
10020 .collect::<Vec<_>>();
10021 let mut writer = Writer::create(&file, "money", fields).expect("new file");
10022 let rows: [i128; 3] = [-1234, 0, 999];
10023 let columns = widths
10024 .iter()
10025 .map(|(width, scale)| {
10026 let values = rows
10027 .iter()
10028 .map(|unscaled| Value::Decimal {
10029 unscaled: *unscaled,
10030 width: *width,
10031 scale: *scale,
10032 })
10033 .collect::<Vec<_>>();
10034 Vector::from_values(
10035 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10036 &values,
10037 )
10038 .expect("a decimal column")
10039 })
10040 .collect::<Vec<_>>();
10041 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
10042 writer.finish().expect("commit");
10043
10044 let reader = Reader::open(&file).expect("a committed file");
10045 for (index, (width, scale)) in widths.iter().enumerate() {
10046 assert_eq!(
10047 reader.table().fields()[index].ty,
10048 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10049 "column {index} came back as another type"
10050 );
10051 let column = reader.read(0, &[index]).expect("the column");
10052 for (row, unscaled) in rows.iter().enumerate() {
10053 assert_eq!(
10054 column.value_at(row, 0),
10055 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
10056 "column {index} row {row}"
10057 );
10058 }
10059 }
10060 fs::remove_file(file).expect("remove scratch file");
10061 }
10062
10063 #[test]
10064 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
10065 let file = path("two-of-a-name");
10066 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
10067 .expect("new file");
10068 let error = writer
10069 .next("t", vec![Field::new("a", LogicalType::BigInt)])
10070 .expect_err("the same name twice");
10071 assert!(error.message().contains("same name"), "{}", error.message());
10072 fs::remove_file(file).expect("remove scratch file");
10073 }
10074
10075 #[test]
10076 fn opening_the_catalog_reads_no_table_directory() {
10077 let file = path("catalog-only");
10078 three_tables(&file);
10079 let catalog = Catalog::open(&file).expect("a committed catalog");
10080 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
10083 assert_eq!(catalog.names().len(), 3);
10084 fs::remove_file(file).expect("remove scratch file");
10085 }
10086
10087 #[test]
10098 fn the_checksum_answers_what_it_has_always_answered() {
10099 let bytes: Vec<u8> =
10100 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
10101 for (length, expected) in [
10102 (0, 0xef46_db37_51d8_e999),
10103 (1, 0xa96c_7f0c_e858_bbb7),
10104 (3, 0x56e6_9576_32a4_87f9),
10105 (4, 0xc60d_15b1_e3ff_8f04),
10106 (5, 0x8088_1585_8624_dd4e),
10107 (7, 0xafbe_fc3d_6c6f_9a8e),
10108 (8, 0x3da5_c7aa_2696_83e0),
10109 (9, 0x465e_c429_b13c_3892),
10110 (15, 0xdee8_9d8a_065a_6233),
10111 (16, 0x1330_489a_7767_9c80),
10112 (31, 0x3391_303d_485e_846e),
10113 (32, 0x40b7_aff7_5d45_bbc8),
10114 (33, 0x4997_cae4_951c_17a5),
10115 (39, 0x5807_28fd_5c14_5739),
10116 (40, 0xf95c_f6f5_c08a_3d3b),
10117 (63, 0x2944_b4da_fc69_b206),
10118 (64, 0xbb76_f6ef_19bd_5a1b),
10119 (65, 0x814e_0c65_4a9f_d640),
10120 (127, 0x00de_aab1_31cf_f89b),
10121 (1000, 0x9e33_00c1_cde3_c58d),
10122 ] {
10123 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
10124 }
10125 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
10126 }
10127 #[test]
10134 fn a_declared_order_comes_back_out_of_the_file() {
10135 let path = path("clustered");
10136 let shipped = vec![
10137 Field::new("key", LogicalType::BigInt),
10138 Field::new("line", LogicalType::Integer),
10139 Field::new("shipdate", LogicalType::Date),
10140 ];
10141 let plain = vec![Field::new("a", LogicalType::Integer)];
10142 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
10143
10144 let mut writer = Writer::create(&path, "lineitem", shipped)
10145 .expect("new file")
10146 .declare(stage_zero.clone())
10147 .expect("the columns are the table's");
10148 let column = |ty: LogicalType, values: &[Value]| {
10149 Vector::from_values(ty, values).expect("the values match the type")
10150 };
10151 writer
10152 .append(
10153 &Chunk::new(vec![
10154 column(
10155 LogicalType::BigInt,
10156 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
10157 ),
10158 column(
10159 LogicalType::Integer,
10160 &[
10161 Value::Integer(1),
10162 Value::Integer(1),
10163 Value::Integer(1),
10164 Value::Integer(1),
10165 ],
10166 ),
10167 column(
10168 LogicalType::Date,
10169 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
10170 ),
10171 ])
10172 .expect("three columns"),
10173 )
10174 .expect("four rows");
10175 let mut writer = writer.next("nation", plain).expect("a second table");
10176 writer
10177 .append(
10178 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
10179 .expect("one column"),
10180 )
10181 .expect("one row");
10182 writer.finish().expect("commit");
10183
10184 let catalog = Catalog::open(&path).expect("reopen");
10185 let lineitem = catalog.table("lineitem").expect("the clustered table");
10186 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
10187 let nation = catalog.table("nation").expect("the plain table");
10188 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
10189
10190 assert_eq!(lineitem.table().rows(), 4);
10193 assert_eq!(nation.table().rows(), 1);
10194 fs::remove_file(&path).ok();
10195 }
10196
10197 #[test]
10199 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
10200 let path = path("clustered-bad");
10201 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10202 .expect("new file");
10203 let four =
10204 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
10205 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
10206 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
10207 fs::remove_file(&path).ok();
10208 }
10209}