1#![forbid(unsafe_code)]
34
35use std::cmp::Ordering;
36use std::collections::{HashMap, VecDeque};
37use std::fs::{File, OpenOptions};
38use std::io::{Read, Seek, SeekFrom};
39use std::mem::{size_of, size_of_val};
40use std::path::Path;
41use std::slice;
42use std::sync::atomic::{AtomicUsize, Ordering as Atomic};
43use std::sync::{Arc, Mutex, OnceLock};
44
45use rudb_common::bounds::{self, Bound, Op, scaled_as};
46use rudb_common::{Clustering, Error, Field, LogicalType, PhysicalType, Result, Value, Width};
47use rudb_encoding::{bitpack, chooser, integer, string};
48use rudb_storage::sieve::Sieve;
49use rudb_storage::{Probe, Range, Zone};
50use rudb_vector::string::StringColumn;
51use rudb_vector::validity::Validity;
52use rudb_vector::{Buffer, Chunk, Data, Packed, TextSource, Vector, search_below};
53
54pub mod graph;
55pub mod section;
56pub mod stats;
57mod zones;
58
59pub use section::Section;
60pub use zones::{Common, Stripes, distincts};
61
62const MAGIC: &[u8; 8] = b"RUDBNV10";
63const DIRECTORY: &[u8; 8] = b"RUDBDI10";
64const CATALOG: &[u8; 8] = b"RUDBCA10";
65const FORMAT: u32 = 25;
66
67const READABLE: &[u32] = &[22, 23, 24, FORMAT];
86
87const HEADER: u64 = 80;
88const SLOT_BYTES: usize = 28;
89const MAX_PAGE: usize = 256 * 1024 * 1024;
90const MAX_DIRECTORY: usize = 128 * 1024 * 1024;
91const FREQUENCIES: &[u8; 8] = b"RUDBFQ2\0";
92const CLUSTERING: &[u8; 8] = b"RUDBCL1\0";
99const SECTIONS: &[u8; 8] = b"RUDBSE1\0";
107
108const MAX_SECTIONS: usize = 4096;
115const FREQUENCY_CANDIDATES: usize = 32_768;
116const FREQUENCY_ENTRIES: usize = 512;
117const FREQUENCY_BUILD_RANK: usize = 10;
118const FREQUENCY_ORDINALS: usize = 65_536;
119const MAX_FREQUENCY_WORKERS: usize = 32;
126
127const MAX_ENCODE_WORKERS: usize = 32;
134
135const SIEVE_BUDGET: usize = 8 * 1024;
143
144const PART_BOUND_BYTES: usize = 24;
153
154fn io(error: std::io::Error) -> Error {
155 Error::io(error.to_string())
156}
157
158fn invalid(message: &str) -> Error {
159 Error::invalid_input(format!("invalid rudb native file: {message}"))
160}
161
162fn sum(counts: impl Iterator<Item = u64>) -> u64 {
164 counts.fold(0, u64::saturating_add)
165}
166
167fn span_bytes(spans: &[Span], at: usize) -> u64 {
169 spans.get(at).map_or(0, |span| u64::from(span.length))
170}
171
172fn page_bytes(pages: &[Option<Page>], at: usize) -> u64 {
174 pages.get(at).and_then(Option::as_ref).map_or(0, Page::bytes)
175}
176
177fn checksum(bytes: &[u8]) -> u64 {
187 const P1: u64 = 11_400_714_785_074_694_791;
188 const P2: u64 = 14_029_467_366_897_019_727;
189 const P3: u64 = 1_609_587_929_392_839_161;
190 const P4: u64 = 9_650_029_242_287_828_579;
191 const P5: u64 = 2_870_177_450_012_600_261;
192 let round = |state: u64, word: u64| {
193 state.wrapping_add(word.wrapping_mul(P2)).rotate_left(31).wrapping_mul(P1)
194 };
195 let merge = |state: u64, lane: u64| (state ^ round(0, lane)).wrapping_mul(P1).wrapping_add(P4);
196 let word = |chunk: &[u8]| u64::from_le_bytes(chunk.try_into().expect("eight checksum bytes"));
197
198 let mut blocks = bytes.chunks_exact(32);
201 let mut rest = blocks.remainder();
202 let mut hash = if bytes.len() >= 32 {
203 let mut one = P1.wrapping_add(P2);
204 let mut two = P2;
205 let mut three = 0;
206 let mut four = 0_u64.wrapping_sub(P1);
207 for block in blocks.by_ref() {
208 one = round(one, word(&block[..8]));
209 two = round(two, word(&block[8..16]));
210 three = round(three, word(&block[16..24]));
211 four = round(four, word(&block[24..]));
212 }
213 let combined = one
214 .rotate_left(1)
215 .wrapping_add(two.rotate_left(7))
216 .wrapping_add(three.rotate_left(12))
217 .wrapping_add(four.rotate_left(18));
218 merge(merge(merge(merge(combined, one), two), three), four)
219 } else {
220 P5
221 };
222 hash = hash.wrapping_add(bytes.len() as u64);
223 let mut words = rest.chunks_exact(8);
224 for chunk in words.by_ref() {
225 hash ^= round(0, word(chunk));
226 hash = hash.rotate_left(27).wrapping_mul(P1).wrapping_add(P4);
227 }
228 rest = words.remainder();
229 if rest.len() >= 4 {
230 let (head, tail) = rest.split_at(4);
231 let quarter = u32::from_le_bytes(head.try_into().expect("four checksum bytes"));
232 hash ^= u64::from(quarter).wrapping_mul(P1);
233 hash = hash.rotate_left(23).wrapping_mul(P2).wrapping_add(P3);
234 rest = tail;
235 }
236 for &byte in rest {
237 hash ^= u64::from(byte).wrapping_mul(P5);
238 hash = hash.rotate_left(11).wrapping_mul(P1);
239 }
240 hash ^= hash >> 33;
241 hash = hash.wrapping_mul(P2);
242 hash ^= hash >> 29;
243 hash = hash.wrapping_mul(P3);
244 hash ^ (hash >> 32)
245}
246
247#[derive(Debug, Clone, Copy)]
248struct Slot {
249 offset: u64,
250 length: u32,
251 generation: u64,
252 hash: u64,
253}
254
255impl Slot {
256 fn bytes(self) -> [u8; SLOT_BYTES] {
257 let mut result = [0; SLOT_BYTES];
258 result[..8].copy_from_slice(&self.offset.to_le_bytes());
259 result[8..12].copy_from_slice(&self.length.to_le_bytes());
260 result[12..20].copy_from_slice(&self.generation.to_le_bytes());
261 result[20..28].copy_from_slice(&self.hash.to_le_bytes());
262 result
263 }
264
265 fn read(bytes: &[u8]) -> Self {
266 Self {
267 offset: u64::from_le_bytes(bytes[..8].try_into().expect("eight bytes")),
268 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
269 generation: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
270 hash: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
271 }
272 }
273}
274
275#[derive(Debug, Clone, Copy)]
276struct Page {
277 offset: u64,
278 length: u32,
279 hash: u64,
280}
281
282impl Page {
283 fn bytes(&self) -> u64 {
285 u64::from(self.length)
286 }
287}
288
289#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
290enum FrequencyValue {
291 Null,
292 Integer(i128),
293 Code(u32),
294}
295
296#[derive(Debug, Clone)]
297struct FrequencyEntry {
298 value: FrequencyValue,
299 count: u64,
300}
301
302#[derive(Debug, Clone)]
307struct FrequencySummary {
308 entries: Vec<FrequencyEntry>,
309 omitted_max: u64,
310 ordinals: Vec<u64>,
311}
312
313#[derive(Debug, Clone)]
318pub struct FrequencyPrefix {
319 pub entries: Vec<(Value, u64)>,
321 pub omitted_max: u64,
323}
324
325#[derive(Debug, Clone, PartialEq, Eq)]
327pub struct FrequencyOccurrences {
328 pub omitted_max: u64,
330 pub ordinals: Vec<u64>,
332}
333
334#[derive(Debug, Clone, Copy, Default)]
341struct Span {
342 offset: u64,
343 length: u32,
344}
345
346#[derive(Debug, Clone)]
348pub struct Stripe {
349 rows: usize,
350 parts: Vec<u32>,
353 index: Span,
357 pages: Vec<Span>,
358 memberships: Vec<Option<Page>>,
359 sieves: Vec<Option<Page>>,
362 part_ranges: Vec<Option<Page>>,
373 zone: Zone,
374}
375
376impl Stripe {
377 #[must_use]
379 pub fn rows(&self) -> usize {
380 self.rows
381 }
382
383 #[must_use]
385 pub fn parts(&self) -> usize {
386 self.parts.len()
387 }
388
389 #[must_use]
395 pub fn zone(&self) -> &Zone {
396 &self.zone
397 }
398}
399
400#[derive(Debug, Clone)]
402pub struct Table {
403 name: String,
404 fields: Vec<Field>,
405 stripes: Vec<Stripe>,
406 rows: usize,
407 dictionaries: Vec<Option<Page>>,
408 frequencies: Vec<Option<FrequencySummary>>,
409 distincts: Vec<Option<u64>>,
419 clustering: Option<Clustering>,
427 generation: u64,
441 sections: Vec<Section>,
448}
449
450impl Table {
451 #[must_use]
453 pub fn name(&self) -> &str {
454 &self.name
455 }
456
457 #[must_use]
459 pub fn fields(&self) -> &[Field] {
460 &self.fields
461 }
462
463 #[must_use]
465 pub fn rows(&self) -> usize {
466 self.rows
467 }
468
469 #[must_use]
471 pub fn stripes(&self) -> &[Stripe] {
472 &self.stripes
473 }
474
475 #[must_use]
477 pub fn clustering(&self) -> Option<&Clustering> {
478 self.clustering.as_ref()
479 }
480
481 #[must_use]
486 pub fn generation(&self) -> u64 {
487 self.generation
488 }
489
490 #[must_use]
497 pub fn sections(&self) -> &[Section] {
498 &self.sections
499 }
500}
501
502#[derive(Debug, Clone)]
514struct Entry {
515 name: String,
516 fields: Vec<Field>,
517 rows: usize,
518 directory: Page,
520}
521
522#[derive(Debug, Clone, PartialEq, Eq)]
535pub struct ViewEntry {
536 pub name: String,
538 pub sql: String,
540 pub statement: String,
542 pub aliases: Vec<String>,
544 pub columns: Vec<Field>,
546}
547
548#[derive(Debug, Clone)]
550pub struct ColumnLayout {
551 pub name: String,
553 pub kind: String,
555 pub pages: u64,
557 pub memberships: u64,
559 pub sieves: u64,
561 pub part_ranges: u64,
563 pub dictionary: u64,
565}
566
567impl ColumnLayout {
568 #[must_use]
570 pub fn total(&self) -> u64 {
571 self.pages
572 .saturating_add(self.memberships)
573 .saturating_add(self.sieves)
574 .saturating_add(self.part_ranges)
575 .saturating_add(self.dictionary)
576 }
577}
578
579#[derive(Debug, Clone)]
590pub struct Layout {
591 pub file: u64,
593 pub rows: usize,
595 pub stripes: usize,
597 pub parts: usize,
599 pub columns: Vec<ColumnLayout>,
601 pub indexes: u64,
604 pub directory: u64,
606 pub header: u64,
608}
609
610impl Layout {
611 #[must_use]
613 pub fn columns_total(&self) -> u64 {
614 self.columns.iter().map(ColumnLayout::total).fold(0, u64::saturating_add)
615 }
616
617 #[must_use]
623 pub fn unaccounted(&self) -> u64 {
624 self.file
625 .saturating_sub(self.columns_total())
626 .saturating_sub(self.indexes)
627 .saturating_sub(self.directory)
628 .saturating_sub(self.header)
629 }
630}
631
632#[derive(Debug, Clone)]
643pub struct StoredPart {
644 pub stripe: usize,
646 pub part: usize,
648 pub row: usize,
650 pub rows: usize,
652 pub encoding: String,
654 pub bytes: u64,
656 pub page: u64,
658 pub offset: u64,
660 pub low: Option<Value>,
662 pub high: Option<Value>,
664 pub nulls: Option<usize>,
666}
667
668#[derive(Debug)]
670struct GlobalDictionary {
671 primary: HashMap<u64, u32>,
672 collisions: HashMap<u64, Vec<u32>>,
673 offsets: Vec<u32>,
674 payload: Vec<u8>,
675 counts: Vec<u64>,
676 nulls: u64,
677}
678
679impl GlobalDictionary {
680 fn new() -> Self {
681 Self {
682 primary: HashMap::new(),
683 collisions: HashMap::new(),
684 offsets: vec![0],
685 payload: Vec::new(),
686 counts: Vec::new(),
687 nulls: 0,
688 }
689 }
690
691 fn bytes(&self, code: u32) -> Option<&[u8]> {
692 let start = *self.offsets.get(code as usize)? as usize;
693 let end = *self.offsets.get(code as usize + 1)? as usize;
694 self.payload.get(start..end)
695 }
696
697 fn code(&mut self, text: &str) -> Result<u32> {
698 let hash = checksum(text.as_bytes());
699 if let Some(&code) = self.primary.get(&hash) {
700 if self.bytes(code) == Some(text.as_bytes()) {
701 return Ok(code);
702 }
703 if let Some(codes) = self.collisions.get(&hash) {
704 if let Some(code) =
705 codes.iter().copied().find(|&code| self.bytes(code) == Some(text.as_bytes()))
706 {
707 return Ok(code);
708 }
709 }
710 let code = self.insert(text)?;
711 self.collisions.entry(hash).or_default().push(code);
712 return Ok(code);
713 }
714 let code = self.insert(text)?;
715 self.primary.insert(hash, code);
716 Ok(code)
717 }
718
719 fn insert(&mut self, text: &str) -> Result<u32> {
720 let code = u32::try_from(self.offsets.len() - 1)
721 .map_err(|_| invalid("global dictionary has too many values"))?;
722 self.payload.extend_from_slice(text.as_bytes());
723 self.offsets.push(
724 u32::try_from(self.payload.len())
725 .map_err(|_| invalid("global dictionary payload exceeds 4 GiB"))?,
726 );
727 self.counts.push(0);
728 Ok(code)
729 }
730
731 fn ranked(&self) -> Vec<(u64, u32)> {
751 let count = self.offsets.len() - 1;
752 let mut ranked = (0..count)
753 .map(|code| {
754 let code = code as u32;
755 (head(self.bytes(code).unwrap_or_default()), code)
756 })
757 .collect::<Vec<_>>();
758 ranked.sort_unstable_by(|left, right| {
759 left.0.cmp(&right.0).then_with(|| self.bytes(left.1).cmp(&self.bytes(right.1)))
760 });
761 ranked
762 }
763
764 fn observe(&mut self, code: u32, null: bool) -> Result<()> {
765 if null {
766 self.nulls = self.nulls.saturating_add(1);
767 return Ok(());
768 }
769 let count = self
770 .counts
771 .get_mut(code as usize)
772 .ok_or_else(|| invalid("global dictionary count code is out of range"))?;
773 *count = count.saturating_add(1);
774 Ok(())
775 }
776}
777
778#[derive(Debug)]
786pub struct Writer {
787 file: File,
788 at: u64,
796 table: Table,
797 generation: u64,
798 order: Vec<((u64, u64), (u64, u64))>,
801 next_order: u64,
802 dictionaries: Vec<Option<GlobalDictionary>>,
803 pending: Vec<PendingChunk>,
804 closed: Vec<Entry>,
806 views: Vec<ViewEntry>,
811}
812
813#[derive(Debug)]
821struct PendingChunk {
822 order: (u64, u64),
823 chunk: Chunk,
824}
825
826#[derive(Debug)]
832struct ColumnStripe {
833 pages: Vec<Vec<u8>>,
834 codes: Vec<Option<Vec<u32>>>,
835 sieves: Vec<Option<Sieve>>,
836 ranges: Vec<Range>,
837}
838
839fn weight(ty: &LogicalType) -> usize {
847 match ty {
848 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => 64,
849 LogicalType::HugeInt
850 | LogicalType::UHugeInt
851 | LogicalType::Uuid
852 | LogicalType::Interval => 16,
853 LogicalType::BigInt
854 | LogicalType::UBigInt
855 | LogicalType::Timestamp
856 | LogicalType::Time
857 | LogicalType::TimeTz
858 | LogicalType::TimestampTz
859 | LogicalType::TimestampS
860 | LogicalType::TimestampMs
861 | LogicalType::TimestampNs
862 | LogicalType::Double
863 | LogicalType::Decimal { .. } => 8,
864 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date | LogicalType::Float => 4,
865 LogicalType::SmallInt | LogicalType::USmallInt => 2,
866 _ => 1,
867 }
868}
869
870pub const STRIPE_PARTS: usize = 64;
877
878const DICTIONARY_DECIDE_ROWS: usize = 4_096;
886
887const DICTIONARY_DISTINCT_IN_TEN: usize = 9;
903
904const INDEX_ENTRY: usize = size_of::<u32>() + size_of::<u64>();
906
907fn index_section(parts: usize) -> Result<usize> {
909 parts
910 .checked_mul(INDEX_ENTRY)
911 .and_then(|bytes| bytes.checked_add(size_of::<u64>()))
912 .ok_or_else(|| invalid("index page length overflow"))
913}
914
915impl Writer {
916 pub fn open(
935 path: impl AsRef<Path>,
936 name: impl Into<String>,
937 fields: Vec<Field>,
938 ) -> Result<Self> {
939 for field in &fields {
940 type_tag(&field.ty)?;
941 }
942 let name = name.into();
943 let path = path.as_ref();
944 let (_, size, slot, bytes, _) = slot_bytes(path)?;
945 let (mut closed, views) = decode_catalog(&bytes, size)?;
946 if let Some(at) = closed.iter().position(|held| held.name == name) {
957 if closed[at].rows > 0 {
958 return Err(invalid("two tables in one native file have the same name"));
959 }
960 closed.remove(at);
961 }
962 let generation = slot
967 .generation
968 .checked_add(1)
969 .ok_or_else(|| invalid("native file generation overflow"))?;
970 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
971 Ok(Self {
972 file,
973 at: size,
976 dictionaries: fields
977 .iter()
978 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
979 .collect(),
980 table: Table {
981 name,
982 dictionaries: vec![None; fields.len()],
983 distincts: vec![None; fields.len()],
984 fields,
985 stripes: Vec::new(),
986 rows: 0,
987 frequencies: Vec::new(),
988 clustering: None,
989 generation,
990 sections: Vec::new(),
991 },
992 generation,
993 order: Vec::new(),
994 next_order: 0,
995 pending: Vec::with_capacity(STRIPE_PARTS),
996 closed,
997 views,
998 })
999 }
1000
1001 pub fn create(
1007 path: impl AsRef<Path>,
1008 name: impl Into<String>,
1009 fields: Vec<Field>,
1010 ) -> Result<Self> {
1011 for field in &fields {
1012 type_tag(&field.ty)?;
1013 }
1014 let file =
1015 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1016 let mut header = [0; HEADER as usize];
1017 header[..8].copy_from_slice(MAGIC);
1018 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1019 write_at(&file, 0, &header)?;
1020 Ok(Self {
1021 file,
1022 at: HEADER,
1023 dictionaries: fields
1024 .iter()
1025 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1026 .collect(),
1027 table: Table {
1028 name: name.into(),
1029 dictionaries: vec![None; fields.len()],
1030 distincts: vec![None; fields.len()],
1031 fields,
1032 stripes: Vec::new(),
1033 rows: 0,
1034 frequencies: Vec::new(),
1035 clustering: None,
1036 generation: 1,
1037 sections: Vec::new(),
1038 },
1039 generation: 1,
1040 order: Vec::new(),
1041 next_order: 0,
1042 pending: Vec::with_capacity(STRIPE_PARTS),
1043 closed: Vec::new(),
1044 views: Vec::new(),
1045 })
1046 }
1047
1048 pub fn empty(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1070 let file =
1071 OpenOptions::new().write(true).read(true).create_new(true).open(path).map_err(io)?;
1072 let mut header = [0; HEADER as usize];
1073 header[..8].copy_from_slice(MAGIC);
1074 header[8..12].copy_from_slice(&FORMAT.to_le_bytes());
1075 write_at(&file, 0, &header)?;
1076 let catalog = encode_catalog(&[], views)?;
1077 write_at(&file, HEADER, &catalog)?;
1078 file.sync_all().map_err(io)?;
1082 let slot = Slot {
1083 offset: HEADER,
1084 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1085 generation: 1,
1086 hash: checksum(&catalog),
1087 };
1088 write_at(&file, slot_offset(1), &slot.bytes())?;
1089 file.sync_all().map_err(io)?;
1090 Ok(())
1091 }
1092
1093 pub fn next(mut self, name: impl Into<String>, fields: Vec<Field>) -> Result<Self> {
1104 for field in &fields {
1105 type_tag(&field.ty)?;
1106 }
1107 let name = name.into();
1108 let entry = self.close()?;
1109 if self.closed.iter().chain(std::iter::once(&entry)).any(|held| held.name == name) {
1110 return Err(invalid("two tables in one native file have the same name"));
1111 }
1112 let Self { file, at, generation, mut closed, views, .. } = self;
1113 closed.push(entry);
1114 Ok(Self {
1115 file,
1116 at,
1117 generation,
1118 closed,
1119 views,
1120 dictionaries: fields
1121 .iter()
1122 .map(|field| (field.ty == LogicalType::Varchar).then(GlobalDictionary::new))
1123 .collect(),
1124 table: Table {
1125 name,
1126 dictionaries: vec![None; fields.len()],
1127 distincts: vec![None; fields.len()],
1128 fields,
1129 stripes: Vec::new(),
1130 rows: 0,
1131 frequencies: Vec::new(),
1132 clustering: None,
1133 generation,
1134 sections: Vec::new(),
1135 },
1136 order: Vec::new(),
1137 next_order: 0,
1138 pending: Vec::with_capacity(STRIPE_PARTS),
1139 })
1140 }
1141
1142 #[must_use]
1152 pub fn with_views(mut self, views: Vec<ViewEntry>) -> Self {
1153 self.views = views;
1154 self
1155 }
1156
1157 pub fn declare(mut self, clustering: Clustering) -> Result<Self> {
1172 self.table.clustering = Some(Clustering::new(
1175 clustering.columns().to_vec(),
1176 clustering.width(),
1177 &self.table.fields,
1178 )?);
1179 Ok(self)
1180 }
1181
1182 fn put(&mut self, bytes: &[u8]) -> Result<()> {
1187 write_at(&self.file, self.at, bytes)?;
1188 self.at = self
1189 .at
1190 .checked_add(bytes.len() as u64)
1191 .ok_or_else(|| invalid("native file length overflow"))?;
1192 Ok(())
1193 }
1194
1195 pub fn append(&mut self, chunk: &Chunk) -> Result<()> {
1201 let order = (self.next_order, 0);
1202 self.next_order = self.next_order.saturating_add(1);
1203 self.append_at(order, chunk)
1204 }
1205
1206 pub fn append_at(&mut self, order: (u64, u64), chunk: &Chunk) -> Result<()> {
1217 if chunk.is_empty() {
1218 return Ok(());
1219 }
1220 self.admit(chunk)?;
1221 if self.pending.last().is_some_and(|last| last.order > order) {
1222 self.flush_pending()?;
1223 }
1224 self.pending.push(PendingChunk { order, chunk: chunk.clone() });
1229 if self.pending.len() == STRIPE_PARTS {
1230 self.flush_pending()?;
1231 }
1232 Ok(())
1233 }
1234
1235 pub fn append_stripe(&mut self, parts: Vec<((u64, u64), Chunk)>) -> Result<()> {
1251 if parts.len() > STRIPE_PARTS {
1252 return Err(invalid("a stripe was handed more parts than it holds"));
1253 }
1254 self.flush_pending()?;
1257 for (order, chunk) in parts {
1258 if chunk.is_empty() {
1259 continue;
1260 }
1261 self.admit(&chunk)?;
1262 self.pending.push(PendingChunk { order, chunk });
1263 }
1264 self.flush_pending()
1265 }
1266
1267 fn admit(&mut self, chunk: &Chunk) -> Result<()> {
1269 if chunk.width() != self.table.fields.len() {
1270 return Err(invalid("chunk width differs from table schema"));
1271 }
1272 for (index, field) in self.table.fields.iter().enumerate() {
1273 if chunk.column(index)?.logical_type() != &field.ty {
1274 return Err(invalid("chunk type differs from table schema"));
1275 }
1276 }
1277 self.table.rows = self
1278 .table
1279 .rows
1280 .checked_add(chunk.len())
1281 .ok_or_else(|| invalid("row count overflow"))?;
1282 Ok(())
1283 }
1284
1285 fn encode_column(
1316 index: usize,
1317 held: &[PendingChunk],
1318 dictionary: &mut Option<GlobalDictionary>,
1319 ) -> Result<ColumnStripe> {
1320 let deciding = dictionary.as_ref().is_some_and(|held| held.offsets.len() == 1);
1323 let stripe = Self::encode_pages(index, held, dictionary.as_mut())?;
1324 if !deciding {
1325 return Ok(stripe);
1326 }
1327 let rows: usize = held.iter().map(|pending| pending.chunk.len()).sum();
1328 let distinct = dictionary.as_ref().map_or(0, |held| held.offsets.len() - 1);
1329 if rows < DICTIONARY_DECIDE_ROWS
1330 || distinct.saturating_mul(10) <= rows.saturating_mul(DICTIONARY_DISTINCT_IN_TEN)
1331 {
1332 return Ok(stripe);
1333 }
1334 *dictionary = None;
1335 Self::encode_pages(index, held, None)
1336 }
1337
1338 fn encode_pages(
1340 index: usize,
1341 held: &[PendingChunk],
1342 mut dictionary: Option<&mut GlobalDictionary>,
1343 ) -> Result<ColumnStripe> {
1344 let mut stripe = ColumnStripe {
1345 pages: Vec::with_capacity(held.len()),
1346 codes: Vec::with_capacity(held.len()),
1347 sieves: Vec::with_capacity(held.len()),
1348 ranges: Vec::with_capacity(held.len()),
1349 };
1350 for pending in held {
1351 let column = pending.chunk.column(index)?;
1352 let (bytes, unique) = encode(column, dictionary.as_deref_mut())?;
1353 if bytes.len() > MAX_PAGE {
1354 return Err(invalid("column page exceeds the configured bound"));
1355 }
1356 let range = Range::of(column);
1359 let sieve = match dictionary {
1372 Some(_) => None,
1373 None => Sieve::of(column, &range, SIEVE_BUDGET)
1374 .filter(|sieve| sieve.len() < bytes.len()),
1375 };
1376 stripe.pages.push(bytes);
1377 stripe.codes.push(unique);
1378 stripe.sieves.push(sieve);
1379 stripe.ranges.push(range);
1380 }
1381 Ok(stripe)
1382 }
1383
1384 fn encode_columns(&mut self, held: &[PendingChunk]) -> Result<Vec<ColumnStripe>> {
1393 let width = self.table.fields.len();
1394 let workers = std::thread::available_parallelism()
1395 .map_or(1, usize::from)
1396 .min(MAX_ENCODE_WORKERS)
1397 .min(width);
1398 if workers <= 1 || held.len() <= 1 {
1399 return self
1400 .dictionaries
1401 .iter_mut()
1402 .enumerate()
1403 .map(|(index, dictionary)| Self::encode_column(index, held, dictionary))
1404 .collect();
1405 }
1406 let mut jobs: Vec<(usize, Option<GlobalDictionary>)> =
1409 std::mem::take(&mut self.dictionaries).into_iter().enumerate().collect();
1410 jobs.sort_by_key(|(index, _)| weight(&self.table.fields[*index].ty));
1412 let queue = Mutex::new(jobs);
1413 let pieces = std::thread::scope(|scope| {
1414 (0..workers)
1415 .map(|_| {
1416 scope.spawn(|| {
1417 let mut mine = Vec::new();
1418 loop {
1419 let taken = queue
1420 .lock()
1421 .map_err(|_| Error::internal("a native encode worker panicked"))?
1422 .pop();
1423 let Some((index, mut dictionary)) = taken else { break };
1424 let encoded = Self::encode_column(index, held, &mut dictionary)?;
1425 mine.push((index, dictionary, encoded));
1426 }
1427 Ok(mine)
1428 })
1429 })
1430 .collect::<Vec<_>>()
1431 .into_iter()
1432 .map(|handle| {
1433 handle.join().map_err(|_| Error::internal("a native encode worker panicked"))?
1434 })
1435 .collect::<Result<Vec<_>>>()
1436 })?;
1437 let mut dictionaries: Vec<Option<GlobalDictionary>> = (0..width).map(|_| None).collect();
1438 let mut encoded: Vec<Option<ColumnStripe>> = (0..width).map(|_| None).collect();
1439 for piece in pieces {
1440 for (index, dictionary, stripe) in piece {
1441 dictionaries[index] = dictionary;
1442 encoded[index] = Some(stripe);
1443 }
1444 }
1445 self.dictionaries = dictionaries;
1446 encoded
1447 .into_iter()
1448 .map(|stripe| stripe.ok_or_else(|| Error::internal("a column was never encoded")))
1449 .collect()
1450 }
1451
1452 fn flush_pending(&mut self) -> Result<()> {
1454 if self.pending.is_empty() {
1455 return Ok(());
1456 }
1457 let width = self.table.fields.len();
1458 let mut held = std::mem::take(&mut self.pending);
1461 let parts = held.len();
1462 let encoded = self.encode_columns(&held)?;
1463 let mut pages = Vec::with_capacity(width);
1464 let mut memberships = vec![None; width];
1465 let mut ranges = Vec::with_capacity(width);
1466 let mut index = Vec::with_capacity(width.saturating_mul(index_section(parts)?));
1467 for stripe in &encoded {
1468 let offset = self.at;
1469 let section = index.len();
1470 let mut length = 0_usize;
1471 for bytes in &stripe.pages {
1472 write_at(&self.file, self.at + length as u64, bytes)?;
1473 put_u32(
1474 &mut index,
1475 u32::try_from(bytes.len()).map_err(|_| invalid("part length overflow"))?,
1476 );
1477 put_u64(&mut index, checksum(bytes));
1478 length = length
1479 .checked_add(bytes.len())
1480 .ok_or_else(|| invalid("column page length overflow"))?;
1481 }
1482 let hash = checksum(&index[section..]);
1483 put_u64(&mut index, hash);
1484 if length > MAX_PAGE {
1485 return Err(invalid("column page exceeds the configured bound"));
1486 }
1487 self.at = self
1488 .at
1489 .checked_add(length as u64)
1490 .ok_or_else(|| invalid("native file length overflow"))?;
1491 pages.push(Span {
1492 offset,
1493 length: u32::try_from(length).map_err(|_| invalid("page length overflow"))?,
1494 });
1495 ranges.push(merged_range(stripe.ranges.iter().cloned()));
1496 }
1497 for (membership, stripe) in memberships.iter_mut().zip(&encoded) {
1498 if stripe.codes.iter().all(Option::is_none) {
1499 continue;
1500 }
1501 let lists = stripe
1502 .codes
1503 .iter()
1504 .map(|codes| codes.clone().unwrap_or_default())
1505 .collect::<Vec<_>>();
1506 let bytes = encode_membership(&merged_codes(lists));
1507 let offset = self.at;
1508 self.put(&bytes)?;
1509 *membership = Some(Page {
1510 offset,
1511 length: u32::try_from(bytes.len())
1512 .map_err(|_| invalid("membership page length overflow"))?,
1513 hash: checksum(&bytes),
1514 });
1515 }
1516 let mut sieves = vec![None; width];
1517 for (page, stripe) in sieves.iter_mut().zip(&encoded) {
1518 if stripe.sieves.iter().all(Option::is_none) {
1519 continue;
1520 }
1521 let bytes = encode_sieves(stripe.sieves.iter())?;
1522 let offset = self.at;
1523 self.put(&bytes)?;
1524 *page = Some(Page {
1525 offset,
1526 length: u32::try_from(bytes.len())
1527 .map_err(|_| invalid("sieve page length overflow"))?,
1528 hash: checksum(&bytes),
1529 });
1530 }
1531 let mut part_ranges = vec![None; width];
1537 if parts > 1 {
1538 for ((page, stripe), span) in part_ranges.iter_mut().zip(&encoded).zip(&pages) {
1539 let bytes = encode_part_ranges(&stripe.ranges)?;
1540 if bytes.len() >= span.length as usize {
1541 continue;
1542 }
1543 let offset = self.at;
1544 self.put(&bytes)?;
1545 *page = Some(Page {
1546 offset,
1547 length: u32::try_from(bytes.len())
1548 .map_err(|_| invalid("part range page length overflow"))?,
1549 hash: checksum(&bytes),
1550 });
1551 }
1552 }
1553 let offset = self.at;
1554 self.put(&index)?;
1555 let index = Span {
1556 offset,
1557 length: u32::try_from(index.len())
1558 .map_err(|_| invalid("index page length overflow"))?,
1559 };
1560 let mut rows = 0_usize;
1561 let mut lengths = Vec::with_capacity(parts);
1562 let mut span = None;
1563 for pending in held.drain(..) {
1564 let part = pending.chunk.len();
1565 rows = rows.checked_add(part).ok_or_else(|| invalid("row count overflow"))?;
1566 lengths.push(u32::try_from(part).map_err(|_| invalid("part row count overflow"))?);
1567 span = Some(
1568 span.map_or((pending.order, pending.order), |(first, _)| (first, pending.order)),
1569 );
1570 }
1571 self.order.push(span.ok_or_else(|| invalid("a stripe was flushed with no parts"))?);
1572 self.table.stripes.push(Stripe {
1573 rows,
1574 parts: lengths,
1575 index,
1576 pages,
1577 memberships,
1578 sieves,
1579 part_ranges,
1580 zone: Zone::from_ranges(ranges),
1581 });
1582 self.pending = held;
1584 Ok(())
1585 }
1586
1587 fn numeric_frequency(&self, column: usize) -> Result<Option<FrequencySummary>> {
1591 let ty = &self.table.fields[column].ty;
1592 if !matches!(
1593 ty,
1594 LogicalType::TinyInt
1595 | LogicalType::SmallInt
1596 | LogicalType::Integer
1597 | LogicalType::BigInt
1598 | LogicalType::UTinyInt
1599 | LogicalType::USmallInt
1600 | LogicalType::UInteger
1601 | LogicalType::UBigInt
1602 | LogicalType::Date
1603 | LogicalType::Timestamp
1604 ) {
1605 return Ok(None);
1606 }
1607 let mut candidates: HashMap<FrequencyValue, u32> = HashMap::new();
1608 let mut decrements = 0_u64;
1609 self.visit_numeric(column, |_, value| {
1610 if let Some(count) = candidates.get_mut(&value) {
1611 *count = count.saturating_add(1);
1612 } else if candidates.len() < FREQUENCY_CANDIDATES {
1613 candidates.insert(value, 1);
1614 } else {
1615 candidates.retain(|_, count| {
1616 *count -= 1;
1617 *count != 0
1618 });
1619 decrements = decrements.saturating_add(1);
1620 }
1621 })?;
1622 let (exact, ordinals) = if decrements == 0 {
1623 (
1624 candidates
1625 .into_iter()
1626 .map(|(value, count)| (value, u64::from(count)))
1627 .collect::<HashMap<_, _>>(),
1628 Vec::new(),
1629 )
1630 } else {
1631 let mut lower = candidates.values().copied().collect::<Vec<_>>();
1632 lower.sort_unstable_by(|left, right| right.cmp(left));
1633 if lower.len() < FREQUENCY_BUILD_RANK
1634 || u64::from(lower[FREQUENCY_BUILD_RANK - 1]) <= decrements
1635 {
1636 return Ok(None);
1637 }
1638 let mut exact =
1639 candidates.into_keys().map(|value| (value, 0_u64)).collect::<HashMap<_, _>>();
1640 let mut ordinals = Vec::new();
1641 let mut exceeded = false;
1642 self.visit_numeric(column, |ordinal, value| {
1643 if let Some(count) = exact.get_mut(&value) {
1644 *count = count.saturating_add(1);
1645 if !exceeded {
1646 if ordinals.len() < FREQUENCY_ORDINALS {
1647 ordinals.push(ordinal);
1648 } else {
1649 ordinals.clear();
1650 exceeded = true;
1651 }
1652 }
1653 }
1654 })?;
1655 (exact, ordinals)
1656 };
1657 let mut entries = exact
1658 .into_iter()
1659 .map(|(value, count)| FrequencyEntry { value, count })
1660 .collect::<Vec<_>>();
1661 entries.sort_unstable_by(|left, right| {
1662 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
1663 });
1664 let omitted_max =
1665 entries.get(FREQUENCY_ENTRIES).map_or(decrements, |entry| decrements.max(entry.count));
1666 entries.truncate(FREQUENCY_ENTRIES);
1667 Ok(Some(FrequencySummary { entries, omitted_max, ordinals }))
1668 }
1669
1670 fn visit_numeric(
1671 &self,
1672 column: usize,
1673 mut visit: impl FnMut(u64, FrequencyValue),
1674 ) -> Result<()> {
1675 let ty = &self.table.fields[column].ty;
1676 let mut start = 0_u64;
1677 for stripe in &self.table.stripes {
1678 let spans = read_index(&self.file, stripe, column)?;
1679 let page = stripe.pages[column];
1680 let mut bytes = vec![0; page.length as usize];
1681 read_at(&self.file, page.offset, &mut bytes)?;
1682 for (span, &rows) in spans.iter().zip(&stripe.parts) {
1683 let part = part_bytes(&bytes, *span)?;
1684 if checksum(part) != span.hash {
1685 return Err(invalid("column page checksum differs while building frequencies"));
1686 }
1687 let rows = rows as usize;
1688 let vector = decode(ty, rows, part, None)?;
1689 for row in 0..rows {
1691 let value = if vector.is_null_at(row) {
1692 FrequencyValue::Null
1693 } else {
1694 let widened = match vector.signed_at(row) {
1698 Some(value) => Some(value),
1699 None => match vector.value_at(row) {
1700 Value::UTinyInt(value) => Some(i128::from(value)),
1701 Value::USmallInt(value) => Some(i128::from(value)),
1702 Value::UInteger(value) => Some(i128::from(value)),
1703 Value::UBigInt(value) => Some(i128::from(value)),
1704 _ => None,
1705 },
1706 };
1707 FrequencyValue::Integer(widened.ok_or_else(|| {
1708 invalid("numeric frequency page did not contain an integer value")
1709 })?)
1710 };
1711 visit(start.saturating_add(row as u64), value);
1712 }
1713 start = start.saturating_add(rows as u64);
1714 }
1715 }
1716 Ok(())
1717 }
1718
1719 fn numeric_frequencies(&self) -> Result<Vec<Option<FrequencySummary>>> {
1727 let mut columns = self
1728 .table
1729 .fields
1730 .iter()
1731 .enumerate()
1732 .filter_map(|(column, field)| {
1733 matches!(
1734 field.ty,
1735 LogicalType::TinyInt
1736 | LogicalType::SmallInt
1737 | LogicalType::Integer
1738 | LogicalType::BigInt
1739 | LogicalType::UTinyInt
1740 | LogicalType::USmallInt
1741 | LogicalType::UInteger
1742 | LogicalType::UBigInt
1743 | LogicalType::Date
1744 | LogicalType::Timestamp
1745 )
1746 .then_some(column)
1747 })
1748 .collect::<Vec<_>>();
1749 let workers = std::thread::available_parallelism()
1750 .map_or(1, usize::from)
1751 .min(MAX_FREQUENCY_WORKERS)
1752 .min(columns.len());
1753 if workers <= 1 {
1754 let mut frequencies = vec![None; self.table.fields.len()];
1755 for column in columns {
1756 frequencies[column] = self.numeric_frequency(column)?;
1757 }
1758 return Ok(frequencies);
1759 }
1760 columns.sort_by_key(|&column| weight(&self.table.fields[column].ty));
1763 let queue = Mutex::new(columns);
1764 let pieces = std::thread::scope(|scope| {
1765 (0..workers)
1766 .map(|_| {
1767 scope.spawn(|| {
1768 let mut mine = Vec::new();
1769 loop {
1770 let taken = queue
1771 .lock()
1772 .map_err(|_| Error::internal("a native frequency worker panicked"))?
1773 .pop();
1774 let Some(column) = taken else { break };
1775 mine.push((column, self.numeric_frequency(column)?));
1776 }
1777 Ok(mine)
1778 })
1779 })
1780 .collect::<Vec<_>>()
1781 .into_iter()
1782 .map(|handle| {
1783 handle
1784 .join()
1785 .map_err(|_| Error::internal("a native frequency worker panicked"))?
1786 })
1787 .collect::<Result<Vec<_>>>()
1788 })?;
1789 let mut frequencies = vec![None; self.table.fields.len()];
1790 for piece in pieces {
1791 for (column, summary) in piece {
1792 frequencies[column] = summary;
1793 }
1794 }
1795 Ok(frequencies)
1796 }
1797
1798 fn close(&mut self) -> Result<Entry> {
1809 self.flush_pending()?;
1810 let mut stripes = std::mem::take(&mut self.order)
1811 .into_iter()
1812 .zip(std::mem::take(&mut self.table.stripes))
1813 .collect::<Vec<_>>();
1814 stripes.sort_by_key(|(order, _)| order.0);
1815 let mut previous: Option<(u64, u64)> = None;
1816 for ((first, last), _) in &stripes {
1817 if previous.is_some_and(|previous| previous >= *first) {
1818 return Err(invalid("chunks did not arrive in source order"));
1819 }
1820 previous = Some(*last);
1821 }
1822 self.table.stripes = stripes.into_iter().map(|(_, stripe)| stripe).collect();
1823 self.table.frequencies = self.numeric_frequencies()?;
1824 let dictionaries = std::mem::take(&mut self.dictionaries);
1825 let orders = rankings(&dictionaries)?;
1826 for (index, (dictionary, order)) in dictionaries.into_iter().zip(orders).enumerate() {
1827 let Some(dictionary) = dictionary else { continue };
1828 self.table.distincts[index] =
1832 Some(dictionary.counts.iter().filter(|count| **count != 0).count() as u64);
1833 self.table.frequencies[index] = Some(code_frequency(&dictionary));
1834 let encoded = encode_global_dictionary(dictionary, &order)?;
1835 let offset = self.at;
1836 self.put(&encoded.index)?;
1837 self.put(&encoded.ranks)?;
1838 for block in &encoded.payload {
1839 self.put(block)?;
1840 }
1841 let payload_len =
1842 encoded.payload.iter().try_fold(0_usize, |len, block| len.checked_add(block.len()));
1843 let length = payload_len
1844 .and_then(|len| len.checked_add(encoded.index.len()))
1845 .and_then(|len| len.checked_add(encoded.ranks.len()))
1846 .ok_or_else(|| invalid("dictionary page length overflow"))?;
1847 self.table.dictionaries[index] = Some(Page {
1848 offset,
1849 length: u32::try_from(length)
1850 .map_err(|_| invalid("dictionary page length overflow"))?,
1851 hash: checksum(&encoded.index),
1852 });
1853 }
1854 let directory = encode_directory(&self.table)?;
1855 if directory.len() > MAX_DIRECTORY {
1856 return Err(invalid("directory exceeds the configured bound"));
1857 }
1858 let offset = self.at;
1859 self.put(&directory)?;
1860 Ok(Entry {
1861 name: self.table.name.clone(),
1862 fields: self.table.fields.clone(),
1863 rows: self.table.rows,
1864 directory: Page {
1865 offset,
1866 length: u32::try_from(directory.len())
1867 .map_err(|_| invalid("directory length overflow"))?,
1868 hash: checksum(&directory),
1869 },
1870 })
1871 }
1872
1873 pub fn finish(mut self) -> Result<Table> {
1883 let entry = self.close()?;
1884 let mut tables = std::mem::take(&mut self.closed);
1885 tables.push(entry);
1886 let catalog = encode_catalog(&tables, &self.views)?;
1887 if catalog.len() > MAX_DIRECTORY {
1888 return Err(invalid("catalog exceeds the configured bound"));
1889 }
1890 let offset = self.at;
1891 self.put(&catalog)?;
1892 self.file.sync_all().map_err(io)?;
1896 let slot = Slot {
1897 offset,
1898 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1899 generation: self.generation,
1900 hash: checksum(&catalog),
1901 };
1902 write_at(&self.file, slot_offset(self.generation), &slot.bytes())?;
1907 self.file.sync_all().map_err(io)?;
1908 Ok(self.table)
1909 }
1910
1911 pub fn restate(path: impl AsRef<Path>, views: &[ViewEntry]) -> Result<()> {
1928 let path = path.as_ref();
1929 let (_, size, slot, bytes, _) = slot_bytes(path)?;
1930 let (closed, _) = decode_catalog(&bytes, size)?;
1931 let generation = slot
1932 .generation
1933 .checked_add(1)
1934 .ok_or_else(|| invalid("native file generation overflow"))?;
1935 let catalog = encode_catalog(&closed, views)?;
1936 if catalog.len() > MAX_DIRECTORY {
1937 return Err(invalid("catalog exceeds the configured bound"));
1938 }
1939 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
1940 write_at(&file, size, &catalog)?;
1941 file.sync_all().map_err(io)?;
1942 let slot = Slot {
1943 offset: size,
1944 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
1945 generation,
1946 hash: checksum(&catalog),
1947 };
1948 write_at(&file, slot_offset(generation), &slot.bytes())?;
1949 file.sync_all().map_err(io)?;
1950 Ok(())
1951 }
1952}
1953
1954fn append(file: &File, at: &mut u64, bytes: &[u8]) -> Result<u64> {
1960 let offset = *at;
1961 write_at(file, offset, bytes)?;
1962 *at =
1963 at.checked_add(bytes.len() as u64).ok_or_else(|| invalid("native file length overflow"))?;
1964 Ok(offset)
1965}
1966
1967fn write_section(
1973 file: &File,
1974 at: &mut u64,
1975 one: §ion::Attachment<'_>,
1976 generation: u64,
1977) -> Result<Section> {
1978 if one.header_bytes as usize > one.bytes.len() {
1979 return Err(invalid("a section's header is longer than its payload"));
1980 }
1981 let mut extents = Vec::new();
1982 let mut first = 0_u64;
1983 for chunk in one.bytes.chunks(section::MAX_EXTENT as usize) {
1984 let offset = append(file, at, chunk)?;
1985 extents.push(section::Extent {
1986 offset,
1987 length: u32::try_from(chunk.len()).map_err(|_| invalid("extent length overflow"))?,
1988 hash: checksum(chunk),
1989 first,
1990 });
1991 first += chunk.len() as u64;
1992 }
1993 let mut table = Vec::with_capacity(extents.len() * section::EXTENT_BYTES);
1994 section::encode_extents(&extents, &mut table)?;
1995 let extent_page = if table.is_empty() { 0 } else { append(file, at, &table)? };
1999 Ok(Section {
2000 kind: one.kind,
2001 id: one.id,
2002 generation,
2003 extents: u32::try_from(extents.len()).map_err(|_| invalid("too many extents"))?,
2004 extent_page,
2005 extent_bytes: u32::try_from(table.len()).map_err(|_| invalid("extent table overflow"))?,
2006 hash: checksum(&table),
2007 flags: one.flags,
2008 header_bytes: one.header_bytes,
2009 })
2010}
2011
2012pub fn attach(
2036 path: impl AsRef<Path>,
2037 table: &str,
2038 attachments: &[section::Attachment<'_>],
2039) -> Result<Table> {
2040 let path = path.as_ref();
2041 let (_, size, slot, bytes, _) = slot_bytes(path)?;
2042 let (mut entries, views) = decode_catalog(&bytes, size)?;
2043 let at = entries
2044 .iter()
2045 .position(|entry| entry.name == table)
2046 .ok_or_else(|| invalid(&format!("the file holds no table called {table}")))?;
2047 let file = OpenOptions::new().write(true).read(true).open(path).map_err(io)?;
2048 let mut version = [0; 4];
2049 read_at(&file, 8, &mut version)?;
2050 let version = u32::from_le_bytes(version);
2051 if version != FORMAT {
2057 return Err(invalid(&format!(
2058 "the file is format {version} and a graph section needs format {FORMAT}, so it has \
2059 to be written again"
2060 )));
2061 }
2062 let mut directory = vec![0; entries[at].directory.length as usize];
2063 read_at(&file, entries[at].directory.offset, &mut directory)?;
2064 if checksum(&directory) != entries[at].directory.hash {
2065 return Err(invalid(&format!("the directory of table {table} does not checksum")));
2066 }
2067 let mut held = decode_directory(&directory, size)?;
2068 let mut cursor = size;
2069 for one in attachments {
2070 let written = write_section(&file, &mut cursor, one, held.generation)?;
2071 held.sections.retain(|old| !(old.kind == one.kind && old.id == one.id));
2072 held.sections.push(written);
2073 }
2074 if held.sections.len() > MAX_SECTIONS {
2075 return Err(invalid("the table would name more sections than the bound allows"));
2076 }
2077 let encoded = encode_directory(&held)?;
2078 if encoded.len() > MAX_DIRECTORY {
2079 return Err(invalid("directory exceeds the configured bound"));
2080 }
2081 let offset = append(&file, &mut cursor, &encoded)?;
2082 entries[at].directory = Page {
2083 offset,
2084 length: u32::try_from(encoded.len()).map_err(|_| invalid("directory length overflow"))?,
2085 hash: checksum(&encoded),
2086 };
2087 let catalog = encode_catalog(&entries, &views)?;
2090 if catalog.len() > MAX_DIRECTORY {
2091 return Err(invalid("catalog exceeds the configured bound"));
2092 }
2093 let offset = append(&file, &mut cursor, &catalog)?;
2094 file.sync_all().map_err(io)?;
2095 let generation =
2096 slot.generation.checked_add(1).ok_or_else(|| invalid("native file generation overflow"))?;
2097 let committed = Slot {
2098 offset,
2099 length: u32::try_from(catalog.len()).map_err(|_| invalid("catalog length overflow"))?,
2100 generation,
2101 hash: checksum(&catalog),
2102 };
2103 write_at(&file, slot_offset(generation), &committed.bytes())?;
2104 file.sync_all().map_err(io)?;
2105 Ok(held)
2106}
2107
2108#[derive(Debug, Clone)]
2110pub struct Reader {
2111 file: Arc<File>,
2112 table: Arc<Table>,
2113 dictionaries: Arc<Vec<OnceLock<Arc<Vector>>>>,
2114 loading: Arc<Vec<Mutex<()>>>,
2123 opened: Arc<AtomicUsize>,
2127 sieves: Arc<Vec<Vec<SieveSlot>>>,
2131 part_ranges: Arc<Vec<Vec<RangeSlot>>>,
2134 places: Arc<Vec<Place>>,
2136 cache: Arc<Vec<Mutex<Cached>>>,
2137 pages: Arc<AtomicUsize>,
2140 indexes: Arc<AtomicUsize>,
2143 kept: Arc<AtomicUsize>,
2146 size: u64,
2148 directory: u64,
2150 opening: Opening,
2152}
2153
2154#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2166pub struct Opening {
2167 pub reads: u32,
2170 pub bytes: u64,
2172}
2173
2174#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2176pub struct Reads {
2177 pub opening: Opening,
2179 pub pages: usize,
2181 pub indexes: usize,
2183 pub dictionaries: usize,
2186}
2187
2188#[derive(Debug, Clone, Copy)]
2190struct Place {
2191 stripe: u32,
2192 part: u32,
2193 rows: u32,
2194}
2195
2196#[derive(Debug, Clone, Copy)]
2198struct PartSpan {
2199 start: usize,
2200 length: usize,
2201 hash: u64,
2202}
2203
2204#[derive(Debug, Clone)]
2210struct CachedColumn {
2211 stripe: usize,
2212 index: Arc<Vec<PartSpan>>,
2213 page: Option<Arc<Vec<u8>>>,
2214}
2215
2216#[derive(Debug, Default)]
2236struct Cached {
2237 pages: Vec<Option<Arc<Vec<u8>>>>,
2238 order: VecDeque<usize>,
2239 loading: Vec<usize>,
2240 index: Vec<Option<Arc<Vec<PartSpan>>>>,
2241}
2242
2243const CACHED_STRIPES_PER_COLUMN: usize = 4;
2255
2256type SieveSlot = OnceLock<Arc<Vec<Option<Sieve>>>>;
2258
2259type RangeSlot = OnceLock<Arc<Vec<Range>>>;
2260
2261#[derive(Debug)]
2262struct NativeText {
2263 file: Arc<File>,
2264 values: usize,
2266 offsets: Vec<u8>,
2275 offset_bits: usize,
2278 ranks: usize,
2280 rank_at: u64,
2284 rank_ends: Vec<u64>,
2288 rank_hashes: Vec<u64>,
2289 rank_blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2290 code_bits: usize,
2293 code_ranks: OnceLock<Option<Vec<u32>>>,
2300 payload: u64,
2301 ends: Vec<u64>,
2304 hashes: Vec<u64>,
2305 blocks: Vec<OnceLock<Result<Vec<u8>>>>,
2307 keep_budget: usize,
2310 payload_kept: AtomicUsize,
2318 searched: Mutex<HashMap<Vec<u8>, (usize, bool)>>,
2335}
2336
2337const TEXT_SEARCH_MEMO: usize = 64;
2342
2343const TEXT_PAYLOAD_VALUES: usize = 1024;
2359
2360const TEXT_KEEP_BUDGET: usize = 256 * 1024 * 1024;
2381
2382const TEXT_OFFSET_RUN: usize = 512;
2389
2390const DICTIONARY_HEADER: usize = 16;
2393
2394const TEXT_RANK_BLOCK: usize = 512;
2405
2406const RANK_BLOCK_HEADER: usize = size_of::<u64>() + 1;
2420
2421impl NativeText {
2422 fn payload_block(&self, block: usize) -> Result<Option<&[u8]>> {
2429 let Some(slot) = self.blocks.get(block) else { return Ok(None) };
2430 let bytes = slot.get_or_init(|| self.decode_block(block)).as_ref().map_err(Clone::clone)?;
2431 Ok(Some(bytes.as_slice()))
2432 }
2433
2434 fn decode_block(&self, block: usize) -> Result<Vec<u8>> {
2439 let start = if block == 0 { 0 } else { self.ends[block - 1] };
2440 let end = self.ends[block];
2441 let len = end
2442 .checked_sub(start)
2443 .ok_or_else(|| invalid("global dictionary block ends before it starts"))?;
2444 let mut stored = vec![
2445 0;
2446 usize::try_from(len).map_err(|_| invalid(
2447 "global dictionary block does not fit in memory"
2448 ))?
2449 ];
2450 read_at(&self.file, self.payload + start, &mut stored)?;
2451 if checksum(&stored) != self.hashes[block] {
2452 return Err(invalid("global dictionary payload checksum differs"));
2453 }
2454 let first = block * TEXT_PAYLOAD_VALUES;
2455 let last = (first + TEXT_PAYLOAD_VALUES).min(self.values);
2456 let want = self.end_within(last - 1)? as usize;
2457 let values = string::decode_flat(&stored)?;
2458 if values.len() != last - first {
2459 return Err(invalid("global dictionary block holds the wrong value count"));
2460 }
2461 let bytes = values.into_bytes();
2462 if bytes.len() != want {
2463 return Err(invalid("global dictionary block decodes to the wrong length"));
2464 }
2465 Ok(bytes)
2466 }
2467
2468 fn end_within(&self, index: usize) -> Result<u32> {
2470 let run = index / TEXT_OFFSET_RUN;
2471 let bytes = self
2472 .offsets
2473 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2474 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2475 let end = bitpack::tail_at(bytes, self.offset_bits, index % TEXT_OFFSET_RUN)
2476 .map_err(|_| invalid("global dictionary offsets are short"))?;
2477 u32::try_from(end).map_err(|_| invalid("global dictionary offset is past the payload"))
2478 }
2479
2480 fn ends_within(&self, first: usize, last: usize) -> Result<Vec<u64>> {
2493 let mut ends = Vec::with_capacity(last.saturating_sub(first));
2494 let mut at = first;
2495 while at < last {
2496 let run = at / TEXT_OFFSET_RUN;
2497 let stop = ((run + 1) * TEXT_OFFSET_RUN).min(last);
2498 let held = self.values.saturating_sub(run * TEXT_OFFSET_RUN).min(TEXT_OFFSET_RUN);
2499 let bytes = self
2500 .offsets
2501 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2502 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2503 let run_ends = bitpack::unpack_tail(bytes, self.offset_bits, held)
2504 .map_err(|_| invalid("global dictionary offsets are short"))?;
2505 let within = run_ends
2506 .get(at % TEXT_OFFSET_RUN..stop - run * TEXT_OFFSET_RUN)
2507 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2508 ends.extend_from_slice(within);
2509 at = stop;
2510 }
2511 Ok(ends)
2512 }
2513
2514 fn start_within(&self, index: usize) -> Result<u32> {
2517 if index % TEXT_PAYLOAD_VALUES == 0 { Ok(0) } else { self.end_within(index - 1) }
2518 }
2519
2520 fn span_within(&self, index: usize) -> Result<(u32, u32)> {
2528 let within = index % TEXT_OFFSET_RUN;
2529 let (start, end) = if within == 0 {
2530 (self.start_within(index)?, self.end_within(index)?)
2531 } else {
2532 let run = index / TEXT_OFFSET_RUN;
2533 let bytes = self
2534 .offsets
2535 .get(run * TEXT_OFFSET_RUN / 8 * self.offset_bits..)
2536 .ok_or_else(|| invalid("global dictionary offsets are short"))?;
2537 let (start, end) = bitpack::tail_pair(bytes, self.offset_bits, within)
2538 .map_err(|_| invalid("global dictionary offsets are short"))?;
2539 let ends = u32::try_from(end)
2540 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2541 let starts = u32::try_from(start)
2542 .map_err(|_| invalid("global dictionary offset is past the payload"))?;
2543 (starts, ends)
2544 };
2545 if start > end {
2546 return Err(invalid("global dictionary value ends before it starts"));
2547 }
2548 Ok((start, end))
2549 }
2550
2551 fn rank_parts(&self, rank: usize) -> Result<(&[u8], usize)> {
2558 let slot = self
2559 .rank_blocks
2560 .get(rank / TEXT_RANK_BLOCK)
2561 .ok_or_else(|| invalid("global dictionary rank is past the order"))?;
2562 let block = slot
2563 .get_or_init(|| {
2564 let which = rank / TEXT_RANK_BLOCK;
2565 let start = if which == 0 { 0 } else { self.rank_ends[which - 1] };
2566 let end = self.rank_ends[which];
2567 let mut bytes = vec![0; (end - start) as usize];
2568 read_at(&self.file, self.rank_at + start, &mut bytes)?;
2569 if checksum(&bytes)
2570 != *self
2571 .rank_hashes
2572 .get(rank / TEXT_RANK_BLOCK)
2573 .ok_or_else(|| invalid("global dictionary rank block has no checksum"))?
2574 {
2575 return Err(invalid("global dictionary rank checksum differs"));
2576 }
2577 Ok(bytes)
2578 })
2579 .as_ref()
2580 .map_err(Clone::clone)?;
2581 Ok((block.as_slice(), rank % TEXT_RANK_BLOCK))
2582 }
2583
2584 fn head_at(&self, rank: usize) -> Result<u64> {
2586 let (block, within) = self.rank_parts(rank)?;
2587 let (base, width, packed) = rank_heads(block)?;
2588 let above = bitpack::tail_at(packed, width, within)
2589 .map_err(|_| invalid("global dictionary rank block is short of heads"))?;
2590 Ok(base.wrapping_add(above))
2591 }
2592
2593 fn rank_codes<'block>(&self, block: &'block [u8], count: usize) -> Result<&'block [u8]> {
2595 let (_, width, packed) = rank_heads(block)?;
2596 packed
2597 .get(bitpack::tail_len(count, width)..)
2598 .ok_or_else(|| invalid("global dictionary rank block is short of codes"))
2599 }
2600
2601 fn rank_block_len(&self, rank: usize) -> usize {
2603 let first = rank / TEXT_RANK_BLOCK * TEXT_RANK_BLOCK;
2604 TEXT_RANK_BLOCK.min(self.ranks - first)
2605 }
2606}
2607
2608fn rank_heads(block: &[u8]) -> Result<(u64, usize, &[u8])> {
2610 let header = block
2611 .get(..RANK_BLOCK_HEADER)
2612 .ok_or_else(|| invalid("global dictionary rank block is short"))?;
2613 let base = u64::from_le_bytes(header[..8].try_into().expect("eight bytes"));
2614 let width = header[8] as usize;
2615 if width > 64 {
2616 return Err(invalid("global dictionary rank block packs heads past a word"));
2617 }
2618 Ok((base, width, &block[RANK_BLOCK_HEADER..]))
2619}
2620
2621fn offset_width(offsets: &[u32]) -> usize {
2628 let values = offsets.len() - 1;
2629 let mut span = 0;
2630 for first in (0..values).step_by(TEXT_PAYLOAD_VALUES) {
2631 let last = (first + TEXT_PAYLOAD_VALUES).min(values);
2632 span = span.max(offsets[last] - offsets[first]);
2633 }
2634 (u32::BITS - span.leading_zeros()) as usize
2635}
2636
2637fn offset_bytes(values: usize, bits: usize) -> usize {
2640 let full = values / TEXT_OFFSET_RUN;
2641 let rest = values % TEXT_OFFSET_RUN;
2642 full * TEXT_OFFSET_RUN / 8 * bits + bitpack::tail_len(rest, bits)
2643}
2644
2645fn encode_offsets(offsets: &[u32], bits: usize, out: &mut Vec<u8>) -> Result<()> {
2647 let values = offsets.len() - 1;
2648 let mut run = Vec::with_capacity(TEXT_OFFSET_RUN);
2649 for first in (0..values).step_by(TEXT_OFFSET_RUN) {
2650 let last = (first + TEXT_OFFSET_RUN).min(values);
2651 let base = offsets[first / TEXT_PAYLOAD_VALUES * TEXT_PAYLOAD_VALUES];
2652 run.clear();
2653 run.extend((first..last).map(|value| u64::from(offsets[value + 1] - base)));
2654 bitpack::pack_tail(&run, bits, out)
2655 .map_err(|_| invalid("global dictionary offsets do not pack"))?;
2656 }
2657 Ok(())
2658}
2659
2660fn code_width(values: usize) -> usize {
2662 match u64::try_from(values).unwrap_or(u64::MAX) {
2663 0 | 1 => 0,
2664 last => (u64::BITS - (last - 1).leading_zeros()) as usize,
2665 }
2666}
2667
2668impl TextSource for NativeText {
2669 fn len(&self) -> usize {
2670 self.values
2671 }
2672
2673 fn bytes_at(&self, index: usize) -> Result<Option<&[u8]>> {
2674 if index >= self.values {
2675 return Ok(None);
2676 }
2677 let (start, end) = self.span_within(index)?;
2678 if start == end {
2679 return Ok(Some(&[]));
2680 }
2681 let block = index / TEXT_PAYLOAD_VALUES;
2684 let Some(bytes) = self.payload_block(block)? else { return Ok(None) };
2685 Ok(bytes.get(start as usize..end as usize))
2686 }
2687
2688 fn bytes_len_at(&self, index: usize) -> Result<Option<usize>> {
2689 if index >= self.values {
2690 return Ok(None);
2691 }
2692 let (start, end) = self.span_within(index)?;
2693 Ok(Some((end - start) as usize))
2694 }
2695
2696 fn sweep(
2709 &self,
2710 first: usize,
2711 limit: usize,
2712 body: &mut dyn FnMut(usize, &[u8]) -> Result<()>,
2713 ) -> Result<usize> {
2714 let limit = limit.min(self.values);
2715 if first >= limit {
2716 return Ok(first);
2717 }
2718 let block = first / TEXT_PAYLOAD_VALUES;
2719 let last = ((block + 1) * TEXT_PAYLOAD_VALUES).min(limit);
2720 let decoded;
2721 let bytes: &[u8] = match self.blocks.get(block).and_then(OnceLock::get) {
2722 Some(Ok(kept)) => kept,
2723 _ if self.payload_kept.load(Atomic::Relaxed) < self.keep_budget => {
2724 let kept = self
2725 .payload_block(block)?
2726 .ok_or_else(|| invalid("global dictionary block is past the payload"))?;
2727 self.payload_kept.fetch_add(kept.len(), Atomic::Relaxed);
2728 kept
2729 }
2730 _ => {
2731 decoded = self.decode_block(block)?;
2732 &decoded
2733 }
2734 };
2735 let ends = self.ends_within(first, last)?;
2736 if ends.len() != last - first {
2737 return Err(invalid("global dictionary offsets are short"));
2738 }
2739 let mut start = u64::from(self.start_within(first)?);
2740 for (index, &end) in (first..last).zip(&ends) {
2743 let value = usize::try_from(start)
2744 .ok()
2745 .zip(usize::try_from(end).ok())
2746 .and_then(|(from, to)| bytes.get(from..to))
2747 .ok_or_else(|| invalid("global dictionary value is past its block"))?;
2748 body(index, value)?;
2749 start = end;
2750 }
2751 Ok(last)
2752 }
2753
2754 fn ranks(&self) -> Option<usize> {
2755 (self.ranks > 0).then_some(self.ranks)
2756 }
2757
2758 fn below(&self, ranks: usize, wanted: &[u8]) -> Result<(usize, bool)> {
2766 let mut memo = self.searched.lock().map_err(|_| invalid("a poisoned dictionary search"))?;
2767 if let Some(&answer) = memo.get(wanted) {
2768 return Ok(answer);
2769 }
2770 let answer = search_below(self, ranks, wanted)?;
2771 if memo.len() >= TEXT_SEARCH_MEMO {
2772 memo.clear();
2773 }
2774 memo.insert(wanted.to_vec(), answer);
2775 Ok(answer)
2776 }
2777
2778 fn compare_rank(&self, rank: usize, wanted: &[u8]) -> Result<Ordering> {
2779 let settled = self.head_at(rank)?.cmp(&head(wanted));
2783 if settled != Ordering::Equal {
2784 return Ok(settled);
2785 }
2786 let code = self.code_at_rank(rank)?;
2787 let bytes = self
2788 .bytes_at(code as usize)?
2789 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
2790 Ok(bytes.cmp(wanted))
2791 }
2792
2793 fn code_at_rank(&self, rank: usize) -> Result<u32> {
2794 let (block, within) = self.rank_parts(rank)?;
2795 let codes = self.rank_codes(block, self.rank_block_len(rank))?;
2796 let code = bitpack::tail_at(codes, self.code_bits, within)
2797 .map_err(|_| invalid("global dictionary rank block is short of codes"))?;
2798 let code = u32::try_from(code)
2799 .map_err(|_| invalid("global dictionary order names a code it does not have"))?;
2800 if code as usize >= self.len() {
2801 return Err(invalid("global dictionary order names a code it does not have"));
2802 }
2803 Ok(code)
2804 }
2805
2806 fn code_ranks(&self) -> Option<&[u32]> {
2807 if self.ranks == 0 || self.ranks != self.len() {
2811 return None;
2812 }
2813 self.code_ranks
2814 .get_or_init(|| {
2815 let mut ranks = vec![u32::MAX; self.ranks];
2816 for first in (0..self.ranks).step_by(TEXT_RANK_BLOCK) {
2819 let (block, _) = self.rank_parts(first).ok()?;
2820 let count = self.rank_block_len(first);
2821 let codes = self.rank_codes(block, count).ok()?;
2822 for (within, code) in bitpack::unpack_tail(codes, self.code_bits, count)
2823 .ok()?
2824 .into_iter()
2825 .enumerate()
2826 {
2827 let code = usize::try_from(code).ok()?;
2828 *ranks.get_mut(code)? = u32::try_from(first + within).ok()?;
2829 }
2830 }
2831 if ranks.contains(&u32::MAX) {
2832 return None;
2833 }
2834 Some(ranks)
2835 })
2836 .as_deref()
2837 }
2838
2839 fn footprint(&self) -> usize {
2840 self.offsets.capacity()
2841 + self
2842 .code_ranks
2843 .get()
2844 .and_then(Option::as_ref)
2845 .map_or(0, |ranks| ranks.capacity() * size_of::<u32>())
2846 + self.rank_hashes.capacity() * size_of::<u64>()
2847 + self.rank_ends.capacity() * size_of::<u64>()
2848 + self.rank_blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2849 + self
2850 .rank_blocks
2851 .iter()
2852 .filter_map(OnceLock::get)
2853 .filter_map(|result| result.as_ref().ok())
2854 .map(Vec::capacity)
2855 .sum::<usize>()
2856 + self.blocks.capacity() * size_of::<OnceLock<Result<Vec<u8>>>>()
2857 + self.hashes.capacity() * size_of::<u64>()
2858 + self.ends.capacity() * size_of::<u64>()
2859 + self
2860 .blocks
2861 .iter()
2862 .filter_map(OnceLock::get)
2863 .filter_map(|result| result.as_ref().ok())
2864 .map(Vec::capacity)
2865 .sum::<usize>()
2866 }
2867}
2868
2869fn places(table: &Table) -> Result<Vec<Place>> {
2871 let mut places = Vec::with_capacity(table.stripes.len().saturating_mul(STRIPE_PARTS));
2872 for (at, stripe) in table.stripes.iter().enumerate() {
2873 let index = u32::try_from(at).map_err(|_| invalid("too many stripes"))?;
2874 for (part, &rows) in stripe.parts.iter().enumerate() {
2875 places.push(Place {
2876 stripe: index,
2877 part: u32::try_from(part).map_err(|_| invalid("too many parts in a stripe"))?,
2878 rows,
2879 });
2880 }
2881 }
2882 Ok(places)
2883}
2884
2885fn read_index(file: &File, stripe: &Stripe, column: usize) -> Result<Vec<PartSpan>> {
2890 let parts = stripe.parts.len();
2891 let section = index_section(parts)?;
2892 let at = column.checked_mul(section).ok_or_else(|| invalid("index page offset overflow"))?;
2893 let end = at.checked_add(section).ok_or_else(|| invalid("index page offset overflow"))?;
2894 if end > stripe.index.length as usize {
2895 return Err(invalid("index page is shorter than its columns"));
2896 }
2897 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
2898 let mut bytes = vec![0; section];
2899 let offset = stripe
2900 .index
2901 .offset
2902 .checked_add(at as u64)
2903 .ok_or_else(|| invalid("index page offset overflow"))?;
2904 read_at(file, offset, &mut bytes)?;
2905 let entries = section - size_of::<u64>();
2906 let stored = u64::from_le_bytes(bytes[entries..].try_into().expect("eight bytes"));
2907 if checksum(&bytes[..entries]) != stored {
2908 return Err(invalid(&format!(
2911 "index page section checksum differs, column {column} of {parts} parts at {offset}, \
2912 wanted {stored:016x} and got {:016x}",
2913 checksum(&bytes[..entries]),
2914 )));
2915 }
2916 let mut spans = Vec::with_capacity(parts);
2917 let mut start = 0_usize;
2918 for part in 0..parts {
2919 let at = part * INDEX_ENTRY;
2920 let length = u32::from_le_bytes(bytes[at..at + 4].try_into().expect("four bytes")) as usize;
2921 let hash = u64::from_le_bytes(bytes[at + 4..at + 12].try_into().expect("eight bytes"));
2922 spans.push(PartSpan { start, length, hash });
2923 start = start.checked_add(length).ok_or_else(|| invalid("column page length overflow"))?;
2924 }
2925 if start != page.length as usize {
2926 return Err(invalid("column page length differs from its index"));
2927 }
2928 Ok(spans)
2929}
2930
2931fn part_bytes(page: &[u8], span: PartSpan) -> Result<&[u8]> {
2933 let end = span.start.checked_add(span.length).ok_or_else(|| invalid("part range overflow"))?;
2934 page.get(span.start..end).ok_or_else(|| invalid("part exceeds its column page"))
2935}
2936
2937fn remember(cached: &mut Cached, held: &CachedColumn, kept: usize) {
2942 if let Some(slot) = cached.index.get_mut(held.stripe) {
2943 if slot.is_none() {
2944 *slot = Some(Arc::clone(&held.index));
2945 }
2946 }
2947 let Some(page) = held.page.clone() else { return };
2948 let Some(slot) = cached.pages.get_mut(held.stripe) else { return };
2949 if slot.is_none() {
2950 cached.order.push_back(held.stripe);
2951 }
2952 *slot = Some(page);
2953 while cached.order.len() > kept.max(1) {
2954 let Some(oldest) = cached.order.pop_front() else { break };
2955 if let Some(slot) = cached.pages.get_mut(oldest) {
2956 *slot = None;
2957 }
2958 }
2959}
2960
2961#[derive(Debug, Clone)]
2970pub struct Catalog {
2971 file: Arc<File>,
2972 size: u64,
2973 entries: Arc<Vec<Entry>>,
2974 views: Arc<Vec<ViewEntry>>,
2976 opening: Opening,
2977}
2978
2979impl Catalog {
2980 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
2986 let (file, size, _, bytes, opening) = slot_bytes(path)?;
2987 let (entries, views) = decode_catalog(&bytes, size)?;
2988 Ok(Self {
2989 file: Arc::new(file),
2990 size,
2991 entries: Arc::new(entries),
2992 views: Arc::new(views),
2993 opening,
2994 })
2995 }
2996
2997 pub fn names(&self) -> impl ExactSizeIterator<Item = &str> {
2999 self.entries.iter().map(|entry| entry.name.as_str())
3000 }
3001
3002 pub fn rows(&self) -> impl ExactSizeIterator<Item = (&str, usize)> {
3009 self.entries.iter().map(|entry| (entry.name.as_str(), entry.rows))
3010 }
3011
3012 pub fn views(&self) -> impl ExactSizeIterator<Item = &ViewEntry> {
3018 self.views.iter()
3019 }
3020
3021 #[must_use]
3023 pub fn len(&self) -> usize {
3024 self.entries.len()
3025 }
3026
3027 #[must_use]
3030 pub fn is_empty(&self) -> bool {
3031 self.entries.is_empty()
3032 }
3033
3034 pub fn table(&self, name: &str) -> Result<Reader> {
3040 let entry = self
3041 .entries
3042 .iter()
3043 .find(|entry| entry.name == name)
3044 .ok_or_else(|| invalid(&format!("the file holds no table called {name}")))?;
3045 let mut bytes = vec![0; entry.directory.length as usize];
3046 read_at(&self.file, entry.directory.offset, &mut bytes)?;
3047 if checksum(&bytes) != entry.directory.hash {
3048 return Err(invalid(&format!("the directory of table {name} does not checksum")));
3049 }
3050 let mut opening = self.opening;
3051 opening.reads += 1;
3052 opening.bytes += u64::from(entry.directory.length);
3053 Reader::build(
3054 Arc::clone(&self.file),
3055 self.size,
3056 decode_directory(&bytes, self.size)?,
3057 u64::from(entry.directory.length),
3058 opening,
3059 )
3060 }
3061}
3062
3063fn slot_offset(generation: u64) -> u64 {
3068 16 + (generation - 1) % 2 * SLOT_BYTES as u64
3069}
3070
3071fn slot_bytes(path: impl AsRef<Path>) -> Result<(File, u64, Slot, Vec<u8>, Opening)> {
3076 let mut file = File::open(path).map_err(io)?;
3077 let size = file.metadata().map_err(io)?.len();
3078 if size < HEADER {
3079 return Err(invalid("file is shorter than its header"));
3080 }
3081 let mut header = [0; HEADER as usize];
3082 file.read_exact(&mut header).map_err(io)?;
3083 let mut opening = Opening { reads: 1, bytes: HEADER };
3084 let version = u32::from_le_bytes([header[8], header[9], header[10], header[11]]);
3085 if &header[..8] != MAGIC {
3090 return Err(invalid("the header does not begin with a rudb native magic"));
3091 }
3092 if !READABLE.contains(&version) {
3093 return Err(invalid(&format!(
3094 "the file is format {version} and this build reads format {FORMAT}, so it has to \
3095 be written again"
3096 )));
3097 }
3098 let mut selected = None;
3099 for start in [16, 16 + SLOT_BYTES] {
3100 let slot = Slot::read(&header[start..start + SLOT_BYTES]);
3101 if slot.generation == 0 || slot.length == 0 || slot.length as usize > MAX_DIRECTORY {
3102 continue;
3103 }
3104 let Some(end) = slot.offset.checked_add(u64::from(slot.length)) else { continue };
3105 if slot.offset < HEADER || end > size {
3106 continue;
3107 }
3108 let mut bytes = vec![0; slot.length as usize];
3109 file.seek(SeekFrom::Start(slot.offset)).map_err(io)?;
3110 file.read_exact(&mut bytes).map_err(io)?;
3111 opening.reads += 1;
3112 opening.bytes += u64::from(slot.length);
3113 if checksum(&bytes) == slot.hash
3114 && selected
3115 .as_ref()
3116 .is_none_or(|(old, _): &(Slot, Vec<u8>)| old.generation < slot.generation)
3117 {
3118 selected = Some((slot, bytes));
3119 }
3120 }
3121 let (slot, bytes) = selected.ok_or_else(|| invalid("no committed directory slot is valid"))?;
3122 Ok((file, size, slot, bytes, opening))
3123}
3124
3125impl Reader {
3126 pub fn open(path: impl AsRef<Path>) -> Result<Self> {
3133 let catalog = Catalog::open(path)?;
3134 let mut names = catalog.names();
3135 let name = names.next().ok_or_else(|| invalid("the file holds no table"))?.to_string();
3136 if names.next().is_some() {
3137 return Err(invalid(
3138 "the file holds more than one table, so it has to be opened by name",
3139 ));
3140 }
3141 catalog.table(&name)
3142 }
3143
3144 fn build(
3146 file: Arc<File>,
3147 size: u64,
3148 table: Table,
3149 directory: u64,
3150 opening: Opening,
3151 ) -> Result<Self> {
3152 let places = places(&table)?;
3153 let dictionaries = (0..table.fields.len()).map(|_| OnceLock::new()).collect();
3154 let table_fields = table.fields.len();
3155 let stripes = table.stripes.len();
3156 let cache = (0..table.fields.len())
3157 .map(|_| {
3158 Mutex::new(Cached {
3159 pages: (0..stripes).map(|_| None).collect(),
3160 index: (0..stripes).map(|_| None).collect(),
3161 ..Cached::default()
3162 })
3163 })
3164 .collect::<Vec<_>>();
3165 let sieves: Vec<Vec<SieveSlot>> = (0..table.fields.len())
3166 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3167 .collect();
3168 let part_ranges: Vec<Vec<RangeSlot>> = (0..table.fields.len())
3169 .map(|_| table.stripes.iter().map(|_| OnceLock::new()).collect())
3170 .collect();
3171 Ok(Self {
3172 file,
3173 table: Arc::new(table),
3174 dictionaries: Arc::new(dictionaries),
3175 loading: Arc::new((0..table_fields).map(|_| Mutex::new(())).collect()),
3176 opened: Arc::new(AtomicUsize::new(0)),
3177 sieves: Arc::new(sieves),
3178 part_ranges: Arc::new(part_ranges),
3179 places: Arc::new(places),
3180 cache: Arc::new(cache),
3181 pages: Arc::new(AtomicUsize::new(0)),
3182 indexes: Arc::new(AtomicUsize::new(0)),
3183 kept: Arc::new(AtomicUsize::new(CACHED_STRIPES_PER_COLUMN)),
3184 size,
3185 directory,
3186 opening,
3187 })
3188 }
3189
3190 #[must_use]
3197 pub fn reads(&self) -> Reads {
3198 Reads {
3199 opening: self.opening,
3200 pages: self.pages.load(Atomic::Relaxed),
3201 indexes: self.indexes.load(Atomic::Relaxed),
3202 dictionaries: self.opened.load(Atomic::Relaxed),
3203 }
3204 }
3205
3206 #[must_use]
3211 pub fn layout(&self) -> Layout {
3212 let table = &self.table;
3213 let stripes = table.stripes.as_slice();
3214 let columns = table
3215 .fields
3216 .iter()
3217 .enumerate()
3218 .map(|(at, field)| ColumnLayout {
3219 name: field.name.clone(),
3220 kind: field.ty.to_string(),
3221 pages: sum(stripes.iter().map(|stripe| span_bytes(&stripe.pages, at))),
3222 memberships: sum(stripes.iter().map(|stripe| page_bytes(&stripe.memberships, at))),
3223 sieves: sum(stripes.iter().map(|stripe| page_bytes(&stripe.sieves, at))),
3224 part_ranges: sum(stripes.iter().map(|stripe| page_bytes(&stripe.part_ranges, at))),
3225 dictionary: page_bytes(&table.dictionaries, at),
3226 })
3227 .collect();
3228 Layout {
3229 file: self.size,
3230 rows: table.rows,
3231 stripes: stripes.len(),
3232 parts: self.places.len(),
3233 columns,
3234 indexes: sum(stripes.iter().map(|stripe| u64::from(stripe.index.length))),
3235 directory: self.directory,
3236 header: HEADER,
3237 }
3238 }
3239
3240 pub fn stored(&self, column: usize) -> Result<Vec<StoredPart>> {
3257 let field = self
3258 .table
3259 .fields
3260 .get(column)
3261 .ok_or_else(|| invalid("stored column index out of range"))?;
3262 let mut stored = Vec::with_capacity(self.places.len());
3263 let mut row = 0;
3264 for (at, stripe) in self.table.stripes.iter().enumerate() {
3265 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3266 let index = read_index(&self.file, stripe, column)?;
3267 let mut bytes = vec![0; page.length as usize];
3268 read_at(&self.file, page.offset, &mut bytes)?;
3269 let ranges = self.stripe_part_ranges(at, column);
3270 for (part, &rows) in stripe.parts.iter().enumerate() {
3271 let span = *index.get(part).ok_or_else(|| invalid("part index out of range"))?;
3272 let held = part_bytes(&bytes, span)?;
3273 let range = ranges.and_then(|held| held.get(part));
3274 stored.push(StoredPart {
3275 stripe: at,
3276 part,
3277 row,
3278 rows: rows as usize,
3279 encoding: page_encoding(&field.ty, rows as usize, held),
3280 bytes: span.length as u64,
3281 page: page.offset,
3282 offset: span.start as u64,
3283 low: range
3284 .and_then(|range| range.low.clone())
3285 .and_then(|bound| bound.into_value(&field.ty)),
3286 high: range
3287 .and_then(|range| range.high.clone())
3288 .and_then(|bound| bound.into_value(&field.ty)),
3289 nulls: range.map(|range| range.nulls),
3290 });
3291 row += rows as usize;
3292 }
3293 }
3294 Ok(stored)
3295 }
3296
3297 #[must_use]
3299 pub fn parts(&self) -> usize {
3300 self.places.len()
3301 }
3302
3303 #[must_use]
3310 pub fn stripe_parts(&self) -> Vec<std::ops::Range<usize>> {
3311 let mut runs = Vec::with_capacity(self.table.stripes.len());
3312 let mut start = 0;
3313 for stripe in &self.table.stripes {
3314 let end = start + stripe.parts.len();
3315 runs.push(start..end);
3316 start = end;
3317 }
3318 runs
3319 }
3320
3321 #[must_use]
3326 pub fn stripe_rows(&self, stripe: usize) -> usize {
3327 self.table.stripes.get(stripe).map_or(0, |held| held.rows)
3328 }
3329
3330 pub fn keep_stripes(&self, stripes: usize) {
3337 self.kept.fetch_max(stripes, Atomic::Relaxed);
3338 }
3339
3340 #[must_use]
3342 pub fn part_rows(&self, at: usize) -> usize {
3343 self.places.get(at).map_or(0, |place| place.rows as usize)
3344 }
3345
3346 #[must_use]
3348 pub fn table(&self) -> &Table {
3349 &self.table
3350 }
3351
3352 pub fn top_frequencies(&self, column: usize, top: usize) -> Result<Option<Vec<(Value, u64)>>> {
3361 let field = self
3362 .table
3363 .fields
3364 .get(column)
3365 .ok_or_else(|| invalid("frequency column index out of range"))?;
3366 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3367 return Ok(None);
3368 };
3369 if top == 0 || summary.entries.len() < top {
3370 return Ok(None);
3371 }
3372 let boundary = summary.entries[top - 1].count;
3373 if boundary <= summary.omitted_max {
3374 return Ok(None);
3375 }
3376 self.decode_frequencies(column, &field.ty, &summary.entries).map(Some)
3377 }
3378
3379 pub fn exact_frequencies(&self, column: usize) -> Result<Option<Vec<(Value, u64)>>> {
3399 let Some(prefix) = self.frequency_prefix(column)? else {
3400 return Ok(None);
3401 };
3402 Ok((prefix.omitted_max == 0).then_some(prefix.entries))
3403 }
3404
3405 pub fn frequency_prefix(&self, column: usize) -> Result<Option<FrequencyPrefix>> {
3428 let field = self
3429 .table
3430 .fields
3431 .get(column)
3432 .ok_or_else(|| invalid("frequency column index out of range"))?;
3433 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3434 return Ok(None);
3435 };
3436 let entries = self.decode_frequencies(column, &field.ty, &summary.entries)?;
3437 Ok(Some(FrequencyPrefix { entries, omitted_max: summary.omitted_max }))
3438 }
3439
3440 fn decode_frequencies(
3442 &self,
3443 column: usize,
3444 ty: &LogicalType,
3445 entries: &[FrequencyEntry],
3446 ) -> Result<Vec<(Value, u64)>> {
3447 let dictionary = if *ty == LogicalType::Varchar { self.dictionary(column)? } else { None };
3448 let mut out = Vec::with_capacity(entries.len());
3449 for entry in entries {
3450 let value = match entry.value {
3451 FrequencyValue::Null => Value::Null,
3452 FrequencyValue::Integer(value) => match *ty {
3453 LogicalType::TinyInt => Value::TinyInt(
3454 i8::try_from(value)
3455 .map_err(|_| invalid("frequency TINYINT is out of range"))?,
3456 ),
3457 LogicalType::UTinyInt => Value::UTinyInt(
3458 u8::try_from(value)
3459 .map_err(|_| invalid("frequency UTINYINT is out of range"))?,
3460 ),
3461 LogicalType::USmallInt => Value::USmallInt(
3462 u16::try_from(value)
3463 .map_err(|_| invalid("frequency USMALLINT is out of range"))?,
3464 ),
3465 LogicalType::UInteger => Value::UInteger(
3466 u32::try_from(value)
3467 .map_err(|_| invalid("frequency UINTEGER is out of range"))?,
3468 ),
3469 LogicalType::UBigInt => Value::UBigInt(
3470 u64::try_from(value)
3471 .map_err(|_| invalid("frequency UBIGINT is out of range"))?,
3472 ),
3473 LogicalType::SmallInt => Value::SmallInt(
3474 i16::try_from(value)
3475 .map_err(|_| invalid("frequency SMALLINT is out of range"))?,
3476 ),
3477 LogicalType::Integer => Value::Integer(
3478 i32::try_from(value)
3479 .map_err(|_| invalid("frequency INTEGER is out of range"))?,
3480 ),
3481 LogicalType::BigInt => Value::BigInt(
3482 i64::try_from(value)
3483 .map_err(|_| invalid("frequency BIGINT is out of range"))?,
3484 ),
3485 LogicalType::Date => Value::Date(
3486 i32::try_from(value)
3487 .map_err(|_| invalid("frequency DATE is out of range"))?,
3488 ),
3489 LogicalType::Timestamp => Value::Timestamp(
3490 i64::try_from(value)
3491 .map_err(|_| invalid("frequency TIMESTAMP is out of range"))?,
3492 ),
3493 _ => return Err(invalid("integer frequency belongs to another type")),
3494 },
3495 FrequencyValue::Code(code) => dictionary
3496 .as_ref()
3497 .ok_or_else(|| invalid("frequency code has no dictionary"))?
3498 .try_value_at(code as usize)?,
3499 };
3500 out.push((value, entry.count));
3501 }
3502 Ok(out)
3503 }
3504
3505 pub fn frequency_occurrences(&self, column: usize) -> Result<Option<FrequencyOccurrences>> {
3515 self.table
3516 .fields
3517 .get(column)
3518 .ok_or_else(|| invalid("frequency column index out of range"))?;
3519 let Some(summary) = self.table.frequencies.get(column).and_then(Option::as_ref) else {
3520 return Ok(None);
3521 };
3522 if summary.ordinals.is_empty() {
3523 return Ok(None);
3524 }
3525 Ok(Some(FrequencyOccurrences {
3526 omitted_max: summary.omitted_max,
3527 ordinals: summary.ordinals.clone(),
3528 }))
3529 }
3530
3531 pub fn distinct_values(&self, column: usize) -> Result<Option<u64>> {
3555 self.table
3556 .distincts
3557 .get(column)
3558 .copied()
3559 .ok_or_else(|| invalid("distinct column index out of range"))
3560 }
3561
3562 pub fn null_count(&self, column: usize) -> Result<u64> {
3573 if column >= self.table.fields.len() {
3574 return Err(invalid("null count column index out of range"));
3575 }
3576 let mut nulls = 0_u64;
3577 for stripe in &self.table.stripes {
3578 let range = stripe
3579 .zone
3580 .column(column)
3581 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3582 nulls = nulls
3583 .checked_add(range.nulls as u64)
3584 .ok_or_else(|| invalid("null count overflow"))?;
3585 }
3586 Ok(nulls)
3587 }
3588
3589 pub fn text_extremes(&self, column: usize) -> Result<Option<(Value, Value)>> {
3604 if self.null_count(column)? > 0 {
3605 return Ok(None);
3606 }
3607 let Some(dictionary) = self.dictionary(column)? else { return Ok(None) };
3608 let Some(ranks) = dictionary.ranks() else { return Ok(None) };
3609 if ranks == 0 {
3610 return Ok(None);
3611 }
3612 let low = text_at_rank(&dictionary, 0)?;
3613 let high = text_at_rank(&dictionary, ranks - 1)?;
3614 Ok(Some((low, high)))
3615 }
3616
3617 pub fn exact_extremes(&self, column: usize) -> Result<Option<(Bound, Bound)>> {
3640 if column >= self.table.fields.len() {
3641 return Err(invalid("extremes column index out of range"));
3642 }
3643 let mut low: Option<Bound> = None;
3644 let mut high: Option<Bound> = None;
3645 for stripe in &self.table.stripes {
3646 let range = stripe
3647 .zone
3648 .column(column)
3649 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3650 if !range.exact {
3651 return Ok(None);
3652 }
3653 let (Some(small), Some(large)) = (range.low.as_ref(), range.high.as_ref()) else {
3658 if stripe.rows > range.nulls {
3659 return Ok(None);
3660 }
3661 continue;
3662 };
3663 low = Some(low.map_or_else(|| small.clone(), |held| held.smaller(small.clone())));
3664 high = Some(high.map_or_else(|| large.clone(), |held| held.larger(large.clone())));
3665 }
3666 Ok(low.zip(high))
3667 }
3668
3669 pub fn exact_sum(&self, column: usize) -> Result<Option<(i128, u64)>> {
3682 if column >= self.table.fields.len() {
3683 return Err(invalid("sum column index out of range"));
3684 }
3685 let mut total = 0_i128;
3686 let mut rows = 0_u64;
3687 for stripe in &self.table.stripes {
3688 let range = stripe
3689 .zone
3690 .column(column)
3691 .ok_or_else(|| invalid("stripe zone is narrower than the schema"))?;
3692 let Some(part) = range.sum else { return Ok(None) };
3693 let Some(sum) = total.checked_add(part) else { return Ok(None) };
3694 total = sum;
3695 rows = rows.saturating_add(stripe.rows as u64 - range.nulls as u64);
3696 }
3697 Ok(Some((total, rows)))
3698 }
3699
3700 fn dictionary(&self, column: usize) -> Result<Option<Arc<Vector>>> {
3709 let Some(page) = self.table.dictionaries[column] else { return Ok(None) };
3710 if let Some(dictionary) = self.dictionaries[column].get() {
3711 return Ok(Some(Arc::clone(dictionary)));
3712 }
3713 let _queued = self.loading[column].lock().map_err(|_| invalid("a poisoned dictionary"))?;
3714 if let Some(dictionary) = self.dictionaries[column].get() {
3715 return Ok(Some(Arc::clone(dictionary)));
3716 }
3717 self.opened.fetch_add(1, Atomic::Relaxed);
3718 let dictionary = Arc::new(open_global_dictionary(
3719 Arc::clone(&self.file),
3720 page,
3721 &self.table.fields[column].ty,
3722 TEXT_KEEP_BUDGET,
3723 )?);
3724 let _ = self.dictionaries[column].set(Arc::clone(&dictionary));
3725 Ok(Some(dictionary))
3726 }
3727
3728 pub fn extents(&self, of: &Section) -> Result<Vec<section::Extent>> {
3735 if of.extent_bytes == 0 {
3736 return Ok(Vec::new());
3737 }
3738 let mut bytes = vec![0; of.extent_bytes as usize];
3739 read_at(&self.file, of.extent_page, &mut bytes)?;
3740 if checksum(&bytes) != of.hash {
3741 return Err(invalid("a section's extent table does not checksum"));
3742 }
3743 let extents = section::decode_extents(&bytes)?;
3744 if extents.len() != of.extents as usize {
3745 return Err(invalid("a section's extent table is not the length the entry says"));
3746 }
3747 Ok(extents)
3748 }
3749
3750 pub fn extent(&self, of: §ion::Extent) -> Result<Vec<u8>> {
3760 let end = of
3761 .offset
3762 .checked_add(u64::from(of.length))
3763 .ok_or_else(|| invalid("an extent overflows the file"))?;
3764 if of.offset < HEADER || end > self.size {
3765 return Err(invalid("an extent is outside the file"));
3766 }
3767 let mut bytes = vec![0; of.length as usize];
3768 read_at(&self.file, of.offset, &mut bytes)?;
3769 if checksum(&bytes) != of.hash {
3770 return Err(invalid("an extent does not checksum"));
3771 }
3772 Ok(bytes)
3773 }
3774
3775 pub fn payload(&self, of: &Section) -> Result<Vec<u8>> {
3784 let extents = self.extents(of)?;
3785 let mut bytes =
3786 Vec::with_capacity(sum(extents.iter().map(|one| u64::from(one.length))) as usize);
3787 for one in &extents {
3788 if one.first != bytes.len() as u64 {
3789 return Err(invalid("a section's extents do not join up"));
3790 }
3791 bytes.extend_from_slice(&self.extent(one)?);
3792 }
3793 if of.header_bytes as usize > bytes.len() {
3794 return Err(invalid("a section's header is longer than its payload"));
3795 }
3796 Ok(bytes)
3797 }
3798
3799 pub fn read(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3808 self.read_impl(part, columns, true)
3809 }
3810
3811 pub fn read_sparse(&self, part: usize, columns: &[usize]) -> Result<Chunk> {
3821 self.read_impl(part, columns, false)
3822 }
3823
3824 pub fn skips_codes(&self, part: usize, column: usize, candidates: &[u32]) -> Result<bool> {
3831 if candidates.is_empty() {
3832 return Ok(true);
3833 }
3834 if candidates.windows(2).any(|pair| pair[0] >= pair[1]) {
3835 return Err(Error::internal("native code candidates are not sorted and unique"));
3836 }
3837 let stripe = self.stripe_of(part)?;
3838 let Some(page) = stripe.memberships.get(column).copied().flatten() else {
3839 return Ok(false);
3840 };
3841 let mut bytes = vec![0; page.length as usize];
3842 read_at(&self.file, page.offset, &mut bytes)?;
3843 if checksum(&bytes) != page.hash {
3844 return Err(invalid("membership page checksum differs"));
3845 }
3846 let codes = decode_membership(&bytes)?;
3847 let mut left = 0;
3848 let mut right = 0;
3849 while left < codes.len() && right < candidates.len() {
3850 match codes[left].cmp(&candidates[right]) {
3851 Ordering::Less => left += 1,
3852 Ordering::Greater => right += 1,
3853 Ordering::Equal => return Ok(false),
3854 }
3855 }
3856 Ok(true)
3857 }
3858
3859 fn stripe_of(&self, part: usize) -> Result<&Stripe> {
3860 let place = self.places.get(part).ok_or_else(|| invalid("part index out of range"))?;
3861 self.table
3862 .stripes
3863 .get(place.stripe as usize)
3864 .ok_or_else(|| invalid("stripe index out of range"))
3865 }
3866
3867 fn held(&self, at: usize, stripe: &Stripe, column: usize, whole: bool) -> Result<CachedColumn> {
3884 let cache = self.cache.get(column).ok_or_else(|| invalid("column index out of range"))?;
3885 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3886 let known = cached.index.get(at).and_then(Clone::clone);
3887 let page = cached.pages.get(at).and_then(Clone::clone);
3888 if let Some(index) = known.clone() {
3889 if !whole || page.is_some() {
3890 return Ok(CachedColumn { stripe: at, index, page });
3891 }
3892 }
3893 if cached.loading.contains(&at) {
3894 drop(cached);
3895 if let Some(index) = known {
3899 return Ok(CachedColumn { stripe: at, index, page: None });
3900 }
3901 let held = self.page_of(stripe, column, at, false, None)?;
3902 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3903 remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3904 return Ok(held);
3905 }
3906 cached.loading.push(at);
3907 drop(cached);
3908
3909 let read = self.page_of(stripe, column, at, whole, known);
3910
3911 let mut cached = cache.lock().map_err(|_| invalid("column page cache is poisoned"))?;
3915 if let Some(position) = cached.loading.iter().position(|loading| *loading == at) {
3916 cached.loading.remove(position);
3917 }
3918 let held = read?;
3919 remember(&mut cached, &held, self.kept.load(Atomic::Relaxed));
3920 Ok(held)
3921 }
3922
3923 fn page_of(
3929 &self,
3930 stripe: &Stripe,
3931 column: usize,
3932 at: usize,
3933 whole: bool,
3934 known: Option<Arc<Vec<PartSpan>>>,
3935 ) -> Result<CachedColumn> {
3936 let index = match known {
3937 Some(index) => index,
3938 None => {
3939 self.indexes.fetch_add(1, Atomic::Relaxed);
3940 Arc::new(read_index(&self.file, stripe, column)?)
3941 }
3942 };
3943 let page = if whole {
3944 self.pages.fetch_add(1, Atomic::Relaxed);
3945 let span = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3946 let mut bytes = vec![0; span.length as usize];
3947 read_at(&self.file, span.offset, &mut bytes)?;
3948 Some(Arc::new(bytes))
3949 } else {
3950 None
3951 };
3952 Ok(CachedColumn { stripe: at, index, page })
3953 }
3954
3955 fn read_impl(&self, at: usize, columns: &[usize], whole: bool) -> Result<Chunk> {
3956 let place = *self.places.get(at).ok_or_else(|| invalid("part index out of range"))?;
3957 let index = place.stripe as usize;
3958 let stripe =
3959 self.table.stripes.get(index).ok_or_else(|| invalid("stripe index out of range"))?;
3960 let rows = place.rows as usize;
3961 let mut picked = Vec::with_capacity(columns.len());
3962 for &column in columns {
3963 let field = self
3964 .table
3965 .fields
3966 .get(column)
3967 .ok_or_else(|| invalid("column index out of range"))?;
3968 let page = stripe.pages.get(column).ok_or_else(|| invalid("stripe page is missing"))?;
3969 let held = self.held(index, stripe, column, whole)?;
3970 let span = *held
3971 .index
3972 .get(place.part as usize)
3973 .ok_or_else(|| invalid("part index out of range"))?;
3974 let owned;
3975 let bytes = match &held.page {
3976 Some(held) => part_bytes(held, span)?,
3977 None => {
3978 let offset = page
3979 .offset
3980 .checked_add(span.start as u64)
3981 .ok_or_else(|| invalid("part range overflow"))?;
3982 let mut bytes = vec![0; span.length];
3983 read_at(&self.file, offset, &mut bytes)?;
3984 owned = bytes;
3985 &owned
3986 }
3987 };
3988 if checksum(bytes) != span.hash {
3989 return Err(invalid(&format!(
3990 "column page checksum differs, column {column} part {} at {}+{} of {} bytes, \
3991 wanted {:016x} and got {:016x}",
3992 place.part,
3993 page.offset,
3994 span.start,
3995 span.length,
3996 span.hash,
3997 checksum(bytes),
3998 )));
3999 }
4000 let dictionary = self.dictionary(column)?;
4001 picked.push(decode(&field.ty, rows, bytes, dictionary)?.into_pages());
4007 }
4008 Chunk::with_rows(picked, rows)
4009 }
4010
4011 #[must_use]
4027 pub fn skips(&self, part: usize, probes: &[Probe]) -> bool {
4028 let Some(place) = self.places.get(part).copied() else { return false };
4029 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4030 if stripe.zone.skips(probes) {
4031 return true;
4032 }
4033 probes.iter().any(|probe| self.outside(place, probe) || self.sifted(place, probe))
4034 }
4035
4036 fn outside(&self, place: Place, probe: &Probe) -> bool {
4042 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4043 Some(ranges) => ranges
4044 .get(place.part as usize)
4045 .is_some_and(|range| range.excludes(probe.op, &probe.value)),
4046 None => false,
4047 }
4048 }
4049
4050 fn stripe_part_ranges(&self, stripe: usize, column: usize) -> Option<&[Range]> {
4056 let slot = self.part_ranges.get(column)?.get(stripe)?;
4057 if let Some(held) = slot.get() {
4058 return Some(held);
4059 }
4060 let page = self.table.stripes.get(stripe)?.part_ranges.get(column).copied().flatten()?;
4061 let mut bytes = vec![0; page.length as usize];
4062 read_at(&self.file, page.offset, &mut bytes).ok()?;
4063 if checksum(&bytes) != page.hash {
4064 return None;
4065 }
4066 let ranges = Arc::new(decode_part_ranges(&bytes).ok()?);
4067 let _ = slot.set(ranges);
4068 slot.get().map(|held| held.as_slice())
4069 }
4070
4071 #[must_use]
4088 pub fn certain(&self, part: usize, probes: &[Probe]) -> bool {
4089 let Some(place) = self.places.get(part).copied() else { return false };
4090 let Some(stripe) = self.table.stripes.get(place.stripe as usize) else { return false };
4091 if stripe.zone.certain(probes) {
4092 return true;
4093 }
4094 probes
4095 .iter()
4096 .all(|probe| stripe.zone.certain(slice::from_ref(probe)) || self.inside(place, probe))
4097 }
4098
4099 fn inside(&self, place: Place, probe: &Probe) -> bool {
4105 match self.stripe_part_ranges(place.stripe as usize, probe.column) {
4106 Some(ranges) => ranges
4107 .get(place.part as usize)
4108 .is_some_and(|range| range.certain(probe.op, &probe.value)),
4109 None => false,
4110 }
4111 }
4112
4113 #[must_use]
4124 pub fn stripe_skips(&self, stripe: usize, probes: &[Probe]) -> bool {
4125 self.table.stripes.get(stripe).is_some_and(|held| held.zone.skips(probes))
4126 }
4127
4128 fn sifted(&self, place: Place, probe: &Probe) -> bool {
4134 if probe.op != Op::Equal {
4135 return false;
4136 }
4137 match self.stripe_sieves(place.stripe as usize, probe.column) {
4138 Some(sieves) => sieves
4139 .get(place.part as usize)
4140 .and_then(Option::as_ref)
4141 .is_some_and(|sieve| sieve.excludes(&probe.value)),
4142 None => false,
4143 }
4144 }
4145
4146 fn stripe_sieves(&self, stripe: usize, column: usize) -> Option<&[Option<Sieve>]> {
4153 let slot = self.sieves.get(column)?.get(stripe)?;
4154 if let Some(held) = slot.get() {
4155 return Some(held);
4156 }
4157 let page = self.table.stripes.get(stripe)?.sieves.get(column).copied().flatten()?;
4158 let mut bytes = vec![0; page.length as usize];
4159 read_at(&self.file, page.offset, &mut bytes).ok()?;
4160 if checksum(&bytes) != page.hash {
4161 return None;
4162 }
4163 let sieves = Arc::new(decode_sieves(&bytes).ok()?);
4164 let _ = slot.set(sieves);
4165 slot.get().map(|held| held.as_slice())
4166 }
4167}
4168
4169fn text_at_rank(dictionary: &Vector, rank: usize) -> Result<Value> {
4171 let code = dictionary.code_at_rank(rank)? as usize;
4172 let text = dictionary
4173 .try_text_at(code)?
4174 .ok_or_else(|| invalid("global dictionary order names a code it does not have"))?;
4175 Ok(Value::Varchar(text.into()))
4176}
4177
4178#[cfg(unix)]
4183fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4184 use std::os::unix::fs::FileExt;
4185 while !bytes.is_empty() {
4186 let written = file.write_at(bytes, offset).map_err(io)?;
4187 if written == 0 {
4188 return Err(invalid("a write to the native file wrote nothing"));
4189 }
4190 offset += written as u64;
4191 bytes = &bytes[written..];
4192 }
4193 Ok(())
4194}
4195
4196#[cfg(windows)]
4198fn write_at(file: &File, mut offset: u64, mut bytes: &[u8]) -> Result<()> {
4199 use std::os::windows::fs::FileExt;
4200 while !bytes.is_empty() {
4201 let written = file.seek_write(bytes, offset).map_err(io)?;
4202 if written == 0 {
4203 return Err(invalid("a write to the native file wrote nothing"));
4204 }
4205 offset += written as u64;
4206 bytes = &bytes[written..];
4207 }
4208 Ok(())
4209}
4210
4211#[cfg(not(any(unix, windows)))]
4213fn write_at(file: &File, offset: u64, bytes: &[u8]) -> Result<()> {
4214 use std::io::Write;
4215 let mut file = file.try_clone().map_err(io)?;
4216 file.seek(SeekFrom::Start(offset)).map_err(io)?;
4217 file.write_all(bytes).map_err(io)
4218}
4219
4220#[cfg(unix)]
4230fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4231 use std::os::unix::fs::FileExt;
4232 while !bytes.is_empty() {
4233 let read = file.read_at(bytes, offset).map_err(io)?;
4234 if read == 0 {
4235 return Err(invalid("column page ends before its declared length"));
4236 }
4237 offset += read as u64;
4238 bytes = &mut bytes[read..];
4239 }
4240 Ok(())
4241}
4242
4243#[cfg(windows)]
4249fn read_at(file: &File, mut offset: u64, mut bytes: &mut [u8]) -> Result<()> {
4250 use std::os::windows::fs::FileExt;
4251 while !bytes.is_empty() {
4252 let read = file.seek_read(bytes, offset).map_err(io)?;
4253 if read == 0 {
4254 return Err(invalid("column page ends before its declared length"));
4255 }
4256 offset += read as u64;
4257 bytes = &mut bytes[read..];
4258 }
4259 Ok(())
4260}
4261
4262#[cfg(not(any(unix, windows)))]
4267fn read_at(file: &File, offset: u64, bytes: &mut [u8]) -> Result<()> {
4268 let mut file = file.try_clone().map_err(io)?;
4269 file.seek(SeekFrom::Start(offset)).map_err(io)?;
4270 file.read_exact(bytes).map_err(io)
4271}
4272
4273fn type_tag(ty: &LogicalType) -> Result<u8> {
4280 match ty {
4281 LogicalType::SmallInt => Ok(1),
4282 LogicalType::Integer => Ok(2),
4283 LogicalType::BigInt => Ok(3),
4284 LogicalType::Varchar => Ok(4),
4285 LogicalType::Date => Ok(5),
4286 LogicalType::Timestamp => Ok(6),
4287 LogicalType::Boolean => Ok(7),
4288 LogicalType::TinyInt => Ok(8),
4289 LogicalType::UTinyInt => Ok(9),
4290 LogicalType::USmallInt => Ok(10),
4291 LogicalType::UInteger => Ok(11),
4292 LogicalType::UBigInt => Ok(12),
4293 LogicalType::Decimal { .. } => Ok(13),
4294 LogicalType::Float => Ok(14),
4295 LogicalType::Double => Ok(15),
4296 LogicalType::HugeInt => Ok(16),
4297 LogicalType::UHugeInt => Ok(17),
4298 LogicalType::Time => Ok(18),
4299 LogicalType::TimeTz => Ok(19),
4300 LogicalType::TimestampTz => Ok(20),
4301 LogicalType::Interval => Ok(21),
4302 LogicalType::Uuid => Ok(22),
4303 LogicalType::Blob => Ok(23),
4304 LogicalType::Bit => Ok(24),
4305 LogicalType::TimestampS => Ok(25),
4306 LogicalType::TimestampMs => Ok(26),
4307 LogicalType::TimestampNs => Ok(27),
4308 _ => Err(Error::not_implemented(format!("native storage for {ty}"))),
4309 }
4310}
4311
4312fn put_type(out: &mut Vec<u8>, ty: &LogicalType) -> Result<()> {
4318 out.push(type_tag(ty)?);
4319 if let LogicalType::Decimal { width, scale } = ty {
4320 out.push(*width);
4321 out.push(*scale);
4322 }
4323 Ok(())
4324}
4325
4326fn read_type(cur: &mut Cursor<'_>) -> Result<LogicalType> {
4328 let tag = cur.u8()?;
4329 if tag == 13 {
4330 let width = cur.u8()?;
4331 let scale = cur.u8()?;
4332 return LogicalType::decimal(width, scale)
4333 .map_err(|_| invalid("decimal column width and scale are not a decimal"));
4334 }
4335 tag_type(tag)
4336}
4337
4338fn tag_type(tag: u8) -> Result<LogicalType> {
4339 match tag {
4340 1 => Ok(LogicalType::SmallInt),
4341 2 => Ok(LogicalType::Integer),
4342 3 => Ok(LogicalType::BigInt),
4343 4 => Ok(LogicalType::Varchar),
4344 5 => Ok(LogicalType::Date),
4345 6 => Ok(LogicalType::Timestamp),
4346 7 => Ok(LogicalType::Boolean),
4347 8 => Ok(LogicalType::TinyInt),
4348 9 => Ok(LogicalType::UTinyInt),
4349 10 => Ok(LogicalType::USmallInt),
4350 11 => Ok(LogicalType::UInteger),
4351 12 => Ok(LogicalType::UBigInt),
4352 14 => Ok(LogicalType::Float),
4353 15 => Ok(LogicalType::Double),
4354 16 => Ok(LogicalType::HugeInt),
4355 17 => Ok(LogicalType::UHugeInt),
4356 18 => Ok(LogicalType::Time),
4357 19 => Ok(LogicalType::TimeTz),
4358 20 => Ok(LogicalType::TimestampTz),
4359 21 => Ok(LogicalType::Interval),
4360 22 => Ok(LogicalType::Uuid),
4361 23 => Ok(LogicalType::Blob),
4362 24 => Ok(LogicalType::Bit),
4363 25 => Ok(LogicalType::TimestampS),
4364 26 => Ok(LogicalType::TimestampMs),
4365 27 => Ok(LogicalType::TimestampNs),
4366 _ => Err(invalid("column type tag is unknown")),
4367 }
4368}
4369
4370fn put_u16(out: &mut Vec<u8>, value: u16) {
4371 out.extend_from_slice(&value.to_le_bytes());
4372}
4373fn put_u32(out: &mut Vec<u8>, value: u32) {
4374 out.extend_from_slice(&value.to_le_bytes());
4375}
4376fn put_u64(out: &mut Vec<u8>, value: u64) {
4377 out.extend_from_slice(&value.to_le_bytes());
4378}
4379fn put_var_u64(out: &mut Vec<u8>, mut value: u64) {
4380 while value >= 0x80 {
4381 out.push((value as u8 & 0x7f) | 0x80);
4382 value >>= 7;
4383 }
4384 out.push(value as u8);
4385}
4386
4387fn frequency_order(left: FrequencyValue, right: FrequencyValue) -> Ordering {
4388 match (left, right) {
4389 (FrequencyValue::Null, FrequencyValue::Null) => Ordering::Equal,
4390 (FrequencyValue::Null, _) => Ordering::Less,
4391 (_, FrequencyValue::Null) => Ordering::Greater,
4392 (FrequencyValue::Integer(left), FrequencyValue::Integer(right)) => left.cmp(&right),
4393 (FrequencyValue::Code(left), FrequencyValue::Code(right)) => left.cmp(&right),
4394 (FrequencyValue::Integer(_), FrequencyValue::Code(_)) => Ordering::Less,
4395 (FrequencyValue::Code(_), FrequencyValue::Integer(_)) => Ordering::Greater,
4396 }
4397}
4398
4399fn code_frequency(dictionary: &GlobalDictionary) -> FrequencySummary {
4400 let mut entries = dictionary
4401 .counts
4402 .iter()
4403 .enumerate()
4404 .filter(|(_, count)| **count != 0)
4405 .map(|(code, &count)| FrequencyEntry { value: FrequencyValue::Code(code as u32), count })
4406 .collect::<Vec<_>>();
4407 if dictionary.nulls != 0 {
4408 entries.push(FrequencyEntry { value: FrequencyValue::Null, count: dictionary.nulls });
4409 }
4410 entries.sort_unstable_by(|left, right| {
4411 right.count.cmp(&left.count).then_with(|| frequency_order(left.value, right.value))
4412 });
4413 let omitted_max = entries.get(FREQUENCY_ENTRIES).map_or(0, |entry| entry.count);
4414 entries.truncate(FREQUENCY_ENTRIES);
4415 FrequencySummary { entries, omitted_max, ordinals: Vec::new() }
4416}
4417
4418fn encode_directory(table: &Table) -> Result<Vec<u8>> {
4419 let mut out = DIRECTORY.to_vec();
4420 let name = table.name.as_bytes();
4421 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4422 out.extend_from_slice(name);
4423 put_u16(&mut out, u16::try_from(table.fields.len()).map_err(|_| invalid("too many columns"))?);
4424 for field in &table.fields {
4425 let name = field.name.as_bytes();
4426 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?);
4427 out.extend_from_slice(name);
4428 put_type(&mut out, &field.ty)?;
4429 out.push(u8::from(field.not_null));
4430 }
4431 for dictionary in &table.dictionaries {
4432 match dictionary {
4433 None => out.push(0),
4434 Some(page) => {
4435 out.push(1);
4436 put_u64(&mut out, page.offset);
4437 put_u32(&mut out, page.length);
4438 put_u64(&mut out, page.hash);
4439 }
4440 }
4441 }
4442 for distinct in &table.distincts {
4443 match distinct {
4444 None => out.push(0),
4445 Some(count) => {
4446 out.push(1);
4447 put_u64(&mut out, *count);
4448 }
4449 }
4450 }
4451 put_u64(&mut out, u64::try_from(table.rows).map_err(|_| invalid("row count overflow"))?);
4452 put_u32(&mut out, u32::try_from(table.stripes.len()).map_err(|_| invalid("too many stripes"))?);
4453 for stripe in &table.stripes {
4454 put_u32(
4455 &mut out,
4456 u32::try_from(stripe.parts.len()).map_err(|_| invalid("too many parts in a stripe"))?,
4457 );
4458 for &rows in &stripe.parts {
4459 put_u32(&mut out, rows);
4460 }
4461 put_u64(&mut out, stripe.index.offset);
4462 put_u32(&mut out, stripe.index.length);
4463 for page in &stripe.pages {
4464 put_u64(&mut out, page.offset);
4465 put_u32(&mut out, page.length);
4466 }
4467 for ((field, dictionary), membership) in
4472 table.fields.iter().zip(&table.dictionaries).zip(&stripe.memberships)
4473 {
4474 if field.ty != LogicalType::Varchar || dictionary.is_none() {
4475 continue;
4476 }
4477 let page =
4478 membership.ok_or_else(|| invalid("string page has no code membership index"))?;
4479 put_u64(&mut out, page.offset);
4480 put_u32(&mut out, page.length);
4481 put_u64(&mut out, page.hash);
4482 }
4483 for sieve in &stripe.sieves {
4484 match sieve {
4485 None => out.push(0),
4486 Some(page) => {
4487 out.push(1);
4488 put_u64(&mut out, page.offset);
4489 put_u32(&mut out, page.length);
4490 put_u64(&mut out, page.hash);
4491 }
4492 }
4493 }
4494 for held in &stripe.part_ranges {
4495 match held {
4496 None => out.push(0),
4497 Some(page) => {
4498 out.push(1);
4499 put_u64(&mut out, page.offset);
4500 put_u32(&mut out, page.length);
4501 put_u64(&mut out, page.hash);
4502 }
4503 }
4504 }
4505 for range in stripe.zone.columns() {
4506 put_bound(&mut out, range.low.as_ref())?;
4507 put_bound(&mut out, range.high.as_ref())?;
4508 put_u32(
4509 &mut out,
4510 u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?,
4511 );
4512 out.push(u8::from(range.exact));
4513 match range.sum {
4514 None => out.push(0),
4515 Some(total) => {
4516 out.push(1);
4517 out.extend_from_slice(&total.to_le_bytes());
4518 }
4519 }
4520 }
4521 }
4522 out.extend_from_slice(FREQUENCIES);
4523 put_u16(
4524 &mut out,
4525 u16::try_from(table.frequencies.len())
4526 .map_err(|_| invalid("too many frequency columns"))?,
4527 );
4528 for summary in &table.frequencies {
4529 let Some(summary) = summary else {
4530 out.push(0);
4531 continue;
4532 };
4533 out.push(1);
4534 put_u64(&mut out, summary.omitted_max);
4535 put_u32(
4536 &mut out,
4537 u32::try_from(summary.entries.len())
4538 .map_err(|_| invalid("too many frequency entries"))?,
4539 );
4540 for entry in &summary.entries {
4541 match entry.value {
4542 FrequencyValue::Null => out.push(0),
4543 FrequencyValue::Integer(value) => {
4544 out.push(1);
4545 out.extend_from_slice(&value.to_le_bytes());
4546 }
4547 FrequencyValue::Code(value) => {
4548 out.push(2);
4549 put_u32(&mut out, value);
4550 }
4551 }
4552 put_u64(&mut out, entry.count);
4553 }
4554 put_u32(
4555 &mut out,
4556 u32::try_from(summary.ordinals.len())
4557 .map_err(|_| invalid("too many frequency ordinals"))?,
4558 );
4559 let mut previous = 0_u64;
4560 for (at, &ordinal) in summary.ordinals.iter().enumerate() {
4561 let delta = if at == 0 {
4562 ordinal
4563 } else {
4564 ordinal
4565 .checked_sub(previous)
4566 .ok_or_else(|| invalid("frequency ordinals are not ordered"))?
4567 };
4568 if at != 0 && delta == 0 {
4569 return Err(invalid("frequency ordinals are not unique"));
4570 }
4571 put_var_u64(&mut out, delta);
4572 previous = ordinal;
4573 }
4574 }
4575 if let Some(clustering) = &table.clustering {
4578 out.extend_from_slice(CLUSTERING);
4579 out.push(clustering.width().tag());
4580 put_u16(
4581 &mut out,
4582 u16::try_from(clustering.columns().len())
4583 .map_err(|_| invalid("too many clustering columns"))?,
4584 );
4585 for &column in clustering.columns() {
4586 put_u16(
4587 &mut out,
4588 u16::try_from(column).map_err(|_| invalid("clustering column index overflow"))?,
4589 );
4590 }
4591 }
4592 out.extend_from_slice(SECTIONS);
4598 put_u64(&mut out, table.generation);
4599 put_u16(
4600 &mut out,
4601 u16::try_from(table.sections.len()).map_err(|_| invalid("too many sections"))?,
4602 );
4603 for held in &table.sections {
4604 held.encode(&mut out)?;
4605 }
4606 Ok(out)
4607}
4608
4609fn encode_catalog(entries: &[Entry], views: &[ViewEntry]) -> Result<Vec<u8>> {
4618 let mut out = CATALOG.to_vec();
4619 put_u32(&mut out, u32::try_from(entries.len()).map_err(|_| invalid("too many tables"))?);
4620 for entry in entries {
4621 let name = entry.name.as_bytes();
4622 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("table name too long"))?);
4623 out.extend_from_slice(name);
4624 put_u64(&mut out, u64::try_from(entry.rows).map_err(|_| invalid("row count overflow"))?);
4625 put_u16(
4626 &mut out,
4627 u16::try_from(entry.fields.len()).map_err(|_| invalid("too many columns"))?,
4628 );
4629 for field in &entry.fields {
4630 let name = field.name.as_bytes();
4631 put_u16(
4632 &mut out,
4633 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4634 );
4635 out.extend_from_slice(name);
4636 put_type(&mut out, &field.ty)?;
4637 out.push(u8::from(field.not_null));
4638 }
4639 put_u64(&mut out, entry.directory.offset);
4640 put_u32(&mut out, entry.directory.length);
4641 put_u64(&mut out, entry.directory.hash);
4642 }
4643 put_u32(&mut out, u32::try_from(views.len()).map_err(|_| invalid("too many views"))?);
4644 for view in views {
4645 let name = view.name.as_bytes();
4646 put_u16(&mut out, u16::try_from(name.len()).map_err(|_| invalid("view name too long"))?);
4647 out.extend_from_slice(name);
4648 put_long_text(&mut out, &view.sql, "view body")?;
4649 put_long_text(&mut out, &view.statement, "view statement")?;
4650 put_u16(
4651 &mut out,
4652 u16::try_from(view.aliases.len()).map_err(|_| invalid("too many aliases"))?,
4653 );
4654 for alias in &view.aliases {
4655 let alias = alias.as_bytes();
4656 put_u16(
4657 &mut out,
4658 u16::try_from(alias.len()).map_err(|_| invalid("alias name too long"))?,
4659 );
4660 out.extend_from_slice(alias);
4661 }
4662 put_u16(
4663 &mut out,
4664 u16::try_from(view.columns.len()).map_err(|_| invalid("too many columns"))?,
4665 );
4666 for field in &view.columns {
4667 let name = field.name.as_bytes();
4668 put_u16(
4669 &mut out,
4670 u16::try_from(name.len()).map_err(|_| invalid("column name too long"))?,
4671 );
4672 out.extend_from_slice(name);
4673 put_type(&mut out, &field.ty)?;
4674 out.push(u8::from(field.not_null));
4675 }
4676 }
4677 Ok(out)
4678}
4679
4680fn put_long_text(out: &mut Vec<u8>, text: &str, what: &str) -> Result<()> {
4682 let bytes = text.as_bytes();
4683 put_u32(out, u32::try_from(bytes.len()).map_err(|_| invalid(&format!("{what} too long")))?);
4684 out.extend_from_slice(bytes);
4685 Ok(())
4686}
4687
4688fn decode_catalog(bytes: &[u8], size: u64) -> Result<(Vec<Entry>, Vec<ViewEntry>)> {
4691 let mut cur = Cursor { bytes, at: 0 };
4692 if cur.take(8)? != CATALOG {
4693 return Err(invalid("catalog magic differs"));
4694 }
4695 let count = cur.u32()? as usize;
4696 let mut entries: Vec<Entry> = Vec::with_capacity(count.min(1024));
4697 for _ in 0..count {
4698 let name = cur.text()?;
4699 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4700 let width = cur.u16()? as usize;
4701 let mut fields = Vec::with_capacity(width);
4702 for _ in 0..width {
4703 let name = cur.text()?;
4704 let ty = read_type(&mut cur)?;
4705 let not_null = match cur.u8()? {
4706 0 => false,
4707 1 => true,
4708 _ => return Err(invalid("nullability flag differs")),
4709 };
4710 fields.push(Field { name, ty, not_null });
4711 }
4712 let directory = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4713 let end = directory
4714 .offset
4715 .checked_add(u64::from(directory.length))
4716 .ok_or_else(|| invalid("table directory offset overflow"))?;
4717 if directory.offset < HEADER
4718 || end > size
4719 || directory.length as usize > MAX_DIRECTORY
4720 || directory.length == 0
4721 {
4722 return Err(invalid("table directory range is outside the file"));
4723 }
4724 if entries.iter().any(|held| held.name == name) {
4725 return Err(invalid("two tables in the catalog have the same name"));
4726 }
4727 entries.push(Entry { name, fields, rows, directory });
4728 }
4729 let count = if cur.done() { 0 } else { cur.u32()? as usize };
4734 let mut views: Vec<ViewEntry> = Vec::with_capacity(count.min(1024));
4735 for _ in 0..count {
4736 let name = cur.text()?;
4737 let sql = cur.long_text()?;
4738 let statement = cur.long_text()?;
4739 let width = cur.u16()? as usize;
4740 let mut aliases = Vec::with_capacity(width);
4741 for _ in 0..width {
4742 aliases.push(cur.text()?);
4743 }
4744 let width = cur.u16()? as usize;
4745 let mut columns = Vec::with_capacity(width);
4746 for _ in 0..width {
4747 let name = cur.text()?;
4748 let ty = read_type(&mut cur)?;
4749 let not_null = match cur.u8()? {
4750 0 => false,
4751 1 => true,
4752 _ => return Err(invalid("nullability flag differs")),
4753 };
4754 columns.push(Field { name, ty, not_null });
4755 }
4756 if views.iter().any(|held| held.name == name) {
4760 return Err(invalid("two views in the catalog have the same name"));
4761 }
4762 if entries.iter().any(|held| held.name == name) {
4763 return Err(invalid("a table and a view in the catalog have the same name"));
4764 }
4765 views.push(ViewEntry { name, sql, statement, aliases, columns });
4766 }
4767 Ok((entries, views))
4768}
4769
4770struct Cursor<'a> {
4771 bytes: &'a [u8],
4772 at: usize,
4773}
4774impl<'a> Cursor<'a> {
4775 fn take(&mut self, len: usize) -> Result<&'a [u8]> {
4776 let end = self.at.checked_add(len).ok_or_else(|| invalid("directory offset overflow"))?;
4777 let bytes =
4778 self.bytes.get(self.at..end).ok_or_else(|| invalid("directory is truncated"))?;
4779 self.at = end;
4780 Ok(bytes)
4781 }
4782 fn u8(&mut self) -> Result<u8> {
4783 Ok(self.take(1)?[0])
4784 }
4785 fn u16(&mut self) -> Result<u16> {
4786 Ok(u16::from_le_bytes(self.take(2)?.try_into().expect("two bytes")))
4787 }
4788 fn u32(&mut self) -> Result<u32> {
4789 Ok(u32::from_le_bytes(self.take(4)?.try_into().expect("four bytes")))
4790 }
4791 fn u64(&mut self) -> Result<u64> {
4792 Ok(u64::from_le_bytes(self.take(8)?.try_into().expect("eight bytes")))
4793 }
4794 fn var_u64(&mut self) -> Result<u64> {
4795 let mut value = 0_u64;
4796 for shift in (0..=63).step_by(7) {
4797 let byte = self.u8()?;
4798 let part = u64::from(byte & 0x7f);
4799 if shift == 63 && part > 1 {
4800 return Err(invalid("frequency ordinal varint overflows"));
4801 }
4802 value |= part << shift;
4803 if byte & 0x80 == 0 {
4804 return Ok(value);
4805 }
4806 }
4807 Err(invalid("frequency ordinal varint is too long"))
4808 }
4809 fn bound(&mut self) -> Result<Option<Bound>> {
4815 bounds::get(self.bytes, &mut self.at)
4816 }
4817 fn text(&mut self) -> Result<String> {
4818 let len = self.u16()? as usize;
4819 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("name is not UTF-8"))
4820 }
4821 fn done(&self) -> bool {
4824 self.at >= self.bytes.len()
4825 }
4826 fn long_text(&mut self) -> Result<String> {
4833 let len = self.u32()? as usize;
4834 String::from_utf8(self.take(len)?.to_vec()).map_err(|_| invalid("text is not UTF-8"))
4835 }
4836}
4837
4838fn decode_directory(bytes: &[u8], size: u64) -> Result<Table> {
4839 let mut cur = Cursor { bytes, at: 0 };
4840 if cur.take(8)? != DIRECTORY {
4841 return Err(invalid("directory magic differs"));
4842 }
4843 let name = cur.text()?;
4844 let width = cur.u16()? as usize;
4845 let mut fields = Vec::with_capacity(width);
4846 for _ in 0..width {
4847 let name = cur.text()?;
4848 let ty = read_type(&mut cur)?;
4849 let not_null = match cur.u8()? {
4850 0 => false,
4851 1 => true,
4852 _ => return Err(invalid("nullability flag differs")),
4853 };
4854 fields.push(Field { name, ty, not_null });
4855 }
4856 let mut dictionaries = Vec::with_capacity(width);
4857 for _ in 0..width {
4858 dictionaries.push(match cur.u8()? {
4859 0 => None,
4860 1 => {
4861 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4862 let end = page
4863 .offset
4864 .checked_add(u64::from(page.length))
4865 .ok_or_else(|| invalid("dictionary page offset overflow"))?;
4866 if page.offset < HEADER || end > size {
4871 return Err(invalid("dictionary page range is outside the file"));
4872 }
4873 Some(page)
4874 }
4875 _ => return Err(invalid("dictionary page tag differs")),
4876 });
4877 }
4878 let mut distincts = Vec::with_capacity(width);
4879 for _ in 0..width {
4880 distincts.push(match cur.u8()? {
4881 0 => None,
4882 1 => Some(cur.u64()?),
4883 _ => return Err(invalid("distinct count tag differs")),
4884 });
4885 }
4886 let rows = usize::try_from(cur.u64()?).map_err(|_| invalid("row count does not fit"))?;
4887 let count = cur.u32()? as usize;
4888 let mut stripes = Vec::with_capacity(count);
4889 let mut total = 0_usize;
4890 for _ in 0..count {
4891 let count = cur.u32()? as usize;
4892 if count == 0 || count > STRIPE_PARTS {
4893 return Err(invalid("stripe part count is outside its bound"));
4894 }
4895 let mut parts = Vec::with_capacity(count);
4896 let mut stripe_rows = 0_usize;
4897 for _ in 0..count {
4898 let rows = cur.u32()?;
4899 if rows == 0 {
4900 return Err(invalid("empty part"));
4901 }
4902 parts.push(rows);
4903 stripe_rows = stripe_rows
4904 .checked_add(rows as usize)
4905 .ok_or_else(|| invalid("stripe row count overflow"))?;
4906 }
4907 total =
4908 total.checked_add(stripe_rows).ok_or_else(|| invalid("stripe row count overflow"))?;
4909 let index = Span { offset: cur.u64()?, length: cur.u32()? };
4910 let section = index_section(count)?;
4911 let wanted = section
4912 .checked_mul(width)
4913 .and_then(|bytes| u32::try_from(bytes).ok())
4914 .ok_or_else(|| invalid("index page length overflow"))?;
4915 let end = index
4916 .offset
4917 .checked_add(u64::from(index.length))
4918 .ok_or_else(|| invalid("index page offset overflow"))?;
4919 if index.offset < HEADER || end > size || index.length != wanted {
4920 return Err(invalid("index page range is outside the file"));
4921 }
4922 let mut pages = Vec::with_capacity(width);
4923 for _ in 0..width {
4924 let offset = cur.u64()?;
4925 let length = cur.u32()?;
4926 let end = offset
4927 .checked_add(u64::from(length))
4928 .ok_or_else(|| invalid("page offset overflow"))?;
4929 if offset < HEADER || end > size || length as usize > MAX_PAGE {
4930 return Err(invalid("page range is outside the file"));
4931 }
4932 pages.push(Span { offset, length });
4933 }
4934 let mut memberships = vec![None; width];
4935 for (column, field) in fields.iter().enumerate() {
4936 if field.ty != LogicalType::Varchar || dictionaries[column].is_none() {
4937 continue;
4938 }
4939 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4940 let end = page
4941 .offset
4942 .checked_add(u64::from(page.length))
4943 .ok_or_else(|| invalid("membership page offset overflow"))?;
4944 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4945 return Err(invalid("membership page range is outside the file"));
4946 }
4947 memberships[column] = Some(page);
4948 }
4949 let mut sieves = vec![None; width];
4950 for sieve in sieves.iter_mut().take(width) {
4951 match cur.u8()? {
4952 0 => continue,
4953 1 => {}
4954 _ => return Err(invalid("a sieve page has an unknown tag")),
4955 }
4956 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4957 let end = page
4958 .offset
4959 .checked_add(u64::from(page.length))
4960 .ok_or_else(|| invalid("sieve page offset overflow"))?;
4961 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4962 return Err(invalid("sieve page range is outside the file"));
4963 }
4964 *sieve = Some(page);
4965 }
4966 let mut part_ranges = vec![None; width];
4967 for held in part_ranges.iter_mut().take(width) {
4968 match cur.u8()? {
4969 0 => continue,
4970 1 => {}
4971 _ => return Err(invalid("a part range page has an unknown tag")),
4972 }
4973 let page = Page { offset: cur.u64()?, length: cur.u32()?, hash: cur.u64()? };
4974 let end = page
4975 .offset
4976 .checked_add(u64::from(page.length))
4977 .ok_or_else(|| invalid("part range page offset overflow"))?;
4978 if page.offset < HEADER || end > size || page.length as usize > MAX_PAGE {
4979 return Err(invalid("part range page range is outside the file"));
4980 }
4981 *held = Some(page);
4982 }
4983 let mut ranges = Vec::with_capacity(width);
4984 for column in 0..width {
4985 let low = cur.bound()?;
4986 let high = cur.bound()?;
4987 let nulls = cur.u32()? as usize;
4988 if nulls > stripe_rows {
4989 return Err(invalid("null count exceeds stripe rows"));
4990 }
4991 let exact = cur.u8()? != 0;
4992 let sum = match cur.u8()? {
4993 0 => None,
4994 1 => Some(i128::from_le_bytes(
4995 cur.take(16)?.try_into().map_err(|_| invalid("a stripe sum is truncated"))?,
4996 )),
4997 _ => return Err(invalid("a stripe sum has an unknown tag")),
4998 };
4999 let ty = &fields.get(column).ok_or_else(|| invalid("a stripe range has no column"))?.ty;
5005 let low = low.map(|bound| scaled_as(bound, ty));
5006 let high = high.map(|bound| scaled_as(bound, ty));
5007 ranges.push(Range { low, high, nulls, exact, sum });
5008 }
5009 stripes.push(Stripe {
5010 rows: stripe_rows,
5011 parts,
5012 index,
5013 pages,
5014 memberships,
5015 sieves,
5016 part_ranges,
5017 zone: Zone::from_ranges(ranges),
5018 });
5019 }
5020 if total != rows {
5021 return Err(invalid("table row count differs from stripes"));
5022 }
5023 let frequencies = if cur.at == bytes.len() {
5024 vec![None; width]
5025 } else {
5026 if cur.take(8)? != FREQUENCIES {
5027 return Err(invalid("directory extension magic differs"));
5028 }
5029 if cur.u16()? as usize != width {
5030 return Err(invalid("frequency column count differs"));
5031 }
5032 let mut frequencies = Vec::with_capacity(width);
5033 for field in &fields {
5034 let summary = match cur.u8()? {
5035 0 => None,
5036 1 => {
5037 let omitted_max = cur.u64()?;
5038 let count = cur.u32()? as usize;
5039 if count > FREQUENCY_ENTRIES {
5040 return Err(invalid("frequency entry count exceeds its bound"));
5041 }
5042 let mut entries = Vec::with_capacity(count);
5043 for _ in 0..count {
5045 let value = match cur.u8()? {
5046 0 => FrequencyValue::Null,
5047 1 => FrequencyValue::Integer(i128::from_le_bytes(
5048 cur.take(16)?.try_into().expect("sixteen bytes"),
5049 )),
5050 2 => FrequencyValue::Code(cur.u32()?),
5051 _ => return Err(invalid("frequency value tag differs")),
5052 };
5053 let valid = matches!(
5054 (&field.ty, value),
5055 (_, FrequencyValue::Null)
5056 | (LogicalType::Varchar, FrequencyValue::Code(_))
5057 | (
5058 LogicalType::TinyInt
5059 | LogicalType::SmallInt
5060 | LogicalType::Integer
5061 | LogicalType::BigInt
5062 | LogicalType::UTinyInt
5063 | LogicalType::USmallInt
5064 | LogicalType::UInteger
5065 | LogicalType::UBigInt
5066 | LogicalType::Date
5067 | LogicalType::Timestamp,
5068 FrequencyValue::Integer(_),
5069 )
5070 );
5071 if !valid {
5072 return Err(invalid("frequency value does not match its column"));
5073 }
5074 let count = cur.u64()?;
5075 if count == 0 || count > rows as u64 {
5076 return Err(invalid("frequency count is outside the table"));
5077 }
5078 entries.push(FrequencyEntry { value, count });
5079 }
5080 if entries.windows(2).any(|pair| pair[0].count < pair[1].count) {
5081 return Err(invalid("frequency entries are not descending"));
5082 }
5083 let ordinals = {
5084 let ordinal_count = cur.u32()? as usize;
5085 if ordinal_count > FREQUENCY_ORDINALS || ordinal_count > rows {
5086 return Err(invalid("frequency ordinal count exceeds its bound"));
5087 }
5088 let mut ordinals = Vec::with_capacity(ordinal_count);
5089 let mut previous = 0_u64;
5090 for at in 0..ordinal_count {
5091 let delta = cur.var_u64()?;
5092 if at != 0 && delta == 0 {
5093 return Err(invalid("frequency ordinals are not increasing"));
5094 }
5095 let ordinal = if at == 0 {
5096 delta
5097 } else {
5098 previous
5099 .checked_add(delta)
5100 .ok_or_else(|| invalid("frequency ordinal overflows"))?
5101 };
5102 if ordinal >= rows as u64 {
5103 return Err(invalid("frequency ordinal is outside the table"));
5104 }
5105 ordinals.push(ordinal);
5106 previous = ordinal;
5107 }
5108 ordinals
5109 };
5110 Some(FrequencySummary { entries, omitted_max, ordinals })
5111 }
5112 _ => return Err(invalid("frequency summary tag differs")),
5113 };
5114 frequencies.push(summary);
5115 }
5116 frequencies
5117 };
5118 let mut clustering = None;
5128 let mut sections = Vec::new();
5129 let mut seen_sections = false;
5130 let mut generation = 0;
5133 while cur.at != bytes.len() {
5134 let mut tag = [0u8; 8];
5135 tag.copy_from_slice(cur.take(8)?);
5136 if &tag == CLUSTERING {
5137 if clustering.is_some() {
5138 return Err(invalid("directory names two clustering declarations"));
5139 }
5140 let bucket = Width::from_tag(cur.u8()?)
5141 .ok_or_else(|| invalid("clustering width tag differs"))?;
5142 let count = cur.u16()? as usize;
5143 let mut columns = Vec::with_capacity(count.min(fields.len()));
5144 for _ in 0..count {
5145 columns.push(u32::from(cur.u16()?));
5146 }
5147 clustering = Some(Clustering::new(columns, bucket, &fields).map_err(|_| {
5150 invalid("stored clustering declaration does not match the table it is on")
5151 })?);
5152 } else if &tag == SECTIONS {
5153 if seen_sections {
5154 return Err(invalid("directory names two section tables"));
5155 }
5156 seen_sections = true;
5157 generation = cur.u64()?;
5158 let count = cur.u16()? as usize;
5159 if count > MAX_SECTIONS {
5160 return Err(invalid("section count exceeds its bound"));
5161 }
5162 sections = Vec::with_capacity(count);
5163 for _ in 0..count {
5166 sections.push(Section::decode(cur.take(section::ENTRY_BYTES)?)?);
5167 }
5168 for held in §ions {
5169 let Some(end) = held.extent_page.checked_add(u64::from(held.extent_bytes)) else {
5170 return Err(invalid("a section's extent table overflows the file"));
5171 };
5172 if held.extent_bytes != 0 && (held.extent_page < HEADER || end > size) {
5176 return Err(invalid("a section's extent table is outside the file"));
5177 }
5178 if held.extents == 0 && held.extent_bytes != 0 {
5179 return Err(invalid("a section with no extents names an extent table"));
5180 }
5181 }
5182 } else {
5183 return Err(invalid("directory extension magic differs"));
5184 }
5185 }
5186 if cur.at != bytes.len() {
5187 return Err(invalid("directory has trailing bytes"));
5188 }
5189 Ok(Table {
5190 name,
5191 fields,
5192 stripes,
5193 rows,
5194 dictionaries,
5195 distincts,
5196 frequencies,
5197 clustering,
5198 generation,
5199 sections,
5200 })
5201}
5202
5203fn put_bound(out: &mut Vec<u8>, bound: Option<&Bound>) -> Result<()> {
5205 bounds::put(out, bound)
5206}
5207
5208#[derive(Debug)]
5225struct Codes;
5226
5227impl chooser::Chooser for Codes {
5228 fn name(&self) -> &'static str {
5229 "codes"
5230 }
5231
5232 fn narrow_strings(
5233 &self,
5234 _values: &[&[u8]],
5235 offered: &[string::Kind],
5236 _depth: u8,
5237 ) -> Vec<string::Kind> {
5238 offered.to_vec()
5241 }
5242
5243 fn narrow_integers(
5244 &self,
5245 _values: &[i64],
5246 offered: &[integer::Kind],
5247 depth: u8,
5248 ) -> Vec<integer::Kind> {
5249 let keep: &[integer::Kind] = if depth == 0 {
5250 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Rle]
5251 } else {
5252 &[integer::Kind::Constant, integer::Kind::Packed]
5253 };
5254 let narrowed: Vec<integer::Kind> =
5255 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5256 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5259 }
5260}
5261
5262#[derive(Debug)]
5274struct Fixed;
5275
5276impl chooser::Chooser for Fixed {
5277 fn name(&self) -> &'static str {
5278 "fixed"
5279 }
5280
5281 fn narrow_strings(
5282 &self,
5283 _values: &[&[u8]],
5284 offered: &[string::Kind],
5285 _depth: u8,
5286 ) -> Vec<string::Kind> {
5287 offered.to_vec()
5288 }
5289
5290 fn narrow_integers(
5291 &self,
5292 _values: &[i64],
5293 offered: &[integer::Kind],
5294 depth: u8,
5295 ) -> Vec<integer::Kind> {
5296 let keep: &[integer::Kind] = if depth == 0 {
5297 &[
5298 integer::Kind::Constant,
5299 integer::Kind::Packed,
5300 integer::Kind::Delta,
5301 integer::Kind::Rle,
5302 integer::Kind::Sparse,
5303 integer::Kind::Strided,
5304 ]
5305 } else {
5306 &[integer::Kind::Constant, integer::Kind::Packed, integer::Kind::Delta]
5307 };
5308 let narrowed: Vec<integer::Kind> =
5309 offered.iter().copied().filter(|kind| keep.contains(kind)).collect();
5310 if narrowed.is_empty() { offered.to_vec() } else { narrowed }
5311 }
5312}
5313
5314fn widened(data: &Data) -> Option<Vec<i64>> {
5321 match data {
5322 Data::Int8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5323 Data::UInt8(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5324 Data::Int16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5325 Data::UInt16(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5326 Data::Int32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5327 Data::UInt32(values) => Some(values.iter().map(|value| i64::from(*value)).collect()),
5328 Data::Int64(values) => Some(values.to_vec()),
5329 _ => None,
5330 }
5331}
5332
5333trait Narrow: Copy {
5340 const BIASED: (u32, u64);
5345
5346 fn narrow(value: i64) -> Self;
5348}
5349
5350#[allow(clippy::cast_sign_loss, reason = "a residue is a bit pattern and not a number")]
5367fn residue<T: Narrow>(value: i64) -> u64 {
5368 let (bits, bias) = T::BIASED;
5369 (value as u64).wrapping_add(bias) >> bits
5370}
5371
5372macro_rules! narrows {
5377 ($($ty:ty => $bias:expr),* $(,)?) => {$(
5378 impl Narrow for $ty {
5379 const BIASED: (u32, u64) = (<$ty>::BITS, $bias);
5380
5381 #[allow(
5382 clippy::cast_possible_truncation,
5383 clippy::cast_sign_loss,
5384 reason = "the caller has checked the bits this truncates away"
5385 )]
5386 fn narrow(value: i64) -> Self {
5387 value as Self
5388 }
5389 }
5390 )*};
5391}
5392
5393narrows! {
5394 i8 => 1 << 7,
5395 u8 => 0,
5396 i16 => 1 << 15,
5397 u16 => 0,
5398 i32 => 1 << 31,
5399 u32 => 0,
5400}
5401
5402fn fit<T: Narrow>(values: &[i64]) -> Result<Vec<T>> {
5415 let mut spilled = 0u64;
5416 for value in values {
5417 spilled |= residue::<T>(*value);
5418 }
5419 if spilled != 0 {
5420 return Err(invalid("page value is not of its type"));
5421 }
5422 Ok(values.iter().map(|value| T::narrow(*value)).collect())
5423}
5424
5425fn narrowed(ty: &LogicalType, values: Vec<i64>) -> Result<Data> {
5430 Ok(match ty {
5431 LogicalType::TinyInt => Data::Int8(fit::<i8>(&values)?.into()),
5432 LogicalType::UTinyInt => Data::UInt8(fit::<u8>(&values)?.into()),
5433 LogicalType::SmallInt => Data::Int16(fit::<i16>(&values)?.into()),
5434 LogicalType::USmallInt => Data::UInt16(fit::<u16>(&values)?.into()),
5435 LogicalType::Integer | LogicalType::Date => Data::Int32(fit::<i32>(&values)?.into()),
5436 LogicalType::UInteger => Data::UInt32(fit::<u32>(&values)?.into()),
5437 LogicalType::BigInt
5438 | LogicalType::Timestamp
5439 | LogicalType::Time
5440 | LogicalType::TimeTz
5441 | LogicalType::TimestampTz
5442 | LogicalType::TimestampS
5443 | LogicalType::TimestampMs
5444 | LogicalType::TimestampNs => Data::Int64(values.into()),
5445 LogicalType::Decimal { .. } => match ty.physical() {
5448 PhysicalType::Int16 => Data::Int16(fit::<i16>(&values)?.into()),
5449 PhysicalType::Int32 => Data::Int32(fit::<i32>(&values)?.into()),
5450 PhysicalType::Int64 => Data::Int64(values.into()),
5451 _ => return Err(invalid("cascade codec belongs to a decimal that is not an integer")),
5452 },
5453 _ => return Err(invalid("cascade codec belongs to a page that is not integers")),
5454 })
5455}
5456
5457fn plain_width(ty: &LogicalType) -> Option<usize> {
5460 Some(match ty {
5461 LogicalType::TinyInt | LogicalType::UTinyInt => 1,
5462 LogicalType::SmallInt | LogicalType::USmallInt => 2,
5463 LogicalType::Integer | LogicalType::UInteger | LogicalType::Date => 4,
5464 LogicalType::BigInt
5465 | LogicalType::Timestamp
5466 | LogicalType::Time
5467 | LogicalType::TimeTz
5468 | LogicalType::TimestampTz
5469 | LogicalType::TimestampS
5470 | LogicalType::TimestampMs
5471 | LogicalType::TimestampNs => 8,
5472 LogicalType::Decimal { .. } => match ty.physical() {
5473 PhysicalType::Int16 => 2,
5474 PhysicalType::Int32 => 4,
5475 PhysicalType::Int64 => 8,
5476 _ => return None,
5479 },
5480 _ => return None,
5481 })
5482}
5483
5484fn cascaded(
5490 flat: &Vector,
5491 ty: &LogicalType,
5492 packed: Option<&Packed<'_>>,
5493) -> Result<Option<Vec<u8>>> {
5494 let (Some(width), Some(data)) = (plain_width(ty), flat.data()) else { return Ok(None) };
5495 let Some(values) = widened(data) else { return Ok(None) };
5496 let plain = values.len().saturating_mul(width);
5497 let best = match packed {
5498 Some(packed) => plain.min(21 + size_of_val(packed.words())),
5500 None => plain,
5501 };
5502 let out = integer::encode_with(&values, &Fixed)?;
5503 Ok((out.len() < best).then_some(out))
5504}
5505
5506fn text_compressed(flat: &Vector) -> Result<Option<Vec<u8>>> {
5544 let mut values: Vec<&[u8]> = Vec::with_capacity(flat.len());
5545 let mut payload = 0_usize;
5546 for row in 0..flat.len() {
5547 let text = flat.text_at(row).unwrap_or("").as_bytes();
5548 payload = payload.saturating_add(text.len());
5549 values.push(text);
5550 }
5551 let plain = (flat.len() + 1).saturating_mul(4).saturating_add(payload);
5553 let Some(out) = string::encode_only(string::Kind::Fsst, &values)? else {
5554 return Ok(None);
5555 };
5556 Ok((out.len() < plain).then_some(out))
5557}
5558
5559fn encoded_codes(codes: &[u32]) -> Result<Option<Vec<u8>>> {
5560 let wide: Vec<i64> = codes.iter().map(|code| i64::from(*code)).collect();
5561 let coded = integer::encode_with(&wide, &Codes)?;
5562 let plain = codes.len().saturating_mul(size_of::<u32>());
5563 Ok((coded.len() < plain).then_some(coded))
5564}
5565
5566fn encode(
5567 vector: &Vector,
5568 global: Option<&mut GlobalDictionary>,
5569) -> Result<(Vec<u8>, Option<Vec<u32>>)> {
5570 let ty = vector.logical_type();
5571 let flat = vector.flatten()?;
5573 let mut out = Vec::new();
5574 let mut global_codes = None;
5575 if let Some(global) = global {
5576 let mut codes = Vec::with_capacity(flat.len());
5577 for row in 0..flat.len() {
5578 let text = flat.text_at(row).unwrap_or("");
5579 let code = global.code(text)?;
5580 global.observe(code, flat.is_null_at(row))?;
5581 codes.push(code);
5582 }
5583 global_codes = Some(codes);
5584 }
5585 let membership = global_codes.as_deref().map(unique_codes);
5586 let dictionary = if global_codes.is_none() && ty == &LogicalType::Varchar {
5587 string_dictionary(&flat)?
5588 } else {
5589 None
5590 };
5591 let compressed_text =
5592 if global_codes.is_none() && dictionary.is_none() && ty == &LogicalType::Varchar {
5593 text_compressed(&flat)?
5594 } else {
5595 None
5596 };
5597 let packed_vector = if dictionary.is_none() && global_codes.is_none() {
5598 Some(flat.bit_packed()?)
5599 } else {
5600 None
5601 };
5602 let packed = packed_vector.as_ref().and_then(Vector::packed_parts);
5603 let coded = match global_codes.as_deref() {
5604 Some(codes) => encoded_codes(codes)?,
5605 None => None,
5606 };
5607 let cascade = if dictionary.is_none() && global_codes.is_none() {
5611 cascaded(&flat, ty, packed.as_ref())?
5612 } else {
5613 None
5614 };
5615 out.push(if coded.is_some() {
5616 4
5617 } else if cascade.is_some() {
5618 5
5619 } else if global_codes.is_some() {
5620 3
5621 } else if dictionary.is_some() {
5622 1
5623 } else if compressed_text.is_some() {
5624 6
5625 } else if packed.is_some() {
5626 2
5627 } else {
5628 0
5629 });
5630 let nulls = flat.validity();
5631 let flag = match nulls {
5632 Validity::AllValid => 0,
5633 Validity::AllInvalid => 1,
5634 Validity::Mask(_) => 2,
5635 };
5636 out.push(flag);
5637 if flag == 2 {
5638 for group in (0..vector.len()).step_by(8) {
5639 let mut bits = 0_u8;
5640 for bit in 0..8 {
5641 if group + bit < vector.len() && !flat.is_null_at(group + bit) {
5642 bits |= 1 << bit;
5643 }
5644 }
5645 out.push(bits);
5646 }
5647 }
5648 if let Some(coded) = coded {
5649 out.extend_from_slice(&coded);
5650 return Ok((out, membership));
5651 }
5652 if let Some(cascade) = cascade {
5653 out.extend_from_slice(&cascade);
5654 return Ok((out, membership));
5655 }
5656 if let Some(codes) = global_codes {
5657 for code in codes {
5658 put_u32(&mut out, code);
5659 }
5660 return Ok((out, membership));
5661 }
5662 if let Some(dictionary) = dictionary {
5663 out.extend_from_slice(&dictionary);
5664 return Ok((out, membership));
5665 }
5666 if let Some(compressed_text) = compressed_text {
5667 out.extend_from_slice(&compressed_text);
5668 return Ok((out, membership));
5669 }
5670 if let Some(packed) = packed {
5671 if packed.offset() != 0 {
5672 return Err(invalid("writer received a sliced packed vector"));
5673 }
5674 out.push(u8::try_from(packed.width()).map_err(|_| invalid("packed width overflow"))?);
5675 out.extend_from_slice(&packed.base().to_le_bytes());
5676 put_u32(
5677 &mut out,
5678 u32::try_from(packed.words().len()).map_err(|_| invalid("too many packed words"))?,
5679 );
5680 for word in packed.words() {
5681 put_u64(&mut out, *word);
5682 }
5683 return Ok((out, membership));
5684 }
5685 let data = flat.data().ok_or_else(|| invalid("scalar column did not flatten"))?;
5686 match (ty, data) {
5687 (LogicalType::TinyInt, Data::Int8(values)) => {
5688 for value in &**values {
5689 out.extend_from_slice(&value.to_le_bytes());
5690 }
5691 }
5692 (LogicalType::UTinyInt, Data::UInt8(values)) => {
5693 for value in &**values {
5694 out.extend_from_slice(&value.to_le_bytes());
5695 }
5696 }
5697 (LogicalType::SmallInt, Data::Int16(values)) => {
5698 for value in &**values {
5699 out.extend_from_slice(&value.to_le_bytes());
5700 }
5701 }
5702 (LogicalType::USmallInt, Data::UInt16(values)) => {
5703 for value in &**values {
5704 out.extend_from_slice(&value.to_le_bytes());
5705 }
5706 }
5707 (LogicalType::UInteger, Data::UInt32(values)) => {
5708 for value in &**values {
5709 out.extend_from_slice(&value.to_le_bytes());
5710 }
5711 }
5712 (LogicalType::UBigInt, Data::UInt64(values)) => {
5713 for value in &**values {
5714 out.extend_from_slice(&value.to_le_bytes());
5715 }
5716 }
5717 (LogicalType::Integer | LogicalType::Date, Data::Int32(values)) => {
5718 for value in &**values {
5719 out.extend_from_slice(&value.to_le_bytes());
5720 }
5721 }
5722 (
5723 LogicalType::BigInt
5724 | LogicalType::Timestamp
5725 | LogicalType::Time
5726 | LogicalType::TimeTz
5727 | LogicalType::TimestampTz
5728 | LogicalType::TimestampS
5729 | LogicalType::TimestampMs
5730 | LogicalType::TimestampNs,
5731 Data::Int64(values),
5732 ) => {
5733 for value in &**values {
5734 out.extend_from_slice(&value.to_le_bytes());
5735 }
5736 }
5737 (LogicalType::HugeInt | LogicalType::Uuid, Data::Int128(values)) => {
5740 for value in &**values {
5741 out.extend_from_slice(&value.to_le_bytes());
5742 }
5743 }
5744 (LogicalType::UHugeInt, Data::UInt128(values)) => {
5745 for value in &**values {
5746 out.extend_from_slice(&value.to_le_bytes());
5747 }
5748 }
5749 (LogicalType::Float, Data::Float32(values)) => {
5752 for value in &**values {
5753 out.extend_from_slice(&value.to_le_bytes());
5754 }
5755 }
5756 (LogicalType::Double, Data::Float64(values)) => {
5757 for value in &**values {
5758 out.extend_from_slice(&value.to_le_bytes());
5759 }
5760 }
5761 (LogicalType::Interval, Data::Interval(values)) => {
5765 for (months, days, micros) in &**values {
5766 out.extend_from_slice(&months.to_le_bytes());
5767 out.extend_from_slice(&days.to_le_bytes());
5768 out.extend_from_slice(µs.to_le_bytes());
5769 }
5770 }
5771 (LogicalType::Boolean, Data::Bool(values)) => {
5772 for value in &**values {
5773 out.push(u8::from(*value));
5774 }
5775 }
5776 (LogicalType::Decimal { .. }, Data::Int16(values)) => {
5779 for value in &**values {
5780 out.extend_from_slice(&value.to_le_bytes());
5781 }
5782 }
5783 (LogicalType::Decimal { .. }, Data::Int32(values)) => {
5784 for value in &**values {
5785 out.extend_from_slice(&value.to_le_bytes());
5786 }
5787 }
5788 (LogicalType::Decimal { .. }, Data::Int64(values)) => {
5789 for value in &**values {
5790 out.extend_from_slice(&value.to_le_bytes());
5791 }
5792 }
5793 (LogicalType::Decimal { .. }, Data::Int128(values)) => {
5794 for value in &**values {
5795 out.extend_from_slice(&value.to_le_bytes());
5796 }
5797 }
5798 (LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit, Data::Varlen(values)) => {
5803 let mut bytes = Vec::new();
5804 put_u32(&mut out, 0);
5805 for row in 0..vector.len() {
5806 let value = values.bytes(row).ok_or_else(|| invalid("string view is invalid"))?;
5807 bytes.extend_from_slice(value);
5808 put_u32(
5809 &mut out,
5810 u32::try_from(bytes.len())
5811 .map_err(|_| invalid("string payload exceeds 4GiB"))?,
5812 );
5813 }
5814 out.extend_from_slice(&bytes);
5815 }
5816 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
5817 }
5818 Ok((out, membership))
5819}
5820
5821fn put_varint(out: &mut Vec<u8>, mut value: u32) {
5822 while value >= 0x80 {
5823 out.push((value as u8 & 0x7f) | 0x80);
5824 value >>= 7;
5825 }
5826 out.push(value as u8);
5827}
5828
5829fn unique_codes(codes: &[u32]) -> Vec<u32> {
5831 let mut unique = codes.to_vec();
5832 unique.sort_unstable();
5833 unique.dedup();
5834 unique
5835}
5836
5837fn merged_codes(lists: Vec<Vec<u32>>) -> Vec<u32> {
5843 let mut lists = lists;
5844 while lists.len() > 1 {
5845 let mut next = Vec::with_capacity(lists.len().div_ceil(2));
5846 for pair in lists.chunks(2) {
5847 match pair {
5848 [left, right] => next.push(merged_pair(left, right)),
5849 [only] => next.push(only.clone()),
5850 _ => {}
5851 }
5852 }
5853 lists = next;
5854 }
5855 lists.pop().unwrap_or_default()
5856}
5857
5858fn merged_pair(left: &[u32], right: &[u32]) -> Vec<u32> {
5859 let mut out = Vec::with_capacity(left.len().saturating_add(right.len()));
5860 let mut at = 0;
5861 let mut to = 0;
5862 while at < left.len() && to < right.len() {
5863 match left[at].cmp(&right[to]) {
5864 Ordering::Less => {
5865 out.push(left[at]);
5866 at += 1;
5867 }
5868 Ordering::Greater => {
5869 out.push(right[to]);
5870 to += 1;
5871 }
5872 Ordering::Equal => {
5873 out.push(left[at]);
5874 at += 1;
5875 to += 1;
5876 }
5877 }
5878 }
5879 out.extend_from_slice(&left[at..]);
5880 out.extend_from_slice(&right[to..]);
5881 out
5882}
5883
5884fn merged_range(ranges: impl Iterator<Item = Range>) -> Range {
5889 let mut merged = Range::default();
5890 let mut first = true;
5891 for range in ranges {
5892 merged.nulls = merged.nulls.saturating_add(range.nulls);
5893 merged.sum = match (merged.sum.take(), range.sum) {
5897 (Some(held), Some(next)) if !first => held.checked_add(next),
5898 (_, next) if first => next,
5899 _ => None,
5900 };
5901 merged.exact = if first { range.exact } else { merged.exact && range.exact };
5902 if first {
5903 merged.low = range.low;
5904 merged.high = range.high;
5905 first = false;
5906 continue;
5907 }
5908 merged.low = match (merged.low.take(), range.low) {
5909 (Some(held), Some(next)) => Some(held.smaller(next)),
5910 _ => None,
5911 };
5912 merged.high = match (merged.high.take(), range.high) {
5913 (Some(held), Some(next)) => Some(held.larger(next)),
5914 _ => None,
5915 };
5916 }
5917 merged
5918}
5919
5920fn shortened(bound: Option<Bound>, high: bool) -> Option<Bound> {
5933 match bound {
5934 Some(Bound::Bytes(mut value)) if value.len() > PART_BOUND_BYTES => {
5935 value.truncate(PART_BOUND_BYTES);
5936 if !high {
5937 return Some(Bound::Bytes(value));
5938 }
5939 while let Some(last) = value.pop() {
5940 if last < u8::MAX {
5941 value.push(last + 1);
5942 return Some(Bound::Bytes(value));
5943 }
5944 }
5945 None
5946 }
5947 other => other,
5948 }
5949}
5950
5951fn encode_part_ranges(ranges: &[Range]) -> Result<Vec<u8>> {
5959 let mut out = Vec::new();
5960 put_u32(
5961 &mut out,
5962 u32::try_from(ranges.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5963 );
5964 for range in ranges {
5965 put_bound(&mut out, shortened(range.low.clone(), false).as_ref())?;
5966 put_bound(&mut out, shortened(range.high.clone(), true).as_ref())?;
5967 put_u32(&mut out, u32::try_from(range.nulls).map_err(|_| invalid("null count overflow"))?);
5968 }
5969 Ok(out)
5970}
5971
5972fn decode_part_ranges(bytes: &[u8]) -> Result<Vec<Range>> {
5974 let mut cur = Cursor { bytes, at: 0 };
5975 let parts = cur.u32()? as usize;
5976 let mut out = Vec::new();
5977 for _ in 0..parts {
5978 let low = cur.bound()?;
5979 let high = cur.bound()?;
5980 let nulls = cur.u32()? as usize;
5981 out.push(Range { low, high, nulls, exact: false, sum: None });
5982 }
5983 Ok(out)
5984}
5985
5986fn encode_sieves<'a>(sieves: impl Iterator<Item = &'a Option<Sieve>>) -> Result<Vec<u8>> {
5987 let held: Vec<&Option<Sieve>> = sieves.collect();
5988 let mut out = Vec::new();
5989 put_u32(
5990 &mut out,
5991 u32::try_from(held.len()).map_err(|_| invalid("too many parts in a stripe"))?,
5992 );
5993 for sieve in &held {
5994 let length = sieve.as_ref().map_or(0, Sieve::len);
5995 put_u32(&mut out, u32::try_from(length).map_err(|_| invalid("sieve length overflow"))?);
5996 }
5997 for sieve in held.into_iter().flatten() {
5999 out.extend_from_slice(&sieve.to_bytes());
6000 }
6001 Ok(out)
6002}
6003
6004fn decode_sieves(bytes: &[u8]) -> Result<Vec<Option<Sieve>>> {
6010 let parts = u32::from_le_bytes(
6011 bytes
6012 .get(..4)
6013 .ok_or_else(|| invalid("sieve page is truncated"))?
6014 .try_into()
6015 .map_err(|_| invalid("sieve page is truncated"))?,
6016 ) as usize;
6017 let mut lengths = Vec::with_capacity(parts);
6018 for part in 0..parts {
6019 let at = 4 + part * 4;
6020 let field = bytes.get(at..at + 4).ok_or_else(|| invalid("sieve page is truncated"))?;
6021 lengths.push(u32::from_le_bytes(
6022 field.try_into().map_err(|_| invalid("sieve page is truncated"))?,
6023 ) as usize);
6024 }
6025 let mut at = 4 + parts * 4;
6026 let mut out = Vec::with_capacity(parts);
6027 for length in lengths {
6028 if length == 0 {
6029 out.push(None);
6030 continue;
6031 }
6032 let end = at.checked_add(length).ok_or_else(|| invalid("sieve page is truncated"))?;
6033 let field = bytes.get(at..end).ok_or_else(|| invalid("sieve page is truncated"))?;
6034 out.push(Sieve::from_bytes(field));
6035 at = end;
6036 }
6037 if at != bytes.len() {
6038 return Err(invalid("sieve page has trailing bytes"));
6039 }
6040 Ok(out)
6041}
6042
6043fn encode_membership(unique: &[u32]) -> Vec<u8> {
6049 let mut out = Vec::with_capacity(unique.len().saturating_mul(2).saturating_add(5));
6050 put_varint(&mut out, u32::try_from(unique.len()).unwrap_or(u32::MAX));
6051 let mut previous = 0;
6052 for (at, &code) in unique.iter().enumerate() {
6053 put_varint(&mut out, if at == 0 { code } else { code - previous });
6054 previous = code;
6055 }
6056 out
6057}
6058
6059fn take_varint(bytes: &[u8], at: &mut usize) -> Result<u32> {
6060 let mut value = 0_u32;
6061 for shift in (0..35).step_by(7) {
6062 let byte = *bytes.get(*at).ok_or_else(|| invalid("membership varint is truncated"))?;
6063 *at += 1;
6064 let part = u32::from(byte & 0x7f);
6065 if shift == 28 && part > 0x0f {
6066 return Err(invalid("membership varint overflow"));
6067 }
6068 value = value
6069 .checked_add(
6070 part.checked_shl(shift).ok_or_else(|| invalid("membership varint overflow"))?,
6071 )
6072 .ok_or_else(|| invalid("membership varint overflow"))?;
6073 if byte & 0x80 == 0 {
6074 return Ok(value);
6075 }
6076 }
6077 Err(invalid("membership varint is too long"))
6078}
6079
6080fn decode_membership(bytes: &[u8]) -> Result<Vec<u32>> {
6081 let mut at = 0;
6082 let count = take_varint(bytes, &mut at)? as usize;
6083 let mut codes = Vec::with_capacity(count);
6084 let mut previous = 0_u32;
6085 for index in 0..count {
6086 let delta = take_varint(bytes, &mut at)?;
6087 let code = if index == 0 {
6088 delta
6089 } else {
6090 previous.checked_add(delta).ok_or_else(|| invalid("membership code overflow"))?
6091 };
6092 if index > 0 && code <= previous {
6093 return Err(invalid("membership codes are not increasing"));
6094 }
6095 codes.push(code);
6096 previous = code;
6097 }
6098 if at != bytes.len() {
6099 return Err(invalid("membership page has trailing bytes"));
6100 }
6101 Ok(codes)
6102}
6103
6104fn string_dictionary(vector: &Vector) -> Result<Option<Vec<u8>>> {
6105 let mut by_text = HashMap::new();
6106 let mut values = Vec::new();
6107 let mut codes = Vec::with_capacity(vector.len());
6108 let mut plain_bytes = 0_usize;
6109 for row in 0..vector.len() {
6110 let text = vector.text_at(row).unwrap_or("");
6111 plain_bytes = plain_bytes.saturating_add(text.len());
6112 let code = match by_text.get(text) {
6113 Some(&code) => code,
6114 None => {
6115 let code = u32::try_from(values.len())
6116 .map_err(|_| invalid("too many dictionary values"))?;
6117 by_text.insert(text, code);
6118 values.push(text);
6119 code
6120 }
6121 };
6122 codes.push(code);
6123 }
6124 let dictionary_bytes = values.iter().map(|value| value.len()).sum::<usize>();
6125 let encoded = 8_usize
6126 .saturating_add((values.len() + 1).saturating_mul(4))
6127 .saturating_add(dictionary_bytes)
6128 .saturating_add(codes.len().saturating_mul(4));
6129 let plain = (vector.len() + 1).saturating_mul(4).saturating_add(plain_bytes);
6130 if encoded >= plain {
6131 return Ok(None);
6132 }
6133 let mut out = Vec::with_capacity(encoded);
6134 put_u32(
6135 &mut out,
6136 u32::try_from(values.len()).map_err(|_| invalid("too many dictionary values"))?,
6137 );
6138 put_u32(
6139 &mut out,
6140 u32::try_from(dictionary_bytes).map_err(|_| invalid("dictionary payload exceeds 4GiB"))?,
6141 );
6142 let mut offset = 0_u32;
6143 put_u32(&mut out, offset);
6144 for value in &values {
6145 offset = offset
6146 .checked_add(
6147 u32::try_from(value.len()).map_err(|_| invalid("dictionary value is too long"))?,
6148 )
6149 .ok_or_else(|| invalid("dictionary payload exceeds 4GiB"))?;
6150 put_u32(&mut out, offset);
6151 }
6152 for value in values {
6153 out.extend_from_slice(value.as_bytes());
6154 }
6155 for code in codes {
6156 put_u32(&mut out, code);
6157 }
6158 Ok(Some(out))
6159}
6160
6161struct EncodedDictionary {
6162 index: Vec<u8>,
6163 ranks: Vec<u8>,
6164 payload: Vec<Vec<u8>>,
6167}
6168
6169fn head(bytes: &[u8]) -> u64 {
6171 let mut word = [0; 8];
6172 let take = bytes.len().min(8);
6173 word[..take].copy_from_slice(&bytes[..take]);
6174 u64::from_be_bytes(word)
6175}
6176
6177fn rankings(dictionaries: &[Option<GlobalDictionary>]) -> Result<Vec<Vec<(u64, u32)>>> {
6185 let present =
6186 dictionaries.iter().enumerate().filter(|(_, held)| held.is_some()).map(|(at, _)| at);
6187 let present = present.collect::<Vec<_>>();
6188 let mut orders = vec![Vec::new(); dictionaries.len()];
6189 let workers = std::thread::available_parallelism()
6190 .map_or(1, usize::from)
6191 .min(MAX_FREQUENCY_WORKERS)
6192 .min(present.len());
6193 if workers <= 1 {
6194 for at in present {
6195 if let Some(dictionary) = &dictionaries[at] {
6196 orders[at] = dictionary.ranked();
6197 }
6198 }
6199 return Ok(orders);
6200 }
6201 let width = present.len().div_ceil(workers);
6202 let pieces = std::thread::scope(|scope| {
6203 present
6204 .chunks(width)
6205 .map(|columns| {
6206 scope.spawn(|| {
6207 columns
6208 .iter()
6209 .filter_map(|&at| dictionaries[at].as_ref().map(|held| (at, held.ranked())))
6210 .collect::<Vec<_>>()
6211 })
6212 })
6213 .collect::<Vec<_>>()
6214 .into_iter()
6215 .map(|handle| {
6216 handle.join().map_err(|_| Error::internal("a dictionary sort worker panicked"))
6217 })
6218 .collect::<Result<Vec<_>>>()
6219 })?;
6220 for piece in pieces {
6221 for (at, order) in piece {
6222 orders[at] = order;
6223 }
6224 }
6225 Ok(orders)
6226}
6227
6228fn encode_global_dictionary(
6229 dictionary: GlobalDictionary,
6230 order: &[(u64, u32)],
6231) -> Result<EncodedDictionary> {
6232 let values = dictionary.offsets.len() - 1;
6233 if order.len() != values {
6234 return Err(invalid("global dictionary order does not cover its values"));
6235 }
6236 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6237 let payload = encode_payload(&dictionary)?;
6238 if payload.len() != blocks {
6239 return Err(invalid("global dictionary payload is not the blocks it says it is"));
6240 }
6241 let (ranks, rank_ends) = encode_ranks(order, code_width(values))?;
6242 let rank_blocks = values.div_ceil(TEXT_RANK_BLOCK);
6243 let offset_bits = offset_width(&dictionary.offsets);
6244 let mut index = Vec::with_capacity(
6245 DICTIONARY_HEADER + offset_bytes(values, offset_bits) + (blocks + rank_blocks) * 16,
6246 );
6247 put_u32(
6248 &mut index,
6249 u32::try_from(values).map_err(|_| invalid("global dictionary has too many values"))?,
6250 );
6251 put_u32(&mut index, TEXT_PAYLOAD_VALUES as u32);
6252 put_u32(
6253 &mut index,
6254 u32::try_from(blocks).map_err(|_| invalid("global dictionary has too many blocks"))?,
6255 );
6256 put_u32(&mut index, offset_bits as u32);
6257 encode_offsets(&dictionary.offsets, offset_bits, &mut index)?;
6258 let mut at = 0_u64;
6262 for block in &payload {
6263 at = at
6264 .checked_add(block.len() as u64)
6265 .ok_or_else(|| invalid("global dictionary payload overflow"))?;
6266 put_u64(&mut index, at);
6267 }
6268 for block in &payload {
6269 put_u64(&mut index, checksum(block));
6270 }
6271 if rank_ends.len() != rank_blocks {
6274 return Err(invalid("global dictionary order is not the blocks it says it is"));
6275 }
6276 for end in &rank_ends {
6277 put_u64(&mut index, *end);
6278 }
6279 let mut at = 0_usize;
6280 for end in &rank_ends {
6281 let end = usize::try_from(*end).map_err(|_| invalid("global dictionary order overflow"))?;
6282 put_u64(&mut index, checksum(&ranks[at..end]));
6283 at = end;
6284 }
6285 Ok(EncodedDictionary { index, ranks, payload })
6286}
6287
6288const PAYLOAD_SAMPLE_BLOCKS: usize = 8;
6295
6296fn payload_shapes() -> Vec<chooser::Settled> {
6322 let integers = vec![integer::Kind::Packed];
6323 [
6324 vec![string::Kind::Front, string::Kind::Lz],
6325 vec![string::Kind::Lz, string::Kind::Fsst],
6326 vec![string::Kind::Lz, string::Kind::Plain],
6327 vec![string::Kind::Fsst],
6328 vec![string::Kind::Plain],
6329 ]
6330 .into_iter()
6331 .map(|strings| chooser::Settled::new(strings, integers.clone()))
6332 .collect()
6333}
6334
6335fn encode_payload(dictionary: &GlobalDictionary) -> Result<Vec<Vec<u8>>> {
6341 let values = dictionary.offsets.len() - 1;
6342 let blocks = values.div_ceil(TEXT_PAYLOAD_VALUES);
6343 let run = |block: usize| {
6344 let first = block * TEXT_PAYLOAD_VALUES;
6345 let last = (first + TEXT_PAYLOAD_VALUES).min(values);
6346 (first..last)
6347 .map(|value| {
6348 let from = dictionary.offsets[value] as usize;
6349 let to = dictionary.offsets[value + 1] as usize;
6350 &dictionary.payload[from..to]
6351 })
6352 .collect::<Vec<_>>()
6353 };
6354 let shape = (blocks > PAYLOAD_SAMPLE_BLOCKS).then(|| settle_shape(&run, blocks)).transpose()?;
6357 let one = |block: usize| match &shape {
6358 Some(shape) => string::encode_with(&run(block), shape),
6359 None => string::encode(&run(block)),
6360 };
6361 let workers = std::thread::available_parallelism()
6362 .map_or(1, usize::from)
6363 .min(MAX_FREQUENCY_WORKERS)
6364 .min(blocks);
6365 if workers <= 1 {
6366 return (0..blocks).map(one).collect();
6367 }
6368 let next = AtomicUsize::new(0);
6369 let pieces = std::thread::scope(|scope| {
6370 (0..workers)
6371 .map(|_| {
6372 scope.spawn(|| {
6373 let mut mine = Vec::new();
6374 loop {
6375 let block = next.fetch_add(1, Atomic::Relaxed);
6376 if block >= blocks {
6377 break;
6378 }
6379 mine.push((block, one(block)?));
6380 }
6381 Ok(mine)
6382 })
6383 })
6384 .collect::<Vec<_>>()
6385 .into_iter()
6386 .map(|handle| {
6387 handle.join().map_err(|_| Error::internal("a dictionary encode worker panicked"))?
6388 })
6389 .collect::<Result<Vec<_>>>()
6390 })?;
6391 let mut payload = vec![Vec::new(); blocks];
6392 for piece in pieces {
6393 for (block, bytes) in piece {
6394 payload[block] = bytes;
6395 }
6396 }
6397 Ok(payload)
6398}
6399
6400fn settle_shape<'a>(
6408 run: &dyn Fn(usize) -> Vec<&'a [u8]>,
6409 blocks: usize,
6410) -> Result<chooser::Settled> {
6411 let last = blocks - 1;
6412 let sample = (0..PAYLOAD_SAMPLE_BLOCKS)
6413 .map(|region| run(region * last / (PAYLOAD_SAMPLE_BLOCKS - 1)))
6414 .collect::<Vec<_>>();
6415 let mut best: Option<(chooser::Settled, usize)> = None;
6416 for shape in payload_shapes() {
6417 let mut size = 0;
6418 for block in &sample {
6419 size += string::encode_with(block, &shape)?.len();
6420 }
6421 if best.as_ref().is_none_or(|(_, smallest)| size < *smallest) {
6422 best = Some((shape, size));
6423 }
6424 }
6425 best.map(|(shape, _)| shape)
6426 .ok_or_else(|| invalid("no shape applies to a global dictionary payload"))
6427}
6428
6429fn encode_ranks(order: &[(u64, u32)], code_bits: usize) -> Result<(Vec<u8>, Vec<u64>)> {
6436 let mut out = Vec::with_capacity(order.len() * 4);
6437 let mut ends = Vec::with_capacity(order.len().div_ceil(TEXT_RANK_BLOCK));
6438 let mut heads = Vec::with_capacity(TEXT_RANK_BLOCK);
6439 let mut codes = Vec::with_capacity(TEXT_RANK_BLOCK);
6440 for block in order.chunks(TEXT_RANK_BLOCK) {
6441 let base = block.first().map_or(0, |&(head, _)| head);
6444 let span = block.last().map_or(0, |&(head, _)| head.wrapping_sub(base));
6445 let width = (u64::BITS - span.leading_zeros()) as usize;
6446 heads.clear();
6447 codes.clear();
6448 for &(head, code) in block {
6449 heads.push(head.wrapping_sub(base));
6450 codes.push(u64::from(code));
6451 }
6452 put_u64(&mut out, base);
6453 out.push(width as u8);
6454 bitpack::pack_tail(&heads, width, &mut out)
6455 .map_err(|_| invalid("global dictionary heads do not pack"))?;
6456 bitpack::pack_tail(&codes, code_bits, &mut out)
6457 .map_err(|_| invalid("global dictionary codes do not pack"))?;
6458 ends.push(out.len() as u64);
6459 }
6460 Ok((out, ends))
6461}
6462
6463fn open_global_dictionary(
6470 file: Arc<File>,
6471 page: Page,
6472 ty: &LogicalType,
6473 keep_budget: usize,
6474) -> Result<Vector> {
6475 if ty != &LogicalType::Varchar {
6476 return Err(invalid("global dictionary belongs to a non-string column"));
6477 }
6478 let mut header = [0; DICTIONARY_HEADER];
6479 read_at(&file, page.offset, &mut header)?;
6480 let count = u32::from_le_bytes(header[0..4].try_into().expect("four bytes")) as usize;
6481 let per_block = u32::from_le_bytes(header[4..8].try_into().expect("four bytes")) as usize;
6482 let blocks = u32::from_le_bytes(header[8..12].try_into().expect("four bytes")) as usize;
6483 let offset_bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
6484 if per_block != TEXT_PAYLOAD_VALUES {
6485 return Err(invalid("global dictionary block width differs"));
6486 }
6487 if blocks != count.div_ceil(TEXT_PAYLOAD_VALUES) {
6488 return Err(invalid("global dictionary block count differs from its value count"));
6489 }
6490 if offset_bits > u32::BITS as usize {
6491 return Err(invalid("global dictionary packs offsets past a payload"));
6492 }
6493 let offset_len = offset_bytes(count, offset_bits);
6494 let ranks = count;
6499 let rank_blocks = ranks.div_ceil(TEXT_RANK_BLOCK);
6500 let hash_len = blocks
6503 .checked_add(rank_blocks)
6504 .and_then(|words| words.checked_mul(16))
6505 .ok_or_else(|| invalid("global dictionary block count overflow"))?;
6506 let index_len = DICTIONARY_HEADER
6507 .checked_add(offset_len)
6508 .and_then(|len| len.checked_add(hash_len))
6509 .ok_or_else(|| invalid("global dictionary header overflow"))?;
6510 if index_len > page.length as usize {
6511 return Err(invalid("global dictionary offset index exceeds its page"));
6512 }
6513 let mut index = vec![0; index_len];
6514 index[..DICTIONARY_HEADER].copy_from_slice(&header);
6515 read_at(&file, page.offset + DICTIONARY_HEADER as u64, &mut index[DICTIONARY_HEADER..])?;
6516 if checksum(&index) != page.hash {
6517 return Err(invalid("global dictionary index checksum differs"));
6518 }
6519 let offsets = index[DICTIONARY_HEADER..DICTIONARY_HEADER + offset_len].to_vec();
6520 let mut words = index[DICTIONARY_HEADER + offset_len..]
6521 .chunks_exact(8)
6522 .map(|part| u64::from_le_bytes(part.try_into().expect("eight bytes")))
6523 .collect::<Vec<_>>();
6524 let mut hashes = words.split_off(blocks);
6525 let mut rank_ends = hashes.split_off(blocks);
6526 let rank_hashes = rank_ends.split_off(rank_blocks);
6527 let ends = words;
6528 if rank_ends.windows(2).any(|pair| pair[0] >= pair[1]) {
6531 return Err(invalid("global dictionary order blocks do not rise"));
6532 }
6533 let rank_len = usize::try_from(rank_ends.last().copied().unwrap_or_default())
6534 .map_err(|_| invalid("global dictionary rank overflow"))?;
6535 let body_len = index_len
6536 .checked_add(rank_len)
6537 .ok_or_else(|| invalid("global dictionary header overflow"))?;
6538 if body_len > page.length as usize {
6539 return Err(invalid("global dictionary order exceeds its page"));
6540 }
6541 let stored_len = page.length as usize - body_len;
6544 if ends.last().copied().unwrap_or_default() as usize != stored_len
6545 || ends.windows(2).any(|pair| pair[0] > pair[1])
6546 {
6547 return Err(invalid("global dictionary blocks do not bound the payload"));
6548 }
6549 Vector::external_text(
6550 LogicalType::Varchar,
6551 Arc::new(NativeText {
6552 file,
6553 values: count,
6554 offsets,
6555 offset_bits,
6556 ranks,
6557 rank_at: page.offset + index_len as u64,
6558 rank_ends,
6559 rank_hashes,
6560 rank_blocks: (0..rank_blocks).map(|_| OnceLock::new()).collect(),
6561 code_bits: code_width(count),
6562 code_ranks: OnceLock::new(),
6563 payload: page.offset + body_len as u64,
6564 ends,
6565 hashes,
6566 blocks: (0..blocks).map(|_| OnceLock::new()).collect(),
6567 keep_budget,
6568 payload_kept: AtomicUsize::new(0),
6569 searched: Mutex::new(HashMap::new()),
6570 }),
6571 )
6572}
6573
6574fn page_encoding(ty: &LogicalType, rows: usize, bytes: &[u8]) -> String {
6587 fn cascade_at(rows: usize, bytes: &[u8]) -> Result<(u8, usize)> {
6589 let mut cur = Cursor { bytes, at: 0 };
6590 let codec = cur.u8()?;
6591 if cur.u8()? == 2 {
6592 cur.take(rows.div_ceil(8))?;
6593 }
6594 Ok((codec, cur.at))
6595 }
6596 let Ok((codec, at)) = cascade_at(rows, bytes) else {
6597 return "UNREADABLE".to_string();
6598 };
6599 let tail = &bytes[at..];
6600 let described = |described: Result<String>| described.unwrap_or_else(|_| "UNREADABLE".into());
6601 match codec {
6602 0 => match ty {
6603 LogicalType::Varchar | LogicalType::Blob => "PLAIN".to_string(),
6604 _ => "FIXED".to_string(),
6605 },
6606 1 => "DICT(PLAIN)".to_string(),
6607 2 => "FOR+BITPACK".to_string(),
6608 3 => "TABLE DICT".to_string(),
6609 4 => format!("TABLE DICT({})", described(integer::describe(tail))),
6610 5 => described(integer::describe(tail)),
6611 6 => described(string::describe(tail)),
6612 other => format!("CODEC {other}"),
6613 }
6614}
6615
6616fn decode(
6617 ty: &LogicalType,
6618 rows: usize,
6619 bytes: &[u8],
6620 global: Option<Arc<Vector>>,
6621) -> Result<Vector> {
6622 let mut cur = Cursor { bytes, at: 0 };
6623 let codec = cur.u8()?;
6624 let flag = cur.u8()?;
6625 let validity = match flag {
6626 0 => Validity::AllValid,
6627 1 => Validity::AllInvalid,
6628 2 => {
6629 let mask = cur.take(rows.div_ceil(8))?;
6630 Validity::from_iter(rows, |row| mask[row / 8] >> (row % 8) & 1 == 1)
6631 }
6632 _ => return Err(invalid("page validity tag differs")),
6633 };
6634 if codec == 1 {
6635 if ty != &LogicalType::Varchar {
6636 return Err(invalid("dictionary codec belongs to a non-string page"));
6637 }
6638 let count = cur.u32()? as usize;
6639 let payload_len = cur.u32()? as usize;
6640 let offset_bytes = cur.take(
6641 (count + 1)
6642 .checked_mul(4)
6643 .ok_or_else(|| invalid("dictionary offset count overflow"))?,
6644 )?;
6645 let offsets = offset_bytes
6646 .chunks_exact(4)
6647 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6648 .collect::<Vec<_>>();
6649 let payload = cur.take(payload_len)?.to_vec();
6650 if offsets.first() != Some(&0)
6651 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6652 || offsets.windows(2).any(|pair| pair[0] > pair[1])
6653 {
6654 return Err(invalid("dictionary offsets do not bound the payload"));
6655 }
6656 let mut strings = StringColumn::over(Buffer::from_vec(payload).into_page());
6659 for pair in offsets.windows(2) {
6660 strings.push_in_place(pair[0] as usize, (pair[1] - pair[0]) as usize)?;
6661 }
6662 let mut codes = Vec::with_capacity(rows);
6663 for _ in 0..rows {
6664 codes.push(cur.u32()?);
6665 }
6666 if codes.iter().any(|code| *code as usize >= count) {
6667 return Err(invalid("dictionary code is out of range"));
6668 }
6669 if cur.at != bytes.len() {
6670 return Err(invalid("dictionary page has trailing bytes"));
6671 }
6672 let dictionary = Vector::flat(LogicalType::Varchar, Data::Varlen(strings))?;
6673 return Ok(Vector::dictionary(codes, dictionary)?.with_validity(validity));
6674 }
6675 if codec == 3 || codec == 4 {
6676 let dictionary = global.ok_or_else(|| invalid("global code page has no dictionary"))?;
6677 let codes = if codec == 4 {
6678 let wide = integer::decode(&bytes[cur.at..])?;
6681 if wide.len() != rows {
6682 return Err(invalid("encoded code page holds the wrong number of rows"));
6683 }
6684 let mut codes = Vec::with_capacity(wide.len());
6691 let mut seen = 0_i64;
6692 for &code in &wide {
6693 seen |= code;
6694 codes.push(code as u32);
6695 }
6696 if seen < 0 || seen > i64::from(u32::MAX) {
6697 return Err(invalid("code is not a code"));
6698 }
6699 codes
6700 } else {
6701 let mut codes = Vec::with_capacity(rows);
6702 for _ in 0..rows {
6703 codes.push(cur.u32()?);
6704 }
6705 if cur.at != bytes.len() {
6706 return Err(invalid("global code page has trailing bytes"));
6707 }
6708 codes
6709 };
6710 let highest = codes.iter().copied().max();
6711 return Ok(Vector::stable_dictionary_validated(codes, dictionary, highest)?
6712 .with_validity(validity));
6713 }
6714 if codec == 6 {
6715 if ty != &LogicalType::Varchar {
6716 return Err(invalid("compressed text codec belongs to a non-string page"));
6717 }
6718 let (payload, ends) = string::decode_flat(&bytes[cur.at..])?.into_parts();
6722 if ends.len() != rows {
6723 return Err(invalid("compressed text page holds the wrong number of rows"));
6724 }
6725 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6728 let mut start = 0;
6729 for end in ends {
6730 let len = end
6731 .checked_sub(start)
6732 .ok_or_else(|| invalid("compressed text value ends before it starts"))?;
6733 values.push_in_place(start, len)?;
6734 start = end;
6735 }
6736 return Ok(Vector::flat(ty.clone(), Data::Varlen(values))?.with_validity(validity));
6737 }
6738 if codec == 5 {
6739 let values = integer::decode(&bytes[cur.at..])?;
6741 if values.len() != rows {
6742 return Err(invalid("cascade page holds the wrong number of rows"));
6743 }
6744 let data = narrowed(ty, values)?;
6745 return Ok(Vector::flat(ty.clone(), data)?.with_validity(validity));
6746 }
6747 if codec == 2 {
6748 let width = u32::from(cur.u8()?);
6749 let base = i128::from_le_bytes(cur.take(16)?.try_into().expect("sixteen bytes"));
6750 let count = cur.u32()? as usize;
6751 let mut words = Vec::with_capacity(count);
6752 for _ in 0..count {
6753 words.push(cur.u64()?);
6754 }
6755 if cur.at != bytes.len() {
6756 return Err(invalid("packed page has trailing bytes"));
6757 }
6758 return Ok(Vector::packed(ty.clone(), words, width, base, rows)?.with_validity(validity));
6759 }
6760 if codec != 0 {
6761 return Err(invalid("page codec is unknown"));
6762 }
6763 let data = match ty {
6764 LogicalType::TinyInt => {
6765 let values = cur.take(rows)?;
6766 Data::Int8(values.iter().map(|item| *item as i8).collect::<Vec<_>>().into())
6767 }
6768 LogicalType::UTinyInt => Data::UInt8(cur.take(rows)?.to_vec().into()),
6769 LogicalType::SmallInt => {
6770 let values =
6771 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6772 Data::Int16(
6773 values
6774 .chunks_exact(2)
6775 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6776 .collect::<Vec<_>>()
6777 .into(),
6778 )
6779 }
6780 LogicalType::USmallInt => {
6781 let values =
6782 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6783 Data::UInt16(
6784 values
6785 .chunks_exact(2)
6786 .map(|item| u16::from_le_bytes(item.try_into().expect("two bytes")))
6787 .collect::<Vec<_>>()
6788 .into(),
6789 )
6790 }
6791 LogicalType::UInteger => {
6792 let values =
6793 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6794 Data::UInt32(
6795 values
6796 .chunks_exact(4)
6797 .map(|item| u32::from_le_bytes(item.try_into().expect("four bytes")))
6798 .collect::<Vec<_>>()
6799 .into(),
6800 )
6801 }
6802 LogicalType::UBigInt => {
6803 let values =
6804 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6805 Data::UInt64(
6806 values
6807 .chunks_exact(8)
6808 .map(|item| u64::from_le_bytes(item.try_into().expect("eight bytes")))
6809 .collect::<Vec<_>>()
6810 .into(),
6811 )
6812 }
6813 LogicalType::Integer | LogicalType::Date => {
6814 let values =
6815 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6816 Data::Int32(
6817 values
6818 .chunks_exact(4)
6819 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6820 .collect::<Vec<_>>()
6821 .into(),
6822 )
6823 }
6824 LogicalType::BigInt
6825 | LogicalType::Timestamp
6826 | LogicalType::Time
6827 | LogicalType::TimeTz
6828 | LogicalType::TimestampTz
6829 | LogicalType::TimestampS
6830 | LogicalType::TimestampMs
6831 | LogicalType::TimestampNs => {
6832 let values =
6833 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6834 Data::Int64(
6835 values
6836 .chunks_exact(8)
6837 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6838 .collect::<Vec<_>>()
6839 .into(),
6840 )
6841 }
6842 LogicalType::HugeInt | LogicalType::Uuid => {
6843 let values =
6844 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6845 Data::Int128(
6846 values
6847 .chunks_exact(16)
6848 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6849 .collect::<Vec<_>>()
6850 .into(),
6851 )
6852 }
6853 LogicalType::UHugeInt => {
6854 let values =
6855 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6856 Data::UInt128(
6857 values
6858 .chunks_exact(16)
6859 .map(|item| u128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6860 .collect::<Vec<_>>()
6861 .into(),
6862 )
6863 }
6864 LogicalType::Float => {
6865 let values =
6866 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6867 Data::Float32(
6868 values
6869 .chunks_exact(4)
6870 .map(|item| f32::from_le_bytes(item.try_into().expect("four bytes")))
6871 .collect::<Vec<_>>()
6872 .into(),
6873 )
6874 }
6875 LogicalType::Double => {
6876 let values =
6877 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6878 Data::Float64(
6879 values
6880 .chunks_exact(8)
6881 .map(|item| f64::from_le_bytes(item.try_into().expect("eight bytes")))
6882 .collect::<Vec<_>>()
6883 .into(),
6884 )
6885 }
6886 LogicalType::Interval => {
6887 let values =
6888 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6889 Data::Interval(
6890 values
6891 .chunks_exact(16)
6892 .map(|item| {
6893 (
6894 i32::from_le_bytes(item[..4].try_into().expect("four bytes")),
6895 i32::from_le_bytes(item[4..8].try_into().expect("four bytes")),
6896 i64::from_le_bytes(item[8..].try_into().expect("eight bytes")),
6897 )
6898 })
6899 .collect::<Vec<_>>()
6900 .into(),
6901 )
6902 }
6903 LogicalType::Boolean => {
6904 let values = cur.take(rows)?;
6905 if values.iter().any(|value| *value > 1) {
6906 return Err(invalid("boolean page has another value"));
6907 }
6908 Data::Bool(values.iter().map(|value| *value == 1).collect::<Vec<_>>().into())
6909 }
6910 LogicalType::Decimal { .. } => match ty.physical() {
6913 PhysicalType::Int16 => {
6914 let values =
6915 cur.take(rows.checked_mul(2).ok_or_else(|| invalid("page size overflow"))?)?;
6916 Data::Int16(
6917 values
6918 .chunks_exact(2)
6919 .map(|item| i16::from_le_bytes(item.try_into().expect("two bytes")))
6920 .collect::<Vec<_>>()
6921 .into(),
6922 )
6923 }
6924 PhysicalType::Int32 => {
6925 let values =
6926 cur.take(rows.checked_mul(4).ok_or_else(|| invalid("page size overflow"))?)?;
6927 Data::Int32(
6928 values
6929 .chunks_exact(4)
6930 .map(|item| i32::from_le_bytes(item.try_into().expect("four bytes")))
6931 .collect::<Vec<_>>()
6932 .into(),
6933 )
6934 }
6935 PhysicalType::Int64 => {
6936 let values =
6937 cur.take(rows.checked_mul(8).ok_or_else(|| invalid("page size overflow"))?)?;
6938 Data::Int64(
6939 values
6940 .chunks_exact(8)
6941 .map(|item| i64::from_le_bytes(item.try_into().expect("eight bytes")))
6942 .collect::<Vec<_>>()
6943 .into(),
6944 )
6945 }
6946 _ => {
6947 let values =
6948 cur.take(rows.checked_mul(16).ok_or_else(|| invalid("page size overflow"))?)?;
6949 Data::Int128(
6950 values
6951 .chunks_exact(16)
6952 .map(|item| i128::from_le_bytes(item.try_into().expect("sixteen bytes")))
6953 .collect::<Vec<_>>()
6954 .into(),
6955 )
6956 }
6957 },
6958 LogicalType::Varchar | LogicalType::Blob | LogicalType::Bit => {
6959 let offset_bytes = cur
6960 .take((rows + 1).checked_mul(4).ok_or_else(|| invalid("offset count overflow"))?)?;
6961 let offsets = offset_bytes
6962 .chunks_exact(4)
6963 .map(|part| u32::from_le_bytes(part.try_into().expect("four bytes")))
6964 .collect::<Vec<_>>();
6965 let payload = cur.take(bytes.len() - cur.at)?.to_vec();
6966 if offsets.first() != Some(&0)
6967 || offsets.last().copied().map(|last| last as usize) != Some(payload.len())
6968 || offsets.windows(2).any(|pair| pair[0] > pair[1])
6969 {
6970 return Err(invalid("string offsets do not bound the payload"));
6971 }
6972 let mut values = StringColumn::over(Buffer::from_vec(payload).into_page());
6980 let text = ty == &LogicalType::Varchar;
6981 for pair in offsets.windows(2) {
6982 let (at, len) = (pair[0] as usize, (pair[1] - pair[0]) as usize);
6983 if text {
6984 values.push_in_place(at, len)?;
6985 } else {
6986 values.push_bytes_in_place(at, len)?;
6987 }
6988 }
6989 Data::Varlen(values)
6990 }
6991 _ => return Err(Error::not_implemented(format!("native page for {ty}"))),
6992 };
6993 if cur.at != bytes.len() {
6994 return Err(invalid("page has trailing bytes"));
6995 }
6996 Ok(Vector::flat(ty.clone(), data)?.with_validity(validity))
6997}
6998
6999#[cfg(test)]
7000mod tests {
7001 use std::fs;
7002 use std::io::{Seek, SeekFrom, Write};
7003 use std::path::PathBuf;
7004 use std::time::{SystemTime, UNIX_EPOCH};
7005
7006 use rudb_common::Stat;
7007 use rudb_common::Value;
7008 use rudb_common::bounds::{Frequencies, Op, Remainder, Zones};
7009 use rudb_common::stat::Provenance;
7010
7011 use super::*;
7012
7013 #[test]
7014 fn checksum_matches_fixed_vectors() {
7015 assert_eq!(checksum(b""), 0xef46_db37_51d8_e999);
7016 assert_eq!(checksum(b"a"), 0xd24e_c4f1_a98c_6e5b);
7017 assert_eq!(checksum(b"abc"), 0x44bc_2cf5_ad77_0999);
7018 }
7019
7020 fn path(label: &str) -> PathBuf {
7021 let stamp = SystemTime::now().duration_since(UNIX_EPOCH).expect("time advances").as_nanos();
7022 std::env::temp_dir().join(format!("rudb-native-{label}-{}-{stamp}.rdb", std::process::id()))
7023 }
7024
7025 #[test]
7027 fn a_read_at_an_offset_ignores_where_another_thread_left_the_cursor() {
7028 const SPANS: usize = 64;
7029 const SPAN: usize = 512;
7030 let path = path("positional");
7031 let content: Vec<u8> =
7032 (0..SPANS).flat_map(|span| std::iter::repeat_n(span as u8, SPAN)).collect();
7033 fs::write(&path, &content).expect("the file is written");
7034 let file = Arc::new(File::open(&path).expect("the file opens"));
7035 std::thread::scope(|scope| {
7036 for _ in 0..8 {
7037 let file = Arc::clone(&file);
7038 scope.spawn(move || {
7039 for _ in 0..64 {
7040 for span in 0..SPANS {
7041 let mut bytes = [0_u8; SPAN];
7042 read_at(&file, (span * SPAN) as u64, &mut bytes)
7043 .expect("the span reads");
7044 assert!(
7045 bytes.iter().all(|byte| *byte == span as u8),
7046 "span {span} came back as {}",
7047 bytes[0],
7048 );
7049 }
7050 }
7051 });
7052 }
7053 });
7054 let mut past = [0_u8; SPAN];
7055 let end = (SPANS * SPAN) as u64;
7056 let error = read_at(&file, end, &mut past).expect_err("a read past the end is refused");
7057 assert!(error.message().contains("ends before its declared length"), "{error}");
7058 drop(file);
7059 let _ = fs::remove_file(&path);
7060 }
7061
7062 #[test]
7068 fn a_writer_puts_a_page_where_it_said_it_did_wherever_the_cursor_has_got_to() {
7069 let path = path("cursor");
7070 let mut writer = Writer::create(
7071 &path,
7072 "items",
7073 vec![
7074 Field::required("id", LogicalType::Integer),
7075 Field::new("text", LogicalType::Varchar),
7076 ],
7077 )
7078 .expect("new file");
7079 writer.append(&sample()).expect("first part");
7080 writer.file.seek(SeekFrom::Start(0)).expect("the cursor goes back to the header");
7081 writer.append(&sample()).expect("second part");
7082 writer.file.seek(SeekFrom::Start(1)).expect("and somewhere useless again");
7083 writer.finish().expect("commit");
7084 let reader = Reader::open(&path).expect("reopen from disk");
7085 assert_eq!(reader.table().rows(), 6);
7086 let ids = reader.read(0, &[0]).expect("the integer page reads back");
7087 assert_eq!(ids.value_at(0, 0), Value::Integer(4));
7088 assert_eq!(ids.value_at(2, 0), Value::Integer(-2));
7089 let text = reader.read(1, &[1]).expect("the text page reads back");
7090 assert_eq!(text.value_at(1, 0), Value::Null);
7091 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7092 let end = reader.table().stripes().iter().flat_map(|stripe| {
7095 stripe
7096 .pages
7097 .iter()
7098 .map(|page| page.offset + u64::from(page.length))
7099 .chain(std::iter::once(stripe.index.offset + u64::from(stripe.index.length)))
7100 });
7101 let last = end.fold(HEADER, u64::max);
7102 let directory = fs::metadata(&path).expect("the file is there").len();
7103 assert!(last <= directory, "a page runs to {last} in a file of {directory} bytes");
7104 fs::remove_file(path).expect("remove scratch file");
7105 }
7106
7107 fn dictionary_index_len(header: &[u8; DICTIONARY_HEADER]) -> u64 {
7113 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7114 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7115 let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7116 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7117 DICTIONARY_HEADER as u64
7118 + offset_bytes(count as usize, bits) as u64
7119 + (blocks + rank_blocks) * 16
7120 }
7121
7122 fn last_rank_end(file: &File, offset: u64, header: &[u8; DICTIONARY_HEADER]) -> u64 {
7124 let count = u64::from(u32::from_le_bytes(header[0..4].try_into().expect("four bytes")));
7125 let blocks = u64::from(u32::from_le_bytes(header[8..12].try_into().expect("four bytes")));
7126 let bits = u32::from_le_bytes(header[12..16].try_into().expect("four bytes")) as usize;
7127 let rank_blocks = count.div_ceil(TEXT_RANK_BLOCK as u64);
7128 let at = offset
7129 + DICTIONARY_HEADER as u64
7130 + offset_bytes(count as usize, bits) as u64
7131 + blocks * 16
7132 + (rank_blocks - 1) * 8;
7133 let mut end = [0; 8];
7134 read_at(file, at, &mut end).expect("the last rank block end");
7135 u64::from_le_bytes(end)
7136 }
7137
7138 fn sample() -> Chunk {
7139 Chunk::new(vec![
7140 Vector::from_values(
7141 LogicalType::Integer,
7142 &[Value::Integer(4), Value::Integer(9), Value::Integer(-2)],
7143 )
7144 .expect("integers"),
7145 Vector::from_values(
7146 LogicalType::Varchar,
7147 &[
7148 Value::Varchar("alpha".into()),
7149 Value::Null,
7150 Value::Varchar("long text after a slash".into()),
7151 ],
7152 )
7153 .expect("strings"),
7154 ])
7155 .expect("matching rows")
7156 }
7157
7158 fn sample_ids() -> Chunk {
7159 Chunk::new(vec![
7160 Vector::flat(LogicalType::Integer, Data::Int32(vec![7, 8, 9].into()))
7161 .expect("integers"),
7162 ])
7163 .expect("one column")
7164 }
7165
7166 #[test]
7167 fn the_planner_gets_the_null_count_off_the_same_directory_the_bounds_are_in() {
7168 let path = path("nulls_for_the_planner");
7171 let mut writer =
7172 Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
7173 .expect("new file");
7174 let rows = Chunk::new(vec![
7175 Vector::from_values(
7176 LogicalType::Integer,
7177 &[
7178 Value::Integer(4),
7179 Value::Null,
7180 Value::Integer(9),
7181 Value::Null,
7182 Value::Integer(1),
7183 Value::Integer(2),
7184 ],
7185 )
7186 .expect("integers"),
7187 ])
7188 .expect("one column");
7189 writer.append(&rows).expect("the only part");
7190 writer.finish().expect("commit");
7191 let reader = Reader::open(&path).expect("reopen from disk");
7192 let stripes = Stripes::new(reader);
7193 let column = stripes.column("a").expect("the file has that column");
7194 assert_eq!(stripes.nulls(column), Stat::exact(2, Provenance::NullCount));
7195 assert_eq!(stripes.nulls(column + 1), Stat::Unknown);
7198 fs::remove_file(&path).expect("clean up");
7199 }
7200
7201 #[test]
7202 fn the_planner_gets_a_row_count_per_value_off_a_complete_synopsis() {
7203 let path = path("frequencies_for_the_planner");
7208 let mut writer =
7209 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7210 .expect("new file");
7211 let rows = Chunk::new(vec![
7212 Vector::from_values(
7213 LogicalType::Integer,
7214 &[
7215 Value::Integer(4),
7216 Value::Integer(4),
7217 Value::Integer(4),
7218 Value::Integer(9),
7219 Value::Integer(9),
7220 Value::Integer(1),
7221 ],
7222 )
7223 .expect("integers"),
7224 ])
7225 .expect("one column");
7226 writer.append(&rows).expect("the only part");
7227 writer.finish().expect("commit");
7228 let reader = Reader::open(&path).expect("reopen from disk");
7229 let common = Common::new(reader);
7230 assert_eq!(common.rows(), 6);
7231 let column = common.column("id").expect("the file has that column");
7232 assert_eq!(common.column("nothing"), None);
7233 assert_eq!(
7234 common.rows_with(column, &Bound::Int(4)),
7235 Stat::exact(3, Provenance::FrequencySynopsis)
7236 );
7237 assert_eq!(
7239 common.rows_with(column, &Bound::Int(7)),
7240 Stat::exact(0, Provenance::FrequencySynopsis)
7241 );
7242 assert_eq!(common.rows_with(column, &Bound::Bytes(b"four".to_vec())), Stat::Unknown);
7245 assert_eq!(common.remainder(column), None);
7248 fs::remove_file(&path).expect("clean up");
7249 }
7250
7251 fn bare_table(sections: Vec<Section>) -> Table {
7256 Table {
7257 name: "linked".to_owned(),
7258 fields: vec![Field::required("id", LogicalType::Integer)],
7259 stripes: Vec::new(),
7260 rows: 0,
7261 dictionaries: vec![None],
7262 distincts: vec![None],
7263 frequencies: vec![None],
7264 clustering: None,
7265 generation: 1,
7266 sections,
7267 }
7268 }
7269
7270 fn a_key_map_section() -> Section {
7271 Section {
7272 kind: *section::KEY_MAP,
7273 id: 1,
7274 generation: 3,
7275 extents: 1,
7276 extent_page: HEADER,
7277 extent_bytes: section::EXTENT_BYTES as u32,
7278 hash: 0x1234_5678_9abc_def0,
7279 flags: 0,
7280 header_bytes: 24,
7281 }
7282 }
7283
7284 #[test]
7285 fn a_section_table_round_trips_through_a_directory() {
7286 let mut later = a_key_map_section();
7287 later.kind = *b"RUDBZZ9\0";
7288 later.id = 2;
7289 let table = bare_table(vec![a_key_map_section(), later]);
7290 let directory = encode_directory(&table).expect("directory");
7291 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7292 assert_eq!(decoded.sections(), &[a_key_map_section(), later]);
7293 assert!(decoded.sections()[0].known());
7297 assert!(!decoded.sections()[1].known());
7298 }
7299
7300 #[test]
7301 fn a_directory_written_before_the_section_table_reads_as_a_table_with_none() {
7302 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7306 let block = SECTIONS.len() + size_of::<u64>() + size_of::<u16>();
7307 let older = &directory[..directory.len() - block];
7308 let decoded = decode_directory(older, 1 << 20).expect("a directory from before sections");
7309 assert!(decoded.sections().is_empty());
7310 assert_eq!(decoded.generation(), 0, "a format 22 table recorded no generation");
7311 assert_eq!(decoded.name(), "linked");
7312 assert_eq!(decoded.fields().len(), 1, "everything before the block still decodes");
7313 }
7314
7315 #[test]
7316 fn a_file_stamped_with_the_previous_format_still_opens_and_reads() {
7317 let path = path("format_twenty_two");
7324 let mut writer =
7325 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7326 .expect("new file");
7327 let rows = Chunk::new(vec![
7328 Vector::from_values(
7329 LogicalType::Integer,
7330 &[Value::Integer(1), Value::Integer(2), Value::Integer(3)],
7331 )
7332 .expect("integers"),
7333 ])
7334 .expect("one column");
7335 writer.append(&rows).expect("the only part");
7336 writer.finish().expect("commit");
7337
7338 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7339 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7340 drop(file);
7341
7342 let reader = Reader::open(&path).expect("a format 22 file opens unchanged");
7343 assert_eq!(reader.table().rows(), 3);
7344 assert!(reader.table().sections().is_empty());
7345
7346 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7349 write_at(&file, 8, &21_u32.to_le_bytes()).expect("stamp an unreadable format");
7350 drop(file);
7351 let error = Reader::open(&path).expect_err("format 21 is not readable");
7352 assert!(error.to_string().contains("format 21"), "{error}");
7353
7354 fs::remove_file(&path).expect("clean up");
7355 }
7356
7357 #[test]
7358 fn a_section_whose_extent_table_is_outside_the_file_is_refused() {
7359 let mut past = a_key_map_section();
7364 past.extent_page = 1 << 30;
7365 let directory = encode_directory(&bare_table(vec![past])).expect("directory");
7366 let error = decode_directory(&directory, 1 << 20).expect_err("refused");
7367 assert!(error.to_string().contains("outside the file"), "{error}");
7368
7369 let mut inside_the_header = a_key_map_section();
7370 inside_the_header.extent_page = 8;
7371 let directory = encode_directory(&bare_table(vec![inside_the_header])).expect("directory");
7372 assert!(
7373 decode_directory(&directory, 1 << 20).is_err(),
7374 "a section may not overlap a header"
7375 );
7376 }
7377
7378 #[test]
7379 fn a_section_recorded_as_not_built_is_legal_and_names_no_bytes() {
7380 let not_built = Section {
7384 kind: *section::FORWARD_LINK,
7385 id: 9,
7386 generation: 3,
7387 extents: 0,
7388 extent_page: 0,
7389 extent_bytes: 0,
7390 hash: 0,
7391 flags: 0,
7392 header_bytes: 0,
7393 };
7394 let directory = encode_directory(&bare_table(vec![not_built])).expect("directory");
7395 let decoded = decode_directory(&directory, 1 << 20).expect("reopen");
7396 assert_eq!(decoded.sections(), &[not_built]);
7397
7398 let mut incoherent = not_built;
7401 incoherent.extent_bytes = 28;
7402 incoherent.extent_page = HEADER;
7403 let directory = encode_directory(&bare_table(vec![incoherent])).expect("directory");
7404 assert!(decode_directory(&directory, 1 << 20).is_err());
7405 }
7406
7407 #[test]
7408 fn a_directory_naming_more_sections_than_the_bound_is_refused() {
7409 let directory = encode_directory(&bare_table(Vec::new())).expect("directory");
7410 let mut torn = directory.clone();
7411 let count_at = torn.len() - size_of::<u16>();
7412 torn[count_at..].copy_from_slice(&u16::MAX.to_le_bytes());
7413 assert!(decode_directory(&torn, 1 << 20).is_err());
7416 }
7417
7418 fn linked_file(label: &str, rows: i32) -> PathBuf {
7420 let path = path(label);
7421 let mut writer =
7422 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7423 .expect("new file");
7424 let values = (0..rows).map(Value::Integer).collect::<Vec<_>>();
7425 let chunk =
7426 Chunk::new(vec![Vector::from_values(LogicalType::Integer, &values).expect("integers")])
7427 .expect("one column");
7428 writer.append(&chunk).expect("the only part");
7429 writer.finish().expect("commit");
7430 path
7431 }
7432
7433 fn a_key_map_payload() -> Vec<u8> {
7434 (0..512_u32).flat_map(u32::to_le_bytes).collect()
7437 }
7438
7439 #[test]
7440 fn a_section_attached_to_a_committed_file_reads_back_byte_for_byte() {
7441 let path = linked_file("attach", 64);
7442 let payload = a_key_map_payload();
7443 let table = attach(
7444 &path,
7445 "items",
7446 &[section::Attachment {
7447 kind: *section::KEY_MAP,
7448 id: 0,
7449 flags: 2,
7450 header_bytes: 40,
7451 bytes: &payload,
7452 }],
7453 )
7454 .expect("attach a key map");
7455 assert_eq!(table.sections().len(), 1);
7456
7457 let reader = Reader::open(&path).expect("reopen after the attach");
7458 let held = reader.table().sections();
7459 assert_eq!(held.len(), 1);
7460 assert_eq!(held[0].kind, *section::KEY_MAP);
7461 assert_eq!(held[0].flags, 2, "the form a reader must not have to guess");
7462 assert_eq!(held[0].header_bytes, 40);
7463 assert_eq!(held[0].generation, 1);
7467 assert!(held[0].usable(reader.table().generation()));
7468 assert_eq!(reader.payload(&held[0]).expect("read the payload"), payload);
7469 assert_eq!(reader.extents(&held[0]).expect("extent table").len(), 1);
7470
7471 fs::remove_file(&path).expect("clean up");
7472 }
7473
7474 #[test]
7475 fn attaching_a_section_answers_every_row_exactly_as_before() {
7476 let path = linked_file("attach_changes_nothing", 300);
7481 let before = Reader::open(&path).expect("open before");
7482 let rows = before.table().rows();
7483 let first = before.read(0, &[0]).expect("read before");
7484 let values = (0..rows).map(|at| first.value_at(at, 0)).collect::<Vec<_>>();
7485 let layout = before.layout().columns_total();
7486 drop(before);
7487
7488 let payload = a_key_map_payload();
7489 attach(
7490 &path,
7491 "items",
7492 &[section::Attachment {
7493 kind: *section::KEY_MAP,
7494 id: 0,
7495 flags: 0,
7496 header_bytes: 0,
7497 bytes: &payload,
7498 }],
7499 )
7500 .expect("attach");
7501
7502 let after = Reader::open(&path).expect("open after");
7503 assert_eq!(after.table().rows(), rows);
7504 let read = after.read(0, &[0]).expect("read after");
7505 for (at, value) in values.iter().enumerate() {
7506 assert_eq!(&read.value_at(at, 0), value, "row {at} moved");
7507 }
7508 assert_eq!(
7509 after.layout().columns_total(),
7510 layout,
7511 "an attach appends and does not rewrite a column page"
7512 );
7513
7514 fs::remove_file(&path).expect("clean up");
7515 }
7516
7517 #[test]
7518 fn a_rebuilt_section_replaces_the_one_it_supersedes() {
7519 let path = linked_file("attach_twice", 32);
7523 let one = a_key_map_payload();
7524 let two = vec![7_u8; 1024];
7525 let entry = |bytes| section::Attachment {
7526 kind: *section::KEY_MAP,
7527 id: 4,
7528 flags: 1,
7529 header_bytes: 0,
7530 bytes,
7531 };
7532 attach(&path, "items", &[entry(&one)]).expect("first build");
7533 attach(&path, "items", &[entry(&two)]).expect("rebuild");
7534
7535 let reader = Reader::open(&path).expect("reopen");
7536 let held = reader.table().sections();
7537 assert_eq!(held.len(), 1, "one map per column and not one per build");
7538 assert_eq!(reader.payload(&held[0]).expect("payload"), two);
7539
7540 fs::remove_file(&path).expect("clean up");
7541 }
7542
7543 #[test]
7544 fn an_attach_carries_through_a_kind_it_does_not_know() {
7545 let path = linked_file("attach_unknown", 16);
7549 let payload = vec![3_u8; 96];
7550 attach(
7551 &path,
7552 "items",
7553 &[section::Attachment {
7554 kind: *b"RUDBZZ9\0",
7555 id: 1,
7556 flags: 0,
7557 header_bytes: 0,
7558 bytes: &payload,
7559 }],
7560 )
7561 .expect("a kind this build does not know still writes");
7562 let key_map = a_key_map_payload();
7563 attach(
7564 &path,
7565 "items",
7566 &[section::Attachment {
7567 kind: *section::KEY_MAP,
7568 id: 0,
7569 flags: 0,
7570 header_bytes: 0,
7571 bytes: &key_map,
7572 }],
7573 )
7574 .expect("attach beside it");
7575
7576 let reader = Reader::open(&path).expect("reopen");
7577 let held = reader.table().sections();
7578 assert_eq!(held.len(), 2, "the unfamiliar entry survived a directory rewrite");
7579 let unknown = held.iter().find(|one| !one.known()).expect("the unfamiliar one");
7580 assert_eq!(reader.payload(unknown).expect("its bytes are still there"), payload);
7581
7582 fs::remove_file(&path).expect("clean up");
7583 }
7584
7585 #[test]
7586 fn a_payload_of_nothing_is_a_relationship_recorded_as_not_built() {
7587 let path = linked_file("attach_not_built", 8);
7588 attach(
7589 &path,
7590 "items",
7591 &[section::Attachment {
7592 kind: *section::FORWARD_LINK,
7593 id: 2,
7594 flags: 0,
7595 header_bytes: 0,
7596 bytes: &[],
7597 }],
7598 )
7599 .expect("record a link that did not fit the budget");
7600
7601 let reader = Reader::open(&path).expect("reopen");
7602 let held = reader.table().sections();
7603 assert_eq!(held.len(), 1);
7604 assert_eq!(held[0].extents, 0);
7605 assert_eq!(held[0].extent_page, 0, "an entry that names no bytes points at none");
7606 assert!(reader.extents(&held[0]).expect("no extent table").is_empty());
7607 assert!(reader.payload(&held[0]).expect("no payload").is_empty());
7608
7609 fs::remove_file(&path).expect("clean up");
7610 }
7611
7612 #[test]
7613 fn a_payload_past_one_extent_is_split_and_joined_back() {
7614 let path = linked_file("attach_two_extents", 8);
7618 let payload = vec![0x5a_u8; section::MAX_EXTENT as usize + 1];
7619 attach(
7620 &path,
7621 "items",
7622 &[section::Attachment {
7623 kind: *section::KEY_MAP,
7624 id: 0,
7625 flags: 0,
7626 header_bytes: 0,
7627 bytes: &payload,
7628 }],
7629 )
7630 .expect("attach a payload past the bound");
7631
7632 let reader = Reader::open(&path).expect("reopen");
7633 let held = reader.table().sections();
7634 let extents = reader.extents(&held[0]).expect("extent table");
7635 assert_eq!(extents.len(), 2, "one byte past the bound is two extents");
7636 assert_eq!(extents[0].length, section::MAX_EXTENT);
7637 assert_eq!(extents[1].length, 1);
7638 assert_eq!(extents[1].first, u64::from(section::MAX_EXTENT));
7639 assert_eq!(reader.extent(&extents[1]).expect("the last extent"), vec![0x5a]);
7641 assert_eq!(reader.payload(&held[0]).expect("the whole payload").len(), payload.len());
7642
7643 fs::remove_file(&path).expect("clean up");
7644 }
7645
7646 #[test]
7647 fn a_torn_extent_is_refused_rather_than_decoded() {
7648 let path = linked_file("attach_torn", 8);
7649 let payload = a_key_map_payload();
7650 attach(
7651 &path,
7652 "items",
7653 &[section::Attachment {
7654 kind: *section::KEY_MAP,
7655 id: 0,
7656 flags: 0,
7657 header_bytes: 0,
7658 bytes: &payload,
7659 }],
7660 )
7661 .expect("attach");
7662
7663 let reader = Reader::open(&path).expect("reopen");
7664 let extent = reader.extents(&reader.table().sections()[0]).expect("extent table")[0];
7665 let file = OpenOptions::new().write(true).open(&path).expect("reopen to corrupt");
7666 write_at(&file, extent.offset + 7, &[0xff]).expect("flip a byte of the payload");
7667 drop(file);
7668
7669 let reader = Reader::open(&path).expect("the table still opens");
7670 let error = reader
7671 .payload(&reader.table().sections()[0])
7672 .expect_err("a corrupt payload is not handed out");
7673 assert!(error.to_string().contains("checksum"), "{error}");
7674 assert_eq!(reader.read(0, &[0]).expect("the column is untouched").width(), 1);
7677
7678 fs::remove_file(&path).expect("clean up");
7679 }
7680
7681 #[test]
7682 fn attaching_to_a_file_of_the_previous_format_is_refused_rather_than_done() {
7683 let path = linked_file("attach_old_format", 8);
7686 let file = OpenOptions::new().write(true).open(&path).expect("reopen to patch");
7687 write_at(&file, 8, &22_u32.to_le_bytes()).expect("stamp the older format");
7688 drop(file);
7689
7690 let payload = a_key_map_payload();
7691 let error = attach(
7692 &path,
7693 "items",
7694 &[section::Attachment {
7695 kind: *section::KEY_MAP,
7696 id: 0,
7697 flags: 0,
7698 header_bytes: 0,
7699 bytes: &payload,
7700 }],
7701 )
7702 .expect_err("format 22 cannot gain a section");
7703 assert!(error.to_string().contains("format 22"), "{error}");
7704 assert!(Reader::open(&path).expect("and the file is untouched").table().rows() == 8);
7705
7706 fs::remove_file(&path).expect("clean up");
7707 }
7708
7709 #[test]
7710 fn a_section_header_longer_than_its_payload_is_refused_at_the_write() {
7711 let path = linked_file("attach_bad_header", 8);
7712 let error = attach(
7713 &path,
7714 "items",
7715 &[section::Attachment {
7716 kind: *section::KEY_MAP,
7717 id: 0,
7718 flags: 0,
7719 header_bytes: 40,
7720 bytes: &[1, 2, 3],
7721 }],
7722 )
7723 .expect_err("a writer's bug stops at the write");
7724 assert!(error.to_string().contains("header is longer"), "{error}");
7725
7726 fs::remove_file(&path).expect("clean up");
7727 }
7728
7729 #[test]
7730 fn attaching_to_a_name_the_file_does_not_hold_says_so() {
7731 let path = linked_file("attach_wrong_name", 8);
7732 let error = attach(&path, "orders", &[]).expect_err("no such table");
7733 assert!(error.to_string().contains("orders"), "{error}");
7734 fs::remove_file(&path).expect("clean up");
7735 }
7736
7737 #[test]
7738 fn the_planner_gets_an_exact_count_for_a_leading_value_of_an_incomplete_synopsis() {
7739 let path = path("frequency_prefix_for_the_planner");
7746 let mut writer =
7747 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7748 .expect("new file");
7749 let mut values = vec![Value::Integer(1); 10_000];
7750 for _ in 0..10 {
7751 values.extend((0..600).map(|tail| Value::Integer(1_000 + tail)));
7752 }
7753 for part in values.chunks(8_000) {
7756 let rows = Chunk::new(vec![
7757 Vector::from_values(LogicalType::Integer, part).expect("integers"),
7758 ])
7759 .expect("one column");
7760 writer.append(&rows).expect("a part");
7761 }
7762 writer.finish().expect("commit");
7763 let reader = Reader::open(&path).expect("reopen from disk");
7764 let prefix =
7765 reader.frequency_prefix(0).expect("a readable synopsis").expect("the column has one");
7766 assert_eq!(prefix.entries.len(), 512);
7769 assert_eq!(prefix.omitted_max, 10);
7770 let common = Common::new(reader);
7771 assert_eq!(common.rows(), 16_000);
7772 let column = common.column("id").expect("the file has that column");
7773 assert_eq!(
7774 common.rows_with(column, &Bound::Int(1)),
7775 Stat::exact(10_000, Provenance::FrequencySynopsis)
7776 );
7777 assert_eq!(
7779 common.rows_with(column, &Bound::Int(1_100)),
7780 Stat::exact(10, Provenance::FrequencySynopsis)
7781 );
7782 assert_eq!(common.rows_with(column, &Bound::Int(1_550)), Stat::Unknown);
7785 assert_eq!(common.rows_with(column, &Bound::Int(9_999)), Stat::Unknown);
7788 let remainder = common.remainder(column).expect("the list is a prefix");
7792 assert_eq!(remainder, Remainder { rows: 890, listed: 512, most: 10 });
7793 assert_eq!(remainder.rows / (601 - remainder.listed), 10);
7794 fs::remove_file(&path).expect("clean up");
7795 }
7796
7797 #[test]
7799 fn a_file_holding_no_table_commits_and_opens_and_a_table_can_be_added_to_it() {
7800 let path = path("empty");
7801 Writer::empty(&path, &[]).expect("a file with nothing in it");
7802 let catalog = Catalog::open(&path).expect("the empty file opens");
7803 assert_eq!(catalog.len(), 0);
7804 assert!(catalog.is_empty());
7805 assert_eq!(catalog.names().count(), 0);
7806 let mut writer =
7809 Writer::open(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7810 .expect("a table goes into the empty file");
7811 writer.append(&sample_ids()).expect("rows");
7812 writer.finish().expect("commit");
7813 let catalog = Catalog::open(&path).expect("the file opens again");
7814 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7815 fs::remove_file(&path).expect("clean up");
7816 }
7817
7818 #[test]
7828 fn a_committed_empty_table_gives_up_its_name_and_one_with_rows_does_not() {
7829 let path = path("empty-name");
7830 let field = || vec![Field::required("id", LogicalType::Integer)];
7831 Writer::create(&path, "items", field()).expect("new file").finish().expect("commit");
7832 let catalog = Catalog::open(&path).expect("the file opens");
7833 assert_eq!(catalog.rows().collect::<Vec<_>>(), vec![("items", 0)]);
7834
7835 let mut writer = Writer::open(&path, "items", field()).expect("the empty name is free");
7836 writer.append(&sample_ids()).expect("rows");
7837 writer.finish().expect("commit");
7838 let catalog = Catalog::open(&path).expect("the file opens again");
7839 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7841 let held = catalog.rows().collect::<Vec<_>>();
7842 assert_eq!(held.len(), 1);
7843 assert!(held[0].1 > 0, "the rows that were appended are the ones the catalog counts");
7844
7845 let error = Writer::open(&path, "items", field()).expect_err("a name with rows is taken");
7847 assert!(error.to_string().contains("same name"), "{error}");
7848 fs::remove_file(&path).expect("clean up");
7849 }
7850
7851 fn sample_view(name: &str) -> ViewEntry {
7853 ViewEntry {
7854 name: name.to_string(),
7855 sql: "SELECT id FROM items WHERE id > 0".to_string(),
7856 statement: format!("CREATE VIEW {name} AS SELECT id FROM items WHERE (id > 0);"),
7857 aliases: vec!["n".to_string()],
7858 columns: vec![Field::new("n", LogicalType::Integer)],
7859 }
7860 }
7861
7862 #[test]
7863 fn a_view_written_into_the_catalog_comes_back_whole() {
7864 let path = path("views");
7865 let mut writer =
7866 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7867 .expect("new file");
7868 writer.append(&sample_ids()).expect("rows");
7869 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7870 let catalog = Catalog::open(&path).expect("reopen");
7871 assert_eq!(catalog.views().cloned().collect::<Vec<_>>(), vec![sample_view("v")]);
7872 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7875 fs::remove_file(&path).expect("clean up");
7876 }
7877
7878 #[test]
7880 fn appending_a_table_carries_the_views_forward() {
7881 let path = path("viewscarry");
7882 let mut writer =
7883 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7884 .expect("new file");
7885 writer.append(&sample_ids()).expect("rows");
7886 writer.with_views(vec![sample_view("v")]).finish().expect("commit");
7887 let mut writer =
7888 Writer::open(&path, "other", vec![Field::required("id", LogicalType::Integer)])
7889 .expect("a second table");
7890 writer.append(&sample_ids()).expect("rows");
7891 writer.finish().expect("commit");
7892 let catalog = Catalog::open(&path).expect("reopen");
7893 assert_eq!(catalog.views().count(), 1);
7894 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items", "other"]);
7895 fs::remove_file(&path).expect("clean up");
7896 }
7897
7898 #[test]
7900 fn restating_the_views_leaves_every_table_where_it_was() {
7901 let path = path("restate");
7902 let mut writer =
7903 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
7904 .expect("new file");
7905 writer.append(&sample_ids()).expect("rows");
7906 writer.finish().expect("commit");
7907 let before = fs::metadata(&path).expect("the file is there").len();
7908 Writer::restate(&path, &[sample_view("v"), sample_view("w")]).expect("two views");
7909 let catalog = Catalog::open(&path).expect("reopen");
7910 assert_eq!(catalog.views().count(), 2);
7911 assert_eq!(catalog.names().collect::<Vec<_>>(), vec!["items"]);
7912 let after = fs::metadata(&path).expect("the file is there").len();
7915 assert!(after > before, "a generation was written");
7916 assert!(after - before < before, "the table was not written again");
7917 let reader = Catalog::open(&path).expect("reopen").table("items").expect("the table");
7920 assert_eq!(reader.table().rows, 3);
7921 Writer::restate(&path, &[]).expect("no views at all");
7924 assert_eq!(Catalog::open(&path).expect("reopen").views().count(), 0);
7925 fs::remove_file(&path).expect("clean up");
7926 }
7927
7928 #[test]
7930 fn a_view_named_after_a_table_is_refused_when_the_catalog_is_read() {
7931 let bytes = encode_catalog(
7932 &[Entry {
7933 name: "items".to_string(),
7934 fields: vec![Field::required("id", LogicalType::Integer)],
7935 rows: 1,
7936 directory: Page { offset: HEADER, length: 8, hash: 0 },
7937 }],
7938 &[sample_view("items")],
7939 )
7940 .expect("it encodes, because encoding does not look");
7941 let error = decode_catalog(&bytes, HEADER + 8).expect_err("and decoding does");
7942 assert!(error.to_string().contains("same name"), "{error}");
7943 }
7944
7945 #[test]
7946 fn committed_file_reopens_and_reads_only_requested_columns() {
7947 let path = path("reopen");
7948 let mut writer = Writer::create(
7949 &path,
7950 "items",
7951 vec![
7952 Field::required("id", LogicalType::Integer),
7953 Field::new("text", LogicalType::Varchar),
7954 ],
7955 )
7956 .expect("new file");
7957 writer.append(&sample()).expect("first part");
7958 writer.append(&sample()).expect("second part");
7959 writer.finish().expect("commit");
7960 let reader = Reader::open(&path).expect("reopen from disk");
7961 assert_eq!(reader.table().rows(), 6);
7962 assert_eq!(reader.table().stripes().len(), 1);
7965 assert_eq!(reader.parts(), 2);
7966 assert_eq!(reader.part_rows(0), 3);
7967 assert_eq!(reader.part_rows(1), 3);
7968 let text = reader.read(1, &[1]).expect("only text page");
7969 assert_eq!(text.width(), 1);
7970 assert_eq!(text.value_at(1, 0), Value::Null);
7971 assert_eq!(text.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7972 let sparse = reader.read_sparse(1, &[1]).expect("one part without its whole page");
7973 assert_eq!(sparse.width(), 1);
7974 assert_eq!(sparse.value_at(1, 0), Value::Null);
7975 assert_eq!(sparse.value_at(2, 0), Value::Varchar("long text after a slash".into()));
7976 assert!(!reader.skips_codes(0, 1, &[0]).expect("alpha is in the stripe"));
7977 assert!(!reader.skips_codes(0, 1, &[2]).expect("long text is in the stripe"));
7978 assert!(reader.skips_codes(0, 1, &[3]).expect("unknown code is absent"));
7979 let count = reader.read(0, &[]).expect("no page is needed for count");
7980 assert_eq!(count.len(), 3);
7981 assert!(reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }]));
7982 assert!(!reader.skips(0, &[Probe { column: 0, op: Op::Greater, value: Bound::Int(0) }]));
7983 let integers = reader.top_frequencies(0, 1).expect("valid integer synopsis").expect("kept");
7984 assert_eq!(
7985 integers,
7986 vec![(Value::Integer(-2), 2), (Value::Integer(4), 2), (Value::Integer(9), 2),]
7987 );
7988 let strings = reader.top_frequencies(1, 1).expect("valid string synopsis").expect("kept");
7989 assert_eq!(strings.len(), 3);
7990 assert!(strings.contains(&(Value::Null, 2)));
7991 assert!(strings.contains(&(Value::Varchar("alpha".into()), 2)));
7992 assert!(strings.contains(&(Value::Varchar("long text after a slash".into()), 2)));
7993 fs::remove_file(path).expect("remove scratch file");
7994 }
7995
7996 #[test]
8004 fn runs_handed_over_out_of_order_still_read_back_in_source_order() {
8005 let path = path("interleaved-runs");
8006 let mut writer =
8007 Writer::create(&path, "interleaved", vec![Field::new("v", LogicalType::BigInt)])
8008 .expect("new file");
8009 for morsel in [2_u64, 0, 3, 1] {
8010 let parts = (0..4_u64)
8011 .map(|chunk| {
8012 let first = i64::try_from(morsel * 32 + chunk * 8).expect("small");
8013 let values =
8014 (0..8_i64).map(|row| Value::BigInt(first + row)).collect::<Vec<_>>();
8015 let column =
8016 Vector::from_values(LogicalType::BigInt, &values).expect("a column");
8017 ((morsel, chunk), Chunk::new(vec![column]).expect("one column"))
8018 })
8019 .collect::<Vec<_>>();
8020 writer.append_stripe(parts).expect("a stripe");
8021 }
8022 writer.finish().expect("commit");
8023
8024 let reader = Reader::open(&path).expect("valid directory");
8025 assert_eq!(reader.table().stripes().len(), 4, "a run is a stripe of its own");
8026 assert_eq!(reader.table().rows(), 128);
8027 for part in 0..16_usize {
8028 let read = reader.read(part, &[0]).expect("a part back");
8029 for row in 0..8_usize {
8030 let want = i64::try_from(part * 8 + row).expect("small");
8031 assert_eq!(read.value_at(row, 0), Value::BigInt(want), "part {part} row {row}");
8032 }
8033 }
8034 fs::remove_file(path).expect("remove scratch file");
8035 }
8036
8037 #[test]
8040 fn runs_that_overlap_each_other_are_refused_at_commit() {
8041 let path = path("overlapping-runs");
8042 let mut writer =
8043 Writer::create(&path, "overlapping", vec![Field::new("v", LogicalType::BigInt)])
8044 .expect("new file");
8045 let one = |order: (u64, u64)| {
8046 let column =
8047 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)]).expect("a column");
8048 (order, Chunk::new(vec![column]).expect("one column"))
8049 };
8050 writer.append_stripe(vec![one((0, 0)), one((0, 2))]).expect("a stripe");
8053 writer.append_stripe(vec![one((0, 1))]).expect("a stripe");
8054 let error = writer.finish().expect_err("the runs overlap");
8055 assert!(error.message().contains("source order"), "{error}");
8056 fs::remove_file(path).expect("remove scratch file");
8057 }
8058
8059 #[test]
8062 fn a_run_longer_than_a_stripe_is_refused() {
8063 let path = path("overlong-run");
8064 let mut writer =
8065 Writer::create(&path, "overlong", vec![Field::new("v", LogicalType::BigInt)])
8066 .expect("new file");
8067 let parts = (0..=STRIPE_PARTS)
8068 .map(|at| {
8069 let column = Vector::from_values(LogicalType::BigInt, &[Value::BigInt(1)])
8070 .expect("a column");
8071 let chunk = Chunk::new(vec![column]).expect("one column");
8072 ((0, u64::try_from(at).expect("small")), chunk)
8073 })
8074 .collect::<Vec<_>>();
8075 let error = writer.append_stripe(parts).expect_err("one part too many");
8076 assert!(error.message().contains("more parts than it holds"), "{error}");
8077 fs::remove_file(path).expect("remove scratch file");
8078 }
8079
8080 #[test]
8086 fn parts_past_the_stripe_bound_start_a_new_stripe() {
8087 let path = path("stripe-bound");
8088 let mut writer = Writer::create(
8089 &path,
8090 "items",
8091 vec![
8092 Field::required("id", LogicalType::Integer),
8093 Field::new("text", LogicalType::Varchar),
8094 ],
8095 )
8096 .expect("new file");
8097 let parts = STRIPE_PARTS * 2 + 3;
8098 for part in 0..parts {
8099 let id = part as i32;
8100 let chunk = Chunk::new(vec![
8101 Vector::from_values(
8102 LogicalType::Integer,
8103 &[Value::Integer(id), Value::Integer(-id)],
8104 )
8105 .expect("integers"),
8106 Vector::from_values(
8107 LogicalType::Varchar,
8108 &[Value::Varchar(format!("value {part}")), Value::Null],
8109 )
8110 .expect("strings"),
8111 ])
8112 .expect("matching rows");
8113 writer.append(&chunk).expect("one part");
8114 }
8115 writer.finish().expect("commit");
8116
8117 let reader = Reader::open(&path).expect("reopen from disk");
8118 assert_eq!(reader.parts(), parts);
8119 assert_eq!(reader.table().rows(), parts * 2);
8120 assert_eq!(reader.table().stripes().len(), parts.div_ceil(STRIPE_PARTS));
8121 assert_eq!(reader.table().stripes()[0].parts(), STRIPE_PARTS);
8122 assert_eq!(reader.table().stripes()[0].rows(), STRIPE_PARTS * 2);
8123 assert_eq!(reader.table().stripes()[2].parts(), 3);
8124 for part in (0..parts).rev() {
8127 let dense = reader.read(part, &[0, 1]).expect("a whole page read");
8128 let sparse = reader.read_sparse(part, &[0, 1]).expect("one part read");
8129 for chunk in [&dense, &sparse] {
8130 assert_eq!(chunk.len(), 2, "part {part} has its own row count");
8131 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8132 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8133 assert_eq!(chunk.value_at(0, 1), Value::Varchar(format!("value {part}")));
8134 assert_eq!(chunk.value_at(1, 1), Value::Null);
8135 }
8136 }
8137 let above = [Probe { column: 0, op: Op::Greater, value: Bound::Int(100) }];
8140 assert!(reader.skips(0, &above), "the first stripe stops at 63");
8141 assert!(!reader.skips(STRIPE_PARTS * 2, &above), "the third stripe reaches 130");
8142 fs::remove_file(path).expect("remove scratch file");
8143 }
8144
8145 fn scattered(n: i64) -> i64 {
8147 n.wrapping_mul(-7_046_029_254_386_353_131)
8148 }
8149
8150 #[test]
8156 fn a_part_is_skipped_when_its_sieve_does_not_hold_the_constant() {
8157 let path = path("sieve-skip");
8158 let mut writer =
8159 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8160 .expect("new file");
8161 let parts = STRIPE_PARTS + 3;
8162 let per_part = 128;
8166 for part in 0..parts {
8167 let held: Vec<Value> = (0..per_part)
8168 .map(|row| Value::BigInt(scattered((part * per_part + row) as i64)))
8169 .collect();
8170 let chunk =
8171 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8172 .expect("one column");
8173 writer.append(&chunk).expect("one part");
8174 }
8175 writer.finish().expect("commit");
8176
8177 let reader = Reader::open(&path).expect("reopen from disk");
8178 let probe = |value: i64| Probe {
8179 column: 0,
8180 op: Op::Equal,
8181 value: Bound::Int(i128::from(scattered(value))),
8182 };
8183 for wanted in [0_i64, (per_part + 1) as i64, (parts * per_part - 1) as i64] {
8184 let tests = [probe(wanted)];
8185 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &tests)).collect();
8186 let home = wanted as usize / per_part;
8187 assert!(kept.contains(&home), "the part holding {wanted} is read");
8188 assert!(kept.len() <= 2, "{wanted} keeps {kept:?}, which is more than one stray part");
8192 }
8193 let absent = [probe((parts * per_part) as i64 + 1)];
8194 let kept = (0..parts).filter(|&part| !reader.skips(part, &absent)).count();
8195 assert!(kept <= 1, "{kept} parts of {parts} kept a value no part holds");
8196 let tests = [probe(0)];
8199 assert!(
8200 reader.table().stripes().iter().all(|stripe| !stripe.zone.skips(&tests)),
8201 "the bounds rule out no stripe at all"
8202 );
8203 fs::remove_file(path).expect("remove scratch file");
8204 }
8205
8206 #[test]
8212 fn a_part_is_skipped_when_its_own_bounds_rule_out_a_comparison_the_stripe_keeps() {
8213 let path = path("part-range-skip");
8214 let mut writer =
8215 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8216 .expect("new file");
8217 let parts = STRIPE_PARTS + 3;
8218 let per_part = 128;
8219 for part in 0..parts {
8220 let held: Vec<Value> = (0..per_part)
8224 .map(|row| {
8225 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8226 })
8227 .collect();
8228 let chunk =
8229 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8230 .expect("one column");
8231 writer.append(&chunk).expect("one part");
8232 }
8233 writer.finish().expect("commit");
8234
8235 let reader = Reader::open(&path).expect("reopen from disk");
8236 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8237 let kept: Vec<usize> = (0..parts).filter(|&part| !reader.skips(part, &under)).collect();
8238 assert_eq!(kept, vec![0, 1, 2], "only the three parts that start under three thousand");
8239 assert!(!reader.stripe_skips(0, &under), "the stripe reaches from zero and keeps itself");
8241 fs::remove_file(path).expect("remove scratch file");
8242 }
8243
8244 #[test]
8248 fn a_part_is_waved_through_when_its_own_bounds_pass_a_comparison_the_stripe_cannot() {
8249 let path = path("part-range-certain");
8250 let mut writer =
8251 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8252 .expect("new file");
8253 let parts = STRIPE_PARTS + 3;
8254 let per_part = 128;
8255 for part in 0..parts {
8256 let held: Vec<Value> = (0..per_part)
8257 .map(|row| {
8258 Value::BigInt((part * 1_000) as i64 + (scattered(row as i64).rem_euclid(900)))
8259 })
8260 .collect();
8261 let chunk =
8262 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8263 .expect("one column");
8264 writer.append(&chunk).expect("one part");
8265 }
8266 writer.finish().expect("commit");
8267
8268 let reader = Reader::open(&path).expect("reopen from disk");
8269 let under = [Probe { column: 0, op: Op::Less, value: Bound::Int(3_000) }];
8270 let waved: Vec<usize> = (0..parts).filter(|&part| reader.certain(part, &under)).collect();
8271 assert_eq!(waved, vec![0, 1, 2], "the three parts that end under three thousand");
8272 assert!(!reader.stripe_skips(0, &under), "the stripe straddles the comparison");
8275 fs::remove_file(path).expect("remove scratch file");
8276 }
8277
8278 #[test]
8281 fn a_stripe_of_one_part_writes_no_range_page_and_a_stripe_of_many_does() {
8282 for (parts, wanted) in [(1_usize, false), (STRIPE_PARTS, true)] {
8283 let path = path("part-range-page");
8284 let mut writer =
8285 Writer::create(&path, "hits", vec![Field::required("at", LogicalType::BigInt)])
8286 .expect("new file");
8287 for part in 0..parts {
8288 let held: Vec<Value> = (0..128)
8289 .map(|row| {
8290 Value::BigInt((part * 1_000) as i64 + scattered(row as i64).rem_euclid(900))
8291 })
8292 .collect();
8293 let chunk = Chunk::new(vec![
8294 Vector::from_values(LogicalType::BigInt, &held).expect("numbers"),
8295 ])
8296 .expect("one column");
8297 writer.append(&chunk).expect("one part");
8298 }
8299 writer.finish().expect("commit");
8300 let reader = Reader::open(&path).expect("reopen from disk");
8301 let bytes = reader.layout().columns[0].part_ranges;
8302 assert_eq!(bytes > 0, wanted, "{parts} parts wrote {bytes} bytes of ranges");
8303 fs::remove_file(path).expect("remove scratch file");
8304 }
8305 }
8306
8307 #[test]
8310 fn a_string_end_that_is_cut_down_still_covers_the_value_it_came_from() {
8311 let long = vec![b'a'; PART_BOUND_BYTES * 2];
8312 let low = shortened(Some(Bound::Bytes(long.clone())), false).expect("a low end");
8313 let high = shortened(Some(Bound::Bytes(long.clone())), true).expect("a high end");
8314 let Bound::Bytes(low) = low else { panic!("a string stays a string") };
8315 let Bound::Bytes(high) = high else { panic!("a string stays a string") };
8316 assert!(low.len() <= PART_BOUND_BYTES && high.len() <= PART_BOUND_BYTES);
8317 assert!(low.as_slice() <= long.as_slice(), "the low end is at or under the value");
8318 assert!(high.as_slice() >= long.as_slice(), "the high end is at or over the value");
8319 }
8320
8321 #[test]
8324 fn a_string_end_with_no_room_to_step_up_gives_up_the_bound() {
8325 let long = vec![u8::MAX; PART_BOUND_BYTES * 2];
8326 assert_eq!(shortened(Some(Bound::Bytes(long.clone())), true), None);
8327 let low = shortened(Some(Bound::Bytes(long)), false).expect("a low end is still a prefix");
8328 assert_eq!(low, Bound::Bytes(vec![u8::MAX; PART_BOUND_BYTES]));
8329 }
8330
8331 #[test]
8343 fn what_a_column_is_stored_as_follows_the_order_the_rows_were_written_in() {
8344 let parts = 4;
8345 let per_part = 1024;
8346 let rows = parts * per_part;
8347 let written = |name: &str, keys: &[i64]| {
8348 let path = path(name);
8349 let fields = vec![Field::required("key", LogicalType::BigInt)];
8350 let mut writer = Writer::create(&path, "keys", fields).expect("new file");
8351 for part in 0..parts {
8352 let values: Vec<Value> = keys[part * per_part..(part + 1) * per_part]
8353 .iter()
8354 .map(|key| Value::BigInt(*key))
8355 .collect();
8356 let chunk = Chunk::new(vec![
8357 Vector::from_values(LogicalType::BigInt, &values).expect("numbers"),
8358 ])
8359 .expect("one column");
8360 writer.append(&chunk).expect("one part");
8361 }
8362 writer.finish().expect("commit");
8363 path
8364 };
8365 let climbing = |step: &dyn Fn(usize) -> i64| {
8368 let mut key = 0;
8369 (0..rows)
8370 .map(|row| {
8371 key += step(row);
8372 key
8373 })
8374 .collect::<Vec<i64>>()
8375 };
8376 let ascending = climbing(&|row| (row % 3) as i64);
8377 let sparse = climbing(&|row| ((row * 2_654_435_761) % 4096) as i64);
8381 let near_path = written("stored-near", &ascending);
8382 let far_path = written("stored-far", &sparse);
8383
8384 let one = Reader::open(&near_path).expect("reopen from disk");
8385 let other = Reader::open(&far_path).expect("reopen from disk");
8386 let near = one.stored(0).expect("the column is stored");
8387 let far = other.stored(0).expect("the column is stored");
8388 assert_eq!(near.len(), parts, "one row per part");
8389 assert_eq!(far.len(), parts);
8390 let total = |stored: &[StoredPart]| stored.iter().map(|part| part.bytes).sum::<u64>();
8393 assert_eq!(total(&near), one.layout().columns[0].pages);
8394 assert_eq!(total(&far), other.layout().columns[0].pages);
8395 assert!(
8396 total(&near) * 2 < total(&far),
8397 "the sparse keys cost more, {} against {}",
8398 total(&far),
8399 total(&near)
8400 );
8401 for (at, part) in near.iter().enumerate() {
8403 assert_eq!(part.part, at);
8404 assert_eq!(part.row, at * per_part);
8405 assert_eq!(part.rows, per_part);
8406 let held = &ascending[at * per_part..(at + 1) * per_part];
8407 assert_eq!(part.low, Some(Value::BigInt(held[0])));
8408 assert_eq!(part.high, Some(Value::BigInt(held[per_part - 1])));
8409 assert_eq!(part.nulls, Some(0));
8410 }
8411 assert!(near[0].encoding.contains("DELTA"), "{}", near[0].encoding);
8414 assert!(far[0].encoding.contains("DELTA"), "{}", far[0].encoding);
8415 assert_ne!(near[0].encoding, far[0].encoding);
8416 fs::remove_file(near_path).expect("remove scratch file");
8417 fs::remove_file(far_path).expect("remove scratch file");
8418 }
8419
8420 #[test]
8430 fn a_sieve_larger_than_the_part_it_indexes_is_not_written() {
8431 let path = path("sieve-pays");
8432 let fields = vec![
8433 Field::required("spread", LogicalType::BigInt),
8434 Field::required("repeated", LogicalType::BigInt),
8435 ];
8436 let mut writer = Writer::create(&path, "hits", fields).expect("new file");
8437 let parts = 3;
8438 let per_part = 1024;
8439 for part in 0..parts {
8440 let base = (part * per_part) as i64;
8441 let spread: Vec<Value> =
8442 (0..per_part).map(|row| Value::BigInt(scattered(base + row as i64))).collect();
8443 let repeated: Vec<Value> =
8444 (0..per_part).map(|row| Value::BigInt(scattered((row / 256) as i64))).collect();
8445 let chunk = Chunk::new(vec![
8446 Vector::from_values(LogicalType::BigInt, &spread).expect("numbers"),
8447 Vector::from_values(LogicalType::BigInt, &repeated).expect("numbers"),
8448 ])
8449 .expect("two columns");
8450 writer.append(&chunk).expect("one part");
8451 }
8452 writer.finish().expect("commit");
8453
8454 let reader = Reader::open(&path).expect("reopen from disk");
8455 let layout = reader.layout();
8456 let spread = &layout.columns[0];
8457 let repeated = &layout.columns[1];
8458 assert!(spread.sieves > 0, "a column whose parts are worth a filter keeps one");
8459 assert_eq!(
8460 repeated.sieves, 0,
8461 "a column whose filter costs more than its parts keeps none"
8462 );
8463 for column in &layout.columns {
8466 assert!(
8467 column.sieves < column.pages,
8468 "{} spends {} on sieves over {} of data",
8469 column.name,
8470 column.sieves,
8471 column.pages
8472 );
8473 }
8474 let absent = [Probe {
8476 column: 0,
8477 op: Op::Equal,
8478 value: Bound::Int(i128::from(scattered((parts * per_part) as i64 + 1))),
8479 }];
8480 assert!((0..parts).all(|part| reader.skips(part, &absent)), "no part holds it");
8481 fs::remove_file(path).expect("remove scratch file");
8482 }
8483
8484 #[test]
8490 fn a_damaged_sieve_page_is_read_through_rather_than_refused() {
8491 let path = path("sieve-damaged");
8492 let mut writer =
8493 Writer::create(&path, "hits", vec![Field::required("id", LogicalType::BigInt)])
8494 .expect("new file");
8495 let rows = 128;
8496 let held: Vec<Value> = (0..rows).map(|row| Value::BigInt(scattered(row))).collect();
8497 let chunk =
8498 Chunk::new(vec![Vector::from_values(LogicalType::BigInt, &held).expect("numbers")])
8499 .expect("one column");
8500 writer.append(&chunk).expect("one part");
8501 writer.finish().expect("commit");
8502
8503 let page =
8504 Reader::open(&path).expect("reopen").table.stripes[0].sieves[0].expect("a sieve page");
8505 let mut file = OpenOptions::new().write(true).open(&path).expect("open the sieve page");
8506 file.seek(SeekFrom::Start(page.offset + u64::from(page.length) - 1)).expect("seek");
8507 file.write_all(&[0xff]).expect("damage one byte");
8508 drop(file);
8509
8510 let reader = Reader::open(&path).expect("reopen the damaged file");
8511 let absent =
8512 [Probe { column: 0, op: Op::Equal, value: Bound::Int(i128::from(scattered(99))) }];
8513 assert!(!reader.skips(0, &absent), "a sieve that cannot be read skips nothing");
8514 assert_eq!(
8515 reader.read(0, &[0]).expect("the rows are untouched").len(),
8516 usize::try_from(rows).expect("a small count")
8517 );
8518 fs::remove_file(path).expect("remove scratch file");
8519 }
8520
8521 #[test]
8532 fn workers_that_want_the_same_stripe_read_it_once() {
8533 let path = path("single-flight");
8534 let mut writer =
8535 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8536 .expect("new file");
8537 for part in 0..STRIPE_PARTS {
8538 let id = part as i32;
8539 let chunk = Chunk::new(vec![
8540 Vector::from_values(
8541 LogicalType::Integer,
8542 &[Value::Integer(id), Value::Integer(-id)],
8543 )
8544 .expect("integers"),
8545 ])
8546 .expect("matching rows");
8547 writer.append(&chunk).expect("one part");
8548 }
8549 writer.finish().expect("commit");
8550
8551 let reader = Reader::open(&path).expect("reopen from disk");
8552 assert_eq!(reader.table().stripes().len(), 1, "one stripe is the point of the test");
8553 let barrier = std::sync::Barrier::new(8);
8554 std::thread::scope(|scope| {
8555 for worker in 0..8 {
8556 let reader = &reader;
8557 let barrier = &barrier;
8558 scope.spawn(move || {
8559 barrier.wait();
8560 for part in (worker..STRIPE_PARTS).step_by(8) {
8561 let chunk = reader.read(part, &[0]).expect("a whole page read");
8562 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8563 assert_eq!(chunk.value_at(1, 0), Value::Integer(-(part as i32)));
8564 }
8565 });
8566 }
8567 });
8568 assert_eq!(reader.pages.load(Atomic::Relaxed), 1, "one stripe, one page read, whoever won");
8569 fs::remove_file(path).expect("remove scratch file");
8570 }
8571
8572 #[test]
8585 fn opening_costs_the_same_over_a_thousand_times_the_rows() {
8586 let opened = |label: &str, rows_per_part: i32| {
8587 let path = path(label);
8588 let mut writer =
8589 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8590 .expect("new file");
8591 for part in 0..STRIPE_PARTS * 3 {
8592 let values = (0..rows_per_part)
8596 .map(|row| {
8597 Value::Integer((part as i32 * rows_per_part + row).wrapping_mul(2_654_435))
8598 })
8599 .collect::<Vec<_>>();
8600 let chunk = Chunk::new(vec![
8601 Vector::from_values(LogicalType::Integer, &values).expect("integers"),
8602 ])
8603 .expect("matching rows");
8604 writer.append(&chunk).expect("one part");
8605 }
8606 writer.finish().expect("commit");
8607 let reader = Reader::open(&path).expect("reopen from disk");
8608 let size = fs::metadata(&path).expect("the file is there").len();
8609 let out = (reader.reads(), reader.table().stripes().len(), size);
8610 fs::remove_file(path).expect("remove scratch file");
8611 out
8612 };
8613
8614 let (thin, thin_stripes, thin_size) = opened("open-thin", 1);
8615 let (fat, fat_stripes, fat_size) = opened("open-fat", 1000);
8616 assert_eq!(
8617 thin_stripes, fat_stripes,
8618 "the same stripe count is what makes this a fair ask"
8619 );
8620 assert!(
8621 fat_size > thin_size * 50,
8622 "the fat file has to actually be larger, and it is {fat_size} against {thin_size}"
8623 );
8624
8625 assert_eq!(thin.opening.reads, fat.opening.reads, "the same reads either way");
8626 assert_eq!(thin.pages, 0, "opening read a page");
8627 assert_eq!(fat.pages, 0, "opening read a page");
8628 assert_eq!(thin.indexes, 0, "opening read an index");
8629 assert_eq!(fat.indexes, 0, "opening read an index");
8630 assert!(
8633 fat.opening.bytes < thin.opening.bytes * 2,
8634 "opening the thin file read {} bytes and the fat one read {}",
8635 thin.opening.bytes,
8636 fat.opening.bytes
8637 );
8638 }
8639
8640 #[test]
8648 fn two_opens_of_one_file_cost_the_same_and_the_second_is_not_cheaper() {
8649 let path = path("open-twice");
8650 let mut writer =
8651 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8652 .expect("new file");
8653 for part in 0..STRIPE_PARTS * 3 {
8654 let chunk = Chunk::new(vec![
8655 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8656 .expect("integers"),
8657 ])
8658 .expect("matching rows");
8659 writer.append(&chunk).expect("one part");
8660 }
8661 writer.finish().expect("commit");
8662
8663 let first = Reader::open(&path).expect("open");
8664 for part in 0..first.parts() {
8667 first.read(part, &[0]).expect("a part");
8668 }
8669 assert!(first.reads().pages > 0, "the scan has to have read something");
8670 let second = Reader::open(&path).expect("open again");
8671
8672 assert_eq!(first.reads().opening, second.reads().opening);
8673 assert_eq!(
8674 second.reads().pages,
8675 0,
8676 "the second open read a page off the back of the first"
8677 );
8678 assert_eq!(second.reads().indexes, 0, "the second open read an index it inherited");
8679 fs::remove_file(path).expect("remove scratch file");
8680 }
8681
8682 #[test]
8690 fn an_index_is_read_once_per_stripe_however_often_the_page_is_evicted() {
8691 let path = path("index-cache");
8692 let mut writer =
8693 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8694 .expect("new file");
8695 let parts = STRIPE_PARTS * (CACHED_STRIPES_PER_COLUMN + 2);
8696 for part in 0..parts {
8697 let id = part as i32;
8698 let chunk = Chunk::new(vec![
8699 Vector::from_values(LogicalType::Integer, &[Value::Integer(id)]).expect("integers"),
8700 ])
8701 .expect("matching rows");
8702 writer.append(&chunk).expect("one part");
8703 }
8704 writer.finish().expect("commit");
8705
8706 let reader = Reader::open(&path).expect("reopen from disk");
8707 let stripes = reader.table().stripes().len();
8708 assert!(stripes > CACHED_STRIPES_PER_COLUMN, "the page cache has to be too small for this");
8709 for _ in 0..2 {
8711 for part in 0..parts {
8712 let chunk = reader.read(part, &[0]).expect("a part");
8713 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8714 }
8715 }
8716 assert_eq!(reader.indexes.load(Atomic::Relaxed), stripes, "one index read per stripe");
8717 assert!(
8718 reader.pages.load(Atomic::Relaxed) > stripes,
8719 "the pages are the ones that get read again, which is what makes the index count mean \
8720 something"
8721 );
8722 fs::remove_file(path).expect("remove scratch file");
8723 }
8724
8725 #[test]
8734 fn a_worker_per_stripe_reads_its_page_once_when_the_cache_was_told_to_expect_it() {
8735 let workers = CACHED_STRIPES_PER_COLUMN + 4;
8736 let path = path("stripe-per-worker");
8737 let mut writer =
8738 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8739 .expect("new file");
8740 for part in 0..STRIPE_PARTS * workers {
8741 let chunk = Chunk::new(vec![
8742 Vector::from_values(LogicalType::Integer, &[Value::Integer(part as i32)])
8743 .expect("integers"),
8744 ])
8745 .expect("matching rows");
8746 writer.append(&chunk).expect("one part");
8747 }
8748 writer.finish().expect("commit");
8749
8750 let read = |told: bool| {
8751 let reader = Reader::open(&path).expect("reopen from disk");
8752 assert_eq!(reader.table().stripes().len(), workers, "a stripe per worker");
8753 if told {
8754 reader.keep_stripes(workers);
8755 }
8756 let barrier = std::sync::Barrier::new(workers);
8757 std::thread::scope(|scope| {
8758 for (worker, run) in reader.stripe_parts().into_iter().enumerate() {
8759 let reader = &reader;
8760 let barrier = &barrier;
8761 scope.spawn(move || {
8762 for part in run {
8763 barrier.wait();
8764 let chunk = reader.read(part, &[0]).expect("a part of my own stripe");
8765 assert_eq!(chunk.value_at(0, 0), Value::Integer(part as i32));
8766 }
8767 assert!(worker < workers);
8768 });
8769 }
8770 });
8771 reader.pages.load(Atomic::Relaxed)
8772 };
8773
8774 assert_eq!(read(true), workers, "one page read per stripe and no more");
8775 assert!(read(false) > workers, "a cache that small is read again on every part");
8776 fs::remove_file(path).expect("remove scratch file");
8777 }
8778
8779 #[test]
8784 fn a_damaged_index_page_is_an_error() {
8785 let path = path("damaged-index");
8786 let mut writer =
8787 Writer::create(&path, "items", vec![Field::required("id", LogicalType::Integer)])
8788 .expect("new file");
8789 writer.append(&sample_ids()).expect("first part");
8790 writer.append(&sample_ids()).expect("second part");
8791 writer.finish().expect("commit");
8792
8793 let reader = Reader::open(&path).expect("valid directory");
8794 let index = reader.table.stripes[0].index;
8795 let mut byte = [0; 1];
8796 read_at(&reader.file, index.offset, &mut byte).expect("the first part length");
8797 let mut file = OpenOptions::new().write(true).open(&path).expect("open index page");
8798 file.seek(SeekFrom::Start(index.offset)).expect("index start");
8799 file.write_all(&[!byte[0]]).expect("damage the first part length");
8800 let error = reader.read(1, &[0]).expect_err("a damaged index must not be used");
8801 assert!(error.message().contains("index page section checksum differs"), "{error}");
8802 fs::remove_file(path).expect("remove scratch file");
8803 }
8804
8805 #[test]
8812 fn every_integer_width_round_trips_through_a_page() {
8813 let path = path("integer-widths");
8814 let columns = [
8815 (LogicalType::TinyInt, vec![Value::TinyInt(i8::MIN), Value::TinyInt(i8::MAX)]),
8816 (LogicalType::UTinyInt, vec![Value::UTinyInt(0), Value::UTinyInt(u8::MAX)]),
8817 (LogicalType::SmallInt, vec![Value::SmallInt(i16::MIN), Value::SmallInt(i16::MAX)]),
8818 (LogicalType::USmallInt, vec![Value::USmallInt(0), Value::USmallInt(u16::MAX)]),
8819 (LogicalType::Integer, vec![Value::Integer(i32::MIN), Value::Integer(i32::MAX)]),
8820 (LogicalType::UInteger, vec![Value::UInteger(0), Value::UInteger(u32::MAX)]),
8821 (LogicalType::BigInt, vec![Value::BigInt(i64::MIN), Value::BigInt(i64::MAX)]),
8822 (LogicalType::UBigInt, vec![Value::UBigInt(0), Value::UBigInt(u64::MAX)]),
8823 ];
8824 let fields = columns
8825 .iter()
8826 .enumerate()
8827 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8828 .collect::<Vec<_>>();
8829 let vectors = columns
8830 .iter()
8831 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8832 .collect::<Vec<_>>();
8833 let mut writer = Writer::create(&path, "widths", fields).expect("new file");
8834 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8835 writer.finish().expect("commit");
8836
8837 let reader = Reader::open(&path).expect("reopen from disk");
8838 let wanted = (0..columns.len()).collect::<Vec<_>>();
8839 let read = reader.read(0, &wanted).expect("every column");
8840 assert_eq!(read.len(), 2);
8841 for (at, (ty, values)) in columns.iter().enumerate() {
8843 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8844 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8845 }
8846 fs::remove_file(path).expect("remove scratch file");
8847 }
8848
8849 #[test]
8860 fn every_other_type_the_format_knows_round_trips_through_a_page() {
8861 let path = path("other-types");
8862 let columns = [
8863 (LogicalType::Float, vec![Value::Float(f32::MIN), Value::Float(-0.0)]),
8864 (LogicalType::Double, vec![Value::Double(f64::MIN), Value::Double(f64::MAX)]),
8865 (LogicalType::HugeInt, vec![Value::HugeInt(i128::MIN), Value::HugeInt(i128::MAX)]),
8866 (LogicalType::UHugeInt, vec![Value::UHugeInt(0), Value::UHugeInt(u128::MAX)]),
8867 (LogicalType::Time, vec![Value::Time(0), Value::Time(86_399_999_999)]),
8868 (LogicalType::TimeTz, vec![Value::TimeTz(-50_400_000_000), Value::TimeTz(0)]),
8869 (
8870 LogicalType::TimestampTz,
8871 vec![Value::TimestampTz(i64::MIN + 1), Value::TimestampTz(i64::MAX)],
8872 ),
8873 (
8874 LogicalType::Interval,
8875 vec![
8876 Value::Interval { months: i32::MIN, days: i32::MAX, micros: i64::MIN },
8877 Value::Interval { months: 13, days: -1, micros: 1 },
8878 ],
8879 ),
8880 (
8881 LogicalType::Blob,
8882 vec![Value::Blob(vec![0, 0xff, 0x80, 0xfe]), Value::Blob(Vec::new())],
8883 ),
8884 ];
8885 let fields = columns
8886 .iter()
8887 .enumerate()
8888 .map(|(at, (ty, _))| Field::required(format!("c{at}"), ty.clone()))
8889 .collect::<Vec<_>>();
8890 let vectors = columns
8891 .iter()
8892 .map(|(ty, values)| Vector::from_values(ty.clone(), values).expect("a vector"))
8893 .collect::<Vec<_>>();
8894 let mut writer = Writer::create(&path, "others", fields).expect("new file");
8895 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8896 writer.finish().expect("commit");
8897
8898 let reader = Reader::open(&path).expect("reopen from disk");
8899 let wanted = (0..columns.len()).collect::<Vec<_>>();
8900 let read = reader.read(0, &wanted).expect("every column");
8901 assert_eq!(read.len(), 2);
8902 for (at, (ty, values)) in columns.iter().enumerate() {
8903 assert_eq!(read.value_at(0, at), values[0], "the low end of {ty}");
8904 assert_eq!(read.value_at(1, at), values[1], "the high end of {ty}");
8905 }
8906 let Value::Float(zero) = read.value_at(1, 0) else { panic!("a float stays a float") };
8909 assert!(zero.is_sign_negative(), "a negative zero came back as {zero}");
8910
8911 fs::remove_file(path).expect("remove scratch file");
8912 }
8913
8914 #[test]
8920 fn a_nan_survives_being_written_down() {
8921 let path = path("nan");
8922 let nan = Vector::from_values(LogicalType::Double, &[Value::Double(f64::NAN)])
8923 .expect("a NaN vector");
8924 let mut writer =
8925 Writer::create(&path, "nan", vec![Field::required("d", LogicalType::Double)])
8926 .expect("new file");
8927 writer.append(&Chunk::new(vec![nan]).expect("one column")).expect("one stripe");
8928 writer.finish().expect("commit");
8929 let read = Reader::open(&path).expect("reopen").read(0, &[0]).expect("the column");
8930 let Value::Double(back) = read.value_at(0, 0) else { panic!("a double stays a double") };
8931 assert!(back.is_nan(), "a NaN came back as {back}");
8932 fs::remove_file(path).expect("remove scratch file");
8933 }
8934
8935 #[test]
8942 fn a_uuid_and_a_bit_string_come_back_as_the_bits_that_went_in() {
8943 let path = path("uuid-and-bit");
8944 let uuids = vec![0_i128, i128::MIN, -1];
8945 let mut bits = StringColumn::new();
8946 for value in [&b"\x02\xff"[..], &b""[..], &b"\x00\x01\x02\x03\x04\x05"[..]] {
8947 bits.push_bytes(value);
8948 }
8949 let expected = bits.clone();
8950 let fields =
8951 vec![Field::required("u", LogicalType::Uuid), Field::required("b", LogicalType::Bit)];
8952 let vectors = vec![
8953 Vector::flat(LogicalType::Uuid, Data::Int128(uuids.clone().into())).expect("uuids"),
8954 Vector::flat(LogicalType::Bit, Data::Varlen(bits)).expect("bit strings"),
8955 ];
8956 let mut writer = Writer::create(&path, "ids", fields).expect("new file");
8957 writer.append(&Chunk::new(vectors).expect("matching rows")).expect("one stripe");
8958 writer.finish().expect("commit");
8959
8960 let reader = Reader::open(&path).expect("reopen from disk");
8961 let read = reader.read(0, &[0, 1]).expect("both columns").flatten().expect("flat");
8962 let Some(Data::Int128(back)) = read.column(0).expect("the uuids").data() else {
8963 panic!("a uuid column is the 128 bit lane")
8964 };
8965 assert_eq!(back.as_slice(), uuids.as_slice());
8966 let Some(Data::Varlen(back)) = read.column(1).expect("the bits").data() else {
8967 panic!("a bit column is bytes")
8968 };
8969 for row in 0..expected.len() {
8970 assert_eq!(back.bytes(row), expected.bytes(row), "row {row} of the bit column");
8971 }
8972 fs::remove_file(path).expect("remove scratch file");
8973 }
8974
8975 #[test]
8976 fn numeric_frequency_candidates_keep_bounded_row_ordinals() {
8977 let path = path("frequency-ordinals");
8978 let mut writer =
8979 Writer::create(&path, "items", vec![Field::required("id", LogicalType::BigInt)])
8980 .expect("new file");
8981 let mut values = Vec::new();
8982 for leader in 0..10_i64 {
8983 values.extend(std::iter::repeat_n(leader, 100));
8984 }
8985 values.extend(1_000_i64..41_000);
8986 for part in values.chunks(1_024) {
8987 let vector = Vector::flat(LogicalType::BigInt, Data::Int64(part.to_vec().into()))
8988 .expect("big integers");
8989 writer.append(&Chunk::new(vec![vector]).expect("one column")).expect("one stripe");
8990 }
8991 writer.finish().expect("commit");
8992
8993 let reader = Reader::open(&path).expect("reopen from disk");
8994 let occurrences =
8995 reader.frequency_occurrences(0).expect("valid metadata").expect("bounded ordinals");
8996 assert!(occurrences.omitted_max < 100);
8997 assert!(occurrences.ordinals.len() <= FREQUENCY_ORDINALS);
8998 assert!(occurrences.ordinals.windows(2).all(|pair| pair[0] < pair[1]));
8999 assert_eq!(&occurrences.ordinals[..1_000], &(0_u64..1_000).collect::<Vec<_>>());
9000 fs::remove_file(path).expect("remove scratch file");
9001 }
9002
9003 #[test]
9009 fn a_file_from_another_format_says_which_format_it_is() {
9010 let older = path("older-format");
9011 let mut writer =
9012 Writer::create(&older, "items", vec![Field::new("id", LogicalType::Integer)])
9013 .expect("new file");
9014 let chunk = Chunk::new(vec![
9015 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9016 .expect("integers"),
9017 ])
9018 .expect("chunk");
9019 writer.append(&chunk).expect("page written");
9020 writer.finish().expect("commit");
9021
9022 let unreadable =
9026 READABLE.iter().copied().min().expect("at least one format is readable") - 1;
9027 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9028 file.seek(SeekFrom::Start(8)).expect("the version follows the magic");
9029 file.write_all(&unreadable.to_le_bytes()).expect("write an older version");
9030 drop(file);
9031 let complaint = Reader::open(&older).expect_err("an older format is refused").to_string();
9032 assert!(complaint.contains(&format!("format {unreadable}")), "{complaint}");
9033 assert!(complaint.contains(&format!("format {FORMAT}")), "{complaint}");
9034
9035 let mut file = OpenOptions::new().write(true).open(&older).expect("open for the header");
9036 file.seek(SeekFrom::Start(0)).expect("the magic is first");
9037 file.write_all(b"NOTRUDB!").expect("write another engine's magic");
9038 drop(file);
9039 let complaint = Reader::open(&older).expect_err("a foreign file is refused").to_string();
9040 assert!(complaint.contains("magic"), "{complaint}");
9041 assert!(!complaint.contains("format"), "a version has nothing to do with it: {complaint}");
9042 fs::remove_file(older).expect("remove scratch file");
9043 }
9044
9045 #[test]
9046 fn an_unfinished_or_damaged_file_does_not_answer_with_partial_rows() {
9047 let unfinished = path("unfinished");
9048 let mut writer =
9049 Writer::create(&unfinished, "items", vec![Field::new("id", LogicalType::Integer)])
9050 .expect("new file");
9051 let chunk = Chunk::new(vec![
9052 Vector::flat(LogicalType::Integer, Data::Int32(vec![1, 2, 3].into()))
9053 .expect("integers"),
9054 ])
9055 .expect("chunk");
9056 writer.append(&chunk).expect("page written");
9057 drop(writer);
9058 assert!(Reader::open(&unfinished).is_err(), "no directory was committed");
9059 fs::remove_file(unfinished).expect("remove scratch file");
9060
9061 let damaged = path("damaged");
9062 let mut writer =
9063 Writer::create(&damaged, "items", vec![Field::new("id", LogicalType::Integer)])
9064 .expect("new file");
9065 writer.append(&chunk).expect("page written");
9066 writer.finish().expect("commit");
9067 let reader = Reader::open(&damaged).expect("valid directory");
9068 let mut file =
9069 OpenOptions::new().write(true).open(&damaged).expect("open for a damaged page");
9070 file.seek(SeekFrom::Start(HEADER + 1)).expect("inside first page");
9071 file.write_all(&[255]).expect("damage one byte");
9072 assert!(reader.read(0, &[0]).is_err(), "page checksum rejects corruption");
9073 fs::remove_file(damaged).expect("remove scratch file");
9074 }
9075
9076 #[test]
9077 fn damaged_lazy_dictionary_payload_is_an_error() {
9078 let path = path("damaged-dictionary");
9079 let mut writer = Writer::create(
9080 &path,
9081 "items",
9082 vec![
9083 Field::required("id", LogicalType::Integer),
9084 Field::new("text", LogicalType::Varchar),
9085 ],
9086 )
9087 .expect("new file");
9088 writer.append(&sample()).expect("stripe written");
9089 writer.finish().expect("commit");
9090
9091 let reader = Reader::open(&path).expect("valid directory");
9092 let dictionary = reader.table.dictionaries[1].expect("string dictionary page");
9093 let mut header = [0; DICTIONARY_HEADER];
9096 read_at(&reader.file, dictionary.offset, &mut header).expect("dictionary header");
9097 let index_len = dictionary_index_len(&header);
9098 let rank_len = last_rank_end(&reader.file, dictionary.offset, &header);
9099 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9100 file.seek(SeekFrom::Start(dictionary.offset + index_len + rank_len))
9101 .expect("inside dictionary payload");
9102 file.write_all(&[255]).expect("damage dictionary payload");
9103
9104 let chunk = reader.read(0, &[1]).expect("code page and dictionary index remain valid");
9105 let error =
9106 chunk.validate_external().expect_err("payload corruption must reach the caller");
9107 assert!(error.message().contains("payload checksum differs"), "{error}");
9108 fs::remove_file(path).expect("remove scratch file");
9109 }
9110
9111 #[test]
9121 fn a_column_of_all_different_values_is_written_without_a_dictionary() {
9122 let path = path("dictionary-decide");
9123 let rows = 20_000;
9124 let unique =
9126 |row: usize| format!("{row:09} a value that appears exactly once in the table");
9127 let repeated = |row: usize| unique(row / 40);
9129 let mut writer = Writer::create(
9130 &path,
9131 "items",
9132 vec![
9133 Field::required("unique", LogicalType::Varchar),
9134 Field::required("repeated", LogicalType::Varchar),
9135 ],
9136 )
9137 .expect("new file");
9138 for part in (0..rows).step_by(1_000) {
9139 let span = part..(part + 1_000).min(rows);
9140 let left = span.clone().map(|row| Value::Varchar(unique(row))).collect::<Vec<_>>();
9141 let right = span.map(|row| Value::Varchar(repeated(row))).collect::<Vec<_>>();
9142 writer
9143 .append(
9144 &Chunk::new(vec![
9145 Vector::from_values(LogicalType::Varchar, &left).expect("strings"),
9146 Vector::from_values(LogicalType::Varchar, &right).expect("strings"),
9147 ])
9148 .expect("two columns"),
9149 )
9150 .expect("a part");
9151 }
9152 writer.finish().expect("commit");
9153
9154 let reader = Reader::open(&path).expect("reopen from disk");
9155 assert!(
9156 reader.table.dictionaries[0].is_none(),
9157 "a column with no repeats has nothing to say twice"
9158 );
9159 assert!(
9160 reader.table.dictionaries[1].is_some(),
9161 "a column whose values come round again keeps its dictionary"
9162 );
9163 let mut first = 0;
9164 for part in 0..reader.parts() {
9165 let chunk = reader.read(part, &[0, 1]).expect("a part");
9166 for row in 0..chunk.len() {
9167 assert_eq!(chunk.value_at(row, 0), Value::Varchar(unique(first + row)));
9168 assert_eq!(chunk.value_at(row, 1), Value::Varchar(repeated(first + row)));
9169 }
9170 first += chunk.len();
9171 }
9172 assert_eq!(first, rows, "every row was read back");
9173 let raw = (0..rows).map(|row| unique(row).len()).sum::<usize>();
9174 let size = fs::metadata(&path).expect("the file is there").len() as usize;
9175 assert!(size < raw, "a column without a dictionary is still encoded: {size} against {raw}");
9176 fs::remove_file(path).expect("remove scratch file");
9177 }
9178
9179 #[test]
9192 fn a_dictionary_over_many_blocks_checks_every_block_of_it() {
9193 let path = path("dictionary-blocks");
9194 let value = |row: usize| {
9195 let row = row.saturating_sub(8_000);
9196 format!("{row:07} a value long enough to be worth a payload block")
9197 };
9198 let parts = 40;
9199 let per_part = 1000;
9200 let mut writer =
9201 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9202 .expect("new file");
9203 for part in 0..parts {
9204 let values = (0..per_part)
9205 .map(|row| Value::Varchar(value(part * per_part + row)))
9206 .collect::<Vec<_>>();
9207 let chunk = Chunk::new(vec![
9208 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9209 ])
9210 .expect("matching rows");
9211 writer.append(&chunk).expect("a part");
9212 }
9213 writer.finish().expect("commit");
9214
9215 let reader = Reader::open(&path).expect("reopen from disk");
9216 let dictionary = reader.table.dictionaries[0].expect("string dictionary page");
9217 assert!(
9218 parts * per_part > TEXT_PAYLOAD_VALUES * 4,
9219 "the dictionary has to be several blocks for this to be testing anything"
9220 );
9221 for part in [0, parts - 1] {
9222 let chunk = reader.read(part, &[0]).expect("a part");
9223 chunk.validate_external().expect("every payload block checks out");
9224 assert_eq!(chunk.value_at(0, 0), Value::Varchar(value(part * per_part)));
9225 }
9226
9227 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9228 file.seek(SeekFrom::Start(dictionary.offset + u64::from(dictionary.length) - 4))
9229 .expect("the last bytes of the page are payload");
9230 file.write_all(&[255]).expect("damage the last payload block");
9231 let reader = Reader::open(&path).expect("the directory and the index are untouched");
9232 let chunk = reader.read(parts - 1, &[0]).expect("the code page remains valid");
9233 let error = chunk.validate_external().expect_err("the damage must reach the caller");
9234 assert!(error.message().contains("payload checksum differs"), "{error}");
9235 fs::remove_file(path).expect("remove scratch file");
9236 }
9237
9238 #[test]
9252 fn values_of_different_lengths_read_back_out_of_packed_offsets() {
9253 let path = path("dictionary-offsets");
9254 let value = |row: usize| {
9255 let row = row % 5_000;
9256 if row % 511 == 3 { String::new() } else { "x".repeat(row % 97) + &format!("{row:05}") }
9257 };
9258 let rows = 6_000;
9259 let mut writer =
9260 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9261 .expect("new file");
9262 let values = (0..rows).map(|row| Value::Varchar(value(row))).collect::<Vec<_>>();
9263 for part in values.chunks(1_000) {
9264 let chunk =
9265 Chunk::new(vec![Vector::from_values(LogicalType::Varchar, part).expect("strings")])
9266 .expect("matching rows");
9267 writer.append(&chunk).expect("a part");
9268 }
9269 writer.finish().expect("commit");
9270
9271 let reader = Reader::open(&path).expect("reopen from disk");
9272 assert!(
9273 rows > TEXT_PAYLOAD_VALUES * 4,
9274 "the dictionary has to be several blocks for this to be testing anything"
9275 );
9276 for part in 0..rows / 1_000 {
9277 let chunk = reader.read(part, &[0]).expect("a part");
9278 for row in 0..1_000 {
9279 let row = part * 1_000 + row;
9280 assert_eq!(
9281 chunk.value_at(row % 1_000, 0),
9282 Value::Varchar(value(row)),
9283 "value {row}"
9284 );
9285 }
9286 }
9287 fs::remove_file(path).expect("remove scratch file");
9288 }
9289
9290 #[test]
9302 fn a_global_dictionary_is_opened_once_however_many_workers_ask_at_once() {
9303 let path = path("dictionary-once");
9304 let parts = 8;
9305 let per_part = 500;
9306 let value =
9307 |row: usize| format!("{row:07} a value long enough to be worth a payload block");
9308 let mut writer =
9309 Writer::create(&path, "items", vec![Field::required("text", LogicalType::Varchar)])
9310 .expect("new file");
9311 for part in 0..parts {
9312 let values = (0..per_part)
9313 .map(|row| Value::Varchar(value(part * per_part + row)))
9314 .collect::<Vec<_>>();
9315 let chunk = Chunk::new(vec![
9316 Vector::from_values(LogicalType::Varchar, &values).expect("strings"),
9317 ])
9318 .expect("matching rows");
9319 writer.append(&chunk).expect("a part");
9320 }
9321 writer.finish().expect("commit");
9322
9323 let reader = Reader::open(&path).expect("reopen from disk");
9324 assert!(reader.table.dictionaries[0].is_some(), "the column has to have one to share");
9325 assert_eq!(reader.reads().dictionaries, 0, "opening the file does not open a dictionary");
9326
9327 let workers = 16;
9328 let gate = std::sync::Barrier::new(workers);
9329 std::thread::scope(|scope| {
9330 for worker in 0..workers {
9331 let reader = reader.clone();
9332 let gate = &gate;
9333 scope.spawn(move || {
9334 gate.wait();
9335 let chunk = reader.read(worker % parts, &[0]).expect("a part");
9336 assert_eq!(
9337 chunk.value_at(0, 0),
9338 Value::Varchar(value((worker % parts) * per_part))
9339 );
9340 });
9341 }
9342 });
9343
9344 assert_eq!(reader.reads().dictionaries, 1, "sixteen workers, one dictionary, one open");
9345 fs::remove_file(path).expect("remove scratch file");
9346 }
9347
9348 #[test]
9353 fn a_damaged_sorted_order_is_an_error() {
9354 let path = path("damaged-order");
9355 let mut writer = Writer::create(
9356 &path,
9357 "items",
9358 vec![
9359 Field::required("id", LogicalType::Integer),
9360 Field::new("text", LogicalType::Varchar),
9361 ],
9362 )
9363 .expect("new file");
9364 writer.append(&sample()).expect("stripe written");
9365 writer.finish().expect("commit");
9366
9367 let reader = Reader::open(&path).expect("valid directory");
9368 let page = reader.table.dictionaries[1].expect("string dictionary page");
9369 let mut header = [0; DICTIONARY_HEADER];
9370 read_at(&reader.file, page.offset, &mut header).expect("dictionary header");
9371 let index_len = dictionary_index_len(&header);
9372 let mut file = OpenOptions::new().write(true).open(&path).expect("open dictionary page");
9373 file.seek(SeekFrom::Start(page.offset + index_len)).expect("the first head");
9374 file.write_all(&[255]).expect("damage the order");
9375
9376 let dictionary = reader.dictionary(1).expect("read").expect("a string column has one");
9377 let error = dictionary.compare_rank(0, b"anything").expect_err("a damaged order is caught");
9378 assert!(error.message().contains("rank checksum differs"), "{error}");
9379 fs::remove_file(path).expect("remove scratch file");
9380 }
9381
9382 #[test]
9386 fn a_global_dictionary_carries_the_sorted_order_of_its_values() {
9387 let spellings = ["overlong1z", "b", "", "overlong1a", "overlong", "ab", "a", "overlong1"];
9390 let path = path("dictionary-order");
9391 let mut writer =
9392 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9393 .expect("new file");
9394 writer
9395 .append(
9396 &Chunk::new(vec![
9397 Vector::from_values(
9398 LogicalType::Varchar,
9399 &spellings.map(|text| Value::Varchar(text.into())),
9400 )
9401 .expect("strings"),
9402 ])
9403 .expect("one column"),
9404 )
9405 .expect("stripe written");
9406 writer.finish().expect("commit");
9407
9408 let reader = Reader::open(&path).expect("valid directory");
9409 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9410 let count = dictionary.ranks().expect("a v10 file stores one");
9411 assert_eq!(count, spellings.len(), "every distinct value has a rank");
9412 let order = (0..count)
9413 .map(|rank| dictionary.code_at_rank(rank).expect("a code"))
9414 .collect::<Vec<_>>();
9415 let mut seen = order.clone();
9416 seen.sort_unstable();
9417 assert_eq!(seen, (0..spellings.len() as u32).collect::<Vec<_>>(), "a permutation of codes");
9418
9419 let ranked = order
9420 .iter()
9421 .map(|&code| {
9422 dictionary.try_bytes_at(code as usize).expect("read").expect("a value").to_vec()
9423 })
9424 .collect::<Vec<_>>();
9425 let mut expected = spellings.map(|text| text.as_bytes().to_vec()).to_vec();
9426 expected.sort();
9427 assert_eq!(ranked, expected, "rank order is value order");
9428
9429 for (rank, value) in expected.iter().enumerate() {
9432 assert_eq!(
9433 dictionary.compare_rank(rank, value).expect("compare"),
9434 Ordering::Equal,
9435 "rank {rank} is its own value"
9436 );
9437 if rank > 0 {
9438 assert_eq!(
9439 dictionary.compare_rank(rank - 1, value).expect("compare"),
9440 Ordering::Less,
9441 "rank {rank} follows the one before it"
9442 );
9443 }
9444 }
9445 fs::remove_file(path).expect("remove scratch file");
9446 }
9447
9448 #[test]
9456 fn a_dictionary_sweep_reads_every_value_and_keeps_it_under_the_budget() {
9457 let path = path("dictionary-sweep");
9458 let spellings = (0..2_500)
9461 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9462 .collect::<Vec<_>>();
9463 let mut writer =
9464 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9465 .expect("new file");
9466 for part in spellings.chunks(1_024) {
9469 writer
9470 .append(
9471 &Chunk::new(vec![
9472 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9473 ])
9474 .expect("one column"),
9475 )
9476 .expect("stripe written");
9477 }
9478 writer.finish().expect("commit");
9479
9480 let reader = Reader::open(&path).expect("valid directory");
9481 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9482 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9483
9484 let resting = dictionary.footprint();
9485 let mut swept: Vec<Vec<u8>> = Vec::new();
9486 let mut at = 0;
9487 let mut calls = 0;
9488 while at < dictionary.len() {
9489 let stopped = dictionary
9490 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9491 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9492 swept.push(text.to_vec());
9493 Ok(())
9494 })
9495 .expect("a sweep reads");
9496 assert!(stopped > at, "a sweep moves");
9497 at = stopped;
9498 calls += 1;
9499 }
9500 assert_eq!(calls, 3, "a sweep hands over one block at a time");
9501 let after = dictionary.footprint();
9502 assert!(after > resting, "a sweep under the budget keeps what it decoded");
9503
9504 let read = (0..dictionary.len())
9505 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9506 .collect::<Vec<_>>();
9507 assert_eq!(swept, read, "a sweep answers what a point read answers");
9508 assert_eq!(dictionary.footprint(), after, "a point read of a kept block decodes nothing");
9509 fs::remove_file(path).expect("remove scratch file");
9510 }
9511
9512 #[test]
9523 fn a_sweep_over_a_block_with_a_short_second_run_reads_what_a_point_read_reads() {
9524 let path = path("dictionary-sweep-short-run");
9525 let spellings = (0..2_800)
9526 .map(|index| Value::Varchar(format!("value {index:08} {}", "x".repeat(index % 40))))
9527 .collect::<Vec<_>>();
9528 let mut writer =
9529 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9530 .expect("new file");
9531 for part in spellings.chunks(1_024) {
9532 writer
9533 .append(
9534 &Chunk::new(vec![
9535 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9536 ])
9537 .expect("one column"),
9538 )
9539 .expect("stripe written");
9540 }
9541 writer.finish().expect("commit");
9542
9543 let reader = Reader::open(&path).expect("valid directory");
9544 let dictionary = reader.dictionary(0).expect("read").expect("a string column has one");
9545 assert_eq!(dictionary.len(), spellings.len(), "every value is distinct");
9546 let last = dictionary.len() % TEXT_PAYLOAD_VALUES;
9547 assert!(last > TEXT_OFFSET_RUN, "the last block has to reach into a second run of offsets");
9548 assert!(last < TEXT_PAYLOAD_VALUES, "and that second run has to be short of a whole one");
9549
9550 let mut swept: Vec<Vec<u8>> = Vec::new();
9551 let mut at = 0;
9552 while at < dictionary.len() {
9553 let stopped = dictionary
9554 .sweep_text(at, dictionary.len(), &mut |index: usize, text: &[u8]| {
9555 assert_eq!(index, swept.len(), "a sweep hands its values over in order");
9556 swept.push(text.to_vec());
9557 Ok(())
9558 })
9559 .expect("a sweep reads");
9560 assert!(stopped > at, "a sweep moves");
9561 at = stopped;
9562 }
9563 let read = (0..dictionary.len())
9564 .map(|code| dictionary.try_bytes_at(code).expect("read").expect("a value").to_vec())
9565 .collect::<Vec<_>>();
9566 assert_eq!(swept, read, "a sweep answers what a point read answers");
9567 fs::remove_file(path).expect("remove scratch file");
9568 }
9569
9570 #[test]
9580 fn narrowing_a_page_takes_what_fits_and_refuses_what_does_not() {
9581 assert_eq!(fit::<i8>(&[]).expect("an empty page fits anything"), Vec::<i8>::new());
9582 assert_eq!(fit::<i8>(&[-128, 0, 127]).expect("the edges fit"), vec![-128_i8, 0, 127]);
9583 fit::<i8>(&[128]).expect_err("one past the top does not fit");
9584 fit::<i8>(&[-129]).expect_err("one past the bottom does not fit");
9585 assert_eq!(fit::<u8>(&[0, 255]).expect("the edges fit"), vec![0_u8, 255]);
9586 fit::<u8>(&[256]).expect_err("one past the top does not fit");
9587 fit::<u8>(&[-1]).expect_err("a negative does not fit an unsigned page");
9588 assert_eq!(
9589 fit::<i16>(&[-32_768, 0, 32_767]).expect("the edges fit"),
9590 vec![-32_768_i16, 0, 32_767]
9591 );
9592 fit::<i16>(&[32_768]).expect_err("one past the top does not fit");
9593 fit::<i16>(&[-32_769]).expect_err("one past the bottom does not fit");
9594 assert_eq!(fit::<u16>(&[0, 65_535]).expect("the edges fit"), vec![0_u16, 65_535]);
9595 fit::<u16>(&[65_536]).expect_err("one past the top does not fit");
9596 fit::<u16>(&[-1]).expect_err("a negative does not fit an unsigned page");
9597 assert_eq!(
9598 fit::<i32>(&[i64::from(i32::MIN), 0, i64::from(i32::MAX)]).expect("the edges fit"),
9599 vec![i32::MIN, 0, i32::MAX]
9600 );
9601 fit::<i32>(&[i64::from(i32::MAX) + 1]).expect_err("one past the top does not fit");
9602 fit::<i32>(&[i64::from(i32::MIN) - 1]).expect_err("one past the bottom does not fit");
9603 assert_eq!(
9604 fit::<u32>(&[0, 4_294_967_295]).expect("the edges fit"),
9605 vec![0_u32, 4_294_967_295]
9606 );
9607 fit::<u32>(&[4_294_967_296]).expect_err("one past the top does not fit");
9608 fit::<u32>(&[-1]).expect_err("a negative does not fit an unsigned page");
9609
9610 fit::<i8>(&[0, 1, 2, 128, 3]).expect_err("one bad value spoils the page");
9613 }
9614
9615 #[test]
9622 fn the_residue_agrees_with_a_checked_conversion_everywhere() {
9623 for value in -70_000_i64..70_000 {
9624 assert_eq!(fit::<i8>(&[value]).is_ok(), i8::try_from(value).is_ok(), "{value} as i8");
9625 assert_eq!(fit::<u8>(&[value]).is_ok(), u8::try_from(value).is_ok(), "{value} as u8");
9626 assert_eq!(fit::<i16>(&[value]).is_ok(), i16::try_from(value).is_ok(), "{value} i16");
9627 assert_eq!(fit::<u16>(&[value]).is_ok(), u16::try_from(value).is_ok(), "{value} u16");
9628 }
9629 let wide = [i64::MIN, i64::MIN + 1, i64::from(i32::MIN), 0, i64::from(u32::MAX), i64::MAX];
9630 for edge in wide {
9631 for step in -2_i64..=2 {
9632 let value = edge.saturating_add(step);
9633 assert_eq!(
9634 fit::<i32>(&[value]).is_ok(),
9635 i32::try_from(value).is_ok(),
9636 "{value} as i32"
9637 );
9638 assert_eq!(
9639 fit::<u32>(&[value]).is_ok(),
9640 u32::try_from(value).is_ok(),
9641 "{value} as u32"
9642 );
9643 }
9644 }
9645 }
9646
9647 #[test]
9655 fn a_dictionary_at_its_budget_sweeps_without_keeping() {
9656 let path = path("dictionary-budget");
9657 let spellings = (0..2_500)
9658 .map(|index| Value::Varchar(format!("value {index:08} {}", "y".repeat(index % 40))))
9659 .collect::<Vec<_>>();
9660 let mut writer =
9661 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9662 .expect("new file");
9663 for part in spellings.chunks(1_024) {
9664 writer
9665 .append(
9666 &Chunk::new(vec![
9667 Vector::from_values(LogicalType::Varchar, part).expect("strings"),
9668 ])
9669 .expect("one column"),
9670 )
9671 .expect("stripe written");
9672 }
9673 writer.finish().expect("commit");
9674
9675 let reader = Reader::open(&path).expect("valid directory");
9676 let page = reader.table.dictionaries[0].expect("a string column has one");
9677 let file = Arc::clone(&reader.file);
9678 let starved = open_global_dictionary(file, page, &LogicalType::Varchar, 0)
9679 .expect("a dictionary opens whatever it may keep");
9680
9681 let resting = starved.footprint();
9682 let mut swept: Vec<Vec<u8>> = Vec::new();
9683 let mut at = 0;
9684 while at < starved.len() {
9685 at = starved
9686 .sweep_text(at, starved.len(), &mut |_index: usize, text: &[u8]| {
9687 swept.push(text.to_vec());
9688 Ok(())
9689 })
9690 .expect("a sweep reads");
9691 }
9692 assert_eq!(swept.len(), spellings.len(), "a starved sweep still reads every value");
9693 assert_eq!(starved.footprint(), resting, "and keeps no block it decoded");
9694
9695 let generous = reader.dictionary(0).expect("read").expect("a string column has one");
9696 let read = (0..generous.len())
9697 .map(|code| generous.try_bytes_at(code).expect("read").expect("a value").to_vec())
9698 .collect::<Vec<_>>();
9699 assert_eq!(swept, read, "a starved sweep answers what a point read answers");
9700 fs::remove_file(path).expect("remove scratch file");
9701 }
9702
9703 #[test]
9704 fn damaged_membership_cannot_skip_a_string_page() {
9705 let path = path("damaged-membership");
9706 let mut writer = Writer::create(
9707 &path,
9708 "items",
9709 vec![
9710 Field::required("id", LogicalType::Integer),
9711 Field::new("text", LogicalType::Varchar),
9712 ],
9713 )
9714 .expect("new file");
9715 writer.append(&sample()).expect("stripe written");
9716 writer.finish().expect("commit");
9717
9718 let reader = Reader::open(&path).expect("valid directory");
9719 let membership = reader.table.stripes[0].memberships[1].expect("string membership");
9720 let mut file = OpenOptions::new().write(true).open(&path).expect("open membership page");
9721 file.seek(SeekFrom::Start(membership.offset)).expect("membership start");
9722 file.write_all(&[255]).expect("damage membership");
9723 let error = reader.skips_codes(0, 1, &[3]).expect_err("corruption must not skip rows");
9724 assert!(error.message().contains("membership page checksum differs"), "{error}");
9725 fs::remove_file(path).expect("remove scratch file");
9726 }
9727
9728 #[test]
9729 fn membership_delta_stream_is_sorted_exact_and_bounded() {
9730 let unique = unique_codes(&[900, 4, 4, 72, 9, u32::MAX]);
9731 assert_eq!(unique, [4, 9, 72, 900, u32::MAX]);
9732 let encoded = encode_membership(&unique);
9733 assert_eq!(
9734 decode_membership(&encoded).expect("valid membership"),
9735 [4, 9, 72, 900, u32::MAX]
9736 );
9737 let merged = merged_codes(vec![vec![4, 900], vec![9, 900, u32::MAX], vec![72]]);
9740 assert_eq!(merged, [4, 9, 72, 900, u32::MAX]);
9741 assert_eq!(
9742 decode_membership(&encode_membership(&merged)).expect("valid membership"),
9743 unique
9744 );
9745 assert!(decode_membership(&[1, 0x80]).is_err(), "a truncated varint is invalid");
9746 assert!(
9747 decode_membership(&[1, 0xff, 0xff, 0xff, 0xff, 0x10]).is_err(),
9748 "a value past u32 is invalid"
9749 );
9750 }
9751
9752 #[test]
9753 fn a_global_dictionary_may_be_larger_than_one_column_page() {
9754 let dictionary = Page {
9755 offset: HEADER,
9756 length: u32::try_from(MAX_PAGE + 1).expect("the page bound fits on disk"),
9757 hash: 0,
9758 };
9759 let table = Table {
9760 name: "items".to_owned(),
9761 fields: vec![Field::new("text", LogicalType::Varchar)],
9762 stripes: Vec::new(),
9763 rows: 0,
9764 dictionaries: vec![Some(dictionary)],
9765 distincts: vec![None],
9766 frequencies: vec![None],
9767 clustering: None,
9768 generation: 1,
9769 sections: Vec::new(),
9770 };
9771 let directory = encode_directory(&table).expect("directory");
9772 let file_size = dictionary.offset + u64::from(dictionary.length) + 1;
9773
9774 let decoded = decode_directory(&directory, file_size).expect("large lazy dictionary");
9775 assert_eq!(decoded.dictionaries[0].expect("dictionary").length, dictionary.length);
9776 }
9777
9778 #[test]
9779 fn a_column_with_one_value_everywhere_costs_almost_nothing_a_row() {
9780 let path = path("constant-codes");
9781 let mut writer =
9782 Writer::create(&path, "items", vec![Field::new("text", LogicalType::Varchar)])
9783 .expect("new file");
9784 let empty = vec![Value::Varchar(String::new()); 1024];
9785 for _ in 0..4 {
9786 let column = Vector::from_values(LogicalType::Varchar, &empty).expect("strings");
9787 writer.append(&Chunk::new(vec![column]).expect("one column")).expect("a part");
9788 }
9789 writer.finish().expect("commit");
9790
9791 let reader = Reader::open(&path).expect("valid directory");
9792 let pages = reader.layout().columns.first().expect("one column").pages;
9793 assert!(pages < 256, "{pages} bytes of pages for 4,096 rows of one value");
9797 let read = reader.read(3, &[0]).expect("the last part back");
9798 assert_eq!(read.value_at(0, 0), Value::Varchar(String::new()));
9799 assert_eq!(read.value_at(1023, 0), Value::Varchar(String::new()));
9800 fs::remove_file(path).expect("remove scratch file");
9801 }
9802
9803 #[test]
9804 fn a_cascade_value_too_wide_for_its_column_is_refused_rather_than_cut() {
9805 let over = vec![i64::from(i32::MAX) + 1];
9808 let error = narrowed(&LogicalType::Integer, over).expect_err("a page that disagrees");
9809 assert!(format!("{error}").contains("not of its type"), "{error}");
9810 assert!(narrowed(&LogicalType::BigInt, vec![i64::MIN]).is_ok(), "bigint holds all of i64");
9811 assert!(narrowed(&LogicalType::Varchar, vec![0]).is_err(), "strings are not integers");
9812 }
9813
9814 #[test]
9815 fn a_code_stream_the_cascade_cannot_shrink_is_left_alone() {
9816 let mut state: u32 = 0x9e37_79b9;
9820 let spread: Vec<u32> = (0..1024)
9821 .map(|_| {
9822 state ^= state << 13;
9823 state ^= state >> 17;
9824 state ^= state << 5;
9825 state
9826 })
9827 .collect();
9828 assert_eq!(encoded_codes(&spread).expect("no failure"), None);
9829 let near: Vec<u32> = (0..1024).collect();
9830 let coded = encoded_codes(&near).expect("no failure").expect("counting up is packable");
9831 assert!(coded.len() < near.len() * 4, "{} bytes for a run of 1,024", coded.len());
9832 }
9833
9834 #[test]
9840 fn two_writes_of_the_same_rows_give_the_same_bytes() {
9841 fn written(path: &PathBuf) {
9842 let fields = (0..40)
9843 .map(|column| {
9844 let ty =
9845 if column % 4 == 0 { LogicalType::Varchar } else { LogicalType::BigInt };
9846 Field::new(format!("c{column}"), ty)
9847 })
9848 .collect::<Vec<_>>();
9849 let mut writer = Writer::create(path, "wide", fields).expect("new file");
9850 for part in 0..70_u64 {
9851 let columns = (0..40)
9852 .map(|column| {
9853 let values = (0..64_u64)
9854 .map(|row| {
9855 let seed = part.wrapping_mul(31).wrapping_add(row);
9856 if column % 4 == 0 {
9857 Value::Varchar(format!("v{}", seed % 17))
9858 } else {
9859 Value::BigInt(i64::try_from(seed % 97).expect("small"))
9860 }
9861 })
9862 .collect::<Vec<_>>();
9863 let ty = if column % 4 == 0 {
9864 LogicalType::Varchar
9865 } else {
9866 LogicalType::BigInt
9867 };
9868 Vector::from_values(ty, &values).expect("a column")
9869 })
9870 .collect::<Vec<_>>();
9871 writer.append(&Chunk::new(columns).expect("forty columns")).expect("a part");
9872 }
9873 writer.finish().expect("commit");
9874 }
9875
9876 let first = path("repeatable-one");
9877 let second = path("repeatable-two");
9878 written(&first);
9879 written(&second);
9880 let left = fs::read(&first).expect("the first file");
9881 let right = fs::read(&second).expect("the second file");
9882 assert_eq!(left.len(), right.len(), "two writes of the same rows differ in length");
9883 assert!(left == right, "two writes of the same rows differ in their bytes");
9884
9885 let reader = Reader::open(&first).expect("valid directory");
9888 assert_eq!(reader.table().rows(), 70 * 64);
9889 let read = reader.read(0, &[0, 1]).expect("the first part back");
9890 assert_eq!(read.value_at(0, 0), Value::Varchar("v0".to_owned()));
9891 assert_eq!(read.value_at(0, 1), Value::BigInt(0));
9892 fs::remove_file(first).expect("remove scratch file");
9893 fs::remove_file(second).expect("remove scratch file");
9894 }
9895
9896 fn three_tables(path: &PathBuf) {
9898 let writer = Writer::create(
9899 path,
9900 "region",
9901 vec![
9902 Field::new("r_key", LogicalType::Integer),
9903 Field::new("r_name", LogicalType::Varchar),
9904 ],
9905 )
9906 .expect("new file");
9907 let mut writer = writer;
9908 writer
9909 .append(
9910 &Chunk::new(vec![
9911 Vector::from_values(
9912 LogicalType::Integer,
9913 &[Value::Integer(0), Value::Integer(1)],
9914 )
9915 .expect("keys"),
9916 Vector::from_values(
9917 LogicalType::Varchar,
9918 &[Value::Varchar("AFRICA".to_owned()), Value::Varchar("ASIA".to_owned())],
9919 )
9920 .expect("names"),
9921 ])
9922 .expect("two columns"),
9923 )
9924 .expect("a part");
9925 let mut writer = writer
9926 .next("empty", vec![Field::new("nothing", LogicalType::BigInt)])
9927 .expect("a second table");
9928 writer
9929 .append(
9930 &Chunk::new(vec![
9931 Vector::from_values(LogicalType::BigInt, &[Value::BigInt(7)]).expect("a row"),
9932 ])
9933 .expect("one column"),
9934 )
9935 .expect("a part");
9936 let mut writer =
9937 writer.next("wide", vec![Field::new("n", LogicalType::BigInt)]).expect("a third table");
9938 for part in 0..70_i64 {
9939 let values = (0..64).map(|row| Value::BigInt(part * 64 + row)).collect::<Vec<_>>();
9940 writer
9941 .append(
9942 &Chunk::new(vec![
9943 Vector::from_values(LogicalType::BigInt, &values).expect("a column"),
9944 ])
9945 .expect("one column"),
9946 )
9947 .expect("a part");
9948 }
9949 writer.finish().expect("commit");
9950 }
9951
9952 #[test]
9953 fn three_tables_in_one_file_read_back_by_name() {
9954 let file = path("three-tables");
9955 three_tables(&file);
9956 let catalog = Catalog::open(&file).expect("a committed catalog");
9957 assert_eq!(catalog.names().collect::<Vec<_>>(), ["region", "empty", "wide"]);
9958
9959 let region = catalog.table("region").expect("the first table");
9960 assert_eq!(region.table().rows(), 2);
9961 assert_eq!(
9962 region.read(0, &[1]).expect("names").value_at(1, 0),
9963 Value::Varchar("ASIA".to_owned())
9964 );
9965
9966 let wide = catalog.table("wide").expect("the third table");
9967 assert_eq!(wide.table().rows(), 70 * 64);
9968 assert_eq!(wide.read(0, &[0]).expect("the first part").value_at(0, 0), Value::BigInt(0));
9969
9970 let empty = catalog.table("empty").expect("the second table");
9973 assert_eq!(empty.table().rows(), 1);
9974 assert_eq!(empty.read(0, &[0]).expect("the row").value_at(0, 0), Value::BigInt(7));
9975
9976 fs::remove_file(file).expect("remove scratch file");
9977 }
9978
9979 #[test]
9980 fn a_name_the_file_does_not_hold_is_an_error_rather_than_the_first_table() {
9981 let file = path("three-tables-missing");
9982 three_tables(&file);
9983 let catalog = Catalog::open(&file).expect("a committed catalog");
9984 let error = catalog.table("nation").expect_err("no such table");
9985 assert!(error.message().contains("nation"), "{}", error.message());
9986 fs::remove_file(file).expect("remove scratch file");
9987 }
9988
9989 #[test]
9990 fn a_file_of_three_tables_will_not_open_as_one() {
9991 let file = path("three-tables-unnamed");
9992 three_tables(&file);
9993 let error = Reader::open(&file).expect_err("more than one table");
9994 assert!(error.message().contains("more than one table"), "{}", error.message());
9995 fs::remove_file(file).expect("remove scratch file");
9996 }
9997
9998 #[test]
10000 fn decimals_of_every_storage_width_round_trip() {
10001 let file = path("decimals");
10002 let widths = [(4_u8, 2_u8), (9, 2), (18, 4), (38, 6)];
10003 let fields = widths
10004 .iter()
10005 .enumerate()
10006 .map(|(index, (width, scale))| {
10007 Field::new(
10008 format!("d{index}"),
10009 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10010 )
10011 })
10012 .collect::<Vec<_>>();
10013 let mut writer = Writer::create(&file, "money", fields).expect("new file");
10014 let rows: [i128; 3] = [-1234, 0, 999];
10015 let columns = widths
10016 .iter()
10017 .map(|(width, scale)| {
10018 let values = rows
10019 .iter()
10020 .map(|unscaled| Value::Decimal {
10021 unscaled: *unscaled,
10022 width: *width,
10023 scale: *scale,
10024 })
10025 .collect::<Vec<_>>();
10026 Vector::from_values(
10027 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10028 &values,
10029 )
10030 .expect("a decimal column")
10031 })
10032 .collect::<Vec<_>>();
10033 writer.append(&Chunk::new(columns).expect("four columns")).expect("a part");
10034 writer.finish().expect("commit");
10035
10036 let reader = Reader::open(&file).expect("a committed file");
10037 for (index, (width, scale)) in widths.iter().enumerate() {
10038 assert_eq!(
10039 reader.table().fields()[index].ty,
10040 LogicalType::decimal(*width, *scale).expect("a decimal type"),
10041 "column {index} came back as another type"
10042 );
10043 let column = reader.read(0, &[index]).expect("the column");
10044 for (row, unscaled) in rows.iter().enumerate() {
10045 assert_eq!(
10046 column.value_at(row, 0),
10047 Value::Decimal { unscaled: *unscaled, width: *width, scale: *scale },
10048 "column {index} row {row}"
10049 );
10050 }
10051 }
10052 fs::remove_file(file).expect("remove scratch file");
10053 }
10054
10055 #[test]
10056 fn two_tables_of_one_name_are_refused_before_anything_is_committed() {
10057 let file = path("two-of-a-name");
10058 let writer = Writer::create(&file, "t", vec![Field::new("a", LogicalType::BigInt)])
10059 .expect("new file");
10060 let error = writer
10061 .next("t", vec![Field::new("a", LogicalType::BigInt)])
10062 .expect_err("the same name twice");
10063 assert!(error.message().contains("same name"), "{}", error.message());
10064 fs::remove_file(file).expect("remove scratch file");
10065 }
10066
10067 #[test]
10068 fn opening_the_catalog_reads_no_table_directory() {
10069 let file = path("catalog-only");
10070 three_tables(&file);
10071 let catalog = Catalog::open(&file).expect("a committed catalog");
10072 assert_eq!(catalog.opening.reads, 2, "opening the catalog read more than the slot");
10075 assert_eq!(catalog.names().len(), 3);
10076 fs::remove_file(file).expect("remove scratch file");
10077 }
10078
10079 #[test]
10090 fn the_checksum_answers_what_it_has_always_answered() {
10091 let bytes: Vec<u8> =
10092 (0..1000_u32).map(|at| (at.wrapping_mul(31).wrapping_add(7) % 251) as u8).collect();
10093 for (length, expected) in [
10094 (0, 0xef46_db37_51d8_e999),
10095 (1, 0xa96c_7f0c_e858_bbb7),
10096 (3, 0x56e6_9576_32a4_87f9),
10097 (4, 0xc60d_15b1_e3ff_8f04),
10098 (5, 0x8088_1585_8624_dd4e),
10099 (7, 0xafbe_fc3d_6c6f_9a8e),
10100 (8, 0x3da5_c7aa_2696_83e0),
10101 (9, 0x465e_c429_b13c_3892),
10102 (15, 0xdee8_9d8a_065a_6233),
10103 (16, 0x1330_489a_7767_9c80),
10104 (31, 0x3391_303d_485e_846e),
10105 (32, 0x40b7_aff7_5d45_bbc8),
10106 (33, 0x4997_cae4_951c_17a5),
10107 (39, 0x5807_28fd_5c14_5739),
10108 (40, 0xf95c_f6f5_c08a_3d3b),
10109 (63, 0x2944_b4da_fc69_b206),
10110 (64, 0xbb76_f6ef_19bd_5a1b),
10111 (65, 0x814e_0c65_4a9f_d640),
10112 (127, 0x00de_aab1_31cf_f89b),
10113 (1000, 0x9e33_00c1_cde3_c58d),
10114 ] {
10115 assert_eq!(checksum(&bytes[..length]), expected, "the checksum of {length} bytes");
10116 }
10117 assert_eq!(checksum(b"the quick brown fox jumps over the lazy dog"), 0xed71_4233_c5a9_a792);
10118 }
10119 #[test]
10126 fn a_declared_order_comes_back_out_of_the_file() {
10127 let path = path("clustered");
10128 let shipped = vec![
10129 Field::new("key", LogicalType::BigInt),
10130 Field::new("line", LogicalType::Integer),
10131 Field::new("shipdate", LogicalType::Date),
10132 ];
10133 let plain = vec![Field::new("a", LogicalType::Integer)];
10134 let stage_zero = Clustering::new(vec![2, 0, 1], Width::Month, &shipped).expect("valid");
10135
10136 let mut writer = Writer::create(&path, "lineitem", shipped)
10137 .expect("new file")
10138 .declare(stage_zero.clone())
10139 .expect("the columns are the table's");
10140 let column = |ty: LogicalType, values: &[Value]| {
10141 Vector::from_values(ty, values).expect("the values match the type")
10142 };
10143 writer
10144 .append(
10145 &Chunk::new(vec![
10146 column(
10147 LogicalType::BigInt,
10148 &[Value::BigInt(0), Value::BigInt(1), Value::BigInt(2), Value::BigInt(3)],
10149 ),
10150 column(
10151 LogicalType::Integer,
10152 &[
10153 Value::Integer(1),
10154 Value::Integer(1),
10155 Value::Integer(1),
10156 Value::Integer(1),
10157 ],
10158 ),
10159 column(
10160 LogicalType::Date,
10161 &[Value::Date(0), Value::Date(1), Value::Date(2), Value::Date(3)],
10162 ),
10163 ])
10164 .expect("three columns"),
10165 )
10166 .expect("four rows");
10167 let mut writer = writer.next("nation", plain).expect("a second table");
10168 writer
10169 .append(
10170 &Chunk::new(vec![column(LogicalType::Integer, &[Value::Integer(7)])])
10171 .expect("one column"),
10172 )
10173 .expect("one row");
10174 writer.finish().expect("commit");
10175
10176 let catalog = Catalog::open(&path).expect("reopen");
10177 let lineitem = catalog.table("lineitem").expect("the clustered table");
10178 assert_eq!(lineitem.table().clustering(), Some(&stage_zero));
10179 let nation = catalog.table("nation").expect("the plain table");
10180 assert_eq!(nation.table().clustering(), None, "nobody declared one here");
10181
10182 assert_eq!(lineitem.table().rows(), 4);
10185 assert_eq!(nation.table().rows(), 1);
10186 fs::remove_file(&path).ok();
10187 }
10188
10189 #[test]
10191 fn a_declaration_off_the_end_of_the_table_never_reaches_the_file() {
10192 let path = path("clustered-bad");
10193 let writer = Writer::create(&path, "items", vec![Field::new("a", LogicalType::Integer)])
10194 .expect("new file");
10195 let four =
10196 (0..4).map(|at| Field::new(format!("c{at}"), LogicalType::Integer)).collect::<Vec<_>>();
10197 let wrong = Clustering::new(vec![3], Width::Exact, &four).expect("valid against four");
10198 assert!(writer.declare(wrong).is_err(), "the table has one column, not four");
10199 fs::remove_file(&path).ok();
10200 }
10201}