rudb_native/section.rs
1//! The section table: one general mechanism for carrying a graph structure in a rudb file.
2//!
3//! spec/graph/03-the-file-format.md section 3.2 asks for one mechanism and three section kinds
4//! rather than three mechanisms. A section is an opaque payload with a kind, an identity, a
5//! generation stamp and a list of extents, and this module is the whole of what the format knows
6//! about one. What a key map or a forward link *means* lives in `rudb-graph` at rank 5, which is
7//! below the format on purpose: a key map that could see a page would be a key map that could only
8//! be tested through a file.
9//!
10//! Three rules make the mechanism the last one the format needs.
11//!
12//! A reader ignores a kind it does not know. That is what [`Section::kind`] being eight opaque
13//! bytes rather than an enum is for: a build that meets `RUDBAJ1\0` before backward adjacency
14//! exists carries the entry through, does not read the payload, and answers the query without it.
15//! Section 3.1 guarantees the answer is the same either way, so ignoring is always available and no
16//! future section kind needs another format bump.
17//!
18//! A section is a list of extents of at most [`MAX_EXTENT`] bytes, each independently checksummed
19//! and readable. Issue #745 is what this rule is for: a single buffer works until it does not, and
20//! an SF100 `lineitem` neighbour array is two gigabytes. Splitting is not an optimization here, it
21//! is the difference between a structure that exists at scale and one that does not.
22//!
23//! Sections are written before the directory and committed by the two-generation header swap the
24//! format already performs. So a crash during a section build leaves unreferenced trailing bytes in
25//! the file and nothing else, and there is no new recovery path to write or to test.
26
27use rudb_common::{Error, Result};
28
29/// Bytes one section table entry takes on disk.
30///
31/// Fifty six, per section 3.2, and fixed rather than variable because the entry list is walked at
32/// open to decide which sections this build understands and a fixed stride makes that a multiply.
33pub(crate) const ENTRY_BYTES: usize = 56;
34
35/// The largest one extent may be.
36///
37/// Sixty four megabytes. Small enough that a reader can hold one while it checksums it, and large
38/// enough that even an SF100 `lineitem` forward link is tens of extents rather than thousands.
39pub const MAX_EXTENT: u32 = 64 * 1024 * 1024;
40
41/// The most extents one section may have.
42///
43/// Sixty four megabytes each, so this bounds a section at a terabyte. The bound exists so that a
44/// torn directory naming four billion extents is refused at decode rather than turned into an
45/// allocation.
46pub const MAX_EXTENTS: u32 = 16 * 1024;
47
48/// A key map, per section 3.3.
49pub const KEY_MAP: &[u8; 8] = b"RUDBKM1\0";
50
51/// A forward link column, per section 3.4.
52pub const FORWARD_LINK: &[u8; 8] = b"RUDBFL1\0";
53
54/// A backward adjacency list, per section 3.5.
55pub const ADJACENCY: &[u8; 8] = b"RUDBAJ1\0";
56
57/// A column summary, per `spec/stats/03-the-file-format.md` section 3.3.
58///
59/// The first kind here that is not from the graph document, which is the point of the mechanism
60/// rather than a complication of it. A statistics section is carried, stamped, split and ignored by
61/// exactly the rules above, and adding it took two constants and one arm below.
62pub const SUMMARY: &[u8; 8] = b"RUDBCS1\0";
63
64/// A column's sketches, per `spec/stats/03-the-file-format.md` section 3.4.
65pub const SKETCHES: &[u8; 8] = b"RUDBSK1\0";
66
67/// A relationship's degree distribution and certificates, per `spec/stats/07-graph-statistics.md`.
68///
69/// Written by the graph layer, because it comes out of the pass the forward link build is already
70/// making, and owned by the statistics document, because nothing in it is needed to resolve a
71/// relationship. Its id is the child column, the same as the forward link it describes, so the two
72/// are found the same way and a rebuild replaces both.
73pub const DEGREES: &[u8; 8] = b"RUDBGD1\0";
74
75/// Rows sorted by one column and covering a second column. The payload holds row values,
76/// not grouped counts. A changed table generation makes the section stale.
77pub const SORTED_PROJECTION: &[u8; 8] = b"RUDBSP1\0";
78
79/// Row-preserving run encoding of a projection ordered by one signed integer column.
80pub const RUN_PROJECTION: &[u8; 8] = b"RUDBRP1\0";
81
82/// The kinds the graph document owns, which share its ten percent of the column bytes.
83pub const GRAPH_KINDS: &[&[u8; 8]] = &[KEY_MAP, FORWARD_LINK, ADJACENCY];
84
85/// The kinds the statistics document owns, which share its two percent.
86///
87/// Ownership here is about which budget pays, not about which builder writes. [`DEGREES`] is
88/// written by the link build and is on this list, because it is a planning hint that a reader can
89/// drop without losing a relationship, which is the line the two documents are divided along.
90///
91/// Two lists rather than one because the two budgets are separate, and separate means each counts
92/// only what it owns. A statistics build that counted the key maps as already spent would be a
93/// statistics budget the graph layer eats: a TPC-H SF10 file's key maps are 7.7 MB against a two
94/// percent allowance of 54 MB, so a seventh of the statistics budget would go to sections that have
95/// their own.
96///
97/// A kind in neither list is one a later build wrote, and it counts against neither. There is no
98/// better answer available, since this build cannot know which document invented it, and charging
99/// it to both would make every budget here tighter than the document says by an amount that depends
100/// on what some other build did.
101pub const STATISTICS_KINDS: &[&[u8; 8]] = &[SUMMARY, SKETCHES, DEGREES];
102
103/// One entry in a table's section table.
104///
105/// The payload is not here. This is the entry that says where the payload is, what it is, and
106/// whether it is still current, and it is all a reader needs to decide whether to read the payload
107/// at all.
108#[derive(Debug, Clone, Copy, PartialEq, Eq)]
109pub struct Section {
110 /// Which kind of structure this is: one of [`KEY_MAP`], [`FORWARD_LINK`], [`ADJACENCY`], or
111 /// something a later build wrote that this one carries through untouched.
112 pub kind: [u8; 8],
113 /// Which structure of that kind. For a key map this identifies the column, for a forward link
114 /// the relationship. The format does not interpret it; `rudb-graph` assigns it.
115 pub id: u64,
116 /// The table generation this section was built against.
117 ///
118 /// A section whose stamp does not match the table's is stale, and section 3.1 says stale means
119 /// ignored rather than repaired. So this field is the whole of the maintenance story: there is
120 /// no repair path in this crate because a mismatch here removes the section from consideration
121 /// and the query runs the way it ran before the section existed.
122 pub generation: u64,
123 /// How many extents the payload is split into.
124 pub extents: u32,
125 /// Where the extent table starts.
126 pub extent_page: u64,
127 /// How many bytes the extent table takes.
128 pub extent_bytes: u32,
129 /// Checksum over the extent table, so a torn one is found before it is believed.
130 pub hash: u64,
131 /// Kind-specific flags. For a key map this carries which of the three forms was chosen, which
132 /// is why a reader never has to guess a form.
133 pub flags: u32,
134 /// Bytes of kind-specific header at the front of the first extent, or, when there are no
135 /// extents, what the structure would have cost. See [`Self::refused`].
136 pub header_bytes: u32,
137}
138
139impl Section {
140 /// Appends this entry's fifty six bytes.
141 ///
142 /// # Errors
143 ///
144 /// If the entry describes something that cannot exist: more extents than [`MAX_EXTENTS`], or an
145 /// extent table larger than one extent. Both are caught here rather than at decode because a
146 /// writer that produced one has a bug, and the bug should stop at the write.
147 pub(crate) fn encode(&self, out: &mut Vec<u8>) -> Result<()> {
148 if self.extents > MAX_EXTENTS {
149 return Err(malformed(format!(
150 "a section of {} extents exceeds the bound of {MAX_EXTENTS}",
151 self.extents
152 )));
153 }
154 if self.extent_bytes > MAX_EXTENT {
155 return Err(malformed("a section's extent table is larger than one extent"));
156 }
157 let before = out.len();
158 out.extend_from_slice(&self.kind);
159 out.extend_from_slice(&self.id.to_le_bytes());
160 out.extend_from_slice(&self.generation.to_le_bytes());
161 out.extend_from_slice(&self.extents.to_le_bytes());
162 out.extend_from_slice(&self.extent_page.to_le_bytes());
163 out.extend_from_slice(&self.extent_bytes.to_le_bytes());
164 out.extend_from_slice(&self.hash.to_le_bytes());
165 out.extend_from_slice(&self.flags.to_le_bytes());
166 out.extend_from_slice(&self.header_bytes.to_le_bytes());
167 debug_assert_eq!(out.len() - before, ENTRY_BYTES, "a section entry is fifty six bytes");
168 Ok(())
169 }
170
171 /// Reads one entry from exactly [`ENTRY_BYTES`] bytes.
172 ///
173 /// # Errors
174 ///
175 /// If the slice is the wrong length, or if the entry names more extents than [`MAX_EXTENTS`] or
176 /// an extent table larger than one extent. A bad entry is an error and not a panic because the
177 /// caller's answer to one is to drop the section and open the table anyway.
178 pub(crate) fn decode(bytes: &[u8]) -> Result<Self> {
179 if bytes.len() != ENTRY_BYTES {
180 return Err(malformed("a section entry is not fifty six bytes"));
181 }
182 let section = Self {
183 kind: bytes[0..8].try_into().expect("eight bytes"),
184 id: u64::from_le_bytes(bytes[8..16].try_into().expect("eight bytes")),
185 generation: u64::from_le_bytes(bytes[16..24].try_into().expect("eight bytes")),
186 extents: u32::from_le_bytes(bytes[24..28].try_into().expect("four bytes")),
187 extent_page: u64::from_le_bytes(bytes[28..36].try_into().expect("eight bytes")),
188 extent_bytes: u32::from_le_bytes(bytes[36..40].try_into().expect("four bytes")),
189 hash: u64::from_le_bytes(bytes[40..48].try_into().expect("eight bytes")),
190 flags: u32::from_le_bytes(bytes[48..52].try_into().expect("four bytes")),
191 header_bytes: u32::from_le_bytes(bytes[52..56].try_into().expect("four bytes")),
192 };
193 if section.extents > MAX_EXTENTS {
194 return Err(malformed("a section names more extents than the bound allows"));
195 }
196 if section.extent_bytes > MAX_EXTENT {
197 return Err(malformed("a section's extent table is larger than one extent"));
198 }
199 Ok(section)
200 }
201
202 /// Whether this build understands this section's kind.
203 ///
204 /// The five it knows are the three the graph document's section 3.2 names and the two the
205 /// statistics document's sections 3.3 and 3.4 name. Everything else is a section a later build
206 /// wrote, and the answer is to leave it alone: the entry is carried through a rewrite so that
207 /// opening a file with an old build and closing it does not silently discard work, and the
208 /// payload is never read.
209 #[must_use]
210 pub fn known(&self) -> bool {
211 matches!(
212 &self.kind,
213 KEY_MAP
214 | FORWARD_LINK
215 | ADJACENCY
216 | SUMMARY
217 | SKETCHES
218 | DEGREES
219 | SORTED_PROJECTION
220 | RUN_PROJECTION
221 )
222 }
223
224 /// Whether this section's kind is one of these, which is how a budget finds what it owns.
225 #[must_use]
226 pub fn among(&self, kinds: &[&[u8; 8]]) -> bool {
227 kinds.iter().any(|kind| self.kind == **kind)
228 }
229
230 /// Whether this section was built against this table generation.
231 #[must_use]
232 pub fn current(&self, generation: u64) -> bool {
233 self.generation == generation
234 }
235
236 /// Whether this section is one this build should read: a kind it knows, at the current
237 /// generation.
238 #[must_use]
239 pub fn usable(&self, generation: u64) -> bool {
240 self.known() && self.current(generation)
241 }
242
243 /// What this structure would have cost, when the entry is a record of one that did not fit.
244 ///
245 /// Section 3.7 asks for a relationship that did not fit the budget to be recorded with its size
246 /// rather than forgotten, so that raising `graph_budget` is a decision somebody can make from a
247 /// number. An entry with no extents is that record, and the number is in [`Self::header_bytes`],
248 /// which has nothing else to mean when there is no first extent to have a header at the front
249 /// of. [`Self::flags`] keeps the meaning it has for a built section of the same kind, so a
250 /// record says which form the structure would have taken as well as what it would have cost.
251 ///
252 /// `None` for a section that is in the file, which is the ordinary case and is the one where
253 /// the size is the payload's own length.
254 ///
255 /// A size past four gigabytes saturates, because the field is a `u32`. The largest structure
256 /// this project expects to refuse is a packed forward link over an SF100 `lineitem`, which is
257 /// about 2.1 GB, so the saturation is a bound rather than a rounding, and a saturated record
258 /// still says *far more than the budget* correctly.
259 #[must_use]
260 pub fn refused(&self) -> Option<u64> {
261 (self.extents == 0).then(|| u64::from(self.header_bytes))
262 }
263}
264
265/// One section to be written into a file, handed to [`crate::attach`].
266///
267/// The payload is bytes and the format keeps it that way. Which of the three key map forms is in
268/// `flags`, and what the first `header_bytes` bytes mean, are questions `rudb-graph` answers and
269/// this crate never asks, which is what makes the first of section 3.2's three rules true rather
270/// than intended: a mechanism that had to understand a payload could not carry one it had never
271/// heard of.
272#[derive(Debug, Clone, Copy)]
273pub struct Attachment<'a> {
274 /// Which kind of structure this is, usually one of [`KEY_MAP`], [`FORWARD_LINK`],
275 /// [`ADJACENCY`].
276 pub kind: [u8; 8],
277 /// Which structure of that kind. An attachment replaces any section already in the table with
278 /// the same kind and id, which is what makes rebuilding a key map a write rather than a
279 /// question about what to do with the old one.
280 pub id: u64,
281 /// Kind-specific flags, copied into the entry and not interpreted.
282 pub flags: u32,
283 /// How many bytes at the front of `bytes` are the kind's own header.
284 pub header_bytes: u32,
285 /// The payload. Empty is legal and is how section 3.7 records a relationship that did not fit
286 /// the budget: an entry with no extents, its size reported by `rudb_links()`, and nothing in
287 /// the file to read.
288 pub bytes: &'a [u8],
289}
290
291/// Where one extent of a section's payload lives.
292///
293/// Each carries its own checksum, which is the second of section 3.2's three rules: an extent is
294/// independently readable, so a reduction that only needs the third extent of a forward link reads
295/// and verifies one extent rather than two gigabytes.
296#[derive(Debug, Clone, Copy, PartialEq, Eq)]
297pub struct Extent {
298 /// Where the extent's bytes start.
299 pub offset: u64,
300 /// How many bytes it holds, at most [`MAX_EXTENT`].
301 pub length: u32,
302 /// Checksum over those bytes.
303 pub hash: u64,
304 /// How many logical elements precede this extent, so that a random access can find the extent
305 /// holding an element without reading any of them.
306 pub first: u64,
307}
308
309/// Bytes one extent entry takes in an extent table.
310pub const EXTENT_BYTES: usize = 28;
311
312impl Extent {
313 /// Appends this extent's twenty eight bytes.
314 ///
315 /// # Errors
316 ///
317 /// If the extent is larger than [`MAX_EXTENT`], which is the rule the split exists to keep.
318 pub(crate) fn encode(&self, out: &mut Vec<u8>) -> Result<()> {
319 if self.length > MAX_EXTENT {
320 return Err(malformed(format!(
321 "an extent of {} bytes exceeds the maximum of {MAX_EXTENT}",
322 self.length
323 )));
324 }
325 out.extend_from_slice(&self.offset.to_le_bytes());
326 out.extend_from_slice(&self.length.to_le_bytes());
327 out.extend_from_slice(&self.hash.to_le_bytes());
328 out.extend_from_slice(&self.first.to_le_bytes());
329 Ok(())
330 }
331
332 /// Reads one extent from exactly [`EXTENT_BYTES`] bytes.
333 ///
334 /// # Errors
335 ///
336 /// If the slice is the wrong length or the extent is oversized.
337 pub(crate) fn decode(bytes: &[u8]) -> Result<Self> {
338 if bytes.len() != EXTENT_BYTES {
339 return Err(malformed("an extent entry is not twenty eight bytes"));
340 }
341 let extent = Self {
342 offset: u64::from_le_bytes(bytes[0..8].try_into().expect("eight bytes")),
343 length: u32::from_le_bytes(bytes[8..12].try_into().expect("four bytes")),
344 hash: u64::from_le_bytes(bytes[12..20].try_into().expect("eight bytes")),
345 first: u64::from_le_bytes(bytes[20..28].try_into().expect("eight bytes")),
346 };
347 if extent.length > MAX_EXTENT {
348 return Err(malformed("an extent is larger than the maximum extent"));
349 }
350 Ok(extent)
351 }
352}
353
354/// Encodes a whole extent table, checking that it describes a contiguous run of elements.
355///
356/// # Errors
357///
358/// If an extent is oversized, if the `first` counts are not increasing, or if there are more
359/// extents than [`MAX_EXTENTS`]. The increasing check is what makes a binary search over the table
360/// meaningful, and an unchecked one would be a search that silently returned the wrong extent.
361pub fn encode_extents(extents: &[Extent], out: &mut Vec<u8>) -> Result<()> {
362 if extents.len() > MAX_EXTENTS as usize {
363 return Err(malformed("a section names more extents than the bound allows"));
364 }
365 for (at, extent) in extents.iter().enumerate() {
366 if at == 0 {
367 if extent.first != 0 {
368 return Err(malformed("a section's first extent does not start at element zero"));
369 }
370 } else if extent.first <= extents[at - 1].first {
371 return Err(malformed("a section's extents are not in element order"));
372 }
373 extent.encode(out)?;
374 }
375 Ok(())
376}
377
378/// Decodes a whole extent table.
379///
380/// # Errors
381///
382/// If the byte count is not a multiple of an entry, if an entry is malformed, or if the entries are
383/// not in element order.
384pub fn decode_extents(bytes: &[u8]) -> Result<Vec<Extent>> {
385 if bytes.len() % EXTENT_BYTES != 0 {
386 return Err(malformed("an extent table is not a whole number of entries"));
387 }
388 let mut extents: Vec<Extent> = Vec::with_capacity(bytes.len() / EXTENT_BYTES);
389 for chunk in bytes.chunks(EXTENT_BYTES) {
390 let extent = Extent::decode(chunk)?;
391 match extents.last() {
392 None if extent.first != 0 => {
393 return Err(malformed("a section's first extent does not start at element zero"));
394 }
395 Some(previous) if extent.first <= previous.first => {
396 return Err(malformed("a section's extents are not in element order"));
397 }
398 _ => {}
399 }
400 extents.push(extent);
401 }
402 Ok(extents)
403}
404
405/// Which extent holds a given logical element, by binary search over the table.
406///
407/// Returns the index into `extents` and the element's offset within that extent's elements, or
408/// `None` when there are no extents at all, which is the not-built entry of section 3.7.
409///
410/// It does not bound the element from above, because an extent table cannot: the last extent's
411/// length is in bytes and only the caller knows how many elements a byte holds. So an element past
412/// the end answers with an offset past the end of the last extent, and the caller checks that
413/// against the count it already has. `None` rather than an error for the empty case because a
414/// stale link may name a structure that is no longer there, and section 3.1 wants staleness
415/// ignored.
416#[must_use]
417pub fn locate(extents: &[Extent], element: u64) -> Option<(usize, u64)> {
418 let at = extents.partition_point(|extent| extent.first <= element);
419 if at == 0 {
420 return None;
421 }
422 Some((at - 1, element - extents[at - 1].first))
423}
424
425fn malformed(message: impl Into<String>) -> Error {
426 Error::invalid_input(format!("invalid rudb section table: {}", message.into()))
427}
428
429#[cfg(test)]
430mod tests {
431 use super::*;
432
433 fn entry() -> Section {
434 Section {
435 kind: *KEY_MAP,
436 id: 7,
437 generation: 42,
438 extents: 3,
439 extent_page: 1 << 20,
440 extent_bytes: 84,
441 hash: 0xdead_beef_cafe_f00d,
442 flags: 2,
443 header_bytes: 24,
444 }
445 }
446
447 #[test]
448 fn an_entry_takes_fifty_six_bytes_and_round_trips() {
449 let mut bytes = Vec::new();
450 entry().encode(&mut bytes).expect("encode");
451 assert_eq!(bytes.len(), ENTRY_BYTES, "section 3.2 says fifty six");
452 assert_eq!(Section::decode(&bytes).expect("decode"), entry());
453 }
454
455 #[test]
456 fn an_unknown_kind_is_carried_and_not_read() {
457 // The rule that makes this the last bump the mechanism needs. A build that met this entry
458 // before the kind existed has to be able to hold it, report it as not understood, and open
459 // the table anyway.
460 let mut unknown = entry();
461 unknown.kind = *b"RUDBZZ9\0";
462 let mut bytes = Vec::new();
463 unknown.encode(&mut bytes).expect("an unknown kind still encodes");
464 let read = Section::decode(&bytes).expect("an unknown kind still decodes");
465 assert_eq!(read, unknown, "the entry survives a build that does not know it");
466 assert!(!read.known());
467 assert!(!read.usable(42), "a kind this build does not know is never read");
468 }
469
470 #[test]
471 fn the_kinds_the_two_documents_name_are_known() {
472 for kind in [KEY_MAP, FORWARD_LINK, ADJACENCY, SUMMARY, SKETCHES] {
473 let mut section = entry();
474 section.kind = *kind;
475 assert!(section.known(), "{}", String::from_utf8_lossy(kind));
476 }
477 }
478
479 #[test]
480 fn no_two_kinds_share_a_tag() {
481 // Worth a test now that two documents assign them. A collision would mean one kind's payload
482 // read by the other's decoder, which is the one thing an opaque payload cannot defend
483 // against by itself.
484 let all = [KEY_MAP, FORWARD_LINK, ADJACENCY, SUMMARY, SKETCHES];
485 for (at, one) in all.iter().enumerate() {
486 for other in &all[at + 1..] {
487 assert_ne!(one, other, "{}", String::from_utf8_lossy(*one));
488 }
489 }
490 }
491
492 #[test]
493 fn a_stale_section_is_ignored_rather_than_repaired() {
494 // Section 3.1's staleness rule, which is the whole of the maintenance story: the generation
495 // stamp not matching removes the section from consideration, and there is no third state
496 // between usable and ignored for a repair path to live in.
497 let section = entry();
498 assert!(section.usable(42));
499 assert!(!section.usable(43), "a rewrite invalidates rather than corrupts");
500 assert!(section.known(), "staleness is not the same question as familiarity");
501 }
502
503 #[test]
504 fn an_entry_naming_more_extents_than_the_bound_is_refused_at_both_ends() {
505 let mut oversized = entry();
506 oversized.extents = MAX_EXTENTS + 1;
507 assert!(oversized.encode(&mut Vec::new()).is_err(), "a writer's bug stops at the write");
508
509 let mut bytes = Vec::new();
510 entry().encode(&mut bytes).expect("encode");
511 bytes[24..28].copy_from_slice(&(MAX_EXTENTS + 1).to_le_bytes());
512 assert!(Section::decode(&bytes).is_err(), "a torn count is not turned into an allocation");
513 }
514
515 #[test]
516 fn a_short_entry_is_refused_rather_than_read_past() {
517 let mut bytes = Vec::new();
518 entry().encode(&mut bytes).expect("encode");
519 bytes.pop();
520 assert!(Section::decode(&bytes).is_err());
521 assert!(Section::decode(&[]).is_err());
522 }
523
524 #[test]
525 fn an_extent_at_the_maximum_is_allowed_and_one_past_it_is_not() {
526 // The bound is the point of the split, so the boundary is the case worth pinning: sixty
527 // four megabytes exactly has to work, because a payload that is a multiple of it would
528 // otherwise be unwritable.
529 let at_bound = Extent { offset: 4096, length: MAX_EXTENT, hash: 9, first: 0 };
530 let mut bytes = Vec::new();
531 at_bound.encode(&mut bytes).expect("an extent at the bound encodes");
532 assert_eq!(bytes.len(), EXTENT_BYTES);
533 assert_eq!(Extent::decode(&bytes).expect("decode"), at_bound);
534
535 let past = Extent { offset: 4096, length: MAX_EXTENT + 1, hash: 9, first: 0 };
536 assert!(past.encode(&mut Vec::new()).is_err());
537 }
538
539 fn table() -> Vec<Extent> {
540 vec![
541 Extent { offset: 1024, length: MAX_EXTENT, hash: 1, first: 0 },
542 Extent {
543 offset: 1024 + u64::from(MAX_EXTENT),
544 length: MAX_EXTENT,
545 hash: 2,
546 first: 100,
547 },
548 Extent { offset: 1024 + 2 * u64::from(MAX_EXTENT), length: 512, hash: 3, first: 250 },
549 ]
550 }
551
552 #[test]
553 fn an_extent_table_round_trips() {
554 let mut bytes = Vec::new();
555 encode_extents(&table(), &mut bytes).expect("encode");
556 assert_eq!(bytes.len(), 3 * EXTENT_BYTES);
557 assert_eq!(decode_extents(&bytes).expect("decode"), table());
558 }
559
560 #[test]
561 fn an_extent_table_out_of_element_order_is_refused() {
562 // The order is what makes the binary search in `locate` mean anything, so an unordered
563 // table has to be refused rather than searched: a search over one would return a plausible
564 // extent holding the wrong elements.
565 let mut out_of_order = table();
566 out_of_order.swap(1, 2);
567 assert!(encode_extents(&out_of_order, &mut Vec::new()).is_err());
568
569 let mut bytes = Vec::new();
570 encode_extents(&table(), &mut bytes).expect("encode");
571 bytes[EXTENT_BYTES + 20..EXTENT_BYTES + 28].copy_from_slice(&0_u64.to_le_bytes());
572 assert!(decode_extents(&bytes).is_err(), "a torn element order is refused");
573 }
574
575 #[test]
576 fn an_extent_table_not_starting_at_element_zero_is_refused() {
577 let mut shifted = table();
578 shifted[0].first = 1;
579 assert!(encode_extents(&shifted, &mut Vec::new()).is_err());
580 }
581
582 #[test]
583 fn a_partial_extent_table_is_refused_rather_than_truncated() {
584 let mut bytes = Vec::new();
585 encode_extents(&table(), &mut bytes).expect("encode");
586 bytes.truncate(bytes.len() - 1);
587 assert!(decode_extents(&bytes).is_err());
588 }
589
590 #[test]
591 fn an_empty_extent_table_is_a_section_with_no_payload() {
592 // A relationship recorded as not built, per section 3.7, is an entry with no extents. It
593 // has to be legal, because that is how `rudb_links()` reports what a larger budget would
594 // buy.
595 let mut bytes = Vec::new();
596 encode_extents(&[] as &[Extent], &mut bytes).expect("encode");
597 assert!(bytes.is_empty());
598 assert!(decode_extents(&bytes).expect("decode").is_empty());
599 assert_eq!(locate(&[], 0), None);
600 }
601
602 #[test]
603 fn an_element_resolves_to_the_extent_holding_it() {
604 let extents = table();
605 assert_eq!(locate(&extents, 0), Some((0, 0)));
606 assert_eq!(locate(&extents, 99), Some((0, 99)));
607 assert_eq!(locate(&extents, 100), Some((1, 0)), "the first element of the second extent");
608 assert_eq!(locate(&extents, 249), Some((1, 149)));
609 assert_eq!(locate(&extents, 250), Some((2, 0)));
610 assert_eq!(locate(&extents, 1_000_000), Some((2, 999_750)), "past the end of the elements");
611 }
612
613 #[test]
614 fn a_two_gigabyte_payload_is_tens_of_extents_and_not_one_buffer() {
615 // The arithmetic issue #745 is about, and the reason the split is a rule rather than an
616 // option. An SF100 lineitem forward link is 600,037,902 rows at 28 bits, which is 2.10 GB,
617 // and no reader should be asked to hold that in one buffer to checksum it.
618 let payload = 600_037_902_u64 * 28 / 8;
619 let extents = payload.div_ceil(u64::from(MAX_EXTENT));
620 assert!(extents > 30, "{extents} extents");
621 assert!(extents < u64::from(MAX_EXTENTS), "{extents} extents is inside the bound");
622 }
623}