rudb_vector/chunk.rs
1//! A batch of columns, which is the unit every operator passes to the next one.
2//!
3//! A chunk is some vectors of the same length plus that length. It is not a table and it is not a
4//! result set: it is at most [`VECTOR_SIZE`] rows, because the whole point of the number in
5//! `spec/04-architecture.md` section 4.3 is that a batch of this width stays in L1 while an
6//! operator works on it, and a type that can hold ten times that many rows is a type that lets an
7//! operator quietly stop being vectorized.
8//!
9//! The row count is stored rather than derived, which matters for the one case that looks like a
10//! mistake and is not. `SELECT count(*) FROM t` scans no columns, so the chunk the scan produces
11//! has no vectors in it and still has to say how many rows went past, and a chunk that derived its
12//! length from its first column would say zero.
13
14use rudb_common::{Error, LogicalType, Result, Value};
15
16use crate::selection::Selection;
17use crate::vector::{Form, VECTOR_SIZE, Vector};
18
19/// A batch of columns of equal length.
20#[derive(Debug, Clone, PartialEq)]
21pub struct Chunk {
22 columns: Vec<Vector>,
23 rows: usize,
24 /// The rows a filter kept, when it marked them on the whole chunk rather than cutting them out.
25 /// See [`Chunk::marked`].
26 marked: Option<Selection>,
27}
28
29/// The scheduler's half of the data plane contract, imposed now rather than at layer eight.
30///
31/// `spec/engine/03-data-plane.md` section 3.10. A chunk is what one thread hands another, so a chunk
32/// is `Send`, and that is not free: it rules out an `Rc` anywhere in a vector, it rules out a borrow
33/// of thread local state, and it is what the pin handle in [`Buffer`](crate::Buffer) is protecting
34/// against a lifetime parameter.
35///
36/// It is a static assertion rather than a comment because the failure mode is quiet. Every one of
37/// those mistakes compiles perfectly well on its own and is only a problem the day a chunk is put in
38/// a queue, which is eight layers from here and far too late to be told. This way the build breaks
39/// on the commit that introduces it.
40const _: () = {
41 const fn assert_send<T: Send>() {}
42 assert_send::<Chunk>();
43};
44
45impl Chunk {
46 /// A chunk of `columns`, taking the row count from the first of them.
47 ///
48 /// # Errors
49 ///
50 /// If the columns are not all the same length, or if there are more rows than [`VECTOR_SIZE`].
51 pub fn new(columns: Vec<Vector>) -> Result<Self> {
52 let rows = columns.first().map_or(0, Vector::len);
53 Self::with_rows(columns, rows)
54 }
55
56 /// A chunk of `columns` that is `rows` long, for the case where there are no columns to take
57 /// the count from.
58 ///
59 /// # Errors
60 ///
61 /// If any column is not `rows` long, or if `rows` is more than [`VECTOR_SIZE`].
62 pub fn with_rows(columns: Vec<Vector>, rows: usize) -> Result<Self> {
63 if rows > VECTOR_SIZE {
64 return Err(Error::internal(format!(
65 "a chunk of {rows} rows is longer than the {VECTOR_SIZE} row vector"
66 )));
67 }
68 for (index, column) in columns.iter().enumerate() {
69 if column.len() != rows {
70 return Err(Error::internal(format!(
71 "column {index} of a chunk is {} rows and the chunk is {rows}",
72 column.len()
73 )));
74 }
75 }
76 Ok(Self { columns, rows, marked: None })
77 }
78
79 /// A chunk of the given types with no rows in it.
80 ///
81 /// What a scan of an empty table returns and what an operator returns when it is done. The
82 /// types are kept, because a consumer asks a chunk what its columns are before it asks whether
83 /// there are any.
84 #[must_use]
85 pub fn empty(types: &[LogicalType]) -> Self {
86 let columns =
87 types.iter().map(|ty| Vector::constant(ty.clone(), Value::Null, 0)).collect::<Vec<_>>();
88 Self { columns, rows: 0, marked: None }
89 }
90
91 /// The columns.
92 #[must_use]
93 pub fn columns(&self) -> &[Vector] {
94 &self.columns
95 }
96
97 /// One column.
98 ///
99 /// # Errors
100 ///
101 /// If there is no column at `index`.
102 pub fn column(&self, index: usize) -> Result<&Vector> {
103 self.columns.get(index).ok_or_else(|| {
104 Error::internal(format!(
105 "column {index} of a chunk that has {} columns",
106 self.columns.len()
107 ))
108 })
109 }
110
111 /// The same rows with every column [`Vector::loosened`].
112 #[must_use]
113 pub fn loosened(mut self) -> Self {
114 self.columns = self.columns.into_iter().map(Vector::loosened).collect();
115 self
116 }
117
118 /// The columns, given up.
119 #[must_use]
120 pub fn into_columns(self) -> Vec<Vector> {
121 self.columns
122 }
123
124 /// How many columns.
125 #[must_use]
126 pub fn width(&self) -> usize {
127 self.columns.len()
128 }
129
130 /// How many rows.
131 #[must_use]
132 pub fn len(&self) -> usize {
133 self.rows
134 }
135
136 /// Whether there are no rows.
137 #[must_use]
138 pub fn is_empty(&self) -> bool {
139 self.rows == 0
140 }
141
142 /// How many bytes of memory this chunk is holding.
143 ///
144 /// What the memory limit charges for a chunk somebody kept. A chunk handed from one operator to
145 /// the next and dropped is not charged at all, because charging it would count the same
146 /// megabyte once per level of the tree, and the levels of the tree are not where a query runs
147 /// out of memory.
148 ///
149 /// Every size in this workspace counts the thing itself as well as what it owns, so a column's
150 /// own bytes are already in its own number and are not added again here.
151 #[must_use]
152 pub fn footprint(&self) -> usize {
153 size_of::<Self>() + self.columns.iter().map(Vector::footprint).sum::<usize>()
154 }
155
156 /// This chunk with every column's payload held as a page, so that a copy of it is free.
157 ///
158 /// For a chunk that is going to be stored and handed out many times, which is what an in memory
159 /// table's chunks are. See [`Vector::into_pages`] for what it does to each form.
160 #[must_use]
161 pub fn into_pages(self) -> Self {
162 Self {
163 columns: self.columns.into_iter().map(Vector::into_pages).collect(),
164 rows: self.rows,
165 marked: self.marked,
166 }
167 }
168
169 /// This chunk with the rows a filter kept marked on it, and every row still in its columns.
170 ///
171 /// Cutting the kept rows out of a chunk copies every column, and when a filter keeps nearly
172 /// all of them that copy is most of what the filter costs. In TPC-H q01 the date filter keeps
173 /// 98 percent of lineitem and the copy was a fifth of the query. An aggregate can read the
174 /// whole columns instead and count a dropped row into no group, so a filter that feeds one
175 /// directly marks the rows rather than cutting them. Only a consumer that asked for this is
176 /// handed a marked chunk, and [`Chunk::len`] is still the count of every row in the columns.
177 /// Anything that wants the kept rows alone calls [`Chunk::settled`].
178 #[must_use]
179 pub fn marked(mut self, kept: Selection) -> Self {
180 self.marked = Some(kept);
181 self
182 }
183
184 /// The rows a filter kept, when it marked them instead of cutting them out.
185 #[must_use]
186 pub fn kept(&self) -> Option<&Selection> {
187 self.marked.as_ref()
188 }
189
190 /// How many rows the chunk stands for, which is the kept rows of a marked chunk.
191 #[must_use]
192 pub fn live(&self) -> usize {
193 self.marked.as_ref().map_or(self.rows, Selection::len)
194 }
195
196 /// The kept rows alone, cut out the way a filter that did not mark them would have cut them.
197 ///
198 /// # Errors
199 ///
200 /// Whatever [`Chunk::select`] raises.
201 pub fn settled(mut self) -> Result<Self> {
202 match self.marked.take() {
203 Some(kept) => self.select(&kept),
204 None => Ok(self),
205 }
206 }
207
208 /// The type of each column.
209 #[must_use]
210 pub fn types(&self) -> Vec<LogicalType> {
211 self.columns.iter().map(|column| column.logical_type().clone()).collect()
212 }
213
214 /// Validate every storage-backed value reachable from this chunk.
215 pub fn validate_external(&self) -> Result<()> {
216 self.columns.iter().try_for_each(Vector::validate_external)
217 }
218
219 /// The value at a row and a column, or null if either is past the end.
220 ///
221 /// The slow path, same as [`Vector::value_at`]. It is what a result set is read out with and
222 /// what a test asserts on.
223 #[must_use]
224 pub fn value_at(&self, row: usize, column: usize) -> Value {
225 match self.columns.get(column) {
226 Some(held) => held.value_at(row),
227 None => Value::Null,
228 }
229 }
230
231 /// The value at a row and column, preserving storage read and validation failures.
232 pub fn try_value_at(&self, row: usize, column: usize) -> Result<Value> {
233 match self.columns.get(column) {
234 Some(held) => held.try_value_at(row),
235 None => Ok(Value::Null),
236 }
237 }
238
239 /// One row, left to right.
240 pub fn row(&self, row: usize) -> impl Iterator<Item = Value> + '_ {
241 self.columns.iter().map(move |column| column.value_at(row))
242 }
243
244 /// The rows a selection kept, without moving any of the values.
245 ///
246 /// Every column becomes a dictionary vector whose codes are the selection, which is the form
247 /// `spec/07-execution.md` section 7.1 asks a filter to produce rather than compacting. It takes
248 /// the chunk by value because that is what makes it free: the payload is moved into the new
249 /// vector rather than copied, so a filter that keeps one row in a thousand still costs the
250 /// selection and nothing else.
251 ///
252 /// A column that is already a stable dictionary is the one exception, and it composes the two
253 /// levels of codes instead of stacking them. Stacking is just as cheap here and it hides the
254 /// thing that matters: a stable dictionary is a promise that codes from separate chunks name the
255 /// same values, and the aggregate, the group key store and the string kernels all read that
256 /// promise off the outermost body. Wrapping it in a second dictionary breaks the promise, so a
257 /// `GROUP BY SearchPhrase` behind a `WHERE SearchPhrase <> ''` fell off the code path and hashed
258 /// strings instead, which measured at 30 ms of processor time against 4 ms for the same group by
259 /// with nothing in front of it. Composing costs one lookup per kept row and keeps the promise.
260 ///
261 /// # Errors
262 ///
263 /// If the selection points past the end of the chunk.
264 pub fn select(self, selection: &Selection) -> Result<Self> {
265 marked_twice(self.marked.as_ref())?;
266 // See `below` for why this is not the largest position, and every filtered chunk comes
267 // through here.
268 if !crate::vector::below(selection.indices(), self.rows) {
269 let bad = selection.indices().iter().max().copied().unwrap_or_default();
270 return Err(Error::internal(format!(
271 "a selection keeps row {bad} of a chunk that has {} rows",
272 self.rows
273 )));
274 }
275 let rows = selection.len();
276 let codes = selection.indices();
277 let mut columns = Vec::with_capacity(self.columns.len());
278 for column in self.columns {
279 // A packed column read through a dictionary is unpacked again by every reader, and
280 // unpacking it once here costs what one of those reads does. So it is copied out, which
281 // is never worse than selecting it once anything reads it and better as soon as two do.
282 if column.stable_dictionary_parts().is_some() || column.form() == Form::BitPacked {
283 columns.push(column.gather(codes)?);
284 } else {
285 columns.push(Vector::dictionary(codes.to_vec(), column)?);
286 }
287 }
288 Self::with_rows(columns, rows)
289 }
290
291 /// The rows a selection kept, copied, so that nothing downstream reads through an indirection.
292 ///
293 /// The copying counterpart to [`Self::select`], and the two exist because neither one is right
294 /// twice. Which one to call is measured rather than argued, and the measurement says something
295 /// other than what the argument does, so here is both.
296 ///
297 /// The argument is that selecting pays nothing now and one redirection on every later read of
298 /// every kept row, while compacting pays a copy now and nothing afterwards, so the deciding
299 /// variable is selectivity: keep a few rows and select, keep most of them and compact. The
300 /// measurement says the deciding variable is not selectivity at all, it is how many times the
301 /// rows are read again afterwards, and selectivity barely moves the line. On server3, over a
302 /// chunk of two integer columns, compacting loses to selecting at every selectivity from one
303 /// percent to a hundred when there is one later pass over the kept rows, and beats it at every
304 /// selectivity from one percent to a hundred when there are sixteen. With four later passes the
305 /// two are within a few percent of each other everywhere. Put a varchar column in the chunk and
306 /// compaction loses almost everywhere, because copying string bytes is most of what it costs and
307 /// the dictionary it avoids is most of what it saves.
308 ///
309 /// Which is why nothing in the streaming pipeline calls this yet. A filter today feeds an
310 /// aggregate or a projection and that is one pass or two, and end to end on two million rows
311 /// `SELECT sum(a), sum(b), count(*) FROM t WHERE a > ?` measures the same either way at one
312 /// percent selectivity and fifty percent slower compacting at fifty percent selectivity. The
313 /// operators that will want this are the ones that hold chunks rather than pass them on, the
314 /// hash join build side and the sort, because a chunk that is kept alive as a selection keeps
315 /// the whole chunk it was selected from alive with it, and that is a hundred to one on memory
316 /// rather than a few percent on time.
317 ///
318 /// Takes the chunk by value like [`Self::select`] does, even though the payload is copied rather
319 /// than moved, because a caller that still wanted the original after compacting it would be
320 /// holding both copies and should say so.
321 ///
322 /// # Errors
323 ///
324 /// If the selection points past the end of the chunk, or if a column has a type there is no
325 /// vector for, which today means `ARRAY` and `UNION`.
326 pub fn compact(self, selection: &Selection) -> Result<Self> {
327 marked_twice(self.marked.as_ref())?;
328 if let Some(bad) = selection.iter().find(|&index| index >= self.rows) {
329 return Err(Error::internal(format!(
330 "a selection keeps row {bad} of a chunk that has {} rows",
331 self.rows
332 )));
333 }
334 let rows = selection.len();
335 let indices = selection.indices();
336 let mut columns = Vec::with_capacity(self.columns.len());
337 for column in &self.columns {
338 columns.push(column.gather(indices)?);
339 }
340 Self::with_rows(columns, rows)
341 }
342
343 /// The columns at the given positions, in that order.
344 ///
345 /// A position may appear twice, which is what `SELECT x, x FROM t` is, and the second one costs
346 /// a copy. Every other position is moved.
347 ///
348 /// # Errors
349 ///
350 /// If a position is past the end of the chunk.
351 pub fn project(self, positions: &[usize]) -> Result<Self> {
352 let width = self.columns.len();
353 if let Some(&bad) = positions.iter().find(|&&position| position >= width) {
354 return Err(Error::internal(format!(
355 "column {bad} of a chunk that has {width} columns"
356 )));
357 }
358 let rows = self.rows;
359 let mut sources: Vec<Option<Vector>> = self.columns.into_iter().map(Some).collect();
360 let mut columns = Vec::with_capacity(positions.len());
361 for (at, &position) in positions.iter().enumerate() {
362 let last_use = !positions[at + 1..].contains(&position);
363 let taken = if last_use { sources[position].take() } else { sources[position].clone() };
364 match taken {
365 Some(column) => columns.push(column),
366 // Only reachable if the last-use bookkeeping above is wrong, since a position is
367 // taken on its last appearance and cloned on every earlier one.
368 None => {
369 return Err(Error::internal(format!("column {position} was taken twice")));
370 }
371 }
372 }
373 Self::with_rows(columns, rows)
374 }
375
376 /// The same rows with every column in flat form.
377 ///
378 /// Costs a copy per column that was not already flat. It is here for the result set at the top
379 /// of a query, where the dictionary vectors a filter left behind would otherwise be handed to a
380 /// caller who has to understand them.
381 ///
382 /// # Errors
383 ///
384 /// If a column has a type there is no vector for, which today means `ARRAY` and `UNION`. A `LIST`
385 /// and a `MAP` flatten to themselves and a `STRUCT` to a struct of flattened fields, since none of
386 /// the three has a data slice for a caller to read and there is nothing flatter to become.
387 pub fn flatten(&self) -> Result<Self> {
388 let mut columns = Vec::with_capacity(self.columns.len());
389 for column in &self.columns {
390 // flatten: this is the chunk wide version of the vector call and it exists so that the
391 // one caller at the top of a query can say it once instead of per column. Whether the
392 // copy is deserved is decided where this is called from, which today is one line in
393 // `rudb::database`, and that line says why.
394 columns.push(column.flatten()?);
395 }
396 Self::with_rows(columns, self.rows)
397 }
398
399 /// The same rows in flat form, taking the chunk rather than borrowing it.
400 ///
401 /// The same answer [`Self::flatten`] gives and it costs less for the column that is already
402 /// flat, which is most of them: that column is moved out of this chunk and into the new one
403 /// rather than copied. Borrowing had no way to do that, so flattening a chunk of four flat
404 /// columns of eight thousand rows copied every value for nothing, and at the top of a query of
405 /// six million rows that was a hundred and sixty megabytes copied to produce the bytes it
406 /// already had.
407 ///
408 /// # Errors
409 ///
410 /// The same as [`Self::flatten`].
411 pub fn into_flat(self) -> Result<Self> {
412 let rows = self.rows;
413 let mut columns = Vec::with_capacity(self.columns.len());
414 for column in self.columns {
415 columns.push(column.into_flat()?);
416 }
417 Self::with_rows(columns, rows)
418 }
419}
420
421/// A selection over a chunk that is marked already would pick rows of the whole columns rather than
422/// of the kept ones, so it is refused rather than answered wrong. See [`Chunk::settled`].
423fn marked_twice(marked: Option<&Selection>) -> Result<()> {
424 match marked {
425 Some(_) => Err(Error::internal("a marked chunk was cut before it was settled".to_string())),
426 None => Ok(()),
427 }
428}
429
430#[cfg(test)]
431mod tests {
432 use std::sync::Arc;
433
434 use rudb_common::LogicalType;
435
436 use super::*;
437 use crate::vector::{Data, Form};
438
439 fn integers(values: &[i32]) -> Vector {
440 Vector::flat(LogicalType::Integer, Data::Int32(values.to_vec().into()))
441 .expect("integers are an i32 layout")
442 }
443
444 #[test]
445 fn a_chunk_takes_its_length_from_its_columns() {
446 let chunk = Chunk::new(vec![integers(&[1, 2, 3]), integers(&[4, 5, 6])])
447 .expect("two columns of three");
448 assert_eq!(chunk.len(), 3);
449 assert_eq!(chunk.width(), 2);
450 assert_eq!(chunk.value_at(2, 1), Value::Integer(6));
451 }
452
453 #[test]
454 fn a_ragged_chunk_is_caught() {
455 let error = Chunk::new(vec![integers(&[1, 2, 3]), integers(&[4])])
456 .expect_err("a chunk is not ragged");
457 assert!(error.message().contains("column 1"), "{error}");
458 }
459
460 /// `SELECT count(*) FROM t` scans no columns and the row count still has to survive, which is
461 /// the reason the length is a field rather than the first column's length.
462 #[test]
463 fn a_chunk_with_no_columns_can_still_have_rows() {
464 let chunk = Chunk::with_rows(Vec::new(), 900).expect("no columns and nine hundred rows");
465 assert_eq!(chunk.len(), 900);
466 assert_eq!(chunk.width(), 0);
467 assert!(!chunk.is_empty(), "nine hundred rows is not empty");
468 }
469
470 #[test]
471 fn a_chunk_longer_than_a_vector_is_caught() {
472 let error = Chunk::with_rows(Vec::new(), VECTOR_SIZE + 1).expect_err("too long");
473 assert!(error.message().contains("longer than"), "{error}");
474 }
475
476 #[test]
477 fn an_empty_chunk_keeps_its_types() {
478 let chunk = Chunk::empty(&[LogicalType::Integer, LogicalType::Varchar]);
479 assert_eq!(chunk.len(), 0);
480 assert_eq!(chunk.types(), vec![LogicalType::Integer, LogicalType::Varchar]);
481 }
482
483 #[test]
484 fn selecting_keeps_the_rows_it_selected_and_no_others() {
485 let chunk = Chunk::new(vec![integers(&[10, 20, 30, 40]), integers(&[1, 2, 3, 4])])
486 .expect("four rows");
487 let kept = Selection::from_predicate(4, |index| index % 2 == 1);
488 let chunk = chunk.select(&kept).expect("rows one and three exist");
489 assert_eq!(chunk.len(), 2);
490 assert_eq!(chunk.row(0).collect::<Vec<_>>(), vec![Value::Integer(20), Value::Integer(2)]);
491 assert_eq!(chunk.row(1).collect::<Vec<_>>(), vec![Value::Integer(40), Value::Integer(4)]);
492 }
493
494 /// The reason `select` takes the chunk by value. If it copied the payload then a filter would
495 /// cost the same as a compaction and the selection would be a pure loss.
496 #[test]
497 fn selecting_leaves_the_values_where_they_were() {
498 let chunk = Chunk::new(vec![integers(&[10, 20, 30, 40])]).expect("four rows");
499 let kept = Selection::from_predicate(4, |index| index == 0);
500 let chunk = chunk.select(&kept).expect("row zero exists");
501 assert_eq!(chunk.column(0).expect("one column").form(), Form::Dictionary);
502 }
503
504 /// The promise a stable dictionary makes is about the outermost body, so a filter in front of a
505 /// group by has to compose the codes rather than stack a second dictionary on top of them.
506 #[test]
507 fn selecting_a_stable_dictionary_composes_the_codes_instead_of_stacking_them() {
508 let values = Arc::new(integers(&[10, 20, 30]));
509 let column = Vector::stable_dictionary(vec![2, 0, 1, 2], values).expect("three codes");
510 let chunk = Chunk::new(vec![column]).expect("four rows");
511 let kept = Selection::from_predicate(4, |index| index % 2 == 1);
512 let chunk = chunk.select(&kept).expect("rows one and three exist");
513 let column = chunk.column(0).expect("one column");
514 let (codes, values) = column.stable_dictionary_parts().expect("still a stable dictionary");
515 assert_eq!(codes, [0, 2]);
516 assert_eq!(values.len(), 3);
517 assert_eq!(column.value_at(0), Value::Integer(10));
518 assert_eq!(column.value_at(1), Value::Integer(30));
519 }
520
521 #[test]
522 fn a_selection_past_the_end_is_caught() {
523 let chunk = Chunk::new(vec![integers(&[1, 2])]).expect("two rows");
524 let mut kept = Selection::empty();
525 kept.push(7);
526 let error = chunk.select(&kept).expect_err("row seven does not exist");
527 assert!(error.message().contains("row 7"), "{error}");
528 }
529
530 /// The two halves of section 7.1's decision have to answer the same question the same way, or
531 /// the threshold between them is a place where a query changes its answer.
532 #[test]
533 fn compacting_keeps_the_same_rows_selecting_does_and_leaves_no_indirection() {
534 let chunk = Chunk::new(vec![integers(&[10, 20, 30, 40]), integers(&[1, 2, 3, 4])])
535 .expect("four rows");
536 let kept = Selection::from_predicate(4, |index| index % 2 == 1);
537 let selected = chunk.clone().select(&kept).expect("rows one and three exist");
538 let compacted = chunk.compact(&kept).expect("rows one and three exist");
539 assert_eq!(compacted.len(), selected.len());
540 for row in 0..compacted.len() {
541 assert_eq!(
542 compacted.row(row).collect::<Vec<_>>(),
543 selected.row(row).collect::<Vec<_>>()
544 );
545 }
546 assert_eq!(compacted.column(0).expect("one column").form(), Form::Flat);
547 }
548
549 #[test]
550 fn a_selection_past_the_end_is_caught_by_compacting_too() {
551 let chunk = Chunk::new(vec![integers(&[1, 2])]).expect("two rows");
552 let mut kept = Selection::empty();
553 kept.push(7);
554 let error = chunk.compact(&kept).expect_err("row seven does not exist");
555 assert!(error.message().contains("row 7"), "{error}");
556 }
557
558 #[test]
559 fn projecting_reorders_and_can_repeat_a_column() {
560 let chunk = Chunk::new(vec![integers(&[1, 2]), integers(&[3, 4])]).expect("two by two");
561 let chunk = chunk.project(&[1, 0, 1]).expect("both columns exist");
562 assert_eq!(chunk.width(), 3);
563 assert_eq!(
564 chunk.row(0).collect::<Vec<_>>(),
565 vec![Value::Integer(3), Value::Integer(1), Value::Integer(3)]
566 );
567 }
568
569 #[test]
570 fn projecting_a_column_that_is_not_there_is_caught() {
571 let chunk = Chunk::new(vec![integers(&[1, 2])]).expect("one column");
572 let error = chunk.project(&[0, 4]).expect_err("there is no column four");
573 assert!(error.message().contains("column 4"), "{error}");
574 }
575
576 #[test]
577 fn flattening_a_selected_chunk_gives_the_same_values() {
578 let chunk = Chunk::new(vec![integers(&[10, 20, 30])]).expect("three rows");
579 let kept = Selection::from_predicate(3, |index| index != 1);
580 let selected = chunk.select(&kept).expect("rows zero and two exist");
581 let flat = selected.flatten().expect("integers flatten");
582 assert_eq!(flat.column(0).expect("one column").form(), Form::Flat);
583 for row in 0..flat.len() {
584 assert_eq!(flat.value_at(row, 0), selected.value_at(row, 0), "row {row}");
585 }
586 }
587
588 /// Taking the chunk rather than borrowing it, which is the same flatten and is the one that
589 /// gets to move a column that is already flat instead of copying it.
590 #[test]
591 fn flattening_a_chunk_of_mixed_forms_moves_the_column_that_is_already_flat() {
592 let flat = integers(&[10, 20, 30]);
593 let address = |vector: &Vector| match vector.data() {
594 Some(Data::Int32(values)) => values.as_slice().as_ptr() as usize,
595 _ => panic!("the layout changed under the test"),
596 };
597 let stored = address(&flat);
598 let coded = Vector::dictionary(vec![2, 1, 0], integers(&[1, 2, 3])).expect("three codes");
599 let chunk = Chunk::new(vec![flat, coded]).expect("three rows of two columns");
600 let want: Vec<Vec<_>> = (0..3).map(|row| chunk.row(row).collect()).collect();
601 let flattened = chunk.into_flat().expect("integers flatten");
602 assert_eq!(address(flattened.column(0).expect("the first column")), stored);
603 for column in flattened.columns() {
604 assert_eq!(column.form(), Form::Flat);
605 }
606 let got: Vec<Vec<_>> = (0..3).map(|row| flattened.row(row).collect()).collect();
607 assert_eq!(got, want);
608 }
609
610 #[test]
611 fn a_chunk_costs_what_its_columns_cost() {
612 let chunk = Chunk::new(vec![integers(&[1; 1000]), integers(&[2; 1000])])
613 .expect("two columns of a thousand");
614 let columns: usize = chunk.columns().iter().map(Vector::footprint).sum();
615 assert_eq!(chunk.footprint(), size_of::<Chunk>() + columns);
616 assert!(chunk.footprint() >= 8000, "two thousand i32: {}", chunk.footprint());
617 }
618}