1use crate::statistics::collect_lazy;
2use color_eyre::Result;
3use color_eyre::eyre::Report;
4use polars::chunked_array::cast::CastOptions;
5use polars::prelude::*;
6use std::collections::BTreeMap;
7use std::sync::Arc;
8
9const DEFAULT_SAMPLE_ROWS: usize = 10_000;
11const DEFAULT_CHUNK_ROWS: usize = 1_000_000;
12const QUALITY_WINDOW_START: &str = "__datui_quality_window_start";
13pub const QUALITY_SOURCE_FILE_COLUMN: &str = "__datui_quality_source_file";
14pub const KEY_LIKE_UNIQUENESS: f64 = 0.95;
21const MAX_EVIDENCE_FILES: usize = 20;
24const MAX_CONFLICT_EXAMPLES: usize = 5;
25pub const MAX_FINDING_EXAMPLES: usize = 3;
28pub const QUALITY_WINDOW_WIDTHS: [&str; 4] = ["1h", "1d", "1w", "1mo"];
30
31#[derive(Debug, Clone, PartialEq, Eq, Default)]
32pub enum QualityScope {
33 #[default]
34 CurrentView,
35 WholeSource,
36 FirstRows(usize),
37 ViewRows {
38 start: usize,
39 end: usize,
40 },
41 SourceFiles(Vec<usize>),
42 SourcePartition {
43 column: String,
44 value: String,
45 },
46 SourceTimeRange {
47 column: String,
48 start: String,
49 end: String,
50 },
51}
52
53impl QualityScope {
54 pub fn label(&self) -> String {
55 match self {
56 Self::CurrentView => "current view".to_string(),
57 Self::WholeSource => "whole source".to_string(),
58 Self::FirstRows(rows) => format!(
59 "first {} rows of the view",
60 crate::numfmt::group_chrome(*rows)
61 ),
62 Self::ViewRows { start, end } => format!(
63 "view rows {}-{}",
64 crate::numfmt::group_chrome(*start),
65 crate::numfmt::group_chrome(*end)
66 ),
67 Self::SourceFiles(indices) => format!(
68 "source files {}",
69 indices
70 .iter()
71 .map(usize::to_string)
72 .collect::<Vec<_>>()
73 .join(",")
74 ),
75 Self::SourcePartition { column, value } => format!("source {column}={value}"),
76 Self::SourceTimeRange { column, start, end } => {
77 format!("source {column} {start}..{end}")
78 }
79 }
80 }
81
82 pub fn uses_source(&self) -> bool {
83 matches!(
84 self,
85 Self::WholeSource
86 | Self::SourceFiles(_)
87 | Self::SourcePartition { .. }
88 | Self::SourceTimeRange { .. }
89 )
90 }
91
92 pub fn command(&self) -> String {
93 match self {
94 Self::CurrentView => "view".to_string(),
95 Self::WholeSource => "source".to_string(),
96 Self::FirstRows(rows) => format!("rows 1..{rows}"),
97 Self::ViewRows { start, end } => format!("rows {start}..{end}"),
98 Self::SourceFiles(indices) => format!(
99 "files {}",
100 indices
101 .iter()
102 .map(usize::to_string)
103 .collect::<Vec<_>>()
104 .join(",")
105 ),
106 Self::SourcePartition { column, value } => format!("partition {column}={value}"),
107 Self::SourceTimeRange { column, start, end } => format!("time {column}={start}..{end}"),
108 }
109 }
110
111 pub fn parse_command(text: &str) -> Result<Self> {
112 let value = text.trim();
113 if value == "view" {
114 return Ok(Self::CurrentView);
115 }
116 if value == "source" {
117 return Ok(Self::WholeSource);
118 }
119 if let Some(range) = value.strip_prefix("rows ") {
120 let (start, end) = range
121 .split_once("..")
122 .ok_or_else(|| color_eyre::eyre::eyre!("use rows START..END"))?;
123 let start = start.parse::<usize>()?;
124 let end = end.parse::<usize>()?;
125 if start == 0 || end < start {
126 return Err(color_eyre::eyre::eyre!(
127 "row range must be 1-based with END >= START"
128 ));
129 }
130 if start == 1 {
131 return Ok(Self::FirstRows(end));
134 }
135 return Ok(Self::ViewRows { start, end });
136 }
137 if let Some(files) = value.strip_prefix("files ") {
138 let indices = files
139 .split(',')
140 .map(|part| part.trim().parse::<usize>())
141 .collect::<std::result::Result<Vec<_>, _>>()?;
142 if indices.is_empty() || indices.contains(&0) {
143 return Err(color_eyre::eyre::eyre!(
144 "use 1-based file numbers, for example files 1,3"
145 ));
146 }
147 let mut indices = indices;
148 indices.sort_unstable();
149 indices.dedup();
150 return Ok(Self::SourceFiles(indices));
151 }
152 if let Some(partition) = value.strip_prefix("partition ") {
153 let (column, value) = partition
154 .split_once('=')
155 .ok_or_else(|| color_eyre::eyre::eyre!("use partition COLUMN=VALUE"))?;
156 if column.trim().is_empty() || value.trim().is_empty() {
157 return Err(color_eyre::eyre::eyre!(
158 "partition column and value are required"
159 ));
160 }
161 return Ok(Self::SourcePartition {
162 column: column.trim().to_string(),
163 value: value.trim().to_string(),
164 });
165 }
166 if let Some(time) = value.strip_prefix("time ") {
167 let (column, range) = time
168 .split_once('=')
169 .ok_or_else(|| color_eyre::eyre::eyre!("use time COLUMN=START..END"))?;
170 let (start, end) = range
171 .split_once("..")
172 .ok_or_else(|| color_eyre::eyre::eyre!("use time COLUMN=START..END"))?;
173 let (start, end) = (start.trim(), end.trim());
174 if column.trim().is_empty()
175 || parse_scope_time(start).is_none()
176 || parse_scope_time(end).is_none()
177 || parse_scope_time(end) <= parse_scope_time(start)
178 {
179 return Err(color_eyre::eyre::eyre!(
180 "time range needs a column and increasing ISO dates or UTC timestamps"
181 ));
182 }
183 return Ok(Self::SourceTimeRange {
184 column: column.trim().to_string(),
185 start: start.to_string(),
186 end: end.to_string(),
187 });
188 }
189 Err(color_eyre::eyre::eyre!(
190 "use view, source, rows, files, partition, or time"
191 ))
192 }
193}
194
195pub(crate) fn parse_scope_time(text: &str) -> Option<i64> {
196 chrono::DateTime::parse_from_rfc3339(text)
197 .ok()
198 .map(|value| value.timestamp_micros())
199 .or_else(|| {
200 chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d")
201 .ok()
202 .and_then(|value| value.and_hms_opt(0, 0, 0))
203 .map(|value| value.and_utc().timestamp_micros())
204 })
205}
206
207#[derive(Debug, Clone, Default)]
208pub struct QualitySourceContext {
209 pub file_names: Vec<String>,
210 pub file_starts: Vec<usize>,
211 pub row_index_column: String,
212 pub file_group: Vec<u32>,
215 pub drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
218 pub file_omitted: Vec<Vec<(PlSmallStr, DataType)>>,
222 pub dataset_rows: usize,
224 pub footers_read: usize,
228 pub conflict_scan: Option<QualityConflictScan>,
232}
233
234impl QualitySourceContext {
235 fn group_of_file(&self, file: usize) -> Option<&crate::schema_union::DriftGroup> {
238 let group = *self.file_group.get(file)? as usize;
239 self.drift_groups.get(group)
240 }
241
242 fn drifting_files(&self) -> impl Iterator<Item = (usize, &crate::schema_union::DriftGroup)> {
245 (0..self.file_names.len()).filter_map(move |file| {
246 let group = self.group_of_file(file)?;
247 (!group.is_empty()).then_some((file, group))
248 })
249 }
250
251 fn file_rows(&self, file: usize) -> usize {
253 let Some(start) = self.file_starts.get(file) else {
254 return 0;
255 };
256 self.file_starts
257 .get(file + 1)
258 .copied()
259 .unwrap_or(self.dataset_rows)
260 .saturating_sub(*start)
261 }
262
263 fn stored_type(&self, file: usize, column: &str) -> Option<&DataType> {
266 self.file_omitted
267 .get(file)?
268 .iter()
269 .find(|(name, _)| name.as_str() == column)
270 .map(|(_, dtype)| dtype)
271 }
272}
273
274pub fn prepare_source_quality_scan(
277 lf: LazyFrame,
278 source: Option<&QualitySourceContext>,
279) -> Result<LazyFrame> {
280 let schema = lf.clone().collect_schema()?;
281 let expressions = schema
282 .iter()
283 .filter_map(|(name, dtype)| {
284 let column = name.as_str();
285 if column == crate::schema_union::DRIFT_COLUMN
286 && !source.is_some_and(|context| context.row_index_column == column)
287 {
288 return None;
289 }
290 Some(if matches!(dtype, DataType::Binary) {
291 lit(crate::widgets::datatable::binary_stub()).alias(column)
292 } else {
293 col(column)
294 })
295 })
296 .collect::<Vec<_>>();
297 let lf = lf.select(expressions);
298 Ok(
299 if source.is_some_and(|context| context.row_index_column == "__datui_quality_row") {
300 lf.with_row_index("__datui_quality_row", None)
301 } else {
302 lf
303 },
304 )
305}
306
307fn partition_predicate(column: &str, value: &str, schema: &Schema) -> Result<Expr> {
313 let dtype = schema
314 .get(column)
315 .ok_or_else(|| color_eyre::eyre::eyre!("partition column {column:?} is unavailable"))?;
316 let read = |text: &str| {
317 crate::typed_value::parse(text, dtype)
318 .map(lit)
319 .map_err(|why| color_eyre::eyre::eyre!("{column}: {why}"))
320 };
321 if let Some((start, end)) = value.split_once("..") {
322 let (start, end) = (start.trim(), end.trim());
323 if start.is_empty() || end.is_empty() {
324 return Err(color_eyre::eyre::eyre!(
325 "a partition range needs both ends, for example year=2020..2022"
326 ));
327 }
328 return Ok(col(column)
329 .gt_eq(read(start)?)
330 .and(col(column).lt_eq(read(end)?)));
331 }
332 value
333 .split(',')
334 .map(str::trim)
335 .filter(|value| !value.is_empty())
336 .map(|value| {
337 Ok(if value == "∅" {
338 col(column).is_null()
339 } else {
340 col(column).eq(read(value)?)
341 })
342 })
343 .reduce(|all, one| Ok(all?.or(one?)))
344 .ok_or_else(|| color_eyre::eyre::eyre!("name at least one partition value"))?
345}
346
347pub fn apply_quality_scope(
348 lf: LazyFrame,
349 scope: &QualityScope,
350 source: Option<&QualitySourceContext>,
351) -> Result<LazyFrame> {
352 match scope {
353 QualityScope::CurrentView | QualityScope::WholeSource => Ok(lf),
354 QualityScope::FirstRows(rows) => Ok(lf.slice(0, (*rows).min(u32::MAX as usize) as u32)),
355 QualityScope::ViewRows { start, end } => {
356 if *start == 0 || end < start {
357 return Err(color_eyre::eyre::eyre!("invalid 1-based view row range"));
358 }
359 let offset = i64::try_from(start - 1)?;
360 let length = end
361 .saturating_sub(*start)
362 .saturating_add(1)
363 .min(u32::MAX as usize) as u32;
364 Ok(lf.slice(offset, length))
365 }
366 QualityScope::SourceFiles(indices) => {
367 let source = source
368 .ok_or_else(|| color_eyre::eyre::eyre!("source-file positions are unavailable"))?;
369 let mut predicate: Option<Expr> = None;
370 for index in indices {
371 let file = index
372 .checked_sub(1)
373 .ok_or_else(|| color_eyre::eyre::eyre!("source file numbers start at 1"))?;
374 let start = *source.file_starts.get(file).ok_or_else(|| {
375 color_eyre::eyre::eyre!("source file #{index} is unavailable")
376 })?;
377 let start = u32::try_from(start)?;
378 let mut range = col(&source.row_index_column).gt_eq(lit(start));
379 if let Some(end) = source.file_starts.get(*index) {
380 range = range.and(col(&source.row_index_column).lt(lit(u32::try_from(*end)?)));
381 }
382 predicate = Some(match predicate {
383 Some(previous) => previous.or(range),
384 None => range,
385 });
386 }
387 Ok(lf.filter(
388 predicate.ok_or_else(|| color_eyre::eyre::eyre!("select at least one file"))?,
389 ))
390 }
391 QualityScope::SourcePartition { column, value } => {
392 let schema = lf.clone().collect_schema()?;
393 if !schema.contains(column.as_str()) {
394 return Err(color_eyre::eyre::eyre!(
395 "partition column {column:?} is unavailable"
396 ));
397 }
398 Ok(lf.filter(partition_predicate(column, value, &schema)?))
399 }
400 QualityScope::SourceTimeRange { column, start, end } => {
401 let schema = lf.clone().collect_schema()?;
402 let dtype = schema
403 .get(column.as_str())
404 .ok_or_else(|| color_eyre::eyre::eyre!("time column {column:?} is unavailable"))?;
405 if !matches!(dtype, DataType::Date | DataType::Datetime(..)) {
406 return Err(color_eyre::eyre::eyre!(
407 "{column:?} is not a date or datetime column"
408 ));
409 }
410 let start = parse_scope_time(start)
411 .ok_or_else(|| color_eyre::eyre::eyre!("invalid start time"))?;
412 let end =
413 parse_scope_time(end).ok_or_else(|| color_eyre::eyre::eyre!("invalid end time"))?;
414 if end <= start {
415 return Err(color_eyre::eyre::eyre!("time end must be after start"));
416 }
417 let value = col(column).cast(DataType::Datetime(TimeUnit::Microseconds, None));
418 Ok(lf.filter(
419 value
420 .clone()
421 .gt_eq(lit(start).cast(DataType::Datetime(TimeUnit::Microseconds, None)))
422 .and(value.lt(lit(end).cast(DataType::Datetime(TimeUnit::Microseconds, None)))),
423 ))
424 }
425 }
426}
427
428#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
429pub enum QualityPage {
430 #[default]
433 Setup,
434 Overview,
435 Columns,
436 Segments,
437 Trends,
438 Detail,
439 SegmentDetail,
441 Intervals,
443 IntervalDetail,
445 TimeRoles,
446 IntervalPairs,
448 TrendDetail,
451 Gaps,
453 ExpectedWindows,
455 Intent,
457}
458
459impl QualityPage {
460 pub const TABS: [Self; 5] = [
463 Self::Overview,
464 Self::Columns,
465 Self::Segments,
466 Self::Trends,
467 Self::Intervals,
468 ];
469
470 pub fn tab(self) -> Self {
471 match self {
472 Self::Detail => Self::Columns,
473 Self::SegmentDetail => Self::Segments,
474 Self::IntervalDetail => Self::Intervals,
475 Self::TrendDetail | Self::Gaps => Self::Trends,
476 Self::TimeRoles | Self::IntervalPairs | Self::ExpectedWindows | Self::Intent => {
477 Self::Setup
478 }
479 page => page,
480 }
481 }
482
483 pub fn is_setup(self) -> bool {
485 self.tab() == Self::Setup
486 }
487
488 pub fn title(self) -> &'static str {
489 match self.tab() {
490 Self::Overview => "Overview",
491 Self::Columns => "Columns",
492 Self::Segments => "Segments",
493 Self::Trends => "Trends",
494 Self::Intervals => "Intervals",
495 _ => "Setup",
496 }
497 }
498}
499
500#[derive(Debug, Clone, Copy, PartialEq, Eq)]
502pub enum QualitySetup {
503 Grain,
504 TimeRoles,
505 Intervals,
506}
507
508impl QualitySetup {
509 pub fn label(self) -> &'static str {
510 match self {
511 Self::Grain => "Set Grain",
512 Self::TimeRoles => "Time Roles",
513 Self::Intervals => "Intervals",
514 }
515 }
516}
517
518pub fn shows_trend(plan: &DataQualityPlan, results: &DataQualityResults) -> bool {
521 matches!(
522 plan.grain,
523 QualityGrain::RowChunks(_) | QualityGrain::TimeWindows { .. } | QualityGrain::Partition(_)
524 ) && results.segments.len() + results.unsampled_segments.len() > 1
525}
526
527pub fn page_setup(
530 page: QualityPage,
531 plan: &DataQualityPlan,
532 results: Option<&DataQualityResults>,
533 has_time_columns: bool,
534) -> Option<QualitySetup> {
535 let results = results?;
536 match page {
537 QualityPage::Segments if plan.grain == QualityGrain::Dataset => Some(QualitySetup::Grain),
538 QualityPage::Trends if !shows_trend(plan, results) => Some(QualitySetup::Grain),
539 QualityPage::Intervals if results.temporal.is_empty() && has_time_columns => {
543 if plan.candidate_pairs().is_empty() {
544 Some(QualitySetup::TimeRoles)
545 } else if plan.interval_pairs().is_empty() {
546 Some(QualitySetup::Intervals)
547 } else {
548 None
549 }
550 }
551 _ => None,
552 }
553}
554
555#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
556pub enum QualityCompute {
557 Metadata,
558 #[default]
559 Sample,
560 Full,
561}
562
563impl QualityCompute {
564 pub fn label(self) -> &'static str {
565 match self {
566 Self::Metadata => "metadata",
567 Self::Sample => "sample",
568 Self::Full => "full",
569 }
570 }
571}
572
573#[derive(Debug, Clone, PartialEq, Eq, Default)]
574pub enum QualityGrain {
575 #[default]
576 Dataset,
577 File,
578 Partition(String),
579 RowChunks(usize),
580 TimeWindows {
581 column: String,
582 every: String,
583 },
584}
585
586impl QualityGrain {
587 pub fn label(&self) -> String {
589 match self {
590 Self::Dataset => "whole dataset".to_string(),
591 Self::File => "by file".to_string(),
592 Self::Partition(column) => format!("by {column}"),
593 Self::RowChunks(rows) => {
594 format!("in chunks of {} rows", crate::numfmt::group_chrome(*rows))
595 }
596 Self::TimeWindows { column, every } => {
597 let unit = match every.as_str() {
598 "1h" => "hour",
599 "1d" => "day",
600 "1w" => "week",
601 "1mo" => "month",
602 other => other,
603 };
604 if column.eq_ignore_ascii_case(unit) {
606 format!("by {unit} of the {column} column")
607 } else {
608 format!("by {unit} of {column}")
609 }
610 }
611 }
612 }
613}
614
615#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
616pub enum QualityComparison {
617 #[default]
618 None,
619 Previous,
620 Baseline,
621}
622
623impl QualityComparison {
624 pub fn choice_label(self) -> &'static str {
626 match self {
627 Self::None => "none",
628 Self::Previous => "the segment before",
629 Self::Baseline => "a baseline segment (the first, or b on Segments)",
630 }
631 }
632
633 pub fn label(self) -> &'static str {
634 match self {
635 Self::None => "none",
636 Self::Previous => "previous",
637 Self::Baseline => "baseline (first)",
638 }
639 }
640}
641
642#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
643pub enum TemporalRole {
644 Event,
645 Effective,
646 PeriodEnd,
647 Created,
648 Published,
649 Received,
650 Processed,
651 ValidFrom,
652 ValidTo,
653}
654
655impl TemporalRole {
656 pub const ALL: [Self; 9] = [
657 Self::Event,
658 Self::Effective,
659 Self::PeriodEnd,
660 Self::Created,
661 Self::Published,
662 Self::Received,
663 Self::Processed,
664 Self::ValidFrom,
665 Self::ValidTo,
666 ];
667
668 pub fn label(self) -> &'static str {
669 match self {
670 Self::Event => "event",
671 Self::Effective => "effective/as-of",
672 Self::PeriodEnd => "period end",
673 Self::Created => "created",
674 Self::Published => "published",
675 Self::Received => "received",
676 Self::Processed => "processed",
677 Self::ValidFrom => "valid from",
678 Self::ValidTo => "valid to",
679 }
680 }
681}
682
683#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct TemporalRoleAssignment {
685 pub role: TemporalRole,
686 pub column: String,
687 pub timezone: Option<String>,
688}
689
690pub const INTERVAL_PAIRS: [(TemporalRole, TemporalRole); 7] = [
695 (TemporalRole::Event, TemporalRole::Published),
696 (TemporalRole::Event, TemporalRole::Received),
697 (TemporalRole::PeriodEnd, TemporalRole::Published),
698 (TemporalRole::Published, TemporalRole::Received),
699 (TemporalRole::Received, TemporalRole::Processed),
700 (TemporalRole::Event, TemporalRole::Processed),
701 (TemporalRole::ValidFrom, TemporalRole::ValidTo),
702];
703
704pub fn interval_label((start, end): (TemporalRole, TemporalRole)) -> String {
706 format!("{} to {}", start.label(), end.label())
707}
708
709#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default)]
713pub enum IntervalClock {
714 #[default]
715 Grain,
716 Start,
717 End,
718}
719
720impl IntervalClock {
721 pub const ALL: [Self; 3] = [Self::Grain, Self::Start, Self::End];
722
723 pub fn label(self) -> &'static str {
724 match self {
725 Self::Grain => "the grain's column",
726 Self::Start => "each interval's start",
727 Self::End => "each interval's end",
728 }
729 }
730}
731
732#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
734pub enum TimeKind {
735 Date,
736 Datetime,
737}
738
739impl TimeKind {
740 pub fn label(self) -> &'static str {
741 match self {
742 Self::Date => "date",
743 Self::Datetime => "datetime",
744 }
745 }
746}
747
748pub const TIME_FORMATS: [(TimeKind, &str); 16] = [
752 (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S"),
753 (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S"),
754 (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S%.f"),
755 (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S%.f"),
756 (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S%.f%#z"),
759 (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S%.f%#z"),
760 (TimeKind::Datetime, "%Y-%m-%d %H:%M"),
761 (TimeKind::Date, "%Y-%m-%d"),
762 (TimeKind::Date, "%Y%m%d"),
763 (TimeKind::Datetime, "%m/%d/%Y %H:%M:%S"),
764 (TimeKind::Datetime, "%m/%d/%Y %I:%M:%S %p"),
765 (TimeKind::Datetime, "%d/%m/%Y %H:%M:%S"),
766 (TimeKind::Datetime, "%d.%m.%Y %H:%M:%S"),
767 (TimeKind::Date, "%m/%d/%Y"),
768 (TimeKind::Date, "%d/%m/%Y"),
769 (TimeKind::Date, "%d.%m.%Y"),
770];
771
772#[derive(Debug, Clone, PartialEq, Eq, Hash)]
777pub struct TimeInterpretation {
778 pub column: String,
779 pub kind: TimeKind,
780 pub format: String,
783}
784
785impl TimeInterpretation {
786 pub fn zoned(&self) -> bool {
789 self.format.contains('z')
790 }
791
792 pub fn label(&self) -> String {
794 format!("{} {}", self.kind.label(), self.format)
795 }
796
797 pub fn expr(&self) -> Expr {
799 let options = StrptimeOptions {
800 format: Some(PlSmallStr::from(self.format.as_str())),
801 strict: false,
802 exact: true,
803 cache: true,
804 };
805 let text = col(self.column.as_str()).cast(DataType::String).str();
807 match self.kind {
808 TimeKind::Date => text.to_date(options),
809 TimeKind::Datetime => text.to_datetime(
810 Some(TimeUnit::Microseconds),
811 None,
812 options,
813 lit(PlSmallStr::from_static("raise")),
814 ),
815 }
816 }
817
818 pub fn unparsed(&self) -> Expr {
820 col(self.column.as_str())
821 .is_not_null()
822 .and(self.expr().is_null())
823 }
824
825 pub fn reads(&self, value: &str) -> bool {
828 match self.kind {
829 TimeKind::Date => chrono::NaiveDate::parse_from_str(value, &self.format).is_ok(),
830 TimeKind::Datetime if self.zoned() => {
832 chrono::DateTime::parse_from_str(value, &self.format).is_ok()
833 }
834 TimeKind::Datetime => {
835 chrono::NaiveDateTime::parse_from_str(value, &self.format).is_ok()
836 }
837 }
838 }
839}
840
841#[derive(Debug, Clone, Copy, PartialEq, Eq)]
844pub enum QualityStage {
845 Preparing,
846 CopyingSource,
847 ReusingSample,
848 ReadingSample,
849 CountingRows,
850 CountingSegments,
851 ProfilingColumns,
852 CheckingDuplicates,
853 CheckingKey,
854 CheckingSpellings,
855 ReadingConflicts,
856 ProfilingSegments,
857 ComputingIntervals,
858 CheckingSharedNulls,
859 CheckingSignal,
861 Assembling,
862}
863
864impl QualityStage {
865 pub fn label(self) -> &'static str {
866 match self {
867 Self::Preparing => "Preparing the plan",
868 Self::CopyingSource => "Copying the source locally",
869 Self::ReusingSample => "Reusing the retained sample",
870 Self::ReadingSample => "Reading the sample",
871 Self::CountingRows => "Counting rows",
872 Self::CountingSegments => "Counting segment rows",
873 Self::ProfilingColumns => "Profiling columns",
874 Self::CheckingDuplicates => "Checking duplicate rows",
875 Self::CheckingKey => "Checking the declared key",
876 Self::CheckingSpellings => "Checking category spellings",
877 Self::ReadingConflicts => "Reading conflicting values",
878 Self::ProfilingSegments => "Profiling segments",
879 Self::ComputingIntervals => "Computing intervals",
880 Self::CheckingSharedNulls => "Checking columns missing together",
881 Self::CheckingSignal => "Checking the signal",
882 Self::Assembling => "Assembling the report",
883 }
884 }
885}
886
887#[derive(Debug, Clone, Copy, PartialEq, Eq)]
890pub struct QualityPhase {
891 pub stage: QualityStage,
892 pub reads_source: bool,
893 pub interruptible: bool,
896}
897
898#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
901pub struct ObservedReads {
902 pub reads: usize,
904 pub counted: usize,
906 pub rows: usize,
908 pub copy: Option<CopyRead>,
910}
911
912#[derive(Debug, Clone, Copy, PartialEq, Eq)]
914pub struct CopyRead {
915 pub bytes: u64,
916 pub objects: usize,
917 pub fetched: bool,
919}
920
921#[derive(Clone, Default)]
925pub struct QualityWatch {
926 read: crate::sampling::ReadWatch,
927 report: Option<Arc<dyn Fn(QualityPhase) + Send + Sync>>,
928 last: Arc<std::sync::Mutex<(Option<QualityPhase>, ObservedReads)>>,
930 copy: Arc<std::sync::OnceLock<CopyRead>>,
932}
933
934impl std::fmt::Debug for QualityWatch {
935 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
936 f.debug_struct("QualityWatch")
937 .field("read", &self.read)
938 .finish_non_exhaustive()
939 }
940}
941
942impl QualityWatch {
943 pub fn new(report: impl Fn(QualityPhase) + Send + Sync + 'static) -> Self {
945 Self {
946 report: Some(Arc::new(report)),
947 ..Self::default()
948 }
949 }
950
951 pub fn cancel(&self) {
952 self.read.stop();
953 }
954
955 pub fn cancelled(&self) -> bool {
956 self.read.stopped()
957 }
958
959 pub fn read(&self) -> &crate::sampling::ReadWatch {
961 &self.read
962 }
963
964 pub fn observed(&self) -> ObservedReads {
966 let Ok(last) = self.last.lock() else {
967 return ObservedReads::default();
968 };
969 let (phase, mut observed) = *last;
970 if phase.is_some_and(|phase| phase.reads_source) {
971 observed.reads += 1;
972 if let Some(rows) = self.read.rows_seen() {
973 observed.counted += 1;
974 observed.rows += rows;
975 }
976 }
977 observed.copy = self.copy.get().copied();
978 observed
979 }
980
981 pub(crate) fn use_copy(&self, copy: CopyRead) {
983 let _ = self.copy.set(copy);
984 }
985
986 fn scope_reads(&self, reads: bool) -> bool {
988 reads && self.copy.get().is_none()
989 }
990
991 fn watched(&self, lf: &LazyFrame) -> LazyFrame {
997 let read = self.read.clone();
998 lf.clone().map(
999 move |df: DataFrame| {
1000 if read.stopped() {
1001 return Err(PolarsError::ComputeError(crate::sampling::CANCELLED.into()));
1002 }
1003 read.saw(df.height());
1004 Ok(df)
1005 },
1006 OptFlags::PROJECTION_PUSHDOWN | OptFlags::PREDICATE_PUSHDOWN | OptFlags::STREAMING,
1007 None,
1008 Some("quality watch"),
1009 )
1010 }
1011
1012 pub(crate) fn stage(
1016 &self,
1017 stage: QualityStage,
1018 reads_source: bool,
1019 interruptible: bool,
1020 ) -> Result<()> {
1021 self.read.check()?;
1022 let phase = QualityPhase {
1023 stage,
1024 reads_source,
1025 interruptible,
1026 };
1027 let mut last = self
1028 .last
1029 .lock()
1030 .map_err(|_| Report::msg("quality progress lock failed"))?;
1031 let (previous, observed) = &mut *last;
1032 if *previous != Some(phase) {
1033 let seen = self.read.restart();
1035 if previous.is_some_and(|phase| phase.reads_source) {
1036 observed.reads += 1;
1037 if let Some(rows) = seen {
1038 observed.counted += 1;
1039 observed.rows += rows;
1040 }
1041 }
1042 *previous = Some(phase);
1043 if let Some(report) = &self.report {
1044 report(phase);
1045 }
1046 }
1047 Ok(())
1048 }
1049
1050 fn failed(&self, error: impl Into<Report>) -> Report {
1053 if self.cancelled() {
1054 Report::msg(crate::sampling::CANCELLED)
1055 } else {
1056 error.into()
1057 }
1058 }
1059}
1060
1061#[derive(Debug, Clone, PartialEq, Eq)]
1062pub struct DataQualityPlan {
1063 pub scope: QualityScope,
1064 pub compute: QualityCompute,
1065 pub method: crate::sampling::SampleMethod,
1067 pub dataset_rows: usize,
1069 pub sample_seed: u64,
1070 pub grain: QualityGrain,
1071 pub comparison: QualityComparison,
1072 pub baseline_segment: Option<String>,
1073 pub temporal_roles: Vec<TemporalRoleAssignment>,
1074 pub intervals: Option<Vec<(TemporalRole, TemporalRole)>>,
1077 pub interval_clock: IntervalClock,
1079 pub latency_threshold_seconds: Option<i64>,
1080 pub time_formats: Vec<TimeInterpretation>,
1082 pub expected: Option<ExpectedWindows>,
1086 pub intent: crate::quality_intent::DeclaredIntent,
1089}
1090
1091#[derive(Debug, Clone, PartialEq, Eq, Default)]
1095pub struct ExpectedWindows {
1096 pub weekdays: bool,
1098 pub from: Option<String>,
1101 pub before: Option<String>,
1104}
1105
1106impl ExpectedWindows {
1107 pub fn weekdays_apply(every: &str) -> bool {
1110 matches!(every, "1h" | "1d")
1111 }
1112
1113 pub fn cadence_label(&self, every: &str) -> String {
1115 if self.weekdays && Self::weekdays_apply(every) {
1116 "weekdays".to_string()
1117 } else {
1118 let unit = match every {
1119 "1h" => "hour",
1120 "1d" => "day",
1121 "1w" => "week",
1122 "1mo" => "month",
1123 other => other,
1124 };
1125 format!("every {unit}")
1126 }
1127 }
1128
1129 pub fn range_label(&self) -> String {
1132 match (self.from.as_deref(), self.before.as_deref()) {
1133 (None, None) => "first to last window found".to_string(),
1134 (Some(from), None) => format!("{from} to the last window found"),
1135 (None, Some(before)) => format!("first window found to before {before}"),
1136 (Some(from), Some(before)) => format!("{from} to before {before}"),
1137 }
1138 }
1139
1140 pub fn problem(&self) -> Option<String> {
1142 let read = |text: &Option<String>| match text.as_deref() {
1143 None => Ok(None),
1144 Some(text) => parse_scope_time(text)
1145 .map(Some)
1146 .ok_or_else(|| format!("{text} is not a date or UTC timestamp")),
1147 };
1148 match (read(&self.from), read(&self.before)) {
1149 (Err(problem), _) | (_, Err(problem)) => Some(problem),
1150 (Ok(Some(from)), Ok(Some(before))) if before <= from => {
1151 Some("Before must be after From".to_string())
1152 }
1153 _ => None,
1154 }
1155 }
1156
1157 pub fn bounds(&self) -> (Option<i64>, Option<i64>) {
1160 (
1161 self.from.as_deref().and_then(parse_scope_time),
1162 self.before.as_deref().and_then(parse_scope_time),
1163 )
1164 }
1165}
1166
1167impl Default for DataQualityPlan {
1168 fn default() -> Self {
1169 Self {
1170 scope: QualityScope::CurrentView,
1171 compute: QualityCompute::Sample,
1172 method: crate::sampling::SampleMethod::Spread,
1173 dataset_rows: DEFAULT_SAMPLE_ROWS,
1174 sample_seed: 42_891,
1175 grain: QualityGrain::Dataset,
1176 comparison: QualityComparison::None,
1177 baseline_segment: None,
1178 temporal_roles: Vec::new(),
1179 intervals: None,
1180 interval_clock: IntervalClock::Grain,
1181 latency_threshold_seconds: None,
1182 time_formats: Vec::new(),
1183 expected: None,
1184 intent: crate::quality_intent::DeclaredIntent::default(),
1185 }
1186 }
1187}
1188
1189impl DataQualityPlan {
1190 pub fn requires_confirmation(&self) -> bool {
1193 self.compute == QualityCompute::Full
1194 }
1195
1196 pub fn comparison_label(&self) -> String {
1197 if self.comparison == QualityComparison::Baseline {
1198 self.baseline_segment
1199 .as_ref()
1200 .map(|label| format!("baseline: {label}"))
1201 .unwrap_or_else(|| self.comparison.label().to_string())
1202 } else {
1203 self.comparison.label().to_string()
1204 }
1205 }
1206
1207 pub fn sample(&self) -> crate::sampling::Sample {
1209 crate::sampling::Sample {
1210 scope: self.scope.clone(),
1211 method: self.method.clone(),
1212 rows: self.dataset_rows,
1213 seed: self.sample_seed,
1214 }
1215 }
1216
1217 pub fn adopt_sample(&mut self, sample: &crate::sampling::Sample) {
1225 if self.scope != sample.scope {
1226 self.baseline_segment = None;
1227 }
1228 self.scope = sample.scope.clone();
1229 self.sample_seed = sample.seed;
1230 self.dataset_rows = sample.rows;
1231 if self.compute != QualityCompute::Metadata {
1232 self.compute = if sample.method == crate::sampling::SampleMethod::EveryRow {
1233 QualityCompute::Full
1234 } else {
1235 QualityCompute::Sample
1236 };
1237 }
1238 if let crate::sampling::SampleMethod::PerPartition { column } = &sample.method
1239 && self.method != sample.method
1240 && self.grain == QualityGrain::Dataset
1241 {
1242 self.grain = QualityGrain::Partition(column.clone());
1243 self.baseline_segment = None;
1244 }
1245 self.method = sample.method.clone();
1246 }
1247
1248 pub fn time_format(&self, column: &str) -> Option<&TimeInterpretation> {
1250 self.time_formats
1251 .iter()
1252 .find(|interpretation| interpretation.column == column)
1253 }
1254
1255 pub fn time_value(&self, column: &str) -> Expr {
1258 self.time_format(column)
1259 .map(TimeInterpretation::expr)
1260 .unwrap_or_else(|| col(column))
1261 }
1262
1263 pub fn reads_as_time(&self, column: &str, schema: &Schema) -> bool {
1266 self.time_format(column).is_some() || schema.get(column).is_some_and(DataType::is_temporal)
1267 }
1268
1269 pub fn role_column(&self, role: TemporalRole) -> Option<&str> {
1271 self.temporal_roles
1272 .iter()
1273 .find(|assignment| assignment.role == role)
1274 .map(|assignment| assignment.column.as_str())
1275 }
1276
1277 pub fn interval_pairs(&self) -> Vec<(TemporalRole, TemporalRole)> {
1280 let assigned = |(start, end): &(TemporalRole, TemporalRole)| {
1281 self.role_column(*start).is_some() && self.role_column(*end).is_some()
1282 };
1283 match &self.intervals {
1284 None => INTERVAL_PAIRS.into_iter().filter(assigned).collect(),
1285 Some(chosen) => chosen.iter().copied().filter(assigned).collect(),
1286 }
1287 }
1288
1289 pub fn candidate_pairs(&self) -> Vec<(TemporalRole, TemporalRole)> {
1292 let roles = TemporalRole::ALL
1293 .into_iter()
1294 .filter(|role| self.role_column(*role).is_some())
1295 .collect::<Vec<_>>();
1296 let mut pairs = INTERVAL_PAIRS
1297 .into_iter()
1298 .filter(|(start, end)| roles.contains(start) && roles.contains(end))
1299 .collect::<Vec<_>>();
1300 for start in &roles {
1301 for end in &roles {
1302 if start != end && !pairs.contains(&(*start, *end)) {
1303 pairs.push((*start, *end));
1304 }
1305 }
1306 }
1307 pairs
1308 }
1309
1310 pub fn toggle_interval(&mut self, pair: (TemporalRole, TemporalRole)) {
1313 let mut chosen = self.interval_pairs();
1314 match chosen.iter().position(|chosen| *chosen == pair) {
1315 Some(index) => {
1316 chosen.remove(index);
1317 }
1318 None => chosen.push(pair),
1319 }
1320 self.intervals = Some(chosen);
1321 }
1322
1323 pub fn unpaired_roles(&self) -> Vec<TemporalRole> {
1325 let pairs = self.interval_pairs();
1326 TemporalRole::ALL
1327 .into_iter()
1328 .filter(|role| {
1329 self.role_column(*role).is_some()
1330 && !pairs
1331 .iter()
1332 .any(|(start, end)| start == role || end == role)
1333 })
1334 .collect()
1335 }
1336
1337 pub fn windows_intervals(&self) -> bool {
1339 matches!(self.grain, QualityGrain::TimeWindows { .. }) && !self.interval_pairs().is_empty()
1340 }
1341
1342 pub fn interval_grain(&self, start: &str, end: &str) -> QualityGrain {
1345 match (&self.grain, self.interval_clock) {
1346 (QualityGrain::TimeWindows { every, .. }, IntervalClock::Start) => {
1347 QualityGrain::TimeWindows {
1348 column: start.to_string(),
1349 every: every.clone(),
1350 }
1351 }
1352 (QualityGrain::TimeWindows { every, .. }, IntervalClock::End) => {
1353 QualityGrain::TimeWindows {
1354 column: end.to_string(),
1355 every: every.clone(),
1356 }
1357 }
1358 (grain, _) => grain.clone(),
1359 }
1360 }
1361
1362 pub fn zoned(&self, column: &str, schema: &Schema) -> Option<bool> {
1365 if let Some(format) = self.time_format(column) {
1366 return Some(format.zoned());
1367 }
1368 match schema.get(column)? {
1369 DataType::Datetime(_, zone) => Some(zone.is_some()),
1370 DataType::Date => Some(false),
1371 _ => None,
1372 }
1373 }
1374
1375 pub fn same_measurement(&self, other: &Self) -> bool {
1380 let measured = |plan: &Self| Self {
1381 expected: None,
1382 comparison: QualityComparison::None,
1383 baseline_segment: None,
1384 ..plan.clone()
1385 };
1386 measured(self) == measured(other)
1387 }
1388
1389 pub fn compares_differently(&self, other: &Self) -> bool {
1391 self.comparison != other.comparison || self.baseline_segment != other.baseline_segment
1392 }
1393
1394 pub fn expected_windows(&self) -> Option<&ExpectedWindows> {
1396 matches!(self.grain, QualityGrain::TimeWindows { .. })
1397 .then_some(self.expected.as_ref())
1398 .flatten()
1399 }
1400
1401 pub fn coarser_grain(&self) -> Option<QualityGrain> {
1405 match &self.grain {
1406 QualityGrain::TimeWindows { column, every } => {
1407 let coarser = match every.as_str() {
1408 "1h" => "1d",
1409 "1d" => "1w",
1410 "1w" => "1mo",
1411 _ => return None,
1412 };
1413 Some(QualityGrain::TimeWindows {
1414 column: column.clone(),
1415 every: coarser.to_string(),
1416 })
1417 }
1418 QualityGrain::RowChunks(rows) if *rows < DEFAULT_CHUNK_ROWS => {
1419 Some(QualityGrain::RowChunks(DEFAULT_CHUNK_ROWS))
1420 }
1421 _ => None,
1422 }
1423 }
1424
1425 pub fn set_row_chunks(&mut self) {
1426 self.grain = QualityGrain::RowChunks(DEFAULT_CHUNK_ROWS);
1427 }
1428
1429 pub fn compact_summary(&self) -> String {
1430 format!(
1431 "scope {} -> grain {} -> compute {} -> compare {}",
1432 self.scope.label(),
1433 self.grain.label(),
1434 match self.compute {
1435 QualityCompute::Sample => format!(
1436 "{} rows {}",
1437 self.dataset_rows,
1438 self.method.label().to_lowercase()
1439 ),
1440 other => other.label().to_string(),
1441 },
1442 self.comparison_label()
1443 )
1444 }
1445}
1446
1447#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1448pub enum QualityPrecision {
1449 Metadata,
1450 Sampled,
1451 Exact,
1452 Estimated,
1453}
1454
1455impl QualityPrecision {
1456 pub fn label(self) -> &'static str {
1457 match self {
1458 Self::Metadata => "metadata",
1459 Self::Sampled => "sampled",
1460 Self::Exact => "exact",
1461 Self::Estimated => "estimated",
1462 }
1463 }
1464}
1465
1466#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1467pub enum QualityMetric {
1468 #[default]
1469 NullRate,
1470 EmptyRate,
1471 WhitespaceRate,
1472 NonFiniteRate,
1473 DistinctShare,
1474 IntegerParseShare,
1475 DecimalParseShare,
1476}
1477
1478impl QualityMetric {
1479 pub const ALL: [Self; 7] = [
1480 Self::NullRate,
1481 Self::EmptyRate,
1482 Self::WhitespaceRate,
1483 Self::NonFiniteRate,
1484 Self::DistinctShare,
1485 Self::IntegerParseShare,
1486 Self::DecimalParseShare,
1487 ];
1488
1489 pub fn label(self) -> &'static str {
1490 match self {
1491 Self::NullRate => "Null rate",
1492 Self::EmptyRate => "Empty rate",
1493 Self::WhitespaceRate => "Whitespace rate",
1494 Self::NonFiniteRate => "Non-finite rate",
1495 Self::DistinctShare => "Distinct share",
1496 Self::IntegerParseShare => "Integer parse share",
1497 Self::DecimalParseShare => "Decimal parse share",
1498 }
1499 }
1500
1501 pub fn denominator(self, column: &ColumnQualityProfile) -> usize {
1503 match self {
1504 Self::NullRate | Self::EmptyRate | Self::WhitespaceRate | Self::NonFiniteRate => {
1505 column.evaluated_rows
1506 }
1507 Self::DistinctShare | Self::IntegerParseShare | Self::DecimalParseShare => {
1508 column.non_null_rows()
1509 }
1510 }
1511 }
1512
1513 pub fn short_label(self) -> &'static str {
1515 match self {
1516 Self::NullRate => "nulls",
1517 Self::EmptyRate => "empty",
1518 Self::WhitespaceRate => "blank",
1519 Self::NonFiniteRate => "NaN/inf",
1520 Self::DistinctShare => "distinct",
1521 Self::IntegerParseShare => "integer parse",
1522 Self::DecimalParseShare => "decimal parse",
1523 }
1524 }
1525
1526 pub fn value(self, column: &ColumnQualityProfile) -> Option<f64> {
1527 let ratio = |numerator: usize, denominator: usize| {
1528 (denominator > 0).then(|| numerator as f64 / denominator as f64)
1529 };
1530 match self {
1531 Self::NullRate => ratio(column.null_count, column.evaluated_rows),
1532 Self::EmptyRate => ratio(column.empty_count?, column.evaluated_rows),
1533 Self::WhitespaceRate => ratio(column.whitespace_count?, column.evaluated_rows),
1534 Self::NonFiniteRate => ratio(
1535 column.nan_count?
1536 + column.positive_infinity_count?
1537 + column.negative_infinity_count?,
1538 column.evaluated_rows,
1539 ),
1540 Self::DistinctShare => ratio(column.distinct_count?, column.non_null_rows()),
1541 Self::IntegerParseShare => ratio(column.integer_parse_count?, column.non_null_rows()),
1542 Self::DecimalParseShare => ratio(column.decimal_parse_count?, column.non_null_rows()),
1543 }
1544 }
1545}
1546
1547#[derive(Debug, Clone)]
1548pub struct ColumnQualityProfile {
1549 pub name: String,
1550 pub dtype: DataType,
1551 pub evaluated_rows: usize,
1552 pub null_count: usize,
1553 pub empty_count: Option<usize>,
1554 pub whitespace_count: Option<usize>,
1555 pub nan_count: Option<usize>,
1556 pub positive_infinity_count: Option<usize>,
1557 pub negative_infinity_count: Option<usize>,
1558 pub distinct_count: Option<usize>,
1559 pub min: Option<String>,
1560 pub max: Option<String>,
1561 pub integer_parse_count: Option<usize>,
1562 pub decimal_parse_count: Option<usize>,
1563 pub date_parse_count: Option<usize>,
1564 pub datetime_parse_count: Option<usize>,
1565 pub leading_zero_count: Option<usize>,
1568 pub dominant_value: Option<String>,
1569 pub dominant_count: Option<usize>,
1570 pub min_length: Option<usize>,
1571 pub max_length: Option<usize>,
1572}
1573
1574impl ColumnQualityProfile {
1575 pub fn null_rate(&self) -> f64 {
1576 rate(self.null_count, self.evaluated_rows)
1577 }
1578
1579 pub fn non_null_rows(&self) -> usize {
1580 self.evaluated_rows.saturating_sub(self.null_count)
1581 }
1582
1583 pub fn uniqueness_rate(&self) -> Option<f64> {
1584 self.distinct_count
1585 .map(|count| rate(count, self.non_null_rows()))
1586 }
1587}
1588
1589#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1590pub enum ObservationKind {
1591 Nulls,
1592 Empty,
1593 Whitespace,
1594 NonFinite,
1595 Constant,
1596 ParseableText,
1597 DuplicateRows,
1598 CategoryVariants,
1599 Absent,
1602 TypeConflict,
1605 KeyLike,
1608 UnparsedTime,
1610 KeyRepeated,
1612 KeyMissing,
1614 RequiredMissing,
1616 NotAllowed,
1618 OutOfRange,
1620 UnparsedNumber,
1622 Clipping,
1624 ZeroRuns,
1626 DcOffset,
1628}
1629
1630impl ObservationKind {
1631 pub fn label(self) -> &'static str {
1632 match self {
1633 Self::Nulls => "Null",
1634 Self::Empty => "Empty",
1635 Self::Whitespace => "Whitespace",
1636 Self::NonFinite => "Non-finite",
1637 Self::Constant => "Constant",
1638 Self::ParseableText => "Stored as text",
1639 Self::DuplicateRows => "Duplicate rows",
1640 Self::CategoryVariants => "Category variants",
1641 Self::Absent => "Absent",
1642 Self::TypeConflict => "Type conflict",
1643 Self::KeyLike => "Key-like",
1644 Self::UnparsedTime => "Unparsed time",
1645 Self::KeyRepeated => "Repeated key",
1646 Self::KeyMissing => "Incomplete key",
1647 Self::RequiredMissing => "Required, missing",
1648 Self::NotAllowed => "Not allowed",
1649 Self::OutOfRange => "Out of range",
1650 Self::UnparsedNumber => "Unparsed number",
1651 Self::Clipping => "Clipping",
1652 Self::ZeroRuns => "Zero runs",
1653 Self::DcOffset => "DC offset",
1654 }
1655 }
1656
1657 pub fn definition(self) -> &'static str {
1659 match self {
1660 Self::Nulls => "Null values / evaluated rows",
1661 Self::Empty => "Exact empty strings / evaluated rows",
1662 Self::Whitespace => "Nonempty strings that trim to empty / evaluated rows",
1663 Self::NonFinite => "NaN or positive/negative infinity / evaluated rows",
1664 Self::Constant => "One distinct non-null value in evaluated rows",
1665 Self::ParseableText => "Values parseable as a typed value, stored as text",
1666 Self::DuplicateRows => "Equal complete rows; extras = sum(group size - 1)",
1667 Self::CategoryVariants => "Distinct originals equal after trim and lowercase",
1668 Self::Absent => "Rows in files whose footer has no such column / source rows",
1669 Self::TypeConflict => {
1670 "Rows in files holding the column in an unreadable type / source rows"
1671 }
1672 Self::KeyLike => {
1673 "Non-null rows - distinct values, where distinct >= 95% of non-null rows"
1674 }
1675 Self::UnparsedTime => {
1676 "Non-null text the chosen time format does not read / non-null values"
1677 }
1678 Self::KeyRepeated => "Rows sharing a declared key value / rows checked",
1679 Self::KeyMissing => "Rows with no value in part of the declared key / rows checked",
1680 Self::RequiredMissing => "Null values in a required column / rows checked",
1681 Self::NotAllowed => "Values not in the declared set / non-null values",
1682 Self::OutOfRange => "Values below the minimum or above the maximum / values read",
1683 Self::UnparsedNumber => {
1684 "Non-null text that does not read as the number / non-null values"
1685 }
1686 Self::Clipping => "Samples in runs of 3 or more at full scale / samples",
1687 Self::ZeroRuns => {
1688 "Samples in runs of exact zeros 10 ms or longer (16 samples at least) / samples"
1689 }
1690 Self::DcOffset => "The channel's mean / full scale; noted from 1%",
1691 }
1692 }
1693}
1694
1695#[derive(Debug, Clone, PartialEq, Eq)]
1697pub struct QualityFileEvidence {
1698 pub number: usize,
1700 pub name: String,
1701 pub rows: usize,
1703 pub stored_type: Option<String>,
1706 pub examples: Vec<String>,
1709}
1710
1711#[derive(Debug, Clone)]
1712pub struct QualityObservation {
1713 pub kind: ObservationKind,
1714 pub column: String,
1715 pub affected_rows: usize,
1716 pub evaluated_rows: usize,
1717 pub fact: String,
1718 pub normalized_category: Option<String>,
1719 pub files: Vec<QualityFileEvidence>,
1723 pub time_format: Option<TimeInterpretation>,
1725 pub full_scale: Option<(f64, f64)>,
1728}
1729
1730impl QualityObservation {
1731 pub fn evidence_scope(&self) -> Option<QualityScope> {
1735 if !matches!(
1736 self.kind,
1737 ObservationKind::Absent | ObservationKind::TypeConflict
1738 ) || self.files.is_empty()
1739 {
1740 return None;
1741 }
1742 Some(QualityScope::SourceFiles(
1743 self.files.iter().map(|file| file.number).collect(),
1744 ))
1745 }
1746
1747 pub fn evidence_predicate(&self) -> Option<Expr> {
1748 let value = col(&self.column);
1749 match self.kind {
1750 ObservationKind::Nulls => Some(value.is_null()),
1751 ObservationKind::Empty => Some(value.eq(lit(""))),
1752 ObservationKind::Whitespace => Some(
1753 value
1754 .clone()
1755 .cast(DataType::String)
1756 .str()
1757 .strip_chars(lit(LiteralValue::untyped_null()))
1758 .eq(lit(""))
1759 .and(value.neq(lit(""))),
1760 ),
1761 ObservationKind::NonFinite => Some(
1762 value
1763 .clone()
1764 .is_nan()
1765 .or(value.clone().eq(lit(f64::INFINITY)))
1766 .or(value.eq(lit(f64::NEG_INFINITY))),
1767 ),
1768 ObservationKind::Constant => Some(value.is_not_null()),
1769 ObservationKind::CategoryVariants => Some(
1770 value
1771 .cast(DataType::String)
1772 .str()
1773 .strip_chars(lit(LiteralValue::untyped_null()))
1774 .str()
1775 .to_lowercase()
1776 .eq(lit(self.normalized_category.clone()?)),
1777 ),
1778 ObservationKind::KeyLike => {
1782 Some(value.clone().is_duplicated().and(value.is_not_null()))
1783 }
1784 ObservationKind::UnparsedTime => {
1785 self.time_format.as_ref().map(TimeInterpretation::unparsed)
1786 }
1787 ObservationKind::Clipping => self
1790 .full_scale
1791 .map(|(low, high)| value.clone().lt_eq(lit(low)).or(value.gt_eq(lit(high)))),
1792 ObservationKind::ZeroRuns => Some(value.eq(lit(0))),
1793 ObservationKind::ParseableText
1798 | ObservationKind::DuplicateRows
1799 | ObservationKind::Absent
1800 | ObservationKind::TypeConflict
1801 | ObservationKind::KeyRepeated
1802 | ObservationKind::KeyMissing
1803 | ObservationKind::RequiredMissing
1804 | ObservationKind::NotAllowed
1805 | ObservationKind::OutOfRange
1806 | ObservationKind::UnparsedNumber
1807 | ObservationKind::DcOffset => None,
1808 }
1809 }
1810}
1811
1812#[derive(Debug, Clone)]
1813pub struct IdentityProfile {
1814 pub duplicate_groups: usize,
1815 pub extra_rows: usize,
1816 pub rows_involved: usize,
1817 pub evaluated_rows: usize,
1818 pub precision: QualityPrecision,
1819 pub examples: Vec<DuplicateExample>,
1822}
1823
1824#[derive(Debug, Clone, PartialEq, Eq)]
1826pub struct DuplicateExample {
1827 pub copies: usize,
1828 pub values: Vec<String>,
1830}
1831
1832#[derive(Debug, Clone, PartialEq, Eq)]
1834pub struct FindingExamples {
1835 pub kind: ObservationKind,
1836 pub column: String,
1837 pub values: Vec<String>,
1839}
1840
1841#[derive(Debug, Clone)]
1842pub struct CategoryVariantGroup {
1843 pub column: String,
1844 pub normalized: String,
1845 pub variants: Vec<(String, usize)>,
1846 pub rows_involved: usize,
1847 pub complete: bool,
1848}
1849
1850#[derive(Debug, Clone)]
1851pub struct SegmentQualityProfile {
1852 pub label: String,
1853 pub total_rows: Option<usize>,
1854 pub evaluated_rows: usize,
1855 pub columns: Vec<ColumnQualityProfile>,
1856 pub null_cells: usize,
1857 pub null_rate: f64,
1858 pub compared_with: Option<String>,
1859 pub largest_change: Option<String>,
1860 pub change_size: Option<f64>,
1863}
1864
1865#[derive(Debug, Clone, PartialEq)]
1867pub struct SegmentChange {
1868 pub column: String,
1869 pub metric: QualityMetric,
1870 pub before: Option<f64>,
1871 pub now: f64,
1872 pub clear: bool,
1875}
1876
1877impl SegmentChange {
1878 pub fn change(&self) -> Option<f64> {
1880 self.before.map(|before| (self.now - before) * 100.0)
1881 }
1882}
1883
1884pub fn segment_order(results: &DataQualityResults, by_change: bool) -> Vec<usize> {
1887 let mut order = (0..results.segments.len()).collect::<Vec<_>>();
1888 if by_change {
1889 order.sort_by(|&left, &right| {
1890 let size = |index: usize| results.segments[index].change_size.unwrap_or(-1.0);
1891 size(right).total_cmp(&size(left))
1892 });
1893 }
1894 order
1895}
1896
1897pub fn segment_changes(results: &DataQualityResults, index: usize) -> Vec<SegmentChange> {
1901 let Some(segment) = results.segments.get(index) else {
1902 return Vec::new();
1903 };
1904 let compared = segment
1905 .compared_with
1906 .as_ref()
1907 .and_then(|label| results.segments.iter().find(|other| &other.label == label));
1908 let mut changes = Vec::new();
1909 for column in &segment.columns {
1910 let prior = compared.and_then(|other| other.columns.iter().find(|c| c.name == column.name));
1911 for metric in CHANGE_MEASURES {
1912 let Some(now) = metric.value(column) else {
1913 continue;
1914 };
1915 let before = prior.and_then(|prior| metric.value(prior));
1916 if now == 0.0 && before.unwrap_or(0.0) == 0.0 {
1917 continue;
1918 }
1919 let clear = match (prior, before) {
1920 (Some(prior), Some(before)) => {
1921 (now - before).abs() * 100.0 >= MATERIAL_CHANGE_PP
1922 && (results.precision == QualityPrecision::Exact
1923 || beyond_noise(
1924 now,
1925 metric.denominator(column),
1926 before,
1927 metric.denominator(prior),
1928 ))
1929 }
1930 _ => false,
1931 };
1932 changes.push(SegmentChange {
1933 column: column.name.clone(),
1934 metric,
1935 before,
1936 now,
1937 clear,
1938 });
1939 }
1940 }
1941 if compared.is_some() {
1942 changes.sort_by(|left, right| {
1944 let size = |change: &SegmentChange| change.change().unwrap_or(0.0).abs();
1945 right
1946 .clear
1947 .cmp(&left.clear)
1948 .then_with(|| size(right).total_cmp(&size(left)))
1949 });
1950 } else {
1951 changes.sort_by(|left, right| right.now.total_cmp(&left.now));
1952 }
1953 changes
1954}
1955
1956#[derive(Debug, Clone)]
1957pub struct TemporalLatencyProfile {
1958 pub segment: String,
1959 pub start_role: TemporalRole,
1960 pub end_role: TemporalRole,
1961 pub start_column: String,
1962 pub end_column: String,
1963 pub evaluated_rows: usize,
1965 pub paired_rows: usize,
1969 pub missing_start: usize,
1970 pub missing_end: usize,
1971 pub unparsed_start: usize,
1973 pub unparsed_end: usize,
1974 pub negative_count: usize,
1976 pub zero_count: usize,
1978 pub p50_seconds: Option<i64>,
1979 pub p90_seconds: Option<i64>,
1980 pub p95_seconds: Option<i64>,
1981 pub p99_seconds: Option<i64>,
1982 pub max_seconds: Option<i64>,
1983 pub threshold_seconds: Option<i64>,
1986 pub above_threshold_count: Option<usize>,
1987}
1988
1989impl TemporalLatencyProfile {
1990 pub fn pair(&self) -> (TemporalRole, TemporalRole) {
1991 (self.start_role, self.end_role)
1992 }
1993
1994 pub fn label(&self) -> String {
1996 interval_label(self.pair())
1997 }
1998
1999 pub fn is_validity(&self) -> bool {
2002 self.pair() == (TemporalRole::ValidFrom, TemporalRole::ValidTo)
2003 }
2004
2005 pub fn count(&self, fact: IntervalFact, plan: &DataQualityPlan) -> Option<(usize, usize)> {
2008 let rows = self.evaluated_rows;
2009 let paired = self.paired_rows;
2010 match fact {
2011 IntervalFact::MissingStart => Some((self.missing_start, rows)),
2012 IntervalFact::MissingEnd => Some((self.missing_end, rows)),
2013 IntervalFact::UnparsedStart => plan
2014 .time_format(&self.start_column)
2015 .map(|_| (self.unparsed_start, rows)),
2016 IntervalFact::UnparsedEnd => plan
2017 .time_format(&self.end_column)
2018 .map(|_| (self.unparsed_end, rows)),
2019 IntervalFact::Negative => Some((self.negative_count, paired)),
2020 IntervalFact::Zero => Some((self.zero_count, paired)),
2021 IntervalFact::OverThreshold => self.above_threshold_count.map(|count| (count, paired)),
2022 }
2023 }
2024
2025 pub fn segment_opens(&self, plan: &DataQualityPlan) -> bool {
2028 let grain = plan.interval_grain(&self.start_column, &self.end_column);
2029 segment_predicate(plan, &grain, &self.segment, None).is_some()
2030 }
2031
2032 pub fn evidence_predicate(
2038 &self,
2039 fact: IntervalFact,
2040 plan: &DataQualityPlan,
2041 schema: Option<&Schema>,
2042 ) -> Option<Expr> {
2043 self.count(fact, plan)?;
2044 let micros = || interval_micros(plan, &self.start_column, &self.end_column);
2045 let rows = match fact {
2046 IntervalFact::MissingStart => col(self.start_column.as_str()).is_null(),
2047 IntervalFact::MissingEnd => col(self.end_column.as_str()).is_null(),
2048 IntervalFact::UnparsedStart => plan.time_format(&self.start_column)?.unparsed(),
2049 IntervalFact::UnparsedEnd => plan.time_format(&self.end_column)?.unparsed(),
2050 IntervalFact::Negative => micros().lt(lit(0i64)),
2051 IntervalFact::Zero => micros().eq(lit(0i64)),
2052 IntervalFact::OverThreshold => {
2053 micros().gt(lit(self.threshold_seconds?.saturating_mul(1_000_000)))
2054 }
2055 };
2056 let grain = plan.interval_grain(&self.start_column, &self.end_column);
2057 Some(
2058 match segment_predicate(plan, &grain, &self.segment, schema)? {
2059 Some(segment) => segment.and(rows),
2060 None => rows,
2061 },
2062 )
2063 }
2064}
2065
2066#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2068pub enum IntervalFact {
2069 MissingStart,
2070 MissingEnd,
2071 UnparsedStart,
2072 UnparsedEnd,
2073 Negative,
2074 Zero,
2075 OverThreshold,
2076}
2077
2078impl IntervalFact {
2079 pub const ALL: [Self; 7] = [
2080 Self::MissingStart,
2081 Self::MissingEnd,
2082 Self::UnparsedStart,
2083 Self::UnparsedEnd,
2084 Self::Negative,
2085 Self::Zero,
2086 Self::OverThreshold,
2087 ];
2088
2089 pub fn label(self, profile: &TemporalLatencyProfile) -> String {
2092 let validity = profile.is_validity();
2093 match self {
2094 Self::MissingStart => "Missing start".to_string(),
2095 Self::MissingEnd if validity => "Open, no end".to_string(),
2096 Self::MissingEnd => "Missing end".to_string(),
2097 Self::UnparsedStart => "Unparsed start".to_string(),
2098 Self::UnparsedEnd => "Unparsed end".to_string(),
2099 Self::Negative if validity => "Ends first".to_string(),
2100 Self::Negative => "Negative".to_string(),
2101 Self::Zero => "Zero".to_string(),
2102 Self::OverThreshold => format!(
2103 "Over {}",
2104 crate::analysis_modal::threshold_label(profile.threshold_seconds)
2105 ),
2106 }
2107 }
2108
2109 pub fn short(self) -> &'static str {
2111 match self {
2112 Self::MissingStart => "missing start",
2113 Self::MissingEnd => "missing end",
2114 Self::UnparsedStart => "unparsed start",
2115 Self::UnparsedEnd => "unparsed end",
2116 Self::Negative => "negative",
2117 Self::Zero => "zero",
2118 Self::OverThreshold => "over threshold",
2119 }
2120 }
2121}
2122
2123fn label_text(column: &str) -> Expr {
2131 col(column).map(
2132 |values| {
2133 let text = (0..values.len())
2134 .map(|row| {
2135 let value = values.get(row)?;
2136 Ok((!value.is_null()).then(|| crate::exact::str_value(&value).into_owned()))
2137 })
2138 .collect::<PolarsResult<StringChunked>>()?;
2139 Ok(text.with_name(values.name().clone()).into_column())
2140 },
2141 |_, field| Ok(Field::new(field.name().clone(), DataType::String)),
2142 )
2143}
2144
2145fn partition_label_predicate(column: &str, value: &str, schema: Option<&Schema>) -> Expr {
2150 let writes_each_once = |dtype: &&DataType| {
2151 dtype.is_integer()
2152 || matches!(
2153 dtype,
2154 DataType::String
2155 | DataType::Boolean
2156 | DataType::Date
2157 | DataType::Decimal(..)
2158 | DataType::Categorical(..)
2159 | DataType::Enum(..)
2160 )
2161 };
2162 let native = schema
2163 .and_then(|schema| schema.get(column))
2164 .filter(writes_each_once)
2165 .and_then(|dtype| crate::typed_value::parse(value, dtype).ok())
2166 .filter(|scalar| crate::exact::str_value(scalar.value()) == value);
2167 match native {
2168 Some(scalar) => col(column).eq(lit(scalar)),
2169 None => label_text(column).eq(lit(value.to_string())),
2170 }
2171}
2172
2173fn segment_predicate(
2174 plan: &DataQualityPlan,
2175 grain: &QualityGrain,
2176 label: &str,
2177 schema: Option<&Schema>,
2178) -> Option<Option<Expr>> {
2179 match grain {
2180 QualityGrain::Dataset => Some(None),
2181 QualityGrain::Partition(column) => {
2182 let value = label.strip_prefix(&format!("{column}="))?;
2183 Some(Some(if value == "∅" {
2184 col(column.as_str()).is_null()
2185 } else {
2186 partition_label_predicate(column, value, schema)
2187 }))
2188 }
2189 QualityGrain::TimeWindows { column, every } => {
2190 let value = plan.time_value(column);
2191 if label == time_window_label(column, every, None) {
2193 return Some(Some(time_window_start(value, every).is_null()));
2194 }
2195 let date = |text: &str| chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d").ok();
2196 let start = match every.as_str() {
2197 "1h" => chrono::NaiveDateTime::parse_from_str(label, "%Y-%m-%d %H:%M").ok()?,
2198 "1d" => date(label)?.and_hms_opt(0, 0, 0)?,
2199 "1w" => date(label.strip_prefix("week of ")?)?.and_hms_opt(0, 0, 0)?,
2200 "1mo" => date(&format!("{label}-01"))?.and_hms_opt(0, 0, 0)?,
2201 _ => return None,
2202 };
2203 Some(Some(
2204 time_window_start(value, every).eq(lit(start.and_utc().timestamp_micros())
2205 .cast(DataType::Datetime(TimeUnit::Microseconds, None))),
2206 ))
2207 }
2208 QualityGrain::RowChunks(_) | QualityGrain::File => None,
2209 }
2210}
2211#[derive(Debug, Clone, PartialEq, Eq)]
2215pub struct SharedNulls {
2216 pub columns: Vec<String>,
2217 pub null_rows: usize,
2218 pub rows_null_in_all: usize,
2219}
2220
2221impl SharedNulls {
2222 pub fn same_rows(&self) -> bool {
2223 self.rows_null_in_all == self.null_rows
2224 }
2225}
2226
2227#[derive(Debug, Clone)]
2228pub struct DataQualityResults {
2229 pub total_rows: Option<usize>,
2230 pub evaluated_rows: usize,
2231 pub precision: QualityPrecision,
2232 pub sample_seed: u64,
2233 pub columns: Vec<ColumnQualityProfile>,
2234 pub observations: Vec<QualityObservation>,
2235 pub segments: Vec<SegmentQualityProfile>,
2236 pub temporal: Vec<TemporalLatencyProfile>,
2237 pub identity: Option<IdentityProfile>,
2238 pub category_variants: Vec<CategoryVariantGroup>,
2239 pub shared_nulls: Vec<SharedNulls>,
2240 pub source_files: Option<usize>,
2243 pub per_value: Option<usize>,
2245 pub footers_read: Option<usize>,
2248 pub reads: Option<ObservedReads>,
2251 pub examples: Vec<FindingExamples>,
2254 pub unsampled_segments: Vec<UnsampledSegment>,
2258 pub intent: Option<Box<crate::quality_intent::IntentResults>>,
2260 pub source: Option<Box<crate::quality_export::SourceIdentity>>,
2262}
2263
2264#[derive(Debug, Clone, PartialEq, Eq)]
2266pub struct UnsampledSegment {
2267 pub label: String,
2268 pub total_rows: usize,
2270}
2271
2272impl DataQualityResults {
2273 pub fn estimated_bytes(&self) -> usize {
2277 let profile = |column: &ColumnQualityProfile| {
2278 std::mem::size_of::<ColumnQualityProfile>()
2279 + column.name.len()
2280 + column.min.as_ref().map_or(0, String::len)
2281 + column.max.as_ref().map_or(0, String::len)
2282 + column.dominant_value.as_ref().map_or(0, String::len)
2283 };
2284 let segments = self
2285 .segments
2286 .iter()
2287 .map(|segment| {
2288 std::mem::size_of::<SegmentQualityProfile>()
2289 + segment.label.len()
2290 + segment.columns.iter().map(profile).sum::<usize>()
2291 })
2292 .sum::<usize>();
2293 let observations = self
2294 .observations
2295 .iter()
2296 .map(|observation| {
2297 std::mem::size_of::<QualityObservation>()
2298 + observation.fact.len()
2299 + observation.column.len()
2300 + observation.normalized_category.as_ref().map_or(0, String::len)
2301 + observation
2303 .files
2304 .iter()
2305 .map(|file| {
2306 std::mem::size_of::<QualityFileEvidence>()
2307 + file.name.len()
2308 + file.stored_type.as_ref().map_or(0, String::len)
2309 + file.examples.iter().map(String::len).sum::<usize>()
2310 })
2311 .sum::<usize>()
2312 })
2313 .sum::<usize>();
2314 let unsampled = self
2315 .unsampled_segments
2316 .iter()
2317 .map(|segment| std::mem::size_of::<UnsampledSegment>() + segment.label.len())
2318 .sum::<usize>();
2319 let texts = |values: &[String]| {
2320 values
2321 .iter()
2322 .map(|value| std::mem::size_of::<String>() + value.len())
2323 .sum::<usize>()
2324 };
2325 let spellings = self
2327 .category_variants
2328 .iter()
2329 .map(|group| {
2330 std::mem::size_of::<CategoryVariantGroup>()
2331 + group.column.len()
2332 + group.normalized.len()
2333 + group
2334 .variants
2335 .iter()
2336 .map(|(variant, _)| std::mem::size_of::<(String, usize)>() + variant.len())
2337 .sum::<usize>()
2338 })
2339 .sum::<usize>();
2340 let examples = self
2341 .examples
2342 .iter()
2343 .map(|found| std::mem::size_of::<FindingExamples>() + texts(&found.values))
2344 .sum::<usize>()
2345 + self.identity.as_ref().map_or(0, |identity| {
2346 identity
2347 .examples
2348 .iter()
2349 .map(|example| std::mem::size_of::<DuplicateExample>() + texts(&example.values))
2350 .sum()
2351 });
2352 let temporal = self
2353 .temporal
2354 .iter()
2355 .map(|latency| {
2356 std::mem::size_of::<TemporalLatencyProfile>()
2357 + latency.segment.len()
2358 + latency.start_column.len()
2359 + latency.end_column.len()
2360 })
2361 .sum::<usize>();
2362 let shared = self
2363 .shared_nulls
2364 .iter()
2365 .map(|shared| std::mem::size_of::<SharedNulls>() + texts(&shared.columns))
2366 .sum::<usize>();
2367 let intent = self.intent.as_ref().map_or(0, |intent| {
2369 let counted = |values: &[(String, usize)]| {
2370 values
2371 .iter()
2372 .map(|(value, _)| std::mem::size_of::<(String, usize)>() + value.len())
2373 .sum::<usize>()
2374 };
2375 std::mem::size_of::<crate::quality_intent::IntentResults>()
2376 + intent
2377 .columns
2378 .iter()
2379 .map(|check| {
2380 std::mem::size_of::<crate::quality_intent::ColumnCheck>()
2381 + check.lowest.as_ref().map_or(0, String::len)
2382 + check.highest.as_ref().map_or(0, String::len)
2383 + counted(&check.outside_examples)
2384 + counted(&check.unparsed_examples)
2385 })
2386 .sum::<usize>()
2387 });
2388 std::mem::size_of::<Self>()
2389 + self.columns.iter().map(profile).sum::<usize>()
2390 + segments
2391 + unsampled
2392 + observations
2393 + temporal
2394 + spellings
2395 + examples
2396 + shared
2397 + intent
2398 }
2399
2400 pub fn compare_segments(&mut self, plan: &DataQualityPlan) {
2401 apply_comparisons(
2402 &mut self.segments,
2403 plan.comparison,
2404 plan.baseline_segment.as_deref(),
2405 self.precision,
2406 );
2407 }
2408
2409 pub fn empty(total_rows: Option<usize>, plan: &DataQualityPlan, schema: &Schema) -> Self {
2410 Self {
2411 total_rows,
2412 evaluated_rows: 0,
2413 precision: QualityPrecision::Metadata,
2414 sample_seed: plan.sample_seed,
2415 columns: schema
2416 .iter()
2417 .map(|(name, dtype)| ColumnQualityProfile {
2418 name: name.to_string(),
2419 dtype: dtype.clone(),
2420 evaluated_rows: 0,
2421 null_count: 0,
2422 empty_count: None,
2423 whitespace_count: None,
2424 nan_count: None,
2425 positive_infinity_count: None,
2426 negative_infinity_count: None,
2427 distinct_count: None,
2428 min: None,
2429 max: None,
2430 integer_parse_count: None,
2431 decimal_parse_count: None,
2432 date_parse_count: None,
2433 datetime_parse_count: None,
2434 leading_zero_count: None,
2435 dominant_value: None,
2436 dominant_count: None,
2437 min_length: None,
2438 max_length: None,
2439 })
2440 .collect(),
2441 observations: Vec::new(),
2442 segments: Vec::new(),
2443 temporal: Vec::new(),
2444 identity: None,
2445 category_variants: Vec::new(),
2446 shared_nulls: Vec::new(),
2447 source_files: None,
2448 per_value: None,
2449 footers_read: None,
2450 reads: None,
2451 examples: Vec::new(),
2452 unsampled_segments: Vec::new(),
2453 intent: None,
2454 source: None,
2455 }
2456 }
2457
2458 pub fn examples_of(&self, kind: ObservationKind, column: &str) -> &[String] {
2460 self.examples
2461 .iter()
2462 .find(|examples| examples.kind == kind && examples.column == column)
2463 .map(|examples| examples.values.as_slice())
2464 .unwrap_or_default()
2465 }
2466}
2467
2468#[derive(Debug, Clone)]
2477pub struct QualitySample {
2478 df: DataFrame,
2479 positions: Vec<IdxSize>,
2481 precision: QualityPrecision,
2482 total_rows: Option<usize>,
2483 per_value: Option<crate::sampling::PerValue>,
2484 counted: Vec<(SegmentKey, SegmentCounts)>,
2488 too_many: Vec<SegmentKey>,
2491}
2492
2493type SegmentCounts = BTreeMap<Option<String>, usize>;
2495
2496type SegmentKey = (QualityGrain, Option<TimeInterpretation>);
2498
2499fn segment_key(plan: &DataQualityPlan) -> SegmentKey {
2500 let format = match &plan.grain {
2501 QualityGrain::TimeWindows { column, .. } => plan.time_format(column).cloned(),
2502 _ => None,
2503 };
2504 (plan.grain.clone(), format)
2505}
2506
2507#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2511pub enum CopyPlan {
2512 #[default]
2514 NotApplicable,
2515 Passes(NoCopy),
2517 Fetch { bytes: u64, objects: usize },
2519 Kept { bytes: u64, objects: usize },
2521}
2522
2523#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2525pub enum NoCopy {
2526 Off,
2528 SizeUnknown,
2530 Unusable,
2532 PartOfTheSource,
2535 TooLarge { bytes: u64, limit: u64 },
2537 NoRoom { bytes: u64, free: Option<u64> },
2539}
2540
2541#[derive(Debug, Clone, PartialEq, Eq, Default)]
2543pub enum SegmentCount {
2544 #[default]
2547 NotNeeded,
2548 PerValue,
2550 InSamplePass,
2552 Retained,
2554 RolledUp(String),
2557 CountPass,
2559 TooMany,
2561}
2562
2563impl SegmentCount {
2564 pub fn reads(&self) -> bool {
2566 *self == Self::CountPass
2567 }
2568}
2569
2570pub fn window_cadence(every: &str) -> &str {
2572 match every {
2573 "1h" => "hourly",
2574 "1d" => "daily",
2575 "1w" => "weekly",
2576 "1mo" => "monthly",
2577 other => other,
2578 }
2579}
2580
2581pub fn window_nests(fine: &str, coarse: &str) -> bool {
2587 matches!(
2588 (fine, coarse),
2589 ("1h", "1d" | "1w" | "1mo") | ("1d", "1w" | "1mo")
2590 )
2591}
2592
2593impl QualitySample {
2594 pub fn needs_segment_count(&self, plan: &DataQualityPlan) -> bool {
2598 self.segment_count(plan).reads()
2599 }
2600
2601 pub fn segment_count(&self, plan: &DataQualityPlan) -> SegmentCount {
2603 if self.precision != QualityPrecision::Sampled || !segments_need_count(plan) {
2604 return SegmentCount::NotNeeded;
2605 }
2606 if per_value_counts(plan, self.per_value.as_ref()) {
2607 return SegmentCount::PerValue;
2608 }
2609 let key = segment_key(plan);
2610 if self.too_many.contains(&key) {
2611 return SegmentCount::TooMany;
2612 }
2613 if self.counted.iter().any(|(counted, _)| *counted == key) {
2614 return SegmentCount::Retained;
2615 }
2616 match self.finer_count(&key) {
2617 Some(((QualityGrain::TimeWindows { every, .. }, _), _)) => {
2618 SegmentCount::RolledUp(every.clone())
2619 }
2620 _ => SegmentCount::CountPass,
2621 }
2622 }
2623
2624 fn finer_count(&self, key: &SegmentKey) -> Option<&(SegmentKey, SegmentCounts)> {
2627 let (QualityGrain::TimeWindows { column, every }, format) = key else {
2628 return None;
2629 };
2630 self.counted.iter().find(|((grain, counted_format), _)| {
2631 matches!(
2632 grain,
2633 QualityGrain::TimeWindows { column: counted, every: fine }
2634 if counted == column && window_nests(fine, every)
2635 ) && counted_format == format
2636 })
2637 }
2638
2639 pub fn df(&self) -> &DataFrame {
2641 &self.df
2642 }
2643
2644 pub fn estimated_bytes(&self) -> usize {
2647 let counts = self
2648 .counted
2649 .iter()
2650 .flat_map(|(_, counts)| counts.keys())
2651 .map(|key| key.as_ref().map_or(0, String::len) + 64)
2652 .sum::<usize>();
2653 let per_value = self.per_value.as_ref().map_or(0, |per_value| {
2654 per_value
2655 .totals
2656 .keys()
2657 .map(|key| key.as_ref().map_or(0, String::len) + 64)
2658 .sum()
2659 });
2660 self.df.estimated_size()
2661 + self.positions.len() * std::mem::size_of::<IdxSize>()
2662 + counts
2663 + per_value
2664 }
2665
2666 pub fn analysis_rows(&self, df: DataFrame) -> crate::statistics::AnalysisRows {
2668 crate::statistics::AnalysisRows {
2669 sample_size: (self.precision == QualityPrecision::Sampled).then_some(df.height()),
2670 total_rows: self.total_rows.unwrap_or(df.height()),
2671 per_value: self.per_value.clone(),
2672 df,
2673 }
2674 }
2675}
2676
2677pub fn segments_need_count(plan: &DataQualityPlan) -> bool {
2680 matches!(
2681 plan.grain,
2682 QualityGrain::Partition(_) | QualityGrain::TimeWindows { .. }
2683 )
2684}
2685
2686pub fn sampler_counts_segments(plan: &DataQualityPlan) -> bool {
2690 matches!(
2691 (&plan.grain, &plan.method),
2692 (
2693 QualityGrain::Partition(column),
2694 crate::sampling::SampleMethod::PerPartition { column: sampled },
2695 ) if column == sampled
2696 )
2697}
2698
2699pub fn fresh_segment_count(plan: &DataQualityPlan, may_read_blocks: bool) -> SegmentCount {
2704 if plan.compute != QualityCompute::Sample || !segments_need_count(plan) {
2705 return SegmentCount::NotNeeded;
2706 }
2707 if sampler_counts_segments(plan) {
2708 return SegmentCount::PerValue;
2709 }
2710 match plan.method {
2711 crate::sampling::SampleMethod::FirstRows => SegmentCount::CountPass,
2712 crate::sampling::SampleMethod::Spread if may_read_blocks => SegmentCount::CountPass,
2713 _ => SegmentCount::InSamplePass,
2714 }
2715}
2716
2717fn segment_count_key(plan: &DataQualityPlan) -> Option<Expr> {
2720 match &plan.grain {
2721 QualityGrain::Partition(column) => Some(col(column.as_str())),
2722 QualityGrain::TimeWindows { column, every } => {
2723 Some(time_window_start(plan.time_value(column), every))
2724 }
2725 _ => None,
2726 }
2727}
2728
2729fn per_value_counts(plan: &DataQualityPlan, per_value: Option<&crate::sampling::PerValue>) -> bool {
2732 sampler_counts_segments(plan) && per_value.is_some()
2733}
2734
2735pub fn compute_data_quality(
2736 lf: &LazyFrame,
2737 total_rows: Option<usize>,
2738 plan: &DataQualityPlan,
2739 source: Option<&QualitySourceContext>,
2740 polars_streaming: bool,
2741) -> Result<DataQualityResults> {
2742 compute_data_quality_kept(lf, total_rows, plan, source, polars_streaming, None)
2743 .map(|(results, _)| results)
2744}
2745
2746pub fn compute_data_quality_kept(
2749 lf: &LazyFrame,
2750 total_rows: Option<usize>,
2751 plan: &DataQualityPlan,
2752 source: Option<&QualitySourceContext>,
2753 polars_streaming: bool,
2754 kept: Option<&QualitySample>,
2755) -> Result<(DataQualityResults, Option<QualitySample>)> {
2756 let (results, kept) = compute_data_quality_watched(
2757 lf,
2758 total_rows,
2759 plan,
2760 source,
2761 polars_streaming,
2762 kept,
2763 &QualityWatch::default(),
2764 );
2765 results.map(|results| (results, kept))
2766}
2767
2768pub fn compute_data_quality_watched(
2774 lf: &LazyFrame,
2775 total_rows: Option<usize>,
2776 plan: &DataQualityPlan,
2777 source: Option<&QualitySourceContext>,
2778 polars_streaming: bool,
2779 kept: Option<&QualitySample>,
2780 watch: &QualityWatch,
2781) -> (Result<DataQualityResults>, Option<QualitySample>) {
2782 let mut acquired = None;
2783 let inputs = QualityInputs {
2784 lf,
2785 total_rows,
2786 plan,
2787 source,
2788 polars_streaming: polars_streaming && cfg!(feature = "streaming"),
2791 watch,
2792 };
2793 let results = profile_quality(inputs, kept, &mut acquired);
2794 (results, acquired)
2795}
2796
2797#[derive(Clone, Copy)]
2799struct QualityInputs<'a> {
2800 lf: &'a LazyFrame,
2801 total_rows: Option<usize>,
2802 plan: &'a DataQualityPlan,
2803 source: Option<&'a QualitySourceContext>,
2804 polars_streaming: bool,
2805 watch: &'a QualityWatch,
2806}
2807
2808fn profile_quality(
2811 inputs: QualityInputs<'_>,
2812 kept: Option<&QualitySample>,
2813 acquired: &mut Option<QualitySample>,
2814) -> Result<DataQualityResults> {
2815 let QualityInputs {
2816 lf,
2817 total_rows,
2818 plan,
2819 source,
2820 polars_streaming,
2821 watch,
2822 } = inputs;
2823 watch.stage(QualityStage::Preparing, false, false)?;
2824 let collected_schema = lf.clone().collect_schema()?;
2825 let schema = visible_schema(&collected_schema, source);
2826 if plan.compute == QualityCompute::Metadata {
2829 watch.stage(QualityStage::Assembling, false, false)?;
2830 let mut results = DataQualityResults::empty(total_rows, plan, &schema);
2831 if let Some(source) = source {
2832 results.observations = drift_observations(source, None, polars_streaming, watch);
2833 }
2834 results.source_files = source.map(|source| source.file_names.len());
2835 results.footers_read = source.map(|source| source.footers_read);
2836 results.reads = Some(watch.observed());
2837 results.intent =
2838 crate::quality_intent::IntentResults::unmeasured(plan, &schema).map(Box::new);
2839 return Ok(results);
2840 }
2841 let grain_column = match &plan.grain {
2842 QualityGrain::Partition(column) | QualityGrain::TimeWindows { column, .. } => Some(column),
2843 _ => None,
2844 };
2845 if let Some(column) = grain_column
2846 && collected_schema.get(column).is_none()
2847 {
2848 return Err(Report::msg(format!(
2849 "Grain column {column} is not in scope {}; choose another grain or scope",
2850 plan.scope.label()
2851 )));
2852 }
2853 if let QualityGrain::TimeWindows { column, .. } = &plan.grain
2854 && !plan.reads_as_time(column, &collected_schema)
2855 {
2856 return Err(Report::msg(format!(
2857 "Grain column {column} is text; choose a format for it under Text as time"
2858 )));
2859 }
2860 if plan.compute == QualityCompute::Full {
2861 let total_rows = match total_rows {
2862 Some(rows) => rows,
2863 None => {
2864 watch.stage(QualityStage::CountingRows, watch.scope_reads(true), false)?;
2867 let count = collect_lazy(
2868 crate::widgets::datatable::row_count_lf(lf),
2869 polars_streaming,
2870 )
2871 .map_err(Report::from)?;
2872 let count_values = count
2873 .get(0)
2874 .ok_or_else(|| Report::msg("Data quality row count was not returned"))?;
2875 let Some(AnyValue::UInt64(rows)) = count_values.first() else {
2876 return Err(Report::msg("Data quality row count was not UInt64"));
2877 };
2878 *rows as usize
2879 }
2880 };
2881 if total_rows == 0 && plan.scope != QualityScope::CurrentView {
2882 return Err(crate::sampling::no_rows_error(&plan.scope));
2883 }
2884 return compute_full_quality(
2885 lf,
2886 total_rows,
2887 plan,
2888 source,
2889 &schema,
2890 polars_streaming,
2891 watch,
2892 );
2893 }
2894
2895 let kept = acquired.insert(match kept {
2900 Some(kept) => {
2901 watch.stage(QualityStage::ReusingSample, false, false)?;
2902 kept.clone()
2903 }
2904 None => {
2905 let interruptible = plan.method != crate::sampling::SampleMethod::FirstRows
2910 && cfg!(feature = "streaming");
2911 watch.stage(QualityStage::ReadingSample, true, interruptible)?;
2912 read_quality_sample(lf, total_rows, plan, polars_streaming, watch)?
2913 }
2914 });
2915 let profile_df = kept.df.clone();
2916 let sample_positions = kept.positions.clone();
2917 let evaluated_rows = profile_df.height();
2918 let precision = kept.precision;
2919 let total_rows = kept.total_rows;
2920
2921 if total_rows == Some(0) && plan.scope != QualityScope::CurrentView {
2924 return Err(crate::sampling::no_rows_error(&plan.scope));
2925 }
2926 let profile_df = attach_source_file(profile_df, source)?;
2927 watch.stage(QualityStage::ProfilingColumns, false, false)?;
2928 let mut columns = profile_columns(&profile_df, &schema, polars_streaming)?;
2929 let profile_lf = profile_df.clone().lazy();
2933 add_dominance_lazy(&profile_lf, &mut columns, polars_streaming)?;
2934 let mut formats = interpretation_exprs(plan, &collected_schema);
2937 formats.extend(crate::quality_intent::intent_exprs(plan, &schema));
2938 let unparsed = if formats.is_empty() {
2939 DataFrame::default()
2940 } else {
2941 collect_lazy(profile_lf.clone().select(formats), polars_streaming).map_err(Report::from)?
2942 };
2943 watch.stage(QualityStage::CheckingDuplicates, false, false)?;
2944 let identity = profile_identity_lazy(
2945 &profile_lf,
2946 &schema,
2947 evaluated_rows,
2948 precision,
2949 polars_streaming,
2950 )?;
2951 let repeats = crate::quality_intent::key_repeats(&profile_lf, plan, &schema, polars_streaming)?;
2954 let intent = crate::quality_intent::IntentResults::from_counts(
2955 plan,
2956 &schema,
2957 &unparsed,
2958 repeats,
2959 evaluated_rows,
2960 precision,
2961 Some(&profile_lf),
2962 )?
2963 .map(Box::new);
2964 watch.stage(QualityStage::CheckingSpellings, false, false)?;
2965 let category_variants = profile_category_variants_lazy(&profile_lf, &schema, polars_streaming)?;
2966 let mut observations = observations_from_profiles(&columns, precision);
2967 observations.extend(interpretation_observations(
2968 &unparsed,
2969 plan,
2970 &collected_schema,
2971 ));
2972 observations.extend(identity_observations(&identity, &category_variants));
2973 if let Some(intent) = &intent {
2974 observations.extend(intent.observations());
2975 }
2976 crate::quality_intent::supersede(&mut observations, plan);
2977 let mut identity = identity;
2980 if identity.duplicate_groups > 0 {
2981 identity.examples = duplicate_examples(&profile_lf, &schema, polars_streaming)?;
2982 }
2983 let examples = finding_examples(&profile_lf, &columns, &observations, polars_streaming)?;
2984 if let Some(source) = source {
2987 observations.extend(drift_observations(source, None, polars_streaming, watch));
2988 }
2989 let totals = {
2990 let mut totals = known_segment_totals(plan, total_rows, source);
2991 if precision == QualityPrecision::Sampled {
2992 totals.extend(sampled_segment_totals(
2993 lf,
2994 plan,
2995 kept,
2996 polars_streaming,
2997 watch,
2998 )?);
2999 }
3000 totals
3001 };
3002 watch.stage(QualityStage::ProfilingSegments, false, false)?;
3003 let (segments, unsampled_segments) = profile_segments(
3004 &profile_df,
3005 total_rows,
3006 plan,
3007 precision,
3008 &schema,
3009 SegmentSampleProvenance {
3010 positions: Some(sample_positions.as_slice()),
3011 totals: &totals,
3012 },
3013 polars_streaming,
3014 )?;
3015 watch.stage(QualityStage::ComputingIntervals, false, false)?;
3016 let temporal = profile_temporal(&profile_df, plan, Some(sample_positions.as_slice()))?;
3017 watch.stage(QualityStage::CheckingSharedNulls, false, false)?;
3018 let shared_nulls = profile_shared_nulls(&profile_df.lazy(), &columns, polars_streaming)?;
3019 let per_value = kept.per_value.as_ref().map(|per_value| per_value.kept);
3020 watch.stage(QualityStage::Assembling, false, false)?;
3021
3022 let results = DataQualityResults {
3023 total_rows,
3024 evaluated_rows,
3025 precision,
3026 sample_seed: plan.sample_seed,
3027 columns,
3028 observations,
3029 segments,
3030 temporal,
3031 identity: Some(identity),
3032 category_variants,
3033 shared_nulls,
3034 source_files: source.map(|source| source.file_names.len()),
3035 per_value,
3036 footers_read: source.map(|source| source.footers_read),
3037 reads: Some(watch.observed()),
3038 examples,
3039 unsampled_segments,
3040 intent,
3041 source: None,
3042 };
3043 Ok(results)
3044}
3045
3046fn read_quality_sample(
3049 lf: &LazyFrame,
3050 total_rows: Option<usize>,
3051 plan: &DataQualityPlan,
3052 polars_streaming: bool,
3053 watch: &QualityWatch,
3054) -> Result<QualitySample> {
3055 let sample = crate::sampling::Sample {
3056 scope: QualityScope::CurrentView,
3057 method: plan.method.clone(),
3058 rows: plan.dataset_rows,
3059 seed: plan.sample_seed,
3060 };
3061 let count = if sampler_counts_segments(plan) {
3062 None
3063 } else {
3064 segment_count_key(plan)
3065 };
3066 let sampled = crate::sampling::acquire(
3067 lf,
3068 &sample,
3069 total_rows,
3070 polars_streaming,
3071 Some(watch.read()),
3072 count.as_ref(),
3073 )?;
3074 let precision = if sampled.rows.sample_size.is_some() {
3075 QualityPrecision::Sampled
3076 } else {
3077 QualityPrecision::Exact
3078 };
3079 let mut kept = QualitySample {
3080 df: sampled.rows.df,
3081 positions: sampled.positions,
3082 precision,
3083 total_rows: Some(sampled.rows.total_rows),
3084 per_value: sampled.rows.per_value,
3085 counted: Vec::new(),
3086 too_many: Vec::new(),
3087 };
3088 match sampled.counted {
3089 Some(crate::sampling::Counted::Totals(totals)) => {
3090 kept.counted.push((segment_key(plan), totals));
3091 }
3092 Some(crate::sampling::Counted::TooMany) => kept.too_many.push(segment_key(plan)),
3093 None => {}
3094 }
3095 Ok(kept)
3096}
3097
3098fn sampled_segment_totals(
3107 lf: &LazyFrame,
3108 plan: &DataQualityPlan,
3109 kept: &mut QualitySample,
3110 polars_streaming: bool,
3111 watch: &QualityWatch,
3112) -> Result<BTreeMap<String, usize>> {
3113 if !segments_need_count(plan) {
3114 return Ok(BTreeMap::new());
3115 }
3116 let labeled = |counts: &SegmentCounts| {
3117 counts
3118 .iter()
3119 .map(|(raw, rows)| (segment_label(&plan.grain, raw.as_deref()), *rows))
3120 .collect::<BTreeMap<_, _>>()
3121 };
3122 if per_value_counts(plan, kept.per_value.as_ref())
3123 && let Some(per_value) = &kept.per_value
3124 {
3125 return Ok(labeled(&per_value.totals));
3126 }
3127 let key = segment_key(plan);
3128 if kept.too_many.contains(&key) {
3129 return Err(too_many_segments(plan));
3130 }
3131 if let Some((_, counts)) = kept.counted.iter().find(|(counted, _)| *counted == key) {
3132 return Ok(labeled(counts));
3133 }
3134 if let Some((_, finer)) = kept.finer_count(&key)
3135 && let QualityGrain::TimeWindows { every, .. } = &plan.grain
3136 {
3137 let counts = roll_up_windows(finer, every)?;
3138 let totals = labeled(&counts);
3139 kept.counted.push((key, counts));
3140 return Ok(totals);
3141 }
3142 watch.stage(QualityStage::CountingSegments, true, polars_streaming)?;
3143 let counts = counted_segment_totals(&watch.watched(lf), plan, polars_streaming)
3144 .map_err(|error| watch.failed(error))?;
3145 if counts.len() > crate::sampling::MAX_COUNTED_KEYS {
3146 kept.too_many.push(key);
3147 return Err(too_many_segments(plan));
3148 }
3149 let totals = labeled(&counts);
3150 kept.counted.push((key, counts));
3151 Ok(totals)
3152}
3153
3154fn too_many_segments(plan: &DataQualityPlan) -> Report {
3155 Report::msg(format!(
3156 "More than {} segments {}; choose a coarser grain",
3157 crate::numfmt::group_chrome(crate::sampling::MAX_COUNTED_KEYS),
3158 plan.grain.label()
3159 ))
3160}
3161
3162fn roll_up_windows(finer: &SegmentCounts, every: &str) -> Result<SegmentCounts> {
3166 let mut rolled = SegmentCounts::new();
3167 let mut starts = Vec::with_capacity(finer.len());
3168 let mut rows = Vec::with_capacity(finer.len());
3169 for (raw, count) in finer {
3170 match raw {
3171 Some(raw) => {
3172 let start = chrono::NaiveDateTime::parse_from_str(raw, "%Y-%m-%d %H:%M:%S%.f")
3173 .map_err(|_| Report::msg(format!("Window start {raw:?} is not a time")))?;
3174 starts.push(start.and_utc().timestamp_micros());
3175 rows.push(*count as u64);
3176 }
3177 None => *rolled.entry(None).or_default() += count,
3179 }
3180 }
3181 let finer = DataFrame::new(
3182 starts.len(),
3183 vec![
3184 Column::new("start".into(), starts)
3185 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))?,
3186 Column::new("rows".into(), rows),
3187 ],
3188 )?;
3189 let coarse = finer
3190 .lazy()
3191 .select([time_window_start(col("start"), every), col("rows")])
3192 .collect()?;
3193 let (starts, rows) = (coarse.column("start")?, coarse.column("rows")?.u64()?);
3194 for (row, count) in rows.into_no_null_iter().enumerate() {
3195 let start = starts.get(row)?;
3196 let key = (!start.is_null()).then(|| crate::exact::str_value(&start).into_owned());
3197 *rolled.entry(key).or_default() += count as usize;
3198 }
3199 Ok(rolled)
3200}
3201
3202fn compute_full_quality(
3203 lf: &LazyFrame,
3204 total_rows: usize,
3205 plan: &DataQualityPlan,
3206 source: Option<&QualitySourceContext>,
3207 schema: &Schema,
3208 polars_streaming: bool,
3209 watch: &QualityWatch,
3210) -> Result<DataQualityResults> {
3211 let full_schema = lf.clone().collect_schema()?;
3212 let lf = &watch.watched(lf);
3215 let failed = |error: Report| watch.failed(error);
3216 watch.stage(
3217 QualityStage::ProfilingColumns,
3218 watch.scope_reads(true),
3219 polars_streaming,
3220 )?;
3221 let mut exprs = build_profile_exprs(schema);
3223 exprs.extend(interpretation_exprs(plan, &full_schema));
3224 exprs.extend(crate::quality_intent::intent_exprs(plan, schema));
3226 let aggregate = collect_lazy(lf.clone().select(exprs), polars_streaming)
3227 .map_err(|error| watch.failed(error))?;
3228 let mut columns = parse_profiles(&aggregate, schema, total_rows);
3229 add_dominance_lazy(lf, &mut columns, polars_streaming).map_err(failed)?;
3230 watch.stage(
3231 QualityStage::CheckingDuplicates,
3232 watch.scope_reads(true),
3233 polars_streaming,
3234 )?;
3235 let identity = profile_identity_lazy(
3236 lf,
3237 schema,
3238 total_rows,
3239 QualityPrecision::Exact,
3240 polars_streaming,
3241 )
3242 .map_err(failed)?;
3243 let texts = schema
3244 .iter_values()
3245 .any(|dtype| matches!(dtype, DataType::String | DataType::Categorical(..)));
3246 watch.stage(
3247 QualityStage::CheckingSpellings,
3248 watch.scope_reads(texts),
3249 polars_streaming,
3250 )?;
3251 let category_variants =
3252 profile_category_variants_lazy(lf, schema, polars_streaming).map_err(failed)?;
3253 let keyed = !plan.intent.key.is_empty();
3256 watch.stage(
3257 QualityStage::CheckingKey,
3258 watch.scope_reads(keyed),
3259 polars_streaming,
3260 )?;
3261 let repeats =
3262 crate::quality_intent::key_repeats(lf, plan, schema, polars_streaming).map_err(failed)?;
3263 let intent = crate::quality_intent::IntentResults::from_counts(
3264 plan,
3265 schema,
3266 &aggregate,
3267 repeats,
3268 total_rows,
3269 QualityPrecision::Exact,
3270 None,
3271 )?
3272 .map(Box::new);
3273 let mut observations = observations_from_profiles(&columns, QualityPrecision::Exact);
3274 observations.extend(interpretation_observations(&aggregate, plan, &full_schema));
3275 observations.extend(identity_observations(&identity, &category_variants));
3276 if let Some(intent) = &intent {
3277 observations.extend(intent.observations());
3278 }
3279 crate::quality_intent::supersede(&mut observations, plan);
3280 if let Some(source) = source {
3284 if source.conflict_scan.is_some() {
3285 watch.stage(QualityStage::ReadingConflicts, true, true)?;
3286 }
3287 observations.extend(drift_observations(
3288 source,
3289 source.conflict_scan.as_ref(),
3290 polars_streaming,
3291 watch,
3292 ));
3293 }
3294 let whole = unsegmented(plan, source);
3295 watch.stage(
3296 QualityStage::ProfilingSegments,
3297 watch.scope_reads(!whole),
3298 polars_streaming,
3299 )?;
3300 let segments = if whole {
3301 vec![whole_segment(plan, total_rows, &columns, schema.len())]
3304 } else {
3305 profile_segments_lazy(lf, total_rows, plan, source, schema, polars_streaming)
3306 .map_err(failed)?
3307 };
3308 let intervals = !resolved_intervals(plan, &full_schema).is_empty();
3309 watch.stage(
3310 QualityStage::ComputingIntervals,
3311 watch.scope_reads(intervals),
3312 polars_streaming,
3313 )?;
3314 let temporal = profile_temporal_lazy(lf, plan, source, polars_streaming).map_err(failed)?;
3315 let shared = !shared_null_groups(&columns).is_empty();
3316 watch.stage(
3317 QualityStage::CheckingSharedNulls,
3318 watch.scope_reads(shared),
3319 polars_streaming,
3320 )?;
3321 let shared_nulls = profile_shared_nulls(lf, &columns, polars_streaming).map_err(failed)?;
3322 watch.stage(QualityStage::Assembling, false, false)?;
3323 Ok(DataQualityResults {
3324 total_rows: Some(total_rows),
3325 evaluated_rows: total_rows,
3326 precision: QualityPrecision::Exact,
3327 sample_seed: plan.sample_seed,
3328 columns,
3329 observations,
3330 segments,
3331 temporal,
3332 identity: Some(identity),
3333 category_variants,
3334 shared_nulls,
3335 source_files: source.map(|source| source.file_names.len()),
3336 per_value: None,
3337 footers_read: source.map(|source| source.footers_read),
3338 reads: Some(watch.observed()),
3339 examples: Vec::new(),
3340 unsampled_segments: Vec::new(),
3341 intent,
3342 source: None,
3343 })
3344}
3345
3346fn shared_null_groups(columns: &[ColumnQualityProfile]) -> Vec<(usize, Vec<String>)> {
3349 let mut by_count = BTreeMap::<usize, Vec<String>>::new();
3350 for profile in columns.iter().filter(|profile| profile.null_count > 0) {
3351 by_count
3352 .entry(profile.null_count)
3353 .or_default()
3354 .push(profile.name.clone());
3355 }
3356 by_count
3357 .into_iter()
3358 .filter(|(_, names)| names.len() > 1)
3359 .collect()
3360}
3361
3362fn profile_shared_nulls(
3368 lf: &LazyFrame,
3369 columns: &[ColumnQualityProfile],
3370 polars_streaming: bool,
3371) -> Result<Vec<SharedNulls>> {
3372 let groups = shared_null_groups(columns);
3373 if groups.is_empty() {
3374 return Ok(Vec::new());
3375 }
3376 let exprs = groups
3377 .iter()
3378 .enumerate()
3379 .map(|(index, (_, names))| {
3380 names
3381 .iter()
3382 .map(|name| col(name.as_str()).is_null())
3383 .reduce(Expr::and)
3384 .expect("a group has two columns")
3385 .sum()
3386 .alias(format!("__quality_shared_null_{index}"))
3387 })
3388 .collect::<Vec<_>>();
3389 let counts = collect_lazy(lf.clone().select(exprs), polars_streaming).map_err(Report::from)?;
3390 Ok(groups
3391 .into_iter()
3392 .enumerate()
3393 .map(|(index, (null_rows, columns))| SharedNulls {
3394 columns,
3395 null_rows,
3396 rows_null_in_all: usize_value(&counts, &format!("__quality_shared_null_{index}")),
3397 })
3398 .collect())
3399}
3400
3401fn add_dominance_lazy(
3405 lf: &LazyFrame,
3406 profiles: &mut [ColumnQualityProfile],
3407 polars_streaming: bool,
3408) -> Result<()> {
3409 if profiles.is_empty() {
3410 return Ok(());
3411 }
3412 const COUNT: &str = "__quality_value_count";
3413 let exprs = profiles
3414 .iter()
3415 .enumerate()
3416 .map(|(index, profile)| {
3417 col(&profile.name)
3418 .drop_nulls()
3419 .value_counts(true, true, COUNT, false)
3420 .first()
3421 .alias(format!("__quality_dominant_{index}"))
3422 })
3423 .collect::<Vec<_>>();
3424 let top = collect_lazy(lf.clone().select(exprs), polars_streaming).map_err(Report::from)?;
3425 for (index, profile) in profiles.iter_mut().enumerate() {
3426 let Ok(column) = top.column(&format!("__quality_dominant_{index}")) else {
3427 continue;
3428 };
3429 let Ok(fields) = column.struct_() else {
3430 continue;
3431 };
3432 let Ok(value) = fields.field_by_name(&profile.name) else {
3433 continue;
3434 };
3435 let Ok(counts) = fields.field_by_name(COUNT) else {
3436 continue;
3437 };
3438 profile.dominant_value = value
3439 .get(0)
3440 .ok()
3441 .filter(|value| !value.is_null())
3442 .map(|value| crate::exact::str_value(&value).to_string());
3443 profile.dominant_count = counts
3444 .get(0)
3445 .ok()
3446 .and_then(|value| value.try_extract::<u64>().ok())
3447 .map(|count| count as usize);
3448 }
3449 Ok(())
3450}
3451
3452fn profile_category_variants_lazy(
3453 lf: &LazyFrame,
3454 schema: &Schema,
3455 polars_streaming: bool,
3456) -> Result<Vec<CategoryVariantGroup>> {
3457 let normalized_name = "__quality_normalized";
3458 let original_name = "__quality_original";
3459 let count_name = "__quality_variant_rows";
3460 let variant_count_name = "__quality_variant_count";
3461 let mut result = Vec::new();
3462 for (name, dtype) in schema.iter() {
3463 if !matches!(dtype, DataType::String | DataType::Categorical(..)) {
3464 continue;
3465 }
3466 let original = text_expr(col(name.as_str()), dtype);
3467 let normalized = original
3468 .clone()
3469 .str()
3470 .strip_chars(lit(LiteralValue::untyped_null()))
3471 .str()
3472 .to_lowercase();
3473 let variant_count = col(original_name)
3474 .n_unique()
3475 .over([col(normalized_name)])?
3476 .alias(variant_count_name);
3477 let query = lf
3478 .clone()
3479 .filter(original.clone().is_not_null())
3480 .select([
3481 normalized.alias(normalized_name),
3482 original.alias(original_name),
3483 ])
3484 .group_by([col(normalized_name), col(original_name)])
3485 .agg([len().alias(count_name)])
3486 .with_columns([variant_count])
3487 .filter(col(variant_count_name).gt(lit(1u32)))
3488 .limit(1_001);
3489 let groups = collect_lazy(query, polars_streaming).map_err(Report::from)?;
3490 let complete = groups.height() <= 1_000;
3491 let mut by_normalized = BTreeMap::<String, Vec<(String, usize)>>::new();
3492 for row in 0..groups.height().min(1_000) {
3493 let Some(normalized) = string_value_at(&groups, normalized_name, row) else {
3494 continue;
3495 };
3496 let Some(original) = string_value_at(&groups, original_name, row) else {
3497 continue;
3498 };
3499 let count = usize_value_at(&groups, count_name, row);
3500 by_normalized
3501 .entry(normalized)
3502 .or_default()
3503 .push((original, count));
3504 }
3505 for (normalized, variants) in by_normalized {
3506 if variants.len() < 2 {
3507 continue;
3508 }
3509 let rows_involved = variants.iter().map(|(_, count)| count).sum();
3510 result.push(CategoryVariantGroup {
3511 column: name.to_string(),
3512 normalized,
3513 variants,
3514 rows_involved,
3515 complete,
3516 });
3517 if result.len() >= 100 {
3518 return Ok(result);
3519 }
3520 }
3521 }
3522 Ok(result)
3523}
3524
3525fn profile_identity_lazy(
3526 lf: &LazyFrame,
3527 schema: &Schema,
3528 total_rows: usize,
3529 precision: QualityPrecision,
3530 polars_streaming: bool,
3531) -> Result<IdentityProfile> {
3532 let keys = schema
3533 .iter_names()
3534 .map(|name| col(name.as_str()))
3535 .collect::<Vec<_>>();
3536 let duplicate_count = "__quality_duplicate_count";
3537 let grouped = lf
3538 .clone()
3539 .group_by(keys)
3540 .agg([len().alias(duplicate_count)])
3541 .filter(col(duplicate_count).gt(lit(1u32)))
3542 .select([
3543 len().alias("duplicate_groups"),
3544 (col(duplicate_count) - lit(1u32)).sum().alias("extra_rows"),
3545 col(duplicate_count).sum().alias("rows_involved"),
3546 ]);
3547 let summary = collect_lazy(grouped, polars_streaming).map_err(Report::from)?;
3548 Ok(IdentityProfile {
3549 duplicate_groups: usize_value(&summary, "duplicate_groups"),
3550 extra_rows: usize_value(&summary, "extra_rows"),
3551 rows_involved: usize_value(&summary, "rows_involved"),
3552 evaluated_rows: total_rows,
3553 precision,
3554 examples: Vec::new(),
3555 })
3556}
3557
3558const DUPLICATE_COPIES: &str = "__datui_quality_copies";
3559
3560fn duplicate_groups(lf: LazyFrame, keys: &[PlSmallStr]) -> LazyFrame {
3564 lf.group_by_stable(keys.iter().map(|key| col(key.clone())).collect::<Vec<_>>())
3565 .agg([len().alias(DUPLICATE_COPIES)])
3566 .filter(col(DUPLICATE_COPIES).gt(lit(1u32)))
3567 .sort(
3568 [DUPLICATE_COPIES],
3569 SortMultipleOptions::default()
3570 .with_order_descending(true)
3571 .with_maintain_order(true),
3572 )
3573}
3574
3575pub fn duplicate_rows(
3581 lf: LazyFrame,
3582 keys: &[PlSmallStr],
3583 polars_streaming: bool,
3584) -> Result<DataFrame> {
3585 let groups =
3586 collect_lazy(duplicate_groups(lf, keys), polars_streaming).map_err(Report::from)?;
3587 let copies = groups
3588 .column(DUPLICATE_COPIES)?
3589 .cast(&DataType::UInt64)?
3590 .u64()?
3591 .into_no_null_iter()
3592 .collect::<Vec<_>>();
3593 let mut take = Vec::with_capacity(copies.iter().sum::<u64>() as usize);
3594 for (group, copies) in copies.into_iter().enumerate() {
3595 take.extend(std::iter::repeat_n(group as IdxSize, copies as usize));
3596 }
3597 let rows = groups.drop(DUPLICATE_COPIES)?;
3598 Ok(rows.take(&IdxCa::from_vec(PlSmallStr::EMPTY, take))?)
3599}
3600
3601fn duplicate_examples(
3603 lf: &LazyFrame,
3604 schema: &Schema,
3605 polars_streaming: bool,
3606) -> Result<Vec<DuplicateExample>> {
3607 let keys = schema.iter_names().cloned().collect::<Vec<_>>();
3608 let groups = collect_lazy(
3609 duplicate_groups(lf.clone(), &keys).limit(MAX_FINDING_EXAMPLES as IdxSize),
3610 polars_streaming,
3611 )
3612 .map_err(Report::from)?;
3613 Ok((0..groups.height())
3614 .map(|row| DuplicateExample {
3615 copies: usize_value_at(&groups, DUPLICATE_COPIES, row),
3616 values: keys
3617 .iter()
3618 .map(|key| {
3619 groups
3620 .column(key)
3621 .and_then(|column| column.get(row))
3622 .map(|value| example_text(&value))
3623 .unwrap_or_else(|_| "null".to_string())
3624 })
3625 .collect(),
3626 })
3627 .collect())
3628}
3629
3630fn example_text(value: &AnyValue<'_>) -> String {
3632 match value {
3633 AnyValue::Null => "null".to_string(),
3634 AnyValue::String(text) => crate::quality_report::quoted(text, 24),
3635 AnyValue::StringOwned(text) => crate::quality_report::quoted(text, 24),
3636 other => {
3637 let text = crate::exact::str_value(other).to_string();
3638 if crate::glyphs::display_width(&text) > 24 {
3639 format!(
3640 "{}{}",
3641 crate::glyphs::take_columns(&text, 23),
3642 crate::glyphs::get().ellipsis
3643 )
3644 } else {
3645 text
3646 }
3647 }
3648 }
3649}
3650
3651fn finding_examples(
3654 lf: &LazyFrame,
3655 columns: &[ColumnQualityProfile],
3656 observations: &[QualityObservation],
3657 polars_streaming: bool,
3658) -> Result<Vec<FindingExamples>> {
3659 let mut examples = Vec::new();
3660 for observation in observations {
3661 let profile = columns
3662 .iter()
3663 .find(|profile| profile.name == observation.column);
3664 let failed = match observation.kind {
3665 ObservationKind::ParseableText => profile.and_then(unparsed_text),
3666 ObservationKind::UnparsedTime => observation
3667 .time_format
3668 .as_ref()
3669 .map(TimeInterpretation::unparsed),
3670 _ => None,
3671 };
3672 let (Some(failed), Some(profile)) = (failed, profile) else {
3673 continue;
3674 };
3675 let values = text_expr(col(observation.column.as_str()), &profile.dtype)
3678 .filter(failed)
3679 .head(Some(256))
3680 .alias("values");
3681 let found =
3682 collect_lazy(lf.clone().select([values]), polars_streaming).map_err(Report::from)?;
3683 let mut values = Vec::new();
3684 for value in (0..found.height()).filter_map(|row| string_value_at(&found, "values", row)) {
3685 let value = crate::quality_report::quoted(&value, 24);
3686 if !values.contains(&value) {
3687 values.push(value);
3688 }
3689 if values.len() == MAX_FINDING_EXAMPLES {
3690 break;
3691 }
3692 }
3693 if !values.is_empty() {
3694 examples.push(FindingExamples {
3695 kind: observation.kind,
3696 column: observation.column.clone(),
3697 values,
3698 });
3699 }
3700 }
3701 Ok(examples)
3702}
3703
3704fn visible_schema(schema: &Schema, source: Option<&QualitySourceContext>) -> Schema {
3705 let mut visible = Schema::with_capacity(schema.len());
3706 for (name, dtype) in schema.iter() {
3707 if source.is_some_and(|context| name.as_str() == context.row_index_column) {
3708 continue;
3709 }
3710 visible.insert(name.clone(), dtype.clone());
3711 }
3712 visible
3713}
3714
3715fn attach_source_file(
3716 mut df: DataFrame,
3717 source: Option<&QualitySourceContext>,
3718) -> Result<DataFrame> {
3719 let Some(source) = source else {
3720 return Ok(df);
3721 };
3722 let rows = df.drop_in_place(&source.row_index_column)?;
3723 let rows = rows.u32()?;
3724 let names: Vec<Option<&str>> = rows
3725 .iter()
3726 .map(|row| {
3727 let row = row? as usize;
3728 let file = source
3729 .file_starts
3730 .partition_point(|start| *start <= row)
3731 .saturating_sub(1);
3732 source.file_names.get(file).map(String::as_str)
3733 })
3734 .collect();
3735 df.with_column(Column::new(QUALITY_SOURCE_FILE_COLUMN.into(), names))?;
3736 Ok(df)
3737}
3738
3739fn profile_columns(
3740 df: &DataFrame,
3741 schema: &Schema,
3742 polars_streaming: bool,
3743) -> Result<Vec<ColumnQualityProfile>> {
3744 let aggregate = collect_lazy(
3745 df.clone().lazy().select(build_profile_exprs(schema)),
3746 polars_streaming,
3747 )
3748 .map_err(Report::from)?;
3749 Ok(parse_profiles(&aggregate, schema, df.height()))
3750}
3751
3752fn identity_observations(
3753 identity: &IdentityProfile,
3754 variants: &[CategoryVariantGroup],
3755) -> Vec<QualityObservation> {
3756 let mut observations = Vec::new();
3757 if identity.duplicate_groups > 0 {
3758 observations.push(QualityObservation {
3759 kind: ObservationKind::DuplicateRows,
3760 column: "all columns".to_string(),
3761 affected_rows: identity.rows_involved,
3762 evaluated_rows: identity.evaluated_rows,
3763 fact: format!(
3764 "{} groups; {} extra rows ({})",
3765 identity.duplicate_groups,
3766 identity.extra_rows,
3767 identity.precision.label()
3768 ),
3769 normalized_category: None,
3770 files: Vec::new(),
3771 time_format: None,
3772 full_scale: None,
3773 });
3774 }
3775 observations.extend(variants.iter().map(|group| QualityObservation {
3776 kind: ObservationKind::CategoryVariants,
3777 column: group.column.clone(),
3778 affected_rows: group.rows_involved,
3779 evaluated_rows: identity.evaluated_rows,
3780 fact: format!(
3781 "{}{} variants normalize to {:?}",
3782 if group.complete { "" } else { "at least " },
3783 group.variants.len(),
3784 group.normalized
3785 ),
3786 normalized_category: Some(group.normalized.clone()),
3787 files: Vec::new(),
3788 time_format: None,
3789 full_scale: None,
3790 }));
3791 observations
3792}
3793
3794#[derive(Debug)]
3795struct SegmentRows {
3796 label: String,
3797 indices: Vec<u32>,
3798}
3799
3800fn segment_rows(
3801 df: &DataFrame,
3802 plan: &DataQualityPlan,
3803 sample_positions: Option<&[IdxSize]>,
3804) -> Result<Vec<SegmentRows>> {
3805 let all_rows = || SegmentRows {
3806 label: "current view".to_string(),
3807 indices: (0..df.height() as u32).collect(),
3808 };
3809 let groups = match &plan.grain {
3810 QualityGrain::Dataset => vec![all_rows()],
3811 QualityGrain::RowChunks(size) => {
3812 let size = (*size).max(1);
3813 let mut chunks = BTreeMap::<usize, Vec<u32>>::new();
3814 for row in 0..df.height() {
3815 let position = sample_positions
3816 .and_then(|positions| positions.get(row))
3817 .copied()
3818 .unwrap_or(row as IdxSize) as usize;
3819 chunks.entry(position / size).or_default().push(row as u32);
3820 }
3821 chunks
3822 .into_iter()
3823 .map(|(chunk, indices)| SegmentRows {
3824 label: format!(
3825 "rows {}-{}",
3826 chunk.saturating_mul(size) + 1,
3827 (chunk + 1).saturating_mul(size)
3828 ),
3829 indices,
3830 })
3831 .collect()
3832 }
3833 QualityGrain::Partition(column) => group_by_value(df, column, &format!("{column}="))?,
3834 QualityGrain::TimeWindows { column, every } => {
3835 group_by_time_window(df, plan, column, every)?
3836 }
3837 QualityGrain::File => {
3838 if df.column(QUALITY_SOURCE_FILE_COLUMN).is_ok() {
3839 group_by_value(df, QUALITY_SOURCE_FILE_COLUMN, "file ")?
3840 } else {
3841 vec![SegmentRows {
3842 label: "file mapping unavailable for this view".to_string(),
3843 indices: (0..df.height() as u32).collect(),
3844 }]
3845 }
3846 }
3847 };
3848 Ok(groups)
3849}
3850
3851fn group_by_value(df: &DataFrame, column: &str, prefix: &str) -> Result<Vec<SegmentRows>> {
3852 let values = df.column(column)?;
3853 let mut groups: BTreeMap<String, Vec<u32>> = BTreeMap::new();
3854 let mut missing = Vec::new();
3855 for row in 0..df.height() {
3856 let value = values.get(row)?;
3857 if value.is_null() {
3858 missing.push(row as u32);
3859 } else {
3860 groups
3861 .entry(format!("{prefix}{}", crate::exact::str_value(&value)))
3862 .or_default()
3863 .push(row as u32);
3864 }
3865 }
3866 let mut result: Vec<SegmentRows> = groups
3867 .into_iter()
3868 .map(|(label, indices)| SegmentRows { label, indices })
3869 .collect();
3870 if !missing.is_empty() {
3874 result.push(SegmentRows {
3875 label: format!("{prefix}∅"),
3876 indices: missing,
3877 });
3878 }
3879 Ok(result)
3880}
3881
3882fn time_window_start(value: Expr, every: &str) -> Expr {
3887 value
3888 .map(
3889 |c| {
3890 Ok(
3891 crate::exact::calendar_without_out_of_range(c.as_materialized_series())?
3892 .map_or(c, Column::from),
3893 )
3894 },
3895 |_, field| Ok(field.clone()),
3896 )
3897 .cast(DataType::Datetime(TimeUnit::Microseconds, None))
3898 .dt()
3899 .truncate(lit(every.to_string()))
3900}
3901
3902fn group_by_time_window(
3903 df: &DataFrame,
3904 plan: &DataQualityPlan,
3905 column: &str,
3906 every: &str,
3907) -> Result<Vec<SegmentRows>> {
3908 let starts = df
3909 .clone()
3910 .lazy()
3911 .select([time_window_start(plan.time_value(column), every).alias(QUALITY_WINDOW_START)])
3912 .collect()?;
3913 let starts = starts.column(QUALITY_WINDOW_START)?;
3914 let mut groups: BTreeMap<String, Vec<u32>> = BTreeMap::new();
3915 let mut missing = Vec::new();
3916 for row in 0..df.height() {
3917 let value = starts.get(row)?;
3918 if value.is_null() {
3919 missing.push(row as u32);
3920 } else {
3921 groups
3922 .entry(crate::exact::str_value(&value).into_owned())
3923 .or_default()
3924 .push(row as u32);
3925 }
3926 }
3927 let mut result: Vec<SegmentRows> = groups
3928 .into_iter()
3929 .map(|(start, indices)| SegmentRows {
3930 label: time_window_label(column, every, Some(&start)),
3931 indices,
3932 })
3933 .collect();
3934 if !missing.is_empty() {
3935 result.push(SegmentRows {
3936 label: time_window_label(column, every, None),
3937 indices: missing,
3938 });
3939 }
3940 Ok(result)
3941}
3942
3943pub(crate) fn time_window_label(column: &str, every: &str, start: Option<&str>) -> String {
3946 let Some(start) = start else {
3947 return format!("{column} ∅");
3948 };
3949 let prefix = |length: usize| start.get(..length).unwrap_or(start).to_string();
3950 match every {
3951 "1h" => prefix(16),
3952 "1d" => prefix(10),
3953 "1w" => format!("week of {}", prefix(10)),
3954 "1mo" => prefix(7),
3955 _ => format!("{start} / {every}"),
3956 }
3957}
3958
3959fn value_epoch_micros(value: AnyValue<'_>) -> Option<i64> {
3960 match value.as_borrowed() {
3962 AnyValue::Date(days) => Some(i64::from(days) * 86_400_000_000),
3963 AnyValue::Datetime(value, TimeUnit::Nanoseconds, _) => Some(value / 1_000),
3964 AnyValue::Datetime(value, TimeUnit::Microseconds, _) => Some(value),
3965 AnyValue::Datetime(value, TimeUnit::Milliseconds, _) => Some(value * 1_000),
3966 _ => None,
3967 }
3968}
3969
3970fn take_rows(df: &DataFrame, indices: &[u32]) -> PolarsResult<DataFrame> {
3971 df.take(&UInt32Chunked::new("quality_rows".into(), indices.to_vec()))
3972}
3973
3974struct SegmentSampleProvenance<'a> {
3975 positions: Option<&'a [IdxSize]>,
3976 totals: &'a BTreeMap<String, usize>,
3977}
3978
3979fn counted_segment_totals(
3987 lf: &LazyFrame,
3988 plan: &DataQualityPlan,
3989 polars_streaming: bool,
3990) -> Result<SegmentCounts> {
3991 const KEY: &str = "__quality_count_key";
3992 const ROWS: &str = "__quality_count_rows";
3993 let Some(key) = segment_count_key(plan) else {
3994 return Ok(SegmentCounts::new());
3995 };
3996 let counts = collect_lazy(
3997 lf.clone()
3998 .select([key.alias(KEY)])
3999 .group_by([col(KEY)])
4000 .agg([len().alias(ROWS)]),
4001 polars_streaming,
4002 )
4003 .map_err(Report::from)?;
4004 let keys = counts.column(KEY)?;
4005 let mut totals = BTreeMap::new();
4006 for row in 0..counts.height() {
4007 let raw = keys.get(row)?;
4008 let raw = (!raw.is_null()).then(|| crate::exact::str_value(&raw).into_owned());
4011 totals.insert(raw, usize_value_at(&counts, ROWS, row));
4012 }
4013 Ok(totals)
4014}
4015
4016fn known_segment_totals(
4017 plan: &DataQualityPlan,
4018 total_rows: Option<usize>,
4019 source: Option<&QualitySourceContext>,
4020) -> BTreeMap<String, usize> {
4021 let mut totals = BTreeMap::new();
4022 match &plan.grain {
4023 QualityGrain::File => {
4024 let Some(source) = source else {
4025 return totals;
4026 };
4027 let whole_files = matches!(plan.scope, QualityScope::SourceFiles(_))
4028 || total_rows == Some(source.dataset_rows);
4029 if whole_files {
4030 for (index, name) in source.file_names.iter().enumerate() {
4031 totals.insert(format!("file {name}"), source.file_rows(index));
4032 }
4033 }
4034 }
4035 QualityGrain::RowChunks(size) => {
4036 let (Some(total), size) = (total_rows, (*size).max(1)) else {
4037 return totals;
4038 };
4039 for chunk in 0..total.div_ceil(size) {
4040 let start = chunk * size;
4041 totals.insert(
4042 format!("rows {}-{}", start + 1, (chunk + 1).saturating_mul(size)),
4043 size.min(total - start),
4044 );
4045 }
4046 }
4047 _ => {}
4048 }
4049 totals
4050}
4051
4052fn profile_segments(
4053 df: &DataFrame,
4054 total_rows: Option<usize>,
4055 plan: &DataQualityPlan,
4056 precision: QualityPrecision,
4057 schema: &Schema,
4058 sample: SegmentSampleProvenance<'_>,
4059 polars_streaming: bool,
4060) -> Result<(Vec<SegmentQualityProfile>, Vec<UnsampledSegment>)> {
4061 let groups = segment_rows(df, plan, sample.positions)?;
4062 let mut segment_of = vec![0u32; df.height()];
4066 for (index, group) in groups.iter().enumerate() {
4067 for row in &group.indices {
4068 segment_of[*row as usize] = index as u32;
4069 }
4070 }
4071 const SEGMENT: &str = "__quality_segment_index";
4072 let mut keyed = df.clone();
4073 keyed.with_column(Column::new(SEGMENT.into(), segment_of))?;
4074 let grouped = collect_lazy(
4075 keyed
4076 .lazy()
4077 .group_by([col(SEGMENT)])
4078 .agg(build_profile_exprs(schema)),
4079 polars_streaming,
4080 )
4081 .map_err(Report::from)?;
4082 let mut by_segment = vec![None; groups.len()];
4083 for row in 0..grouped.height() {
4084 let index = usize_value_at(&grouped, SEGMENT, row);
4085 if let Some(slot) = by_segment.get_mut(index) {
4086 *slot = Some(row);
4087 }
4088 }
4089 let mut profiles = Vec::with_capacity(groups.len());
4090 for (group, row) in groups.into_iter().zip(by_segment) {
4091 let evaluated_rows = group.indices.len();
4092 let Some(row) = row else {
4093 continue;
4094 };
4095 let columns = parse_profiles_at(&grouped, schema, evaluated_rows, row);
4096 let null_cells = columns
4097 .iter()
4098 .map(|column| column.null_count)
4099 .sum::<usize>();
4100 let denominator = evaluated_rows.saturating_mul(columns.len());
4101 let known_segment_rows = sample.totals.get(&group.label);
4102 profiles.push(SegmentQualityProfile {
4103 label: group.label,
4104 total_rows: if let Some(total) = known_segment_rows {
4105 Some(*total)
4106 } else if matches!(plan.grain, QualityGrain::Dataset) {
4107 total_rows
4108 } else if precision == QualityPrecision::Exact {
4109 Some(evaluated_rows)
4110 } else {
4111 None
4112 },
4113 evaluated_rows,
4114 columns,
4115 null_cells,
4116 null_rate: rate(null_cells, denominator),
4117 compared_with: None,
4118 largest_change: None,
4119 change_size: None,
4120 });
4121 }
4122 order_segments(&mut profiles);
4123 apply_comparisons(
4124 &mut profiles,
4125 plan.comparison,
4126 plan.baseline_segment.as_deref(),
4127 precision,
4128 );
4129 let drawn = profiles
4132 .iter()
4133 .map(|profile| profile.label.as_str())
4134 .collect::<std::collections::HashSet<_>>();
4135 let mut unsampled = sample
4136 .totals
4137 .iter()
4138 .filter(|(label, rows)| **rows > 0 && !drawn.contains(label.as_str()))
4139 .map(|(label, rows)| UnsampledSegment {
4140 label: label.clone(),
4141 total_rows: *rows,
4142 })
4143 .collect::<Vec<_>>();
4144 unsampled.sort_by(|left, right| segment_cmp(&left.label, &right.label));
4145 Ok((profiles, unsampled))
4146}
4147
4148fn order_segments(segments: &mut [SegmentQualityProfile]) {
4152 segments.sort_by(|left, right| segment_cmp(&left.label, &right.label));
4153}
4154
4155pub(crate) fn segment_cmp(left: &str, right: &str) -> std::cmp::Ordering {
4157 left.ends_with('∅')
4158 .cmp(&right.ends_with('∅'))
4159 .then_with(|| natural_cmp(left, right))
4160}
4161
4162fn natural_cmp(left: &str, right: &str) -> std::cmp::Ordering {
4164 use std::cmp::Ordering;
4165 let (mut left, mut right) = (left, right);
4166 loop {
4167 let (Some(l), Some(r)) = (left.chars().next(), right.chars().next()) else {
4168 return left.len().cmp(&right.len());
4169 };
4170 if l.is_ascii_digit() && r.is_ascii_digit() {
4171 let digits = |text: &str| {
4172 text.find(|c: char| !c.is_ascii_digit())
4173 .unwrap_or(text.len())
4174 };
4175 let (l_end, r_end) = (digits(left), digits(right));
4176 let (l_num, r_num) = (
4177 left[..l_end].trim_start_matches('0'),
4178 right[..r_end].trim_start_matches('0'),
4179 );
4180 let order = l_num.len().cmp(&r_num.len()).then_with(|| l_num.cmp(r_num));
4181 if order != Ordering::Equal {
4182 return order;
4183 }
4184 left = &left[l_end..];
4185 right = &right[r_end..];
4186 } else {
4187 if l != r {
4188 return l.cmp(&r);
4189 }
4190 left = &left[l.len_utf8()..];
4191 right = &right[r.len_utf8()..];
4192 }
4193 }
4194}
4195
4196fn profile_segments_lazy(
4197 lf: &LazyFrame,
4198 total_rows: usize,
4199 plan: &DataQualityPlan,
4200 source: Option<&QualitySourceContext>,
4201 schema: &Schema,
4202 polars_streaming: bool,
4203) -> Result<Vec<SegmentQualityProfile>> {
4204 if unsegmented(plan, source) {
4205 let aggregate = collect_lazy(
4206 lf.clone().select(build_profile_exprs(schema)),
4207 polars_streaming,
4208 )
4209 .map_err(Report::from)?;
4210 let columns = parse_profiles(&aggregate, schema, total_rows);
4211 return Ok(vec![whole_segment(
4212 plan,
4213 total_rows,
4214 &columns,
4215 schema.len(),
4216 )]);
4217 }
4218
4219 let (grouped_lf, group) = grouped_frame(lf, plan, source)?;
4220 let mut aggregates = vec![len().alias("__quality_segment_rows")];
4221 aggregates.extend(build_profile_exprs(schema));
4222 let grouped = collect_lazy(
4223 grouped_lf
4224 .group_by([group.alias("__quality_segment")])
4225 .agg(aggregates),
4226 polars_streaming,
4227 )
4228 .map_err(Report::from)?;
4229 let mut segments = Vec::with_capacity(grouped.height());
4230 let mut unassigned = Vec::with_capacity(grouped.height());
4231 for row in 0..grouped.height() {
4232 let evaluated_rows = usize_value_at(&grouped, "__quality_segment_rows", row);
4233 let columns = parse_profiles_at(&grouped, schema, evaluated_rows, row);
4234 let null_cells = columns
4235 .iter()
4236 .map(|column| column.null_count)
4237 .sum::<usize>();
4238 let denominator = evaluated_rows.saturating_mul(schema.len());
4239 let raw_label = string_value_at(&grouped, "__quality_segment", row);
4240 unassigned.push(raw_label.is_none());
4241 segments.push(SegmentQualityProfile {
4242 label: segment_label(&plan.grain, raw_label.as_deref()),
4243 total_rows: Some(evaluated_rows),
4244 evaluated_rows,
4245 columns,
4246 null_cells,
4247 null_rate: rate(null_cells, denominator),
4248 compared_with: None,
4249 largest_change: None,
4250 change_size: None,
4251 });
4252 }
4253 let mut ordered = unassigned.into_iter().zip(segments).collect::<Vec<_>>();
4255 ordered.sort_by(|left, right| {
4256 left.0
4257 .cmp(&right.0)
4258 .then_with(|| natural_cmp(&left.1.label, &right.1.label))
4259 });
4260 let mut segments = ordered
4261 .into_iter()
4262 .map(|(_, segment)| segment)
4263 .collect::<Vec<_>>();
4264 if matches!(plan.grain, QualityGrain::RowChunks(_)) {
4265 for segment in &mut segments {
4266 segment.label = pretty_chunk_label(&segment.label);
4267 }
4268 }
4269 apply_comparisons(
4270 &mut segments,
4271 plan.comparison,
4272 plan.baseline_segment.as_deref(),
4273 QualityPrecision::Exact,
4274 );
4275 Ok(segments)
4276}
4277
4278fn unsegmented(plan: &DataQualityPlan, source: Option<&QualitySourceContext>) -> bool {
4281 matches!(plan.grain, QualityGrain::Dataset)
4282 || matches!(plan.grain, QualityGrain::File) && source.is_none()
4283}
4284
4285fn whole_segment(
4287 plan: &DataQualityPlan,
4288 total_rows: usize,
4289 columns: &[ColumnQualityProfile],
4290 column_count: usize,
4291) -> SegmentQualityProfile {
4292 let null_cells = columns
4293 .iter()
4294 .map(|column| column.null_count)
4295 .sum::<usize>();
4296 let denominator = total_rows.saturating_mul(column_count);
4297 SegmentQualityProfile {
4298 label: if matches!(plan.grain, QualityGrain::File) {
4299 "file mapping unavailable for this view".to_string()
4300 } else {
4301 "current view".to_string()
4302 },
4303 total_rows: Some(total_rows),
4304 evaluated_rows: total_rows,
4305 columns: columns.to_vec(),
4306 null_cells,
4307 null_rate: rate(null_cells, denominator),
4308 compared_with: None,
4309 largest_change: None,
4310 change_size: None,
4311 }
4312}
4313
4314fn grouped_frame(
4315 lf: &LazyFrame,
4316 plan: &DataQualityPlan,
4317 source: Option<&QualitySourceContext>,
4318) -> Result<(LazyFrame, Expr)> {
4319 match &plan.grain {
4320 QualityGrain::Dataset => Err(color_eyre::eyre::eyre!(
4321 "dataset grain does not need grouping"
4322 )),
4323 QualityGrain::Partition(column) => Ok((lf.clone(), col(column))),
4324 QualityGrain::RowChunks(size) => {
4325 let row = "__datui_quality_row";
4326 Ok((
4327 lf.clone().with_row_index(row, None),
4328 col(row).cast(DataType::UInt64) / lit((*size).max(1) as u64),
4329 ))
4330 }
4331 QualityGrain::TimeWindows { column, every } => Ok((
4332 lf.clone(),
4333 time_window_start(plan.time_value(column), every),
4334 )),
4335 QualityGrain::File => {
4336 let source = source
4337 .ok_or_else(|| color_eyre::eyre::eyre!("source-file mapping is unavailable"))?;
4338 let mut file = lit("unknown");
4339 for (start, name) in source.file_starts.iter().zip(source.file_names.iter()) {
4340 file = when(col(&source.row_index_column).gt_eq(lit(*start as u32)))
4341 .then(lit(name.clone()))
4342 .otherwise(file);
4343 }
4344 Ok((lf.clone(), file))
4345 }
4346 }
4347}
4348
4349fn pretty_chunk_label(label: &str) -> String {
4350 let Some(range) = label.strip_prefix("rows ") else {
4351 return label.to_string();
4352 };
4353 let Some((start, end)) = range.split_once('-') else {
4354 return label.to_string();
4355 };
4356 let start = start.trim_start_matches('0');
4357 let end = end.trim_start_matches('0');
4358 format!(
4359 "rows {}-{}",
4360 if start.is_empty() { "0" } else { start },
4361 if end.is_empty() { "0" } else { end }
4362 )
4363}
4364
4365fn segment_label(grain: &QualityGrain, raw: Option<&str>) -> String {
4366 match grain {
4367 QualityGrain::RowChunks(size) => {
4368 let raw = raw.unwrap_or("∅");
4369 raw.parse::<usize>()
4370 .map(|chunk| {
4371 let start = chunk.saturating_mul(*size) + 1;
4372 let end = start.saturating_add(*size).saturating_sub(1);
4373 format!("rows {start:012}-{end:012}")
4374 })
4375 .unwrap_or_else(|_| format!("rows {raw}"))
4376 }
4377 QualityGrain::Partition(column) => format!("{column}={}", raw.unwrap_or("∅")),
4378 QualityGrain::TimeWindows { column, every } => time_window_label(column, every, raw),
4379 QualityGrain::File => format!("file {}", raw.unwrap_or("∅")),
4380 QualityGrain::Dataset => "current view".to_string(),
4381 }
4382}
4383
4384fn apply_comparisons(
4385 segments: &mut [SegmentQualityProfile],
4386 comparison: QualityComparison,
4387 baseline_segment: Option<&str>,
4388 precision: QualityPrecision,
4389) {
4390 for segment in segments.iter_mut() {
4391 segment.compared_with = None;
4392 segment.largest_change = None;
4393 segment.change_size = None;
4394 }
4395 let baseline_index = baseline_segment
4396 .and_then(|label| segments.iter().position(|segment| segment.label == label))
4397 .or_else(|| baseline_segment.is_none().then_some(0));
4398 if comparison == QualityComparison::Baseline && baseline_index.is_none() {
4399 for segment in segments {
4400 segment.largest_change = Some("selected baseline unavailable".to_string());
4401 }
4402 return;
4403 }
4404 for index in 0..segments.len() {
4405 let compared = match comparison {
4406 QualityComparison::None => None,
4407 QualityComparison::Previous if index > 0 => Some(index - 1),
4408 QualityComparison::Baseline if Some(index) != baseline_index => baseline_index,
4409 QualityComparison::Previous | QualityComparison::Baseline => None,
4410 };
4411 if let Some(other) = compared {
4412 let change = largest_material_change(&segments[index], &segments[other], precision);
4413 segments[index].compared_with = Some(segments[other].label.clone());
4414 if let Some((what, size)) = change {
4415 segments[index].largest_change = Some(what);
4416 segments[index].change_size = Some(size);
4417 }
4418 }
4419 }
4420}
4421
4422pub(crate) const MATERIAL_CHANGE_PP: f64 = 1.0;
4425
4426const NOISE_Z: f64 = 4.0;
4430
4431pub fn beyond_noise(a: f64, n_a: usize, b: f64, n_b: usize) -> bool {
4434 if n_a == 0 || n_b == 0 {
4435 return false;
4436 }
4437 let (n_a, n_b) = (n_a as f64, n_b as f64);
4438 let pooled = (a * n_a + b * n_b) / (n_a + n_b);
4439 let error = (pooled * (1.0 - pooled) * (1.0 / n_a + 1.0 / n_b)).sqrt();
4440 error > 0.0 && (a - b).abs() / error >= NOISE_Z
4441}
4442
4443const CHANGE_MEASURES: [QualityMetric; 4] = [
4447 QualityMetric::NullRate,
4448 QualityMetric::EmptyRate,
4449 QualityMetric::WhitespaceRate,
4450 QualityMetric::NonFiniteRate,
4451];
4452
4453fn largest_material_change(
4463 segment: &SegmentQualityProfile,
4464 baseline: &SegmentQualityProfile,
4465 precision: QualityPrecision,
4466) -> Option<(String, f64)> {
4467 if let (Some(now), Some(before)) = (segment.total_rows, baseline.total_rows)
4468 && before > 0
4469 {
4470 let ratio = now as f64 / before as f64;
4471 if !(0.5..2.0).contains(&ratio) {
4472 let percent = (ratio - 1.0) * 100.0;
4473 return Some((
4474 format!("rows {} ({percent:+.0}%)", crate::numfmt::group_chrome(now)),
4475 percent.abs(),
4476 ));
4477 }
4478 }
4479 let sampled = precision != QualityPrecision::Exact;
4480 let mut largest: Option<(f64, String)> = None;
4481 let mut range: Option<String> = None;
4482 for (index, column) in segment.columns.iter().enumerate() {
4483 let Some(prior) = baseline
4488 .columns
4489 .get(index)
4490 .filter(|other| other.name == column.name)
4491 .or_else(|| {
4492 baseline
4493 .columns
4494 .iter()
4495 .find(|other| other.name == column.name)
4496 })
4497 else {
4498 continue;
4499 };
4500 for metric in CHANGE_MEASURES {
4501 let (Some(now), Some(before)) = (metric.value(column), metric.value(prior)) else {
4502 continue;
4503 };
4504 let change = (now - before) * 100.0;
4505 if change.abs() < MATERIAL_CHANGE_PP
4506 || sampled
4507 && !beyond_noise(
4508 now,
4509 metric.denominator(column),
4510 before,
4511 metric.denominator(prior),
4512 )
4513 {
4514 continue;
4515 }
4516 if largest
4517 .as_ref()
4518 .is_none_or(|(most, _)| change.abs() > most.abs())
4519 {
4520 largest = Some((change, format!("{} {}", column.name, metric.short_label())));
4521 }
4522 }
4523 if !sampled && range.is_none() && (column.min != prior.min || column.max != prior.max) {
4524 range = Some(format!(
4525 "{} range {} -> {}",
4526 column.name,
4527 range_label(prior),
4528 range_label(column)
4529 ));
4530 }
4531 }
4532 match (largest, range) {
4533 (Some((change, what)), _) => Some((format!("{what} {change:+.1} pp"), change.abs())),
4534 (None, Some(moved)) => Some((moved, 0.0)),
4535 (None, None) => None,
4536 }
4537}
4538
4539fn range_label(column: &ColumnQualityProfile) -> String {
4540 match (&column.min, &column.max) {
4541 (Some(min), Some(max)) => format!("{min}..{max}"),
4542 (Some(min), None) => format!("{min}.."),
4543 (None, Some(max)) => format!("..{max}"),
4544 (None, None) => "none".to_string(),
4545 }
4546}
4547
4548struct TimedColumn {
4551 name: String,
4552 values: String,
4553 unparsed: Option<String>,
4554}
4555
4556struct ResolvedInterval {
4559 start_role: TemporalRole,
4560 end_role: TemporalRole,
4561 start: String,
4562 end: String,
4563 grain: QualityGrain,
4564}
4565
4566fn resolved_intervals(plan: &DataQualityPlan, schema: &Schema) -> Vec<ResolvedInterval> {
4570 let usable = |role| {
4571 plan.role_column(role)
4572 .filter(|column| plan.reads_as_time(column, schema))
4573 .map(str::to_string)
4574 };
4575 plan.interval_pairs()
4576 .into_iter()
4577 .filter_map(|(start_role, end_role)| {
4578 let (start, end) = (usable(start_role)?, usable(end_role)?);
4579 let grain = plan.interval_grain(&start, &end);
4580 Some(ResolvedInterval {
4581 start_role,
4582 end_role,
4583 start,
4584 end,
4585 grain,
4586 })
4587 })
4588 .collect()
4589}
4590
4591fn interval_grains(intervals: &[ResolvedInterval]) -> Vec<QualityGrain> {
4594 let mut grains = Vec::new();
4595 for interval in intervals {
4596 if !grains.contains(&interval.grain) {
4597 grains.push(interval.grain.clone());
4598 }
4599 }
4600 grains
4601}
4602
4603pub fn interval_passes(plan: &DataQualityPlan, schema: &Schema) -> usize {
4606 interval_grains(&resolved_intervals(plan, schema)).len()
4607}
4608
4609fn interval_duration(plan: &DataQualityPlan, start: &str, end: &str) -> Expr {
4613 let as_time = |column: &str| {
4614 plan.time_value(column)
4615 .cast(DataType::Datetime(TimeUnit::Microseconds, None))
4616 };
4617 as_time(end) - as_time(start)
4618}
4619
4620fn interval_micros(plan: &DataQualityPlan, start: &str, end: &str) -> Expr {
4622 interval_duration(plan, start, end)
4623 .dt()
4624 .total_microseconds(false)
4625}
4626
4627fn profile_temporal(
4628 df: &DataFrame,
4629 plan: &DataQualityPlan,
4630 sample_positions: Option<&[IdxSize]>,
4631) -> Result<Vec<TemporalLatencyProfile>> {
4632 let resolved = resolved_intervals(plan, df.schema());
4636 if resolved.is_empty() {
4637 return Ok(Vec::new());
4638 }
4639 let mut parsed = Vec::new();
4642 let mut timed = |column: &str| {
4643 let Some(format) = plan.time_format(column) else {
4644 return TimedColumn {
4645 name: column.to_string(),
4646 values: column.to_string(),
4647 unparsed: None,
4648 };
4649 };
4650 let values = format!("__datui_quality_time::{column}");
4651 let unparsed = format!("__datui_quality_unparsed::{column}");
4652 if !parsed
4653 .iter()
4654 .any(|(name, _): &(String, Expr)| *name == values)
4655 {
4656 parsed.push((values.clone(), format.expr()));
4657 parsed.push((unparsed.clone(), format.unparsed()));
4658 }
4659 TimedColumn {
4660 name: column.to_string(),
4661 values,
4662 unparsed: Some(unparsed),
4663 }
4664 };
4665 let intervals = resolved
4666 .iter()
4667 .map(|interval| (interval, timed(&interval.start), timed(&interval.end)))
4668 .collect::<Vec<_>>();
4669 let df = if parsed.is_empty() {
4670 df.clone()
4671 } else {
4672 df.clone()
4673 .lazy()
4674 .with_columns(
4675 parsed
4676 .into_iter()
4677 .map(|(name, expr)| expr.alias(name))
4678 .collect::<Vec<_>>(),
4679 )
4680 .collect()?
4681 };
4682 let mut cut: Vec<(QualityGrain, Vec<(String, DataFrame)>)> = Vec::new();
4684 for grain in interval_grains(&resolved) {
4685 let grain_plan = DataQualityPlan {
4686 grain: grain.clone(),
4687 ..plan.clone()
4688 };
4689 let segments = segment_rows(&df, &grain_plan, sample_positions)?
4690 .into_iter()
4691 .map(|group| Ok((group.label, take_rows(&df, &group.indices)?)))
4692 .collect::<Result<Vec<_>>>()?;
4693 cut.push((grain, segments));
4694 }
4695 let mut profiles = Vec::new();
4697 for (interval, start, end) in &intervals {
4698 let Some((_, segments)) = cut.iter().find(|(grain, _)| *grain == interval.grain) else {
4699 continue;
4700 };
4701 for (label, segment) in segments {
4702 profiles.push(latency_profile(
4703 segment,
4704 label,
4705 (interval.start_role, start),
4706 (interval.end_role, end),
4707 plan.latency_threshold_seconds,
4708 )?);
4709 }
4710 }
4711 Ok(profiles)
4712}
4713
4714fn profile_temporal_lazy(
4715 lf: &LazyFrame,
4716 plan: &DataQualityPlan,
4717 source: Option<&QualitySourceContext>,
4718 polars_streaming: bool,
4719) -> Result<Vec<TemporalLatencyProfile>> {
4720 let schema = lf.clone().collect_schema()?;
4721 let resolved = resolved_intervals(plan, &schema);
4722 if resolved.is_empty() {
4723 return Ok(Vec::new());
4724 }
4725
4726 let unparsed = |column: &str| {
4727 plan.time_format(column)
4728 .map(|format| format.unparsed().sum())
4729 .unwrap_or_else(|| lit(0u32))
4730 };
4731 let mut profiles = (0..resolved.len()).map(|_| Vec::new()).collect::<Vec<_>>();
4732 for grain in interval_grains(&resolved) {
4735 let mut expressions = vec![len().alias("__quality_temporal_rows")];
4736 let members = resolved
4737 .iter()
4738 .enumerate()
4739 .filter(|(_, interval)| interval.grain == grain)
4740 .map(|(index, _)| index)
4741 .collect::<Vec<_>>();
4742 for index in &members {
4743 let interval = &resolved[*index];
4744 let prefix = format!("latency::{index}::");
4745 let micros = interval_micros(plan, &interval.start, &interval.end);
4746 let seconds = interval_duration(plan, &interval.start, &interval.end)
4748 .dt()
4749 .total_seconds(false);
4750 expressions.extend([
4751 col(interval.start.as_str())
4754 .is_null()
4755 .sum()
4756 .alias(format!("{prefix}missing_start")),
4757 col(interval.end.as_str())
4758 .is_null()
4759 .sum()
4760 .alias(format!("{prefix}missing_end")),
4761 unparsed(&interval.start).alias(format!("{prefix}unparsed_start")),
4762 unparsed(&interval.end).alias(format!("{prefix}unparsed_end")),
4763 micros
4764 .clone()
4765 .is_not_null()
4766 .sum()
4767 .alias(format!("{prefix}paired")),
4768 micros
4769 .clone()
4770 .lt(lit(0i64))
4771 .sum()
4772 .alias(format!("{prefix}negative")),
4773 micros
4774 .clone()
4775 .eq(lit(0i64))
4776 .sum()
4777 .alias(format!("{prefix}zero")),
4778 seconds
4779 .clone()
4780 .quantile(lit(0.50), QuantileMethod::Nearest)
4781 .alias(format!("{prefix}p50")),
4782 seconds
4783 .clone()
4784 .quantile(lit(0.90), QuantileMethod::Nearest)
4785 .alias(format!("{prefix}p90")),
4786 seconds
4787 .clone()
4788 .quantile(lit(0.95), QuantileMethod::Nearest)
4789 .alias(format!("{prefix}p95")),
4790 seconds
4791 .clone()
4792 .quantile(lit(0.99), QuantileMethod::Nearest)
4793 .alias(format!("{prefix}p99")),
4794 seconds.max().alias(format!("{prefix}max")),
4795 ]);
4796 if let Some(threshold) = plan.latency_threshold_seconds {
4797 expressions.push(
4798 micros
4799 .gt(lit(threshold.saturating_mul(1_000_000)))
4800 .sum()
4801 .alias(format!("{prefix}above")),
4802 );
4803 }
4804 }
4805
4806 let grain_plan = DataQualityPlan {
4807 grain: grain.clone(),
4808 ..plan.clone()
4809 };
4810 let ungrouped = matches!(grain, QualityGrain::Dataset)
4811 || matches!(grain, QualityGrain::File) && source.is_none();
4812 let aggregate = if ungrouped {
4813 collect_lazy(lf.clone().select(expressions), polars_streaming).map_err(Report::from)?
4814 } else {
4815 let (grouped_lf, group) = grouped_frame(lf, &grain_plan, source)?;
4816 collect_lazy(
4817 grouped_lf
4818 .group_by([group.alias("__quality_segment")])
4819 .agg(expressions),
4820 polars_streaming,
4821 )
4822 .map_err(Report::from)?
4823 };
4824
4825 for row in 0..aggregate.height() {
4826 let segment = if ungrouped {
4827 if matches!(grain, QualityGrain::File) {
4828 "file mapping unavailable for this view".to_string()
4829 } else {
4830 "current view".to_string()
4831 }
4832 } else {
4833 let raw = string_value_at(&aggregate, "__quality_segment", row);
4834 segment_label(&grain, raw.as_deref())
4835 };
4836 let evaluated_rows = usize_value_at(&aggregate, "__quality_temporal_rows", row);
4837 for index in &members {
4838 let interval = &resolved[*index];
4839 let prefix = format!("latency::{index}::");
4840 let count =
4841 |name: &str| usize_value_at(&aggregate, &format!("{prefix}{name}"), row);
4842 let seconds =
4843 |name: &str| optional_i64_at(&aggregate, &format!("{prefix}{name}"), row);
4844 profiles[*index].push(TemporalLatencyProfile {
4845 segment: segment.clone(),
4846 start_role: interval.start_role,
4847 end_role: interval.end_role,
4848 start_column: interval.start.clone(),
4849 end_column: interval.end.clone(),
4850 evaluated_rows,
4851 paired_rows: count("paired"),
4852 missing_start: count("missing_start"),
4853 missing_end: count("missing_end"),
4854 unparsed_start: count("unparsed_start"),
4855 unparsed_end: count("unparsed_end"),
4856 negative_count: count("negative"),
4857 zero_count: count("zero"),
4858 p50_seconds: seconds("p50"),
4859 p90_seconds: seconds("p90"),
4860 p95_seconds: seconds("p95"),
4861 p99_seconds: seconds("p99"),
4862 max_seconds: seconds("max"),
4863 threshold_seconds: plan.latency_threshold_seconds,
4864 above_threshold_count: plan.latency_threshold_seconds.map(|_| count("above")),
4865 });
4866 }
4867 }
4868 }
4869 let mut ordered = Vec::new();
4870 for (interval, mut segments) in resolved.iter().zip(profiles) {
4871 segments.sort_by(|left, right| left.segment.cmp(&right.segment));
4872 if matches!(interval.grain, QualityGrain::RowChunks(_)) {
4876 for profile in &mut segments {
4877 profile.segment = pretty_chunk_label(&profile.segment);
4878 }
4879 }
4880 ordered.extend(segments);
4881 }
4882 Ok(ordered)
4883}
4884
4885fn latency_profile(
4886 df: &DataFrame,
4887 segment: &str,
4888 (start_role, start): (TemporalRole, &TimedColumn),
4889 (end_role, end): (TemporalRole, &TimedColumn),
4890 threshold_seconds: Option<i64>,
4891) -> Result<TemporalLatencyProfile> {
4892 let starts = df.column(&start.values)?;
4893 let ends = df.column(&end.values)?;
4894 let flags = |column: &TimedColumn| {
4895 column
4896 .unparsed
4897 .as_ref()
4898 .map(|name| df.column(name))
4899 .transpose()
4900 };
4901 let (start_flags, end_flags) = (flags(start)?, flags(end)?);
4902 let unread = |flags: Option<&Column>, row: usize| -> Result<bool> {
4903 Ok(match flags {
4904 Some(flags) => flags.get(row)? == AnyValue::Boolean(true),
4905 None => false,
4906 })
4907 };
4908 let mut missing_start = 0;
4909 let mut missing_end = 0;
4910 let mut unparsed_start = 0;
4911 let mut unparsed_end = 0;
4912 let mut micros = Vec::new();
4913 for row in 0..df.height() {
4914 let start_at = value_epoch_micros(starts.get(row)?);
4915 let end_at = value_epoch_micros(ends.get(row)?);
4916 if start_at.is_none() {
4917 if unread(start_flags, row)? {
4918 unparsed_start += 1;
4919 } else {
4920 missing_start += 1;
4921 }
4922 }
4923 if end_at.is_none() {
4924 if unread(end_flags, row)? {
4925 unparsed_end += 1;
4926 } else {
4927 missing_end += 1;
4928 }
4929 }
4930 if let (Some(start_at), Some(end_at)) = (start_at, end_at) {
4931 micros.push(end_at - start_at);
4932 }
4933 }
4934 let negative_count = micros.iter().filter(|value| **value < 0).count();
4937 let zero_count = micros.iter().filter(|value| **value == 0).count();
4938 let above_threshold_count = threshold_seconds.map(|threshold| {
4939 let threshold = threshold.saturating_mul(1_000_000);
4940 micros.iter().filter(|value| **value > threshold).count()
4941 });
4942 let mut seconds = micros
4943 .iter()
4944 .map(|value| value / 1_000_000)
4945 .collect::<Vec<_>>();
4946 seconds.sort_unstable();
4947 let percentile = |percent: usize| {
4948 if seconds.is_empty() {
4949 None
4950 } else {
4951 let index = ((seconds.len() - 1) * percent + 50) / 100;
4952 seconds.get(index).copied()
4953 }
4954 };
4955 Ok(TemporalLatencyProfile {
4956 segment: segment.to_string(),
4957 start_role,
4958 end_role,
4959 start_column: start.name.clone(),
4960 end_column: end.name.clone(),
4961 evaluated_rows: df.height(),
4962 paired_rows: micros.len(),
4963 missing_start,
4964 missing_end,
4965 unparsed_start,
4966 unparsed_end,
4967 negative_count,
4968 zero_count,
4969 p50_seconds: percentile(50),
4970 p90_seconds: percentile(90),
4971 p95_seconds: percentile(95),
4972 p99_seconds: percentile(99),
4973 max_seconds: seconds.last().copied(),
4974 threshold_seconds,
4975 above_threshold_count,
4976 })
4977}
4978
4979fn interpretation_exprs(plan: &DataQualityPlan, schema: &Schema) -> Vec<Expr> {
4982 plan.time_formats
4983 .iter()
4984 .enumerate()
4985 .filter(|(_, format)| schema.get(&format.column).is_some())
4986 .flat_map(|(index, format)| {
4987 [
4988 col(format.column.as_str())
4989 .is_not_null()
4990 .sum()
4991 .alias(format!("__datui_time::{index}::values")),
4992 format
4993 .unparsed()
4994 .sum()
4995 .alias(format!("__datui_time::{index}::unparsed")),
4996 ]
4997 })
4998 .collect()
4999}
5000
5001fn interpretation_observations(
5004 counts: &DataFrame,
5005 plan: &DataQualityPlan,
5006 schema: &Schema,
5007) -> Vec<QualityObservation> {
5008 plan.time_formats
5009 .iter()
5010 .enumerate()
5011 .filter(|(_, format)| schema.get(&format.column).is_some())
5012 .filter_map(|(index, format)| {
5013 let values = optional_usize(counts, &format!("__datui_time::{index}::values"))?;
5014 let unparsed = optional_usize(counts, &format!("__datui_time::{index}::unparsed"))?;
5015 (unparsed > 0).then(|| QualityObservation {
5016 kind: ObservationKind::UnparsedTime,
5017 column: format.column.clone(),
5018 affected_rows: unparsed,
5019 evaluated_rows: values,
5020 fact: format!(
5021 "{} of {} values do not read as {}",
5022 crate::numfmt::group_chrome(unparsed),
5023 crate::numfmt::group_chrome(values),
5024 format.label()
5025 ),
5026 normalized_category: None,
5027 files: Vec::new(),
5028 time_format: Some(format.clone()),
5029 full_scale: None,
5030 })
5031 })
5032 .collect()
5033}
5034
5035fn text_expr(column: Expr, dtype: &DataType) -> Expr {
5038 if matches!(dtype, DataType::Categorical(..)) {
5039 column.cast(DataType::String)
5040 } else {
5041 column
5042 }
5043}
5044
5045fn build_profile_exprs(schema: &Schema) -> Vec<Expr> {
5046 let mut exprs = Vec::new();
5047 for (name, dtype) in schema.iter() {
5048 let column = col(name.as_str());
5049 let prefix = format!("{}::", name);
5050 exprs.push(column.clone().null_count().alias(format!("{prefix}null")));
5051 exprs.push(
5052 column
5053 .clone()
5054 .filter(column.clone().is_not_null())
5055 .n_unique()
5056 .alias(format!("{prefix}distinct")),
5057 );
5058
5059 if supports_range(dtype) {
5060 exprs.push(column.clone().min().alias(format!("{prefix}min")));
5061 exprs.push(column.clone().max().alias(format!("{prefix}max")));
5062 }
5063
5064 if matches!(dtype, DataType::String | DataType::Categorical(..)) {
5065 let text = text_expr(column.clone(), dtype);
5066 let trimmed = text
5067 .clone()
5068 .str()
5069 .strip_chars(lit(LiteralValue::untyped_null()));
5070 exprs.push(
5071 text.clone()
5072 .eq(lit(""))
5073 .sum()
5074 .alias(format!("{prefix}empty")),
5075 );
5076 exprs.push(
5077 trimmed
5078 .eq(lit(""))
5079 .and(text.clone().neq(lit("")))
5080 .sum()
5081 .alias(format!("{prefix}whitespace")),
5082 );
5083 exprs.push(
5084 text.clone()
5085 .cast(DataType::Int64)
5086 .is_not_null()
5087 .and(text.clone().is_not_null())
5088 .sum()
5089 .alias(format!("{prefix}parse_int")),
5090 );
5091 exprs.push(
5092 text.clone()
5093 .str()
5094 .starts_with(lit("0"))
5095 .and(text.clone().str().len_chars().gt(lit(1u32)))
5096 .and(text.clone().cast(DataType::Int64).is_not_null())
5097 .sum()
5098 .alias(format!("{prefix}leading_zero")),
5099 );
5100 for (reading, name) in [
5101 (TextReading::Decimal, "parse_decimal"),
5102 (TextReading::Date, "parse_date"),
5103 (TextReading::Datetime, "parse_datetime"),
5104 ] {
5105 exprs.push(
5106 parses_as(text.clone(), reading)
5107 .and(text.clone().is_not_null())
5108 .sum()
5109 .alias(format!("{prefix}{name}")),
5110 );
5111 }
5112 exprs.push(
5113 text.clone()
5114 .str()
5115 .len_chars()
5116 .min()
5117 .alias(format!("{prefix}min_length")),
5118 );
5119 exprs.push(
5120 text.str()
5121 .len_chars()
5122 .max()
5123 .alias(format!("{prefix}max_length")),
5124 );
5125 }
5126
5127 if matches!(dtype, DataType::List(_)) {
5128 exprs.push(
5129 column
5130 .clone()
5131 .list()
5132 .len()
5133 .min()
5134 .alias(format!("{prefix}min_length")),
5135 );
5136 exprs.push(
5137 column
5138 .clone()
5139 .list()
5140 .len()
5141 .max()
5142 .alias(format!("{prefix}max_length")),
5143 );
5144 }
5145
5146 if dtype.is_float() {
5147 let float = column.cast(DataType::Float64);
5148 exprs.push(float.clone().is_nan().sum().alias(format!("{prefix}nan")));
5149 exprs.push(
5150 float
5151 .clone()
5152 .eq(lit(f64::INFINITY))
5153 .sum()
5154 .alias(format!("{prefix}pos_inf")),
5155 );
5156 exprs.push(
5157 float
5158 .eq(lit(f64::NEG_INFINITY))
5159 .sum()
5160 .alias(format!("{prefix}neg_inf")),
5161 );
5162 }
5163 }
5164 exprs
5165}
5166
5167fn parses_as(text: Expr, reading: TextReading) -> Expr {
5171 let strptime = |format: &str| StrptimeOptions {
5174 format: Some(PlSmallStr::from(format)),
5175 strict: false,
5176 exact: true,
5177 cache: true,
5178 };
5179 match reading {
5180 TextReading::WholeNumber | TextReading::Decimal => {
5181 text.cast(DataType::Float64).is_not_null()
5182 }
5183 TextReading::Date => text.str().to_date(strptime("%Y-%m-%d")).is_not_null(),
5184 TextReading::Datetime => [
5185 "%Y-%m-%d %H:%M:%S%.f",
5186 "%Y-%m-%dT%H:%M:%S%.f%#z",
5187 "%Y-%m-%dT%H:%M:%S%.f",
5188 "%Y-%m-%d %H:%M:%S",
5189 "%Y-%m-%dT%H:%M:%S%#z",
5190 "%Y-%m-%dT%H:%M:%S",
5191 ]
5192 .into_iter()
5193 .map(|format| {
5194 text.clone()
5195 .str()
5196 .to_datetime(
5197 Some(TimeUnit::Microseconds),
5198 None,
5199 strptime(format),
5200 lit(PlSmallStr::from_static("raise")),
5201 )
5202 .is_not_null()
5203 })
5204 .reduce(Expr::or)
5205 .expect("at least one datetime format"),
5206 }
5207}
5208
5209pub fn unparsed_text(profile: &ColumnQualityProfile) -> Option<Expr> {
5212 let (_, reading) = text_reading(profile)?;
5213 let text = text_expr(col(profile.name.as_str()), &profile.dtype);
5214 Some(
5215 text.clone()
5216 .is_not_null()
5217 .and(parses_as(text, reading).not()),
5218 )
5219}
5220
5221fn supports_range(dtype: &DataType) -> bool {
5222 dtype.is_numeric()
5223 || dtype.is_temporal()
5224 || matches!(
5225 dtype,
5226 DataType::String | DataType::Categorical(..) | DataType::Boolean
5227 )
5228}
5229
5230fn parse_profiles(
5231 aggregate: &DataFrame,
5232 schema: &Schema,
5233 evaluated_rows: usize,
5234) -> Vec<ColumnQualityProfile> {
5235 parse_profiles_at(aggregate, schema, evaluated_rows, 0)
5236}
5237
5238fn parse_profiles_at(
5239 aggregate: &DataFrame,
5240 schema: &Schema,
5241 evaluated_rows: usize,
5242 row: usize,
5243) -> Vec<ColumnQualityProfile> {
5244 schema
5245 .iter()
5246 .map(|(name, dtype)| {
5247 let prefix = format!("{}::", name);
5248 ColumnQualityProfile {
5249 name: name.to_string(),
5250 dtype: dtype.clone(),
5251 evaluated_rows,
5252 null_count: usize_value_at(aggregate, &format!("{prefix}null"), row),
5253 empty_count: optional_usize_at(aggregate, &format!("{prefix}empty"), row),
5254 whitespace_count: optional_usize_at(aggregate, &format!("{prefix}whitespace"), row),
5255 nan_count: optional_usize_at(aggregate, &format!("{prefix}nan"), row),
5256 positive_infinity_count: optional_usize_at(
5257 aggregate,
5258 &format!("{prefix}pos_inf"),
5259 row,
5260 ),
5261 negative_infinity_count: optional_usize_at(
5262 aggregate,
5263 &format!("{prefix}neg_inf"),
5264 row,
5265 ),
5266 distinct_count: optional_usize_at(aggregate, &format!("{prefix}distinct"), row),
5267 min: string_value_at(aggregate, &format!("{prefix}min"), row),
5268 max: string_value_at(aggregate, &format!("{prefix}max"), row),
5269 integer_parse_count: optional_usize_at(
5270 aggregate,
5271 &format!("{prefix}parse_int"),
5272 row,
5273 ),
5274 decimal_parse_count: optional_usize_at(
5275 aggregate,
5276 &format!("{prefix}parse_decimal"),
5277 row,
5278 ),
5279 date_parse_count: optional_usize_at(aggregate, &format!("{prefix}parse_date"), row),
5280 datetime_parse_count: optional_usize_at(
5281 aggregate,
5282 &format!("{prefix}parse_datetime"),
5283 row,
5284 ),
5285 leading_zero_count: optional_usize_at(
5286 aggregate,
5287 &format!("{prefix}leading_zero"),
5288 row,
5289 ),
5290 dominant_value: None,
5291 dominant_count: None,
5292 min_length: optional_usize_at(aggregate, &format!("{prefix}min_length"), row),
5293 max_length: optional_usize_at(aggregate, &format!("{prefix}max_length"), row),
5294 }
5295 })
5296 .collect()
5297}
5298
5299pub const TEXT_READING_SHARE: f64 = 0.95;
5303
5304#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5306pub enum TextReading {
5307 WholeNumber,
5308 Decimal,
5309 Datetime,
5310 Date,
5311}
5312
5313impl TextReading {
5314 pub fn label(self) -> &'static str {
5315 match self {
5316 Self::WholeNumber => "whole numbers",
5317 Self::Decimal => "decimal numbers",
5318 Self::Datetime => "ISO datetimes",
5319 Self::Date => "ISO dates",
5320 }
5321 }
5322
5323 pub fn is_number(self) -> bool {
5324 matches!(self, Self::WholeNumber | Self::Decimal)
5325 }
5326}
5327
5328pub fn text_reading(profile: &ColumnQualityProfile) -> Option<(usize, TextReading)> {
5334 let non_null = profile.non_null_rows();
5335 if non_null == 0 {
5336 return None;
5337 }
5338 let enough = |count: Option<usize>| {
5339 count.filter(|parsed| *parsed as f64 >= non_null as f64 * TEXT_READING_SHARE)
5340 };
5341 if let Some(parsed) = enough(profile.decimal_parse_count) {
5342 let reading = if profile.integer_parse_count == Some(parsed) {
5343 TextReading::WholeNumber
5344 } else {
5345 TextReading::Decimal
5346 };
5347 return Some((parsed, reading));
5348 }
5349 [
5350 (profile.datetime_parse_count, TextReading::Datetime),
5351 (profile.date_parse_count, TextReading::Date),
5352 ]
5353 .into_iter()
5354 .find_map(|(count, reading)| enough(count).map(|parsed| (parsed, reading)))
5355}
5356
5357fn observations_from_profiles(
5358 columns: &[ColumnQualityProfile],
5359 precision: QualityPrecision,
5360) -> Vec<QualityObservation> {
5361 let mut observations = Vec::new();
5362 for profile in columns {
5363 if profile.null_count > 0 {
5364 observations.push(observation(
5365 ObservationKind::Nulls,
5366 profile,
5367 profile.null_count,
5368 format!("{:.2}% null", profile.null_rate() * 100.0),
5369 ));
5370 }
5371 if let Some(count) = profile.empty_count.filter(|count| *count > 0) {
5372 observations.push(observation(
5373 ObservationKind::Empty,
5374 profile,
5375 count,
5376 format!("{:.2}% empty", rate(count, profile.evaluated_rows) * 100.0),
5377 ));
5378 }
5379 if let Some(count) = profile.whitespace_count.filter(|count| *count > 0) {
5380 observations.push(observation(
5381 ObservationKind::Whitespace,
5382 profile,
5383 count,
5384 format!(
5385 "{:.2}% whitespace only",
5386 rate(count, profile.evaluated_rows) * 100.0
5387 ),
5388 ));
5389 }
5390 let non_finite = profile.nan_count.unwrap_or(0)
5391 + profile.positive_infinity_count.unwrap_or(0)
5392 + profile.negative_infinity_count.unwrap_or(0);
5393 if non_finite > 0 {
5394 observations.push(observation(
5395 ObservationKind::NonFinite,
5396 profile,
5397 non_finite,
5398 format!("{non_finite} NaN or infinite"),
5399 ));
5400 }
5401 if profile.distinct_count == Some(1) && profile.non_null_rows() > 0 {
5402 observations.push(observation(
5403 ObservationKind::Constant,
5404 profile,
5405 profile.non_null_rows(),
5406 "one non-null value".to_string(),
5407 ));
5408 }
5409 if let Some((parsed, reading)) = text_reading(profile) {
5410 observations.push(observation(
5411 ObservationKind::ParseableText,
5412 profile,
5413 parsed,
5414 format!(
5415 "{:.2}% parse as {}",
5416 rate(parsed, profile.non_null_rows()) * 100.0,
5417 reading.label()
5418 ),
5419 ));
5420 }
5421 if precision == QualityPrecision::Exact
5432 && (profile.dtype.is_integer()
5433 || matches!(profile.dtype, DataType::String | DataType::Categorical(..)))
5434 && let (Some(distinct), Some(uniqueness)) =
5435 (profile.distinct_count, profile.uniqueness_rate())
5436 && (KEY_LIKE_UNIQUENESS..1.0).contains(&uniqueness)
5437 {
5438 let extras = profile.non_null_rows().saturating_sub(distinct);
5442 if extras > 0 {
5443 let example = match (&profile.dominant_value, profile.dominant_count) {
5444 (Some(value), Some(count)) if count > 1 => {
5445 format!("; {value:?} appears {count} times")
5446 }
5447 _ => String::new(),
5448 };
5449 observations.push(observation(
5450 ObservationKind::KeyLike,
5451 profile,
5452 extras,
5453 format!(
5454 "{distinct} distinct over {} non-null rows ({:.4}%); {extras} rows beyond one per value{example}",
5455 profile.non_null_rows(),
5456 uniqueness * 100.0,
5457 ),
5458 ));
5459 }
5460 }
5461 }
5462 observations
5463}
5464
5465pub(crate) fn conflict_reads(
5472 file_group: &[u32],
5473 groups: &[crate::schema_union::DriftGroup],
5474) -> usize {
5475 let mut per_column = BTreeMap::<&str, usize>::new();
5476 for group in file_group {
5477 let Some(group) = groups.get(*group as usize) else {
5478 continue;
5479 };
5480 for column in &group.unread {
5481 *per_column.entry(column.as_str()).or_default() += 1;
5482 }
5483 }
5484 per_column
5485 .values()
5486 .map(|files| (*files).min(MAX_EVIDENCE_FILES))
5487 .sum()
5488}
5489
5490#[derive(Clone)]
5494pub struct QualityConflictScan(pub crate::widgets::datatable::FileScan);
5495
5496impl std::fmt::Debug for QualityConflictScan {
5497 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
5498 f.write_str("QualityConflictScan")
5499 }
5500}
5501
5502#[derive(Default)]
5516struct DriftTally {
5517 files: usize,
5518 rows: usize,
5519 named: Vec<QualityFileEvidence>,
5520}
5521
5522impl DriftTally {
5523 fn add(&mut self, evidence: QualityFileEvidence) {
5524 self.files += 1;
5525 self.rows += evidence.rows;
5526 self.named.push(evidence);
5527 if self.named.len() > MAX_EVIDENCE_FILES * 2 {
5528 self.prune();
5529 }
5530 }
5531
5532 fn prune(&mut self) {
5535 self.named.sort_by(|left, right| {
5536 right
5537 .rows
5538 .cmp(&left.rows)
5539 .then_with(|| left.number.cmp(&right.number))
5540 });
5541 self.named.truncate(MAX_EVIDENCE_FILES);
5542 }
5543}
5544
5545fn drift_observations(
5546 source: &QualitySourceContext,
5547 conflicts: Option<&QualityConflictScan>,
5548 polars_streaming: bool,
5549 watch: &QualityWatch,
5550) -> Vec<QualityObservation> {
5551 let mut absent = BTreeMap::<String, DriftTally>::new();
5552 let mut unread = BTreeMap::<String, DriftTally>::new();
5553 for (file, group) in source.drifting_files() {
5554 let evidence = |stored_type: Option<String>| QualityFileEvidence {
5555 number: file + 1,
5556 name: source
5557 .file_names
5558 .get(file)
5559 .cloned()
5560 .unwrap_or_else(|| format!("file {}", file + 1)),
5561 rows: source.file_rows(file),
5562 stored_type,
5563 examples: Vec::new(),
5564 };
5565 for column in &group.absent {
5566 absent
5567 .entry(column.to_string())
5568 .or_default()
5569 .add(evidence(None));
5570 }
5571 for column in &group.unread {
5572 let stored = source
5573 .stored_type(file, column)
5574 .map(|dtype| dtype.to_string());
5575 unread
5576 .entry(column.to_string())
5577 .or_default()
5578 .add(evidence(stored));
5579 }
5580 }
5581
5582 let mut observations = Vec::new();
5583 for (kind, columns) in [
5584 (ObservationKind::Absent, absent),
5585 (ObservationKind::TypeConflict, unread),
5586 ] {
5587 for (column, mut tally) in columns {
5588 tally.prune();
5589 let mut files = tally.named;
5590 if kind == ObservationKind::TypeConflict
5591 && let Some(scan) = conflicts
5592 {
5593 read_conflict_examples(scan, &column, &mut files, polars_streaming, watch);
5594 }
5595 let named = if tally.files > files.len() {
5596 format!(", largest {} named", files.len())
5597 } else {
5598 String::new()
5599 };
5600 let verb = match (kind, tally.files) {
5601 (ObservationKind::Absent, 1) => "has no such column",
5602 (ObservationKind::Absent, _) => "have no such column",
5603 (_, 1) => "holds a type the scan cannot read",
5604 (_, _) => "hold a type the scan cannot read",
5605 };
5606 let sampled = if source.footers_read < source.file_names.len() {
5609 format!(", from {} footers read", source.footers_read)
5610 } else {
5611 String::new()
5612 };
5613 observations.push(QualityObservation {
5614 kind,
5615 column,
5616 affected_rows: tally.rows,
5617 evaluated_rows: source.dataset_rows,
5618 fact: format!(
5619 "{} of {} files {verb}{sampled}{named}",
5620 tally.files,
5621 source.file_names.len()
5622 ),
5623 normalized_category: None,
5624 files,
5625 time_format: None,
5626 full_scale: None,
5627 });
5628 }
5629 }
5630 observations
5631}
5632
5633fn read_conflict_examples(
5640 scan: &QualityConflictScan,
5641 column: &str,
5642 files: &mut [QualityFileEvidence],
5643 polars_streaming: bool,
5644 watch: &QualityWatch,
5645) {
5646 let name = PlSmallStr::from(column);
5647 for file in files.iter_mut() {
5648 if watch.cancelled() {
5649 return;
5650 }
5651 let Ok(lf) = (scan.0)(
5652 std::slice::from_ref(&file.name),
5653 std::slice::from_ref(&name),
5654 ) else {
5655 continue;
5656 };
5657 let query = lf
5658 .select([crate::past_calendar::text_expr(
5659 col(column),
5660 CastOptions::NonStrict,
5661 )])
5662 .drop_nulls(None)
5663 .limit(MAX_CONFLICT_EXAMPLES as u32);
5664 let Ok(values) = collect_lazy(query, polars_streaming) else {
5665 continue;
5666 };
5667 file.examples = (0..values.height())
5668 .filter_map(|row| string_value_at(&values, column, row))
5669 .collect();
5670 }
5671}
5672
5673pub fn add_signal_observations(
5678 results: &mut DataQualityResults,
5679 audio: &crate::audio::AudioSource,
5680 watch: &QualityWatch,
5681) -> Result<()> {
5682 watch.stage(QualityStage::CheckingSignal, true, true)?;
5683 let reports = audio
5684 .signal_report(&|| watch.cancelled())?
5685 .ok_or_else(|| Report::msg(crate::sampling::CANCELLED))?;
5686 results
5687 .observations
5688 .extend(signal_observations(&reports, audio.header().sample_rate));
5689 Ok(())
5690}
5691
5692pub fn signal_observations(
5695 reports: &[crate::audio::SignalReport],
5696 sample_rate: f64,
5697) -> Vec<QualityObservation> {
5698 let mut observations = Vec::new();
5699 let samples = |n: u64| {
5700 format!(
5701 "{} {}",
5702 crate::numfmt::group_chrome(n as usize),
5703 if n == 1 { "sample" } else { "samples" }
5704 )
5705 };
5706 let runs = |n: u64| if n == 1 { "run" } else { "runs" };
5707 for report in reports {
5708 let evaluated = report.frames as usize;
5709 let push = |observations: &mut Vec<QualityObservation>,
5710 kind: ObservationKind,
5711 affected: u64,
5712 fact: String,
5713 full_scale: Option<(f64, f64)>| {
5714 observations.push(QualityObservation {
5715 kind,
5716 column: report.channel.clone(),
5717 affected_rows: affected as usize,
5718 evaluated_rows: evaluated,
5719 fact,
5720 normalized_category: None,
5721 files: Vec::new(),
5722 time_format: None,
5723 full_scale,
5724 });
5725 };
5726 if report.clip_runs > 0 {
5727 push(
5728 &mut observations,
5729 ObservationKind::Clipping,
5730 report.in_clip_runs,
5731 format!(
5732 "{} {} of {}+ samples at full scale; longest {}",
5733 crate::numfmt::group_chrome(report.clip_runs as usize),
5734 runs(report.clip_runs),
5735 report.clip_run_min,
5736 samples(report.longest_clip)
5737 ),
5738 Some(report.full_scale),
5739 );
5740 }
5741 if report.zero_runs > 0 {
5742 push(
5743 &mut observations,
5744 ObservationKind::ZeroRuns,
5745 report.in_zero_runs,
5746 format!(
5747 "{} {} of exact zeros, {}+ samples; longest {} ({})",
5748 crate::numfmt::group_chrome(report.zero_runs as usize),
5749 runs(report.zero_runs),
5750 report.zero_run_min,
5751 samples(report.longest_zeros),
5752 crate::widgets::info::clock(report.longest_zeros as f64 / sample_rate)
5753 ),
5754 None,
5755 );
5756 }
5757 let (low, high) = report.full_scale;
5758 let half_range = (high - low) / 2.0;
5759 let share = if half_range > 0.0 {
5760 report.mean.abs() / half_range
5761 } else {
5762 0.0
5763 };
5764 if share >= DC_OFFSET_SHARE {
5765 let mean = if half_range > 2.0 {
5767 format!("{:+.1}", report.mean)
5768 } else {
5769 format!("{:+.4}", report.mean)
5770 };
5771 push(
5772 &mut observations,
5773 ObservationKind::DcOffset,
5774 report.frames,
5775 format!("mean {mean} ({:.1}% of full scale)", share * 100.0),
5776 None,
5777 );
5778 }
5779 }
5780 observations
5781}
5782
5783const DC_OFFSET_SHARE: f64 = 0.01;
5786
5787fn observation(
5788 kind: ObservationKind,
5789 profile: &ColumnQualityProfile,
5790 affected_rows: usize,
5791 fact: String,
5792) -> QualityObservation {
5793 QualityObservation {
5794 kind,
5795 column: profile.name.clone(),
5796 affected_rows,
5797 evaluated_rows: profile.evaluated_rows,
5798 fact,
5799 normalized_category: None,
5800 files: Vec::new(),
5801 time_format: None,
5802 full_scale: None,
5803 }
5804}
5805
5806fn rate(numerator: usize, denominator: usize) -> f64 {
5807 if denominator == 0 {
5808 0.0
5809 } else {
5810 numerator as f64 / denominator as f64
5811 }
5812}
5813
5814fn optional_usize(df: &DataFrame, name: &str) -> Option<usize> {
5815 optional_usize_at(df, name, 0)
5816}
5817
5818fn optional_usize_at(df: &DataFrame, name: &str, row: usize) -> Option<usize> {
5819 let value = df.column(name).ok()?.get(row).ok()?;
5820 match value {
5821 AnyValue::UInt32(value) => Some(value as usize),
5822 AnyValue::UInt64(value) => Some(value as usize),
5823 AnyValue::Int32(value) => usize::try_from(value).ok(),
5824 AnyValue::Int64(value) => usize::try_from(value).ok(),
5825 _ => None,
5826 }
5827}
5828
5829fn usize_value(df: &DataFrame, name: &str) -> usize {
5830 optional_usize(df, name).unwrap_or(0)
5831}
5832
5833fn usize_value_at(df: &DataFrame, name: &str, row: usize) -> usize {
5834 optional_usize_at(df, name, row).unwrap_or(0)
5835}
5836
5837fn optional_i64_at(df: &DataFrame, name: &str, row: usize) -> Option<i64> {
5838 let value = df.column(name).ok()?.get(row).ok()?;
5839 match value {
5840 AnyValue::Int64(value) => Some(value),
5841 AnyValue::Int32(value) => Some(i64::from(value)),
5842 AnyValue::UInt64(value) => i64::try_from(value).ok(),
5843 AnyValue::UInt32(value) => Some(i64::from(value)),
5844 AnyValue::Float64(value) if value.is_finite() => Some(value.round() as i64),
5845 AnyValue::Float32(value) if value.is_finite() => Some(value.round() as i64),
5846 _ => None,
5847 }
5848}
5849
5850fn string_value_at(df: &DataFrame, name: &str, row: usize) -> Option<String> {
5851 let value = df.column(name).ok()?.get(row).ok()?;
5852 if value.is_null() {
5853 None
5854 } else {
5855 Some(crate::exact::str_value(&value).to_string())
5856 }
5857}
5858
5859#[cfg(test)]
5860mod tests {
5861 use super::*;
5862
5863 fn fixture() -> LazyFrame {
5864 df!(
5865 "id" => &[1i64, 2, 3, 4],
5866 "amount" => &[1.0f64, f64::NAN, f64::INFINITY, 4.0],
5867 "constant" => &["x", "x", "x", "x"],
5868 "text_number" => &[Some("1"), Some("2.5"), Some("bad"), None],
5869 "dirty" => &[Some(""), Some(" "), Some("ok"), None],
5870 )
5871 .unwrap()
5872 .lazy()
5873 }
5874
5875 #[test]
5881 fn a_grain_label_never_reads_day_of_day() {
5882 let grain = |column: &str| QualityGrain::TimeWindows {
5883 column: column.to_string(),
5884 every: "1d".to_string(),
5885 };
5886 assert_eq!(grain("date").label(), "by day of date");
5887 assert_eq!(grain("day").label(), "by day of the day column");
5888 }
5889
5890 #[test]
5891 fn dates_past_the_calendar_fall_in_segments_without_a_panic() {
5892 let edges = [i64::MIN + 1, 0, i64::MAX];
5893 let lf = DataFrame::new(
5894 3,
5895 vec![
5896 Column::new("id".into(), [1i64, 2, 3]),
5897 Series::new("t".into(), edges)
5898 .cast(&DataType::Datetime(TimeUnit::Milliseconds, None))
5899 .unwrap()
5900 .into_column(),
5901 Series::new("d".into(), [i32::MIN, 0, i32::MAX])
5902 .cast(&DataType::Date)
5903 .unwrap()
5904 .into_column(),
5905 ],
5906 )
5907 .unwrap()
5908 .lazy();
5909 for grain in [
5910 QualityGrain::Partition("t".into()),
5911 QualityGrain::Partition("d".into()),
5912 QualityGrain::TimeWindows {
5913 column: "t".into(),
5914 every: "1d".into(),
5915 },
5916 QualityGrain::TimeWindows {
5917 column: "d".into(),
5918 every: "1w".into(),
5919 },
5920 ] {
5921 let plan = DataQualityPlan {
5922 compute: QualityCompute::Full,
5923 grain: grain.clone(),
5924 ..DataQualityPlan::default()
5925 };
5926 let results = compute_data_quality(&lf, Some(3), &plan, None, false).unwrap();
5927 let labels: Vec<&str> = results.segments.iter().map(|s| s.label.as_str()).collect();
5928 if let QualityGrain::Partition(column) = &grain {
5929 assert!(
5930 labels.iter().any(|l| l.starts_with(&format!("{column}=-"))
5931 && l.contains(" since 1970-01-01")),
5932 "{grain:?}: {labels:?}"
5933 );
5934 } else {
5935 assert_eq!(labels.len(), 2, "{grain:?}: {labels:?}");
5936 }
5937 let schema = lf.clone().collect_schema().unwrap();
5938 for segment in &results.segments {
5939 for schema in [Some(schema.as_ref()), None] {
5941 let predicate = segment_predicate(&plan, &grain, &segment.label, schema)
5942 .unwrap()
5943 .unwrap();
5944 let rows = lf.clone().filter(predicate).collect().unwrap().height();
5945 assert_eq!(
5946 Some(rows),
5947 segment.total_rows,
5948 "{grain:?} {}",
5949 segment.label
5950 );
5951 }
5952 }
5953 }
5954 }
5955
5956 #[test]
5959 fn partition_segments_find_the_same_rows_by_type_as_by_label() {
5960 let lf = df!(
5961 "region" => &[Some("west"), Some("east"), None, Some("west")],
5962 "year" => &[2020i16, 2021, 2020, 2020],
5963 "price" => &["1.50", "2.00", "1.50", "3.25"],
5964 "share" => &[1.0 / 3.0, 0.333_333_3, 0.5, 1.0 / 3.0],
5965 )
5966 .unwrap()
5967 .lazy()
5968 .with_column(col("price").cast(DataType::Decimal(10, 2)));
5969 let schema = lf.clone().collect_schema().unwrap();
5970 for column in ["region", "year", "price", "share"] {
5971 let grain = QualityGrain::Partition(column.into());
5972 let plan = DataQualityPlan {
5973 compute: QualityCompute::Full,
5974 grain: grain.clone(),
5975 ..DataQualityPlan::default()
5976 };
5977 let results = compute_data_quality(&lf, Some(4), &plan, None, false).unwrap();
5978 for segment in &results.segments {
5979 for schema in [Some(schema.as_ref()), None] {
5980 let predicate = segment_predicate(&plan, &grain, &segment.label, schema)
5981 .unwrap()
5982 .unwrap();
5983 let rows = lf.clone().filter(predicate).collect().unwrap().height();
5984 assert_eq!(Some(rows), segment.total_rows, "{}", segment.label);
5985 }
5986 }
5987 }
5988 }
5989
5990 #[test]
5993 fn a_full_whole_scope_segment_is_the_column_profile() {
5994 let plan = DataQualityPlan {
5995 compute: QualityCompute::Full,
5996 ..DataQualityPlan::default()
5997 };
5998 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
5999 let schema = fixture().collect_schema().unwrap();
6000 let read = profile_segments_lazy(&fixture(), 4, &plan, None, &schema, false).unwrap();
6001 assert_eq!(results.segments.len(), 1);
6002 let (reused, read) = (&results.segments[0], &read[0]);
6003 assert_eq!(reused.label, read.label);
6004 assert_eq!(reused.total_rows, read.total_rows);
6005 assert_eq!(reused.evaluated_rows, read.evaluated_rows);
6006 assert_eq!(reused.null_cells, read.null_cells);
6007 assert_eq!(reused.null_rate, read.null_rate);
6008 let counts = |segment: &SegmentQualityProfile| {
6009 segment
6010 .columns
6011 .iter()
6012 .map(|column| {
6013 (
6014 column.name.clone(),
6015 column.null_count,
6016 column.distinct_count,
6017 )
6018 })
6019 .collect::<Vec<_>>()
6020 };
6021 assert_eq!(counts(reused), counts(read));
6022 }
6023
6024 #[test]
6025 fn full_profile_reports_core_counts() {
6026 let plan = DataQualityPlan {
6027 compute: QualityCompute::Full,
6028 ..DataQualityPlan::default()
6029 };
6030 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6031
6032 assert_eq!(results.precision, QualityPrecision::Exact);
6033 assert_eq!(results.evaluated_rows, 4);
6034 let dirty = results
6035 .columns
6036 .iter()
6037 .find(|profile| profile.name == "dirty")
6038 .unwrap();
6039 assert_eq!(dirty.null_count, 1);
6040 assert_eq!(dirty.distinct_count, Some(3));
6041 assert_eq!(dirty.empty_count, Some(1));
6042 assert_eq!(dirty.whitespace_count, Some(1));
6043
6044 let amount = results
6045 .columns
6046 .iter()
6047 .find(|profile| profile.name == "amount")
6048 .unwrap();
6049 assert_eq!(amount.nan_count, Some(1));
6050 assert_eq!(amount.positive_infinity_count, Some(1));
6051
6052 let constant = results
6053 .columns
6054 .iter()
6055 .find(|profile| profile.name == "constant")
6056 .unwrap();
6057 assert_eq!(constant.distinct_count, Some(1));
6058 assert!(
6059 results
6060 .observations
6061 .iter()
6062 .any(|item| item.kind == ObservationKind::Constant)
6063 );
6064 }
6065
6066 #[test]
6067 fn exact_observation_predicates_select_matching_rows() {
6068 let examples = [
6069 (ObservationKind::Nulls, "dirty", 1),
6070 (ObservationKind::Empty, "dirty", 1),
6071 (ObservationKind::Whitespace, "dirty", 1),
6072 (ObservationKind::NonFinite, "amount", 2),
6073 (ObservationKind::Constant, "constant", 4),
6074 ];
6075 for (kind, column, expected) in examples {
6076 let observation = QualityObservation {
6077 kind,
6078 column: column.to_string(),
6079 affected_rows: expected,
6080 evaluated_rows: 4,
6081 fact: String::new(),
6082 normalized_category: None,
6083 files: Vec::new(),
6084 time_format: None,
6085 full_scale: None,
6086 };
6087 let rows = fixture()
6088 .filter(observation.evidence_predicate().unwrap())
6089 .collect()
6090 .unwrap();
6091 assert_eq!(rows.height(), expected, "{}", kind.label());
6092 }
6093 let category = QualityObservation {
6094 kind: ObservationKind::CategoryVariants,
6095 column: "category".to_string(),
6096 affected_rows: 3,
6097 evaluated_rows: 3,
6098 fact: String::new(),
6099 normalized_category: Some("north".to_string()),
6100 files: Vec::new(),
6101 time_format: None,
6102 full_scale: None,
6103 };
6104 let rows = df!("category" => &["North", " north ", "NORTH"])
6105 .unwrap()
6106 .lazy()
6107 .filter(category.evidence_predicate().unwrap())
6108 .collect()
6109 .unwrap();
6110 assert_eq!(rows.height(), 3);
6111 }
6112
6113 #[test]
6117 fn duplicate_and_parse_failure_evidence_match_their_counts() {
6118 let lf = df!(
6119 "id" => &[1i64, 2, 1, 3, 2, 1, 4],
6120 "code" => &["10", "20", "10", "3x", "20", "10", "n/a"],
6121 )
6122 .unwrap()
6123 .lazy();
6124 let plan = DataQualityPlan {
6125 compute: QualityCompute::Sample,
6126 dataset_rows: 100,
6127 ..DataQualityPlan::default()
6128 };
6129 let (results, kept) =
6130 compute_data_quality_kept(&lf, Some(7), &plan, None, false, None).unwrap();
6131 assert!(kept.is_some(), "the rows the run read are kept");
6132 let identity = results.identity.as_ref().unwrap();
6133 assert_eq!((identity.duplicate_groups, identity.rows_involved), (2, 5));
6134 let rows = duplicate_rows(lf.clone(), &["id".into(), "code".into()], false).unwrap();
6135 assert_eq!(rows.height(), identity.rows_involved);
6136 let ids = rows
6137 .column("id")
6138 .unwrap()
6139 .i64()
6140 .unwrap()
6141 .into_no_null_iter()
6142 .collect::<Vec<_>>();
6143 assert_eq!(ids, [1, 1, 1, 2, 2], "copies together, most copied first");
6144 let streamed = duplicate_rows(lf.clone(), &["id".into(), "code".into()], true).unwrap();
6145 assert!(streamed.equals(&rows), "the streaming engine agrees");
6146 assert_eq!(
6147 identity.examples,
6148 [
6149 DuplicateExample {
6150 copies: 3,
6151 values: vec!["1".to_string(), "\"10\"".to_string()],
6152 },
6153 DuplicateExample {
6154 copies: 2,
6155 values: vec!["2".to_string(), "\"20\"".to_string()],
6156 },
6157 ]
6158 );
6159
6160 let codes = df!("code" => (0..40).map(|n| n.to_string()).chain(["n/a".to_string()]).collect::<Vec<_>>())
6163 .unwrap()
6164 .lazy();
6165 let (results, _) =
6166 compute_data_quality_kept(&codes, Some(41), &plan, None, false, None).unwrap();
6167 let profile = &results.columns[0];
6168 let (parsed, reading) = text_reading(profile).unwrap();
6169 assert_eq!((parsed, reading), (40, TextReading::WholeNumber));
6170 let failed = codes
6171 .filter(unparsed_text(profile).unwrap())
6172 .collect()
6173 .unwrap();
6174 assert_eq!(failed.height(), profile.non_null_rows() - parsed);
6175 assert_eq!(
6176 results.examples_of(ObservationKind::ParseableText, "code"),
6177 ["\"n/a\""]
6178 );
6179 }
6180
6181 #[test]
6182 fn sample_is_disclosed_and_bounded() {
6183 let plan = DataQualityPlan {
6184 compute: QualityCompute::Sample,
6185 dataset_rows: 2,
6186 sample_seed: 7,
6187 ..DataQualityPlan::default()
6188 };
6189 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6190 assert_eq!(results.precision, QualityPrecision::Sampled);
6191 assert_eq!(results.total_rows, Some(4));
6192 assert_eq!(results.evaluated_rows, 2);
6193 }
6194
6195 #[test]
6196 fn a_sampled_run_knows_the_total_it_was_drawn_from() {
6199 let frame = DataFrame::new(
6200 100,
6201 vec![Column::new("id".into(), (0..100).collect::<Vec<_>>())],
6202 )
6203 .unwrap()
6204 .lazy();
6205 let plan = DataQualityPlan {
6206 dataset_rows: 10,
6207 ..DataQualityPlan::default()
6208 };
6209 let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6210 assert_eq!(results.total_rows, Some(100));
6211 assert_eq!(results.evaluated_rows, 10);
6212 assert_eq!(results.precision, QualityPrecision::Sampled);
6213 assert_eq!(results.segments[0].total_rows, Some(100));
6214
6215 let short = frame.clone().limit(8);
6216 let results = compute_data_quality(&short, None, &plan, None, false).unwrap();
6217 assert_eq!(results.total_rows, Some(8));
6218 assert_eq!(results.evaluated_rows, 8);
6219 assert_eq!(results.precision, QualityPrecision::Exact);
6220
6221 let metadata = DataQualityPlan {
6222 compute: QualityCompute::Metadata,
6223 ..plan
6224 };
6225 let results = compute_data_quality(&frame, None, &metadata, None, false).unwrap();
6226 assert_eq!(results.total_rows, None);
6227 assert_eq!(results.evaluated_rows, 0);
6228 }
6229
6230 #[test]
6234 fn a_dataset_sample_spreads_across_a_sorted_table() {
6235 let rows = 40_000;
6236 let frame = df!(
6237 "id" => (0..rows as i64).collect::<Vec<_>>(),
6238 "year" => (0..rows).map(|row| 2020 + (row * 4 / rows) as i32).collect::<Vec<_>>(),
6239 )
6240 .unwrap()
6241 .lazy();
6242 let plan = DataQualityPlan {
6243 dataset_rows: 1_000,
6244 ..DataQualityPlan::default()
6245 };
6246 let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6247 assert_eq!(results.precision, QualityPrecision::Sampled);
6248 assert_eq!(results.evaluated_rows, 1_000);
6249 assert_eq!(results.total_rows, Some(rows));
6250 let year = results
6251 .columns
6252 .iter()
6253 .find(|profile| profile.name == "year")
6254 .unwrap();
6255 assert_eq!(year.distinct_count, Some(4), "every year is in the sample");
6256 assert!(
6257 !results
6258 .observations
6259 .iter()
6260 .any(|observation| observation.kind == ObservationKind::Constant)
6261 );
6262
6263 let ids = |seed| {
6265 let plan = DataQualityPlan {
6266 sample_seed: seed,
6267 ..plan.clone()
6268 };
6269 let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6270 let id = results
6271 .columns
6272 .iter()
6273 .find(|profile| profile.name == "id")
6274 .unwrap();
6275 (id.min.clone(), id.max.clone())
6276 };
6277 assert_eq!(ids(1), ids(1));
6278 assert_ne!(ids(1), ids(2));
6279 }
6280
6281 #[test]
6285 fn partitions_compare_with_the_one_before_in_value_order() {
6286 let years = (0..300)
6287 .map(|row| [9i64, 10, 11][row / 100])
6288 .collect::<Vec<_>>();
6289 let price = (0..300)
6291 .map(|row| (!(100..110).contains(&row)).then_some(row as f64))
6292 .collect::<Vec<_>>();
6293 let frame = df!("year" => years, "price" => price).unwrap().lazy();
6294 let plan = DataQualityPlan {
6295 compute: QualityCompute::Full,
6296 grain: QualityGrain::Partition("year".to_string()),
6297 comparison: QualityComparison::Previous,
6298 ..DataQualityPlan::default()
6299 };
6300 let results = compute_data_quality(&frame, Some(300), &plan, None, false).unwrap();
6301 let labels = results
6302 .segments
6303 .iter()
6304 .map(|segment| segment.label.as_str())
6305 .collect::<Vec<_>>();
6306 assert_eq!(labels, ["year=9", "year=10", "year=11"]);
6307 assert_eq!(results.segments[0].compared_with, None);
6308 assert_eq!(results.segments[1].compared_with.as_deref(), Some("year=9"));
6309 assert_eq!(
6310 results.segments[1].largest_change.as_deref(),
6311 Some("price nulls +10.0 pp")
6312 );
6313 let changes = segment_changes(&results, 1);
6314 assert_eq!(changes[0].column, "price");
6315 assert_eq!(changes[0].metric, QualityMetric::NullRate);
6316 assert_eq!(changes[0].before, Some(0.0));
6317 assert!((changes[0].change().unwrap() - 10.0).abs() < 1e-9);
6318 assert!(natural_cmp("part-2", "part-10").is_lt());
6319 assert!(natural_cmp("year=2024", "year=2025").is_lt());
6320 }
6321
6322 #[test]
6326 fn a_daily_sample_names_real_changes_and_counts_every_day() {
6327 let mut day = Vec::new();
6328 let (mut switched, mut noisy, mut twin) = (Vec::new(), Vec::new(), Vec::new());
6329 for d in 0..200i32 {
6330 let rows = if d == 150 { 20 } else { 50 };
6332 for r in 0..rows {
6333 let key = d * 50 + r;
6334 day.push(d);
6335 switched.push((d < 100).then_some(1i64));
6337 let gap = (key * 7919) % 10 < 3;
6339 noisy.push((!gap).then_some(1i64));
6340 twin.push((!gap).then_some(2i64));
6341 }
6342 }
6343 let total = day.len();
6344 let frame = df!("day" => day, "switched" => switched, "noisy" => noisy, "twin" => twin)
6345 .unwrap()
6346 .lazy()
6347 .with_column(col("day").cast(DataType::Date));
6348 let plan = DataQualityPlan {
6349 dataset_rows: 5_000,
6350 grain: QualityGrain::TimeWindows {
6351 column: "day".to_string(),
6352 every: "1d".to_string(),
6353 },
6354 comparison: QualityComparison::Previous,
6355 ..DataQualityPlan::default()
6356 };
6357 let results = compute_data_quality(&frame, Some(total), &plan, None, false).unwrap();
6358 assert_eq!(results.precision, QualityPrecision::Sampled);
6359 assert_eq!(results.segments.len(), 200);
6360 assert_eq!(
6361 results.segments[0].total_rows,
6362 Some(50),
6363 "counted, not sampled"
6364 );
6365 assert_eq!(
6366 results.segments[100].largest_change.as_deref(),
6367 Some("switched nulls +100.0 pp")
6368 );
6369 assert_eq!(
6370 results.segments[150].largest_change.as_deref(),
6371 Some("rows 20 (-60%)")
6372 );
6373 let named = results
6374 .segments
6375 .iter()
6376 .filter_map(|segment| segment.largest_change.as_deref())
6377 .collect::<Vec<_>>();
6378 assert!(
6379 named.iter().all(|change| !change.starts_with("noisy")),
6380 "a steady rate is never named: {named:?}"
6381 );
6382 let order = segment_order(&results, true);
6384 assert!(order[..3].contains(&100) && order[..3].contains(&150));
6385
6386 let view = crate::quality_trends::trend_view(&results, QualityMetric::NullRate, 20);
6387 let (rows, per_bar) = (view.lines, view.per_bar);
6388 assert_eq!(per_bar, 10);
6389 assert_eq!(rows[0].names, ["rows"]);
6390 assert_eq!(rows[0].bars[0], Some(50.0));
6391 assert_eq!(
6392 rows[1].names,
6393 ["sampled rows"],
6394 "the sample's reach beside it"
6395 );
6396 assert_eq!(rows[2].names, ["switched"], "the column that moved leads");
6397 assert_eq!(rows[2].bars[0], Some(0.0));
6398 assert_eq!(rows[2].bars[19], Some(1.0));
6399 assert!(
6400 rows.iter()
6401 .any(|row| row.names == ["noisy".to_string(), "twin".to_string()]),
6402 "columns missing together are one line"
6403 );
6404 }
6405
6406 #[test]
6409 fn every_segment_keeps_its_own_counts() {
6410 let days = 400i32;
6411 let day = (0..days * 3).map(|row| row / 3).collect::<Vec<_>>();
6412 let price = (0..days * 3)
6413 .map(|row| (!(row % 3 == 0 && (row / 3) % 2 == 0)).then_some(f64::from(row)))
6414 .collect::<Vec<_>>();
6415 let frame = df!("day" => day, "price" => price)
6416 .unwrap()
6417 .lazy()
6418 .with_column(col("day").cast(DataType::Date));
6419 let plan = DataQualityPlan {
6420 dataset_rows: 10_000,
6421 grain: QualityGrain::TimeWindows {
6422 column: "day".to_string(),
6423 every: "1d".to_string(),
6424 },
6425 ..DataQualityPlan::default()
6426 };
6427 let results =
6428 compute_data_quality(&frame, Some(days as usize * 3), &plan, None, false).unwrap();
6429 assert_eq!(results.segments.len(), days as usize);
6430 for (index, segment) in results.segments.iter().enumerate() {
6431 assert_eq!(segment.evaluated_rows, 3, "{}", segment.label);
6432 let price = segment.columns.iter().find(|c| c.name == "price").unwrap();
6433 assert_eq!(
6434 price.null_count,
6435 usize::from(index % 2 == 0),
6436 "{}",
6437 segment.label
6438 );
6439 }
6440 }
6441
6442 #[test]
6445 fn row_chunks_cut_the_shared_sample_where_its_rows_sat() {
6446 let frame = DataFrame::new(
6447 100,
6448 vec![Column::new("id".into(), (0..100i64).collect::<Vec<_>>())],
6449 )
6450 .unwrap()
6451 .lazy();
6452 let plan = DataQualityPlan {
6453 dataset_rows: 30,
6454 sample_seed: 1,
6455 grain: QualityGrain::RowChunks(10),
6456 ..DataQualityPlan::default()
6457 };
6458 let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
6459 assert_eq!(results.total_rows, Some(100));
6460 assert_eq!(results.evaluated_rows, 30);
6461 assert_eq!(results.precision, QualityPrecision::Sampled);
6462 assert!(!plan.requires_confirmation(), "a sample never asks first");
6463 assert_eq!(
6464 results
6465 .segments
6466 .iter()
6467 .map(|segment| segment.evaluated_rows)
6468 .sum::<usize>(),
6469 30
6470 );
6471 for segment in &results.segments {
6472 assert_eq!(segment.total_rows, Some(10), "{}", segment.label);
6473 let (start, end) = segment
6475 .label
6476 .trim_start_matches("rows ")
6477 .split_once('-')
6478 .map(|(a, b)| (a.parse::<i64>().unwrap(), b.parse::<i64>().unwrap()))
6479 .unwrap();
6480 let id = segment.columns.iter().find(|c| c.name == "id").unwrap();
6481 let min = id.min.as_deref().unwrap().parse::<i64>().unwrap() + 1;
6482 let max = id.max.as_deref().unwrap().parse::<i64>().unwrap() + 1;
6483 assert!(
6484 start <= min && max <= end,
6485 "{} holds {min}..{max}",
6486 segment.label
6487 );
6488 }
6489 let again = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
6490 assert_eq!(
6491 results
6492 .segments
6493 .iter()
6494 .map(|s| s.label.clone())
6495 .collect::<Vec<_>>(),
6496 again
6497 .segments
6498 .iter()
6499 .map(|s| s.label.clone())
6500 .collect::<Vec<_>>(),
6501 "seeded"
6502 );
6503 }
6504
6505 #[test]
6510 fn a_grain_change_cuts_the_kept_sample() {
6511 let frame = df!(
6512 "id" => (0..120i64).collect::<Vec<_>>(),
6513 "region" => (0..120)
6514 .map(|row| if row < 100 { "big" } else { "small" })
6515 .collect::<Vec<_>>(),
6516 "kind" => (0..120)
6517 .map(|row| if row % 2 == 0 { "x" } else { "y" })
6518 .collect::<Vec<_>>(),
6519 )
6520 .unwrap()
6521 .lazy();
6522 let poisoned = frame.clone().filter(
6524 (col("id") + lit(1_000i64))
6525 .strict_cast(DataType::UInt8)
6526 .is_not_null(),
6527 );
6528 let totals = |results: &DataQualityResults| {
6529 results
6530 .segments
6531 .iter()
6532 .map(|segment| (segment.label.clone(), segment.total_rows))
6533 .collect::<Vec<_>>()
6534 };
6535 let whole = DataQualityPlan {
6536 method: crate::sampling::SampleMethod::PerPartition {
6537 column: "region".into(),
6538 },
6539 dataset_rows: 5,
6540 ..DataQualityPlan::default()
6541 };
6542 let (_, kept) = compute_data_quality_kept(&frame, None, &whole, None, false, None).unwrap();
6543 let kept = kept.unwrap();
6544
6545 let by_region = DataQualityPlan {
6546 grain: QualityGrain::Partition("region".into()),
6547 ..whole.clone()
6548 };
6549 let (results, again) =
6550 compute_data_quality_kept(&poisoned, None, &by_region, None, false, Some(&kept))
6551 .unwrap();
6552 assert_eq!(
6553 totals(&results),
6554 [
6555 ("region=big".to_string(), Some(100)),
6556 ("region=small".to_string(), Some(20))
6557 ]
6558 );
6559 assert!(again.unwrap().counted.is_empty(), "counted by the sampler");
6560 let fresh = compute_data_quality(&frame, None, &by_region, None, false).unwrap();
6561 assert_eq!(
6562 format!("{:?}", results.segments),
6563 format!("{:?}", fresh.segments),
6564 "the same as reading afresh"
6565 );
6566
6567 let by_kind = DataQualityPlan {
6568 grain: QualityGrain::Partition("kind".into()),
6569 ..whole
6570 };
6571 let (counted, kept) =
6572 compute_data_quality_kept(&frame, None, &by_kind, None, false, Some(&kept)).unwrap();
6573 let (recut, _) =
6574 compute_data_quality_kept(&poisoned, None, &by_kind, None, false, kept.as_ref())
6575 .unwrap();
6576 assert_eq!(
6577 totals(&recut),
6578 [
6579 ("kind=x".to_string(), Some(60)),
6580 ("kind=y".to_string(), Some(60))
6581 ]
6582 );
6583 assert_eq!(totals(&recut), totals(&counted));
6584 }
6585
6586 #[test]
6590 fn segments_are_the_shared_sample_split() {
6591 let frame = df!(
6592 "id" => (0..120i32).collect::<Vec<_>>(),
6593 "region" => (0..120)
6594 .map(|row| if row < 100 { "big" } else { "small" })
6595 .collect::<Vec<_>>(),
6596 )
6597 .unwrap()
6598 .lazy();
6599 let grain = QualityGrain::Partition("region".into());
6600 let random = DataQualityPlan {
6601 dataset_rows: 24,
6602 grain: grain.clone(),
6603 ..DataQualityPlan::default()
6604 };
6605 let results = compute_data_quality(&frame, None, &random, None, false).unwrap();
6606 assert_eq!(results.total_rows, Some(120));
6607 assert_eq!(results.evaluated_rows, 24);
6608 assert_eq!(
6609 results
6610 .segments
6611 .iter()
6612 .map(|segment| segment.total_rows)
6613 .collect::<Vec<_>>(),
6614 [Some(100), Some(20)],
6615 "a partition's size is counted beside the sample, not guessed from it"
6616 );
6617
6618 let equal = DataQualityPlan {
6619 method: crate::sampling::SampleMethod::PerPartition {
6620 column: "region".into(),
6621 },
6622 dataset_rows: 5,
6623 grain,
6624 ..DataQualityPlan::default()
6625 };
6626 let results = compute_data_quality(&frame, None, &equal, None, false).unwrap();
6627 assert_eq!(results.evaluated_rows, 10);
6628 assert_eq!(
6629 results
6630 .segments
6631 .iter()
6632 .map(|segment| (segment.label.as_str(), segment.evaluated_rows))
6633 .collect::<Vec<_>>(),
6634 vec![("region=big", 5), ("region=small", 5)]
6635 );
6636 }
6637
6638 #[test]
6639 fn metadata_mode_does_not_evaluate_values() {
6640 let plan = DataQualityPlan {
6641 compute: QualityCompute::Metadata,
6642 ..DataQualityPlan::default()
6643 };
6644 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6645 assert_eq!(results.precision, QualityPrecision::Metadata);
6646 assert_eq!(results.evaluated_rows, 0);
6647 assert_eq!(results.columns.len(), 5);
6648 }
6649
6650 #[test]
6651 fn compact_summary_keeps_plan_dimensions_visible() {
6652 let mut plan = DataQualityPlan::default();
6653 plan.set_row_chunks();
6654 plan.comparison = QualityComparison::Previous;
6655 assert_eq!(
6656 plan.compact_summary(),
6657 "scope current view -> grain in chunks of 1,000,000 rows -> compute 10000 rows random -> compare previous"
6658 );
6659 }
6660
6661 #[test]
6662 fn source_projection_preserves_rows_without_binary_payloads() {
6663 let source = QualitySourceContext {
6664 file_names: vec!["one.parquet".to_string()],
6665 file_starts: vec![0],
6666 row_index_column: "__datui_quality_row".to_string(),
6667 ..QualitySourceContext::default()
6668 };
6669 let frame = df!(
6670 "value" => &[1i64, 2, 3],
6671 "blob" => &[&b"one"[..], &b"two"[..], &b"three"[..]],
6672 )
6673 .unwrap()
6674 .lazy();
6675 let prepared = prepare_source_quality_scan(frame, Some(&source)).unwrap();
6676 let collected = prepared.collect().unwrap();
6677 assert_eq!(collected.height(), 3);
6678 assert_eq!(
6679 collected
6680 .column("__datui_quality_row")
6681 .unwrap()
6682 .u32()
6683 .unwrap()
6684 .get(2),
6685 Some(2)
6686 );
6687 assert_eq!(
6688 collected.column("blob").unwrap().str().unwrap().get(0),
6689 Some(crate::widgets::datatable::binary_stub())
6690 );
6691 }
6692
6693 #[test]
6694 fn scope_commands_round_trip_and_reject_invalid_ranges() {
6695 for command in [
6696 "view",
6697 "source",
6698 "rows 2..9",
6699 "files 1,3",
6700 "partition region=west",
6701 "time event=2024-01-01..2024-02-01",
6702 ] {
6703 let scope = QualityScope::parse_command(command).unwrap();
6704 assert_eq!(
6705 QualityScope::parse_command(&scope.command()).unwrap(),
6706 scope
6707 );
6708 }
6709 assert!(QualityScope::parse_command("rows 0..10").is_err());
6710 assert!(QualityScope::parse_command("rows 10..2").is_err());
6711 assert!(QualityScope::parse_command("files 0").is_err());
6712 assert!(QualityScope::parse_command("time event=2024-03-01..2024-01-01").is_err());
6713 }
6714
6715 #[test]
6716 fn scoped_frames_select_exact_view_source_file_partition_and_time_rows() {
6717 let frame = df!(
6718 "id" => &[1i32, 2, 3, 4, 5],
6719 "region" => &["west", "east", "west", "east", "west"],
6720 "day" => &[0i32, 1, 2, 3, 4],
6721 )
6722 .unwrap()
6723 .lazy()
6724 .with_columns([col("day").cast(DataType::Date)]);
6725 let ids = |scope: QualityScope, frame: LazyFrame, source: Option<&QualitySourceContext>| {
6726 let df = apply_quality_scope(frame, &scope, source)
6727 .unwrap()
6728 .collect()
6729 .unwrap();
6730 df.column("id")
6731 .unwrap()
6732 .i32()
6733 .unwrap()
6734 .into_no_null_iter()
6735 .collect::<Vec<_>>()
6736 };
6737 assert_eq!(
6738 ids(
6739 QualityScope::ViewRows { start: 2, end: 4 },
6740 frame.clone(),
6741 None
6742 ),
6743 vec![2, 3, 4]
6744 );
6745 let source = QualitySourceContext {
6746 file_names: vec!["one".into(), "two".into(), "three".into()],
6747 file_starts: vec![0, 2, 4],
6748 row_index_column: "__row".into(),
6749 ..QualitySourceContext::default()
6750 };
6751 assert_eq!(
6752 ids(
6753 QualityScope::SourceFiles(vec![1, 3]),
6754 frame.clone().with_row_index("__row", None),
6755 Some(&source)
6756 ),
6757 vec![1, 2, 5]
6758 );
6759 assert_eq!(
6760 ids(
6761 QualityScope::SourcePartition {
6762 column: "region".into(),
6763 value: "west".into()
6764 },
6765 frame.clone(),
6766 None
6767 ),
6768 vec![1, 3, 5]
6769 );
6770 assert_eq!(
6771 ids(
6772 QualityScope::SourceTimeRange {
6773 column: "day".into(),
6774 start: "1970-01-02".into(),
6775 end: "1970-01-04".into()
6776 },
6777 frame,
6778 None
6779 ),
6780 vec![2, 3]
6781 );
6782 }
6783
6784 #[test]
6787 fn partition_values_compare_in_the_columns_type() {
6788 let frame = df!(
6789 "id" => &[1i32, 2, 3, 4],
6790 "year" => &[2019i64, 2020, 2021, 2020],
6791 "share" => &[0.5f64, 0.25, 0.5, 1.0],
6792 "price" => &["1.50", "2.00", "1.50", "3.25"],
6793 "day" => &[19723i32, 19724, 19723, -800_000],
6794 "at" => &[0i64, 3_600_000_000, 0, 7_200_000_000],
6795 )
6796 .unwrap()
6797 .lazy()
6798 .with_columns([
6799 col("price").cast(DataType::Decimal(10, 2)),
6800 col("day").cast(DataType::Date),
6801 col("at").cast(DataType::Datetime(TimeUnit::Microseconds, None)),
6802 ]);
6803 let schema = frame.clone().collect_schema().unwrap();
6804 let ids = |column: &str, value: &str| -> Vec<i32> {
6805 let predicate = partition_predicate(column, value, &schema).unwrap();
6806 let df = frame.clone().filter(predicate).collect().unwrap();
6807 df.column("id")
6808 .unwrap()
6809 .i32()
6810 .unwrap()
6811 .into_no_null_iter()
6812 .collect()
6813 };
6814 assert_eq!(ids("year", "2020"), [2, 4]);
6815 assert_eq!(ids("year", "2019, 2021"), [1, 3]);
6816 assert_eq!(ids("year", "2020..2021"), [2, 3, 4]);
6817 assert_eq!(ids("share", "0.5"), [1, 3]);
6818 assert_eq!(ids("price", "1.5"), [1, 3], "1.5 is the column's 1.50");
6819 assert_eq!(ids("day", "2024-01-01"), [1, 3]);
6820 assert_eq!(ids("day", "-800000 days since 1970-01-01"), [4]);
6822 assert_eq!(ids("at", "1970-01-01 01:00"), [2]);
6823 assert_eq!(
6824 ids("at", "1970-01-01T00:00:00, 1970-01-01 02:00"),
6825 [1, 3, 4]
6826 );
6827 let error = partition_predicate("day", "2024-13-01", &schema).unwrap_err();
6828 assert_eq!(
6829 error.to_string(),
6830 "day: \"2024-13-01\" is not a date written YYYY-MM-DD"
6831 );
6832 assert!(partition_predicate("year", "2020..soon", &schema).is_err());
6833 }
6834
6835 #[test]
6836 fn time_scope_accepts_timezone_aware_datetime_bounds() {
6837 let frame = df!("id" => &[1i32, 2, 3], "ts" => &[0i64, 1_000_000, 2_000_000])
6838 .unwrap()
6839 .lazy()
6840 .with_columns([col("ts").cast(DataType::Datetime(
6841 TimeUnit::Microseconds,
6842 Some(TimeZone::UTC),
6843 ))]);
6844 let scope =
6845 QualityScope::parse_command("time ts=1970-01-01T01:00:01+01:00..1970-01-01T00:00:02Z")
6846 .unwrap();
6847 let result = apply_quality_scope(frame, &scope, None)
6848 .unwrap()
6849 .collect()
6850 .unwrap();
6851 assert_eq!(result.column("id").unwrap().i32().unwrap().get(0), Some(2));
6852 assert_eq!(result.height(), 1);
6853 }
6854
6855 #[test]
6856 fn full_profile_accepts_list_columns() {
6857 let lists = Column::new(
6858 "items".into(),
6859 &[
6860 Series::new("".into(), &[1i32, 2]),
6861 Series::new("".into(), &[3i32]),
6862 ],
6863 );
6864 let frame = DataFrame::new(2, vec![lists]).unwrap().lazy();
6865 let plan = DataQualityPlan {
6866 compute: QualityCompute::Full,
6867 ..DataQualityPlan::default()
6868 };
6869 let result = compute_data_quality(&frame, Some(2), &plan, None, false).unwrap();
6870 assert_eq!(result.columns[0].null_count, 0);
6871 assert_eq!(result.columns[0].min_length, Some(1));
6872 assert_eq!(result.columns[0].max_length, Some(2));
6873 let sampled =
6874 compute_data_quality(&frame, Some(2), &DataQualityPlan::default(), None, false)
6875 .unwrap();
6876 assert_eq!(sampled.columns[0].min_length, Some(1));
6877 assert_eq!(sampled.columns[0].max_length, Some(2));
6878 }
6879
6880 #[test]
6881 fn row_chunks_keep_denominators_and_compare_previous() {
6882 let plan = DataQualityPlan {
6883 compute: QualityCompute::Full,
6884 grain: QualityGrain::RowChunks(2),
6885 comparison: QualityComparison::Previous,
6886 ..DataQualityPlan::default()
6887 };
6888 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6889 assert_eq!(results.segments.len(), 2);
6890 assert_eq!(results.segments[0].evaluated_rows, 2);
6891 let first_dirty = results.segments[0]
6892 .columns
6893 .iter()
6894 .find(|column| column.name == "dirty")
6895 .unwrap();
6896 let second_dirty = results.segments[1]
6897 .columns
6898 .iter()
6899 .find(|column| column.name == "dirty")
6900 .unwrap();
6901 assert_eq!(QualityMetric::EmptyRate.value(first_dirty), Some(0.5));
6902 assert_eq!(QualityMetric::EmptyRate.value(second_dirty), Some(0.0));
6903 assert_eq!(QualityMetric::NullRate.value(second_dirty), Some(0.5));
6904 assert_eq!(
6905 results.segments[1].compared_with.as_deref(),
6906 Some("rows 1-2")
6907 );
6908 assert!(results.segments[1].largest_change.is_some());
6909
6910 let mut selected = results;
6911 let mut baseline_plan = plan.clone();
6912 baseline_plan.comparison = QualityComparison::Baseline;
6913 baseline_plan.baseline_segment = Some("rows 3-4".to_string());
6914 selected.compare_segments(&baseline_plan);
6915 assert_eq!(
6916 selected.segments[0].compared_with.as_deref(),
6917 Some("rows 3-4")
6918 );
6919 assert!(selected.segments[1].compared_with.is_none());
6920 }
6921
6922 #[test]
6927 fn a_comparison_from_held_segments_matches_a_fresh_run() {
6928 let rows = 2_000usize;
6929 let region = |row: usize| match row % 10 {
6932 0..=4 => "a",
6933 5..=7 => "b",
6934 8 => "c",
6935 _ => "d",
6936 };
6937 let df = df!(
6938 "id" => (0..rows as i64).collect::<Vec<_>>(),
6939 "region" => (0..rows).map(region).collect::<Vec<_>>(),
6940 "amount" => (0..rows)
6941 .map(|row| (region(row) != "c" || row % 3 != 0).then_some(row as f64))
6942 .collect::<Vec<_>>(),
6943 "note" => (0..rows)
6944 .map(|row| if region(row) == "d" { "" } else { "ok" })
6945 .collect::<Vec<_>>(),
6946 )
6947 .unwrap()
6948 .lazy();
6949 let comparisons = [
6950 (QualityComparison::Previous, None),
6951 (QualityComparison::Baseline, None),
6952 (QualityComparison::Baseline, Some("region=c")),
6953 (QualityComparison::Baseline, Some("region=z")),
6954 ];
6955 for compute in [QualityCompute::Full, QualityCompute::Sample] {
6956 let base = DataQualityPlan {
6957 compute,
6958 dataset_rows: 1_000,
6959 sample_seed: 11,
6960 grain: QualityGrain::Partition("region".into()),
6961 ..DataQualityPlan::default()
6962 };
6963 let held = compute_data_quality(&df, Some(rows), &base, None, false).unwrap();
6964 assert_eq!(held.segments.len(), 4, "{compute:?}");
6965 for (comparison, baseline) in comparisons {
6966 let plan = DataQualityPlan {
6967 comparison,
6968 baseline_segment: baseline.map(str::to_string),
6969 ..base.clone()
6970 };
6971 let fresh = compute_data_quality(&df, Some(rows), &plan, None, false).unwrap();
6972 let mut derived = held.clone();
6973 derived.compare_segments(&plan);
6974 let compared = |results: &DataQualityResults| {
6975 results
6976 .segments
6977 .iter()
6978 .map(|segment| {
6979 (
6980 segment.label.clone(),
6981 segment.compared_with.clone(),
6982 segment.largest_change.clone(),
6983 segment.change_size.map(f64::to_bits),
6984 )
6985 })
6986 .collect::<Vec<_>>()
6987 };
6988 assert_eq!(
6989 compared(&derived),
6990 compared(&fresh),
6991 "{compute:?} {comparison:?} {baseline:?}"
6992 );
6993 assert!(
6994 fresh
6995 .segments
6996 .iter()
6997 .any(|segment| segment.largest_change.is_some()),
6998 "{compute:?} {comparison:?} {baseline:?}: something to compare"
6999 );
7000 }
7001 }
7002 }
7003
7004 #[test]
7005 fn temporal_roles_produce_latency_without_name_inference() {
7006 let event = Series::new(
7007 "happened_at".into(),
7008 [Some(0i64), Some(3_600_000_000), None, Some(10_800_000_000)],
7009 )
7010 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7011 .unwrap();
7012 let received = Series::new(
7013 "landed_at".into(),
7014 [
7015 Some(3_600_000_000i64),
7016 Some(1_800_000_000),
7017 Some(7_200_000_000),
7018 None,
7019 ],
7020 )
7021 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7022 .unwrap();
7023 let frame = DataFrame::new(4, vec![event.into(), received.into()])
7024 .unwrap()
7025 .lazy();
7026 let mut plan = DataQualityPlan {
7027 compute: QualityCompute::Full,
7028 ..DataQualityPlan::default()
7029 };
7030
7031 let without_roles = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7032 assert!(without_roles.temporal.is_empty());
7033
7034 plan.temporal_roles = vec![
7035 TemporalRoleAssignment {
7036 role: TemporalRole::Event,
7037 column: "happened_at".to_string(),
7038 timezone: None,
7039 },
7040 TemporalRoleAssignment {
7041 role: TemporalRole::Received,
7042 column: "landed_at".to_string(),
7043 timezone: None,
7044 },
7045 ];
7046 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7047 assert_eq!(results.temporal.len(), 1);
7048 let latency = &results.temporal[0];
7049 assert_eq!(latency.missing_start, 1);
7050 assert_eq!(latency.missing_end, 1);
7051 assert_eq!(latency.negative_count, 1);
7052 assert_eq!(latency.p50_seconds, Some(3_600));
7053 }
7054
7055 #[test]
7058 fn a_column_missing_from_many_files_counts_them_all_and_names_the_largest() {
7059 use crate::schema_union::DriftGroup;
7060
7061 const FILES: usize = 25;
7062 let mut file_starts = Vec::with_capacity(FILES);
7064 let mut row = 0usize;
7065 for file in 0..FILES {
7066 file_starts.push(row);
7067 row += file + 1;
7068 }
7069 let source = QualitySourceContext {
7070 file_names: (0..FILES).map(|file| format!("{file}.parquet")).collect(),
7071 file_starts,
7072 dataset_rows: row,
7073 footers_read: FILES,
7074 file_group: vec![1; FILES],
7076 drift_groups: Arc::new(vec![
7077 DriftGroup::default(),
7078 DriftGroup {
7079 absent: vec!["fee".into()],
7080 unread: Vec::new(),
7081 },
7082 ]),
7083 ..QualitySourceContext::default()
7084 };
7085
7086 let observations = drift_observations(&source, None, false, &QualityWatch::default());
7087 assert_eq!(observations.len(), 1);
7088 let absent = &observations[0];
7089 assert_eq!(absent.kind, ObservationKind::Absent);
7090 assert_eq!(
7091 (absent.affected_rows, absent.evaluated_rows),
7092 (row, row),
7093 "every file is counted, not only the named ones"
7094 );
7095 assert_eq!(absent.files.len(), MAX_EVIDENCE_FILES);
7096 assert_eq!(
7097 absent.files.first().map(|file| file.number),
7098 Some(FILES),
7099 "the largest file first"
7100 );
7101 assert!(
7102 absent
7103 .fact
7104 .starts_with("25 of 25 files have no such column, largest 20 named"),
7105 "{}",
7106 absent.fact
7107 );
7108 assert_eq!(
7109 absent.evidence_scope().map(|scope| match scope {
7110 QualityScope::SourceFiles(files) => files.len(),
7111 _ => 0,
7112 }),
7113 Some(MAX_EVIDENCE_FILES),
7114 "the drill-in opens the files it named"
7115 );
7116 }
7117
7118 #[test]
7121 fn a_nearly_unique_column_that_repeats_is_reported_with_its_repeats() {
7122 let mut ids = (0..98i64).collect::<Vec<_>>();
7123 ids.push(7);
7125 ids.push(11);
7126 let frame = df!(
7127 "id" => &ids,
7128 "region" => &(0..100).map(|row| ["north", "south"][row % 2]).collect::<Vec<_>>(),
7129 )
7130 .unwrap()
7131 .lazy();
7132 let plan = DataQualityPlan {
7133 compute: QualityCompute::Full,
7134 ..DataQualityPlan::default()
7135 };
7136 let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7137 let key_like = results
7138 .observations
7139 .iter()
7140 .filter(|observation| observation.kind == ObservationKind::KeyLike)
7141 .collect::<Vec<_>>();
7142 assert_eq!(
7143 key_like
7144 .iter()
7145 .map(|o| o.column.as_str())
7146 .collect::<Vec<_>>(),
7147 vec!["id"],
7148 "two values in a hundred rows is a category, not a key that slipped"
7149 );
7150 assert_eq!(
7151 (key_like[0].affected_rows, key_like[0].evaluated_rows),
7152 (2, 100),
7153 "rows beyond one per value: non-null rows minus distinct values"
7154 );
7155 assert_eq!(
7156 key_like[0].fact,
7157 "98 distinct over 100 non-null rows (98.0000%); 2 rows beyond one per value; \"7\" appears 2 times"
7158 );
7159 let rows = frame
7163 .clone()
7164 .filter(key_like[0].evidence_predicate().unwrap())
7165 .collect()
7166 .unwrap();
7167 assert_eq!(rows.height(), 4);
7168
7169 let sampled = compute_data_quality(
7172 &frame,
7173 Some(1_000_000),
7174 &DataQualityPlan {
7175 compute: QualityCompute::Sample,
7176 dataset_rows: 10,
7177 ..DataQualityPlan::default()
7178 },
7179 None,
7180 false,
7181 )
7182 .unwrap();
7183 assert_eq!(sampled.precision, QualityPrecision::Sampled);
7184 assert!(
7185 !sampled
7186 .observations
7187 .iter()
7188 .any(|observation| observation.kind == ObservationKind::KeyLike),
7189 "a sampled distinct share cannot say a column is nearly a key"
7190 );
7191 }
7192
7193 #[test]
7196 fn a_sample_is_read_at_its_full_size() {
7197 let rows = 80_000;
7198 let frame = df!(
7199 "id" => (0..rows as i64).collect::<Vec<_>>(),
7200 "tag" => (0..rows).map(|row| ["a", "b", "c"][row % 3]).collect::<Vec<_>>(),
7201 )
7202 .unwrap()
7203 .lazy();
7204 let plan = DataQualityPlan {
7205 dataset_rows: 60_000,
7206 ..DataQualityPlan::default()
7207 };
7208 let results = compute_data_quality(&frame, Some(rows), &plan, None, false).unwrap();
7209 assert_eq!(results.evaluated_rows, 60_000);
7210 assert_eq!(results.precision, QualityPrecision::Sampled);
7211 let tag = results
7212 .columns
7213 .iter()
7214 .find(|profile| profile.name == "tag")
7215 .unwrap();
7216 assert!(
7217 tag.dominant_value.is_some(),
7218 "the most common value is measured"
7219 );
7220 assert_eq!(
7221 results
7222 .identity
7223 .as_ref()
7224 .map(|identity| identity.evaluated_rows),
7225 Some(60_000)
7226 );
7227 }
7228
7229 #[test]
7232 fn an_empty_page_names_the_setting_that_fills_it() {
7233 let mut plan = DataQualityPlan::default();
7234 let results = DataQualityResults::empty(Some(10), &plan, &Schema::default());
7235 let setup =
7236 |page, plan: &DataQualityPlan, dates| page_setup(page, plan, Some(&results), dates);
7237 assert_eq!(
7238 setup(QualityPage::Segments, &plan, false),
7239 Some(QualitySetup::Grain)
7240 );
7241 assert_eq!(
7242 setup(QualityPage::Intervals, &plan, true),
7243 Some(QualitySetup::TimeRoles)
7244 );
7245 assert_eq!(setup(QualityPage::Intervals, &plan, false), None);
7246 assert_eq!(
7247 setup(QualityPage::Trends, &plan, true),
7248 Some(QualitySetup::Grain)
7249 );
7250 assert_eq!(setup(QualityPage::Overview, &plan, true), None);
7251 let mut paired = plan.clone();
7253 paired.temporal_roles = [TemporalRole::Created, TemporalRole::Processed]
7254 .map(|role| TemporalRoleAssignment {
7255 role,
7256 column: "at".to_string(),
7257 timezone: None,
7258 })
7259 .to_vec();
7260 assert_eq!(
7261 setup(QualityPage::Intervals, &paired, true),
7262 Some(QualitySetup::Intervals)
7263 );
7264 paired.toggle_interval((TemporalRole::Created, TemporalRole::Processed));
7267 assert_eq!(setup(QualityPage::Intervals, &paired, true), None);
7268 assert_eq!(
7269 page_setup(QualityPage::Segments, &plan, None, true),
7270 None,
7271 "nothing to set up before a run"
7272 );
7273 plan.grain = QualityGrain::RowChunks(5);
7274 assert_eq!(setup(QualityPage::Segments, &plan, false), None);
7275 }
7276
7277 #[test]
7280 fn a_partition_scope_takes_a_value_a_list_or_a_range() {
7281 let frame = df!(
7282 "year" => [Some(8i64), Some(9), Some(10), Some(11), None],
7283 "id" => [1i64, 2, 3, 4, 5],
7284 )
7285 .unwrap()
7286 .lazy();
7287 let ids = |value: &str| {
7288 let scope = QualityScope::parse_command(&format!("partition year={value}")).unwrap();
7289 let rows = apply_quality_scope(frame.clone(), &scope, None)
7290 .unwrap()
7291 .collect()
7292 .unwrap();
7293 rows.column("id")
7294 .unwrap()
7295 .i64()
7296 .unwrap()
7297 .into_no_null_iter()
7298 .collect::<Vec<_>>()
7299 };
7300 assert_eq!(ids("9"), vec![2]);
7301 assert_eq!(ids("8,11"), vec![1, 4]);
7302 assert_eq!(ids("9..10"), vec![2, 3]);
7303 assert_eq!(ids("∅"), vec![5]);
7304 }
7305
7306 #[test]
7309 fn a_nearly_unique_float_is_not_a_key() {
7310 let mut prices = (0..98).map(|row| row as f64 + 0.5).collect::<Vec<_>>();
7311 prices.push(7.5);
7312 prices.push(11.5);
7313 let frame = df!("price" => &prices).unwrap().lazy();
7314 let plan = DataQualityPlan {
7315 compute: QualityCompute::Full,
7316 ..DataQualityPlan::default()
7317 };
7318 let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7319 assert!(
7320 !results
7321 .observations
7322 .iter()
7323 .any(|observation| observation.kind == ObservationKind::KeyLike)
7324 );
7325 }
7326
7327 #[test]
7330 fn text_is_read_as_numbers_only_when_nearly_all_of_it_parses() {
7331 let mut names = (0..97).map(|row| format!("name {row}")).collect::<Vec<_>>();
7332 names.extend(["1", "2", "3"].map(String::from));
7333 let codes = (0..100)
7334 .map(|row| format!("{:04}", row * 37))
7335 .collect::<Vec<_>>();
7336 let amounts = (0..100).map(|row| format!("{row}.25")).collect::<Vec<_>>();
7337 let frame = df!("name" => &names, "code" => &codes, "amount" => &amounts)
7338 .unwrap()
7339 .lazy();
7340 let plan = DataQualityPlan {
7341 compute: QualityCompute::Full,
7342 ..DataQualityPlan::default()
7343 };
7344 let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7345 let readings = results
7346 .observations
7347 .iter()
7348 .filter(|observation| observation.kind == ObservationKind::ParseableText)
7349 .map(|observation| (observation.column.as_str(), observation.fact.as_str()))
7350 .collect::<Vec<_>>();
7351 assert_eq!(
7352 readings,
7353 vec![
7354 ("code", "100.00% parse as whole numbers"),
7355 ("amount", "100.00% parse as decimal numbers"),
7356 ]
7357 );
7358 let code = results
7359 .columns
7360 .iter()
7361 .find(|profile| profile.name == "code")
7362 .unwrap();
7363 assert_eq!(code.leading_zero_count, Some(28));
7365 }
7366
7367 #[test]
7370 fn columns_missing_together_are_found_to_share_their_rows() {
7371 let missing = |rows: &[usize]| {
7372 (0..10)
7373 .map(|row| (!rows.contains(&row)).then_some(row as f64))
7374 .collect::<Vec<_>>()
7375 };
7376 let frame = df!(
7377 "open" => missing(&[2, 5]),
7378 "close" => missing(&[2, 5]),
7379 "volume" => missing(&[3, 8]),
7380 "note" => missing(&[3, 9]),
7381 )
7382 .unwrap()
7383 .lazy();
7384 for compute in [QualityCompute::Sample, QualityCompute::Full] {
7385 let plan = DataQualityPlan {
7386 compute,
7387 ..DataQualityPlan::default()
7388 };
7389 let results = compute_data_quality(&frame, Some(10), &plan, None, false).unwrap();
7390 assert_eq!(
7391 results.shared_nulls,
7392 vec![SharedNulls {
7393 columns: ["open", "close", "volume", "note"]
7394 .map(String::from)
7395 .to_vec(),
7396 null_rows: 2,
7397 rows_null_in_all: 0,
7398 }],
7399 "{compute:?}: four columns with two nulls each share none of them all"
7400 );
7401 }
7402 let frame = df!("open" => missing(&[2, 5]), "close" => missing(&[2, 5]))
7403 .unwrap()
7404 .lazy();
7405 let results =
7406 compute_data_quality(&frame, Some(10), &DataQualityPlan::default(), None, false)
7407 .unwrap();
7408 assert!(results.shared_nulls[0].same_rows());
7409 }
7410
7411 #[test]
7414 fn the_largest_change_names_the_column_and_measurement_that_moved() {
7415 let frame = df!(
7416 "steady" => &[1i64, 2, 3, 4],
7417 "fee" => &[Some(1.5f64), Some(2.5), None, None],
7418 )
7419 .unwrap()
7420 .lazy();
7421 let plan = DataQualityPlan {
7422 compute: QualityCompute::Full,
7423 grain: QualityGrain::RowChunks(2),
7424 comparison: QualityComparison::Previous,
7425 ..DataQualityPlan::default()
7426 };
7427 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7428 assert_eq!(results.segments.len(), 2);
7429 assert_eq!(
7430 results.segments[0].largest_change, None,
7431 "the first chunk has nothing to compare against"
7432 );
7433 let change = results.segments[1]
7434 .largest_change
7435 .as_deref()
7436 .expect("the second chunk compares with the first");
7437 assert!(
7438 change == "fee nulls +100.0 pp",
7439 "the column and the measurement that moved: {change}"
7440 );
7441 }
7442
7443 #[test]
7445 fn a_segment_whose_rates_hold_still_reports_the_range_that_moved() {
7446 let frame = df!("reading" => &[1i64, 2, 300, 400]).unwrap().lazy();
7447 let plan = DataQualityPlan {
7448 compute: QualityCompute::Full,
7449 grain: QualityGrain::RowChunks(2),
7450 comparison: QualityComparison::Previous,
7451 ..DataQualityPlan::default()
7452 };
7453 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7454 let change = results.segments[1].largest_change.as_deref().unwrap();
7455 assert!(
7456 change.starts_with("reading range 1..2 -> 300..400"),
7457 "no rate moved, but the values did: {change}"
7458 );
7459 }
7460
7461 #[test]
7464 fn row_chunk_labels_agree_between_trends_and_segments_at_every_budget() {
7465 let frame = df!(
7466 "sent" => &[
7467 "2024-01-01T00:00:00", "2024-01-01T01:00:00",
7468 "2024-01-01T02:00:00", "2024-01-01T03:00:00",
7469 ],
7470 "landed" => &[
7471 "2024-01-01T01:00:00", "2024-01-01T03:00:00",
7472 "2024-01-01T04:00:00", "2024-01-01T06:00:00",
7473 ],
7474 )
7475 .unwrap()
7476 .lazy()
7477 .with_columns([
7478 col("sent")
7479 .str()
7480 .to_datetime(None, None, StrptimeOptions::default(), lit("raise")),
7481 col("landed")
7482 .str()
7483 .to_datetime(None, None, StrptimeOptions::default(), lit("raise")),
7484 ]);
7485 let roles = vec![
7486 TemporalRoleAssignment {
7487 role: TemporalRole::Published,
7488 column: "sent".to_string(),
7489 timezone: None,
7490 },
7491 TemporalRoleAssignment {
7492 role: TemporalRole::Received,
7493 column: "landed".to_string(),
7494 timezone: None,
7495 },
7496 ];
7497 for compute in [QualityCompute::Sample, QualityCompute::Full] {
7498 let plan = DataQualityPlan {
7499 compute,
7500 grain: QualityGrain::RowChunks(2),
7501 temporal_roles: roles.clone(),
7502 ..DataQualityPlan::default()
7503 };
7504 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7505 let segments = results
7506 .segments
7507 .iter()
7508 .map(|segment| segment.label.clone())
7509 .collect::<Vec<_>>();
7510 assert_eq!(
7511 segments,
7512 vec!["rows 1-2", "rows 3-4"],
7513 "{compute:?} segments"
7514 );
7515 let mut trends = results
7516 .temporal
7517 .iter()
7518 .map(|profile| profile.segment.clone())
7519 .collect::<Vec<_>>();
7520 trends.dedup();
7521 assert_eq!(trends, segments, "{compute:?} trends");
7522 }
7523 }
7524
7525 #[test]
7528 fn an_unassigned_plan_reports_no_latency() {
7529 let plan = DataQualityPlan {
7530 compute: QualityCompute::Sample,
7531 grain: QualityGrain::RowChunks(2),
7532 ..DataQualityPlan::default()
7533 };
7534 let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
7535 assert!(results.temporal.is_empty());
7536 }
7537
7538 #[test]
7539 fn source_row_map_produces_file_segments_without_profiling_hidden_columns() {
7540 let frame = df!(
7541 "value" => &[1i64, 2, 3, 4],
7542 "__row" => &[0u32, 1, 2, 3],
7543 )
7544 .unwrap()
7545 .lazy();
7546 let source = QualitySourceContext {
7547 file_names: vec!["a.parquet".to_string(), "b.parquet".to_string()],
7548 file_starts: vec![0, 2],
7549 row_index_column: "__row".to_string(),
7550 ..QualitySourceContext::default()
7551 };
7552 let plan = DataQualityPlan {
7553 compute: QualityCompute::Full,
7554 grain: QualityGrain::File,
7555 ..DataQualityPlan::default()
7556 };
7557 let results = compute_data_quality(&frame, Some(4), &plan, Some(&source), false).unwrap();
7558 assert_eq!(results.columns.len(), 1);
7559 assert_eq!(results.segments.len(), 2);
7560 assert_eq!(results.segments[0].evaluated_rows, 2);
7561 assert!(results.segments[0].label.contains("a.parquet"));
7562 }
7563
7564 #[test]
7565 fn identity_and_category_groups_keep_distinct_duplicate_semantics() {
7566 let frame = df!(
7567 "id" => &[1i64, 1, 2, 3],
7568 "category" => &["North", "North", " north ", "NORTH"],
7569 )
7570 .unwrap()
7571 .lazy();
7572 let plan = DataQualityPlan {
7573 compute: QualityCompute::Full,
7574 ..DataQualityPlan::default()
7575 };
7576 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7577 let identity = results.identity.unwrap();
7578 assert_eq!(identity.duplicate_groups, 1);
7579 assert_eq!(identity.extra_rows, 1);
7580 assert_eq!(identity.rows_involved, 2);
7581 assert_eq!(results.category_variants.len(), 1);
7582 assert_eq!(results.category_variants[0].rows_involved, 4);
7583 let category = results
7584 .columns
7585 .iter()
7586 .find(|profile| profile.name == "category")
7587 .unwrap();
7588 assert_eq!(category.dominant_value.as_deref(), Some("North"));
7589 assert_eq!(category.dominant_count, Some(2));
7590 }
7591
7592 #[test]
7593 fn sample_identity_does_not_equate_null_with_literal_text() {
7594 let frame = df!("value" => &[None, Some("<null>"), None])
7595 .unwrap()
7596 .lazy();
7597 let plan = DataQualityPlan {
7598 dataset_rows: 3,
7599 ..DataQualityPlan::default()
7600 };
7601 let results = compute_data_quality(&frame, Some(3), &plan, None, false).unwrap();
7602 let identity = results.identity.unwrap();
7603 assert_eq!(identity.duplicate_groups, 1);
7604 assert_eq!(identity.rows_involved, 2);
7605 }
7606
7607 #[test]
7608 fn partition_and_time_window_grains_create_ordered_profiles() {
7609 let timestamps = Series::new(
7610 "event_at".into(),
7611 [0i64, 86_400_000_000, 8 * 86_400_000_000],
7612 )
7613 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7614 .unwrap();
7615 let frame = DataFrame::new(
7616 3,
7617 vec![
7618 Column::new("partition".into(), ["a", "a", "b"]),
7619 Column::new("value".into(), [Some(1i64), None, Some(3)]),
7620 timestamps.into(),
7621 ],
7622 )
7623 .unwrap()
7624 .lazy();
7625
7626 let partition_plan = DataQualityPlan {
7627 compute: QualityCompute::Full,
7628 grain: QualityGrain::Partition("partition".to_string()),
7629 ..DataQualityPlan::default()
7630 };
7631 let partitioned =
7632 compute_data_quality(&frame, Some(3), &partition_plan, None, false).unwrap();
7633 assert_eq!(partitioned.segments.len(), 2);
7634 assert_eq!(partitioned.segments[0].evaluated_rows, 2);
7635
7636 let window_plan = DataQualityPlan {
7637 compute: QualityCompute::Full,
7638 grain: QualityGrain::TimeWindows {
7639 column: "event_at".to_string(),
7640 every: "1w".to_string(),
7641 },
7642 ..DataQualityPlan::default()
7643 };
7644 let windowed = compute_data_quality(&frame, Some(3), &window_plan, None, false).unwrap();
7645 assert_eq!(windowed.segments.len(), 2);
7646 assert!(windowed.segments[0].label.starts_with("week of "));
7647 }
7648
7649 #[test]
7650 fn an_exact_run_measures_everything_a_sampled_run_does() {
7651 let frame = df!(
7652 "when" => &["2024-01-01", "2024-01-02", "not a date", "2024-03-09"],
7653 "amount" => &["1", "2.5", "3", "bad"],
7654 )
7655 .unwrap()
7656 .lazy();
7657 let dupes = df!(
7659 "label" => &[Some("x"), Some("x"), Some("x"), Some("y"), None],
7660 "n" => &[7i64, 7, 7, 7, 1],
7661 )
7662 .unwrap()
7663 .lazy();
7664 let measured = |compute| {
7665 let plan = DataQualityPlan {
7666 compute,
7667 ..DataQualityPlan::default()
7668 };
7669 let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7670 results
7671 .columns
7672 .iter()
7673 .map(|column| {
7674 (
7675 column.name.clone(),
7676 column.integer_parse_count,
7677 column.decimal_parse_count,
7678 column.date_parse_count,
7679 column.datetime_parse_count,
7680 )
7681 })
7682 .collect::<Vec<_>>()
7683 };
7684 let sampled = measured(QualityCompute::Sample);
7685 assert_eq!(sampled, measured(QualityCompute::Full));
7686
7687 let dominant = |compute| {
7688 let plan = DataQualityPlan {
7689 compute,
7690 ..DataQualityPlan::default()
7691 };
7692 compute_data_quality(&dupes, Some(5), &plan, None, false)
7693 .unwrap()
7694 .columns
7695 .iter()
7696 .map(|column| (column.dominant_value.clone(), column.dominant_count))
7697 .collect::<Vec<_>>()
7698 };
7699 assert_eq!(
7700 dominant(QualityCompute::Full),
7701 vec![
7702 (Some("x".to_string()), Some(3)),
7703 (Some("7".to_string()), Some(4))
7704 ]
7705 );
7706 assert_eq!(
7707 dominant(QualityCompute::Sample),
7708 dominant(QualityCompute::Full)
7709 );
7710 assert_eq!(sampled[0].3, Some(3), "three of four values are ISO dates");
7712
7713 let stamps = df!("t" => &[
7715 "2024-01-01T00:00:00Z",
7716 "2024-01-01T00:00:00.500Z",
7717 "2024-01-01T00:00:00+01:00",
7718 "2024-01-01 00:00:00",
7719 "2024-01-01T00:00:00",
7720 "garbage",
7721 ])
7722 .unwrap()
7723 .lazy();
7724 for compute in [QualityCompute::Sample, QualityCompute::Full] {
7725 let plan = DataQualityPlan {
7726 compute,
7727 ..DataQualityPlan::default()
7728 };
7729 let results = compute_data_quality(&stamps, Some(6), &plan, None, false).unwrap();
7730 assert_eq!(
7731 results.columns[0].datetime_parse_count,
7732 Some(5),
7733 "{compute:?} should accept every ISO timestamp but the garbage"
7734 );
7735 }
7736 assert_eq!(sampled[1].2, Some(3), "three of four parse as decimal");
7737 assert_eq!(sampled[1].1, Some(2), "two of four parse as integer");
7738
7739 let observed = |compute| {
7740 let plan = DataQualityPlan {
7741 compute,
7742 ..DataQualityPlan::default()
7743 };
7744 let mut kinds = compute_data_quality(&frame, Some(4), &plan, None, false)
7745 .unwrap()
7746 .observations
7747 .iter()
7748 .map(|item| (item.kind, item.column.clone()))
7749 .collect::<Vec<_>>();
7750 kinds.sort_by(|left, right| {
7751 left.1
7752 .cmp(&right.1)
7753 .then(format!("{:?}", left.0).cmp(&format!("{:?}", right.0)))
7754 });
7755 kinds
7756 };
7757 assert_eq!(
7758 observed(QualityCompute::Sample),
7759 observed(QualityCompute::Full)
7760 );
7761 }
7762
7763 #[test]
7764 fn a_plan_naming_a_column_the_scope_lost_does_not_kill_the_run() {
7765 let frame = df!("id" => &[1i64, 2, 3]).unwrap().lazy();
7766 let plan = DataQualityPlan {
7768 compute: QualityCompute::Full,
7769 temporal_roles: vec![
7770 TemporalRoleAssignment {
7771 role: TemporalRole::Event,
7772 column: "gone".to_string(),
7773 timezone: None,
7774 },
7775 TemporalRoleAssignment {
7776 role: TemporalRole::Received,
7777 column: "also_gone".to_string(),
7778 timezone: None,
7779 },
7780 ],
7781 ..DataQualityPlan::default()
7782 };
7783 for compute in [QualityCompute::Sample, QualityCompute::Full] {
7784 let results = compute_data_quality(
7785 &frame,
7786 Some(3),
7787 &DataQualityPlan {
7788 compute,
7789 ..plan.clone()
7790 },
7791 None,
7792 false,
7793 )
7794 .unwrap_or_else(|error| panic!("{compute:?} with a stale role: {error}"));
7795 assert!(results.temporal.is_empty());
7796 assert_eq!(results.columns.len(), 1);
7797 }
7798
7799 let error = compute_data_quality(
7801 &frame,
7802 Some(3),
7803 &DataQualityPlan {
7804 grain: QualityGrain::Partition("region".to_string()),
7805 ..plan
7806 },
7807 None,
7808 false,
7809 )
7810 .expect_err("a grain column that is not in scope must be refused");
7811 assert!(
7812 error.to_string().contains("region") && error.to_string().contains("not in scope"),
7813 "the refusal should name the column: {error}"
7814 );
7815 }
7816
7817 #[test]
7818 fn every_scope_survives_a_trip_through_the_editor() {
7819 for scope in [
7820 QualityScope::CurrentView,
7821 QualityScope::WholeSource,
7822 QualityScope::FirstRows(10_000),
7823 QualityScope::FirstRows(1_000_000),
7824 QualityScope::ViewRows {
7825 start: 100,
7826 end: 200,
7827 },
7828 QualityScope::SourceFiles(vec![1, 3]),
7829 QualityScope::SourcePartition {
7830 column: "region".to_string(),
7831 value: "west".to_string(),
7832 },
7833 QualityScope::SourceTimeRange {
7834 column: "event".to_string(),
7835 start: "2024-01-01".to_string(),
7836 end: "2024-02-01".to_string(),
7837 },
7838 ] {
7839 assert_eq!(
7840 QualityScope::parse_command(&scope.command()).unwrap(),
7841 scope,
7842 "{} should come back as itself",
7843 scope.command()
7844 );
7845 }
7846 assert!(QualityScope::parse_command("rows 1..0").is_err());
7848 assert!(QualityScope::parse_command("rows 0..5").is_err());
7849 }
7850
7851 #[test]
7852 fn metadata_mode_does_not_need_the_grain_column() {
7853 let frame = df!("id" => &[1i64, 2]).unwrap().lazy();
7855 let plan = DataQualityPlan {
7856 compute: QualityCompute::Metadata,
7857 grain: QualityGrain::Partition("gone".to_string()),
7858 ..DataQualityPlan::default()
7859 };
7860 let results = compute_data_quality(&frame, Some(2), &plan, None, false).unwrap();
7861 assert_eq!(results.precision, QualityPrecision::Metadata);
7862 }
7863
7864 #[test]
7865 fn segments_come_back_in_one_order_however_much_was_read() {
7866 let frame = df!(
7869 "region" => &[Some("a"), Some("a"), Some("\u{6771}\u{4eac}"), Some("\u{6771}\u{4eac}"), None, None],
7870 "id" => &[1i64, 2, 3, 4, 5, 6],
7871 )
7872 .unwrap()
7873 .lazy();
7874 let plan = DataQualityPlan {
7875 grain: QualityGrain::Partition("region".to_string()),
7876 ..DataQualityPlan::default()
7877 };
7878 let labels = |compute| {
7879 compute_data_quality(
7880 &frame,
7881 Some(6),
7882 &DataQualityPlan {
7883 compute,
7884 ..plan.clone()
7885 },
7886 None,
7887 false,
7888 )
7889 .unwrap()
7890 .segments
7891 .iter()
7892 .map(|segment| segment.label.clone())
7893 .collect::<Vec<_>>()
7894 };
7895 let full = labels(QualityCompute::Full);
7896 assert_eq!(labels(QualityCompute::Sample), full);
7897 assert_eq!(full.last().unwrap(), "region=\u{2205}");
7898 }
7899
7900 #[test]
7901 fn a_categorical_column_is_profiled_rather_than_failing_the_run() {
7902 let frame = df!(
7903 "label" => &["a", "b", "a", " c "],
7904 "n" => &[1i64, 2, 3, 4],
7905 )
7906 .unwrap()
7907 .lazy()
7908 .with_columns([col("label").cast(DataType::from_categories(Categories::global()))]);
7909 for compute in [QualityCompute::Sample, QualityCompute::Full] {
7910 let plan = DataQualityPlan {
7911 compute,
7912 ..DataQualityPlan::default()
7913 };
7914 let results = compute_data_quality(&frame, Some(4), &plan, None, false)
7915 .unwrap_or_else(|error| panic!("{compute:?} on a categorical column: {error}"));
7916 let label = results
7917 .columns
7918 .iter()
7919 .find(|column| column.name == "label")
7920 .expect("the categorical column is profiled");
7921 assert_eq!(label.null_count, 0);
7922 assert_eq!(label.distinct_count, Some(3));
7923 }
7924 }
7925
7926 #[test]
7927 fn every_offered_window_width_cuts_the_scope_it_names() {
7928 let day = 86_400_000_000i64;
7930 let days = 40i64;
7931 let stamps = Series::new(
7932 "event_at".into(),
7933 (0..days).map(|d| d * day).collect::<Vec<_>>(),
7934 )
7935 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7936 .unwrap();
7937 let frame = DataFrame::new(
7938 days as usize,
7939 vec![
7940 Column::new("value".into(), (0..days).collect::<Vec<_>>()),
7941 stamps.into(),
7942 ],
7943 )
7944 .unwrap()
7945 .lazy();
7946 let expected = [("1h", 40), ("1d", 40), ("1w", 7), ("1mo", 2)];
7948 for (every, segments) in expected {
7949 let plan = DataQualityPlan {
7950 compute: QualityCompute::Full,
7951 grain: QualityGrain::TimeWindows {
7952 column: "event_at".to_string(),
7953 every: every.to_string(),
7954 },
7955 ..DataQualityPlan::default()
7956 };
7957 let results =
7958 compute_data_quality(&frame, Some(days as usize), &plan, None, false).unwrap();
7959 assert_eq!(results.segments.len(), segments, "{every} windows");
7960 assert_eq!(
7961 results
7962 .segments
7963 .iter()
7964 .map(|segment| segment.evaluated_rows)
7965 .sum::<usize>(),
7966 days as usize,
7967 "{every} windows must account for every row"
7968 );
7969 let label = &results.segments[0].label;
7971 match every {
7972 "1h" => assert_eq!(label.len(), "2024-01-01 00:00".len(), "{label}"),
7973 "1d" => assert_eq!(label.len(), "2024-01-01".len(), "{label}"),
7974 "1w" => assert!(label.starts_with("week of "), "{label}"),
7975 _ => assert_eq!(label.len(), "2024-01".len(), "{label}"),
7976 }
7977 }
7978 }
7979
7980 #[test]
7981 fn rows_without_a_window_clock_are_named_and_ordered_the_same_however_much_was_read() {
7982 let timestamps = Series::new(
7983 "event_at".into(),
7984 [Some(0i64), Some(8 * 86_400_000_000), None, None],
7985 )
7986 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7987 .unwrap();
7988 let frame = DataFrame::new(
7989 4,
7990 vec![
7991 Column::new("value".into(), [1i64, 2, 3, 4]),
7992 timestamps.into(),
7993 ],
7994 )
7995 .unwrap()
7996 .lazy();
7997 let plan = DataQualityPlan {
7998 grain: QualityGrain::TimeWindows {
7999 column: "event_at".to_string(),
8000 every: "1w".to_string(),
8001 },
8002 ..DataQualityPlan::default()
8003 };
8004
8005 let sampled = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
8006 let full = compute_data_quality(
8007 &frame,
8008 Some(4),
8009 &DataQualityPlan {
8010 compute: QualityCompute::Full,
8011 ..plan.clone()
8012 },
8013 None,
8014 false,
8015 )
8016 .unwrap();
8017
8018 let labels = |results: &DataQualityResults| {
8019 results
8020 .segments
8021 .iter()
8022 .map(|segment| segment.label.clone())
8023 .collect::<Vec<_>>()
8024 };
8025 assert_eq!(labels(&sampled), labels(&full));
8026 assert_eq!(labels(&full).len(), 3);
8028 assert_eq!(labels(&full)[2], "event_at ∅");
8029 assert!(labels(&full)[0].starts_with("week of "));
8030 }
8031
8032 fn text_times() -> LazyFrame {
8033 df!(
8034 "created" => [
8035 Some("2024-01-01 08:00:00"),
8036 Some("2024-01-01 09:30:00"),
8037 Some("2024-01-02 10:00:00"),
8038 Some("not a time"),
8039 None,
8040 Some("2024-01-03 12:00:00"),
8041 ],
8042 "sent" => [
8043 Some("2024-01-01 09:00:00"),
8044 Some("2024-01-01 09:00:00"),
8045 None,
8046 Some("2024-01-02 11:00:00"),
8047 Some("2024-01-02 11:00:00"),
8048 Some("2024-01-03 12:30:00"),
8049 ],
8050 )
8051 .unwrap()
8052 .lazy()
8053 }
8054
8055 fn read_as_datetime(column: &str) -> TimeInterpretation {
8056 TimeInterpretation {
8057 column: column.to_string(),
8058 kind: TimeKind::Datetime,
8059 format: "%Y-%m-%d %H:%M:%S".to_string(),
8060 }
8061 }
8062
8063 #[test]
8067 fn text_read_as_time_windows_and_measures_intervals() {
8068 let plan = DataQualityPlan {
8069 grain: QualityGrain::TimeWindows {
8070 column: "created".to_string(),
8071 every: "1d".to_string(),
8072 },
8073 temporal_roles: vec![
8074 TemporalRoleAssignment {
8075 role: TemporalRole::Event,
8076 column: "created".to_string(),
8077 timezone: None,
8078 },
8079 TemporalRoleAssignment {
8080 role: TemporalRole::Received,
8081 column: "sent".to_string(),
8082 timezone: None,
8083 },
8084 ],
8085 time_formats: vec![read_as_datetime("created"), read_as_datetime("sent")],
8086 ..DataQualityPlan::default()
8087 };
8088 for compute in [QualityCompute::Sample, QualityCompute::Full] {
8089 let plan = DataQualityPlan {
8090 compute,
8091 ..plan.clone()
8092 };
8093 let results = compute_data_quality(&text_times(), Some(6), &plan, None, false)
8094 .unwrap_or_else(|error| panic!("{compute:?}: {error}"));
8095 let labels = results
8096 .segments
8097 .iter()
8098 .map(|segment| segment.label.clone())
8099 .collect::<Vec<_>>();
8100 assert_eq!(
8101 labels,
8102 ["2024-01-01", "2024-01-02", "2024-01-03", "created ∅"],
8103 "{compute:?}"
8104 );
8105 let (unparsed_start, missing_start) =
8106 results
8107 .temporal
8108 .iter()
8109 .fold((0, 0), |(unparsed, missing), latency| {
8110 (
8111 unparsed + latency.unparsed_start,
8112 missing + latency.missing_start,
8113 )
8114 });
8115 assert_eq!((unparsed_start, missing_start), (1, 1), "{compute:?}");
8116 let first_day = results
8117 .temporal
8118 .iter()
8119 .find(|latency| latency.segment == "2024-01-01")
8120 .unwrap();
8121 assert_eq!(first_day.negative_count, 1, "{compute:?}");
8123 assert_eq!(first_day.max_seconds, Some(3_600), "{compute:?}");
8124
8125 let unparsed = results
8126 .observations
8127 .iter()
8128 .find(|observation| observation.kind == ObservationKind::UnparsedTime)
8129 .unwrap_or_else(|| panic!("{compute:?}: no unparsed finding"));
8130 assert_eq!(unparsed.column, "created");
8131 assert_eq!((unparsed.affected_rows, unparsed.evaluated_rows), (1, 5));
8132 let rows = text_times()
8133 .filter(unparsed.evidence_predicate().unwrap())
8134 .collect()
8135 .unwrap();
8136 assert_eq!(rows.height(), 1, "the evidence is the unread value");
8137 let created = results
8138 .columns
8139 .iter()
8140 .find(|column| column.name == "created")
8141 .unwrap();
8142 assert_eq!(created.dtype, DataType::String, "still text to every check");
8143 assert_eq!(created.null_count, 1);
8144 }
8145 }
8146
8147 #[test]
8150 fn text_without_a_format_is_not_read_as_time() {
8151 let plan = DataQualityPlan {
8152 temporal_roles: vec![
8153 TemporalRoleAssignment {
8154 role: TemporalRole::Event,
8155 column: "created".to_string(),
8156 timezone: None,
8157 },
8158 TemporalRoleAssignment {
8159 role: TemporalRole::Received,
8160 column: "sent".to_string(),
8161 timezone: None,
8162 },
8163 ],
8164 ..DataQualityPlan::default()
8165 };
8166 let results = compute_data_quality(&text_times(), Some(6), &plan, None, false).unwrap();
8167 assert!(results.temporal.is_empty());
8168 let windows = DataQualityPlan {
8169 grain: QualityGrain::TimeWindows {
8170 column: "created".to_string(),
8171 every: "1d".to_string(),
8172 },
8173 ..DataQualityPlan::default()
8174 };
8175 let error = compute_data_quality(&text_times(), Some(6), &windows, None, false)
8176 .unwrap_err()
8177 .to_string();
8178 assert!(error.contains("Text as time"), "{error}");
8179 }
8180
8181 #[test]
8184 fn every_offered_format_reads_its_example_the_same_way_twice() {
8185 let samples = [
8186 "2024-01-31 08:15:00",
8187 "2024-01-31T08:15:00",
8188 "2024-01-31 08:15:00.250",
8189 "2024-01-31T08:15:00.5",
8190 "2024-01-31T08:15:00Z",
8191 "2024-01-31T08:15:00.250+05:00",
8192 "2024-01-31 08:15:00-0500",
8193 "2024-01-31 08:15",
8194 "2024-01-31",
8195 "20240131",
8196 "01/31/2024 08:15:00",
8197 "01/31/2024 08:15:00 AM",
8198 "31/01/2024 08:15:00",
8199 "31.01.2024 08:15:00",
8200 "01/31/2024",
8201 "31/01/2024",
8202 "31.01.2024",
8203 "not a time",
8204 ];
8205 let frame = df!("text" => samples).unwrap().lazy();
8206 for (kind, format) in TIME_FORMATS {
8207 let interpretation = TimeInterpretation {
8208 column: "text".to_string(),
8209 kind,
8210 format: format.to_string(),
8211 };
8212 let parsed = frame
8213 .clone()
8214 .select([interpretation.expr().is_not_null().alias("read")])
8215 .collect()
8216 .unwrap();
8217 let read = parsed.column("read").unwrap().bool().unwrap().clone();
8218 let mut any = false;
8219 for (index, sample) in samples.iter().enumerate() {
8220 let polars = read.get(index).unwrap_or(false);
8221 any |= polars;
8222 assert_eq!(
8223 interpretation.reads(sample),
8224 polars,
8225 "{format} on {sample:?}"
8226 );
8227 }
8228 assert!(any, "{format} reads none of the examples");
8229 }
8230 }
8231
8232 #[test]
8235 fn a_run_reports_its_stages_and_stops_when_cancelled() {
8236 let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8237 let seen = Arc::clone(&stages);
8238 let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8239 let plan = DataQualityPlan {
8240 grain: QualityGrain::Partition("constant".to_string()),
8241 ..DataQualityPlan::default()
8242 };
8243 let (results, kept) =
8244 compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &watch);
8245 results.unwrap();
8246 let stages = stages.lock().unwrap().clone();
8247 assert_eq!(stages.first().unwrap().stage, QualityStage::Preparing);
8248 assert_eq!(stages.last().unwrap().stage, QualityStage::Assembling);
8249 let read = stages
8250 .iter()
8251 .find(|phase| phase.stage == QualityStage::ReadingSample)
8252 .unwrap();
8253 assert!(read.reads_source);
8254 assert!(
8255 stages
8256 .iter()
8257 .filter(|phase| phase.stage == QualityStage::ProfilingColumns)
8258 .all(|phase| !phase.reads_source),
8259 "the sample is profiled in memory"
8260 );
8261 let distinct = stages
8262 .iter()
8263 .map(|phase| phase.stage.label())
8264 .collect::<std::collections::HashSet<_>>();
8265 assert_eq!(
8266 stages.len(),
8267 distinct.len(),
8268 "each stage said once: {stages:?}"
8269 );
8270
8271 let again = Arc::new(std::sync::Mutex::new(Vec::new()));
8273 let seen = Arc::clone(&again);
8274 let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8275 compute_data_quality_watched(
8276 &fixture(),
8277 Some(4),
8278 &plan,
8279 None,
8280 false,
8281 kept.as_ref(),
8282 &watch,
8283 )
8284 .0
8285 .unwrap();
8286 let again = again.lock().unwrap().clone();
8287 assert!(again.iter().all(|phase| !phase.reads_source), "{again:?}");
8288 assert!(
8289 again
8290 .iter()
8291 .any(|phase| phase.stage == QualityStage::ReusingSample)
8292 );
8293
8294 let cancelled = QualityWatch::default();
8295 cancelled.cancel();
8296 let (results, kept) =
8297 compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &cancelled);
8298 assert_eq!(results.unwrap_err().to_string(), crate::sampling::CANCELLED);
8299 assert!(kept.is_none(), "stopped before its read, it read nothing");
8300
8301 let late = QualityWatch::default();
8303 let stopper = late.clone();
8304 let late = QualityWatch {
8305 report: Some(Arc::new(move |phase: QualityPhase| {
8306 if phase.stage == QualityStage::ProfilingColumns {
8307 stopper.cancel();
8308 }
8309 })),
8310 ..late
8311 };
8312 let (results, kept) =
8313 compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &late);
8314 assert!(results.is_err());
8315 assert!(kept.is_some(), "the sample it read comes back");
8316 }
8317
8318 fn counting_table(rows: usize) -> (DataFrame, LazyFrame, Arc<std::sync::atomic::AtomicUsize>) {
8322 let start = chrono::NaiveDate::from_ymd_opt(2023, 12, 18)
8323 .unwrap()
8324 .and_hms_opt(0, 0, 0)
8325 .unwrap();
8326 let at: Vec<chrono::NaiveDateTime> = (0..rows)
8329 .map(|row| start + chrono::Duration::minutes(row as i64 * 97))
8330 .collect();
8331 let micros = |times: &[chrono::NaiveDateTime]| {
8332 times
8333 .iter()
8334 .map(|time| time.and_utc().timestamp_micros())
8335 .collect::<Vec<_>>()
8336 };
8337 let sent: Vec<chrono::NaiveDateTime> = at
8338 .iter()
8339 .enumerate()
8340 .map(|(row, time)| *time + chrono::Duration::seconds(30 + (row % 7) as i64))
8341 .collect();
8342 let datetime = DataType::Datetime(TimeUnit::Microseconds, None);
8343 let df = DataFrame::new(
8344 rows,
8345 vec![
8346 Column::new("id".into(), (0..rows as i64).collect::<Vec<_>>()),
8347 Column::new("at".into(), micros(&at))
8348 .cast(&datetime)
8349 .unwrap(),
8350 Column::new("sent".into(), micros(&sent))
8351 .cast(&datetime)
8352 .unwrap(),
8353 Column::new(
8354 "sent_text".into(),
8355 sent.iter()
8356 .map(|time| time.format("%m/%d/%Y %H:%M:%S").to_string())
8357 .collect::<Vec<_>>(),
8358 ),
8359 Column::new(
8360 "region".into(),
8361 (0..rows)
8362 .map(|row| ["North", "South", "East"][row % 3])
8363 .collect::<Vec<_>>(),
8364 ),
8365 ],
8366 )
8367 .unwrap();
8368 let read = Arc::new(std::sync::atomic::AtomicUsize::new(0));
8369 let counter = Arc::clone(&read);
8370 let lf = df.clone().lazy().filter(col("id").map(
8371 move |column| {
8372 counter.fetch_add(column.len(), std::sync::atomic::Ordering::Relaxed);
8373 Ok(column.is_not_null().into_column())
8374 },
8375 |_, field| Ok(Field::new(field.name().clone(), DataType::Boolean)),
8376 ));
8377 (df, lf, read)
8378 }
8379
8380 fn segment_totals(results: &DataQualityResults) -> BTreeMap<String, Option<usize>> {
8381 results
8382 .segments
8383 .iter()
8384 .map(|segment| (segment.label.clone(), segment.total_rows))
8385 .collect()
8386 }
8387
8388 fn assert_exact(results: &DataQualityResults, df: &DataFrame, plan: &DataQualityPlan) {
8390 let full = DataQualityPlan {
8391 compute: QualityCompute::Full,
8392 ..plan.clone()
8393 };
8394 let exact = segment_totals(
8395 &compute_data_quality(&df.clone().lazy(), None, &full, None, false).unwrap(),
8396 );
8397 assert!(!results.segments.is_empty());
8398 for (label, total) in segment_totals(results) {
8399 assert_eq!(total, exact[&label], "{label} of {:?}", plan.grain);
8400 }
8401 for missed in &results.unsampled_segments {
8404 assert!(
8405 !results
8406 .segments
8407 .iter()
8408 .any(|segment| segment.label == missed.label),
8409 "{} of {:?}",
8410 missed.label,
8411 plan.grain
8412 );
8413 assert_eq!(Some(missed.total_rows), exact[&missed.label]);
8414 }
8415 if results
8416 .segments
8417 .iter()
8418 .all(|segment| segment.total_rows.is_some())
8419 {
8420 assert_eq!(
8421 results.segments.len() + results.unsampled_segments.len(),
8422 exact.len(),
8423 "{:?}",
8424 plan.grain
8425 );
8426 }
8427 }
8428
8429 #[test]
8436 fn each_edit_reads_only_what_it_needs() {
8437 let rows = 3_000;
8438 let (df, lf, read) = counting_table(rows);
8439 let daily = DataQualityPlan {
8440 dataset_rows: 300,
8441 sample_seed: 5,
8442 grain: QualityGrain::TimeWindows {
8443 column: "at".into(),
8444 every: "1d".into(),
8445 },
8446 ..DataQualityPlan::default()
8447 };
8448 let roles = vec![
8449 TemporalRoleAssignment {
8450 role: TemporalRole::Event,
8451 column: "at".into(),
8452 timezone: None,
8453 },
8454 TemporalRoleAssignment {
8455 role: TemporalRole::Received,
8456 column: "sent_text".into(),
8457 timezone: None,
8458 },
8459 ];
8460 let sent_format = TimeInterpretation {
8461 column: "sent_text".into(),
8462 kind: TimeKind::Datetime,
8463 format: "%m/%d/%Y %H:%M:%S".into(),
8464 };
8465 let window = |column: &str, every: &str| QualityGrain::TimeWindows {
8466 column: column.into(),
8467 every: every.into(),
8468 };
8469 let with = |grain: QualityGrain| DataQualityPlan {
8470 grain,
8471 temporal_roles: roles.clone(),
8472 time_formats: vec![sent_format.clone()],
8473 ..daily.clone()
8474 };
8475 let mut kept: Option<QualitySample> = None;
8476 let mut run = |plan: &DataQualityPlan, reuse: bool| {
8477 read.store(0, std::sync::atomic::Ordering::Relaxed);
8478 let (results, acquired) = compute_data_quality_kept(
8479 &lf,
8480 None,
8481 plan,
8482 None,
8483 false,
8484 if reuse { kept.as_ref() } else { None },
8485 )
8486 .unwrap();
8487 kept = acquired;
8488 (results, read.load(std::sync::atomic::Ordering::Relaxed))
8489 };
8490 let passes = |read: usize| read as f64 / rows as f64;
8491
8492 let (first, reads) = run(&daily, false);
8493 assert_eq!(passes(reads), 1.0, "sampled and counted in one pass");
8494 assert_eq!(first.precision, QualityPrecision::Sampled);
8495 assert_exact(&first, &df, &daily);
8496
8497 let roled = DataQualityPlan {
8499 temporal_roles: roles.clone(),
8500 ..daily.clone()
8501 };
8502 let (_, reads) = run(&roled, true);
8503 assert_eq!(reads, 0, "a role edit reads nothing");
8504 let interpreted = with(daily.grain.clone());
8505 let (results, reads) = run(&interpreted, true);
8506 assert_eq!(reads, 0, "an interpretation edit reads nothing");
8507 assert!(!results.temporal.is_empty(), "and measures the interval");
8508
8509 for every in ["1w", "1mo"] {
8511 let plan = with(window("at", every));
8512 let (results, reads) = run(&plan, true);
8513 assert_eq!(reads, 0, "{every} from the daily counts");
8514 assert_exact(&results, &df, &plan);
8515 }
8516 for grain in [window("at", "1h"), QualityGrain::Partition("region".into())] {
8518 let plan = with(grain.clone());
8519 let (results, reads) = run(&plan, true);
8520 assert_eq!(passes(reads), 1.0, "{grain:?} is counted");
8521 assert_exact(&results, &df, &plan);
8522 let (_, reads) = run(&plan, true);
8523 assert_eq!(reads, 0, "{grain:?} is counted once");
8524 }
8525 let plan = with(window("sent_text", "1d"));
8527 let (results, reads) = run(&plan, true);
8528 assert_eq!(passes(reads), 1.0);
8529 assert_exact(&results, &df, &plan);
8530
8531 let chunks = with(QualityGrain::RowChunks(500));
8533 let (results, reads) = run(&chunks, true);
8534 assert_eq!(reads, 0, "row chunks after a first run read nothing");
8535 let fine = with(QualityGrain::RowChunks(10));
8537 let (missed, _) = run(&fine, true);
8538 assert!(!missed.unsampled_segments.is_empty());
8539 assert_exact(&missed, &df, &fine);
8540 let (fresh, _) = run(&chunks, false);
8541 assert_eq!(
8542 format!("{:?}", results.segments),
8543 format!("{:?}", fresh.segments),
8544 "the chunks a chunked read cuts"
8545 );
8546
8547 for plan in [
8550 DataQualityPlan {
8551 sample_seed: 6,
8552 ..daily.clone()
8553 },
8554 DataQualityPlan {
8555 dataset_rows: 400,
8556 ..daily.clone()
8557 },
8558 DataQualityPlan {
8559 scope: QualityScope::FirstRows(2_000),
8560 ..daily.clone()
8561 },
8562 ] {
8563 let scoped = apply_quality_scope(lf.clone(), &plan.scope, None).unwrap();
8564 read.store(0, std::sync::atomic::Ordering::Relaxed);
8565 let (results, _) =
8566 compute_data_quality_kept(&scoped, None, &plan, None, false, None).unwrap();
8567 assert_eq!(passes(read.load(std::sync::atomic::Ordering::Relaxed)), 1.0);
8568 let scoped = apply_quality_scope(df.clone().lazy(), &plan.scope, None)
8569 .unwrap()
8570 .collect()
8571 .unwrap();
8572 assert_exact(&results, &scoped, &plan);
8573 }
8574 }
8575
8576 #[test]
8581 fn counts_in_the_sampling_pass_match_a_full_scan() {
8582 let (df, _, _) = counting_table(3_000);
8583 let gaps = |name: &str| {
8584 when((col("id") % lit(11i64)).eq(lit(0i64)))
8585 .then(lit(NULL))
8586 .otherwise(col(name))
8587 .alias(name)
8588 };
8589 let df = df
8590 .lazy()
8591 .with_columns([gaps("at"), gaps("region")])
8592 .with_columns([
8593 col("at")
8594 .dt()
8595 .replace_time_zone(
8596 TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
8597 lit("earliest"),
8598 NonExistent::Null,
8599 )
8600 .alias("zoned"),
8601 col("at").cast(DataType::Date).alias("day"),
8602 ])
8603 .collect()
8604 .unwrap();
8605 let window = |column: &str, every: &str| QualityGrain::TimeWindows {
8606 column: column.into(),
8607 every: every.into(),
8608 };
8609 for grain in [
8610 window("at", "1h"),
8611 window("zoned", "1d"),
8612 window("day", "1d"),
8613 QualityGrain::Partition("region".into()),
8614 ] {
8615 let plan = DataQualityPlan {
8616 dataset_rows: 200,
8617 sample_seed: 3,
8618 grain: grain.clone(),
8619 ..DataQualityPlan::default()
8620 };
8621 let (results, kept) =
8622 compute_data_quality_kept(&df.clone().lazy(), None, &plan, None, false, None)
8623 .unwrap();
8624 assert_eq!(results.precision, QualityPrecision::Sampled);
8625 assert!(
8626 results
8627 .segments
8628 .iter()
8629 .any(|segment| segment.label.contains('∅')),
8630 "{grain:?} has a null segment"
8631 );
8632 assert_exact(&results, &df, &plan);
8633 let kept = kept.unwrap();
8634 assert_eq!(
8635 kept.segment_count(&plan),
8636 SegmentCount::Retained,
8637 "{grain:?}"
8638 );
8639 if let QualityGrain::TimeWindows { column, .. } = &grain {
8640 let monthly = DataQualityPlan {
8641 grain: window(column, "1mo"),
8642 ..plan.clone()
8643 };
8644 let (results, _) = compute_data_quality_kept(
8645 &df.clone().lazy(),
8646 None,
8647 &monthly,
8648 None,
8649 false,
8650 Some(&kept),
8651 )
8652 .unwrap();
8653 assert_exact(&results, &df, &monthly);
8654 }
8655 }
8656 }
8657
8658 #[test]
8663 fn finer_windows_sum_to_coarser_ones_exactly() {
8664 let (df, _, _) = counting_table(4_000);
8665 let df = df
8666 .lazy()
8667 .with_columns([
8668 col("at")
8669 .dt()
8670 .replace_time_zone(
8671 TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
8672 lit("earliest"),
8673 NonExistent::Null,
8674 )
8675 .alias("zoned"),
8676 col("at").cast(DataType::Date).alias("day"),
8677 ])
8678 .collect()
8679 .unwrap();
8680 let count = |column: &str, every: &str| {
8681 counted_segment_totals(
8682 &df.clone().lazy(),
8683 &DataQualityPlan {
8684 grain: QualityGrain::TimeWindows {
8685 column: column.into(),
8686 every: every.into(),
8687 },
8688 ..DataQualityPlan::default()
8689 },
8690 false,
8691 )
8692 .unwrap()
8693 };
8694 let widths = ["1h", "1d", "1w", "1mo"];
8695 for column in ["at", "zoned", "day"] {
8696 for fine in widths {
8697 for coarse in widths
8698 .into_iter()
8699 .filter(|coarse| window_nests(fine, coarse))
8700 {
8701 assert_eq!(
8702 roll_up_windows(&count(column, fine), coarse).unwrap(),
8703 count(column, coarse),
8704 "{column}: {fine} into {coarse}"
8705 );
8706 }
8707 }
8708 }
8709 assert!(!window_nests("1w", "1mo"));
8710 assert!(!window_nests("1d", "1h"));
8711 assert!(!window_nests("1d", "1d"));
8712 }
8713
8714 #[test]
8717 fn a_sample_says_where_its_segment_totals_come_from() {
8718 let (_, lf, _) = counting_table(2_000);
8719 let daily = DataQualityPlan {
8720 dataset_rows: 100,
8721 grain: QualityGrain::TimeWindows {
8722 column: "at".into(),
8723 every: "1d".into(),
8724 },
8725 ..DataQualityPlan::default()
8726 };
8727 assert_eq!(
8728 fresh_segment_count(&daily, false),
8729 SegmentCount::InSamplePass
8730 );
8731 assert_eq!(fresh_segment_count(&daily, true), SegmentCount::CountPass);
8732 let head = DataQualityPlan {
8733 method: crate::sampling::SampleMethod::FirstRows,
8734 ..daily.clone()
8735 };
8736 assert_eq!(fresh_segment_count(&head, false), SegmentCount::CountPass);
8737 let (_, kept) = compute_data_quality_kept(&lf, None, &daily, None, false, None).unwrap();
8738 let mut kept = kept.unwrap();
8739 let grain = |every: &str| DataQualityPlan {
8740 grain: QualityGrain::TimeWindows {
8741 column: "at".into(),
8742 every: every.into(),
8743 },
8744 ..daily.clone()
8745 };
8746 assert_eq!(kept.segment_count(&daily), SegmentCount::Retained);
8747 assert_eq!(
8748 kept.segment_count(&grain("1w")),
8749 SegmentCount::RolledUp("1d".into())
8750 );
8751 assert_eq!(kept.segment_count(&grain("1h")), SegmentCount::CountPass);
8752 let chunks = DataQualityPlan {
8753 grain: QualityGrain::RowChunks(100),
8754 ..daily.clone()
8755 };
8756 assert_eq!(kept.segment_count(&chunks), SegmentCount::NotNeeded);
8757
8758 let by_region = DataQualityPlan {
8761 grain: QualityGrain::Partition("region".into()),
8762 ..daily.clone()
8763 };
8764 kept.too_many.push(segment_key(&grain("1h")));
8765 kept.too_many.push(segment_key(&by_region));
8766 assert_eq!(kept.segment_count(&grain("1h")), SegmentCount::TooMany);
8767 assert_eq!(kept.segment_count(&by_region), SegmentCount::TooMany);
8768 let error = compute_data_quality_kept(&lf, None, &grain("1h"), None, false, Some(&kept))
8769 .unwrap_err();
8770 assert!(
8771 error.to_string().contains("choose a coarser grain"),
8772 "{error}"
8773 );
8774 }
8775
8776 #[test]
8779 fn a_stopped_stream_is_not_a_sample() {
8780 let watch = crate::sampling::ReadWatch::default();
8781 watch.stop();
8782 let sample = crate::sampling::Sample {
8783 scope: QualityScope::CurrentView,
8784 method: crate::sampling::SampleMethod::PerPartition {
8785 column: "constant".to_string(),
8786 },
8787 rows: 1,
8788 seed: 7,
8789 };
8790 let read =
8791 crate::sampling::read_rows_watched(&fixture(), &sample, None, false, Some(&watch));
8792 let Err(error) = read else {
8793 panic!("a stopped read returned rows");
8794 };
8795 assert_eq!(error.to_string(), crate::sampling::CANCELLED);
8796 }
8797
8798 fn csv_source(rows: usize) -> (tempfile::TempDir, LazyFrame) {
8800 let dir = tempfile::tempdir().unwrap();
8801 let path = dir.path().join("rows.csv");
8802 let ids = (0..rows as i64).collect::<Vec<_>>();
8803 let labels = (0..rows)
8804 .map(|row| if row % 7 == 0 { "b" } else { "a" })
8805 .collect::<Vec<_>>();
8806 let mut df = df!("id" => ids, "label" => labels).unwrap();
8807 CsvWriter::new(std::fs::File::create(&path).unwrap())
8808 .finish(&mut df)
8809 .unwrap();
8810 let lf = LazyCsvReader::new(PlRefPath::try_from_path(&path).unwrap())
8811 .finish()
8812 .unwrap();
8813 (dir, lf)
8814 }
8815
8816 #[cfg(feature = "streaming")]
8819 #[test]
8820 fn a_full_run_stops_inside_its_read() {
8821 const ROWS: usize = 2_000_000;
8822 let (_dir, lf) = csv_source(ROWS);
8823 let plan = DataQualityPlan {
8824 compute: QualityCompute::Full,
8825 ..DataQualityPlan::default()
8826 };
8827 let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8828 let seen = Arc::clone(&stages);
8829 let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8830 let stopper = watch.clone();
8835 let batches = Arc::new(std::sync::atomic::AtomicUsize::new(0));
8836 let lf = lf.map(
8837 move |df: DataFrame| {
8838 if batches.fetch_add(1, std::sync::atomic::Ordering::Relaxed) == 1 {
8839 stopper.cancel();
8840 }
8841 Ok(df)
8842 },
8843 OptFlags::PROJECTION_PUSHDOWN | OptFlags::PREDICATE_PUSHDOWN | OptFlags::STREAMING,
8844 None,
8845 Some("cancel inside the read"),
8846 );
8847 let (results, _) =
8848 compute_data_quality_watched(&lf, Some(ROWS), &plan, None, true, None, &watch);
8849 assert_eq!(results.unwrap_err().to_string(), crate::sampling::CANCELLED);
8850 let stages = stages.lock().unwrap().clone();
8851 let last = stages.last().unwrap();
8852 assert_eq!(last.stage, QualityStage::ProfilingColumns, "{stages:?}");
8853 assert!(last.reads_source && last.interruptible);
8854 let observed = watch.observed();
8855 assert!(
8856 observed.rows < ROWS,
8857 "stopped partway through the first pass: {observed:?}"
8858 );
8859 }
8860
8861 #[cfg(not(feature = "streaming"))]
8864 #[test]
8865 fn without_streaming_no_read_says_it_stops_partway() {
8866 let (_dir, lf) = csv_source(1_000);
8867 for compute in [QualityCompute::Full, QualityCompute::Sample] {
8868 let plan = DataQualityPlan {
8869 compute,
8870 ..DataQualityPlan::default()
8871 };
8872 let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8873 let seen = Arc::clone(&stages);
8874 let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8875 let (results, _) =
8876 compute_data_quality_watched(&lf, Some(1_000), &plan, None, true, None, &watch);
8877 results.unwrap();
8878 let stages = stages.lock().unwrap().clone();
8879 assert!(stages.iter().any(|phase| phase.reads_source), "{stages:?}");
8880 assert!(
8881 stages.iter().all(|phase| !phase.interruptible),
8882 "{stages:?}"
8883 );
8884 }
8885 }
8886
8887 #[test]
8891 fn conflict_examples_give_a_date_past_the_calendar_as_its_stored_number() {
8892 let paris = TimeZone::opt_try_new(Some("Europe/Paris")).unwrap();
8893 let values = move |name: &str| match name {
8894 "d" => Series::new("n".into(), [0, i32::MAX]).cast(&DataType::Date),
8895 "ms" => Series::new("n".into(), [0, i64::MIN + 1])
8896 .cast(&DataType::Datetime(TimeUnit::Milliseconds, None)),
8897 _ => Series::new("n".into(), [0, i64::MIN + 1])
8898 .cast(&DataType::Datetime(TimeUnit::Microseconds, paris.clone())),
8899 };
8900 let scan = QualityConflictScan(Arc::new(move |files, _| {
8901 let n = values(&files[0])?;
8902 Ok(DataFrame::new_infer_height(vec![n.into()])?.lazy())
8903 }));
8904 let mut files: Vec<QualityFileEvidence> = ["d", "ms", "us_tz"]
8905 .into_iter()
8906 .enumerate()
8907 .map(|(i, name)| QualityFileEvidence {
8908 number: i + 1,
8909 name: name.to_string(),
8910 rows: 2,
8911 stored_type: None,
8912 examples: Vec::new(),
8913 })
8914 .collect();
8915 let watch = QualityWatch::new(|_| {});
8916 for streaming in [false, true] {
8917 read_conflict_examples(&scan, "n", &mut files, streaming, &watch);
8918 let examples: Vec<&[String]> = files.iter().map(|f| f.examples.as_slice()).collect();
8919 assert_eq!(
8920 examples,
8921 [
8922 ["1970-01-01", "2147483647 days since 1970-01-01"],
8923 [
8924 "1970-01-01 00:00:00.000",
8925 "-9223372036854775807 ms since 1970-01-01 UTC"
8926 ],
8927 [
8928 "1970-01-01 01:00:00.000000+01:00",
8929 "-9223372036854775807 us since 1970-01-01 UTC"
8930 ],
8931 ]
8932 );
8933 }
8934 }
8935
8936 #[test]
8939 fn the_conflict_read_stops_between_files_on_any_engine() {
8940 let lf = df!("id" => [1i64, 2, 3]).unwrap().lazy();
8941 let source = QualitySourceContext {
8942 conflict_scan: Some(QualityConflictScan(Arc::new(|_, _| {
8943 Ok(df!("id" => ["1"]).unwrap().lazy())
8944 }))),
8945 ..QualitySourceContext::default()
8946 };
8947 let plan = DataQualityPlan {
8948 compute: QualityCompute::Full,
8949 ..DataQualityPlan::default()
8950 };
8951 let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8952 let seen = Arc::clone(&stages);
8953 let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8954 let (results, _) =
8955 compute_data_quality_watched(&lf, Some(3), &plan, Some(&source), false, None, &watch);
8956 results.unwrap();
8957 let stages = stages.lock().unwrap().clone();
8958 let conflicts = stages
8959 .iter()
8960 .find(|phase| phase.stage == QualityStage::ReadingConflicts)
8961 .unwrap_or_else(|| panic!("{stages:?}"));
8962 assert!(conflicts.interruptible, "{stages:?}");
8963 }
8964
8965 #[test]
8968 fn a_run_counts_the_rows_its_reads_traverse() {
8969 let (_dir, lf) = csv_source(10_000);
8970 let full = DataQualityPlan {
8971 compute: QualityCompute::Full,
8972 ..DataQualityPlan::default()
8973 };
8974 let watch = QualityWatch::default();
8975 let (results, _) =
8976 compute_data_quality_watched(&lf, Some(10_000), &full, None, true, None, &watch);
8977 let reads = results.unwrap().reads.unwrap();
8978 assert_eq!(reads.reads, reads.counted, "every pass counted: {reads:?}");
8979 assert!(reads.reads >= 2, "{reads:?}");
8980 assert_eq!(reads.rows % 10_000, 0, "whole passes: {reads:?}");
8981 assert!(reads.rows >= 2 * 10_000, "{reads:?}");
8982
8983 let sampled = DataQualityPlan {
8984 dataset_rows: 100,
8985 ..DataQualityPlan::default()
8986 };
8987 let (results, kept) = compute_data_quality_watched(
8988 &lf,
8989 Some(10_000),
8990 &sampled,
8991 None,
8992 true,
8993 None,
8994 &QualityWatch::default(),
8995 );
8996 assert_eq!(
8997 results.unwrap().reads,
8998 Some(ObservedReads {
8999 reads: 1,
9000 counted: 1,
9001 rows: 10_000,
9002 copy: None,
9003 }),
9004 "the sampler streamed the scope once"
9005 );
9006 let (results, _) = compute_data_quality_watched(
9007 &lf,
9008 Some(10_000),
9009 &sampled,
9010 None,
9011 true,
9012 kept.as_ref(),
9013 &QualityWatch::default(),
9014 );
9015 assert_eq!(results.unwrap().reads, Some(ObservedReads::default()));
9016 }
9017
9018 #[test]
9022 fn a_report_on_wide_text_is_weighed_by_the_text_it_holds() {
9023 let wide = "x".repeat(2_000);
9024 let rows = 600;
9025 let names = (0..rows)
9026 .map(|row| {
9027 let name = format!("Vendor {:04} {wide}", row / 2);
9028 if row % 2 == 0 {
9029 name
9030 } else {
9031 name.to_uppercase()
9032 }
9033 })
9034 .collect::<Vec<_>>();
9035 let df = df!(
9036 "id" => (0..rows as i64).collect::<Vec<_>>(),
9037 "name" => names,
9038 "note" => (0..rows).map(|row| format!("{row} {wide}")).collect::<Vec<_>>(),
9039 )
9040 .unwrap();
9041 for compute in [QualityCompute::Sample, QualityCompute::Full] {
9042 let plan = DataQualityPlan {
9043 compute,
9044 dataset_rows: rows,
9045 ..DataQualityPlan::default()
9046 };
9047 let results =
9048 compute_data_quality(&df.clone().lazy(), None, &plan, None, false).unwrap();
9049 assert_eq!(results.category_variants.len(), 100, "{compute:?}");
9050 let spellings = results
9051 .category_variants
9052 .iter()
9053 .map(|group| {
9054 group.column.len()
9055 + group.normalized.len()
9056 + group
9057 .variants
9058 .iter()
9059 .map(|(variant, _)| variant.len())
9060 .sum::<usize>()
9061 })
9062 .sum::<usize>();
9063 assert!(spellings > 100 * 3 * 2_000, "{spellings}");
9064 let spellings = spellings
9066 + results
9067 .observations
9068 .iter()
9069 .filter_map(|observation| observation.normalized_category.as_ref())
9070 .map(String::len)
9071 .sum::<usize>();
9072 assert!(
9073 results.estimated_bytes() >= spellings,
9074 "{compute:?}: {} bytes budgeted for {spellings} of text",
9075 results.estimated_bytes()
9076 );
9077 for value in results.examples.iter().flat_map(|found| &found.values) {
9078 assert!(crate::glyphs::display_width(value) <= 26, "{value}");
9079 }
9080 }
9081 }
9082
9083 #[test]
9088 fn a_count_past_a_million_keys_gives_up_and_keeps_the_rows() {
9089 let rows = crate::sampling::MAX_COUNTED_KEYS + 1;
9090 let df = df!("id" => (0..rows as i64).collect::<Vec<_>>()).unwrap();
9091 let read = Arc::new(std::sync::atomic::AtomicUsize::new(0));
9092 let counter = Arc::clone(&read);
9093 let lf = df.lazy().filter(col("id").map(
9094 move |column| {
9095 counter.fetch_add(column.len(), std::sync::atomic::Ordering::Relaxed);
9096 Ok(column.is_not_null().into_column())
9097 },
9098 |_, field| Ok(Field::new(field.name().clone(), DataType::Boolean)),
9099 ));
9100 let by_id = DataQualityPlan {
9101 dataset_rows: 1_000,
9102 grain: QualityGrain::Partition("id".into()),
9103 ..DataQualityPlan::default()
9104 };
9105 let watch = QualityWatch::default();
9106 let (results, kept) =
9107 compute_data_quality_watched(&lf, None, &by_id, None, false, None, &watch);
9108 let error = results.unwrap_err().to_string();
9109 assert!(
9110 error.contains("More than 1,000,000 segments") && error.contains("coarser grain"),
9111 "{error}"
9112 );
9113 let kept = kept.expect("the rows the pass read are kept");
9114 assert_eq!(kept.df.height(), 1_000);
9115 assert_eq!(kept.segment_count(&by_id), SegmentCount::TooMany);
9116 assert!(
9117 kept.estimated_bytes() < 1_000_000,
9118 "no map of a million keys kept"
9119 );
9120
9121 read.store(0, std::sync::atomic::Ordering::Relaxed);
9122 let dataset = DataQualityPlan {
9123 grain: QualityGrain::Dataset,
9124 ..by_id.clone()
9125 };
9126 let (results, _) =
9127 compute_data_quality_kept(&lf, None, &dataset, None, false, Some(&kept)).unwrap();
9128 assert_eq!(results.evaluated_rows, 1_000);
9129 assert_eq!(read.load(std::sync::atomic::Ordering::Relaxed), 0);
9130
9131 let head = DataQualityPlan {
9134 method: crate::sampling::SampleMethod::FirstRows,
9135 ..by_id
9136 };
9137 let (results, kept) =
9138 compute_data_quality_watched(&lf, None, &head, None, false, None, &watch);
9139 let error = results.unwrap_err().to_string();
9140 assert!(error.contains("More than 1,000,000 segments"), "{error}");
9141 let kept = kept.expect("the head is kept");
9142 assert_eq!(kept.segment_count(&head), SegmentCount::TooMany);
9143 }
9144}
9145
9146#[cfg(test)]
9148mod temporal_tests {
9149 use super::*;
9150
9151 const HOUR: i64 = 3_600_000_000;
9152
9153 fn datetimes(name: &str, values: &[Option<i64>]) -> Column {
9154 Series::new(name.into(), values)
9155 .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
9156 .unwrap()
9157 .into()
9158 }
9159
9160 fn roles(pairs: &[(TemporalRole, &str)]) -> Vec<TemporalRoleAssignment> {
9161 pairs
9162 .iter()
9163 .map(|(role, column)| TemporalRoleAssignment {
9164 role: *role,
9165 column: column.to_string(),
9166 timezone: None,
9167 })
9168 .collect()
9169 }
9170
9171 fn both_ways(
9173 frame: &LazyFrame,
9174 rows: usize,
9175 plan: &DataQualityPlan,
9176 ) -> [DataQualityResults; 2] {
9177 [QualityCompute::Sample, QualityCompute::Full].map(|compute| {
9178 let plan = DataQualityPlan {
9179 compute,
9180 ..plan.clone()
9181 };
9182 compute_data_quality(frame, Some(rows), &plan, None, false).unwrap()
9183 })
9184 }
9185
9186 fn delays() -> LazyFrame {
9189 let start = [
9190 Some(0),
9191 Some(0),
9192 Some(0),
9193 None,
9194 Some(0),
9195 None,
9196 Some(500_000),
9197 Some(0),
9198 ];
9199 let end = [
9200 Some(HOUR),
9201 Some(HOUR + 1_000_000),
9202 Some(2 * HOUR),
9203 Some(HOUR),
9204 None,
9205 None,
9206 Some(0),
9207 Some(0),
9208 ];
9209 DataFrame::new(8, vec![datetimes("sent", &start), datetimes("seen", &end)])
9210 .unwrap()
9211 .lazy()
9212 }
9213
9214 #[test]
9218 fn breaches_are_out_of_rows_with_both_ends() {
9219 let plan = DataQualityPlan {
9220 temporal_roles: roles(&[
9221 (TemporalRole::Event, "sent"),
9222 (TemporalRole::Received, "seen"),
9223 ]),
9224 latency_threshold_seconds: Some(3_600),
9225 ..DataQualityPlan::default()
9226 };
9227 for results in both_ways(&delays(), 8, &plan) {
9228 let [latency] = results.temporal.as_slice() else {
9229 panic!("one interval: {:?}", results.temporal);
9230 };
9231 assert_eq!(latency.evaluated_rows, 8);
9232 assert_eq!((latency.missing_start, latency.missing_end), (2, 2));
9233 assert_eq!(latency.paired_rows, 5);
9234 assert_ne!(
9235 latency.evaluated_rows - latency.missing_start - latency.missing_end,
9236 latency.paired_rows
9237 );
9238 assert_eq!(latency.threshold_seconds, Some(3_600));
9239 assert_eq!(latency.above_threshold_count, Some(2));
9240 assert_eq!(latency.negative_count, 1);
9242 assert_eq!(latency.zero_count, 1);
9243 assert_eq!(latency.max_seconds, Some(7_200));
9244 assert_eq!(
9245 latency.count(IntervalFact::OverThreshold, &plan),
9246 Some((2, 5))
9247 );
9248 assert_eq!(latency.count(IntervalFact::MissingEnd, &plan), Some((2, 8)));
9249 assert_eq!(latency.count(IntervalFact::UnparsedStart, &plan), None);
9250 }
9251 }
9252
9253 #[test]
9256 fn a_chosen_pair_is_measured_and_an_unpaired_role_is_named() {
9257 let mut plan = DataQualityPlan {
9258 temporal_roles: roles(&[
9259 (TemporalRole::Created, "sent"),
9260 (TemporalRole::Processed, "seen"),
9261 ]),
9262 ..DataQualityPlan::default()
9263 };
9264 assert!(plan.interval_pairs().is_empty(), "no suggested pair");
9265 assert_eq!(
9266 plan.unpaired_roles(),
9267 vec![TemporalRole::Created, TemporalRole::Processed]
9268 );
9269 assert_eq!(
9270 plan.candidate_pairs(),
9271 vec![
9272 (TemporalRole::Created, TemporalRole::Processed),
9273 (TemporalRole::Processed, TemporalRole::Created),
9274 ]
9275 );
9276 plan.toggle_interval((TemporalRole::Created, TemporalRole::Processed));
9277 assert_eq!(
9278 plan.interval_pairs(),
9279 vec![(TemporalRole::Created, TemporalRole::Processed)]
9280 );
9281 assert!(plan.unpaired_roles().is_empty());
9282 for results in both_ways(&delays(), 8, &plan) {
9283 let [latency] = results.temporal.as_slice() else {
9284 panic!("one interval: {:?}", results.temporal);
9285 };
9286 assert_eq!(latency.label(), "created to processed");
9287 assert_eq!(latency.paired_rows, 5);
9288 }
9289
9290 let mut plan = DataQualityPlan {
9292 temporal_roles: roles(&[
9293 (TemporalRole::Event, "sent"),
9294 (TemporalRole::Received, "seen"),
9295 ]),
9296 ..DataQualityPlan::default()
9297 };
9298 plan.toggle_interval((TemporalRole::Event, TemporalRole::Received));
9299 assert_eq!(plan.intervals, Some(Vec::new()));
9300 assert!(plan.interval_pairs().is_empty());
9301 let results = compute_data_quality(&delays(), Some(8), &plan, None, false).unwrap();
9302 assert!(results.temporal.is_empty());
9303 }
9304
9305 #[test]
9308 fn a_validity_period_counts_open_and_backwards_periods() {
9309 let days = |name: &str, values: &[Option<i32>]| -> Column {
9310 Series::new(name.into(), values)
9311 .cast(&DataType::Date)
9312 .unwrap()
9313 .into()
9314 };
9315 let frame = DataFrame::new(
9316 4,
9317 vec![
9318 days(
9319 "from",
9320 &[Some(19_000), Some(19_000), Some(19_010), Some(19_020)],
9321 ),
9322 days("to", &[Some(19_005), None, Some(19_009), Some(19_020)]),
9323 ],
9324 )
9325 .unwrap()
9326 .lazy();
9327 let plan = DataQualityPlan {
9328 temporal_roles: roles(&[
9329 (TemporalRole::ValidFrom, "from"),
9330 (TemporalRole::ValidTo, "to"),
9331 ]),
9332 ..DataQualityPlan::default()
9333 };
9334 for results in both_ways(&frame, 4, &plan) {
9335 let [period] = results.temporal.as_slice() else {
9336 panic!("one interval: {:?}", results.temporal);
9337 };
9338 assert!(period.is_validity());
9339 assert_eq!(period.paired_rows, 3);
9340 assert_eq!(period.missing_end, 1);
9341 assert_eq!(period.negative_count, 1);
9342 assert_eq!(period.zero_count, 1);
9343 assert_eq!(IntervalFact::MissingEnd.label(period), "Open, no end");
9344 assert_eq!(IntervalFact::Negative.label(period), "Ends first");
9345 }
9346 }
9347
9348 #[test]
9351 fn zoned_text_compares_with_naive_time_as_utc() {
9352 let frame = DataFrame::new(
9353 2,
9354 vec![
9355 Column::new(
9356 "stamped".into(),
9357 ["2024-01-01T10:00:00+05:00", "2024-01-01T06:00:00Z"],
9358 ),
9359 datetimes(
9360 "logged",
9361 &[
9362 Some(1_704_088_800_000_000), Some(1_704_088_800_000_000),
9364 ],
9365 ),
9366 ],
9367 )
9368 .unwrap()
9369 .lazy();
9370 let offset = TimeInterpretation {
9371 column: "stamped".to_string(),
9372 kind: TimeKind::Datetime,
9373 format: "%Y-%m-%dT%H:%M:%S%.f%#z".to_string(),
9374 };
9375 assert!(offset.zoned());
9376 assert!(offset.reads("2024-01-01T10:00:00+05:00"));
9377 assert!(offset.reads("2024-01-01T06:00:00Z"));
9378 assert!(
9379 !offset.reads("2024-01-01T06:00:00"),
9380 "no offset, no instant"
9381 );
9382 let plan = DataQualityPlan {
9383 temporal_roles: roles(&[
9384 (TemporalRole::Event, "stamped"),
9385 (TemporalRole::Received, "logged"),
9386 ]),
9387 time_formats: vec![offset],
9388 ..DataQualityPlan::default()
9389 };
9390 let schema = frame.clone().collect_schema().unwrap();
9391 assert_eq!(plan.zoned("stamped", &schema), Some(true));
9392 assert_eq!(plan.zoned("logged", &schema), Some(false));
9393 for results in both_ways(&frame, 2, &plan) {
9394 let [latency] = results.temporal.as_slice() else {
9395 panic!("one interval: {:?}", results.temporal);
9396 };
9397 assert_eq!(latency.paired_rows, 2);
9398 assert_eq!(latency.max_seconds, Some(3_600));
9399 assert_eq!((latency.zero_count, latency.negative_count), (1, 0));
9400 }
9401 }
9402
9403 #[test]
9406 fn the_window_clock_puts_an_interval_on_its_start_or_end() {
9407 let late = 1_704_150_000_000_000; let frame = DataFrame::new(
9409 1,
9410 vec![
9411 datetimes("sent", &[Some(late)]),
9412 datetimes("seen", &[Some(late + 2 * HOUR)]),
9413 ],
9414 )
9415 .unwrap()
9416 .lazy();
9417 let mut plan = DataQualityPlan {
9418 temporal_roles: roles(&[
9419 (TemporalRole::Event, "sent"),
9420 (TemporalRole::Received, "seen"),
9421 ]),
9422 grain: QualityGrain::TimeWindows {
9423 column: "sent".to_string(),
9424 every: "1d".to_string(),
9425 },
9426 ..DataQualityPlan::default()
9427 };
9428 assert!(plan.windows_intervals());
9429 let schema = frame.clone().collect_schema().unwrap();
9430 for (clock, day) in [
9431 (IntervalClock::Grain, "2024-01-01"),
9432 (IntervalClock::Start, "2024-01-01"),
9433 (IntervalClock::End, "2024-01-02"),
9434 ] {
9435 plan.interval_clock = clock;
9436 assert_eq!(interval_passes(&plan, &schema), 1);
9437 for results in both_ways(&frame, 1, &plan) {
9438 let segments = results
9439 .temporal
9440 .iter()
9441 .map(|latency| latency.segment.as_str())
9442 .collect::<Vec<_>>();
9443 assert_eq!(segments, vec![day], "{clock:?}");
9444 }
9445 }
9446 plan.temporal_roles
9449 .extend(roles(&[(TemporalRole::Processed, "seen2")]));
9450 let frame = frame.with_column(col("seen").alias("seen2"));
9451 let schema = frame.clone().collect_schema().unwrap();
9452 assert_eq!(plan.interval_pairs().len(), 3);
9453 assert_eq!(interval_passes(&plan, &schema), 2);
9454 plan.interval_clock = IntervalClock::Grain;
9455 assert_eq!(interval_passes(&plan, &schema), 1);
9456 }
9457
9458 #[test]
9461 fn a_facts_rows_are_the_rows_it_counted() {
9462 let frame = delays().with_column(
9463 when(col("sent").is_null())
9464 .then(lit("2024-01-02"))
9465 .otherwise(lit("2024-01-01"))
9466 .str()
9467 .to_date(StrptimeOptions::default())
9468 .alias("day"),
9469 );
9470 for grain in [
9471 QualityGrain::Dataset,
9472 QualityGrain::Partition("day".to_string()),
9473 QualityGrain::TimeWindows {
9474 column: "day".to_string(),
9475 every: "1d".to_string(),
9476 },
9477 ] {
9478 let plan = DataQualityPlan {
9479 temporal_roles: roles(&[
9480 (TemporalRole::Event, "sent"),
9481 (TemporalRole::Received, "seen"),
9482 ]),
9483 latency_threshold_seconds: Some(3_600),
9484 grain: grain.clone(),
9485 ..DataQualityPlan::default()
9486 };
9487 for results in both_ways(&frame, 8, &plan) {
9488 assert!(!results.temporal.is_empty());
9489 for latency in &results.temporal {
9490 for fact in IntervalFact::ALL {
9491 let Some((count, _)) = latency.count(fact, &plan) else {
9492 assert!(latency.evidence_predicate(fact, &plan, None).is_none());
9493 continue;
9494 };
9495 let predicate = latency
9496 .evidence_predicate(fact, &plan, None)
9497 .unwrap_or_else(|| panic!("{grain:?} {fact:?} opens nothing"));
9498 let rows = frame.clone().filter(predicate).collect().unwrap().height();
9499 assert_eq!(rows, count, "{grain:?} {} {fact:?}", latency.segment);
9500 }
9501 }
9502 }
9503 }
9504 let plan = DataQualityPlan {
9506 temporal_roles: roles(&[
9507 (TemporalRole::Event, "sent"),
9508 (TemporalRole::Received, "seen"),
9509 ]),
9510 grain: QualityGrain::RowChunks(4),
9511 ..DataQualityPlan::default()
9512 };
9513 let results = compute_data_quality(&frame, Some(8), &plan, None, false).unwrap();
9514 assert!(results.temporal.iter().all(|latency| {
9515 latency
9516 .evidence_predicate(IntervalFact::Negative, &plan, None)
9517 .is_none()
9518 }));
9519 }
9520
9521 #[test]
9526 fn a_facts_rows_are_found_by_any_segment_label() {
9527 let spring = 1_710_054_000_000_000i64; let zoned: Column = Series::new(
9529 "zoned".into(),
9530 [0, 3, 20, -1, -10, 40, 0, 960]
9531 .map(|hours| (hours >= 0 || hours == -10).then_some(spring + hours * HOUR)),
9532 )
9533 .cast(&DataType::Datetime(
9534 TimeUnit::Nanoseconds,
9535 TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
9536 ))
9537 .unwrap()
9538 .into();
9539 let mut frame = delays().collect().unwrap();
9540 for column in [
9541 Column::new(
9542 "key=part".into(),
9543 [
9544 Some("a=b"),
9545 Some(" x "),
9546 Some(""),
9547 None,
9548 Some("é"),
9549 Some("a=b"),
9550 Some("1.0"),
9551 Some(" x "),
9552 ],
9553 ),
9554 Column::new("int".into(), [1i64, 2, 3, 1, 2, 3, 1, 2]),
9555 Column::new(
9556 "float".into(),
9557 [0.1f64, 1e20, 2.5, 0.1, 1e20, 2.5, 0.1, 3.0],
9558 ),
9559 Column::new(
9560 "bool".into(),
9561 [true, false, true, false, true, false, true, false],
9562 ),
9563 datetimes("stamp", &[0, HOUR, 0, HOUR, 0, HOUR, 1, 0].map(Some)),
9564 zoned,
9565 ] {
9566 frame.with_column(column).unwrap();
9567 }
9568 let frame = frame.lazy();
9569 let grains = ["key=part", "int", "float", "bool", "stamp"]
9570 .map(|column| QualityGrain::Partition(column.to_string()))
9571 .into_iter()
9572 .chain(
9573 QUALITY_WINDOW_WIDTHS.map(|every| QualityGrain::TimeWindows {
9574 column: "zoned".to_string(),
9575 every: every.to_string(),
9576 }),
9577 );
9578 for grain in grains {
9579 for clock in IntervalClock::ALL {
9580 let plan = DataQualityPlan {
9581 temporal_roles: roles(&[
9582 (TemporalRole::Event, "sent"),
9583 (TemporalRole::Received, "seen"),
9584 ]),
9585 latency_threshold_seconds: Some(3_600),
9586 grain: grain.clone(),
9587 interval_clock: clock,
9588 ..DataQualityPlan::default()
9589 };
9590 for results in both_ways(&frame, 8, &plan) {
9591 for latency in &results.temporal {
9592 for fact in IntervalFact::ALL {
9593 let Some((count, _)) = latency.count(fact, &plan) else {
9594 continue;
9595 };
9596 let predicate = latency.evidence_predicate(fact, &plan, None).unwrap();
9597 let rows = frame.clone().filter(predicate).collect().unwrap();
9598 assert_eq!(
9599 rows.height(),
9600 count,
9601 "{grain:?} {clock:?} {} {fact:?}",
9602 latency.segment
9603 );
9604 }
9605 }
9606 }
9607 }
9608 }
9609 }
9610}