1use crate::analysis::data_quality::{
8 ColumnQualityProfile, DataQualityPlan, DataQualityResults, ObservationKind, QualityObservation,
9 QualityPrecision, QualityScope, TextReading, text_reading,
10};
11use crate::numfmt;
12use polars::prelude::Expr;
13use std::collections::BTreeMap;
14
15#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
16pub enum Severity {
17 Problem,
19 Note,
21 Clean,
23}
24
25impl Severity {
26 pub fn heading(self) -> &'static str {
27 match self {
28 Self::Problem => "Problems",
29 Self::Note => "Notes",
30 Self::Clean => "Clean",
31 }
32 }
33}
34
35#[derive(Debug, Clone, Copy, PartialEq, Eq)]
38pub enum Variant {
39 AlwaysMissing,
41 MostlyMissing,
43 MissingTogether,
45 CodesAsText,
47 DatesAsText,
48}
49
50impl Variant {
51 pub fn title(self) -> &'static str {
52 match self {
53 Self::AlwaysMissing => "Always missing",
54 Self::MostlyMissing => "Mostly missing",
55 Self::MissingTogether => "Missing together",
57 Self::CodesAsText => "Codes as text",
58 Self::DatesAsText => "Dates as text",
59 }
60 }
61
62 fn severity(self) -> Option<Severity> {
63 match self {
64 Self::AlwaysMissing => Some(Severity::Problem),
65 Self::MostlyMissing | Self::MissingTogether | Self::CodesAsText | Self::DatesAsText => {
66 None
67 }
68 }
69 }
70
71 fn rank(self) -> Option<u8> {
72 match self {
73 Self::AlwaysMissing => Some(3),
74 Self::MostlyMissing => Some(9),
75 Self::MissingTogether | Self::CodesAsText | Self::DatesAsText => None,
76 }
77 }
78
79 fn advice(self) -> Option<&'static [&'static str]> {
80 match self {
81 Self::AlwaysMissing => Some(&["Carries nothing; check the load or a rename upstream"]),
82 Self::MostlyMissing => None,
84 Self::MissingTogether => {
85 Some(&["Likely one cause: a join with no match, or a source with gaps"])
86 }
87 Self::CodesAsText => Some(&["Fine as text; cast only for arithmetic"]),
88 Self::DatesAsText => Some(&["Sorts as text; parse as a date to filter by range"]),
89 }
90 }
91}
92
93#[derive(Debug, Clone)]
94pub struct Finding {
95 pub severity: Severity,
96 pub kind: Option<ObservationKind>,
98 pub variant: Option<Variant>,
99 pub title: &'static str,
101 pub columns: Vec<String>,
102 pub observations: Vec<usize>,
104 pub affected_rows: usize,
106 pub evaluated_rows: usize,
107 pub summary: String,
109 pub same_rows: bool,
111}
112
113pub const NO_FINDINGS: &str = "No findings";
115
116impl Finding {
117 pub fn same_as(&self, other: &Finding) -> bool {
119 self.kind == other.kind && self.variant == other.variant && self.columns == other.columns
120 }
121
122 pub fn columns_label(&self, width: usize) -> String {
124 columns_label(&self.columns, width)
125 }
126
127 pub fn evidence_predicate(&self, results: &DataQualityResults) -> Option<Expr> {
130 self.observations
131 .iter()
132 .map(|index| {
133 results
134 .observations
135 .get(*index)?
136 .evidence_predicate(results)
137 })
138 .collect::<Option<Vec<_>>>()?
139 .into_iter()
140 .reduce(Expr::or)
141 }
142
143 pub fn evidence_scope(&self, results: &DataQualityResults) -> Option<QualityScope> {
145 match self.observations.as_slice() {
146 [index] => results.observations.get(*index)?.evidence_scope(),
147 _ => None,
148 }
149 }
150
151 pub fn evidence(&self, results: &DataQualityResults) -> Result<EvidenceRows, String> {
153 if let Some(scope) = self.evidence_scope(results) {
154 return Ok(EvidenceRows::Files(scope));
155 }
156 let Some(kind) = self.kind else {
157 return Err("No rows: every column here passed".to_string());
158 };
159 if results.precision == QualityPrecision::Metadata {
160 return Err("No rows: values were not read".to_string());
161 }
162 match kind.spec().evidence {
163 Evidence::Duplicates => return Ok(EvidenceRows::Duplicates),
164 Evidence::Failures if self.failures(results) == Some(0) => {
165 return Err("No rows: every value parses".to_string());
166 }
167 Evidence::Rows
168 | Evidence::Always
169 | Evidence::Failures
170 | Evidence::SharingValue
171 | Evidence::Uncounted => {}
172 }
173 self.evidence_predicate(results)
174 .map(EvidenceRows::Matching)
175 .ok_or_else(|| "No rows: nothing to filter on".to_string())
176 }
177
178 pub fn evidence_count(&self, results: &DataQualityResults) -> Option<usize> {
182 match self.kind?.spec().evidence {
183 Evidence::Duplicates => {
184 Some(results.identity.as_ref()?.rows_involved).filter(|rows| *rows > 0)
185 }
186 Evidence::Failures => self.failures(results),
187 Evidence::SharingValue | Evidence::Uncounted => None,
188 Evidence::Always => Some(self.affected_rows),
189 Evidence::Rows => {
190 (self.observations.len() == 1 || self.same_rows).then_some(self.affected_rows)
191 }
192 }
193 }
194
195 pub fn failures(&self, results: &DataQualityResults) -> Option<usize> {
197 if self.kind?.spec().evidence != Evidence::Failures {
198 return None;
199 }
200 let profile = results
201 .columns
202 .iter()
203 .find(|profile| Some(&profile.name) == self.columns.first())?;
204 let (parsed, _) = text_reading(profile)?;
205 Some(profile.non_null_rows().saturating_sub(parsed))
206 }
207
208 pub fn check(&self) -> Option<&'static str> {
211 Some(self.kind?.spec().check)
212 }
213
214 pub fn varied(&self) -> bool {
216 self.lists_columns()
217 && self.kind.map(|kind| kind.spec().grouping) == Some(Grouping::Missing)
218 && self.severity == Severity::Note
219 }
220
221 pub fn lists_columns(&self) -> bool {
224 self.kind.is_some_and(|kind| {
225 matches!(kind.spec().grouping, Grouping::Missing | Grouping::ByRows)
226 }) && self.columns.len() > 1
227 && !self.same_rows
228 }
229
230 pub fn breakdown(&self, results: &DataQualityResults) -> Vec<String> {
233 let rows = self
234 .observations
235 .iter()
236 .filter_map(|index| results.observations.get(*index))
237 .collect::<Vec<_>>();
238 let name_width = rows
239 .iter()
240 .map(|row| crate::glyphs::display_width(&row.column))
241 .max()
242 .unwrap_or(0)
243 .min(28);
244 let plural = |count: usize| if count == 1 { "row" } else { "rows" };
245 let mut lines = rows
246 .iter()
247 .map(|row| {
248 let name = columns_label(std::slice::from_ref(&row.column), name_width);
249 let pad = name_width.saturating_sub(crate::glyphs::display_width(&name));
250 format!(
251 "{name}{} {:>7} {} {}",
252 " ".repeat(pad),
253 crate::numfmt::percent_of(row.affected_rows, row.evaluated_rows),
254 numfmt::group_chrome(row.affected_rows),
255 plural(row.affected_rows)
256 )
257 })
258 .collect::<Vec<_>>();
259 let least = rows.iter().map(|row| row.affected_rows).max().unwrap_or(0);
260 let most = rows
261 .iter()
262 .map(|row| row.affected_rows)
263 .sum::<usize>()
264 .min(self.evaluated_rows.max(least));
265 lines.push(if least == most {
266 format!(
267 "Rows with any of them: {} {}",
268 numfmt::group_chrome(least),
269 plural(least)
270 )
271 } else {
272 format!(
273 "Rows with any of them: {} to {}, not counted",
274 numfmt::group_chrome(least),
275 numfmt::group_chrome(most)
276 )
277 });
278 lines
279 }
280}
281
282#[derive(Debug, Clone)]
284pub enum EvidenceRows {
285 Matching(Expr),
287 Files(QualityScope),
289 Duplicates,
292}
293
294#[derive(Debug, Clone, Copy, PartialEq, Eq)]
296pub enum Grouping {
297 Alone,
299 All,
301 ByRows,
303 ByColumn,
305 Missing,
308}
309
310#[derive(Debug, Clone, Copy, PartialEq, Eq)]
312pub enum Evidence {
313 Rows,
316 Always,
319 Duplicates,
321 Failures,
323 SharingValue,
326 Uncounted,
328}
329
330pub struct CheckSpec {
333 pub check: &'static str,
335 pub title: &'static str,
337 pub severity: Severity,
338 pub rank: u8,
340 pub advice: &'static [&'static str],
343 pub grouping: Grouping,
344 pub evidence: Evidence,
345 pub rows_are: &'static str,
347 pub noun: &'static str,
349 variant: fn(&Group<'_>) -> Option<Variant>,
351 summary: fn(&Group<'_>) -> String,
353 headline: fn(&Detail<'_>, &mut Vec<String>) -> String,
355}
356
357impl ObservationKind {
358 pub fn spec(self) -> &'static CheckSpec {
360 match self {
361 Self::Nulls => &CheckSpec {
362 check: "Missing values",
363 title: "Missing values",
364 severity: Severity::Note,
365 rank: 10,
366 advice: &["Check: clustered in some files or dates (Segments, by file or window)"],
367 grouping: Grouping::Missing,
368 evidence: Evidence::Rows,
369 rows_are: "",
370 noun: "null",
371 variant: missing_variant,
372 summary: nulls_summary,
373 headline: nulls_headline,
374 },
375 Self::Empty => &CheckSpec {
376 check: "Blank text",
377 title: "Empty text",
378 severity: Severity::Problem,
379 rank: 7,
380 advice: &["Counted as filled; treat as null if it means missing"],
381 grouping: Grouping::ByRows,
382 evidence: Evidence::Rows,
383 rows_are: "",
384 noun: "empty strings",
385 variant: no_variant,
386 summary: rows_each_summary,
387 headline: per_column_headline,
388 },
389 Self::Whitespace => &CheckSpec {
390 check: "Blank text",
391 title: "Blank text",
392 severity: Severity::Problem,
393 rank: 6,
394 advice: &["Counted as filled; trim to null if it means missing"],
395 grouping: Grouping::ByRows,
396 evidence: Evidence::Rows,
397 rows_are: "",
398 noun: "only spaces or tabs",
399 variant: no_variant,
400 summary: rows_each_summary,
401 headline: per_column_headline,
402 },
403 Self::NonFinite => &CheckSpec {
404 check: "NaN or infinite",
405 title: "NaN or infinite",
406 severity: Severity::Problem,
407 rank: 4,
408 advice: &["Sums and means become NaN; check for division by zero upstream"],
409 grouping: Grouping::Alone,
410 evidence: Evidence::Rows,
411 rows_are: "",
412 noun: "NaN or infinite",
413 variant: no_variant,
414 summary: rows_summary,
415 headline: non_finite_headline,
416 },
417 Self::Constant => &CheckSpec {
418 check: "Single value",
419 title: "Single value",
420 severity: Severity::Note,
421 rank: 13,
422 advice: &["Tells no rows apart; a stuck feed if it should vary"],
423 grouping: Grouping::All,
424 evidence: Evidence::Rows,
425 rows_are: "",
426 noun: "",
427 variant: no_variant,
428 summary: constant_summary,
429 headline: constant_headline,
430 },
431 Self::ParseableText => &CheckSpec {
432 check: "Numbers as text",
433 title: "Numbers as text",
434 severity: Severity::Note,
435 rank: 12,
436 advice: &["Sorts as text (\"10\" before \"9\"); cast to a number to sum"],
437 grouping: Grouping::Alone,
438 evidence: Evidence::Failures,
439 rows_are: " that do not parse",
440 noun: "",
441 variant: reading_variant,
442 summary: parseable_summary,
443 headline: parseable_headline,
444 },
445 Self::DuplicateRows => &CheckSpec {
446 check: "Duplicate rows",
447 title: "Duplicate rows",
448 severity: Severity::Problem,
449 rank: 2,
450 advice: &["Counted more than once; check for a double load or a join fan-out"],
451 grouping: Grouping::Alone,
452 evidence: Evidence::Duplicates,
453 rows_are: " that have a copy",
454 noun: "",
455 variant: no_variant,
456 summary: duplicates_summary,
457 headline: duplicates_headline,
458 },
459 Self::CategoryVariants => &CheckSpec {
460 check: "Mixed spellings",
461 title: "Mixed spellings",
462 severity: Severity::Problem,
463 rank: 5,
464 advice: &["Group-bys and joins split them; trim and normalize case"],
465 grouping: Grouping::ByColumn,
466 evidence: Evidence::Always,
467 rows_are: "",
468 noun: "",
469 variant: no_variant,
470 summary: variants_summary,
471 headline: variants_headline,
472 },
473 Self::Absent => &CheckSpec {
474 check: "Missing in files",
475 title: "Missing in files",
476 severity: Severity::Problem,
477 rank: 1,
478 advice: &["Check: files written before the column existed"],
479 grouping: Grouping::Alone,
480 evidence: Evidence::Uncounted,
481 rows_are: "",
482 noun: "",
483 variant: no_variant,
484 summary: fact_summary,
485 headline: files_headline,
486 },
487 Self::TypeConflict => &CheckSpec {
488 check: "Type mismatch",
489 title: "Type mismatch",
490 severity: Severity::Problem,
491 rank: 0,
492 advice: &["Values dropped, not converted; read as text in Info, or fix the writer"],
493 grouping: Grouping::Alone,
494 evidence: Evidence::Uncounted,
495 rows_are: "",
496 noun: "",
497 variant: no_variant,
498 summary: fact_summary,
499 headline: files_headline,
500 },
501 Self::KeyLike => &CheckSpec {
502 check: "Nearly unique",
503 title: "Nearly unique",
504 severity: Severity::Note,
505 rank: 11,
506 advice: &["Duplicates if it is a key; expected if it is a measurement"],
507 grouping: Grouping::Alone,
508 evidence: Evidence::SharingValue,
509 rows_are: "",
510 noun: "",
511 variant: no_variant,
512 summary: key_like_summary,
513 headline: key_like_headline,
514 },
515 Self::UnparsedTime => &CheckSpec {
516 check: "Unparsed times",
517 title: "Unparsed times",
518 severity: Severity::Problem,
519 rank: 3,
520 advice: &[
521 "Left out of time windows and intervals, not counted as missing",
522 "Check: another format, or values that are not times (Setup, e)",
523 ],
524 grouping: Grouping::Alone,
525 evidence: Evidence::Rows,
526 rows_are: "",
527 noun: "",
528 variant: no_variant,
529 summary: values_summary,
530 headline: unparsed_time_headline,
531 },
532 Self::KeyRepeated => &CheckSpec {
533 check: INTENT_CHECK,
534 title: "Repeated key",
535 severity: Severity::Problem,
536 rank: 2,
537 advice: &[
538 "A key names one row; joins on it fan out and counts double",
539 "Check: a double load, or a key that needs another column",
540 ],
541 grouping: Grouping::All,
542 evidence: Evidence::Always,
543 rows_are: "",
544 noun: "",
545 variant: no_variant,
546 summary: key_repeated_summary,
547 headline: key_repeated_headline,
548 },
549 Self::KeyMissing => &CheckSpec {
550 check: INTENT_CHECK,
551 title: "Incomplete key",
552 severity: Severity::Problem,
553 rank: 2,
554 advice: &["Rows with no key cannot be joined or told apart by it"],
555 grouping: Grouping::All,
556 evidence: Evidence::Always,
557 rows_are: "",
558 noun: "",
559 variant: no_variant,
560 summary: rows_summary,
561 headline: key_missing_headline,
562 },
563 Self::RequiredMissing => &CheckSpec {
564 check: INTENT_CHECK,
565 title: "Required, missing",
566 severity: Severity::Problem,
567 rank: 2,
568 advice: &["Check: the load, or rows the source writes without it"],
569 grouping: Grouping::Alone,
570 evidence: Evidence::Rows,
571 rows_are: "",
572 noun: "",
573 variant: no_variant,
574 summary: rows_summary,
575 headline: required_headline,
576 },
577 Self::NotAllowed => &CheckSpec {
578 check: INTENT_CHECK,
579 title: "Not allowed",
580 severity: Severity::Problem,
581 rank: 2,
582 advice: &["Check: a new value upstream, or the allowed list (Setup, e)"],
583 grouping: Grouping::Alone,
584 evidence: Evidence::Rows,
585 rows_are: "",
586 noun: "",
587 variant: no_variant,
588 summary: values_summary,
589 headline: not_allowed_headline,
590 },
591 Self::OutOfRange => &CheckSpec {
592 check: INTENT_CHECK,
593 title: "Out of range",
594 severity: Severity::Problem,
595 rank: 2,
596 advice: &["Check: units, placeholders such as -1 or 9999, or the range (Setup, e)"],
597 grouping: Grouping::Alone,
598 evidence: Evidence::Rows,
599 rows_are: "",
600 noun: "",
601 variant: no_variant,
602 summary: values_summary,
603 headline: out_of_range_headline,
604 },
605 Self::UnparsedNumber => &CheckSpec {
606 check: INTENT_CHECK,
607 title: "Unparsed numbers",
608 severity: Severity::Problem,
609 rank: 2,
610 advice: &["A cast makes them null; check the values or the reading (Setup, e)"],
611 grouping: Grouping::Alone,
612 evidence: Evidence::Rows,
613 rows_are: "",
614 noun: "",
615 variant: no_variant,
616 summary: values_summary,
617 headline: unparsed_number_headline,
618 },
619 Self::Clipping => &CheckSpec {
620 check: "Clipping",
621 title: "Clipping",
622 severity: Severity::Problem,
623 rank: 4,
624 advice: &[
625 "Cut flat at the limit; lower the gain at the source, nothing restores it",
626 ],
627 grouping: Grouping::Alone,
628 evidence: Evidence::Rows,
629 rows_are: "",
630 noun: "in runs at full scale",
631 variant: no_variant,
632 summary: fact_summary,
633 headline: signal_headline,
634 },
635 Self::ZeroRuns => &CheckSpec {
636 check: "Runs of zeros",
637 title: "Runs of zeros",
638 severity: Severity::Note,
639 rank: 8,
640 advice: &["Dropouts mid-recording; digital silence if at the start or end"],
641 grouping: Grouping::Alone,
642 evidence: Evidence::Rows,
643 rows_are: "",
644 noun: "in runs of exact zeros",
645 variant: no_variant,
646 summary: fact_summary,
647 headline: signal_headline,
648 },
649 Self::DcOffset => &CheckSpec {
650 check: "DC offset",
651 title: "DC offset",
652 severity: Severity::Note,
653 rank: 12,
654 advice: &["A constant bias that eats headroom; a high-pass filter removes it"],
655 grouping: Grouping::Alone,
656 evidence: Evidence::Uncounted,
657 rows_are: "",
658 noun: "",
659 variant: no_variant,
660 summary: fact_summary,
661 headline: dc_offset_headline,
662 },
663 }
664 }
665}
666
667#[derive(Debug, Clone, Default, PartialEq, Eq)]
670pub struct FindingsView {
671 pub column: Option<String>,
673 pub check: Option<&'static str>,
675 pub order: FindingOrder,
676}
677
678#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
680pub enum FindingOrder {
681 #[default]
683 Ranked,
684 Rows,
686 Rate,
688}
689
690impl FindingOrder {
691 pub fn next(self) -> Self {
692 match self {
693 Self::Ranked => Self::Rows,
694 Self::Rows => Self::Rate,
695 Self::Rate => Self::Ranked,
696 }
697 }
698
699 pub fn chip(self) -> &'static str {
701 match self {
702 Self::Ranked => "Ranked",
703 Self::Rows => "By rows",
704 Self::Rate => "By rate",
705 }
706 }
707}
708
709impl FindingsView {
710 pub fn narrowed(&self) -> bool {
711 self.column.is_some() || self.check.is_some()
712 }
713
714 fn admits(&self, finding: &Finding) -> bool {
715 self.column
716 .as_ref()
717 .is_none_or(|column| finding.columns.contains(column))
718 && self
719 .check
720 .is_none_or(|check| finding.check() == Some(check))
721 }
722
723 pub fn shown(&self, report: &QualityReport) -> Vec<usize> {
726 let findings = &report.findings;
727 let mut shown = (0..findings.len())
728 .filter(|index| self.admits(&findings[*index]))
729 .collect::<Vec<_>>();
730 let by = |left: &usize, right: &usize| {
731 let (left, right) = (&findings[*left], &findings[*right]);
732 left.severity
733 .cmp(&right.severity)
734 .then_with(|| match self.order {
735 FindingOrder::Ranked => std::cmp::Ordering::Equal,
736 FindingOrder::Rows => right.affected_rows.cmp(&left.affected_rows),
737 FindingOrder::Rate => (right.affected_rows as u128
739 * left.evaluated_rows.max(1) as u128)
740 .cmp(&(left.affected_rows as u128 * right.evaluated_rows.max(1) as u128)),
741 })
742 };
743 shown.sort_by(by);
744 shown
745 }
746
747 pub fn selected<'a>(&self, report: &'a QualityReport, position: usize) -> Option<&'a Finding> {
749 self.shown(report)
750 .get(position)
751 .and_then(|index| report.findings.get(*index))
752 }
753
754 pub fn describe(&self) -> Vec<String> {
757 let mut parts = Vec::new();
758 if let Some(column) = &self.column {
759 parts.push(format!("column {column}"));
760 }
761 if let Some(check) = self.check {
762 parts.push(check.to_string());
763 }
764 match self.order {
765 FindingOrder::Ranked => {}
766 FindingOrder::Rows => parts.push("by rows".to_string()),
767 FindingOrder::Rate => parts.push("by rate".to_string()),
768 }
769 parts
770 }
771}
772
773pub fn column_choices(
776 report: &QualityReport,
777 results: &DataQualityResults,
778) -> Vec<(String, usize)> {
779 results
780 .columns
781 .iter()
782 .map(|profile| {
783 let count = report
784 .findings
785 .iter()
786 .filter(|finding| finding.kind.is_some() && finding.columns.contains(&profile.name))
787 .count();
788 (profile.name.clone(), count)
789 })
790 .collect()
791}
792
793pub fn check_choices(report: &QualityReport) -> Vec<(&'static str, usize)> {
796 let mut choices: Vec<(&'static str, usize)> = Vec::new();
797 for check in report.findings.iter().filter_map(Finding::check) {
798 match choices.iter_mut().find(|(name, _)| *name == check) {
799 Some((_, count)) => *count += 1,
800 None => choices.push((check, 1)),
801 }
802 }
803 choices
804}
805
806#[derive(Debug, Clone, Default)]
807pub struct QualityReport {
808 pub findings: Vec<Finding>,
809 pub problems: usize,
810 pub notes: usize,
811 pub clean_columns: usize,
812 pub total_columns: usize,
813 pub column_status: Vec<Severity>,
815 pub column_findings: Vec<Vec<&'static str>>,
817 pub metadata_only: bool,
819 pub no_rows: bool,
822}
823
824#[derive(Debug, Default)]
828pub struct ReportCache(std::sync::OnceLock<Built>);
829
830impl Clone for ReportCache {
831 fn clone(&self) -> Self {
832 Self::default()
833 }
834}
835
836#[derive(Debug)]
837struct Built {
838 report: QualityReport,
839 checks: Vec<Check>,
840}
841
842#[cfg(test)]
843thread_local! {
844 pub(crate) static REPORTS_BUILT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
846}
847
848impl ReportCache {
849 fn get(&self, results: &DataQualityResults) -> &Built {
850 self.0.get_or_init(|| {
851 #[cfg(test)]
852 REPORTS_BUILT.with(|built| built.set(built.get() + 1));
853 let report = build_report(results);
854 let checks = checks(results, &report);
855 Built { report, checks }
856 })
857 }
858}
859
860impl DataQualityResults {
861 pub fn edit<R>(&mut self, edit: impl FnOnce(&mut Self) -> R) -> R {
864 let out = edit(self);
865 self.derived = ReportCache::default();
866 out
867 }
868
869 pub fn report(&self) -> &QualityReport {
871 &self.derived.get(self).report
872 }
873
874 pub fn checks(&self) -> &[Check] {
876 &self.derived.get(self).checks
877 }
878}
879
880pub fn build_report(results: &DataQualityResults) -> QualityReport {
881 let mut findings = Vec::new();
882 let mut groups = BTreeMap::<(u8, usize, String), Vec<usize>>::new();
884 for (index, observation) in results.observations.iter().enumerate() {
885 let kind = observation.kind as u8 + 1;
886 let key = match observation.kind.spec().grouping {
887 Grouping::Missing => {
888 let always_missing = observation.evaluated_rows > 0
889 && observation.affected_rows == observation.evaluated_rows;
890 let mostly_missing = observation.affected_rows * 2 > observation.evaluated_rows;
891 let same_rows = results.shared_nulls.iter().any(|shared| {
895 shared.same_rows()
896 && shared.columns.len() > 1
897 && shared.null_rows == observation.affected_rows
898 && shared.columns.contains(&observation.column)
899 });
900 if always_missing {
901 (0, 0, String::new())
902 } else if mostly_missing {
903 (kind, usize::MAX, "mostly".to_string())
904 } else if same_rows {
905 (kind, observation.affected_rows, String::new())
906 } else {
907 (kind, usize::MAX, String::new())
908 }
909 }
910 Grouping::ByRows => (kind, observation.affected_rows, String::new()),
911 Grouping::All => (kind, 0, String::new()),
912 Grouping::ByColumn => (kind, 0, observation.column.clone()),
913 Grouping::Alone => (u8::MAX, index, String::new()),
915 };
916 groups.entry(key).or_default().push(index);
917 }
918 for indices in groups.into_values() {
919 findings.push(finding(results, &indices));
920 }
921
922 let mut status = vec![Severity::Clean; results.columns.len()];
923 let mut titles = vec![Vec::new(); results.columns.len()];
924 let position = results
925 .columns
926 .iter()
927 .enumerate()
928 .map(|(index, profile)| (profile.name.as_str(), index))
929 .collect::<BTreeMap<_, _>>();
930 for finding in &findings {
931 for column in &finding.columns {
932 if let Some(&index) = position.get(column.as_str()) {
933 status[index] = status[index].min(finding.severity);
934 if !titles[index].contains(&finding.title) {
935 titles[index].push(finding.title);
936 }
937 }
938 }
939 }
940
941 findings.sort_by(|left, right| {
942 left.severity
943 .cmp(&right.severity)
944 .then_with(|| rank(left).cmp(&rank(right)))
945 .then_with(|| right.affected_rows.cmp(&left.affected_rows))
946 .then_with(|| left.columns.cmp(&right.columns))
947 });
948 let problems = findings
949 .iter()
950 .filter(|finding| finding.severity == Severity::Problem)
951 .count();
952 let notes = findings.len() - problems;
953 let metadata_only = results.precision == QualityPrecision::Metadata;
954 let no_rows = !metadata_only && results.evaluated_rows == 0;
955 let clean = results
956 .columns
957 .iter()
958 .zip(&status)
959 .filter(|(_, severity)| **severity == Severity::Clean)
960 .map(|(profile, _)| profile.name.clone())
961 .collect::<Vec<_>>();
962 let clean_columns = if no_rows { 0 } else { clean.len() };
963 if !clean.is_empty() && !metadata_only && !no_rows {
964 findings.push(Finding {
965 severity: Severity::Clean,
966 kind: None,
967 variant: None,
968 title: NO_FINDINGS,
969 summary: format!(
970 "{} {}",
971 numfmt::group_chrome(clean.len()),
972 if clean.len() == 1 {
973 "column"
974 } else {
975 "columns"
976 }
977 ),
978 columns: clean,
979 observations: Vec::new(),
980 affected_rows: 0,
981 evaluated_rows: results.evaluated_rows,
982 same_rows: false,
983 });
984 }
985 QualityReport {
986 findings,
987 problems,
988 notes,
989 clean_columns,
990 total_columns: results.columns.len(),
991 column_status: status,
992 column_findings: titles,
993 metadata_only,
994 no_rows,
995 }
996}
997
998fn rank(finding: &Finding) -> u8 {
1000 match (finding.variant.and_then(Variant::rank), finding.kind) {
1001 (Some(rank), _) => rank,
1002 (None, Some(kind)) => kind.spec().rank,
1003 (None, None) => 20,
1004 }
1005}
1006
1007struct Group<'a> {
1009 results: &'a DataQualityResults,
1010 indices: &'a [usize],
1011 first: &'a QualityObservation,
1012 profile: Option<&'a ColumnQualityProfile>,
1013 reading: Option<TextReading>,
1014 grouped: bool,
1016 same_rows: bool,
1017 variant: Option<Variant>,
1018}
1019
1020impl Group<'_> {
1021 fn rows_each(&self, count: usize, each: &str) -> String {
1023 format!(
1024 "{} {}{each} ({})",
1025 numfmt::group_chrome(count),
1026 if count == 1 { "row" } else { "rows" },
1027 crate::numfmt::percent_of(count, self.first.evaluated_rows)
1028 )
1029 }
1030
1031 fn rows(&self) -> String {
1032 self.rows_each(self.first.affected_rows, "")
1033 }
1034}
1035
1036fn finding(results: &DataQualityResults, indices: &[usize]) -> Finding {
1037 let mut indices = indices.to_vec();
1038 let spec = results.observations[indices[0]].kind.spec();
1039 if matches!(spec.grouping, Grouping::Missing | Grouping::ByColumn) {
1042 indices.sort_by_key(|index| std::cmp::Reverse(results.observations[*index].affected_rows));
1043 }
1044 let indices = indices.as_slice();
1045 let first = &results.observations[indices[0]];
1046 let mut columns = Vec::new();
1047 for index in indices {
1048 let column = &results.observations[*index].column;
1049 if !columns.contains(column) {
1050 columns.push(column.clone());
1051 }
1052 }
1053 let grouped = columns.len() > 1;
1054 let profile = results
1055 .columns
1056 .iter()
1057 .find(|profile| profile.name == first.column);
1058 let shared = results.shared_nulls.iter().find(|shared| {
1059 shared.null_rows == first.affected_rows
1060 && columns.iter().all(|column| shared.columns.contains(column))
1061 });
1062 let same_rows =
1063 spec.grouping == Grouping::Missing && grouped && shared.is_some_and(|s| s.same_rows());
1064 let mut group = Group {
1065 results,
1066 indices,
1067 first,
1068 profile,
1069 reading: profile.and_then(text_reading).map(|(_, reading)| reading),
1070 grouped,
1071 same_rows,
1072 variant: None,
1073 };
1074 group.variant = (spec.variant)(&group);
1075 let affected_rows = match spec.grouping {
1076 Grouping::ByColumn => indices
1078 .iter()
1079 .map(|index| results.observations[*index].affected_rows)
1080 .sum(),
1081 Grouping::Alone | Grouping::All | Grouping::ByRows | Grouping::Missing => {
1082 first.affected_rows
1083 }
1084 };
1085 Finding {
1086 severity: group
1087 .variant
1088 .and_then(Variant::severity)
1089 .unwrap_or(spec.severity),
1090 kind: Some(first.kind),
1091 variant: group.variant,
1092 title: group.variant.map_or(spec.title, Variant::title),
1093 columns,
1094 observations: indices.to_vec(),
1095 affected_rows,
1096 evaluated_rows: first.evaluated_rows,
1097 summary: (spec.summary)(&group),
1098 same_rows,
1099 }
1100}
1101
1102fn no_variant(_: &Group<'_>) -> Option<Variant> {
1103 None
1104}
1105
1106fn missing_variant(group: &Group<'_>) -> Option<Variant> {
1107 let first = group.first;
1108 if first.evaluated_rows > 0 && first.affected_rows == first.evaluated_rows {
1109 Some(Variant::AlwaysMissing)
1110 } else if first.affected_rows * 2 > first.evaluated_rows {
1111 Some(Variant::MostlyMissing)
1112 } else if group.same_rows {
1113 Some(Variant::MissingTogether)
1114 } else {
1115 None
1116 }
1117}
1118
1119fn reading_variant(group: &Group<'_>) -> Option<Variant> {
1120 match (group.reading, group.profile) {
1121 (Some(reading), Some(profile)) if reading.is_number() && is_code(profile) => {
1122 Some(Variant::CodesAsText)
1123 }
1124 (Some(TextReading::Datetime | TextReading::Date), _) => Some(Variant::DatesAsText),
1125 (Some(TextReading::WholeNumber | TextReading::Decimal) | None, _) => None,
1126 }
1127}
1128
1129fn nulls_summary(group: &Group<'_>) -> String {
1130 let first = group.first;
1131 if group.variant == Some(Variant::AlwaysMissing) {
1132 "no value in any row".to_string()
1133 } else if group.same_rows {
1134 group.rows()
1135 } else if group.grouped {
1136 let fewest =
1137 group.results.observations[group.indices[group.indices.len() - 1]].affected_rows;
1138 let (low, high) = (
1139 crate::numfmt::percent_of(fewest, first.evaluated_rows),
1140 crate::numfmt::percent_of(first.affected_rows, first.evaluated_rows),
1141 );
1142 if low == high {
1143 group.rows_each(first.affected_rows, " each")
1144 } else {
1145 format!("{low} to {high} per column")
1146 }
1147 } else {
1148 group.rows()
1149 }
1150}
1151
1152fn rows_each_summary(group: &Group<'_>) -> String {
1153 if group.grouped {
1154 group.rows_each(group.first.affected_rows, " each")
1155 } else {
1156 group.rows()
1157 }
1158}
1159
1160fn rows_summary(group: &Group<'_>) -> String {
1161 group.rows()
1162}
1163
1164fn values_summary(group: &Group<'_>) -> String {
1166 let first = group.first;
1167 format!(
1168 "{} {} ({})",
1169 numfmt::group_chrome(first.affected_rows),
1170 if first.affected_rows == 1 {
1171 "value"
1172 } else {
1173 "values"
1174 },
1175 crate::numfmt::percent_of(first.affected_rows, first.evaluated_rows)
1176 )
1177}
1178
1179fn fact_summary(group: &Group<'_>) -> String {
1182 group.first.fact.clone()
1183}
1184
1185fn constant_summary(group: &Group<'_>) -> String {
1186 if group.grouped {
1187 return "one value each".to_string();
1188 }
1189 group
1190 .profile
1191 .and_then(|profile| profile.dominant_value.as_ref())
1192 .map(|value| format!("always {}", quoted(value, 24)))
1193 .unwrap_or_else(|| "one value".to_string())
1194}
1195
1196fn parseable_summary(group: &Group<'_>) -> String {
1197 match (group.reading, group.profile) {
1198 (Some(_), Some(profile)) if is_code(profile) => code_shape(profile),
1199 (Some(reading), Some(profile)) => format!(
1200 "{} parse as {}",
1201 crate::numfmt::percent_of(group.first.affected_rows, profile.non_null_rows()),
1202 reading.label()
1203 ),
1204 (None, _) | (_, None) => group.rows(),
1205 }
1206}
1207
1208fn duplicates_summary(group: &Group<'_>) -> String {
1209 match group.results.identity.as_ref() {
1210 Some(identity) => format!(
1211 "{} extra {}",
1212 numfmt::group_chrome(identity.extra_rows),
1213 if identity.extra_rows == 1 {
1214 "copy"
1215 } else {
1216 "copies"
1217 }
1218 ),
1219 None => group.rows(),
1220 }
1221}
1222
1223fn variants_summary(group: &Group<'_>) -> String {
1224 if group.indices.len() > 1 {
1226 return format!("{} values spelled more than one way", group.indices.len());
1227 }
1228 group
1229 .results
1230 .category_variants
1231 .iter()
1232 .find(|variants| Some(&variants.normalized) == group.first.normalized_category.as_ref())
1233 .map(|variants| {
1234 variants
1235 .variants
1236 .iter()
1237 .take(2)
1238 .map(|(value, _)| quoted(value, 24))
1239 .collect::<Vec<_>>()
1240 .join(" vs ")
1241 })
1242 .unwrap_or_default()
1243}
1244
1245fn key_like_summary(group: &Group<'_>) -> String {
1246 let unique = group
1247 .profile
1248 .and_then(ColumnQualityProfile::uniqueness_rate)
1249 .map(|rate| format!(" ({} unique)", crate::numfmt::percent(rate)))
1250 .unwrap_or_default();
1251 format!(
1252 "{} repeated{unique}",
1253 numfmt::group_chrome(group.first.affected_rows)
1254 )
1255}
1256
1257fn key_repeated_summary(group: &Group<'_>) -> String {
1258 match group.results.intent.as_ref().and_then(|i| i.key.as_ref()) {
1259 Some(key) => format!(
1260 "{} {} repeat; {} extra {}",
1261 numfmt::group_chrome(key.groups),
1262 if key.groups == 1 { "value" } else { "values" },
1263 numfmt::group_chrome(key.extra_rows),
1264 if key.extra_rows == 1 { "row" } else { "rows" }
1265 ),
1266 None => group.rows(),
1267 }
1268}
1269
1270fn is_code(profile: &ColumnQualityProfile) -> bool {
1273 profile.leading_zero_count.is_some_and(|count| count > 0)
1274 || (profile.min_length.is_some()
1275 && profile.min_length == profile.max_length
1276 && profile.min_length.is_some_and(|length| length > 1))
1277}
1278
1279fn code_shape(profile: &ColumnQualityProfile) -> String {
1280 let zeros = profile.leading_zero_count.unwrap_or(0);
1281 match (profile.min_length, profile.max_length) {
1282 (Some(min), Some(max)) if min == max && zeros > 0 => {
1283 format!("{min} digits, leading zeros")
1284 }
1285 (Some(min), Some(max)) if min == max => format!("{min} digits each"),
1286 _ => format!("{} with a leading zero", numfmt::group_chrome(zeros)),
1287 }
1288}
1289
1290pub(crate) fn quoted(value: &str, width: usize) -> String {
1291 let text = format!("{value:?}");
1292 if crate::glyphs::display_width(&text) <= width {
1293 text
1294 } else {
1295 let mut cut = String::new();
1296 for ch in text.chars() {
1297 if crate::glyphs::display_width(&cut) + 2 > width {
1298 break;
1299 }
1300 cut.push(ch);
1301 }
1302 format!("{cut}{}", crate::glyphs::get().ellipsis)
1303 }
1304}
1305
1306pub fn columns_label(columns: &[String], width: usize) -> String {
1308 let mut label = String::new();
1309 for (index, column) in columns.iter().enumerate() {
1310 let candidate = if label.is_empty() {
1311 column.clone()
1312 } else {
1313 format!("{label}, {column}")
1314 };
1315 let rest = columns.len() - index - 1;
1316 let suffix = if rest > 0 {
1317 format!(" +{rest}")
1318 } else {
1319 String::new()
1320 };
1321 let needed =
1322 crate::glyphs::display_width(&candidate) + crate::glyphs::display_width(&suffix);
1323 if !label.is_empty() && needed > width {
1324 return format!("{label} +{}", rest + 1);
1325 }
1326 label = candidate;
1327 }
1328 label
1329}
1330
1331#[derive(Debug, Clone, PartialEq, Eq)]
1334pub enum Outcome {
1335 Passed,
1336 Found {
1337 tier: Severity,
1338 detail: String,
1339 },
1340 Skipped(&'static str),
1342 Unavailable(&'static str),
1345}
1346
1347#[derive(Debug, Clone)]
1350pub struct Check {
1351 pub name: &'static str,
1352 pub looks_for: &'static str,
1353 pub applies_to: String,
1354 pub outcome: Outcome,
1355 pub basis: QualityPrecision,
1357}
1358
1359pub const CHECKS_SHOWN: usize = 6;
1361
1362const NO_ROWS: &str = "no rows to check";
1364
1365pub fn checks(results: &DataQualityResults, report: &QualityReport) -> Vec<Check> {
1367 use polars::prelude::DataType;
1368 let columns = &results.columns;
1369 let count = |filter: &dyn Fn(&ColumnQualityProfile) -> bool| {
1370 columns.iter().filter(|profile| filter(profile)).count()
1371 };
1372 let text = |profile: &ColumnQualityProfile| {
1373 matches!(profile.dtype, DataType::String | DataType::Categorical(..))
1374 };
1375 let all = columns.len();
1376 let floats = count(&|profile| profile.dtype.is_float());
1377 let texts = count(&text);
1378 let keys = count(&|profile| profile.dtype.is_integer() || text(profile));
1379 let reach = |count: usize, kind: &str| {
1380 let noun = if count == 1 { "column" } else { "columns" };
1381 if kind.is_empty() {
1382 format!("{} {noun}", numfmt::group_chrome(count))
1383 } else {
1384 format!("{} {kind} {noun}", numfmt::group_chrome(count))
1385 }
1386 };
1387 let values_read = !report.metadata_only;
1388 let outcome = |check: &str, applies: usize, none: &'static str| {
1391 if applies == 0 {
1392 return Outcome::Skipped(none);
1393 }
1394 if !values_read {
1395 return Outcome::Unavailable("values not read");
1396 }
1397 if report.no_rows {
1398 return Outcome::Unavailable(NO_ROWS);
1399 }
1400 found(report, check)
1401 };
1402 let files = results.source_files.filter(|files| *files > 1);
1403 let by_files = |check: &str| match files {
1404 Some(_) => found(report, check),
1405 None => Outcome::Skipped("needs several files"),
1406 };
1407 let files_reach = match (files, results.footers_read) {
1409 (Some(files), Some(read)) if read < files => format!(
1410 "{} of {} files",
1411 numfmt::group_chrome(read),
1412 numfmt::group_chrome(files)
1413 ),
1414 (Some(files), _) => format!("{} files", numfmt::group_chrome(files)),
1415 (None, _) => "files".to_string(),
1416 };
1417 let values = results.precision;
1418 let mut checks = Vec::new();
1419 if let Some(intent) = &results.intent {
1421 checks.push(Check {
1422 name: INTENT_CHECK,
1423 looks_for: "values against the key and rules declared",
1424 applies_to: reach(intent.declared.len(), "declared"),
1425 outcome: if !intent.measured {
1426 Outcome::Unavailable("values not read")
1427 } else if report.no_rows {
1428 Outcome::Unavailable(NO_ROWS)
1429 } else {
1430 found(report, INTENT_CHECK)
1431 },
1432 basis: intent.precision,
1433 });
1434 }
1435 checks.extend([
1436 Check {
1437 name: "Missing values",
1438 looks_for: "nulls in any column",
1439 applies_to: reach(all, ""),
1440 outcome: outcome("Missing values", all, "no columns"),
1441 basis: values,
1442 },
1443 Check {
1444 name: "NaN or infinite",
1445 looks_for: "NaN or +/-infinity in float columns",
1446 applies_to: reach(floats, "float"),
1447 outcome: outcome("NaN or infinite", floats, "no float columns"),
1448 basis: values,
1449 },
1450 Check {
1451 name: "Duplicate rows",
1452 looks_for: "rows identical in every column",
1453 applies_to: "whole rows".to_string(),
1454 outcome: match results.identity.as_ref() {
1455 _ if !values_read => Outcome::Unavailable("values not read"),
1456 _ if report.no_rows => Outcome::Unavailable(NO_ROWS),
1457 Some(identity) if identity.extra_rows > 0 => Outcome::Found {
1458 tier: Severity::Problem,
1459 detail: format!("{} extra rows", numfmt::group_chrome(identity.extra_rows)),
1460 },
1461 Some(_) => Outcome::Passed,
1462 None => Outcome::Unavailable("not measured"),
1463 },
1464 basis: values,
1465 },
1466 Check {
1467 name: "Blank text",
1468 looks_for: "text that is empty or only whitespace",
1469 applies_to: reach(texts, "text"),
1470 outcome: outcome("Blank text", texts, "no text columns"),
1471 basis: values,
1472 },
1473 Check {
1474 name: "Mixed spellings",
1475 looks_for: "one value in several cases or spacings",
1476 applies_to: reach(texts, "text"),
1477 outcome: outcome("Mixed spellings", texts, "no text columns"),
1478 basis: values,
1479 },
1480 Check {
1481 name: "Type mismatch",
1482 looks_for: "a column typed differently by some files",
1483 applies_to: files_reach.clone(),
1484 outcome: by_files("Type mismatch"),
1485 basis: QualityPrecision::Metadata,
1486 },
1487 Check {
1488 name: "Missing in files",
1489 looks_for: "a column some files do not have",
1490 applies_to: files_reach,
1491 outcome: by_files("Missing in files"),
1492 basis: QualityPrecision::Metadata,
1493 },
1494 Check {
1495 name: "Numbers as text",
1496 looks_for: "text that reads as numbers or dates",
1497 applies_to: reach(texts, "text"),
1498 outcome: outcome("Numbers as text", texts, "no text columns"),
1499 basis: values,
1500 },
1501 Check {
1502 name: "Nearly unique",
1503 looks_for: "a would-be key whose values repeat",
1504 applies_to: reach(keys, "integer/text"),
1505 outcome: if keys > 0 && values_read && results.precision != QualityPrecision::Exact {
1506 Outcome::Unavailable("needs every row checked")
1507 } else {
1508 match outcome("Nearly unique", keys, "no integer or text columns") {
1509 Outcome::Passed if declared_key_repeats(results) => Outcome::Found {
1512 tier: Severity::Problem,
1513 detail: "the declared key repeats".to_string(),
1514 },
1515 outcome => outcome,
1516 }
1517 },
1518 basis: values,
1519 },
1520 Check {
1521 name: "Single value",
1522 looks_for: "a column with one value throughout",
1523 applies_to: reach(all, ""),
1524 outcome: outcome("Single value", all, "no columns"),
1525 basis: values,
1526 },
1527 ]);
1528 checks
1529}
1530
1531fn declared_key_repeats(results: &DataQualityResults) -> bool {
1534 results
1535 .intent
1536 .as_ref()
1537 .and_then(|intent| intent.key.as_ref())
1538 .is_some_and(|key| key.columns.len() == 1 && key.rows_involved > 0)
1539}
1540
1541pub const INTENT_CHECK: &str = "Column intent";
1543
1544fn found(report: &QualityReport, check: &str) -> Outcome {
1546 let mut columns = Vec::new();
1547 let mut tier = Severity::Note;
1548 for finding in report
1549 .findings
1550 .iter()
1551 .filter(|finding| finding.check() == Some(check))
1552 {
1553 tier = tier.min(finding.severity);
1554 for column in &finding.columns {
1555 if !columns.contains(column) {
1556 columns.push(column.clone());
1557 }
1558 }
1559 }
1560 match columns.len() {
1561 0 => Outcome::Passed,
1562 count => Outcome::Found {
1563 tier,
1564 detail: format!(
1565 "{} {}",
1566 numfmt::group_chrome(count),
1567 if count == 1 { "column" } else { "columns" }
1568 ),
1569 },
1570 }
1571}
1572
1573pub const THIN_SEGMENT_ROWS: usize = 30;
1576
1577#[derive(Debug, Clone, Default, PartialEq, Eq)]
1582pub struct Coverage {
1583 pub exact: usize,
1585 pub sampled: usize,
1587 pub metadata: usize,
1589 pub skipped: usize,
1591 pub unavailable: Vec<(&'static str, Vec<&'static str>)>,
1594 pub rows: Vec<String>,
1597 pub limits: Vec<String>,
1600}
1601
1602impl Coverage {
1603 pub fn checks(&self) -> Vec<String> {
1606 let unavailable = self
1607 .unavailable
1608 .iter()
1609 .map(|(_, names)| names.len())
1610 .sum::<usize>();
1611 [
1612 (self.exact, QualityPrecision::Exact.label()),
1613 (self.sampled, QualityPrecision::Sampled.label()),
1614 (self.metadata, QualityPrecision::Metadata.label()),
1615 (self.skipped, "skipped"),
1616 (unavailable, "unavailable"),
1617 ]
1618 .into_iter()
1619 .filter(|(count, _)| *count > 0)
1620 .map(|(count, label)| format!("{} {label}", numfmt::group_chrome(count)))
1621 .collect()
1622 }
1623
1624 pub fn limits(&self) -> Vec<String> {
1626 self.unavailable
1627 .iter()
1628 .map(|(reason, names)| match names.as_slice() {
1629 [name] => format!("{name}: {reason}"),
1630 _ => format!("{} checks: {reason}", names.len()),
1631 })
1632 .chain(self.limits.iter().cloned())
1633 .collect()
1634 }
1635}
1636
1637pub fn coverage(
1638 results: &DataQualityResults,
1639 checks: &[Check],
1640 plan: &DataQualityPlan,
1641) -> Coverage {
1642 let mut coverage = Coverage::default();
1643 for check in checks {
1644 match &check.outcome {
1645 Outcome::Skipped(_) => coverage.skipped += 1,
1646 Outcome::Unavailable(reason) => {
1647 match coverage
1648 .unavailable
1649 .iter_mut()
1650 .find(|(known, _)| known == reason)
1651 {
1652 Some((_, names)) => names.push(check.name),
1653 None => coverage.unavailable.push((reason, vec![check.name])),
1654 }
1655 }
1656 Outcome::Passed | Outcome::Found { .. } => match check.basis {
1657 QualityPrecision::Exact => coverage.exact += 1,
1658 QualityPrecision::Metadata => coverage.metadata += 1,
1659 QualityPrecision::Sampled => coverage.sampled += 1,
1660 },
1661 }
1662 }
1663
1664 let count = numfmt::group_chrome;
1665 let evaluated = results.evaluated_rows;
1666 coverage
1667 .rows
1668 .push(match (results.precision, results.total_rows) {
1669 (QualityPrecision::Metadata, _) => "none read, file metadata only".to_string(),
1670 _ if evaluated == 0 => "none: the scope has no rows".to_string(),
1671 (QualityPrecision::Exact, _) => format!("all {} read, exact", count(evaluated)),
1672 (_, Some(total)) => format!(
1673 "{} of {} sampled ({})",
1674 count(evaluated),
1675 count(total),
1676 crate::numfmt::percent_of(evaluated, total)
1677 ),
1678 (_, None) => format!("{} sampled, total not counted", count(evaluated)),
1679 });
1680 if let Some(each) = results.per_value {
1681 coverage
1682 .rows
1683 .push(format!("up to {} per value", count(each)));
1684 }
1685 if let Some(reads) = results
1687 .reads
1688 .filter(|_| results.precision != QualityPrecision::Metadata)
1689 {
1690 if reads.reads == 0 {
1691 coverage.rows.push("no source read".to_string());
1692 } else if reads.counted == reads.reads {
1693 coverage
1694 .rows
1695 .push(format!("{} traversed", count(reads.rows)));
1696 } else if reads.counted > 0 {
1697 coverage
1698 .rows
1699 .push(format!("at least {} traversed", count(reads.rows)));
1700 }
1701 if let Some(copy) = reads.copy {
1702 let bytes = crate::numfmt::bytes(copy.bytes);
1703 coverage.rows.push(if copy.fetched {
1704 format!("passes read a local copy, fetched once ({bytes})")
1705 } else {
1706 format!("passes read a local copy fetched earlier ({bytes})")
1707 });
1708 }
1709 }
1710
1711 let segments = &results.segments;
1712 if matches!(results.precision, QualityPrecision::Sampled) && segments.len() > 1 {
1713 let thin = segments
1714 .iter()
1715 .filter(|segment| segment.evaluated_rows < THIN_SEGMENT_ROWS)
1716 .count();
1717 if thin > 0 {
1718 coverage.limits.push(format!(
1719 "{} of {} segments under {THIN_SEGMENT_ROWS} sampled rows",
1720 count(thin),
1721 count(segments.len())
1722 ));
1723 }
1724 }
1725 if !results.unsampled_segments.is_empty() {
1726 coverage.limits.push(format!(
1727 "{} segments with rows, none sampled",
1728 count(results.unsampled_segments.len())
1729 ));
1730 }
1731 if let (Some(files), Some(read)) = (results.source_files, results.footers_read)
1732 && read < files
1733 {
1734 coverage.limits.push(format!(
1735 "footers of {} of {} files read",
1736 count(read),
1737 count(files)
1738 ));
1739 }
1740 if let Some(intent) = results.intent.as_ref().filter(|intent| intent.measured) {
1741 if intent.key.is_some() && intent.precision != QualityPrecision::Exact {
1743 coverage.limits.push(format!(
1744 "key repeats among {} sampled rows only",
1745 count(intent.evaluated_rows)
1746 ));
1747 }
1748 }
1749 if let Some(intent) = &results.intent
1750 && !intent.absent.is_empty()
1751 {
1752 coverage.limits.push(format!(
1753 "intent on {}: not in scope",
1754 columns_label(&intent.absent, 24)
1755 ));
1756 }
1757 if !plan.temporal_roles.is_empty() && plan.interval_pairs().is_empty() {
1758 coverage
1759 .limits
1760 .push("time roles form no interval".to_string());
1761 }
1762 coverage
1763}
1764
1765pub fn advice(finding: &Finding) -> Vec<String> {
1767 let Some(kind) = finding.kind else {
1768 return Vec::new();
1769 };
1770 if finding.variant == Some(Variant::MostlyMissing) {
1771 let rest = finding.evaluated_rows.saturating_sub(finding.affected_rows);
1772 return vec![
1773 if finding.columns.len() == 1 {
1774 format!(
1775 "Aggregates and joins see only {}",
1776 crate::numfmt::percent_of(rest, finding.evaluated_rows)
1777 )
1778 } else {
1779 "Aggregates and joins see only the filled rows".to_string()
1780 },
1781 "Check: filled only for some rows, or stopped at some point".to_string(),
1782 ];
1783 }
1784 finding
1785 .variant
1786 .and_then(Variant::advice)
1787 .unwrap_or(kind.spec().advice)
1788 .iter()
1789 .map(|line| line.to_string())
1790 .collect()
1791}
1792
1793struct Detail<'a> {
1795 finding: &'a Finding,
1796 results: &'a DataQualityResults,
1797 of: String,
1799}
1800
1801impl Detail<'_> {
1802 fn profile(&self, name: &str) -> Option<&ColumnQualityProfile> {
1803 self.results
1804 .columns
1805 .iter()
1806 .find(|profile| profile.name == name)
1807 }
1808
1809 fn observation(&self, index: usize) -> &QualityObservation {
1810 &self.results.observations[index]
1811 }
1812
1813 fn values(&self, what: &str) -> String {
1815 let finding = self.finding;
1816 format!(
1817 "{} of {} values ({}) {what}",
1818 numfmt::group_chrome(finding.affected_rows),
1819 numfmt::group_chrome(finding.evaluated_rows),
1820 crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
1821 )
1822 }
1823}
1824
1825pub fn describe(finding: &Finding, results: &DataQualityResults) -> (String, Vec<String>) {
1828 let count = numfmt::group_chrome;
1829 let mut evidence = Vec::new();
1830 let Some(kind) = finding.kind else {
1831 let headline = format!(
1832 "{} {} passed every check",
1833 count(finding.columns.len()),
1834 if finding.columns.len() == 1 {
1835 "column"
1836 } else {
1837 "columns"
1838 }
1839 );
1840 return (headline, evidence);
1841 };
1842 let detail = Detail {
1843 finding,
1844 results,
1845 of: format!(
1846 "{} of {} rows ({})",
1847 count(finding.affected_rows),
1848 count(finding.evaluated_rows),
1849 crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
1850 ),
1851 };
1852 let headline = (kind.spec().headline)(&detail, &mut evidence);
1853 (headline, evidence)
1854}
1855
1856fn nulls_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1857 let (finding, of) = (detail.finding, &detail.of);
1858 if finding.severity == Severity::Problem {
1859 format!(
1860 "Null in all {} rows checked",
1861 numfmt::group_chrome(finding.evaluated_rows)
1862 )
1863 } else if finding.same_rows {
1864 if finding.columns.len() == 2 {
1865 evidence.push("No row misses one without the other".to_string());
1866 format!("{of} null in both columns")
1867 } else {
1868 evidence.push("No row misses one of them without the others".to_string());
1869 format!("{of} null in all {} columns", finding.columns.len())
1870 }
1871 } else if finding.varied() {
1872 evidence.extend(finding.breakdown(detail.results));
1874 format!("Null rate in {} columns:", finding.columns.len())
1875 } else {
1876 per_column_headline(detail, evidence)
1877 }
1878}
1879
1880fn per_column_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1883 let (finding, of) = (detail.finding, &detail.of);
1884 let noun = finding.kind.map_or("", |kind| kind.spec().noun);
1885 if finding.lists_columns() {
1886 evidence.extend(finding.breakdown(detail.results));
1887 format!("{of} {noun} in each column:")
1888 } else {
1889 format!("{of} {noun}")
1890 }
1891}
1892
1893fn non_finite_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1894 let count = numfmt::group_chrome;
1895 if let Some(profile) = detail.profile(&detail.finding.columns[0]) {
1896 evidence.push(format!(
1897 "NaN {}, +inf {}, -inf {}",
1898 count(profile.nan_count.unwrap_or(0)),
1899 count(profile.positive_infinity_count.unwrap_or(0)),
1900 count(profile.negative_infinity_count.unwrap_or(0))
1901 ));
1902 }
1903 format!("{} NaN or infinite", detail.of)
1904}
1905
1906fn constant_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1907 let finding = detail.finding;
1908 let rows = numfmt::group_chrome(finding.evaluated_rows);
1909 if finding.columns.len() > 1 {
1910 for column in &finding.columns {
1911 if let Some(value) = detail
1912 .profile(column)
1913 .and_then(|p| p.dominant_value.as_ref())
1914 {
1915 evidence.push(format!("{column}: always {}", quoted(value, 40)));
1916 }
1917 }
1918 return format!("One value per column in {rows} rows");
1919 }
1920 match detail
1921 .profile(&finding.columns[0])
1922 .and_then(|p| p.dominant_value.as_ref())
1923 {
1924 Some(value) => format!("Always {} in {rows} rows", quoted(value, 40)),
1925 None => format!("One value in {rows} rows"),
1926 }
1927}
1928
1929fn parseable_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1930 let count = numfmt::group_chrome;
1931 let column = &detail.finding.columns[0];
1932 let profile = detail.profile(column);
1933 let (Some(profile), Some((parsed, reading))) = (profile, profile.and_then(text_reading)) else {
1934 return detail.finding.summary.clone();
1935 };
1936 if let (Some(min), Some(max)) = (profile.min_length, profile.max_length) {
1937 evidence.push(if min == max {
1938 format!("{min} characters each")
1939 } else {
1940 format!("{min} to {max} characters")
1941 });
1942 }
1943 if let Some(zeros) = profile.leading_zero_count.filter(|zeros| *zeros > 0) {
1944 evidence.push(format!("{} with a leading zero", count(zeros)));
1945 }
1946 if let (Some(min), Some(max)) = (&profile.min, &profile.max) {
1947 evidence.push(format!("From {} to {}", quoted(min, 24), quoted(max, 24)));
1948 }
1949 let failed = profile.non_null_rows().saturating_sub(parsed);
1950 if failed > 0 {
1951 let examples = detail
1952 .results
1953 .examples_of(ObservationKind::ParseableText, column);
1954 evidence.push(if examples.is_empty() {
1955 format!("{} do not parse", count(failed))
1956 } else {
1957 crate::glyphs::fit(
1958 &format!(
1959 "{} do not parse, such as {}",
1960 count(failed),
1961 examples.join(", ")
1962 ),
1963 EXAMPLE_WIDTH,
1964 )
1965 });
1966 }
1967 format!(
1968 "{} of {} values ({}) parse as {}",
1969 count(parsed),
1970 count(profile.non_null_rows()),
1971 crate::numfmt::percent_of(parsed, profile.non_null_rows()),
1972 reading.label()
1973 )
1974}
1975
1976fn duplicates_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1977 let count = numfmt::group_chrome;
1978 let Some(identity) = detail.results.identity.as_ref() else {
1979 return detail.finding.summary.clone();
1980 };
1981 evidence.push(format!(
1982 "{} of {} rows ({}) have a copy",
1983 count(identity.rows_involved),
1984 count(identity.evaluated_rows),
1985 crate::numfmt::percent_of(identity.rows_involved, identity.evaluated_rows)
1986 ));
1987 if !identity.examples.is_empty() {
1988 evidence.push("Most copied:".to_string());
1989 }
1990 for example in &identity.examples {
1991 evidence.push(crate::glyphs::fit(
1992 &format!(
1993 "{}{} {}",
1994 crate::glyphs::get().times,
1995 example.copies,
1996 example.values.join(", ")
1997 ),
1998 EXAMPLE_WIDTH,
1999 ));
2000 }
2001 format!(
2002 "{} rows repeated; {} extra {}",
2003 count(identity.duplicate_groups),
2004 count(identity.extra_rows),
2005 if identity.extra_rows == 1 {
2006 "copy"
2007 } else {
2008 "copies"
2009 }
2010 )
2011}
2012
2013fn variants_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2014 let count = numfmt::group_chrome;
2015 for index in &detail.finding.observations {
2016 let normalized = detail.observation(*index).normalized_category.as_ref();
2017 let Some(group) = detail
2018 .results
2019 .category_variants
2020 .iter()
2021 .find(|group| Some(&group.normalized) == normalized)
2022 else {
2023 continue;
2024 };
2025 let mut variants = group.variants.iter().collect::<Vec<_>>();
2026 variants.sort_by_key(|(_, rows)| std::cmp::Reverse(*rows));
2027 evidence.push(
2028 variants
2029 .into_iter()
2030 .take(4)
2031 .map(|(value, rows)| format!("{} ({})", quoted(value, 28), count(*rows)))
2032 .collect::<Vec<_>>()
2033 .join(" "),
2034 );
2035 }
2036 let values = detail.finding.observations.len();
2037 format!(
2038 "{} {} spelled more than one way, {}",
2039 count(values),
2040 if values == 1 { "value" } else { "values" },
2041 detail.of
2042 )
2043}
2044
2045fn key_like_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2046 let count = numfmt::group_chrome;
2047 let Some(profile) = detail.profile(&detail.finding.columns[0]) else {
2048 return detail.finding.summary.clone();
2049 };
2050 if let (Some(value), Some(times)) = (&profile.dominant_value, profile.dominant_count) {
2051 evidence.push(format!(
2052 "Most repeated: {} ({} times)",
2053 quoted(value, 32),
2054 count(times)
2055 ));
2056 }
2057 format!(
2058 "{} distinct in {} rows; {} repeats",
2059 count(profile.distinct_count.unwrap_or(0)),
2060 count(profile.non_null_rows()),
2061 count(detail.finding.affected_rows)
2062 )
2063}
2064
2065fn unparsed_time_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2066 let observation = detail.observation(detail.finding.observations[0]);
2067 if let Some(format) = &observation.time_format {
2068 evidence.push(format!("Read as {} for this study only", format.label()));
2069 }
2070 let examples = detail
2071 .results
2072 .examples_of(ObservationKind::UnparsedTime, &observation.column);
2073 if !examples.is_empty() {
2074 evidence.push(crate::glyphs::fit(
2075 &format!("Such as {}", examples.join(", ")),
2076 EXAMPLE_WIDTH,
2077 ));
2078 }
2079 detail.values("do not parse")
2080}
2081
2082fn signal_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2084 signal_evidence(detail, evidence);
2085 let noun = detail.finding.kind.map_or("", |kind| kind.spec().noun);
2086 format!("{} {noun}", detail.of)
2087}
2088
2089fn dc_offset_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2090 signal_evidence(detail, evidence);
2091 "Mean away from zero".to_string()
2092}
2093
2094fn signal_evidence(detail: &Detail<'_>, evidence: &mut Vec<String>) {
2095 for index in &detail.finding.observations {
2096 let observation = detail.observation(*index);
2097 evidence.push(format!("{}: {}", observation.column, observation.fact));
2098 }
2099}
2100
2101fn files_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2102 let count = numfmt::group_chrome;
2103 let finding = detail.finding;
2104 let observation = detail.observation(finding.observations[0]);
2105 evidence.push(format!(
2108 "{} of {} rows of the loaded source ({})",
2109 count(finding.affected_rows),
2110 count(finding.evaluated_rows),
2111 crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
2112 ));
2113 for file in observation.files.iter().take(4) {
2114 let stored = file
2115 .stored_type
2116 .as_ref()
2117 .map(|dtype| format!(" as {dtype}"))
2118 .unwrap_or_default();
2119 let examples = if file.examples.is_empty() {
2120 String::new()
2121 } else {
2122 format!(
2123 ": {}",
2124 file.examples
2125 .iter()
2126 .map(|value| quoted(value, 20))
2127 .collect::<Vec<_>>()
2128 .join(", ")
2129 )
2130 };
2131 evidence.push(format!(
2132 "#{} {} ({} rows){stored}{examples}",
2133 file.number,
2134 file.name,
2135 count(file.rows)
2136 ));
2137 }
2138 if observation.files.len() > 4 {
2139 evidence.push(format!(
2140 "{} more {}",
2141 observation.files.len() - 4,
2142 crate::glyphs::get().ellipsis
2143 ));
2144 }
2145 upper_first(&observation.fact)
2146}
2147
2148const EXAMPLE_WIDTH: usize = 72;
2151
2152struct Declared<'a> {
2156 intent: &'a crate::analysis::quality_intent::IntentResults,
2157 sampled: bool,
2158 rows_word: &'static str,
2160 share: String,
2162 check: Option<&'a crate::analysis::quality_intent::ColumnCheck>,
2163 column: &'a str,
2164}
2165
2166impl<'a> Declared<'a> {
2167 fn of(detail: &Detail<'a>) -> Option<Self> {
2168 let intent = detail.results.intent.as_ref()?;
2169 let finding = detail.finding;
2170 let column = finding.columns.first().map(String::as_str).unwrap_or("");
2171 let sampled = intent.precision != QualityPrecision::Exact;
2172 Some(Self {
2173 intent,
2174 sampled,
2175 rows_word: if sampled { "sampled rows" } else { "rows" },
2176 share: crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows),
2177 check: intent.column(column),
2178 column,
2179 })
2180 }
2181}
2182
2183fn value_examples(values: &[(String, usize)]) -> String {
2185 values
2186 .iter()
2187 .map(|(value, rows)| format!("{} ({})", quoted(value, 24), numfmt::group_chrome(*rows)))
2188 .collect::<Vec<_>>()
2189 .join(" ")
2190}
2191
2192fn key_repeated_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2193 let count = numfmt::group_chrome;
2194 let Some(declared) = Declared::of(detail) else {
2195 return detail.finding.summary.clone();
2196 };
2197 let Some(key) = declared.intent.key.as_ref() else {
2198 return detail.finding.summary.clone();
2199 };
2200 evidence.push(format!("Declared key: {}", key.columns.join(", ")));
2201 evidence.push(format!(
2202 "{} {} held by more than one row; {} rows beyond one per value",
2203 count(key.groups),
2204 if key.groups == 1 { "value" } else { "values" },
2205 count(key.extra_rows)
2206 ));
2207 if declared.sampled {
2208 evidence.push(
2209 "Each repeat here is one in the data; unsampled rows are not checked".to_string(),
2210 );
2211 }
2212 format!(
2213 "{} of {} {} ({}) share their key with another row",
2214 count(key.rows_involved),
2215 count(declared.intent.evaluated_rows),
2216 declared.rows_word,
2217 declared.share
2218 )
2219}
2220
2221fn key_missing_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2222 let count = numfmt::group_chrome;
2223 let Some(declared) = Declared::of(detail) else {
2224 return detail.finding.summary.clone();
2225 };
2226 let Some(key) = declared.intent.key.as_ref() else {
2227 return detail.finding.summary.clone();
2228 };
2229 evidence.push(format!("Declared key: {}", key.columns.join(", ")));
2230 format!(
2231 "{} of {} {} ({}) have no value in part of the key",
2232 count(key.missing),
2233 count(declared.intent.evaluated_rows),
2234 declared.rows_word,
2235 declared.share
2236 )
2237}
2238
2239fn required_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2240 let count = numfmt::group_chrome;
2241 let Some(declared) = Declared::of(detail) else {
2242 return detail.finding.summary.clone();
2243 };
2244 let column = declared.column;
2245 evidence.push(format!("Declared required: {column}"));
2246 format!(
2247 "{} of {} {} ({}) have no {column}",
2248 count(detail.finding.affected_rows),
2249 count(detail.finding.evaluated_rows),
2250 declared.rows_word,
2251 declared.share
2252 )
2253}
2254
2255fn not_allowed_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2256 let Some(declared) = Declared::of(detail) else {
2257 return detail.finding.summary.clone();
2258 };
2259 if let Some(check) = declared.check {
2260 evidence.push(format!("Allowed: {}", check.intent.allowed_label(8)));
2261 if !check.outside_examples.is_empty() {
2262 evidence.push(format!(
2263 "Found: {}",
2264 value_examples(&check.outside_examples)
2265 ));
2266 }
2267 }
2268 detail.values("are not allowed")
2269}
2270
2271fn out_of_range_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2272 let count = numfmt::group_chrome;
2273 let Some(declared) = Declared::of(detail) else {
2274 return detail.finding.summary.clone();
2275 };
2276 if let Some(check) = declared.check {
2277 evidence.push(format!(
2278 "Range: {}",
2279 check.intent.range_label().unwrap_or_default()
2280 ));
2281 if let Some(below) = check.below.filter(|below| *below > 0) {
2282 let lowest = check
2283 .lowest
2284 .as_ref()
2285 .map(|value| format!(", lowest {value}"))
2286 .unwrap_or_default();
2287 evidence.push(format!("Below: {}{lowest}", count(below)));
2288 }
2289 if let Some(above) = check.above.filter(|above| *above > 0) {
2290 let highest = check
2291 .highest
2292 .as_ref()
2293 .map(|value| format!(", highest {value}"))
2294 .unwrap_or_default();
2295 evidence.push(format!("Above: {}{highest}", count(above)));
2296 }
2297 }
2298 detail.values("outside the range")
2299}
2300
2301fn unparsed_number_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2302 let Some(declared) = Declared::of(detail) else {
2303 return detail.finding.summary.clone();
2304 };
2305 let reading = declared.check.and_then(|check| check.intent.number).map_or(
2306 "number",
2307 crate::analysis::quality_intent::NumberReading::label,
2308 );
2309 if let Some(check) = declared
2310 .check
2311 .filter(|check| !check.unparsed_examples.is_empty())
2312 {
2313 evidence.push(format!(
2314 "Such as: {}",
2315 value_examples(&check.unparsed_examples)
2316 ));
2317 }
2318 evidence.push(format!("Read as a {reading} for this study only"));
2319 detail.values(&format!("do not read as a {reading}"))
2320}
2321
2322fn upper_first(text: &str) -> String {
2323 let mut chars = text.chars();
2324 match chars.next() {
2325 Some(first) => first.to_uppercase().chain(chars).collect(),
2326 None => String::new(),
2327 }
2328}
2329
2330pub fn verdict(report: &QualityReport) -> String {
2332 if report.metadata_only {
2335 return match report.problems {
2336 0 => "Values not read: file metadata only".to_string(),
2337 1 => "1 problem in file metadata; values not read".to_string(),
2338 count => format!(
2339 "{} problems in file metadata; values not read",
2340 numfmt::group_chrome(count)
2341 ),
2342 };
2343 }
2344 if report.no_rows {
2346 return match report.problems {
2347 0 => "No rows to check".to_string(),
2348 1 => "1 problem in file metadata; no rows to check".to_string(),
2349 count => format!(
2350 "{} problems in file metadata; no rows to check",
2351 numfmt::group_chrome(count)
2352 ),
2353 };
2354 }
2355 let clean = format!(
2356 "{} of {} columns clean",
2357 numfmt::group_chrome(report.clean_columns),
2358 numfmt::group_chrome(report.total_columns)
2359 );
2360 let notes = match report.notes {
2361 0 => String::new(),
2362 1 => "1 note ".to_string(),
2363 count => format!("{} notes ", numfmt::group_chrome(count)),
2364 };
2365 match report.problems {
2366 0 => format!("No problems found {notes}{clean}"),
2367 1 => format!("1 problem {notes}{clean}"),
2368 count => format!("{} problems {notes}{clean}", numfmt::group_chrome(count)),
2369 }
2370}
2371
2372#[cfg(test)]
2373mod tests {
2374 use super::*;
2375
2376 fn reads(finding: &Finding) -> (Option<ObservationKind>, Option<Variant>) {
2378 (finding.kind, finding.variant)
2379 }
2380 use crate::analysis::data_quality::SharedNulls;
2381 use crate::analysis::data_quality::fixtures::{observation, profile, results_with};
2382 use polars::prelude::DataType;
2383
2384 #[test]
2386 fn nulls_with_one_count_become_one_finding() {
2387 let names = ["open", "high", "low", "close"];
2388 let mut columns = names
2389 .iter()
2390 .map(|name| profile(name, DataType::Float64))
2391 .collect::<Vec<_>>();
2392 columns.push(profile("ticker", DataType::String));
2393 let mut results = results_with(
2394 columns,
2395 names
2396 .iter()
2397 .map(|name| observation(ObservationKind::Nulls, name, 4))
2398 .collect(),
2399 );
2400 results.shared_nulls = vec![SharedNulls {
2401 columns: names.iter().map(|name| name.to_string()).collect(),
2402 null_rows: 4,
2403 rows_null_in_all: 4,
2404 }];
2405 let report = build_report(&results);
2406 assert_eq!(report.findings.len(), 2, "one note and the clean entry");
2407 let missing = &report.findings[0];
2408 assert_eq!(
2409 reads(missing),
2410 (Some(ObservationKind::Nulls), Some(Variant::MissingTogether))
2411 );
2412 assert_eq!(missing.severity, Severity::Note);
2413 assert_eq!(missing.columns.len(), 4);
2414 assert!(missing.same_rows);
2415 assert_eq!(missing.summary, "4 rows (4.0%)");
2416 assert_eq!(report.findings[1].severity, Severity::Clean);
2417 assert_eq!(report.findings[1].columns, vec!["ticker".to_string()]);
2418 assert_eq!(report.problems, 0);
2419 assert!(verdict(&report).starts_with("No problems found"));
2420 }
2421
2422 #[test]
2425 fn missing_values_collapse_and_mostly_missing_leads() {
2426 let names = ["a", "b", "c", "d"];
2427 let results = results_with(
2428 names
2429 .iter()
2430 .map(|name| profile(name, DataType::Float64))
2431 .collect(),
2432 vec![
2433 observation(ObservationKind::Nulls, "a", 3),
2434 observation(ObservationKind::Nulls, "b", 40),
2435 observation(ObservationKind::Nulls, "c", 12),
2436 observation(ObservationKind::Nulls, "d", 80),
2437 ],
2438 );
2439 let report = build_report(&results);
2440 assert_eq!(report.findings.len(), 2);
2441 let mostly = &report.findings[0];
2442 assert_eq!(
2443 reads(mostly),
2444 (Some(ObservationKind::Nulls), Some(Variant::MostlyMissing))
2445 );
2446 assert_eq!(mostly.columns, vec!["d".to_string()]);
2447 let missing = &report.findings[1];
2448 assert_eq!(reads(missing), (Some(ObservationKind::Nulls), None));
2449 assert_eq!(missing.columns, vec!["b", "c", "a"]);
2450 assert_eq!(missing.summary, "3.0% to 40.0% per column");
2451 let (headline, evidence) = describe(missing, &results);
2452 assert_eq!(headline, "Null rate in 3 columns:");
2453 assert_eq!(evidence[0], "b 40.0% 40 rows");
2454 assert_eq!(evidence[2], "a 3.0% 3 rows");
2455 }
2456
2457 #[test]
2458 fn problems_rank_before_notes_and_mark_their_columns() {
2459 let results = results_with(
2460 vec![
2461 profile("price", DataType::Float64),
2462 profile("region", DataType::String),
2463 ],
2464 vec![
2465 observation(ObservationKind::Nulls, "region", 3),
2466 observation(ObservationKind::NonFinite, "price", 2),
2467 ],
2468 );
2469 let report = build_report(&results);
2470 assert_eq!(
2471 reads(&report.findings[0]),
2472 (Some(ObservationKind::NonFinite), None)
2473 );
2474 assert_eq!(report.findings[0].severity, Severity::Problem);
2475 assert_eq!(
2476 reads(&report.findings[1]),
2477 (Some(ObservationKind::Nulls), None)
2478 );
2479 assert_eq!(
2480 report.column_status,
2481 vec![Severity::Problem, Severity::Note]
2482 );
2483 assert_eq!(report.clean_columns, 0);
2484 assert_eq!(verdict(&report), "1 problem 1 note 0 of 2 columns clean");
2485 }
2486
2487 #[test]
2488 fn a_column_with_no_values_at_all_is_a_problem() {
2489 let results = results_with(
2490 vec![profile("legacy", DataType::String)],
2491 vec![observation(ObservationKind::Nulls, "legacy", 100)],
2492 );
2493 let report = build_report(&results);
2494 assert_eq!(
2495 reads(&report.findings[0]),
2496 (Some(ObservationKind::Nulls), Some(Variant::AlwaysMissing))
2497 );
2498 assert_eq!(report.findings[0].severity, Severity::Problem);
2499 }
2500
2501 #[test]
2504 fn zero_padded_digits_read_as_codes() {
2505 let mut code = profile("industry", DataType::String);
2506 code.integer_parse_count = Some(100);
2507 code.decimal_parse_count = Some(100);
2508 code.leading_zero_count = Some(20);
2509 code.min_length = Some(4);
2510 code.max_length = Some(4);
2511 let results = results_with(
2512 vec![code],
2513 vec![observation(ObservationKind::ParseableText, "industry", 100)],
2514 );
2515 let finding = &build_report(&results).findings[0];
2516 assert_eq!(
2517 reads(finding),
2518 (
2519 Some(ObservationKind::ParseableText),
2520 Some(Variant::CodesAsText)
2521 )
2522 );
2523 assert_eq!(finding.summary, "4 digits, leading zeros");
2524 }
2525
2526 #[test]
2529 fn a_metadata_run_reports_footer_problems_and_no_clean_columns() {
2530 let mut results = results_with(
2531 vec![
2532 profile("fee", DataType::Float64),
2533 profile("id", DataType::Int64),
2534 ],
2535 vec![observation(ObservationKind::Absent, "fee", 20)],
2536 );
2537 results.precision = QualityPrecision::Metadata;
2538 let report = build_report(&results);
2539 assert_eq!(report.findings.len(), 1, "no clean entry");
2540 assert_eq!(
2541 reads(&report.findings[0]),
2542 (Some(ObservationKind::Absent), None)
2543 );
2544 assert_eq!(
2545 verdict(&report),
2546 "1 problem in file metadata; values not read"
2547 );
2548 }
2549
2550 #[test]
2553 fn checks_report_reach_findings_and_what_did_not_run() {
2554 let mut results = results_with(
2555 vec![
2556 profile("price", DataType::Float64),
2557 profile("region", DataType::String),
2558 profile("id", DataType::Int64),
2559 ],
2560 vec![observation(ObservationKind::NonFinite, "price", 2)],
2561 );
2562 results.precision = QualityPrecision::Sampled;
2563 let report = build_report(&results);
2564 let list = checks(&results, &report);
2565 let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2566 assert_eq!(list[0].name, "Missing values", "most important first");
2567 assert_eq!(by_name("Missing values").outcome, Outcome::Passed);
2568 assert_eq!(by_name("Missing values").applies_to, "3 columns");
2569 assert_eq!(by_name("NaN or infinite").applies_to, "1 float column");
2570 assert_eq!(
2571 by_name("NaN or infinite").outcome,
2572 Outcome::Found {
2573 tier: Severity::Problem,
2574 detail: "1 column".to_string()
2575 }
2576 );
2577 assert_eq!(
2578 by_name("Nearly unique").outcome,
2579 Outcome::Unavailable("needs every row checked"),
2580 "a sample cannot say a column is nearly a key"
2581 );
2582 assert_eq!(
2583 by_name("Type mismatch").outcome,
2584 Outcome::Skipped("needs several files")
2585 );
2586
2587 results.precision = QualityPrecision::Metadata;
2588 results.source_files = Some(3);
2589 let report = build_report(&results);
2590 let list = checks(&results, &report);
2591 let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2592 assert_eq!(
2593 by_name("Missing values").outcome,
2594 Outcome::Unavailable("values not read")
2595 );
2596 assert_eq!(by_name("Type mismatch").outcome, Outcome::Passed);
2597 assert_eq!(by_name("Type mismatch").applies_to, "3 files");
2598 }
2599
2600 #[test]
2604 fn coverage_separates_checked_skipped_and_unavailable() {
2605 use crate::analysis::data_quality::{
2606 ObservedReads, SegmentQualityProfile, TemporalRole, TemporalRoleAssignment,
2607 };
2608 let columns = vec![
2609 profile("price", DataType::Float64),
2610 profile("region", DataType::String),
2611 profile("id", DataType::Int64),
2612 ];
2613 let plan = DataQualityPlan::default();
2614 let measured = |mut results: DataQualityResults| {
2615 results.identity = Some(crate::analysis::data_quality::IdentityProfile {
2616 duplicate_groups: 0,
2617 extra_rows: 0,
2618 rows_involved: 0,
2619 evaluated_rows: results.evaluated_rows,
2620 examples: Vec::new(),
2621 });
2622 results
2623 };
2624
2625 let mut sampled = measured(results_with(columns.clone(), Vec::new()));
2627 sampled.precision = QualityPrecision::Sampled;
2628 sampled.total_rows = Some(1_000);
2629 sampled.reads = Some(ObservedReads {
2630 reads: 1,
2631 counted: 1,
2632 rows: 1_000,
2633 copy: None,
2634 });
2635 let report = build_report(&sampled);
2636 assert_eq!(report.problems + report.notes, 0, "a clean report");
2637 let found = coverage(&sampled, &checks(&sampled, &report), &plan);
2638 assert_eq!(
2639 found.checks(),
2640 ["7 sampled", "2 skipped", "1 unavailable"],
2641 "{found:?}"
2642 );
2643 assert_eq!(
2644 found.rows,
2645 ["100 of 1,000 sampled (10.0%)", "1,000 traversed"]
2646 );
2647 assert_eq!(found.limits(), ["Nearly unique: needs every row checked"]);
2648
2649 let segment = |label: &str, rows: usize| SegmentQualityProfile {
2651 label: label.to_string(),
2652 total_rows: Some(500),
2653 evaluated_rows: rows,
2654 columns: Vec::new(),
2655 null_cells: 0,
2656 null_rate: 0.0,
2657 compared_with: None,
2658 largest_change: None,
2659 change_size: None,
2660 };
2661 sampled.segments = vec![segment("a", 90), segment("b", 10), segment("c", 0)];
2662 sampled.source_files = Some(400);
2663 sampled.footers_read = Some(100);
2664 let roles = DataQualityPlan {
2665 temporal_roles: vec![TemporalRoleAssignment {
2666 role: TemporalRole::Event,
2667 column: "id".to_string(),
2668 timezone: None,
2669 }],
2670 ..plan.clone()
2671 };
2672 let report = build_report(&sampled);
2673 let list = checks(&sampled, &report);
2674 let type_mismatch = list.iter().find(|c| c.name == "Type mismatch").unwrap();
2675 assert_eq!(type_mismatch.applies_to, "100 of 400 files");
2676 assert_eq!(type_mismatch.basis, QualityPrecision::Metadata);
2677 let found = coverage(&sampled, &list, &roles);
2678 assert_eq!(found.checks(), ["7 sampled", "2 metadata", "1 unavailable"]);
2679 assert_eq!(
2680 found.limits(),
2681 [
2682 "Nearly unique: needs every row checked",
2683 "2 of 3 segments under 30 sampled rows",
2684 "footers of 100 of 400 files read",
2685 "time roles form no interval",
2686 ]
2687 );
2688
2689 let mut full = measured(results_with(columns.clone(), Vec::new()));
2691 full.reads = Some(ObservedReads {
2692 reads: 4,
2693 counted: 3,
2694 rows: 300,
2695 copy: None,
2696 });
2697 let report = build_report(&full);
2698 let found = coverage(&full, &checks(&full, &report), &plan);
2699 assert_eq!(found.checks(), ["8 exact", "2 skipped"]);
2700 assert_eq!(
2701 found.rows,
2702 ["all 100 read, exact", "at least 300 traversed"]
2703 );
2704 assert!(found.limits().is_empty());
2705
2706 let mut metadata = measured(results_with(columns, Vec::new()));
2708 metadata.precision = QualityPrecision::Metadata;
2709 metadata.source_files = Some(3);
2710 metadata.footers_read = Some(3);
2711 metadata.reads = Some(ObservedReads::default());
2712 let report = build_report(&metadata);
2713 let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2714 assert_eq!(found.checks(), ["2 metadata", "8 unavailable"]);
2715 assert_eq!(found.rows, ["none read, file metadata only"]);
2716 assert_eq!(found.limits(), ["8 checks: values not read"]);
2717
2718 let floats_only = vec![profile("price", DataType::Float64)];
2721 let mut metadata = measured(results_with(floats_only.clone(), Vec::new()));
2722 metadata.precision = QualityPrecision::Metadata;
2723 let report = build_report(&metadata);
2724 let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2725 assert_eq!(found.checks(), ["6 skipped", "4 unavailable"], "{found:?}");
2726 let mut sampled = measured(results_with(floats_only, Vec::new()));
2727 sampled.precision = QualityPrecision::Sampled;
2728 let report = build_report(&sampled);
2729 let list = checks(&sampled, &report);
2730 let nearly = list.iter().find(|c| c.name == "Nearly unique").unwrap();
2731 assert_eq!(
2732 nearly.outcome,
2733 Outcome::Skipped("no integer or text columns")
2734 );
2735 }
2736
2737 #[test]
2738 fn column_labels_fit_and_count_the_rest() {
2739 let columns = ["open", "high", "low", "close"].map(String::from);
2740 assert_eq!(columns_label(&columns, 40), "open, high, low, close");
2741 assert_eq!(columns_label(&columns, 14), "open, high +2");
2742 assert_eq!(columns_label(&columns[..1], 2), "open");
2743 }
2744
2745 #[test]
2750 fn findings_narrow_and_order_without_measuring() {
2751 let mut results = results_with(
2752 vec![
2753 profile("price", DataType::Float64),
2754 profile("region", DataType::String),
2755 profile("note", DataType::String),
2756 profile("id", DataType::Int64),
2757 ],
2758 vec![
2759 observation(ObservationKind::NonFinite, "price", 2),
2760 observation(ObservationKind::Nulls, "region", 3),
2761 observation(ObservationKind::Nulls, "note", 30),
2762 observation(ObservationKind::Whitespace, "region", 9),
2763 ],
2764 );
2765 results.observations[0].evaluated_rows = 4;
2767 let report = build_report(&results);
2768 let titles = |view: &FindingsView| {
2769 view.shown(&report)
2770 .into_iter()
2771 .map(|index| reads(&report.findings[index]))
2772 .collect::<Vec<_>>()
2773 };
2774 let ranked = FindingsView::default();
2775 assert_eq!(
2776 titles(&ranked),
2777 [
2778 (Some(ObservationKind::NonFinite), None),
2779 (Some(ObservationKind::Whitespace), None),
2780 (Some(ObservationKind::Nulls), None),
2781 (None, None)
2782 ]
2783 );
2784 let rows = FindingsView {
2785 order: FindingOrder::Rows,
2786 ..FindingsView::default()
2787 };
2788 assert_eq!(
2789 titles(&rows),
2790 [
2791 (Some(ObservationKind::Whitespace), None),
2792 (Some(ObservationKind::NonFinite), None),
2793 (Some(ObservationKind::Nulls), None),
2794 (None, None)
2795 ],
2796 "most rows first, Problems still above Notes"
2797 );
2798 let rate = FindingsView {
2799 order: FindingOrder::Rate,
2800 ..FindingsView::default()
2801 };
2802 assert_eq!(
2803 titles(&rate)[0],
2804 (Some(ObservationKind::NonFinite), None),
2805 "2 of 4 beats 9 of 100"
2806 );
2807
2808 let region = FindingsView {
2809 column: Some("region".to_string()),
2810 ..FindingsView::default()
2811 };
2812 assert_eq!(
2813 titles(®ion),
2814 [
2815 (Some(ObservationKind::Whitespace), None),
2816 (Some(ObservationKind::Nulls), None)
2817 ]
2818 );
2819 assert!(region.narrowed());
2820 let clean = FindingsView {
2821 column: Some("id".to_string()),
2822 ..FindingsView::default()
2823 };
2824 assert_eq!(titles(&clean), [(None, None)], "a clean column is clean");
2825 let missing = FindingsView {
2826 check: Some("Missing values"),
2827 ..FindingsView::default()
2828 };
2829 assert_eq!(titles(&missing), [(Some(ObservationKind::Nulls), None)]);
2830 assert_eq!(
2831 missing.selected(&report, 0).map(reads),
2832 Some((Some(ObservationKind::Nulls), None))
2833 );
2834 assert_eq!(
2835 check_choices(&report),
2836 [
2837 ("NaN or infinite", 1),
2838 ("Blank text", 1),
2839 ("Missing values", 1)
2840 ]
2841 );
2842 assert_eq!(
2843 column_choices(&report, &results)
2844 .into_iter()
2845 .map(|(_, count)| count)
2846 .collect::<Vec<_>>(),
2847 [1, 2, 1, 0]
2848 );
2849 }
2850
2851 #[test]
2854 fn grouped_findings_break_down_by_column_and_bound_the_union() {
2855 let results = results_with(
2856 vec![
2857 profile("a", DataType::String),
2858 profile("b", DataType::String),
2859 ],
2860 vec![
2861 observation(ObservationKind::Empty, "a", 6),
2862 observation(ObservationKind::Empty, "b", 6),
2863 ],
2864 );
2865 let report = build_report(&results);
2866 let finding = &report.findings[0];
2867 assert!(finding.lists_columns());
2868 assert_eq!(
2869 finding.evidence_count(&results),
2870 None,
2871 "a union nobody counted"
2872 );
2873 let (headline, evidence) = describe(finding, &results);
2874 assert_eq!(
2875 headline,
2876 "6 of 100 rows (6.0%) empty strings in each column:"
2877 );
2878 assert_eq!(evidence[0], "a 6.0% 6 rows");
2879 assert_eq!(
2880 evidence.last().unwrap(),
2881 "Rows with any of them: 6 to 12, not counted"
2882 );
2883 }
2884
2885 #[test]
2888 fn parse_failures_are_the_evidence_of_text_that_parses() {
2889 let mut text = profile("amount", DataType::String);
2890 text.null_count = 4;
2891 text.integer_parse_count = Some(95);
2892 text.decimal_parse_count = Some(95);
2893 let mut results = results_with(
2894 vec![text],
2895 vec![observation(ObservationKind::ParseableText, "amount", 95)],
2896 );
2897 results.examples = vec![crate::analysis::data_quality::FindingExamples {
2898 kind: ObservationKind::ParseableText,
2899 column: "amount".to_string(),
2900 values: vec!["\"n/a\"".to_string()],
2901 }];
2902 let report = build_report(&results);
2903 let finding = &report.findings[0];
2904 assert_eq!(finding.check(), Some("Numbers as text"));
2905 assert_eq!(finding.failures(&results), Some(1));
2906 assert_eq!(finding.evidence_count(&results), Some(1));
2907 assert!(matches!(
2908 finding.evidence(&results),
2909 Ok(EvidenceRows::Matching(_))
2910 ));
2911 let (_, evidence) = describe(finding, &results);
2912 assert!(
2913 evidence.contains(&"1 do not parse, such as \"n/a\"".to_string()),
2914 "{evidence:?}"
2915 );
2916
2917 results.columns[0].integer_parse_count = Some(96);
2918 results.columns[0].decimal_parse_count = Some(96);
2919 results.observations[0].affected_rows = 96;
2920 let report = build_report(&results);
2921 let reason = report.findings[0].evidence(&results).unwrap_err();
2922 assert!(reason.contains("every value parses"), "{reason}");
2923 }
2924
2925 #[test]
2927 fn duplicate_rows_open_every_row_with_a_copy() {
2928 let mut results = results_with(
2929 vec![profile("id", DataType::Int64)],
2930 vec![observation(
2931 ObservationKind::DuplicateRows,
2932 "all columns",
2933 5,
2934 )],
2935 );
2936 results.identity = Some(crate::analysis::data_quality::IdentityProfile {
2937 duplicate_groups: 2,
2938 extra_rows: 3,
2939 rows_involved: 5,
2940 evaluated_rows: 100,
2941 examples: vec![crate::analysis::data_quality::DuplicateExample {
2942 copies: 3,
2943 values: vec!["7".to_string()],
2944 }],
2945 });
2946 let report = build_report(&results);
2947 let finding = &report.findings[0];
2948 assert!(matches!(
2949 finding.evidence(&results),
2950 Ok(EvidenceRows::Duplicates)
2951 ));
2952 assert_eq!(finding.evidence_count(&results), Some(5));
2953 let (headline, evidence) = describe(finding, &results);
2954 assert_eq!(headline, "2 rows repeated; 3 extra copies");
2955 assert_eq!(evidence[0], "5 of 100 rows (5.0%) have a copy");
2956 assert_eq!(evidence[2], format!("{}3 7", crate::glyphs::get().times));
2957 }
2958}