1use crate::data_quality::{
9 ColumnQualityProfile, DataQualityPlan, DataQualityResults, ObservationKind, QualityPrecision,
10 QualityScope, TextReading, text_reading,
11};
12use crate::numfmt;
13use polars::prelude::Expr;
14use std::collections::BTreeMap;
15
16#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
17pub enum Severity {
18 Problem,
20 Note,
22 Clean,
24}
25
26impl Severity {
27 pub fn heading(self) -> &'static str {
28 match self {
29 Self::Problem => "Problems",
30 Self::Note => "Notes",
31 Self::Clean => "Clean",
32 }
33 }
34}
35
36#[derive(Debug, Clone)]
37pub struct Finding {
38 pub severity: Severity,
39 pub kind: Option<ObservationKind>,
41 pub title: &'static str,
42 pub columns: Vec<String>,
43 pub observations: Vec<usize>,
45 pub affected_rows: usize,
47 pub evaluated_rows: usize,
48 pub summary: String,
50 pub same_rows: bool,
52}
53
54impl Finding {
55 pub fn columns_label(&self, width: usize) -> String {
57 columns_label(&self.columns, width)
58 }
59
60 pub fn evidence_predicate(&self, results: &DataQualityResults) -> Option<Expr> {
64 self.observations
65 .iter()
66 .map(|index| {
67 let observation = results.observations.get(*index)?;
68 match observation.kind {
69 ObservationKind::ParseableText => {
71 let profile = results
72 .columns
73 .iter()
74 .find(|profile| profile.name == observation.column)?;
75 crate::data_quality::unparsed_text(profile)
76 }
77 _ => observation.evidence_predicate().or_else(|| {
78 results
79 .intent
80 .as_ref()?
81 .evidence(observation.kind, &observation.column)
82 }),
83 }
84 })
85 .collect::<Option<Vec<_>>>()?
86 .into_iter()
87 .reduce(Expr::or)
88 }
89
90 pub fn evidence_scope(&self, results: &DataQualityResults) -> Option<QualityScope> {
92 match self.observations.as_slice() {
93 [index] => results.observations.get(*index)?.evidence_scope(),
94 _ => None,
95 }
96 }
97
98 pub fn evidence(&self, results: &DataQualityResults) -> Result<EvidenceRows, String> {
100 if let Some(scope) = self.evidence_scope(results) {
101 return Ok(EvidenceRows::Files(scope));
102 }
103 match self.kind {
104 None => Err("No rows: every column here passed".to_string()),
105 _ if results.precision == QualityPrecision::Metadata => {
106 Err("No rows: values were not read".to_string())
107 }
108 Some(ObservationKind::DuplicateRows) => Ok(EvidenceRows::Duplicates),
109 Some(ObservationKind::ParseableText) if self.failures(results) == Some(0) => {
110 Err("No rows: every value parses".to_string())
111 }
112 _ => self
113 .evidence_predicate(results)
114 .map(EvidenceRows::Matching)
115 .ok_or_else(|| "No rows: nothing to filter on".to_string()),
116 }
117 }
118
119 pub fn evidence_count(&self, results: &DataQualityResults) -> Option<usize> {
123 match self.kind? {
124 ObservationKind::DuplicateRows => {
125 Some(results.identity.as_ref()?.rows_involved).filter(|rows| *rows > 0)
126 }
127 ObservationKind::ParseableText => self.failures(results),
128 ObservationKind::KeyLike
131 | ObservationKind::Absent
132 | ObservationKind::TypeConflict
133 | ObservationKind::DcOffset => None,
134 ObservationKind::CategoryVariants => Some(self.affected_rows),
135 ObservationKind::KeyRepeated | ObservationKind::KeyMissing => Some(self.affected_rows),
137 _ if self.observations.len() == 1 || self.same_rows => Some(self.affected_rows),
138 _ => None,
139 }
140 }
141
142 pub fn failures(&self, results: &DataQualityResults) -> Option<usize> {
144 if self.kind != Some(ObservationKind::ParseableText) {
145 return None;
146 }
147 let profile = results
148 .columns
149 .iter()
150 .find(|profile| Some(&profile.name) == self.columns.first())?;
151 let (parsed, _) = text_reading(profile)?;
152 Some(profile.non_null_rows().saturating_sub(parsed))
153 }
154
155 pub fn check(&self) -> Option<&'static str> {
158 Some(match self.kind? {
159 ObservationKind::Nulls => "Missing values",
160 ObservationKind::NonFinite => "NaN or infinite",
161 ObservationKind::DuplicateRows => "Duplicate rows",
162 ObservationKind::Empty | ObservationKind::Whitespace => "Blank text",
163 ObservationKind::CategoryVariants => "Mixed spellings",
164 ObservationKind::TypeConflict => "Type mismatch",
165 ObservationKind::Absent => "Missing in files",
166 ObservationKind::ParseableText => "Numbers as text",
167 ObservationKind::KeyLike => "Nearly unique",
168 ObservationKind::Constant => "Single value",
169 ObservationKind::UnparsedTime => "Unparsed times",
170 ObservationKind::Clipping => "Clipping",
171 ObservationKind::ZeroRuns => "Runs of zeros",
172 ObservationKind::DcOffset => "DC offset",
173 ObservationKind::KeyRepeated
174 | ObservationKind::KeyMissing
175 | ObservationKind::RequiredMissing
176 | ObservationKind::NotAllowed
177 | ObservationKind::OutOfRange
178 | ObservationKind::UnparsedNumber => INTENT_CHECK,
179 })
180 }
181
182 pub fn varied(&self) -> bool {
184 self.kind == Some(ObservationKind::Nulls)
185 && self.columns.len() > 1
186 && !self.same_rows
187 && self.severity == Severity::Note
188 }
189
190 pub fn lists_columns(&self) -> bool {
193 matches!(
194 self.kind,
195 Some(ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace)
196 ) && self.columns.len() > 1
197 && !self.same_rows
198 }
199
200 pub fn breakdown(&self, results: &DataQualityResults) -> Vec<String> {
204 let rows = self
205 .observations
206 .iter()
207 .filter_map(|index| results.observations.get(*index))
208 .collect::<Vec<_>>();
209 let name_width = rows
210 .iter()
211 .map(|row| crate::glyphs::display_width(&row.column))
212 .max()
213 .unwrap_or(0)
214 .min(28);
215 let plural = |count: usize| if count == 1 { "row" } else { "rows" };
216 let mut lines = rows
217 .iter()
218 .map(|row| {
219 let name = columns_label(std::slice::from_ref(&row.column), name_width);
220 let pad = name_width.saturating_sub(crate::glyphs::display_width(&name));
221 format!(
222 "{name}{} {:>7} {} {}",
223 " ".repeat(pad),
224 percent(row.affected_rows, row.evaluated_rows),
225 numfmt::group_chrome(row.affected_rows),
226 plural(row.affected_rows)
227 )
228 })
229 .collect::<Vec<_>>();
230 let least = rows.iter().map(|row| row.affected_rows).max().unwrap_or(0);
231 let most = rows
232 .iter()
233 .map(|row| row.affected_rows)
234 .sum::<usize>()
235 .min(self.evaluated_rows.max(least));
236 lines.push(if least == most {
237 format!(
238 "Rows with any of them: {} {}",
239 numfmt::group_chrome(least),
240 plural(least)
241 )
242 } else {
243 format!(
244 "Rows with any of them: {} to {}, not counted",
245 numfmt::group_chrome(least),
246 numfmt::group_chrome(most)
247 )
248 });
249 lines
250 }
251}
252
253#[derive(Debug, Clone)]
255pub enum EvidenceRows {
256 Matching(Expr),
258 Files(QualityScope),
260 Duplicates,
263}
264
265#[derive(Debug, Clone, Default, PartialEq, Eq)]
268pub struct FindingsView {
269 pub column: Option<String>,
271 pub check: Option<&'static str>,
273 pub order: FindingOrder,
274}
275
276#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
278pub enum FindingOrder {
279 #[default]
281 Ranked,
282 Rows,
284 Rate,
286}
287
288impl FindingOrder {
289 pub fn next(self) -> Self {
290 match self {
291 Self::Ranked => Self::Rows,
292 Self::Rows => Self::Rate,
293 Self::Rate => Self::Ranked,
294 }
295 }
296
297 pub fn chip(self) -> &'static str {
299 match self {
300 Self::Ranked => "Ranked",
301 Self::Rows => "By Rows",
302 Self::Rate => "By Rate",
303 }
304 }
305}
306
307impl FindingsView {
308 pub fn narrowed(&self) -> bool {
309 self.column.is_some() || self.check.is_some()
310 }
311
312 fn admits(&self, finding: &Finding) -> bool {
313 self.column
314 .as_ref()
315 .is_none_or(|column| finding.columns.contains(column))
316 && self
317 .check
318 .is_none_or(|check| finding.check() == Some(check))
319 }
320
321 pub fn shown(&self, report: &QualityReport) -> Vec<usize> {
324 let findings = &report.findings;
325 let mut shown = (0..findings.len())
326 .filter(|index| self.admits(&findings[*index]))
327 .collect::<Vec<_>>();
328 let by = |left: &usize, right: &usize| {
329 let (left, right) = (&findings[*left], &findings[*right]);
330 left.severity
331 .cmp(&right.severity)
332 .then_with(|| match self.order {
333 FindingOrder::Ranked => std::cmp::Ordering::Equal,
334 FindingOrder::Rows => right.affected_rows.cmp(&left.affected_rows),
335 FindingOrder::Rate => (right.affected_rows as u128
337 * left.evaluated_rows.max(1) as u128)
338 .cmp(&(left.affected_rows as u128 * right.evaluated_rows.max(1) as u128)),
339 })
340 };
341 shown.sort_by(by);
342 shown
343 }
344
345 pub fn selected<'a>(&self, report: &'a QualityReport, position: usize) -> Option<&'a Finding> {
347 self.shown(report)
348 .get(position)
349 .and_then(|index| report.findings.get(*index))
350 }
351
352 pub fn describe(&self) -> Vec<String> {
355 let mut parts = Vec::new();
356 if let Some(column) = &self.column {
357 parts.push(format!("column {column}"));
358 }
359 if let Some(check) = self.check {
360 parts.push(check.to_string());
361 }
362 match self.order {
363 FindingOrder::Ranked => {}
364 FindingOrder::Rows => parts.push("by rows".to_string()),
365 FindingOrder::Rate => parts.push("by rate".to_string()),
366 }
367 parts
368 }
369}
370
371pub fn column_choices(
374 report: &QualityReport,
375 results: &DataQualityResults,
376) -> Vec<(String, usize)> {
377 results
378 .columns
379 .iter()
380 .map(|profile| {
381 let count = report
382 .findings
383 .iter()
384 .filter(|finding| finding.kind.is_some() && finding.columns.contains(&profile.name))
385 .count();
386 (profile.name.clone(), count)
387 })
388 .collect()
389}
390
391pub fn check_choices(report: &QualityReport) -> Vec<(&'static str, usize)> {
394 let mut choices: Vec<(&'static str, usize)> = Vec::new();
395 for check in report.findings.iter().filter_map(Finding::check) {
396 match choices.iter_mut().find(|(name, _)| *name == check) {
397 Some((_, count)) => *count += 1,
398 None => choices.push((check, 1)),
399 }
400 }
401 choices
402}
403
404#[derive(Debug, Clone, Default)]
405pub struct QualityReport {
406 pub findings: Vec<Finding>,
407 pub problems: usize,
408 pub notes: usize,
409 pub clean_columns: usize,
410 pub total_columns: usize,
411 pub column_status: Vec<Severity>,
413 pub column_findings: Vec<Vec<&'static str>>,
415 pub metadata_only: bool,
417 pub no_rows: bool,
420}
421
422pub fn build_report(results: &DataQualityResults) -> QualityReport {
423 let mut findings = Vec::new();
424 let mut groups = BTreeMap::<(u8, usize, String), Vec<usize>>::new();
426 for (index, observation) in results.observations.iter().enumerate() {
427 let always_missing = observation.kind == ObservationKind::Nulls
428 && observation.evaluated_rows > 0
429 && observation.affected_rows == observation.evaluated_rows;
430 let mostly_missing = observation.affected_rows * 2 > observation.evaluated_rows;
431 let same_rows = results.shared_nulls.iter().any(|shared| {
434 shared.same_rows()
435 && shared.columns.len() > 1
436 && shared.null_rows == observation.affected_rows
437 && shared.columns.contains(&observation.column)
438 });
439 let kind = observation.kind as u8 + 1;
440 let key = match observation.kind {
441 ObservationKind::Nulls if always_missing => (0, 0, String::new()),
442 ObservationKind::Nulls if mostly_missing => (kind, usize::MAX, "mostly".to_string()),
443 ObservationKind::Nulls if same_rows => (kind, observation.affected_rows, String::new()),
444 ObservationKind::Nulls => (kind, usize::MAX, String::new()),
445 ObservationKind::Empty | ObservationKind::Whitespace => {
446 (kind, observation.affected_rows, String::new())
447 }
448 ObservationKind::Constant => (kind, 0, String::new()),
449 ObservationKind::CategoryVariants => (kind, 0, observation.column.clone()),
450 ObservationKind::KeyRepeated | ObservationKind::KeyMissing => (kind, 0, String::new()),
452 _ => (u8::MAX, index, String::new()),
454 };
455 groups.entry(key).or_default().push(index);
456 }
457 for indices in groups.into_values() {
458 findings.push(finding(results, &indices));
459 }
460
461 let mut status = vec![Severity::Clean; results.columns.len()];
462 let mut titles = vec![Vec::new(); results.columns.len()];
463 let position = results
464 .columns
465 .iter()
466 .enumerate()
467 .map(|(index, profile)| (profile.name.as_str(), index))
468 .collect::<BTreeMap<_, _>>();
469 for finding in &findings {
470 for column in &finding.columns {
471 if let Some(&index) = position.get(column.as_str()) {
472 status[index] = status[index].min(finding.severity);
473 if !titles[index].contains(&finding.title) {
474 titles[index].push(finding.title);
475 }
476 }
477 }
478 }
479
480 findings.sort_by(|left, right| {
481 left.severity
482 .cmp(&right.severity)
483 .then_with(|| rank(left).cmp(&rank(right)))
484 .then_with(|| right.affected_rows.cmp(&left.affected_rows))
485 .then_with(|| left.columns.cmp(&right.columns))
486 });
487 let problems = findings
488 .iter()
489 .filter(|finding| finding.severity == Severity::Problem)
490 .count();
491 let notes = findings.len() - problems;
492 let metadata_only = results.precision == QualityPrecision::Metadata;
493 let no_rows = !metadata_only && results.evaluated_rows == 0;
494 let clean = results
495 .columns
496 .iter()
497 .zip(&status)
498 .filter(|(_, severity)| **severity == Severity::Clean)
499 .map(|(profile, _)| profile.name.clone())
500 .collect::<Vec<_>>();
501 let clean_columns = if no_rows { 0 } else { clean.len() };
502 if !clean.is_empty() && !metadata_only && !no_rows {
503 findings.push(Finding {
504 severity: Severity::Clean,
505 kind: None,
506 title: "No findings",
507 summary: format!(
508 "{} {}",
509 numfmt::group_chrome(clean.len()),
510 if clean.len() == 1 {
511 "column"
512 } else {
513 "columns"
514 }
515 ),
516 columns: clean,
517 observations: Vec::new(),
518 affected_rows: 0,
519 evaluated_rows: results.evaluated_rows,
520 same_rows: false,
521 });
522 }
523 QualityReport {
524 findings,
525 problems,
526 notes,
527 clean_columns,
528 total_columns: results.columns.len(),
529 column_status: status,
530 column_findings: titles,
531 metadata_only,
532 no_rows,
533 }
534}
535
536fn rank(finding: &Finding) -> u8 {
538 match finding.kind {
539 Some(ObservationKind::TypeConflict) => 0,
540 Some(ObservationKind::Absent) => 1,
541 Some(ObservationKind::KeyRepeated) => 2,
543 Some(ObservationKind::KeyMissing | ObservationKind::RequiredMissing) => 2,
544 Some(ObservationKind::NotAllowed | ObservationKind::OutOfRange) => 2,
545 Some(ObservationKind::UnparsedNumber) => 2,
546 Some(ObservationKind::DuplicateRows) => 2,
547 Some(ObservationKind::UnparsedTime) => 3,
548 Some(ObservationKind::Nulls) if finding.severity == Severity::Problem => 3,
549 Some(ObservationKind::Nulls) if finding.title == "Mostly missing" => 9,
550 Some(ObservationKind::NonFinite) => 4,
551 Some(ObservationKind::Clipping) => 4,
552 Some(ObservationKind::ZeroRuns) => 8,
553 Some(ObservationKind::DcOffset) => 12,
554 Some(ObservationKind::CategoryVariants) => 5,
555 Some(ObservationKind::Whitespace) => 6,
556 Some(ObservationKind::Empty) => 7,
557 Some(ObservationKind::Nulls) => 10,
558 Some(ObservationKind::KeyLike) => 11,
559 Some(ObservationKind::ParseableText) => 12,
560 Some(ObservationKind::Constant) => 13,
561 None => 20,
562 }
563}
564
565fn finding(results: &DataQualityResults, indices: &[usize]) -> Finding {
566 let mut indices = indices.to_vec();
567 if matches!(
570 results.observations[indices[0]].kind,
571 ObservationKind::Nulls | ObservationKind::CategoryVariants
572 ) {
573 indices.sort_by_key(|index| std::cmp::Reverse(results.observations[*index].affected_rows));
574 }
575 let indices = indices.as_slice();
576 let first = &results.observations[indices[0]];
577 let kind = first.kind;
578 let mut columns = Vec::new();
579 for index in indices {
580 let column = &results.observations[*index].column;
581 if !columns.contains(column) {
582 columns.push(column.clone());
583 }
584 }
585 let grouped = columns.len() > 1;
586 let always_missing = kind == ObservationKind::Nulls
587 && first.evaluated_rows > 0
588 && first.affected_rows == first.evaluated_rows;
589 let mostly_missing = kind == ObservationKind::Nulls
590 && !always_missing
591 && first.affected_rows * 2 > first.evaluated_rows;
592 let severity = match kind {
593 ObservationKind::Nulls if always_missing => Severity::Problem,
594 ObservationKind::Nulls
595 | ObservationKind::Constant
596 | ObservationKind::ParseableText
597 | ObservationKind::KeyLike
598 | ObservationKind::ZeroRuns
599 | ObservationKind::DcOffset => Severity::Note,
600 ObservationKind::Empty
601 | ObservationKind::Whitespace
602 | ObservationKind::NonFinite
603 | ObservationKind::DuplicateRows
604 | ObservationKind::CategoryVariants
605 | ObservationKind::Absent
606 | ObservationKind::TypeConflict
607 | ObservationKind::UnparsedTime
608 | ObservationKind::KeyRepeated
609 | ObservationKind::KeyMissing
610 | ObservationKind::RequiredMissing
611 | ObservationKind::NotAllowed
612 | ObservationKind::OutOfRange
613 | ObservationKind::UnparsedNumber
614 | ObservationKind::Clipping => Severity::Problem,
615 };
616 let profile = results
617 .columns
618 .iter()
619 .find(|profile| profile.name == first.column);
620 let reading = profile.and_then(text_reading).map(|(_, reading)| reading);
621 let shared = results.shared_nulls.iter().find(|shared| {
622 shared.null_rows == first.affected_rows
623 && columns.iter().all(|column| shared.columns.contains(column))
624 });
625 let same_rows =
626 kind == ObservationKind::Nulls && grouped && shared.is_some_and(|s| s.same_rows());
627 let title = match kind {
628 ObservationKind::Nulls if always_missing => "Always missing",
629 ObservationKind::Nulls if mostly_missing => "Mostly missing",
630 ObservationKind::Nulls if same_rows => "Missing together",
632 ObservationKind::Nulls => "Missing values",
633 ObservationKind::Empty => "Empty text",
634 ObservationKind::Whitespace => "Blank text",
635 ObservationKind::NonFinite => "NaN or infinite",
636 ObservationKind::Constant => "Single value",
637 ObservationKind::ParseableText => match (reading, profile) {
638 (Some(reading), Some(profile)) if reading.is_number() && is_code(profile) => {
639 "Codes as text"
640 }
641 (Some(TextReading::Datetime | TextReading::Date), _) => "Dates as text",
642 _ => "Numbers as text",
643 },
644 ObservationKind::DuplicateRows => "Duplicate rows",
645 ObservationKind::CategoryVariants => "Mixed spellings",
646 ObservationKind::Absent => "Missing in files",
647 ObservationKind::TypeConflict => "Type mismatch",
648 ObservationKind::KeyLike => "Nearly unique",
649 ObservationKind::UnparsedTime => "Unparsed times",
650 ObservationKind::KeyRepeated => "Repeated key",
651 ObservationKind::KeyMissing => "Incomplete key",
652 ObservationKind::RequiredMissing => "Required, missing",
653 ObservationKind::NotAllowed => "Not allowed",
654 ObservationKind::OutOfRange => "Out of range",
655 ObservationKind::UnparsedNumber => "Unparsed numbers",
656 ObservationKind::Clipping => "Clipping",
657 ObservationKind::ZeroRuns => "Runs of zeros",
658 ObservationKind::DcOffset => "DC offset",
659 };
660 let affected_rows = match kind {
661 ObservationKind::CategoryVariants => indices
663 .iter()
664 .map(|index| results.observations[*index].affected_rows)
665 .sum(),
666 _ => first.affected_rows,
667 };
668 let rows_each = |count: usize, each: &str| {
669 format!(
670 "{} {}{each} ({})",
671 numfmt::group_chrome(count),
672 if count == 1 { "row" } else { "rows" },
673 percent(count, first.evaluated_rows)
674 )
675 };
676 let rows = |count: usize| rows_each(count, "");
677 let summary = match kind {
678 ObservationKind::Nulls if always_missing => "no value in any row".to_string(),
679 ObservationKind::Nulls if same_rows => rows(first.affected_rows),
680 ObservationKind::Nulls if grouped => {
681 let fewest = results.observations[indices[indices.len() - 1]].affected_rows;
682 let (low, high) = (
683 percent(fewest, first.evaluated_rows),
684 percent(first.affected_rows, first.evaluated_rows),
685 );
686 if low == high {
687 rows_each(first.affected_rows, " each")
688 } else {
689 format!("{low} to {high} per column")
690 }
691 }
692 ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace
693 if grouped =>
694 {
695 rows_each(first.affected_rows, " each")
696 }
697 ObservationKind::Nulls
698 | ObservationKind::Empty
699 | ObservationKind::Whitespace
700 | ObservationKind::NonFinite => rows(first.affected_rows),
701 ObservationKind::UnparsedTime => format!(
702 "{} {} ({})",
703 numfmt::group_chrome(first.affected_rows),
704 if first.affected_rows == 1 {
705 "value"
706 } else {
707 "values"
708 },
709 percent(first.affected_rows, first.evaluated_rows)
710 ),
711 ObservationKind::Constant if grouped => "one value each".to_string(),
712 ObservationKind::Constant => profile
713 .and_then(|profile| profile.dominant_value.as_ref())
714 .map(|value| format!("always {}", quoted(value, 24)))
715 .unwrap_or_else(|| "one value".to_string()),
716 ObservationKind::ParseableText => match (reading, profile) {
717 (Some(_), Some(profile)) if is_code(profile) => code_shape(profile),
718 (Some(reading), Some(profile)) => format!(
719 "{} parse as {}",
720 percent(first.affected_rows, profile.non_null_rows()),
721 reading.label()
722 ),
723 _ => first.fact.clone(),
724 },
725 ObservationKind::DuplicateRows => match results.identity.as_ref() {
726 Some(identity) => format!(
727 "{} extra {}",
728 numfmt::group_chrome(identity.extra_rows),
729 if identity.extra_rows == 1 {
730 "copy"
731 } else {
732 "copies"
733 }
734 ),
735 None => first.fact.clone(),
736 },
737 ObservationKind::CategoryVariants => {
738 let example = results
739 .category_variants
740 .iter()
741 .find(|group| Some(&group.normalized) == first.normalized_category.as_ref())
742 .map(|group| {
743 group
744 .variants
745 .iter()
746 .take(2)
747 .map(|(value, _)| quoted(value, 24))
748 .collect::<Vec<_>>()
749 .join(" vs ")
750 })
751 .unwrap_or_default();
752 if indices.len() == 1 {
754 example
755 } else {
756 format!("{} values spelled more than one way", indices.len())
757 }
758 }
759 ObservationKind::KeyLike => {
760 let unique = profile
761 .and_then(ColumnQualityProfile::uniqueness_rate)
762 .map(|rate| format!(" ({:.1}% unique)", rate * 100.0))
763 .unwrap_or_default();
764 format!(
765 "{} repeated{unique}",
766 numfmt::group_chrome(first.affected_rows)
767 )
768 }
769 ObservationKind::Absent | ObservationKind::TypeConflict => first.fact.clone(),
770 ObservationKind::Clipping | ObservationKind::ZeroRuns | ObservationKind::DcOffset => {
773 first.fact.clone()
774 }
775 ObservationKind::KeyRepeated => {
776 match results.intent.as_ref().and_then(|i| i.key.as_ref()) {
777 Some(key) => format!(
778 "{} {} repeat; {} extra {}",
779 numfmt::group_chrome(key.groups),
780 if key.groups == 1 { "value" } else { "values" },
781 numfmt::group_chrome(key.extra_rows),
782 if key.extra_rows == 1 { "row" } else { "rows" }
783 ),
784 None => first.fact.clone(),
785 }
786 }
787 ObservationKind::KeyMissing | ObservationKind::RequiredMissing => rows(first.affected_rows),
788 ObservationKind::NotAllowed
789 | ObservationKind::OutOfRange
790 | ObservationKind::UnparsedNumber => format!(
791 "{} {} ({})",
792 numfmt::group_chrome(first.affected_rows),
793 if first.affected_rows == 1 {
794 "value"
795 } else {
796 "values"
797 },
798 percent(first.affected_rows, first.evaluated_rows)
799 ),
800 };
801 Finding {
802 severity,
803 kind: Some(kind),
804 title,
805 columns,
806 observations: indices.to_vec(),
807 affected_rows,
808 evaluated_rows: first.evaluated_rows,
809 summary,
810 same_rows,
811 }
812}
813
814fn is_code(profile: &ColumnQualityProfile) -> bool {
817 profile.leading_zero_count.is_some_and(|count| count > 0)
818 || (profile.min_length.is_some()
819 && profile.min_length == profile.max_length
820 && profile.min_length.is_some_and(|length| length > 1))
821}
822
823fn code_shape(profile: &ColumnQualityProfile) -> String {
824 let zeros = profile.leading_zero_count.unwrap_or(0);
825 match (profile.min_length, profile.max_length) {
826 (Some(min), Some(max)) if min == max && zeros > 0 => {
827 format!("{min} digits, leading zeros")
828 }
829 (Some(min), Some(max)) if min == max => format!("{min} digits each"),
830 _ => format!("{} with a leading zero", numfmt::group_chrome(zeros)),
831 }
832}
833
834pub fn percent(count: usize, of: usize) -> String {
835 if of == 0 {
836 return "-".to_string();
837 }
838 let value = count as f64 / of as f64 * 100.0;
839 if value > 0.0 && value < 1.0 {
841 format!("{value:.2}%")
842 } else {
843 format!("{value:.1}%")
844 }
845}
846
847pub(crate) fn quoted(value: &str, width: usize) -> String {
848 let text = format!("{value:?}");
849 if crate::glyphs::display_width(&text) <= width {
850 text
851 } else {
852 let mut cut = String::new();
853 for ch in text.chars() {
854 if crate::glyphs::display_width(&cut) + 2 > width {
855 break;
856 }
857 cut.push(ch);
858 }
859 format!("{cut}{}", crate::glyphs::get().ellipsis)
860 }
861}
862
863pub fn columns_label(columns: &[String], width: usize) -> String {
865 let mut label = String::new();
866 for (index, column) in columns.iter().enumerate() {
867 let candidate = if label.is_empty() {
868 column.clone()
869 } else {
870 format!("{label}, {column}")
871 };
872 let rest = columns.len() - index - 1;
873 let suffix = if rest > 0 {
874 format!(" +{rest}")
875 } else {
876 String::new()
877 };
878 let needed =
879 crate::glyphs::display_width(&candidate) + crate::glyphs::display_width(&suffix);
880 if !label.is_empty() && needed > width {
881 return format!("{label} +{}", rest + 1);
882 }
883 label = candidate;
884 }
885 label
886}
887
888#[derive(Debug, Clone, PartialEq, Eq)]
891pub enum Outcome {
892 Passed,
893 Found {
894 tier: Severity,
895 detail: String,
896 },
897 Skipped(&'static str),
899 Unavailable(&'static str),
902}
903
904impl Outcome {
905 pub fn ran(&self) -> bool {
906 matches!(self, Self::Passed | Self::Found { .. })
907 }
908}
909
910#[derive(Debug, Clone)]
913pub struct Check {
914 pub name: &'static str,
915 pub looks_for: &'static str,
916 pub applies_to: String,
917 pub outcome: Outcome,
918 pub basis: QualityPrecision,
920}
921
922pub const CHECKS_SHOWN: usize = 6;
924
925const NO_ROWS: &str = "no rows to check";
927
928pub fn checks(results: &DataQualityResults, report: &QualityReport) -> Vec<Check> {
930 use polars::prelude::DataType;
931 let columns = &results.columns;
932 let count = |filter: &dyn Fn(&ColumnQualityProfile) -> bool| {
933 columns.iter().filter(|profile| filter(profile)).count()
934 };
935 let text = |profile: &ColumnQualityProfile| {
936 matches!(profile.dtype, DataType::String | DataType::Categorical(..))
937 };
938 let all = columns.len();
939 let floats = count(&|profile| profile.dtype.is_float());
940 let texts = count(&text);
941 let keys = count(&|profile| profile.dtype.is_integer() || text(profile));
942 let reach = |count: usize, kind: &str| {
943 let noun = if count == 1 { "column" } else { "columns" };
944 if kind.is_empty() {
945 format!("{} {noun}", numfmt::group_chrome(count))
946 } else {
947 format!("{} {kind} {noun}", numfmt::group_chrome(count))
948 }
949 };
950 let values_read = !report.metadata_only;
951 let outcome = |titles: &[&str], applies: usize, none: &'static str| {
954 if applies == 0 {
955 return Outcome::Skipped(none);
956 }
957 if !values_read {
958 return Outcome::Unavailable("values not read");
959 }
960 if report.no_rows {
961 return Outcome::Unavailable(NO_ROWS);
962 }
963 found(report, titles)
964 };
965 let files = results.source_files.filter(|files| *files > 1);
966 let by_files = |titles: &[&str]| match files {
967 Some(_) => found(report, titles),
968 None => Outcome::Skipped("needs several files"),
969 };
970 let files_reach = match (files, results.footers_read) {
972 (Some(files), Some(read)) if read < files => format!(
973 "{} of {} files",
974 numfmt::group_chrome(read),
975 numfmt::group_chrome(files)
976 ),
977 (Some(files), _) => format!("{} files", numfmt::group_chrome(files)),
978 (None, _) => "files".to_string(),
979 };
980 let values = results.precision;
981 let mut checks = Vec::new();
982 if let Some(intent) = &results.intent {
984 checks.push(Check {
985 name: INTENT_CHECK,
986 looks_for: "values against the key and rules declared",
987 applies_to: reach(intent.declared.len(), "declared"),
988 outcome: if !intent.measured {
989 Outcome::Unavailable("values not read")
990 } else if report.no_rows {
991 Outcome::Unavailable(NO_ROWS)
992 } else {
993 found(report, &INTENT_TITLES)
994 },
995 basis: intent.precision,
996 });
997 }
998 checks.extend([
999 Check {
1000 name: "Missing values",
1001 looks_for: "nulls in any column",
1002 applies_to: reach(all, ""),
1003 outcome: outcome(
1004 &[
1005 "Missing values",
1006 "Missing together",
1007 "Mostly missing",
1008 "Always missing",
1009 ],
1010 all,
1011 "no columns",
1012 ),
1013 basis: values,
1014 },
1015 Check {
1016 name: "NaN or infinite",
1017 looks_for: "NaN or +/-infinity in float columns",
1018 applies_to: reach(floats, "float"),
1019 outcome: outcome(&["NaN or infinite"], floats, "no float columns"),
1020 basis: values,
1021 },
1022 Check {
1023 name: "Duplicate rows",
1024 looks_for: "rows identical in every column",
1025 applies_to: "whole rows".to_string(),
1026 outcome: match results.identity.as_ref() {
1027 _ if !values_read => Outcome::Unavailable("values not read"),
1028 _ if report.no_rows => Outcome::Unavailable(NO_ROWS),
1029 Some(identity) if identity.extra_rows > 0 => Outcome::Found {
1030 tier: Severity::Problem,
1031 detail: format!("{} extra rows", numfmt::group_chrome(identity.extra_rows)),
1032 },
1033 Some(_) => Outcome::Passed,
1034 None => Outcome::Unavailable("not measured"),
1035 },
1036 basis: values,
1037 },
1038 Check {
1039 name: "Blank text",
1040 looks_for: "text that is empty or only whitespace",
1041 applies_to: reach(texts, "text"),
1042 outcome: outcome(&["Blank text", "Empty text"], texts, "no text columns"),
1043 basis: values,
1044 },
1045 Check {
1046 name: "Mixed spellings",
1047 looks_for: "one value in several cases or spacings",
1048 applies_to: reach(texts, "text"),
1049 outcome: outcome(&["Mixed spellings"], texts, "no text columns"),
1050 basis: values,
1051 },
1052 Check {
1053 name: "Type mismatch",
1054 looks_for: "a column typed differently by some files",
1055 applies_to: files_reach.clone(),
1056 outcome: by_files(&["Type mismatch"]),
1057 basis: QualityPrecision::Metadata,
1058 },
1059 Check {
1060 name: "Missing in files",
1061 looks_for: "a column some files do not have",
1062 applies_to: files_reach,
1063 outcome: by_files(&["Missing in files"]),
1064 basis: QualityPrecision::Metadata,
1065 },
1066 Check {
1067 name: "Numbers as text",
1068 looks_for: "text that reads as numbers or dates",
1069 applies_to: reach(texts, "text"),
1070 outcome: outcome(
1071 &["Numbers as text", "Dates as text", "Codes as text"],
1072 texts,
1073 "no text columns",
1074 ),
1075 basis: values,
1076 },
1077 Check {
1078 name: "Nearly unique",
1079 looks_for: "a would-be key whose values repeat",
1080 applies_to: reach(keys, "integer/text"),
1081 outcome: if keys > 0 && values_read && results.precision != QualityPrecision::Exact {
1082 Outcome::Unavailable("needs every row checked")
1083 } else {
1084 match outcome(&["Nearly unique"], keys, "no integer or text columns") {
1085 Outcome::Passed if declared_key_repeats(results) => Outcome::Found {
1088 tier: Severity::Problem,
1089 detail: "the declared key repeats".to_string(),
1090 },
1091 outcome => outcome,
1092 }
1093 },
1094 basis: values,
1095 },
1096 Check {
1097 name: "Single value",
1098 looks_for: "a column with one value throughout",
1099 applies_to: reach(all, ""),
1100 outcome: outcome(&["Single value"], all, "no columns"),
1101 basis: values,
1102 },
1103 ]);
1104 checks
1105}
1106
1107fn declared_key_repeats(results: &DataQualityResults) -> bool {
1110 results
1111 .intent
1112 .as_ref()
1113 .and_then(|intent| intent.key.as_ref())
1114 .is_some_and(|key| key.columns.len() == 1 && key.rows_involved > 0)
1115}
1116
1117pub const INTENT_CHECK: &str = "Column intent";
1119
1120pub const INTENT_TITLES: [&str; 6] = [
1122 "Repeated key",
1123 "Incomplete key",
1124 "Required, missing",
1125 "Not allowed",
1126 "Out of range",
1127 "Unparsed numbers",
1128];
1129
1130fn found(report: &QualityReport, titles: &[&str]) -> Outcome {
1131 let mut columns = Vec::new();
1132 let mut tier = Severity::Note;
1133 for finding in report
1134 .findings
1135 .iter()
1136 .filter(|finding| finding.kind.is_some() && titles.contains(&finding.title))
1137 {
1138 tier = tier.min(finding.severity);
1139 for column in &finding.columns {
1140 if !columns.contains(column) {
1141 columns.push(column.clone());
1142 }
1143 }
1144 }
1145 match columns.len() {
1146 0 => Outcome::Passed,
1147 count => Outcome::Found {
1148 tier,
1149 detail: format!(
1150 "{} {}",
1151 numfmt::group_chrome(count),
1152 if count == 1 { "column" } else { "columns" }
1153 ),
1154 },
1155 }
1156}
1157
1158pub const THIN_SEGMENT_ROWS: usize = 30;
1161
1162#[derive(Debug, Clone, Default, PartialEq, Eq)]
1167pub struct Coverage {
1168 pub exact: usize,
1170 pub sampled: usize,
1172 pub metadata: usize,
1174 pub skipped: usize,
1176 pub unavailable: Vec<(&'static str, Vec<&'static str>)>,
1179 pub rows: Vec<String>,
1182 pub limits: Vec<String>,
1185}
1186
1187impl Coverage {
1188 pub fn checks(&self) -> Vec<String> {
1191 let unavailable = self
1192 .unavailable
1193 .iter()
1194 .map(|(_, names)| names.len())
1195 .sum::<usize>();
1196 [
1197 (self.exact, QualityPrecision::Exact.label()),
1198 (self.sampled, QualityPrecision::Sampled.label()),
1199 (self.metadata, QualityPrecision::Metadata.label()),
1200 (self.skipped, "skipped"),
1201 (unavailable, "unavailable"),
1202 ]
1203 .into_iter()
1204 .filter(|(count, _)| *count > 0)
1205 .map(|(count, label)| format!("{} {label}", numfmt::group_chrome(count)))
1206 .collect()
1207 }
1208
1209 pub fn limits(&self) -> Vec<String> {
1211 self.unavailable
1212 .iter()
1213 .map(|(reason, names)| match names.as_slice() {
1214 [name] => format!("{name}: {reason}"),
1215 _ => format!("{} checks: {reason}", names.len()),
1216 })
1217 .chain(self.limits.iter().cloned())
1218 .collect()
1219 }
1220}
1221
1222pub fn coverage(
1223 results: &DataQualityResults,
1224 checks: &[Check],
1225 plan: &DataQualityPlan,
1226) -> Coverage {
1227 let mut coverage = Coverage::default();
1228 for check in checks {
1229 match &check.outcome {
1230 Outcome::Skipped(_) => coverage.skipped += 1,
1231 Outcome::Unavailable(reason) => {
1232 match coverage
1233 .unavailable
1234 .iter_mut()
1235 .find(|(known, _)| known == reason)
1236 {
1237 Some((_, names)) => names.push(check.name),
1238 None => coverage.unavailable.push((reason, vec![check.name])),
1239 }
1240 }
1241 Outcome::Passed | Outcome::Found { .. } => match check.basis {
1242 QualityPrecision::Exact => coverage.exact += 1,
1243 QualityPrecision::Metadata => coverage.metadata += 1,
1244 QualityPrecision::Sampled | QualityPrecision::Estimated => coverage.sampled += 1,
1245 },
1246 }
1247 }
1248
1249 let count = numfmt::group_chrome;
1250 let evaluated = results.evaluated_rows;
1251 coverage
1252 .rows
1253 .push(match (results.precision, results.total_rows) {
1254 (QualityPrecision::Metadata, _) => "none read, file metadata only".to_string(),
1255 _ if evaluated == 0 => "none: the scope has no rows".to_string(),
1256 (QualityPrecision::Exact, _) => format!("all {} read, exact", count(evaluated)),
1257 (_, Some(total)) => format!(
1258 "{} of {} sampled ({})",
1259 count(evaluated),
1260 count(total),
1261 percent(evaluated, total)
1262 ),
1263 (_, None) => format!("{} sampled, total not counted", count(evaluated)),
1264 });
1265 if let Some(each) = results.per_value {
1266 coverage
1267 .rows
1268 .push(format!("up to {} per value", count(each)));
1269 }
1270 if let Some(reads) = results
1272 .reads
1273 .filter(|_| results.precision != QualityPrecision::Metadata)
1274 {
1275 if reads.reads == 0 {
1276 coverage.rows.push("no source read".to_string());
1277 } else if reads.counted == reads.reads {
1278 coverage
1279 .rows
1280 .push(format!("{} traversed", count(reads.rows)));
1281 } else if reads.counted > 0 {
1282 coverage
1283 .rows
1284 .push(format!("at least {} traversed", count(reads.rows)));
1285 }
1286 if let Some(copy) = reads.copy {
1287 let bytes = crate::widgets::info::format_bytes(copy.bytes);
1288 coverage.rows.push(if copy.fetched {
1289 format!("passes read a local copy, fetched once ({bytes})")
1290 } else {
1291 format!("passes read a local copy fetched earlier ({bytes})")
1292 });
1293 }
1294 }
1295
1296 let segments = &results.segments;
1297 if matches!(
1298 results.precision,
1299 QualityPrecision::Sampled | QualityPrecision::Estimated
1300 ) && segments.len() > 1
1301 {
1302 let thin = segments
1303 .iter()
1304 .filter(|segment| segment.evaluated_rows < THIN_SEGMENT_ROWS)
1305 .count();
1306 if thin > 0 {
1307 coverage.limits.push(format!(
1308 "{} of {} segments under {THIN_SEGMENT_ROWS} sampled rows",
1309 count(thin),
1310 count(segments.len())
1311 ));
1312 }
1313 }
1314 if !results.unsampled_segments.is_empty() {
1315 coverage.limits.push(format!(
1316 "{} segments with rows, none sampled",
1317 count(results.unsampled_segments.len())
1318 ));
1319 }
1320 if let (Some(files), Some(read)) = (results.source_files, results.footers_read)
1321 && read < files
1322 {
1323 coverage.limits.push(format!(
1324 "footers of {} of {} files read",
1325 count(read),
1326 count(files)
1327 ));
1328 }
1329 if let Some(intent) = results.intent.as_ref().filter(|intent| intent.measured) {
1330 if intent.key.is_some() && intent.precision != QualityPrecision::Exact {
1332 coverage.limits.push(format!(
1333 "key repeats among {} sampled rows only",
1334 count(intent.evaluated_rows)
1335 ));
1336 }
1337 }
1338 if let Some(intent) = &results.intent
1339 && !intent.absent.is_empty()
1340 {
1341 coverage.limits.push(format!(
1342 "intent on {}: not in scope",
1343 columns_label(&intent.absent, 24)
1344 ));
1345 }
1346 if !plan.temporal_roles.is_empty() && plan.interval_pairs().is_empty() {
1347 coverage
1348 .limits
1349 .push("time roles form no interval".to_string());
1350 }
1351 coverage
1352}
1353
1354pub fn advice(finding: &Finding) -> Vec<String> {
1357 let lines: &[&str] = match (finding.kind, finding.title) {
1358 (Some(ObservationKind::Nulls), "Always missing") => {
1359 &["Carries nothing; check the load or a rename upstream"]
1360 }
1361 (Some(ObservationKind::Nulls), "Mostly missing") => {
1362 let rest = finding.evaluated_rows.saturating_sub(finding.affected_rows);
1363 return vec![
1364 if finding.columns.len() == 1 {
1365 format!(
1366 "Aggregates and joins see only {}",
1367 percent(rest, finding.evaluated_rows)
1368 )
1369 } else {
1370 "Aggregates and joins see only the filled rows".to_string()
1371 },
1372 "Check: filled only for some rows, or stopped at some point".to_string(),
1373 ];
1374 }
1375 (Some(ObservationKind::Nulls), "Missing together") => {
1376 &["Likely one cause: a join with no match, or a source with gaps"]
1377 }
1378 (Some(ObservationKind::Nulls), _) => {
1379 &["Check: clustered in some files or dates (Segments, by file or window)"]
1380 }
1381 (Some(ObservationKind::Empty), _) => {
1382 &["Counted as filled; treat as null if it means missing"]
1383 }
1384 (Some(ObservationKind::Whitespace), _) => {
1385 &["Counted as filled; trim to null if it means missing"]
1386 }
1387 (Some(ObservationKind::NonFinite), _) => {
1388 &["Sums and means become NaN; check for division by zero upstream"]
1389 }
1390 (Some(ObservationKind::Constant), _) => {
1391 &["Tells no rows apart; a stuck feed if it should vary"]
1392 }
1393 (Some(ObservationKind::ParseableText), "Codes as text") => {
1394 &["Fine as text; cast only for arithmetic"]
1395 }
1396 (Some(ObservationKind::ParseableText), "Dates as text") => {
1397 &["Sorts as text; parse as a date to filter by range"]
1398 }
1399 (Some(ObservationKind::ParseableText), _) => {
1400 &["Sorts as text (\"10\" before \"9\"); cast to a number to sum"]
1401 }
1402 (Some(ObservationKind::DuplicateRows), _) => {
1403 &["Counted more than once; check for a double load or a join fan-out"]
1404 }
1405 (Some(ObservationKind::CategoryVariants), _) => {
1406 &["Group-bys and joins split them; trim and normalize case"]
1407 }
1408 (Some(ObservationKind::Absent), _) => &["Check: files written before the column existed"],
1409 (Some(ObservationKind::TypeConflict), _) => {
1410 &["Values dropped, not converted; read as text in Info, or fix the writer"]
1411 }
1412 (Some(ObservationKind::KeyLike), _) => {
1413 &["Duplicates if it is a key; expected if it is a measurement"]
1414 }
1415 (Some(ObservationKind::UnparsedTime), _) => &[
1416 "Left out of time windows and intervals, not counted as missing",
1417 "Check: another format, or values that are not times (Setup, e)",
1418 ],
1419 (Some(ObservationKind::KeyRepeated), _) => &[
1420 "A key names one row; joins on it fan out and counts double",
1421 "Check: a double load, or a key that needs another column",
1422 ],
1423 (Some(ObservationKind::KeyMissing), _) => {
1424 &["Rows with no key cannot be joined or told apart by it"]
1425 }
1426 (Some(ObservationKind::RequiredMissing), _) => {
1427 &["Check: the load, or rows the source writes without it"]
1428 }
1429 (Some(ObservationKind::NotAllowed), _) => {
1430 &["Check: a new value upstream, or the allowed list (Setup, e)"]
1431 }
1432 (Some(ObservationKind::OutOfRange), _) => {
1433 &["Check: units, placeholders such as -1 or 9999, or the range (Setup, e)"]
1434 }
1435 (Some(ObservationKind::UnparsedNumber), _) => {
1436 &["A cast makes them null; check the values or the reading (Setup, e)"]
1437 }
1438 (Some(ObservationKind::Clipping), _) => {
1439 &["Cut flat at the limit; lower the gain at the source, nothing restores it"]
1440 }
1441 (Some(ObservationKind::ZeroRuns), _) => {
1442 &["Dropouts mid-recording; digital silence if at the start or end"]
1443 }
1444 (Some(ObservationKind::DcOffset), _) => {
1445 &["A constant bias that eats headroom; a high-pass filter removes it"]
1446 }
1447 (None, _) => &[],
1448 };
1449 lines.iter().map(|line| line.to_string()).collect()
1450}
1451
1452pub fn describe(finding: &Finding, results: &DataQualityResults) -> (String, Vec<String>) {
1455 let count = |value: usize| numfmt::group_chrome(value);
1456 let of = format!(
1457 "{} of {} rows ({})",
1458 count(finding.affected_rows),
1459 count(finding.evaluated_rows),
1460 percent(finding.affected_rows, finding.evaluated_rows)
1461 );
1462 let profile = |name: &str| results.columns.iter().find(|profile| profile.name == name);
1463 let observation = |index: &usize| &results.observations[*index];
1464 let grouped = finding.columns.len() > 1;
1465 let mut evidence = Vec::new();
1466 let headline = match finding.kind {
1467 None => format!(
1468 "{} {} passed every check",
1469 count(finding.columns.len()),
1470 if finding.columns.len() == 1 {
1471 "column"
1472 } else {
1473 "columns"
1474 }
1475 ),
1476 Some(ObservationKind::Nulls) if finding.severity == Severity::Problem => {
1477 format!("Null in all {} rows checked", count(finding.evaluated_rows))
1478 }
1479 Some(ObservationKind::Nulls) if finding.same_rows => {
1480 if finding.columns.len() == 2 {
1481 evidence.push("No row misses one without the other".to_string());
1482 format!("{of} null in both columns")
1483 } else {
1484 evidence.push("No row misses one of them without the others".to_string());
1485 format!("{of} null in all {} columns", finding.columns.len())
1486 }
1487 }
1488 Some(ObservationKind::Nulls) if finding.varied() => {
1489 evidence.extend(finding.breakdown(results));
1491 format!("Null rate in {} columns:", finding.columns.len())
1492 }
1493 Some(ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace)
1494 if finding.lists_columns() =>
1495 {
1496 evidence.extend(finding.breakdown(results));
1497 let what = match finding.kind {
1498 Some(ObservationKind::Nulls) => "null",
1499 Some(ObservationKind::Empty) => "empty strings",
1500 _ => "only spaces or tabs",
1501 };
1502 format!("{of} {what} in each column:")
1503 }
1504 Some(ObservationKind::Nulls) => format!("{of} null"),
1505 Some(ObservationKind::Empty) => format!("{of} empty strings"),
1506 Some(ObservationKind::Whitespace) => format!("{of} only spaces or tabs"),
1507 Some(ObservationKind::NonFinite) => {
1508 if let Some(profile) = profile(&finding.columns[0]) {
1509 evidence.push(format!(
1510 "NaN {}, +inf {}, -inf {}",
1511 count(profile.nan_count.unwrap_or(0)),
1512 count(profile.positive_infinity_count.unwrap_or(0)),
1513 count(profile.negative_infinity_count.unwrap_or(0))
1514 ));
1515 }
1516 format!("{of} NaN or infinite")
1517 }
1518 Some(ObservationKind::Constant) if grouped => {
1519 for column in &finding.columns {
1520 if let Some(value) = profile(column).and_then(|p| p.dominant_value.as_ref()) {
1521 evidence.push(format!("{column}: always {}", quoted(value, 40)));
1522 }
1523 }
1524 format!(
1525 "One value per column in {} rows",
1526 count(finding.evaluated_rows)
1527 )
1528 }
1529 Some(ObservationKind::Constant) => {
1530 match profile(&finding.columns[0]).and_then(|p| p.dominant_value.as_ref()) {
1531 Some(value) => format!(
1532 "Always {} in {} rows",
1533 quoted(value, 40),
1534 count(finding.evaluated_rows)
1535 ),
1536 None => format!("One value in {} rows", count(finding.evaluated_rows)),
1537 }
1538 }
1539 Some(ObservationKind::ParseableText) => {
1540 let column = &finding.columns[0];
1541 let profile = profile(column);
1542 let reading = profile.and_then(text_reading);
1543 if let (Some(profile), Some((parsed, reading))) = (profile, reading) {
1544 if let (Some(min), Some(max)) = (profile.min_length, profile.max_length) {
1545 evidence.push(if min == max {
1546 format!("{min} characters each")
1547 } else {
1548 format!("{min} to {max} characters")
1549 });
1550 }
1551 if let Some(zeros) = profile.leading_zero_count.filter(|zeros| *zeros > 0) {
1552 evidence.push(format!("{} with a leading zero", count(zeros)));
1553 }
1554 if let (Some(min), Some(max)) = (&profile.min, &profile.max) {
1555 evidence.push(format!("From {} to {}", quoted(min, 24), quoted(max, 24)));
1556 }
1557 let failed = profile.non_null_rows().saturating_sub(parsed);
1558 if failed > 0 {
1559 let examples = results.examples_of(ObservationKind::ParseableText, column);
1560 evidence.push(if examples.is_empty() {
1561 format!("{} do not parse", count(failed))
1562 } else {
1563 cut(
1564 &format!(
1565 "{} do not parse, such as {}",
1566 count(failed),
1567 examples.join(", ")
1568 ),
1569 EXAMPLE_WIDTH,
1570 )
1571 });
1572 }
1573 format!(
1574 "{} of {} values ({}) parse as {}",
1575 count(parsed),
1576 count(profile.non_null_rows()),
1577 percent(parsed, profile.non_null_rows()),
1578 reading.label()
1579 )
1580 } else {
1581 observation(&finding.observations[0]).fact.clone()
1582 }
1583 }
1584 Some(ObservationKind::DuplicateRows) => match results.identity.as_ref() {
1585 Some(identity) => {
1586 evidence.push(format!(
1587 "{} of {} rows ({}) have a copy",
1588 count(identity.rows_involved),
1589 count(identity.evaluated_rows),
1590 percent(identity.rows_involved, identity.evaluated_rows)
1591 ));
1592 if !identity.examples.is_empty() {
1593 evidence.push("Most copied:".to_string());
1594 }
1595 for example in &identity.examples {
1596 evidence.push(cut(
1597 &format!(
1598 "{}{} {}",
1599 crate::glyphs::get().times,
1600 example.copies,
1601 example.values.join(", ")
1602 ),
1603 EXAMPLE_WIDTH,
1604 ));
1605 }
1606 format!(
1607 "{} rows repeated; {} extra {}",
1608 count(identity.duplicate_groups),
1609 count(identity.extra_rows),
1610 if identity.extra_rows == 1 {
1611 "copy"
1612 } else {
1613 "copies"
1614 }
1615 )
1616 }
1617 None => observation(&finding.observations[0]).fact.clone(),
1618 },
1619 Some(ObservationKind::CategoryVariants) => {
1620 for index in &finding.observations {
1621 let normalized = observation(index).normalized_category.as_ref();
1622 let Some(group) = results
1623 .category_variants
1624 .iter()
1625 .find(|group| Some(&group.normalized) == normalized)
1626 else {
1627 continue;
1628 };
1629 let mut variants = group.variants.iter().collect::<Vec<_>>();
1630 variants.sort_by_key(|(_, rows)| std::cmp::Reverse(*rows));
1631 evidence.push(
1632 variants
1633 .into_iter()
1634 .take(4)
1635 .map(|(value, rows)| format!("{} ({})", quoted(value, 28), count(*rows)))
1636 .collect::<Vec<_>>()
1637 .join(" "),
1638 );
1639 }
1640 let values = finding.observations.len();
1641 format!(
1642 "{} {} spelled more than one way, {of}",
1643 count(values),
1644 if values == 1 { "value" } else { "values" }
1645 )
1646 }
1647 Some(ObservationKind::KeyLike) => {
1648 let column = &finding.columns[0];
1649 if let Some(profile) = profile(column) {
1650 if let (Some(value), Some(times)) =
1651 (&profile.dominant_value, profile.dominant_count)
1652 {
1653 evidence.push(format!(
1654 "Most repeated: {} ({} times)",
1655 quoted(value, 32),
1656 count(times)
1657 ));
1658 }
1659 format!(
1660 "{} distinct in {} rows; {} repeats",
1661 count(profile.distinct_count.unwrap_or(0)),
1662 count(profile.non_null_rows()),
1663 count(finding.affected_rows)
1664 )
1665 } else {
1666 observation(&finding.observations[0]).fact.clone()
1667 }
1668 }
1669 Some(ObservationKind::UnparsedTime) => {
1670 let observation = observation(&finding.observations[0]);
1671 if let Some(format) = &observation.time_format {
1672 evidence.push(format!("Read as {} for this study only", format.label()));
1673 }
1674 let examples = results.examples_of(ObservationKind::UnparsedTime, &observation.column);
1675 if !examples.is_empty() {
1676 evidence.push(cut(
1677 &format!("Such as {}", examples.join(", ")),
1678 EXAMPLE_WIDTH,
1679 ));
1680 }
1681 format!(
1682 "{} of {} values ({}) do not parse",
1683 count(finding.affected_rows),
1684 count(finding.evaluated_rows),
1685 percent(finding.affected_rows, finding.evaluated_rows)
1686 )
1687 }
1688 Some(
1689 kind @ (ObservationKind::KeyRepeated
1690 | ObservationKind::KeyMissing
1691 | ObservationKind::RequiredMissing
1692 | ObservationKind::NotAllowed
1693 | ObservationKind::OutOfRange
1694 | ObservationKind::UnparsedNumber),
1695 ) => describe_intent(kind, finding, results, &mut evidence),
1696 Some(
1697 kind @ (ObservationKind::Clipping
1698 | ObservationKind::ZeroRuns
1699 | ObservationKind::DcOffset),
1700 ) => {
1701 for index in &finding.observations {
1702 let observation = observation(index);
1703 evidence.push(format!("{}: {}", observation.column, observation.fact));
1704 }
1705 match kind {
1706 ObservationKind::Clipping => format!("{of} in runs at full scale"),
1707 ObservationKind::ZeroRuns => format!("{of} in runs of exact zeros"),
1708 _ => "Mean away from zero".to_string(),
1709 }
1710 }
1711 Some(ObservationKind::Absent | ObservationKind::TypeConflict) => {
1712 let observation = observation(&finding.observations[0]);
1713 evidence.push(format!(
1716 "{} of {} rows of the loaded source ({})",
1717 count(finding.affected_rows),
1718 count(finding.evaluated_rows),
1719 percent(finding.affected_rows, finding.evaluated_rows)
1720 ));
1721 for file in observation.files.iter().take(4) {
1722 let stored = file
1723 .stored_type
1724 .as_ref()
1725 .map(|dtype| format!(" as {dtype}"))
1726 .unwrap_or_default();
1727 let examples = if file.examples.is_empty() {
1728 String::new()
1729 } else {
1730 format!(
1731 ": {}",
1732 file.examples
1733 .iter()
1734 .map(|value| quoted(value, 20))
1735 .collect::<Vec<_>>()
1736 .join(", ")
1737 )
1738 };
1739 evidence.push(format!(
1740 "#{} {} ({} rows){stored}{examples}",
1741 file.number,
1742 file.name,
1743 count(file.rows)
1744 ));
1745 }
1746 if observation.files.len() > 4 {
1747 evidence.push(format!(
1748 "{} more {}",
1749 observation.files.len() - 4,
1750 crate::glyphs::get().ellipsis
1751 ));
1752 }
1753 upper_first(&observation.fact)
1754 }
1755 };
1756 (headline, evidence)
1757}
1758
1759const EXAMPLE_WIDTH: usize = 72;
1762
1763fn cut(text: &str, width: usize) -> String {
1765 if crate::glyphs::display_width(text) <= width {
1766 return text.to_string();
1767 }
1768 let ellipsis = crate::glyphs::get().ellipsis;
1769 format!(
1770 "{}{ellipsis}",
1771 crate::glyphs::take_columns(
1772 text,
1773 width.saturating_sub(crate::glyphs::display_width(ellipsis))
1774 )
1775 )
1776}
1777
1778fn describe_intent(
1781 kind: ObservationKind,
1782 finding: &Finding,
1783 results: &DataQualityResults,
1784 evidence: &mut Vec<String>,
1785) -> String {
1786 let count = numfmt::group_chrome;
1787 let Some(intent) = results.intent.as_ref() else {
1788 return finding.summary.clone();
1789 };
1790 let sampled = intent.precision != QualityPrecision::Exact;
1791 let rows_word = if sampled { "sampled rows" } else { "rows" };
1792 let share = percent(finding.affected_rows, finding.evaluated_rows);
1793 let examples = |values: &[(String, usize)]| {
1794 values
1795 .iter()
1796 .map(|(value, rows)| format!("{} ({})", quoted(value, 24), count(*rows)))
1797 .collect::<Vec<_>>()
1798 .join(" ")
1799 };
1800 let column = finding.columns.first().map(String::as_str).unwrap_or("");
1801 let check = intent.column(column);
1802 match kind {
1803 ObservationKind::KeyRepeated | ObservationKind::KeyMissing => {
1804 let Some(key) = intent.key.as_ref() else {
1805 return finding.summary.clone();
1806 };
1807 evidence.push(format!("Declared key: {}", key.columns.join(", ")));
1808 if kind == ObservationKind::KeyMissing {
1809 return format!(
1810 "{} of {} {rows_word} ({share}) have no value in part of the key",
1811 count(key.missing),
1812 count(intent.evaluated_rows)
1813 );
1814 }
1815 evidence.push(format!(
1816 "{} {} held by more than one row; {} rows beyond one per value",
1817 count(key.groups),
1818 if key.groups == 1 { "value" } else { "values" },
1819 count(key.extra_rows)
1820 ));
1821 if sampled {
1822 evidence.push(
1823 "Each repeat here is one in the data; unsampled rows are not checked"
1824 .to_string(),
1825 );
1826 }
1827 format!(
1828 "{} of {} {rows_word} ({share}) share their key with another row",
1829 count(key.rows_involved),
1830 count(intent.evaluated_rows)
1831 )
1832 }
1833 ObservationKind::RequiredMissing => {
1834 evidence.push(format!("Declared required: {column}"));
1835 format!(
1836 "{} of {} {rows_word} ({share}) have no {column}",
1837 count(finding.affected_rows),
1838 count(finding.evaluated_rows)
1839 )
1840 }
1841 ObservationKind::NotAllowed => {
1842 if let Some(check) = check {
1843 evidence.push(format!("Allowed: {}", check.intent.allowed_label(8)));
1844 if !check.outside_examples.is_empty() {
1845 evidence.push(format!("Found: {}", examples(&check.outside_examples)));
1846 }
1847 }
1848 format!(
1849 "{} of {} values ({share}) are not allowed",
1850 count(finding.affected_rows),
1851 count(finding.evaluated_rows)
1852 )
1853 }
1854 ObservationKind::OutOfRange => {
1855 if let Some(check) = check {
1856 evidence.push(format!(
1857 "Range: {}",
1858 check.intent.range_label().unwrap_or_default()
1859 ));
1860 if let Some(below) = check.below.filter(|below| *below > 0) {
1861 let lowest = check
1862 .lowest
1863 .as_ref()
1864 .map(|value| format!(", lowest {value}"))
1865 .unwrap_or_default();
1866 evidence.push(format!("Below: {}{lowest}", count(below)));
1867 }
1868 if let Some(above) = check.above.filter(|above| *above > 0) {
1869 let highest = check
1870 .highest
1871 .as_ref()
1872 .map(|value| format!(", highest {value}"))
1873 .unwrap_or_default();
1874 evidence.push(format!("Above: {}{highest}", count(above)));
1875 }
1876 }
1877 format!(
1878 "{} of {} values ({share}) outside the range",
1879 count(finding.affected_rows),
1880 count(finding.evaluated_rows)
1881 )
1882 }
1883 _ => {
1884 let reading = check
1885 .and_then(|check| check.intent.number)
1886 .map_or("number", crate::quality_intent::NumberReading::label);
1887 if let Some(check) = check.filter(|check| !check.unparsed_examples.is_empty()) {
1888 evidence.push(format!("Such as: {}", examples(&check.unparsed_examples)));
1889 }
1890 evidence.push(format!("Read as a {reading} for this study only"));
1891 format!(
1892 "{} of {} values ({share}) do not read as a {reading}",
1893 count(finding.affected_rows),
1894 count(finding.evaluated_rows)
1895 )
1896 }
1897 }
1898}
1899
1900fn upper_first(text: &str) -> String {
1901 let mut chars = text.chars();
1902 match chars.next() {
1903 Some(first) => first.to_uppercase().chain(chars).collect(),
1904 None => String::new(),
1905 }
1906}
1907
1908pub fn verdict(report: &QualityReport) -> String {
1910 if report.metadata_only {
1913 return match report.problems {
1914 0 => "Values not read: file metadata only".to_string(),
1915 1 => "1 problem in file metadata; values not read".to_string(),
1916 count => format!(
1917 "{} problems in file metadata; values not read",
1918 numfmt::group_chrome(count)
1919 ),
1920 };
1921 }
1922 if report.no_rows {
1924 return match report.problems {
1925 0 => "No rows to check".to_string(),
1926 1 => "1 problem in file metadata; no rows to check".to_string(),
1927 count => format!(
1928 "{} problems in file metadata; no rows to check",
1929 numfmt::group_chrome(count)
1930 ),
1931 };
1932 }
1933 let clean = format!(
1934 "{} of {} columns clean",
1935 numfmt::group_chrome(report.clean_columns),
1936 numfmt::group_chrome(report.total_columns)
1937 );
1938 let notes = match report.notes {
1939 0 => String::new(),
1940 1 => "1 note ".to_string(),
1941 count => format!("{} notes ", numfmt::group_chrome(count)),
1942 };
1943 match report.problems {
1944 0 => format!("No problems found {notes}{clean}"),
1945 1 => format!("1 problem {notes}{clean}"),
1946 count => format!("{} problems {notes}{clean}", numfmt::group_chrome(count)),
1947 }
1948}
1949
1950#[cfg(test)]
1951mod tests {
1952 use super::*;
1953 use crate::data_quality::{DataQualityPlan, QualityObservation, SharedNulls};
1954 use polars::prelude::DataType;
1955
1956 fn profile(name: &str, dtype: DataType) -> ColumnQualityProfile {
1957 ColumnQualityProfile {
1958 name: name.to_string(),
1959 dtype,
1960 evaluated_rows: 100,
1961 null_count: 0,
1962 empty_count: None,
1963 whitespace_count: None,
1964 nan_count: None,
1965 positive_infinity_count: None,
1966 negative_infinity_count: None,
1967 distinct_count: None,
1968 min: None,
1969 max: None,
1970 integer_parse_count: None,
1971 decimal_parse_count: None,
1972 date_parse_count: None,
1973 datetime_parse_count: None,
1974 leading_zero_count: None,
1975 dominant_value: None,
1976 dominant_count: None,
1977 min_length: None,
1978 max_length: None,
1979 }
1980 }
1981
1982 fn observation(kind: ObservationKind, column: &str, affected: usize) -> QualityObservation {
1983 QualityObservation {
1984 kind,
1985 column: column.to_string(),
1986 affected_rows: affected,
1987 evaluated_rows: 100,
1988 fact: String::new(),
1989 normalized_category: None,
1990 files: Vec::new(),
1991 time_format: None,
1992 full_scale: None,
1993 }
1994 }
1995
1996 fn results(
1997 columns: Vec<ColumnQualityProfile>,
1998 observations: Vec<QualityObservation>,
1999 ) -> DataQualityResults {
2000 let plan = DataQualityPlan::default();
2001 let mut results =
2002 DataQualityResults::empty(Some(100), &plan, &polars::prelude::Schema::default());
2003 results.evaluated_rows = 100;
2004 results.precision = QualityPrecision::Exact;
2005 results.columns = columns;
2006 results.observations = observations;
2007 results
2008 }
2009
2010 #[test]
2012 fn nulls_with_one_count_become_one_finding() {
2013 let names = ["open", "high", "low", "close"];
2014 let mut columns = names
2015 .iter()
2016 .map(|name| profile(name, DataType::Float64))
2017 .collect::<Vec<_>>();
2018 columns.push(profile("ticker", DataType::String));
2019 let mut results = results(
2020 columns,
2021 names
2022 .iter()
2023 .map(|name| observation(ObservationKind::Nulls, name, 4))
2024 .collect(),
2025 );
2026 results.shared_nulls = vec![SharedNulls {
2027 columns: names.iter().map(|name| name.to_string()).collect(),
2028 null_rows: 4,
2029 rows_null_in_all: 4,
2030 }];
2031 let report = build_report(&results);
2032 assert_eq!(report.findings.len(), 2, "one note and the clean entry");
2033 let missing = &report.findings[0];
2034 assert_eq!(missing.title, "Missing together");
2035 assert_eq!(missing.severity, Severity::Note);
2036 assert_eq!(missing.columns.len(), 4);
2037 assert!(missing.same_rows);
2038 assert_eq!(missing.summary, "4 rows (4.0%)");
2039 assert_eq!(report.findings[1].severity, Severity::Clean);
2040 assert_eq!(report.findings[1].columns, vec!["ticker".to_string()]);
2041 assert_eq!(report.problems, 0);
2042 assert!(verdict(&report).starts_with("No problems found"));
2043 }
2044
2045 #[test]
2048 fn missing_values_collapse_and_mostly_missing_leads() {
2049 let names = ["a", "b", "c", "d"];
2050 let results = results(
2051 names
2052 .iter()
2053 .map(|name| profile(name, DataType::Float64))
2054 .collect(),
2055 vec![
2056 observation(ObservationKind::Nulls, "a", 3),
2057 observation(ObservationKind::Nulls, "b", 40),
2058 observation(ObservationKind::Nulls, "c", 12),
2059 observation(ObservationKind::Nulls, "d", 80),
2060 ],
2061 );
2062 let report = build_report(&results);
2063 assert_eq!(report.findings.len(), 2);
2064 let mostly = &report.findings[0];
2065 assert_eq!(mostly.title, "Mostly missing");
2066 assert_eq!(mostly.columns, vec!["d".to_string()]);
2067 let missing = &report.findings[1];
2068 assert_eq!(missing.title, "Missing values");
2069 assert_eq!(missing.columns, vec!["b", "c", "a"]);
2070 assert_eq!(missing.summary, "3.0% to 40.0% per column");
2071 let (headline, evidence) = describe(missing, &results);
2072 assert_eq!(headline, "Null rate in 3 columns:");
2073 assert_eq!(evidence[0], "b 40.0% 40 rows");
2074 assert_eq!(evidence[2], "a 3.0% 3 rows");
2075 }
2076
2077 #[test]
2078 fn problems_rank_before_notes_and_mark_their_columns() {
2079 let results = results(
2080 vec![
2081 profile("price", DataType::Float64),
2082 profile("region", DataType::String),
2083 ],
2084 vec![
2085 observation(ObservationKind::Nulls, "region", 3),
2086 observation(ObservationKind::NonFinite, "price", 2),
2087 ],
2088 );
2089 let report = build_report(&results);
2090 assert_eq!(report.findings[0].title, "NaN or infinite");
2091 assert_eq!(report.findings[0].severity, Severity::Problem);
2092 assert_eq!(report.findings[1].title, "Missing values");
2093 assert_eq!(
2094 report.column_status,
2095 vec![Severity::Problem, Severity::Note]
2096 );
2097 assert_eq!(report.clean_columns, 0);
2098 assert_eq!(verdict(&report), "1 problem 1 note 0 of 2 columns clean");
2099 }
2100
2101 #[test]
2102 fn a_column_with_no_values_at_all_is_a_problem() {
2103 let results = results(
2104 vec![profile("legacy", DataType::String)],
2105 vec![observation(ObservationKind::Nulls, "legacy", 100)],
2106 );
2107 let report = build_report(&results);
2108 assert_eq!(report.findings[0].title, "Always missing");
2109 assert_eq!(report.findings[0].severity, Severity::Problem);
2110 }
2111
2112 #[test]
2115 fn zero_padded_digits_read_as_codes() {
2116 let mut code = profile("industry", DataType::String);
2117 code.integer_parse_count = Some(100);
2118 code.decimal_parse_count = Some(100);
2119 code.leading_zero_count = Some(20);
2120 code.min_length = Some(4);
2121 code.max_length = Some(4);
2122 let results = results(
2123 vec![code],
2124 vec![observation(ObservationKind::ParseableText, "industry", 100)],
2125 );
2126 let finding = &build_report(&results).findings[0];
2127 assert_eq!(finding.title, "Codes as text");
2128 assert_eq!(finding.summary, "4 digits, leading zeros");
2129 }
2130
2131 #[test]
2134 fn a_metadata_run_reports_footer_problems_and_no_clean_columns() {
2135 let mut results = results(
2136 vec![
2137 profile("fee", DataType::Float64),
2138 profile("id", DataType::Int64),
2139 ],
2140 vec![observation(ObservationKind::Absent, "fee", 20)],
2141 );
2142 results.precision = QualityPrecision::Metadata;
2143 let report = build_report(&results);
2144 assert_eq!(report.findings.len(), 1, "no clean entry");
2145 assert_eq!(report.findings[0].title, "Missing in files");
2146 assert_eq!(
2147 verdict(&report),
2148 "1 problem in file metadata; values not read"
2149 );
2150 }
2151
2152 #[test]
2155 fn checks_report_reach_findings_and_what_did_not_run() {
2156 let mut results = results(
2157 vec![
2158 profile("price", DataType::Float64),
2159 profile("region", DataType::String),
2160 profile("id", DataType::Int64),
2161 ],
2162 vec![observation(ObservationKind::NonFinite, "price", 2)],
2163 );
2164 results.precision = QualityPrecision::Sampled;
2165 let report = build_report(&results);
2166 let list = checks(&results, &report);
2167 let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2168 assert_eq!(list[0].name, "Missing values", "most important first");
2169 assert_eq!(by_name("Missing values").outcome, Outcome::Passed);
2170 assert_eq!(by_name("Missing values").applies_to, "3 columns");
2171 assert_eq!(by_name("NaN or infinite").applies_to, "1 float column");
2172 assert_eq!(
2173 by_name("NaN or infinite").outcome,
2174 Outcome::Found {
2175 tier: Severity::Problem,
2176 detail: "1 column".to_string()
2177 }
2178 );
2179 assert_eq!(
2180 by_name("Nearly unique").outcome,
2181 Outcome::Unavailable("needs every row checked"),
2182 "a sample cannot say a column is nearly a key"
2183 );
2184 assert_eq!(
2185 by_name("Type mismatch").outcome,
2186 Outcome::Skipped("needs several files")
2187 );
2188
2189 results.precision = QualityPrecision::Metadata;
2190 results.source_files = Some(3);
2191 let report = build_report(&results);
2192 let list = checks(&results, &report);
2193 let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2194 assert_eq!(
2195 by_name("Missing values").outcome,
2196 Outcome::Unavailable("values not read")
2197 );
2198 assert_eq!(by_name("Type mismatch").outcome, Outcome::Passed);
2199 assert_eq!(by_name("Type mismatch").applies_to, "3 files");
2200 }
2201
2202 #[test]
2206 fn coverage_separates_checked_skipped_and_unavailable() {
2207 use crate::data_quality::{
2208 ObservedReads, SegmentQualityProfile, TemporalRole, TemporalRoleAssignment,
2209 };
2210 let columns = vec![
2211 profile("price", DataType::Float64),
2212 profile("region", DataType::String),
2213 profile("id", DataType::Int64),
2214 ];
2215 let plan = DataQualityPlan::default();
2216 let measured = |mut results: DataQualityResults| {
2217 results.identity = Some(crate::data_quality::IdentityProfile {
2218 duplicate_groups: 0,
2219 extra_rows: 0,
2220 rows_involved: 0,
2221 evaluated_rows: results.evaluated_rows,
2222 precision: results.precision,
2223 examples: Vec::new(),
2224 });
2225 results
2226 };
2227
2228 let mut sampled = measured(results(columns.clone(), Vec::new()));
2230 sampled.precision = QualityPrecision::Sampled;
2231 sampled.total_rows = Some(1_000);
2232 sampled.reads = Some(ObservedReads {
2233 reads: 1,
2234 counted: 1,
2235 rows: 1_000,
2236 copy: None,
2237 });
2238 let report = build_report(&sampled);
2239 assert_eq!(report.problems + report.notes, 0, "a clean report");
2240 let found = coverage(&sampled, &checks(&sampled, &report), &plan);
2241 assert_eq!(
2242 found.checks(),
2243 ["7 sampled", "2 skipped", "1 unavailable"],
2244 "{found:?}"
2245 );
2246 assert_eq!(
2247 found.rows,
2248 ["100 of 1,000 sampled (10.0%)", "1,000 traversed"]
2249 );
2250 assert_eq!(found.limits(), ["Nearly unique: needs every row checked"]);
2251
2252 let segment = |label: &str, rows: usize| SegmentQualityProfile {
2254 label: label.to_string(),
2255 total_rows: Some(500),
2256 evaluated_rows: rows,
2257 columns: Vec::new(),
2258 null_cells: 0,
2259 null_rate: 0.0,
2260 compared_with: None,
2261 largest_change: None,
2262 change_size: None,
2263 };
2264 sampled.segments = vec![segment("a", 90), segment("b", 10), segment("c", 0)];
2265 sampled.source_files = Some(400);
2266 sampled.footers_read = Some(100);
2267 let roles = DataQualityPlan {
2268 temporal_roles: vec![TemporalRoleAssignment {
2269 role: TemporalRole::Event,
2270 column: "id".to_string(),
2271 timezone: None,
2272 }],
2273 ..plan.clone()
2274 };
2275 let report = build_report(&sampled);
2276 let list = checks(&sampled, &report);
2277 let type_mismatch = list.iter().find(|c| c.name == "Type mismatch").unwrap();
2278 assert_eq!(type_mismatch.applies_to, "100 of 400 files");
2279 assert_eq!(type_mismatch.basis, QualityPrecision::Metadata);
2280 let found = coverage(&sampled, &list, &roles);
2281 assert_eq!(found.checks(), ["7 sampled", "2 metadata", "1 unavailable"]);
2282 assert_eq!(
2283 found.limits(),
2284 [
2285 "Nearly unique: needs every row checked",
2286 "2 of 3 segments under 30 sampled rows",
2287 "footers of 100 of 400 files read",
2288 "time roles form no interval",
2289 ]
2290 );
2291
2292 let mut full = measured(results(columns.clone(), Vec::new()));
2294 full.reads = Some(ObservedReads {
2295 reads: 4,
2296 counted: 3,
2297 rows: 300,
2298 copy: None,
2299 });
2300 let report = build_report(&full);
2301 let found = coverage(&full, &checks(&full, &report), &plan);
2302 assert_eq!(found.checks(), ["8 exact", "2 skipped"]);
2303 assert_eq!(
2304 found.rows,
2305 ["all 100 read, exact", "at least 300 traversed"]
2306 );
2307 assert!(found.limits().is_empty());
2308
2309 let mut metadata = measured(results(columns, Vec::new()));
2311 metadata.precision = QualityPrecision::Metadata;
2312 metadata.source_files = Some(3);
2313 metadata.footers_read = Some(3);
2314 metadata.reads = Some(ObservedReads::default());
2315 let report = build_report(&metadata);
2316 let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2317 assert_eq!(found.checks(), ["2 metadata", "8 unavailable"]);
2318 assert_eq!(found.rows, ["none read, file metadata only"]);
2319 assert_eq!(found.limits(), ["8 checks: values not read"]);
2320
2321 let floats_only = vec![profile("price", DataType::Float64)];
2324 let mut metadata = measured(results(floats_only.clone(), Vec::new()));
2325 metadata.precision = QualityPrecision::Metadata;
2326 let report = build_report(&metadata);
2327 let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2328 assert_eq!(found.checks(), ["6 skipped", "4 unavailable"], "{found:?}");
2329 let mut sampled = measured(results(floats_only, Vec::new()));
2330 sampled.precision = QualityPrecision::Sampled;
2331 let report = build_report(&sampled);
2332 let list = checks(&sampled, &report);
2333 let nearly = list.iter().find(|c| c.name == "Nearly unique").unwrap();
2334 assert_eq!(
2335 nearly.outcome,
2336 Outcome::Skipped("no integer or text columns")
2337 );
2338 }
2339
2340 #[test]
2341 fn column_labels_fit_and_count_the_rest() {
2342 let columns = ["open", "high", "low", "close"].map(String::from);
2343 assert_eq!(columns_label(&columns, 40), "open, high, low, close");
2344 assert_eq!(columns_label(&columns, 14), "open, high +2");
2345 assert_eq!(columns_label(&columns[..1], 2), "open");
2346 }
2347
2348 #[test]
2353 fn findings_narrow_and_order_without_measuring() {
2354 let mut results = results(
2355 vec![
2356 profile("price", DataType::Float64),
2357 profile("region", DataType::String),
2358 profile("note", DataType::String),
2359 profile("id", DataType::Int64),
2360 ],
2361 vec![
2362 observation(ObservationKind::NonFinite, "price", 2),
2363 observation(ObservationKind::Nulls, "region", 3),
2364 observation(ObservationKind::Nulls, "note", 30),
2365 observation(ObservationKind::Whitespace, "region", 9),
2366 ],
2367 );
2368 results.observations[0].evaluated_rows = 4;
2370 let report = build_report(&results);
2371 let titles = |view: &FindingsView| {
2372 view.shown(&report)
2373 .into_iter()
2374 .map(|index| report.findings[index].title)
2375 .collect::<Vec<_>>()
2376 };
2377 let ranked = FindingsView::default();
2378 assert_eq!(
2379 titles(&ranked),
2380 [
2381 "NaN or infinite",
2382 "Blank text",
2383 "Missing values",
2384 "No findings"
2385 ]
2386 );
2387 let rows = FindingsView {
2388 order: FindingOrder::Rows,
2389 ..FindingsView::default()
2390 };
2391 assert_eq!(
2392 titles(&rows),
2393 [
2394 "Blank text",
2395 "NaN or infinite",
2396 "Missing values",
2397 "No findings"
2398 ],
2399 "most rows first, Problems still above Notes"
2400 );
2401 let rate = FindingsView {
2402 order: FindingOrder::Rate,
2403 ..FindingsView::default()
2404 };
2405 assert_eq!(titles(&rate)[0], "NaN or infinite", "2 of 4 beats 9 of 100");
2406
2407 let region = FindingsView {
2408 column: Some("region".to_string()),
2409 ..FindingsView::default()
2410 };
2411 assert_eq!(titles(®ion), ["Blank text", "Missing values"]);
2412 assert!(region.narrowed());
2413 let clean = FindingsView {
2414 column: Some("id".to_string()),
2415 ..FindingsView::default()
2416 };
2417 assert_eq!(titles(&clean), ["No findings"], "a clean column is clean");
2418 let missing = FindingsView {
2419 check: Some("Missing values"),
2420 ..FindingsView::default()
2421 };
2422 assert_eq!(titles(&missing), ["Missing values"]);
2423 assert_eq!(
2424 missing.selected(&report, 0).map(|finding| finding.title),
2425 Some("Missing values")
2426 );
2427 assert_eq!(
2428 check_choices(&report),
2429 [
2430 ("NaN or infinite", 1),
2431 ("Blank text", 1),
2432 ("Missing values", 1)
2433 ]
2434 );
2435 assert_eq!(
2436 column_choices(&report, &results)
2437 .into_iter()
2438 .map(|(_, count)| count)
2439 .collect::<Vec<_>>(),
2440 [1, 2, 1, 0]
2441 );
2442 }
2443
2444 #[test]
2447 fn grouped_findings_break_down_by_column_and_bound_the_union() {
2448 let results = results(
2449 vec![
2450 profile("a", DataType::String),
2451 profile("b", DataType::String),
2452 ],
2453 vec![
2454 observation(ObservationKind::Empty, "a", 6),
2455 observation(ObservationKind::Empty, "b", 6),
2456 ],
2457 );
2458 let report = build_report(&results);
2459 let finding = &report.findings[0];
2460 assert!(finding.lists_columns());
2461 assert_eq!(
2462 finding.evidence_count(&results),
2463 None,
2464 "a union nobody counted"
2465 );
2466 let (headline, evidence) = describe(finding, &results);
2467 assert_eq!(
2468 headline,
2469 "6 of 100 rows (6.0%) empty strings in each column:"
2470 );
2471 assert_eq!(evidence[0], "a 6.0% 6 rows");
2472 assert_eq!(
2473 evidence.last().unwrap(),
2474 "Rows with any of them: 6 to 12, not counted"
2475 );
2476 }
2477
2478 #[test]
2481 fn parse_failures_are_the_evidence_of_text_that_parses() {
2482 let mut text = profile("amount", DataType::String);
2483 text.null_count = 4;
2484 text.integer_parse_count = Some(95);
2485 text.decimal_parse_count = Some(95);
2486 let mut results = results(
2487 vec![text],
2488 vec![observation(ObservationKind::ParseableText, "amount", 95)],
2489 );
2490 results.examples = vec![crate::data_quality::FindingExamples {
2491 kind: ObservationKind::ParseableText,
2492 column: "amount".to_string(),
2493 values: vec!["\"n/a\"".to_string()],
2494 }];
2495 let report = build_report(&results);
2496 let finding = &report.findings[0];
2497 assert_eq!(finding.check(), Some("Numbers as text"));
2498 assert_eq!(finding.failures(&results), Some(1));
2499 assert_eq!(finding.evidence_count(&results), Some(1));
2500 assert!(matches!(
2501 finding.evidence(&results),
2502 Ok(EvidenceRows::Matching(_))
2503 ));
2504 let (_, evidence) = describe(finding, &results);
2505 assert!(
2506 evidence.contains(&"1 do not parse, such as \"n/a\"".to_string()),
2507 "{evidence:?}"
2508 );
2509
2510 results.columns[0].integer_parse_count = Some(96);
2511 results.columns[0].decimal_parse_count = Some(96);
2512 results.observations[0].affected_rows = 96;
2513 let report = build_report(&results);
2514 let reason = report.findings[0].evidence(&results).unwrap_err();
2515 assert!(reason.contains("every value parses"), "{reason}");
2516 }
2517
2518 #[test]
2520 fn duplicate_rows_open_every_row_with_a_copy() {
2521 let mut results = results(
2522 vec![profile("id", DataType::Int64)],
2523 vec![observation(
2524 ObservationKind::DuplicateRows,
2525 "all columns",
2526 5,
2527 )],
2528 );
2529 results.identity = Some(crate::data_quality::IdentityProfile {
2530 duplicate_groups: 2,
2531 extra_rows: 3,
2532 rows_involved: 5,
2533 evaluated_rows: 100,
2534 precision: QualityPrecision::Exact,
2535 examples: vec![crate::data_quality::DuplicateExample {
2536 copies: 3,
2537 values: vec!["7".to_string()],
2538 }],
2539 });
2540 let report = build_report(&results);
2541 let finding = &report.findings[0];
2542 assert!(matches!(
2543 finding.evidence(&results),
2544 Ok(EvidenceRows::Duplicates)
2545 ));
2546 assert_eq!(finding.evidence_count(&results), Some(5));
2547 let (headline, evidence) = describe(finding, &results);
2548 assert_eq!(headline, "2 rows repeated; 3 extra copies");
2549 assert_eq!(evidence[0], "5 of 100 rows (5.0%) have a copy");
2550 assert_eq!(evidence[2], format!("{}3 7", crate::glyphs::get().times));
2551 }
2552}