Skip to main content

datui_lib/analysis/
quality_report.rs

1//! Data Quality results as a report: what is likely wrong, what depends on intent, and
2//! which columns have nothing to report. Each check kind is one row of
3//! [`ObservationKind::spec`] (names, severity, rank, advice, grouping, wording). Built
4//! once per result; the engine, cache and drill-in work on observations, which a
5//! finding only groups and ranks.
6
7use crate::analysis::data_quality::{
8    ColumnQualityProfile, DataQualityPlan, DataQualityResults, ObservationKind, QualityObservation,
9    QualityPrecision, QualityScope, TextReading, text_reading,
10};
11use crate::numfmt;
12use polars::prelude::Expr;
13use std::collections::BTreeMap;
14
15#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
16pub enum Severity {
17    /// Likely a defect, and one that changes results without saying so.
18    Problem,
19    /// Fine or not depending on what the column is for.
20    Note,
21    /// Nothing to report.
22    Clean,
23}
24
25impl Severity {
26    pub fn heading(self) -> &'static str {
27        match self {
28            Self::Problem => "Problems",
29            Self::Note => "Notes",
30            Self::Clean => "Clean",
31        }
32    }
33}
34
35/// A kind read more than one way (how much is missing; what parsing text holds), each
36/// with its own title and sometimes severity, rank and advice.
37#[derive(Debug, Clone, Copy, PartialEq, Eq)]
38pub enum Variant {
39    /// Null in every row checked.
40    AlwaysMissing,
41    /// Null in more than half the rows.
42    MostlyMissing,
43    /// Several columns null on exactly the same rows.
44    MissingTogether,
45    /// Whole numbers with leading zeros or one fixed length: codes, not quantities.
46    CodesAsText,
47    DatesAsText,
48}
49
50impl Variant {
51    pub fn title(self) -> &'static str {
52        match self {
53            Self::AlwaysMissing => "Always missing",
54            Self::MostlyMissing => "Mostly missing",
55            // No row lacks one of these columns without lacking them all.
56            Self::MissingTogether => "Missing together",
57            Self::CodesAsText => "Codes as text",
58            Self::DatesAsText => "Dates as text",
59        }
60    }
61
62    fn severity(self) -> Option<Severity> {
63        match self {
64            Self::AlwaysMissing => Some(Severity::Problem),
65            Self::MostlyMissing | Self::MissingTogether | Self::CodesAsText | Self::DatesAsText => {
66                None
67            }
68        }
69    }
70
71    fn rank(self) -> Option<u8> {
72        match self {
73            Self::AlwaysMissing => Some(3),
74            Self::MostlyMissing => Some(9),
75            Self::MissingTogether | Self::CodesAsText | Self::DatesAsText => None,
76        }
77    }
78
79    fn advice(self) -> Option<&'static [&'static str]> {
80        match self {
81            Self::AlwaysMissing => Some(&["Carries nothing; check the load or a rename upstream"]),
82            // Says how much is left, so it is written from the finding; see `advice`.
83            Self::MostlyMissing => None,
84            Self::MissingTogether => {
85                Some(&["Likely one cause: a join with no match, or a source with gaps"])
86            }
87            Self::CodesAsText => Some(&["Fine as text; cast only for arithmetic"]),
88            Self::DatesAsText => Some(&["Sorts as text; parse as a date to filter by range"]),
89        }
90    }
91}
92
93#[derive(Debug, Clone)]
94pub struct Finding {
95    pub severity: Severity,
96    /// `None` for the clean-columns entry.
97    pub kind: Option<ObservationKind>,
98    pub variant: Option<Variant>,
99    /// From the kind and its variant: what the list calls it.
100    pub title: &'static str,
101    pub columns: Vec<String>,
102    /// Indices into [`DataQualityResults::observations`].
103    pub observations: Vec<usize>,
104    /// Rows behind the finding; per column when several columns share the count.
105    pub affected_rows: usize,
106    pub evaluated_rows: usize,
107    /// One short line for the list: counts and the telling detail.
108    pub summary: String,
109    /// Grouped columns that are missing on exactly the same rows.
110    pub same_rows: bool,
111}
112
113/// The clean-columns entry's title.
114pub const NO_FINDINGS: &str = "No findings";
115
116impl Finding {
117    /// The same finding: its kind, how it reads, and its columns.
118    pub fn same_as(&self, other: &Finding) -> bool {
119        self.kind == other.kind && self.variant == other.variant && self.columns == other.columns
120    }
121
122    /// "open, high, low +7", fitted to `width` display columns.
123    pub fn columns_label(&self, width: usize) -> String {
124        columns_label(&self.columns, width)
125    }
126
127    /// Rows matching any grouped observation: every row behind the finding (columns missing
128    /// on the same rows match alike).
129    pub fn evidence_predicate(&self, results: &DataQualityResults) -> Option<Expr> {
130        self.observations
131            .iter()
132            .map(|index| {
133                results
134                    .observations
135                    .get(*index)?
136                    .evidence_predicate(results)
137            })
138            .collect::<Option<Vec<_>>>()?
139            .into_iter()
140            .reduce(Expr::or)
141    }
142
143    /// The files behind an absent column or a type conflict; those stand alone.
144    pub fn evidence_scope(&self, results: &DataQualityResults) -> Option<QualityScope> {
145        match self.observations.as_slice() {
146            [index] => results.observations.get(*index)?.evidence_scope(),
147            _ => None,
148        }
149    }
150
151    /// Which rows are the finding's evidence, or why there are none to show.
152    pub fn evidence(&self, results: &DataQualityResults) -> Result<EvidenceRows, String> {
153        if let Some(scope) = self.evidence_scope(results) {
154            return Ok(EvidenceRows::Files(scope));
155        }
156        let Some(kind) = self.kind else {
157            return Err("No rows: every column here passed".to_string());
158        };
159        if results.precision == QualityPrecision::Metadata {
160            return Err("No rows: values were not read".to_string());
161        }
162        match kind.spec().evidence {
163            Evidence::Duplicates => return Ok(EvidenceRows::Duplicates),
164            Evidence::Failures if self.failures(results) == Some(0) => {
165                return Err("No rows: every value parses".to_string());
166            }
167            Evidence::Rows
168            | Evidence::Always
169            | Evidence::Failures
170            | Evidence::SharingValue
171            | Evidence::Uncounted => {}
172        }
173        self.evidence_predicate(results)
174            .map(EvidenceRows::Matching)
175            .ok_or_else(|| "No rows: nothing to filter on".to_string())
176    }
177
178    /// How many rows Enter shows when one count covers them all (one observation, identical
179    /// rows per column, or non-overlapping spellings). `None` for an uncounted union or
180    /// file rows.
181    pub fn evidence_count(&self, results: &DataQualityResults) -> Option<usize> {
182        match self.kind?.spec().evidence {
183            Evidence::Duplicates => {
184                Some(results.identity.as_ref()?.rows_involved).filter(|rows| *rows > 0)
185            }
186            Evidence::Failures => self.failures(results),
187            Evidence::SharingValue | Evidence::Uncounted => None,
188            Evidence::Always => Some(self.affected_rows),
189            Evidence::Rows => {
190                (self.observations.len() == 1 || self.same_rows).then_some(self.affected_rows)
191            }
192        }
193    }
194
195    /// Non-null text in a parseable-text column that its reading does not parse.
196    pub fn failures(&self, results: &DataQualityResults) -> Option<usize> {
197        if self.kind?.spec().evidence != Evidence::Failures {
198            return None;
199        }
200        let profile = results
201            .columns
202            .iter()
203            .find(|profile| Some(&profile.name) == self.columns.first())?;
204        let (parsed, _) = text_reading(profile)?;
205        Some(profile.non_null_rows().saturating_sub(parsed))
206    }
207
208    /// The check that makes this finding, by the name the Checks list gives it.
209    /// `None` for the clean entry.
210    pub fn check(&self) -> Option<&'static str> {
211        Some(self.kind?.spec().check)
212    }
213
214    /// Missing values grouped across columns that go missing at different rates.
215    pub fn varied(&self) -> bool {
216        self.lists_columns()
217            && self.kind.map(|kind| kind.spec().grouping) == Some(Grouping::Missing)
218            && self.severity == Severity::Note
219    }
220
221    /// A count per column that the detail lists one column a line: several columns
222    /// grouped by what they miss, not on the same rows.
223    pub fn lists_columns(&self) -> bool {
224        self.kind.is_some_and(|kind| {
225            matches!(kind.spec().grouping, Grouping::Missing | Grouping::ByRows)
226        }) && self.columns.len() > 1
227            && !self.same_rows
228    }
229
230    /// Each column's count and rate, worst first, then the rows with any of them (uncounted:
231    /// between the largest column's count and their sum).
232    pub fn breakdown(&self, results: &DataQualityResults) -> Vec<String> {
233        let rows = self
234            .observations
235            .iter()
236            .filter_map(|index| results.observations.get(*index))
237            .collect::<Vec<_>>();
238        let name_width = rows
239            .iter()
240            .map(|row| crate::glyphs::display_width(&row.column))
241            .max()
242            .unwrap_or(0)
243            .min(28);
244        let plural = |count: usize| if count == 1 { "row" } else { "rows" };
245        let mut lines = rows
246            .iter()
247            .map(|row| {
248                let name = columns_label(std::slice::from_ref(&row.column), name_width);
249                let pad = name_width.saturating_sub(crate::glyphs::display_width(&name));
250                format!(
251                    "{name}{}  {:>7}  {} {}",
252                    " ".repeat(pad),
253                    crate::numfmt::percent_of(row.affected_rows, row.evaluated_rows),
254                    numfmt::group_chrome(row.affected_rows),
255                    plural(row.affected_rows)
256                )
257            })
258            .collect::<Vec<_>>();
259        let least = rows.iter().map(|row| row.affected_rows).max().unwrap_or(0);
260        let most = rows
261            .iter()
262            .map(|row| row.affected_rows)
263            .sum::<usize>()
264            .min(self.evaluated_rows.max(least));
265        lines.push(if least == most {
266            format!(
267                "Rows with any of them: {} {}",
268                numfmt::group_chrome(least),
269                plural(least)
270            )
271        } else {
272            format!(
273                "Rows with any of them: {} to {}, not counted",
274                numfmt::group_chrome(least),
275                numfmt::group_chrome(most)
276            )
277        });
278        lines
279    }
280}
281
282/// The rows behind a finding.
283#[derive(Debug, Clone)]
284pub enum EvidenceRows {
285    /// Rows whose values match.
286    Matching(Expr),
287    /// Every row the named files hold.
288    Files(QualityScope),
289    /// Rows equal to another in every column, copies together; see
290    /// [`crate::analysis::data_quality::duplicate_rows`].
291    Duplicates,
292}
293
294/// Which of a kind's observations one finding says together.
295#[derive(Debug, Clone, Copy, PartialEq, Eq)]
296pub enum Grouping {
297    /// Each observation is its own finding.
298    Alone,
299    /// Every column at once: one fact named once per column.
300    All,
301    /// Columns with the same count together.
302    ByRows,
303    /// One finding per column, of all its observations.
304    ByColumn,
305    /// Missing values: columns always missing together, mostly together, on the very same
306    /// rows, and the rest as one finding with per-column rates.
307    Missing,
308}
309
310/// What Enter opens of a finding, and how many rows that is.
311#[derive(Debug, Clone, Copy, PartialEq, Eq)]
312pub enum Evidence {
313    /// The rows counted, when one count is all of them: one observation, or the
314    /// same rows in every column.
315    Rows,
316    /// The rows counted, always: values of one column that never overlap, or one
317    /// fact about rows named once per column.
318    Always,
319    /// Rows equal to another in every column.
320    Duplicates,
321    /// The values its reading does not parse.
322    Failures,
323    /// Every row that shares a repeated value: more than the rows beyond one per
324    /// value the check counts.
325    SharingValue,
326    /// Files, or every sample: nothing counted to show.
327    Uncounted,
328}
329
330/// One kind of check. Everything about a kind but its emitter and the predicate
331/// beside it (`QualityObservation::evidence_predicate`) is here.
332pub struct CheckSpec {
333    /// The check that makes it, by the name the Checks list gives it.
334    pub check: &'static str,
335    /// The finding's title, unless a [`Variant`] reads it another way.
336    pub title: &'static str,
337    pub severity: Severity,
338    /// Order within a severity: what costs the most rows of trust first.
339    pub rank: u8,
340    /// What it means for the data and what to do about it, a fragment a line. The
341    /// reader knows what a null is; this says only what is particular.
342    pub advice: &'static [&'static str],
343    pub grouping: Grouping,
344    pub evidence: Evidence,
345    /// What the rows Enter opens are, after their count: " that have a copy".
346    pub rows_are: &'static str,
347    /// What its rows hold, after "N of M rows (P%)": "null".
348    pub noun: &'static str,
349    /// The finding's reading, when one kind reads several ways.
350    variant: fn(&Group<'_>) -> Option<Variant>,
351    /// The list's line: counts and the telling detail.
352    summary: fn(&Group<'_>) -> String,
353    /// The detail's sentence, with the evidence that makes it concrete.
354    headline: fn(&Detail<'_>, &mut Vec<String>) -> String,
355}
356
357impl ObservationKind {
358    /// The kind's row of the table.
359    pub fn spec(self) -> &'static CheckSpec {
360        match self {
361            Self::Nulls => &CheckSpec {
362                check: "Missing values",
363                title: "Missing values",
364                severity: Severity::Note,
365                rank: 10,
366                advice: &["Check: clustered in some files or dates (Segments, by file or window)"],
367                grouping: Grouping::Missing,
368                evidence: Evidence::Rows,
369                rows_are: "",
370                noun: "null",
371                variant: missing_variant,
372                summary: nulls_summary,
373                headline: nulls_headline,
374            },
375            Self::Empty => &CheckSpec {
376                check: "Blank text",
377                title: "Empty text",
378                severity: Severity::Problem,
379                rank: 7,
380                advice: &["Counted as filled; treat as null if it means missing"],
381                grouping: Grouping::ByRows,
382                evidence: Evidence::Rows,
383                rows_are: "",
384                noun: "empty strings",
385                variant: no_variant,
386                summary: rows_each_summary,
387                headline: per_column_headline,
388            },
389            Self::Whitespace => &CheckSpec {
390                check: "Blank text",
391                title: "Blank text",
392                severity: Severity::Problem,
393                rank: 6,
394                advice: &["Counted as filled; trim to null if it means missing"],
395                grouping: Grouping::ByRows,
396                evidence: Evidence::Rows,
397                rows_are: "",
398                noun: "only spaces or tabs",
399                variant: no_variant,
400                summary: rows_each_summary,
401                headline: per_column_headline,
402            },
403            Self::NonFinite => &CheckSpec {
404                check: "NaN or infinite",
405                title: "NaN or infinite",
406                severity: Severity::Problem,
407                rank: 4,
408                advice: &["Sums and means become NaN; check for division by zero upstream"],
409                grouping: Grouping::Alone,
410                evidence: Evidence::Rows,
411                rows_are: "",
412                noun: "NaN or infinite",
413                variant: no_variant,
414                summary: rows_summary,
415                headline: non_finite_headline,
416            },
417            Self::Constant => &CheckSpec {
418                check: "Single value",
419                title: "Single value",
420                severity: Severity::Note,
421                rank: 13,
422                advice: &["Tells no rows apart; a stuck feed if it should vary"],
423                grouping: Grouping::All,
424                evidence: Evidence::Rows,
425                rows_are: "",
426                noun: "",
427                variant: no_variant,
428                summary: constant_summary,
429                headline: constant_headline,
430            },
431            Self::ParseableText => &CheckSpec {
432                check: "Numbers as text",
433                title: "Numbers as text",
434                severity: Severity::Note,
435                rank: 12,
436                advice: &["Sorts as text (\"10\" before \"9\"); cast to a number to sum"],
437                grouping: Grouping::Alone,
438                evidence: Evidence::Failures,
439                rows_are: " that do not parse",
440                noun: "",
441                variant: reading_variant,
442                summary: parseable_summary,
443                headline: parseable_headline,
444            },
445            Self::DuplicateRows => &CheckSpec {
446                check: "Duplicate rows",
447                title: "Duplicate rows",
448                severity: Severity::Problem,
449                rank: 2,
450                advice: &["Counted more than once; check for a double load or a join fan-out"],
451                grouping: Grouping::Alone,
452                evidence: Evidence::Duplicates,
453                rows_are: " that have a copy",
454                noun: "",
455                variant: no_variant,
456                summary: duplicates_summary,
457                headline: duplicates_headline,
458            },
459            Self::CategoryVariants => &CheckSpec {
460                check: "Mixed spellings",
461                title: "Mixed spellings",
462                severity: Severity::Problem,
463                rank: 5,
464                advice: &["Group-bys and joins split them; trim and normalize case"],
465                grouping: Grouping::ByColumn,
466                evidence: Evidence::Always,
467                rows_are: "",
468                noun: "",
469                variant: no_variant,
470                summary: variants_summary,
471                headline: variants_headline,
472            },
473            Self::Absent => &CheckSpec {
474                check: "Missing in files",
475                title: "Missing in files",
476                severity: Severity::Problem,
477                rank: 1,
478                advice: &["Check: files written before the column existed"],
479                grouping: Grouping::Alone,
480                evidence: Evidence::Uncounted,
481                rows_are: "",
482                noun: "",
483                variant: no_variant,
484                summary: fact_summary,
485                headline: files_headline,
486            },
487            Self::TypeConflict => &CheckSpec {
488                check: "Type mismatch",
489                title: "Type mismatch",
490                severity: Severity::Problem,
491                rank: 0,
492                advice: &["Values dropped, not converted; read as text in Info, or fix the writer"],
493                grouping: Grouping::Alone,
494                evidence: Evidence::Uncounted,
495                rows_are: "",
496                noun: "",
497                variant: no_variant,
498                summary: fact_summary,
499                headline: files_headline,
500            },
501            Self::KeyLike => &CheckSpec {
502                check: "Nearly unique",
503                title: "Nearly unique",
504                severity: Severity::Note,
505                rank: 11,
506                advice: &["Duplicates if it is a key; expected if it is a measurement"],
507                grouping: Grouping::Alone,
508                evidence: Evidence::SharingValue,
509                rows_are: "",
510                noun: "",
511                variant: no_variant,
512                summary: key_like_summary,
513                headline: key_like_headline,
514            },
515            Self::UnparsedTime => &CheckSpec {
516                check: "Unparsed times",
517                title: "Unparsed times",
518                severity: Severity::Problem,
519                rank: 3,
520                advice: &[
521                    "Left out of time windows and intervals, not counted as missing",
522                    "Check: another format, or values that are not times (Setup, e)",
523                ],
524                grouping: Grouping::Alone,
525                evidence: Evidence::Rows,
526                rows_are: "",
527                noun: "",
528                variant: no_variant,
529                summary: values_summary,
530                headline: unparsed_time_headline,
531            },
532            Self::KeyRepeated => &CheckSpec {
533                check: INTENT_CHECK,
534                title: "Repeated key",
535                severity: Severity::Problem,
536                rank: 2,
537                advice: &[
538                    "A key names one row; joins on it fan out and counts double",
539                    "Check: a double load, or a key that needs another column",
540                ],
541                grouping: Grouping::All,
542                evidence: Evidence::Always,
543                rows_are: "",
544                noun: "",
545                variant: no_variant,
546                summary: key_repeated_summary,
547                headline: key_repeated_headline,
548            },
549            Self::KeyMissing => &CheckSpec {
550                check: INTENT_CHECK,
551                title: "Incomplete key",
552                severity: Severity::Problem,
553                rank: 2,
554                advice: &["Rows with no key cannot be joined or told apart by it"],
555                grouping: Grouping::All,
556                evidence: Evidence::Always,
557                rows_are: "",
558                noun: "",
559                variant: no_variant,
560                summary: rows_summary,
561                headline: key_missing_headline,
562            },
563            Self::RequiredMissing => &CheckSpec {
564                check: INTENT_CHECK,
565                title: "Required, missing",
566                severity: Severity::Problem,
567                rank: 2,
568                advice: &["Check: the load, or rows the source writes without it"],
569                grouping: Grouping::Alone,
570                evidence: Evidence::Rows,
571                rows_are: "",
572                noun: "",
573                variant: no_variant,
574                summary: rows_summary,
575                headline: required_headline,
576            },
577            Self::NotAllowed => &CheckSpec {
578                check: INTENT_CHECK,
579                title: "Not allowed",
580                severity: Severity::Problem,
581                rank: 2,
582                advice: &["Check: a new value upstream, or the allowed list (Setup, e)"],
583                grouping: Grouping::Alone,
584                evidence: Evidence::Rows,
585                rows_are: "",
586                noun: "",
587                variant: no_variant,
588                summary: values_summary,
589                headline: not_allowed_headline,
590            },
591            Self::OutOfRange => &CheckSpec {
592                check: INTENT_CHECK,
593                title: "Out of range",
594                severity: Severity::Problem,
595                rank: 2,
596                advice: &["Check: units, placeholders such as -1 or 9999, or the range (Setup, e)"],
597                grouping: Grouping::Alone,
598                evidence: Evidence::Rows,
599                rows_are: "",
600                noun: "",
601                variant: no_variant,
602                summary: values_summary,
603                headline: out_of_range_headline,
604            },
605            Self::UnparsedNumber => &CheckSpec {
606                check: INTENT_CHECK,
607                title: "Unparsed numbers",
608                severity: Severity::Problem,
609                rank: 2,
610                advice: &["A cast makes them null; check the values or the reading (Setup, e)"],
611                grouping: Grouping::Alone,
612                evidence: Evidence::Rows,
613                rows_are: "",
614                noun: "",
615                variant: no_variant,
616                summary: values_summary,
617                headline: unparsed_number_headline,
618            },
619            Self::Clipping => &CheckSpec {
620                check: "Clipping",
621                title: "Clipping",
622                severity: Severity::Problem,
623                rank: 4,
624                advice: &[
625                    "Cut flat at the limit; lower the gain at the source, nothing restores it",
626                ],
627                grouping: Grouping::Alone,
628                evidence: Evidence::Rows,
629                rows_are: "",
630                noun: "in runs at full scale",
631                variant: no_variant,
632                summary: fact_summary,
633                headline: signal_headline,
634            },
635            Self::ZeroRuns => &CheckSpec {
636                check: "Runs of zeros",
637                title: "Runs of zeros",
638                severity: Severity::Note,
639                rank: 8,
640                advice: &["Dropouts mid-recording; digital silence if at the start or end"],
641                grouping: Grouping::Alone,
642                evidence: Evidence::Rows,
643                rows_are: "",
644                noun: "in runs of exact zeros",
645                variant: no_variant,
646                summary: fact_summary,
647                headline: signal_headline,
648            },
649            Self::DcOffset => &CheckSpec {
650                check: "DC offset",
651                title: "DC offset",
652                severity: Severity::Note,
653                rank: 12,
654                advice: &["A constant bias that eats headroom; a high-pass filter removes it"],
655                grouping: Grouping::Alone,
656                evidence: Evidence::Uncounted,
657                rows_are: "",
658                noun: "",
659                variant: no_variant,
660                summary: fact_summary,
661                headline: dc_offset_headline,
662            },
663        }
664    }
665}
666
667/// How the findings list is narrowed and ordered. Presentation only: it reads the
668/// report on screen, and nothing is measured again.
669#[derive(Debug, Clone, Default, PartialEq, Eq)]
670pub struct FindingsView {
671    /// Only findings that name this column.
672    pub column: Option<String>,
673    /// Only findings this check made, by [`Finding::check`].
674    pub check: Option<&'static str>,
675    pub order: FindingOrder,
676}
677
678/// The order findings are listed in within each severity.
679#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
680pub enum FindingOrder {
681    /// What costs the most trust first; see [`build_report`].
682    #[default]
683    Ranked,
684    /// Most rows affected first.
685    Rows,
686    /// Highest share of the rows checked first.
687    Rate,
688}
689
690impl FindingOrder {
691    pub fn next(self) -> Self {
692        match self {
693            Self::Ranked => Self::Rows,
694            Self::Rows => Self::Rate,
695            Self::Rate => Self::Ranked,
696        }
697    }
698
699    /// The chip that switches to this order.
700    pub fn chip(self) -> &'static str {
701        match self {
702            Self::Ranked => "Ranked",
703            Self::Rows => "By rows",
704            Self::Rate => "By rate",
705        }
706    }
707}
708
709impl FindingsView {
710    pub fn narrowed(&self) -> bool {
711        self.column.is_some() || self.check.is_some()
712    }
713
714    fn admits(&self, finding: &Finding) -> bool {
715        self.column
716            .as_ref()
717            .is_none_or(|column| finding.columns.contains(column))
718            && self
719                .check
720                .is_none_or(|check| finding.check() == Some(check))
721    }
722
723    /// Indices into `report.findings`, in the order listed. Severity leads in every
724    /// order, so Problems stay above Notes; within one, a tie keeps the rank.
725    pub fn shown(&self, report: &QualityReport) -> Vec<usize> {
726        let findings = &report.findings;
727        let mut shown = (0..findings.len())
728            .filter(|index| self.admits(&findings[*index]))
729            .collect::<Vec<_>>();
730        let by = |left: &usize, right: &usize| {
731            let (left, right) = (&findings[*left], &findings[*right]);
732            left.severity
733                .cmp(&right.severity)
734                .then_with(|| match self.order {
735                    FindingOrder::Ranked => std::cmp::Ordering::Equal,
736                    FindingOrder::Rows => right.affected_rows.cmp(&left.affected_rows),
737                    // Shares compared exactly: a/b against c/d as a·d against c·b.
738                    FindingOrder::Rate => (right.affected_rows as u128
739                        * left.evaluated_rows.max(1) as u128)
740                        .cmp(&(left.affected_rows as u128 * right.evaluated_rows.max(1) as u128)),
741                })
742        };
743        shown.sort_by(by);
744        shown
745    }
746
747    /// The finding at `position` in the list as shown.
748    pub fn selected<'a>(&self, report: &'a QualityReport, position: usize) -> Option<&'a Finding> {
749        self.shown(report)
750            .get(position)
751            .and_then(|index| report.findings.get(*index))
752    }
753
754    /// What narrows and orders the list, for the line above it: "price · Missing
755    /// values · by rows". Empty when the list is the report as ranked.
756    pub fn describe(&self) -> Vec<String> {
757        let mut parts = Vec::new();
758        if let Some(column) = &self.column {
759            parts.push(format!("column {column}"));
760        }
761        if let Some(check) = self.check {
762            parts.push(check.to_string());
763        }
764        match self.order {
765            FindingOrder::Ranked => {}
766            FindingOrder::Rows => parts.push("by rows".to_string()),
767            FindingOrder::Rate => parts.push("by rate".to_string()),
768        }
769        parts
770    }
771}
772
773/// The columns a report can be narrowed to, each with how many findings name it,
774/// in the results' column order.
775pub fn column_choices(
776    report: &QualityReport,
777    results: &DataQualityResults,
778) -> Vec<(String, usize)> {
779    results
780        .columns
781        .iter()
782        .map(|profile| {
783            let count = report
784                .findings
785                .iter()
786                .filter(|finding| finding.kind.is_some() && finding.columns.contains(&profile.name))
787                .count();
788            (profile.name.clone(), count)
789        })
790        .collect()
791}
792
793/// The checks that made a finding in `report`, each with how many, most important
794/// first.
795pub fn check_choices(report: &QualityReport) -> Vec<(&'static str, usize)> {
796    let mut choices: Vec<(&'static str, usize)> = Vec::new();
797    for check in report.findings.iter().filter_map(Finding::check) {
798        match choices.iter_mut().find(|(name, _)| *name == check) {
799            Some((_, count)) => *count += 1,
800            None => choices.push((check, 1)),
801        }
802    }
803    choices
804}
805
806#[derive(Debug, Clone, Default)]
807pub struct QualityReport {
808    pub findings: Vec<Finding>,
809    pub problems: usize,
810    pub notes: usize,
811    pub clean_columns: usize,
812    pub total_columns: usize,
813    /// Worst severity per column, in the results' column order.
814    pub column_status: Vec<Severity>,
815    /// Finding titles per column, in the results' column order.
816    pub column_findings: Vec<Vec<&'static str>>,
817    /// No values were read, so the absence of a finding means nothing.
818    pub metadata_only: bool,
819    /// Values were to be read and the scope had none: no check had anything to look
820    /// at, so nothing is clean either.
821    pub no_rows: bool,
822}
823
824/// The report and checks a [`DataQualityResults`] reads as, built once on first ask and
825/// kept with the results (their inputs never change after install). A copy starts
826/// empty, so a changed copy reads fresh.
827#[derive(Debug, Default)]
828pub struct ReportCache(std::sync::OnceLock<Built>);
829
830impl Clone for ReportCache {
831    fn clone(&self) -> Self {
832        Self::default()
833    }
834}
835
836#[derive(Debug)]
837struct Built {
838    report: QualityReport,
839    checks: Vec<Check>,
840}
841
842#[cfg(test)]
843thread_local! {
844    /// Reports this thread has built for a cache, for the tests that count them.
845    pub(crate) static REPORTS_BUILT: std::cell::Cell<usize> = const { std::cell::Cell::new(0) };
846}
847
848impl ReportCache {
849    fn get(&self, results: &DataQualityResults) -> &Built {
850        self.0.get_or_init(|| {
851            #[cfg(test)]
852            REPORTS_BUILT.with(|built| built.set(built.get() + 1));
853            let report = build_report(results);
854            let checks = checks(results, &report);
855            Built { report, checks }
856        })
857    }
858}
859
860impl DataQualityResults {
861    /// Change these results in place: the report and checks are built again from
862    /// what `edit` leaves.
863    pub fn edit<R>(&mut self, edit: impl FnOnce(&mut Self) -> R) -> R {
864        let out = edit(self);
865        self.derived = ReportCache::default();
866        out
867    }
868
869    /// The findings these results read as; see [`ReportCache`].
870    pub fn report(&self) -> &QualityReport {
871        &self.derived.get(self).report
872    }
873
874    /// Every check, most important first; see [`ReportCache`].
875    pub fn checks(&self) -> &[Check] {
876        &self.derived.get(self).checks
877    }
878}
879
880pub fn build_report(results: &DataQualityResults) -> QualityReport {
881    let mut findings = Vec::new();
882    // Group observations that say the same thing so the list says it once.
883    let mut groups = BTreeMap::<(u8, usize, String), Vec<usize>>::new();
884    for (index, observation) in results.observations.iter().enumerate() {
885        let kind = observation.kind as u8 + 1;
886        let key = match observation.kind.spec().grouping {
887            Grouping::Missing => {
888                let always_missing = observation.evaluated_rows > 0
889                    && observation.affected_rows == observation.evaluated_rows;
890                let mostly_missing = observation.affected_rows * 2 > observation.evaluated_rows;
891                // Columns missing on the very same rows are one fact about those rows;
892                // any other missing values are one finding, with each column's rate
893                // inside it.
894                let same_rows = results.shared_nulls.iter().any(|shared| {
895                    shared.same_rows()
896                        && shared.columns.len() > 1
897                        && shared.null_rows == observation.affected_rows
898                        && shared.columns.contains(&observation.column)
899                });
900                if always_missing {
901                    (0, 0, String::new())
902                } else if mostly_missing {
903                    (kind, usize::MAX, "mostly".to_string())
904                } else if same_rows {
905                    (kind, observation.affected_rows, String::new())
906                } else {
907                    (kind, usize::MAX, String::new())
908                }
909            }
910            Grouping::ByRows => (kind, observation.affected_rows, String::new()),
911            Grouping::All => (kind, 0, String::new()),
912            Grouping::ByColumn => (kind, 0, observation.column.clone()),
913            // The index keeps it from merging.
914            Grouping::Alone => (u8::MAX, index, String::new()),
915        };
916        groups.entry(key).or_default().push(index);
917    }
918    for indices in groups.into_values() {
919        findings.push(finding(results, &indices));
920    }
921
922    let mut status = vec![Severity::Clean; results.columns.len()];
923    let mut titles = vec![Vec::new(); results.columns.len()];
924    let position = results
925        .columns
926        .iter()
927        .enumerate()
928        .map(|(index, profile)| (profile.name.as_str(), index))
929        .collect::<BTreeMap<_, _>>();
930    for finding in &findings {
931        for column in &finding.columns {
932            if let Some(&index) = position.get(column.as_str()) {
933                status[index] = status[index].min(finding.severity);
934                if !titles[index].contains(&finding.title) {
935                    titles[index].push(finding.title);
936                }
937            }
938        }
939    }
940
941    findings.sort_by(|left, right| {
942        left.severity
943            .cmp(&right.severity)
944            .then_with(|| rank(left).cmp(&rank(right)))
945            .then_with(|| right.affected_rows.cmp(&left.affected_rows))
946            .then_with(|| left.columns.cmp(&right.columns))
947    });
948    let problems = findings
949        .iter()
950        .filter(|finding| finding.severity == Severity::Problem)
951        .count();
952    let notes = findings.len() - problems;
953    let metadata_only = results.precision == QualityPrecision::Metadata;
954    let no_rows = !metadata_only && results.evaluated_rows == 0;
955    let clean = results
956        .columns
957        .iter()
958        .zip(&status)
959        .filter(|(_, severity)| **severity == Severity::Clean)
960        .map(|(profile, _)| profile.name.clone())
961        .collect::<Vec<_>>();
962    let clean_columns = if no_rows { 0 } else { clean.len() };
963    if !clean.is_empty() && !metadata_only && !no_rows {
964        findings.push(Finding {
965            severity: Severity::Clean,
966            kind: None,
967            variant: None,
968            title: NO_FINDINGS,
969            summary: format!(
970                "{} {}",
971                numfmt::group_chrome(clean.len()),
972                if clean.len() == 1 {
973                    "column"
974                } else {
975                    "columns"
976                }
977            ),
978            columns: clean,
979            observations: Vec::new(),
980            affected_rows: 0,
981            evaluated_rows: results.evaluated_rows,
982            same_rows: false,
983        });
984    }
985    QualityReport {
986        findings,
987        problems,
988        notes,
989        clean_columns,
990        total_columns: results.columns.len(),
991        column_status: status,
992        column_findings: titles,
993        metadata_only,
994        no_rows,
995    }
996}
997
998/// Order within a severity: what costs the most rows of trust first.
999fn rank(finding: &Finding) -> u8 {
1000    match (finding.variant.and_then(Variant::rank), finding.kind) {
1001        (Some(rank), _) => rank,
1002        (None, Some(kind)) => kind.spec().rank,
1003        (None, None) => 20,
1004    }
1005}
1006
1007/// One finding's observations, as [`finding`] reads them.
1008struct Group<'a> {
1009    results: &'a DataQualityResults,
1010    indices: &'a [usize],
1011    first: &'a QualityObservation,
1012    profile: Option<&'a ColumnQualityProfile>,
1013    reading: Option<TextReading>,
1014    /// Several columns.
1015    grouped: bool,
1016    same_rows: bool,
1017    variant: Option<Variant>,
1018}
1019
1020impl Group<'_> {
1021    /// "1,234 rows (2.0%)", with `each` after the noun.
1022    fn rows_each(&self, count: usize, each: &str) -> String {
1023        format!(
1024            "{} {}{each} ({})",
1025            numfmt::group_chrome(count),
1026            if count == 1 { "row" } else { "rows" },
1027            crate::numfmt::percent_of(count, self.first.evaluated_rows)
1028        )
1029    }
1030
1031    fn rows(&self) -> String {
1032        self.rows_each(self.first.affected_rows, "")
1033    }
1034}
1035
1036fn finding(results: &DataQualityResults, indices: &[usize]) -> Finding {
1037    let mut indices = indices.to_vec();
1038    let spec = results.observations[indices[0]].kind.spec();
1039    // Missing values list their columns, and mixed spellings their values, worst
1040    // first.
1041    if matches!(spec.grouping, Grouping::Missing | Grouping::ByColumn) {
1042        indices.sort_by_key(|index| std::cmp::Reverse(results.observations[*index].affected_rows));
1043    }
1044    let indices = indices.as_slice();
1045    let first = &results.observations[indices[0]];
1046    let mut columns = Vec::new();
1047    for index in indices {
1048        let column = &results.observations[*index].column;
1049        if !columns.contains(column) {
1050            columns.push(column.clone());
1051        }
1052    }
1053    let grouped = columns.len() > 1;
1054    let profile = results
1055        .columns
1056        .iter()
1057        .find(|profile| profile.name == first.column);
1058    let shared = results.shared_nulls.iter().find(|shared| {
1059        shared.null_rows == first.affected_rows
1060            && columns.iter().all(|column| shared.columns.contains(column))
1061    });
1062    let same_rows =
1063        spec.grouping == Grouping::Missing && grouped && shared.is_some_and(|s| s.same_rows());
1064    let mut group = Group {
1065        results,
1066        indices,
1067        first,
1068        profile,
1069        reading: profile.and_then(text_reading).map(|(_, reading)| reading),
1070        grouped,
1071        same_rows,
1072        variant: None,
1073    };
1074    group.variant = (spec.variant)(&group);
1075    let affected_rows = match spec.grouping {
1076        // One group per normalized value; the column's cost is all of them.
1077        Grouping::ByColumn => indices
1078            .iter()
1079            .map(|index| results.observations[*index].affected_rows)
1080            .sum(),
1081        Grouping::Alone | Grouping::All | Grouping::ByRows | Grouping::Missing => {
1082            first.affected_rows
1083        }
1084    };
1085    Finding {
1086        severity: group
1087            .variant
1088            .and_then(Variant::severity)
1089            .unwrap_or(spec.severity),
1090        kind: Some(first.kind),
1091        variant: group.variant,
1092        title: group.variant.map_or(spec.title, Variant::title),
1093        columns,
1094        observations: indices.to_vec(),
1095        affected_rows,
1096        evaluated_rows: first.evaluated_rows,
1097        summary: (spec.summary)(&group),
1098        same_rows,
1099    }
1100}
1101
1102fn no_variant(_: &Group<'_>) -> Option<Variant> {
1103    None
1104}
1105
1106fn missing_variant(group: &Group<'_>) -> Option<Variant> {
1107    let first = group.first;
1108    if first.evaluated_rows > 0 && first.affected_rows == first.evaluated_rows {
1109        Some(Variant::AlwaysMissing)
1110    } else if first.affected_rows * 2 > first.evaluated_rows {
1111        Some(Variant::MostlyMissing)
1112    } else if group.same_rows {
1113        Some(Variant::MissingTogether)
1114    } else {
1115        None
1116    }
1117}
1118
1119fn reading_variant(group: &Group<'_>) -> Option<Variant> {
1120    match (group.reading, group.profile) {
1121        (Some(reading), Some(profile)) if reading.is_number() && is_code(profile) => {
1122            Some(Variant::CodesAsText)
1123        }
1124        (Some(TextReading::Datetime | TextReading::Date), _) => Some(Variant::DatesAsText),
1125        (Some(TextReading::WholeNumber | TextReading::Decimal) | None, _) => None,
1126    }
1127}
1128
1129fn nulls_summary(group: &Group<'_>) -> String {
1130    let first = group.first;
1131    if group.variant == Some(Variant::AlwaysMissing) {
1132        "no value in any row".to_string()
1133    } else if group.same_rows {
1134        group.rows()
1135    } else if group.grouped {
1136        let fewest =
1137            group.results.observations[group.indices[group.indices.len() - 1]].affected_rows;
1138        let (low, high) = (
1139            crate::numfmt::percent_of(fewest, first.evaluated_rows),
1140            crate::numfmt::percent_of(first.affected_rows, first.evaluated_rows),
1141        );
1142        if low == high {
1143            group.rows_each(first.affected_rows, " each")
1144        } else {
1145            format!("{low} to {high} per column")
1146        }
1147    } else {
1148        group.rows()
1149    }
1150}
1151
1152fn rows_each_summary(group: &Group<'_>) -> String {
1153    if group.grouped {
1154        group.rows_each(group.first.affected_rows, " each")
1155    } else {
1156        group.rows()
1157    }
1158}
1159
1160fn rows_summary(group: &Group<'_>) -> String {
1161    group.rows()
1162}
1163
1164/// "12 values (0.4%)".
1165fn values_summary(group: &Group<'_>) -> String {
1166    let first = group.first;
1167    format!(
1168        "{} {} ({})",
1169        numfmt::group_chrome(first.affected_rows),
1170        if first.affected_rows == 1 {
1171            "value"
1172        } else {
1173            "values"
1174        },
1175        crate::numfmt::percent_of(first.affected_rows, first.evaluated_rows)
1176    )
1177}
1178
1179/// What only the engine can say: the files, or a channel's runs or mean. Across
1180/// channels, the worst channel's.
1181fn fact_summary(group: &Group<'_>) -> String {
1182    group.first.fact.clone()
1183}
1184
1185fn constant_summary(group: &Group<'_>) -> String {
1186    if group.grouped {
1187        return "one value each".to_string();
1188    }
1189    group
1190        .profile
1191        .and_then(|profile| profile.dominant_value.as_ref())
1192        .map(|value| format!("always {}", quoted(value, 24)))
1193        .unwrap_or_else(|| "one value".to_string())
1194}
1195
1196fn parseable_summary(group: &Group<'_>) -> String {
1197    match (group.reading, group.profile) {
1198        (Some(_), Some(profile)) if is_code(profile) => code_shape(profile),
1199        (Some(reading), Some(profile)) => format!(
1200            "{} parse as {}",
1201            crate::numfmt::percent_of(group.first.affected_rows, profile.non_null_rows()),
1202            reading.label()
1203        ),
1204        (None, _) | (_, None) => group.rows(),
1205    }
1206}
1207
1208fn duplicates_summary(group: &Group<'_>) -> String {
1209    match group.results.identity.as_ref() {
1210        Some(identity) => format!(
1211            "{} extra {}",
1212            numfmt::group_chrome(identity.extra_rows),
1213            if identity.extra_rows == 1 {
1214                "copy"
1215            } else {
1216                "copies"
1217            }
1218        ),
1219        None => group.rows(),
1220    }
1221}
1222
1223fn variants_summary(group: &Group<'_>) -> String {
1224    // Several values: the count is the finding, and the detail lists them.
1225    if group.indices.len() > 1 {
1226        return format!("{} values spelled more than one way", group.indices.len());
1227    }
1228    group
1229        .results
1230        .category_variants
1231        .iter()
1232        .find(|variants| Some(&variants.normalized) == group.first.normalized_category.as_ref())
1233        .map(|variants| {
1234            variants
1235                .variants
1236                .iter()
1237                .take(2)
1238                .map(|(value, _)| quoted(value, 24))
1239                .collect::<Vec<_>>()
1240                .join(" vs ")
1241        })
1242        .unwrap_or_default()
1243}
1244
1245fn key_like_summary(group: &Group<'_>) -> String {
1246    let unique = group
1247        .profile
1248        .and_then(ColumnQualityProfile::uniqueness_rate)
1249        .map(|rate| format!(" ({} unique)", crate::numfmt::percent(rate)))
1250        .unwrap_or_default();
1251    format!(
1252        "{} repeated{unique}",
1253        numfmt::group_chrome(group.first.affected_rows)
1254    )
1255}
1256
1257fn key_repeated_summary(group: &Group<'_>) -> String {
1258    match group.results.intent.as_ref().and_then(|i| i.key.as_ref()) {
1259        Some(key) => format!(
1260            "{} {} repeat; {} extra {}",
1261            numfmt::group_chrome(key.groups),
1262            if key.groups == 1 { "value" } else { "values" },
1263            numfmt::group_chrome(key.extra_rows),
1264            if key.extra_rows == 1 { "row" } else { "rows" }
1265        ),
1266        None => group.rows(),
1267    }
1268}
1269
1270/// Whole numbers written with a leading zero keep it only as text, which makes them
1271/// a code rather than a quantity.
1272fn is_code(profile: &ColumnQualityProfile) -> bool {
1273    profile.leading_zero_count.is_some_and(|count| count > 0)
1274        || (profile.min_length.is_some()
1275            && profile.min_length == profile.max_length
1276            && profile.min_length.is_some_and(|length| length > 1))
1277}
1278
1279fn code_shape(profile: &ColumnQualityProfile) -> String {
1280    let zeros = profile.leading_zero_count.unwrap_or(0);
1281    match (profile.min_length, profile.max_length) {
1282        (Some(min), Some(max)) if min == max && zeros > 0 => {
1283            format!("{min} digits, leading zeros")
1284        }
1285        (Some(min), Some(max)) if min == max => format!("{min} digits each"),
1286        _ => format!("{} with a leading zero", numfmt::group_chrome(zeros)),
1287    }
1288}
1289
1290pub(crate) fn quoted(value: &str, width: usize) -> String {
1291    let text = format!("{value:?}");
1292    if crate::glyphs::display_width(&text) <= width {
1293        text
1294    } else {
1295        let mut cut = String::new();
1296        for ch in text.chars() {
1297            if crate::glyphs::display_width(&cut) + 2 > width {
1298                break;
1299            }
1300            cut.push(ch);
1301        }
1302        format!("{cut}{}", crate::glyphs::get().ellipsis)
1303    }
1304}
1305
1306/// Column names joined until `width`, then "+N" for the rest.
1307pub fn columns_label(columns: &[String], width: usize) -> String {
1308    let mut label = String::new();
1309    for (index, column) in columns.iter().enumerate() {
1310        let candidate = if label.is_empty() {
1311            column.clone()
1312        } else {
1313            format!("{label}, {column}")
1314        };
1315        let rest = columns.len() - index - 1;
1316        let suffix = if rest > 0 {
1317            format!(" +{rest}")
1318        } else {
1319            String::new()
1320        };
1321        let needed =
1322            crate::glyphs::display_width(&candidate) + crate::glyphs::display_width(&suffix);
1323        if !label.is_empty() && needed > width {
1324            return format!("{label} +{}", rest + 1);
1325        }
1326        label = candidate;
1327    }
1328    label
1329}
1330
1331/// What one check found: nothing, something, nothing because there was nothing for
1332/// it to look at, or nothing because this run could not look.
1333#[derive(Debug, Clone, PartialEq, Eq)]
1334pub enum Outcome {
1335    Passed,
1336    Found {
1337        tier: Severity,
1338        detail: String,
1339    },
1340    /// Nothing in the data for it: no column of its kind, or one file.
1341    Skipped(&'static str),
1342    /// It applies, and this run could not answer it: no values read, or a sample
1343    /// where the answer needs every row.
1344    Unavailable(&'static str),
1345}
1346
1347/// One check the run makes: what it looks for, how far it reached, what it found.
1348/// The list is the answer to "checked for what?" when nothing turned up.
1349#[derive(Debug, Clone)]
1350pub struct Check {
1351    pub name: &'static str,
1352    pub looks_for: &'static str,
1353    pub applies_to: String,
1354    pub outcome: Outcome,
1355    /// What its numbers were read from: every row, a sample, or file footers.
1356    pub basis: QualityPrecision,
1357}
1358
1359/// How many checks the collapsed list shows before "more".
1360pub const CHECKS_SHOWN: usize = 6;
1361
1362/// Why a value check could not run on a scope with no rows.
1363const NO_ROWS: &str = "no rows to check";
1364
1365/// Every check, most important first.
1366pub fn checks(results: &DataQualityResults, report: &QualityReport) -> Vec<Check> {
1367    use polars::prelude::DataType;
1368    let columns = &results.columns;
1369    let count = |filter: &dyn Fn(&ColumnQualityProfile) -> bool| {
1370        columns.iter().filter(|profile| filter(profile)).count()
1371    };
1372    let text = |profile: &ColumnQualityProfile| {
1373        matches!(profile.dtype, DataType::String | DataType::Categorical(..))
1374    };
1375    let all = columns.len();
1376    let floats = count(&|profile| profile.dtype.is_float());
1377    let texts = count(&text);
1378    let keys = count(&|profile| profile.dtype.is_integer() || text(profile));
1379    let reach = |count: usize, kind: &str| {
1380        let noun = if count == 1 { "column" } else { "columns" };
1381        if kind.is_empty() {
1382            format!("{} {noun}", numfmt::group_chrome(count))
1383        } else {
1384            format!("{} {kind} {noun}", numfmt::group_chrome(count))
1385        }
1386    };
1387    let values_read = !report.metadata_only;
1388    // Columns behind the findings a check makes. Nothing to look at is known from
1389    // the schema, whatever the run read.
1390    let outcome = |check: &str, applies: usize, none: &'static str| {
1391        if applies == 0 {
1392            return Outcome::Skipped(none);
1393        }
1394        if !values_read {
1395            return Outcome::Unavailable("values not read");
1396        }
1397        if report.no_rows {
1398            return Outcome::Unavailable(NO_ROWS);
1399        }
1400        found(report, check)
1401    };
1402    let files = results.source_files.filter(|files| *files > 1);
1403    let by_files = |check: &str| match files {
1404        Some(_) => found(report, check),
1405        None => Outcome::Skipped("needs several files"),
1406    };
1407    // A dataset too large to read every footer is checked over the footers read.
1408    let files_reach = match (files, results.footers_read) {
1409        (Some(files), Some(read)) if read < files => format!(
1410            "{} of {} files",
1411            numfmt::group_chrome(read),
1412            numfmt::group_chrome(files)
1413        ),
1414        (Some(files), _) => format!("{} files", numfmt::group_chrome(files)),
1415        (None, _) => "files".to_string(),
1416    };
1417    let values = results.precision;
1418    let mut checks = Vec::new();
1419    // Declared, so first: the question the user asked before the ones datui asks.
1420    if let Some(intent) = &results.intent {
1421        checks.push(Check {
1422            name: INTENT_CHECK,
1423            looks_for: "values against the key and rules declared",
1424            applies_to: reach(intent.declared.len(), "declared"),
1425            outcome: if !intent.measured {
1426                Outcome::Unavailable("values not read")
1427            } else if report.no_rows {
1428                Outcome::Unavailable(NO_ROWS)
1429            } else {
1430                found(report, INTENT_CHECK)
1431            },
1432            basis: intent.precision,
1433        });
1434    }
1435    checks.extend([
1436        Check {
1437            name: "Missing values",
1438            looks_for: "nulls in any column",
1439            applies_to: reach(all, ""),
1440            outcome: outcome("Missing values", all, "no columns"),
1441            basis: values,
1442        },
1443        Check {
1444            name: "NaN or infinite",
1445            looks_for: "NaN or +/-infinity in float columns",
1446            applies_to: reach(floats, "float"),
1447            outcome: outcome("NaN or infinite", floats, "no float columns"),
1448            basis: values,
1449        },
1450        Check {
1451            name: "Duplicate rows",
1452            looks_for: "rows identical in every column",
1453            applies_to: "whole rows".to_string(),
1454            outcome: match results.identity.as_ref() {
1455                _ if !values_read => Outcome::Unavailable("values not read"),
1456                _ if report.no_rows => Outcome::Unavailable(NO_ROWS),
1457                Some(identity) if identity.extra_rows > 0 => Outcome::Found {
1458                    tier: Severity::Problem,
1459                    detail: format!("{} extra rows", numfmt::group_chrome(identity.extra_rows)),
1460                },
1461                Some(_) => Outcome::Passed,
1462                None => Outcome::Unavailable("not measured"),
1463            },
1464            basis: values,
1465        },
1466        Check {
1467            name: "Blank text",
1468            looks_for: "text that is empty or only whitespace",
1469            applies_to: reach(texts, "text"),
1470            outcome: outcome("Blank text", texts, "no text columns"),
1471            basis: values,
1472        },
1473        Check {
1474            name: "Mixed spellings",
1475            looks_for: "one value in several cases or spacings",
1476            applies_to: reach(texts, "text"),
1477            outcome: outcome("Mixed spellings", texts, "no text columns"),
1478            basis: values,
1479        },
1480        Check {
1481            name: "Type mismatch",
1482            looks_for: "a column typed differently by some files",
1483            applies_to: files_reach.clone(),
1484            outcome: by_files("Type mismatch"),
1485            basis: QualityPrecision::Metadata,
1486        },
1487        Check {
1488            name: "Missing in files",
1489            looks_for: "a column some files do not have",
1490            applies_to: files_reach,
1491            outcome: by_files("Missing in files"),
1492            basis: QualityPrecision::Metadata,
1493        },
1494        Check {
1495            name: "Numbers as text",
1496            looks_for: "text that reads as numbers or dates",
1497            applies_to: reach(texts, "text"),
1498            outcome: outcome("Numbers as text", texts, "no text columns"),
1499            basis: values,
1500        },
1501        Check {
1502            name: "Nearly unique",
1503            looks_for: "a would-be key whose values repeat",
1504            applies_to: reach(keys, "integer/text"),
1505            outcome: if keys > 0 && values_read && results.precision != QualityPrecision::Exact {
1506                Outcome::Unavailable("needs every row checked")
1507            } else {
1508                match outcome("Nearly unique", keys, "no integer or text columns") {
1509                    // A declared key's repeats replace the note on that column; the
1510                    // check found them all the same.
1511                    Outcome::Passed if declared_key_repeats(results) => Outcome::Found {
1512                        tier: Severity::Problem,
1513                        detail: "the declared key repeats".to_string(),
1514                    },
1515                    outcome => outcome,
1516                }
1517            },
1518            basis: values,
1519        },
1520        Check {
1521            name: "Single value",
1522            looks_for: "a column with one value throughout",
1523            applies_to: reach(all, ""),
1524            outcome: outcome("Single value", all, "no columns"),
1525            basis: values,
1526        },
1527    ]);
1528    checks
1529}
1530
1531/// Whether a one-column declared key repeats: the case whose "Nearly unique" note
1532/// the key's own finding replaces.
1533fn declared_key_repeats(results: &DataQualityResults) -> bool {
1534    results
1535        .intent
1536        .as_ref()
1537        .and_then(|intent| intent.key.as_ref())
1538        .is_some_and(|key| key.columns.len() == 1 && key.rows_involved > 0)
1539}
1540
1541/// The check the declared intent makes, by the name the Checks list gives it.
1542pub const INTENT_CHECK: &str = "Column intent";
1543
1544/// What the findings `check` makes found: how many columns, at their worst severity.
1545fn found(report: &QualityReport, check: &str) -> Outcome {
1546    let mut columns = Vec::new();
1547    let mut tier = Severity::Note;
1548    for finding in report
1549        .findings
1550        .iter()
1551        .filter(|finding| finding.check() == Some(check))
1552    {
1553        tier = tier.min(finding.severity);
1554        for column in &finding.columns {
1555            if !columns.contains(column) {
1556                columns.push(column.clone());
1557            }
1558        }
1559    }
1560    match columns.len() {
1561        0 => Outcome::Passed,
1562        count => Outcome::Found {
1563            tier,
1564            detail: format!(
1565                "{} {}",
1566                numfmt::group_chrome(count),
1567                if count == 1 { "column" } else { "columns" }
1568            ),
1569        },
1570    }
1571}
1572
1573/// A segment with fewer sampled rows than this is thin: only a large change in it
1574/// clears the sampling noise, and a clean one says little.
1575pub const THIN_SEGMENT_ROWS: usize = 30;
1576
1577/// How far the findings reach, beside them on every report: the checks that ran
1578/// and what they read, the ones that did not and why, the rows behind the numbers,
1579/// and what else bounds them. From what the run measured and saw; nothing here
1580/// reads.
1581#[derive(Debug, Clone, Default, PartialEq, Eq)]
1582pub struct Coverage {
1583    /// Checks that ran over every row in scope.
1584    pub exact: usize,
1585    /// Checks that ran over a sample.
1586    pub sampled: usize,
1587    /// Checks that ran over file footers.
1588    pub metadata: usize,
1589    /// Checks with nothing in the data to look at.
1590    pub skipped: usize,
1591    /// Checks that apply and this run could not answer: each reason, and the checks
1592    /// it kept from running.
1593    pub unavailable: Vec<(&'static str, Vec<&'static str>)>,
1594    /// The rows the numbers are over, with their denominator, and what the run's
1595    /// reads were seen to traverse.
1596    pub rows: Vec<String>,
1597    /// What else bounds the findings: thin segments, footers not read, time roles
1598    /// that measure nothing.
1599    pub limits: Vec<String>,
1600}
1601
1602impl Coverage {
1603    /// Checks by what they read, then the ones that did not run: "6 sampled ·
1604    /// 2 metadata · 1 skipped · 1 unavailable".
1605    pub fn checks(&self) -> Vec<String> {
1606        let unavailable = self
1607            .unavailable
1608            .iter()
1609            .map(|(_, names)| names.len())
1610            .sum::<usize>();
1611        [
1612            (self.exact, QualityPrecision::Exact.label()),
1613            (self.sampled, QualityPrecision::Sampled.label()),
1614            (self.metadata, QualityPrecision::Metadata.label()),
1615            (self.skipped, "skipped"),
1616            (unavailable, "unavailable"),
1617        ]
1618        .into_iter()
1619        .filter(|(count, _)| *count > 0)
1620        .map(|(count, label)| format!("{} {label}", numfmt::group_chrome(count)))
1621        .collect()
1622    }
1623
1624    /// Why each unavailable check did not run, then the other limits.
1625    pub fn limits(&self) -> Vec<String> {
1626        self.unavailable
1627            .iter()
1628            .map(|(reason, names)| match names.as_slice() {
1629                [name] => format!("{name}: {reason}"),
1630                _ => format!("{} checks: {reason}", names.len()),
1631            })
1632            .chain(self.limits.iter().cloned())
1633            .collect()
1634    }
1635}
1636
1637pub fn coverage(
1638    results: &DataQualityResults,
1639    checks: &[Check],
1640    plan: &DataQualityPlan,
1641) -> Coverage {
1642    let mut coverage = Coverage::default();
1643    for check in checks {
1644        match &check.outcome {
1645            Outcome::Skipped(_) => coverage.skipped += 1,
1646            Outcome::Unavailable(reason) => {
1647                match coverage
1648                    .unavailable
1649                    .iter_mut()
1650                    .find(|(known, _)| known == reason)
1651                {
1652                    Some((_, names)) => names.push(check.name),
1653                    None => coverage.unavailable.push((reason, vec![check.name])),
1654                }
1655            }
1656            Outcome::Passed | Outcome::Found { .. } => match check.basis {
1657                QualityPrecision::Exact => coverage.exact += 1,
1658                QualityPrecision::Metadata => coverage.metadata += 1,
1659                QualityPrecision::Sampled => coverage.sampled += 1,
1660            },
1661        }
1662    }
1663
1664    let count = numfmt::group_chrome;
1665    let evaluated = results.evaluated_rows;
1666    coverage
1667        .rows
1668        .push(match (results.precision, results.total_rows) {
1669            (QualityPrecision::Metadata, _) => "none read, file metadata only".to_string(),
1670            _ if evaluated == 0 => "none: the scope has no rows".to_string(),
1671            (QualityPrecision::Exact, _) => format!("all {} read, exact", count(evaluated)),
1672            (_, Some(total)) => format!(
1673                "{} of {} sampled ({})",
1674                count(evaluated),
1675                count(total),
1676                crate::numfmt::percent_of(evaluated, total)
1677            ),
1678            (_, None) => format!("{} sampled, total not counted", count(evaluated)),
1679        });
1680    if let Some(each) = results.per_value {
1681        coverage
1682            .rows
1683            .push(format!("up to {} per value", count(each)));
1684    }
1685    // Only what the reads were seen to do: a read that counted nothing says nothing.
1686    if let Some(reads) = results
1687        .reads
1688        .filter(|_| results.precision != QualityPrecision::Metadata)
1689    {
1690        if reads.reads == 0 {
1691            coverage.rows.push("no source read".to_string());
1692        } else if reads.counted == reads.reads {
1693            coverage
1694                .rows
1695                .push(format!("{} traversed", count(reads.rows)));
1696        } else if reads.counted > 0 {
1697            coverage
1698                .rows
1699                .push(format!("at least {} traversed", count(reads.rows)));
1700        }
1701        if let Some(copy) = reads.copy {
1702            let bytes = crate::numfmt::bytes(copy.bytes);
1703            coverage.rows.push(if copy.fetched {
1704                format!("passes read a local copy, fetched once ({bytes})")
1705            } else {
1706                format!("passes read a local copy fetched earlier ({bytes})")
1707            });
1708        }
1709    }
1710
1711    let segments = &results.segments;
1712    if matches!(results.precision, QualityPrecision::Sampled) && segments.len() > 1 {
1713        let thin = segments
1714            .iter()
1715            .filter(|segment| segment.evaluated_rows < THIN_SEGMENT_ROWS)
1716            .count();
1717        if thin > 0 {
1718            coverage.limits.push(format!(
1719                "{} of {} segments under {THIN_SEGMENT_ROWS} sampled rows",
1720                count(thin),
1721                count(segments.len())
1722            ));
1723        }
1724    }
1725    if !results.unsampled_segments.is_empty() {
1726        coverage.limits.push(format!(
1727            "{} segments with rows, none sampled",
1728            count(results.unsampled_segments.len())
1729        ));
1730    }
1731    if let (Some(files), Some(read)) = (results.source_files, results.footers_read)
1732        && read < files
1733    {
1734        coverage.limits.push(format!(
1735            "footers of {} of {} files read",
1736            count(read),
1737            count(files)
1738        ));
1739    }
1740    if let Some(intent) = results.intent.as_ref().filter(|intent| intent.measured) {
1741        // A key with no repeat in a sample is unique among those rows, and no more.
1742        if intent.key.is_some() && intent.precision != QualityPrecision::Exact {
1743            coverage.limits.push(format!(
1744                "key repeats among {} sampled rows only",
1745                count(intent.evaluated_rows)
1746            ));
1747        }
1748    }
1749    if let Some(intent) = &results.intent
1750        && !intent.absent.is_empty()
1751    {
1752        coverage.limits.push(format!(
1753            "intent on {}: not in scope",
1754            columns_label(&intent.absent, 24)
1755        ));
1756    }
1757    if !plan.temporal_roles.is_empty() && plan.interval_pairs().is_empty() {
1758        coverage
1759            .limits
1760            .push("time roles form no interval".to_string());
1761    }
1762    coverage
1763}
1764
1765/// What a finding means for the data and what to do about it, a fragment a line.
1766pub fn advice(finding: &Finding) -> Vec<String> {
1767    let Some(kind) = finding.kind else {
1768        return Vec::new();
1769    };
1770    if finding.variant == Some(Variant::MostlyMissing) {
1771        let rest = finding.evaluated_rows.saturating_sub(finding.affected_rows);
1772        return vec![
1773            if finding.columns.len() == 1 {
1774                format!(
1775                    "Aggregates and joins see only {}",
1776                    crate::numfmt::percent_of(rest, finding.evaluated_rows)
1777                )
1778            } else {
1779                "Aggregates and joins see only the filled rows".to_string()
1780            },
1781            "Check: filled only for some rows, or stopped at some point".to_string(),
1782        ];
1783    }
1784    finding
1785        .variant
1786        .and_then(Variant::advice)
1787        .unwrap_or(kind.spec().advice)
1788        .iter()
1789        .map(|line| line.to_string())
1790        .collect()
1791}
1792
1793/// A finding as its detail reads it.
1794struct Detail<'a> {
1795    finding: &'a Finding,
1796    results: &'a DataQualityResults,
1797    /// "1,234 of 61,700 rows (2.0%)".
1798    of: String,
1799}
1800
1801impl Detail<'_> {
1802    fn profile(&self, name: &str) -> Option<&ColumnQualityProfile> {
1803        self.results
1804            .columns
1805            .iter()
1806            .find(|profile| profile.name == name)
1807    }
1808
1809    fn observation(&self, index: usize) -> &QualityObservation {
1810        &self.results.observations[index]
1811    }
1812
1813    /// "1,234 of 61,700 values (2.0%) {what}".
1814    fn values(&self, what: &str) -> String {
1815        let finding = self.finding;
1816        format!(
1817            "{} of {} values ({}) {what}",
1818            numfmt::group_chrome(finding.affected_rows),
1819            numfmt::group_chrome(finding.evaluated_rows),
1820            crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
1821        )
1822    }
1823}
1824
1825/// The finding's numbers in a fragment, then the evidence that makes it concrete:
1826/// the values, the spellings, the files.
1827pub fn describe(finding: &Finding, results: &DataQualityResults) -> (String, Vec<String>) {
1828    let count = numfmt::group_chrome;
1829    let mut evidence = Vec::new();
1830    let Some(kind) = finding.kind else {
1831        let headline = format!(
1832            "{} {} passed every check",
1833            count(finding.columns.len()),
1834            if finding.columns.len() == 1 {
1835                "column"
1836            } else {
1837                "columns"
1838            }
1839        );
1840        return (headline, evidence);
1841    };
1842    let detail = Detail {
1843        finding,
1844        results,
1845        of: format!(
1846            "{} of {} rows ({})",
1847            count(finding.affected_rows),
1848            count(finding.evaluated_rows),
1849            crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
1850        ),
1851    };
1852    let headline = (kind.spec().headline)(&detail, &mut evidence);
1853    (headline, evidence)
1854}
1855
1856fn nulls_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1857    let (finding, of) = (detail.finding, &detail.of);
1858    if finding.severity == Severity::Problem {
1859        format!(
1860            "Null in all {} rows checked",
1861            numfmt::group_chrome(finding.evaluated_rows)
1862        )
1863    } else if finding.same_rows {
1864        if finding.columns.len() == 2 {
1865            evidence.push("No row misses one without the other".to_string());
1866            format!("{of} null in both columns")
1867        } else {
1868            evidence.push("No row misses one of them without the others".to_string());
1869            format!("{of} null in all {} columns", finding.columns.len())
1870        }
1871    } else if finding.varied() {
1872        // Each column's own rate, worst first: the list row gave only the range.
1873        evidence.extend(finding.breakdown(detail.results));
1874        format!("Null rate in {} columns:", finding.columns.len())
1875    } else {
1876        per_column_headline(detail, evidence)
1877    }
1878}
1879
1880/// "{of} {noun}", or the same of each column, one a line, when several columns are
1881/// grouped by what they miss.
1882fn per_column_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1883    let (finding, of) = (detail.finding, &detail.of);
1884    let noun = finding.kind.map_or("", |kind| kind.spec().noun);
1885    if finding.lists_columns() {
1886        evidence.extend(finding.breakdown(detail.results));
1887        format!("{of} {noun} in each column:")
1888    } else {
1889        format!("{of} {noun}")
1890    }
1891}
1892
1893fn non_finite_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1894    let count = numfmt::group_chrome;
1895    if let Some(profile) = detail.profile(&detail.finding.columns[0]) {
1896        evidence.push(format!(
1897            "NaN {}, +inf {}, -inf {}",
1898            count(profile.nan_count.unwrap_or(0)),
1899            count(profile.positive_infinity_count.unwrap_or(0)),
1900            count(profile.negative_infinity_count.unwrap_or(0))
1901        ));
1902    }
1903    format!("{} NaN or infinite", detail.of)
1904}
1905
1906fn constant_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1907    let finding = detail.finding;
1908    let rows = numfmt::group_chrome(finding.evaluated_rows);
1909    if finding.columns.len() > 1 {
1910        for column in &finding.columns {
1911            if let Some(value) = detail
1912                .profile(column)
1913                .and_then(|p| p.dominant_value.as_ref())
1914            {
1915                evidence.push(format!("{column}: always {}", quoted(value, 40)));
1916            }
1917        }
1918        return format!("One value per column in {rows} rows");
1919    }
1920    match detail
1921        .profile(&finding.columns[0])
1922        .and_then(|p| p.dominant_value.as_ref())
1923    {
1924        Some(value) => format!("Always {} in {rows} rows", quoted(value, 40)),
1925        None => format!("One value in {rows} rows"),
1926    }
1927}
1928
1929fn parseable_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1930    let count = numfmt::group_chrome;
1931    let column = &detail.finding.columns[0];
1932    let profile = detail.profile(column);
1933    let (Some(profile), Some((parsed, reading))) = (profile, profile.and_then(text_reading)) else {
1934        return detail.finding.summary.clone();
1935    };
1936    if let (Some(min), Some(max)) = (profile.min_length, profile.max_length) {
1937        evidence.push(if min == max {
1938            format!("{min} characters each")
1939        } else {
1940            format!("{min} to {max} characters")
1941        });
1942    }
1943    if let Some(zeros) = profile.leading_zero_count.filter(|zeros| *zeros > 0) {
1944        evidence.push(format!("{} with a leading zero", count(zeros)));
1945    }
1946    if let (Some(min), Some(max)) = (&profile.min, &profile.max) {
1947        evidence.push(format!("From {} to {}", quoted(min, 24), quoted(max, 24)));
1948    }
1949    let failed = profile.non_null_rows().saturating_sub(parsed);
1950    if failed > 0 {
1951        let examples = detail
1952            .results
1953            .examples_of(ObservationKind::ParseableText, column);
1954        evidence.push(if examples.is_empty() {
1955            format!("{} do not parse", count(failed))
1956        } else {
1957            crate::glyphs::fit(
1958                &format!(
1959                    "{} do not parse, such as {}",
1960                    count(failed),
1961                    examples.join(", ")
1962                ),
1963                EXAMPLE_WIDTH,
1964            )
1965        });
1966    }
1967    format!(
1968        "{} of {} values ({}) parse as {}",
1969        count(parsed),
1970        count(profile.non_null_rows()),
1971        crate::numfmt::percent_of(parsed, profile.non_null_rows()),
1972        reading.label()
1973    )
1974}
1975
1976fn duplicates_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
1977    let count = numfmt::group_chrome;
1978    let Some(identity) = detail.results.identity.as_ref() else {
1979        return detail.finding.summary.clone();
1980    };
1981    evidence.push(format!(
1982        "{} of {} rows ({}) have a copy",
1983        count(identity.rows_involved),
1984        count(identity.evaluated_rows),
1985        crate::numfmt::percent_of(identity.rows_involved, identity.evaluated_rows)
1986    ));
1987    if !identity.examples.is_empty() {
1988        evidence.push("Most copied:".to_string());
1989    }
1990    for example in &identity.examples {
1991        evidence.push(crate::glyphs::fit(
1992            &format!(
1993                "{}{}  {}",
1994                crate::glyphs::get().times,
1995                example.copies,
1996                example.values.join(", ")
1997            ),
1998            EXAMPLE_WIDTH,
1999        ));
2000    }
2001    format!(
2002        "{} rows repeated; {} extra {}",
2003        count(identity.duplicate_groups),
2004        count(identity.extra_rows),
2005        if identity.extra_rows == 1 {
2006            "copy"
2007        } else {
2008            "copies"
2009        }
2010    )
2011}
2012
2013fn variants_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2014    let count = numfmt::group_chrome;
2015    for index in &detail.finding.observations {
2016        let normalized = detail.observation(*index).normalized_category.as_ref();
2017        let Some(group) = detail
2018            .results
2019            .category_variants
2020            .iter()
2021            .find(|group| Some(&group.normalized) == normalized)
2022        else {
2023            continue;
2024        };
2025        let mut variants = group.variants.iter().collect::<Vec<_>>();
2026        variants.sort_by_key(|(_, rows)| std::cmp::Reverse(*rows));
2027        evidence.push(
2028            variants
2029                .into_iter()
2030                .take(4)
2031                .map(|(value, rows)| format!("{} ({})", quoted(value, 28), count(*rows)))
2032                .collect::<Vec<_>>()
2033                .join("  "),
2034        );
2035    }
2036    let values = detail.finding.observations.len();
2037    format!(
2038        "{} {} spelled more than one way, {}",
2039        count(values),
2040        if values == 1 { "value" } else { "values" },
2041        detail.of
2042    )
2043}
2044
2045fn key_like_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2046    let count = numfmt::group_chrome;
2047    let Some(profile) = detail.profile(&detail.finding.columns[0]) else {
2048        return detail.finding.summary.clone();
2049    };
2050    if let (Some(value), Some(times)) = (&profile.dominant_value, profile.dominant_count) {
2051        evidence.push(format!(
2052            "Most repeated: {} ({} times)",
2053            quoted(value, 32),
2054            count(times)
2055        ));
2056    }
2057    format!(
2058        "{} distinct in {} rows; {} repeats",
2059        count(profile.distinct_count.unwrap_or(0)),
2060        count(profile.non_null_rows()),
2061        count(detail.finding.affected_rows)
2062    )
2063}
2064
2065fn unparsed_time_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2066    let observation = detail.observation(detail.finding.observations[0]);
2067    if let Some(format) = &observation.time_format {
2068        evidence.push(format!("Read as {} for this study only", format.label()));
2069    }
2070    let examples = detail
2071        .results
2072        .examples_of(ObservationKind::UnparsedTime, &observation.column);
2073    if !examples.is_empty() {
2074        evidence.push(crate::glyphs::fit(
2075            &format!("Such as {}", examples.join(", ")),
2076            EXAMPLE_WIDTH,
2077        ));
2078    }
2079    detail.values("do not parse")
2080}
2081
2082/// Each channel's runs or mean, then `{of} {noun}`.
2083fn signal_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2084    signal_evidence(detail, evidence);
2085    let noun = detail.finding.kind.map_or("", |kind| kind.spec().noun);
2086    format!("{} {noun}", detail.of)
2087}
2088
2089fn dc_offset_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2090    signal_evidence(detail, evidence);
2091    "Mean away from zero".to_string()
2092}
2093
2094fn signal_evidence(detail: &Detail<'_>, evidence: &mut Vec<String>) {
2095    for index in &detail.finding.observations {
2096        let observation = detail.observation(*index);
2097        evidence.push(format!("{}: {}", observation.column, observation.fact));
2098    }
2099}
2100
2101fn files_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2102    let count = numfmt::group_chrome;
2103    let finding = detail.finding;
2104    let observation = detail.observation(finding.observations[0]);
2105    // Read from the footers, so the denominator is the whole loaded source whatever
2106    // the plan's scope was.
2107    evidence.push(format!(
2108        "{} of {} rows of the loaded source ({})",
2109        count(finding.affected_rows),
2110        count(finding.evaluated_rows),
2111        crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows)
2112    ));
2113    for file in observation.files.iter().take(4) {
2114        let stored = file
2115            .stored_type
2116            .as_ref()
2117            .map(|dtype| format!(" as {dtype}"))
2118            .unwrap_or_default();
2119        let examples = if file.examples.is_empty() {
2120            String::new()
2121        } else {
2122            format!(
2123                ": {}",
2124                file.examples
2125                    .iter()
2126                    .map(|value| quoted(value, 20))
2127                    .collect::<Vec<_>>()
2128                    .join(", ")
2129            )
2130        };
2131        evidence.push(format!(
2132            "#{} {} ({} rows){stored}{examples}",
2133            file.number,
2134            file.name,
2135            count(file.rows)
2136        ));
2137    }
2138    if observation.files.len() > 4 {
2139        evidence.push(format!(
2140            "{} more {}",
2141            observation.files.len() - 4,
2142            crate::glyphs::get().ellipsis
2143        ));
2144    }
2145    upper_first(&observation.fact)
2146}
2147
2148/// How wide a line of examples runs before it is cut: a row of many columns would
2149/// otherwise wrap over the whole detail.
2150const EXAMPLE_WIDTH: usize = 72;
2151
2152/// A declared rule's violation, as the intent measured it: the count against what it
2153/// is out of, the rule as declared, and what the rows in memory showed of it. A
2154/// sample's numbers say so.
2155struct Declared<'a> {
2156    intent: &'a crate::analysis::quality_intent::IntentResults,
2157    sampled: bool,
2158    /// "rows", or "sampled rows".
2159    rows_word: &'static str,
2160    /// The finding's share of what it is out of.
2161    share: String,
2162    check: Option<&'a crate::analysis::quality_intent::ColumnCheck>,
2163    column: &'a str,
2164}
2165
2166impl<'a> Declared<'a> {
2167    fn of(detail: &Detail<'a>) -> Option<Self> {
2168        let intent = detail.results.intent.as_ref()?;
2169        let finding = detail.finding;
2170        let column = finding.columns.first().map(String::as_str).unwrap_or("");
2171        let sampled = intent.precision != QualityPrecision::Exact;
2172        Some(Self {
2173            intent,
2174            sampled,
2175            rows_word: if sampled { "sampled rows" } else { "rows" },
2176            share: crate::numfmt::percent_of(finding.affected_rows, finding.evaluated_rows),
2177            check: intent.column(column),
2178            column,
2179        })
2180    }
2181}
2182
2183/// Values and their rows: `"void" (3)  "x" (1)`.
2184fn value_examples(values: &[(String, usize)]) -> String {
2185    values
2186        .iter()
2187        .map(|(value, rows)| format!("{} ({})", quoted(value, 24), numfmt::group_chrome(*rows)))
2188        .collect::<Vec<_>>()
2189        .join("  ")
2190}
2191
2192fn key_repeated_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2193    let count = numfmt::group_chrome;
2194    let Some(declared) = Declared::of(detail) else {
2195        return detail.finding.summary.clone();
2196    };
2197    let Some(key) = declared.intent.key.as_ref() else {
2198        return detail.finding.summary.clone();
2199    };
2200    evidence.push(format!("Declared key: {}", key.columns.join(", ")));
2201    evidence.push(format!(
2202        "{} {} held by more than one row; {} rows beyond one per value",
2203        count(key.groups),
2204        if key.groups == 1 { "value" } else { "values" },
2205        count(key.extra_rows)
2206    ));
2207    if declared.sampled {
2208        evidence.push(
2209            "Each repeat here is one in the data; unsampled rows are not checked".to_string(),
2210        );
2211    }
2212    format!(
2213        "{} of {} {} ({}) share their key with another row",
2214        count(key.rows_involved),
2215        count(declared.intent.evaluated_rows),
2216        declared.rows_word,
2217        declared.share
2218    )
2219}
2220
2221fn key_missing_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2222    let count = numfmt::group_chrome;
2223    let Some(declared) = Declared::of(detail) else {
2224        return detail.finding.summary.clone();
2225    };
2226    let Some(key) = declared.intent.key.as_ref() else {
2227        return detail.finding.summary.clone();
2228    };
2229    evidence.push(format!("Declared key: {}", key.columns.join(", ")));
2230    format!(
2231        "{} of {} {} ({}) have no value in part of the key",
2232        count(key.missing),
2233        count(declared.intent.evaluated_rows),
2234        declared.rows_word,
2235        declared.share
2236    )
2237}
2238
2239fn required_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2240    let count = numfmt::group_chrome;
2241    let Some(declared) = Declared::of(detail) else {
2242        return detail.finding.summary.clone();
2243    };
2244    let column = declared.column;
2245    evidence.push(format!("Declared required: {column}"));
2246    format!(
2247        "{} of {} {} ({}) have no {column}",
2248        count(detail.finding.affected_rows),
2249        count(detail.finding.evaluated_rows),
2250        declared.rows_word,
2251        declared.share
2252    )
2253}
2254
2255fn not_allowed_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2256    let Some(declared) = Declared::of(detail) else {
2257        return detail.finding.summary.clone();
2258    };
2259    if let Some(check) = declared.check {
2260        evidence.push(format!("Allowed: {}", check.intent.allowed_label(8)));
2261        if !check.outside_examples.is_empty() {
2262            evidence.push(format!(
2263                "Found: {}",
2264                value_examples(&check.outside_examples)
2265            ));
2266        }
2267    }
2268    detail.values("are not allowed")
2269}
2270
2271fn out_of_range_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2272    let count = numfmt::group_chrome;
2273    let Some(declared) = Declared::of(detail) else {
2274        return detail.finding.summary.clone();
2275    };
2276    if let Some(check) = declared.check {
2277        evidence.push(format!(
2278            "Range: {}",
2279            check.intent.range_label().unwrap_or_default()
2280        ));
2281        if let Some(below) = check.below.filter(|below| *below > 0) {
2282            let lowest = check
2283                .lowest
2284                .as_ref()
2285                .map(|value| format!(", lowest {value}"))
2286                .unwrap_or_default();
2287            evidence.push(format!("Below: {}{lowest}", count(below)));
2288        }
2289        if let Some(above) = check.above.filter(|above| *above > 0) {
2290            let highest = check
2291                .highest
2292                .as_ref()
2293                .map(|value| format!(", highest {value}"))
2294                .unwrap_or_default();
2295            evidence.push(format!("Above: {}{highest}", count(above)));
2296        }
2297    }
2298    detail.values("outside the range")
2299}
2300
2301fn unparsed_number_headline(detail: &Detail<'_>, evidence: &mut Vec<String>) -> String {
2302    let Some(declared) = Declared::of(detail) else {
2303        return detail.finding.summary.clone();
2304    };
2305    let reading = declared.check.and_then(|check| check.intent.number).map_or(
2306        "number",
2307        crate::analysis::quality_intent::NumberReading::label,
2308    );
2309    if let Some(check) = declared
2310        .check
2311        .filter(|check| !check.unparsed_examples.is_empty())
2312    {
2313        evidence.push(format!(
2314            "Such as: {}",
2315            value_examples(&check.unparsed_examples)
2316        ));
2317    }
2318    evidence.push(format!("Read as a {reading} for this study only"));
2319    detail.values(&format!("do not read as a {reading}"))
2320}
2321
2322fn upper_first(text: &str) -> String {
2323    let mut chars = text.chars();
2324    match chars.next() {
2325        Some(first) => first.to_uppercase().chain(chars).collect(),
2326        None => String::new(),
2327    }
2328}
2329
2330/// Headline counts, one line: the answer before the evidence.
2331pub fn verdict(report: &QualityReport) -> String {
2332    // Footers still say which files lack a column or hold it in another type, so a
2333    // metadata run can find problems; it cannot call a column clean.
2334    if report.metadata_only {
2335        return match report.problems {
2336            0 => "Values not read: file metadata only".to_string(),
2337            1 => "1 problem in file metadata; values not read".to_string(),
2338            count => format!(
2339                "{} problems in file metadata; values not read",
2340                numfmt::group_chrome(count)
2341            ),
2342        };
2343    }
2344    // No rows is no evidence: nothing passed, and nothing is clean.
2345    if report.no_rows {
2346        return match report.problems {
2347            0 => "No rows to check".to_string(),
2348            1 => "1 problem in file metadata; no rows to check".to_string(),
2349            count => format!(
2350                "{} problems in file metadata; no rows to check",
2351                numfmt::group_chrome(count)
2352            ),
2353        };
2354    }
2355    let clean = format!(
2356        "{} of {} columns clean",
2357        numfmt::group_chrome(report.clean_columns),
2358        numfmt::group_chrome(report.total_columns)
2359    );
2360    let notes = match report.notes {
2361        0 => String::new(),
2362        1 => "1 note  ".to_string(),
2363        count => format!("{} notes  ", numfmt::group_chrome(count)),
2364    };
2365    match report.problems {
2366        0 => format!("No problems found  {notes}{clean}"),
2367        1 => format!("1 problem  {notes}{clean}"),
2368        count => format!("{} problems  {notes}{clean}", numfmt::group_chrome(count)),
2369    }
2370}
2371
2372#[cfg(test)]
2373mod tests {
2374    use super::*;
2375
2376    /// What a finding is, for comparing: its kind and how it reads.
2377    fn reads(finding: &Finding) -> (Option<ObservationKind>, Option<Variant>) {
2378        (finding.kind, finding.variant)
2379    }
2380    use crate::analysis::data_quality::SharedNulls;
2381    use crate::analysis::data_quality::fixtures::{observation, profile, results_with};
2382    use polars::prelude::DataType;
2383
2384    /// Sixteen columns missing on the same rows are one fact, and it says so.
2385    #[test]
2386    fn nulls_with_one_count_become_one_finding() {
2387        let names = ["open", "high", "low", "close"];
2388        let mut columns = names
2389            .iter()
2390            .map(|name| profile(name, DataType::Float64))
2391            .collect::<Vec<_>>();
2392        columns.push(profile("ticker", DataType::String));
2393        let mut results = results_with(
2394            columns,
2395            names
2396                .iter()
2397                .map(|name| observation(ObservationKind::Nulls, name, 4))
2398                .collect(),
2399        );
2400        results.shared_nulls = vec![SharedNulls {
2401            columns: names.iter().map(|name| name.to_string()).collect(),
2402            null_rows: 4,
2403            rows_null_in_all: 4,
2404        }];
2405        let report = build_report(&results);
2406        assert_eq!(report.findings.len(), 2, "one note and the clean entry");
2407        let missing = &report.findings[0];
2408        assert_eq!(
2409            reads(missing),
2410            (Some(ObservationKind::Nulls), Some(Variant::MissingTogether))
2411        );
2412        assert_eq!(missing.severity, Severity::Note);
2413        assert_eq!(missing.columns.len(), 4);
2414        assert!(missing.same_rows);
2415        assert_eq!(missing.summary, "4 rows (4.0%)");
2416        assert_eq!(report.findings[1].severity, Severity::Clean);
2417        assert_eq!(report.findings[1].columns, vec!["ticker".to_string()]);
2418        assert_eq!(report.problems, 0);
2419        assert!(verdict(&report).starts_with("No problems found"));
2420    }
2421
2422    /// Missing values at different rates are one finding listing each column worst
2423    /// first; columns missing in most rows are their own note, ranked above it.
2424    #[test]
2425    fn missing_values_collapse_and_mostly_missing_leads() {
2426        let names = ["a", "b", "c", "d"];
2427        let results = results_with(
2428            names
2429                .iter()
2430                .map(|name| profile(name, DataType::Float64))
2431                .collect(),
2432            vec![
2433                observation(ObservationKind::Nulls, "a", 3),
2434                observation(ObservationKind::Nulls, "b", 40),
2435                observation(ObservationKind::Nulls, "c", 12),
2436                observation(ObservationKind::Nulls, "d", 80),
2437            ],
2438        );
2439        let report = build_report(&results);
2440        assert_eq!(report.findings.len(), 2);
2441        let mostly = &report.findings[0];
2442        assert_eq!(
2443            reads(mostly),
2444            (Some(ObservationKind::Nulls), Some(Variant::MostlyMissing))
2445        );
2446        assert_eq!(mostly.columns, vec!["d".to_string()]);
2447        let missing = &report.findings[1];
2448        assert_eq!(reads(missing), (Some(ObservationKind::Nulls), None));
2449        assert_eq!(missing.columns, vec!["b", "c", "a"]);
2450        assert_eq!(missing.summary, "3.0% to 40.0% per column");
2451        let (headline, evidence) = describe(missing, &results);
2452        assert_eq!(headline, "Null rate in 3 columns:");
2453        assert_eq!(evidence[0], "b    40.0%  40 rows");
2454        assert_eq!(evidence[2], "a     3.0%  3 rows");
2455    }
2456
2457    #[test]
2458    fn problems_rank_before_notes_and_mark_their_columns() {
2459        let results = results_with(
2460            vec![
2461                profile("price", DataType::Float64),
2462                profile("region", DataType::String),
2463            ],
2464            vec![
2465                observation(ObservationKind::Nulls, "region", 3),
2466                observation(ObservationKind::NonFinite, "price", 2),
2467            ],
2468        );
2469        let report = build_report(&results);
2470        assert_eq!(
2471            reads(&report.findings[0]),
2472            (Some(ObservationKind::NonFinite), None)
2473        );
2474        assert_eq!(report.findings[0].severity, Severity::Problem);
2475        assert_eq!(
2476            reads(&report.findings[1]),
2477            (Some(ObservationKind::Nulls), None)
2478        );
2479        assert_eq!(
2480            report.column_status,
2481            vec![Severity::Problem, Severity::Note]
2482        );
2483        assert_eq!(report.clean_columns, 0);
2484        assert_eq!(verdict(&report), "1 problem  1 note  0 of 2 columns clean");
2485    }
2486
2487    #[test]
2488    fn a_column_with_no_values_at_all_is_a_problem() {
2489        let results = results_with(
2490            vec![profile("legacy", DataType::String)],
2491            vec![observation(ObservationKind::Nulls, "legacy", 100)],
2492        );
2493        let report = build_report(&results);
2494        assert_eq!(
2495            reads(&report.findings[0]),
2496            (Some(ObservationKind::Nulls), Some(Variant::AlwaysMissing))
2497        );
2498        assert_eq!(report.findings[0].severity, Severity::Problem);
2499    }
2500
2501    /// Fixed-width digits with leading zeros are a code, and the report says the
2502    /// column is fine as text rather than asking for a cast.
2503    #[test]
2504    fn zero_padded_digits_read_as_codes() {
2505        let mut code = profile("industry", DataType::String);
2506        code.integer_parse_count = Some(100);
2507        code.decimal_parse_count = Some(100);
2508        code.leading_zero_count = Some(20);
2509        code.min_length = Some(4);
2510        code.max_length = Some(4);
2511        let results = results_with(
2512            vec![code],
2513            vec![observation(ObservationKind::ParseableText, "industry", 100)],
2514        );
2515        let finding = &build_report(&results).findings[0];
2516        assert_eq!(
2517            reads(finding),
2518            (
2519                Some(ObservationKind::ParseableText),
2520                Some(Variant::CodesAsText)
2521            )
2522        );
2523        assert_eq!(finding.summary, "4 digits, leading zeros");
2524    }
2525
2526    /// Footers can show a problem without reading a value; nothing can be called
2527    /// clean that way.
2528    #[test]
2529    fn a_metadata_run_reports_footer_problems_and_no_clean_columns() {
2530        let mut results = results_with(
2531            vec![
2532                profile("fee", DataType::Float64),
2533                profile("id", DataType::Int64),
2534            ],
2535            vec![observation(ObservationKind::Absent, "fee", 20)],
2536        );
2537        results.precision = QualityPrecision::Metadata;
2538        let report = build_report(&results);
2539        assert_eq!(report.findings.len(), 1, "no clean entry");
2540        assert_eq!(
2541            reads(&report.findings[0]),
2542            (Some(ObservationKind::Absent), None)
2543        );
2544        assert_eq!(
2545            verdict(&report),
2546            "1 problem in file metadata; values not read"
2547        );
2548    }
2549
2550    /// The checks say what they covered, what they found, and what they could not
2551    /// look at and why, so a clean result is one the reader can trust.
2552    #[test]
2553    fn checks_report_reach_findings_and_what_did_not_run() {
2554        let mut results = results_with(
2555            vec![
2556                profile("price", DataType::Float64),
2557                profile("region", DataType::String),
2558                profile("id", DataType::Int64),
2559            ],
2560            vec![observation(ObservationKind::NonFinite, "price", 2)],
2561        );
2562        results.precision = QualityPrecision::Sampled;
2563        let report = build_report(&results);
2564        let list = checks(&results, &report);
2565        let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2566        assert_eq!(list[0].name, "Missing values", "most important first");
2567        assert_eq!(by_name("Missing values").outcome, Outcome::Passed);
2568        assert_eq!(by_name("Missing values").applies_to, "3 columns");
2569        assert_eq!(by_name("NaN or infinite").applies_to, "1 float column");
2570        assert_eq!(
2571            by_name("NaN or infinite").outcome,
2572            Outcome::Found {
2573                tier: Severity::Problem,
2574                detail: "1 column".to_string()
2575            }
2576        );
2577        assert_eq!(
2578            by_name("Nearly unique").outcome,
2579            Outcome::Unavailable("needs every row checked"),
2580            "a sample cannot say a column is nearly a key"
2581        );
2582        assert_eq!(
2583            by_name("Type mismatch").outcome,
2584            Outcome::Skipped("needs several files")
2585        );
2586
2587        results.precision = QualityPrecision::Metadata;
2588        results.source_files = Some(3);
2589        let report = build_report(&results);
2590        let list = checks(&results, &report);
2591        let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2592        assert_eq!(
2593            by_name("Missing values").outcome,
2594            Outcome::Unavailable("values not read")
2595        );
2596        assert_eq!(by_name("Type mismatch").outcome, Outcome::Passed);
2597        assert_eq!(by_name("Type mismatch").applies_to, "3 files");
2598    }
2599
2600    /// Coverage tells a clean sample from an exhaustive run: which checks ran over
2601    /// what, which had nothing to look at, which this run could not answer and why,
2602    /// and what else bounds the result, each count beside its denominator.
2603    #[test]
2604    fn coverage_separates_checked_skipped_and_unavailable() {
2605        use crate::analysis::data_quality::{
2606            ObservedReads, SegmentQualityProfile, TemporalRole, TemporalRoleAssignment,
2607        };
2608        let columns = vec![
2609            profile("price", DataType::Float64),
2610            profile("region", DataType::String),
2611            profile("id", DataType::Int64),
2612        ];
2613        let plan = DataQualityPlan::default();
2614        let measured = |mut results: DataQualityResults| {
2615            results.identity = Some(crate::analysis::data_quality::IdentityProfile {
2616                duplicate_groups: 0,
2617                extra_rows: 0,
2618                rows_involved: 0,
2619                evaluated_rows: results.evaluated_rows,
2620                examples: Vec::new(),
2621            });
2622            results
2623        };
2624
2625        // A clean sampled run of one file, its sample streamed from 1,000 rows.
2626        let mut sampled = measured(results_with(columns.clone(), Vec::new()));
2627        sampled.precision = QualityPrecision::Sampled;
2628        sampled.total_rows = Some(1_000);
2629        sampled.reads = Some(ObservedReads {
2630            reads: 1,
2631            counted: 1,
2632            rows: 1_000,
2633            copy: None,
2634        });
2635        let report = build_report(&sampled);
2636        assert_eq!(report.problems + report.notes, 0, "a clean report");
2637        let found = coverage(&sampled, &checks(&sampled, &report), &plan);
2638        assert_eq!(
2639            found.checks(),
2640            ["7 sampled", "2 skipped", "1 unavailable"],
2641            "{found:?}"
2642        );
2643        assert_eq!(
2644            found.rows,
2645            ["100 of 1,000 sampled (10.0%)", "1,000 traversed"]
2646        );
2647        assert_eq!(found.limits(), ["Nearly unique: needs every row checked"]);
2648
2649        // Thin segments, footers read for only some files, roles that pair nothing.
2650        let segment = |label: &str, rows: usize| SegmentQualityProfile {
2651            label: label.to_string(),
2652            total_rows: Some(500),
2653            evaluated_rows: rows,
2654            columns: Vec::new(),
2655            null_cells: 0,
2656            null_rate: 0.0,
2657            compared_with: None,
2658            largest_change: None,
2659            change_size: None,
2660        };
2661        sampled.segments = vec![segment("a", 90), segment("b", 10), segment("c", 0)];
2662        sampled.source_files = Some(400);
2663        sampled.footers_read = Some(100);
2664        let roles = DataQualityPlan {
2665            temporal_roles: vec![TemporalRoleAssignment {
2666                role: TemporalRole::Event,
2667                column: "id".to_string(),
2668                timezone: None,
2669            }],
2670            ..plan.clone()
2671        };
2672        let report = build_report(&sampled);
2673        let list = checks(&sampled, &report);
2674        let type_mismatch = list.iter().find(|c| c.name == "Type mismatch").unwrap();
2675        assert_eq!(type_mismatch.applies_to, "100 of 400 files");
2676        assert_eq!(type_mismatch.basis, QualityPrecision::Metadata);
2677        let found = coverage(&sampled, &list, &roles);
2678        assert_eq!(found.checks(), ["7 sampled", "2 metadata", "1 unavailable"]);
2679        assert_eq!(
2680            found.limits(),
2681            [
2682                "Nearly unique: needs every row checked",
2683                "2 of 3 segments under 30 sampled rows",
2684                "footers of 100 of 400 files read",
2685                "time roles form no interval",
2686            ]
2687        );
2688
2689        // Every row read: nothing unavailable, and the passes' rows beside the total.
2690        let mut full = measured(results_with(columns.clone(), Vec::new()));
2691        full.reads = Some(ObservedReads {
2692            reads: 4,
2693            counted: 3,
2694            rows: 300,
2695            copy: None,
2696        });
2697        let report = build_report(&full);
2698        let found = coverage(&full, &checks(&full, &report), &plan);
2699        assert_eq!(found.checks(), ["8 exact", "2 skipped"]);
2700        assert_eq!(
2701            found.rows,
2702            ["all 100 read, exact", "at least 300 traversed"]
2703        );
2704        assert!(found.limits().is_empty());
2705
2706        // No values read: the footers are all that was checked.
2707        let mut metadata = measured(results_with(columns, Vec::new()));
2708        metadata.precision = QualityPrecision::Metadata;
2709        metadata.source_files = Some(3);
2710        metadata.footers_read = Some(3);
2711        metadata.reads = Some(ObservedReads::default());
2712        let report = build_report(&metadata);
2713        let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2714        assert_eq!(found.checks(), ["2 metadata", "8 unavailable"]);
2715        assert_eq!(found.rows, ["none read, file metadata only"]);
2716        assert_eq!(found.limits(), ["8 checks: values not read"]);
2717
2718        // A check with no column of its kind is skipped whatever was read: the schema
2719        // says so without the values.
2720        let floats_only = vec![profile("price", DataType::Float64)];
2721        let mut metadata = measured(results_with(floats_only.clone(), Vec::new()));
2722        metadata.precision = QualityPrecision::Metadata;
2723        let report = build_report(&metadata);
2724        let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2725        assert_eq!(found.checks(), ["6 skipped", "4 unavailable"], "{found:?}");
2726        let mut sampled = measured(results_with(floats_only, Vec::new()));
2727        sampled.precision = QualityPrecision::Sampled;
2728        let report = build_report(&sampled);
2729        let list = checks(&sampled, &report);
2730        let nearly = list.iter().find(|c| c.name == "Nearly unique").unwrap();
2731        assert_eq!(
2732            nearly.outcome,
2733            Outcome::Skipped("no integer or text columns")
2734        );
2735    }
2736
2737    #[test]
2738    fn column_labels_fit_and_count_the_rest() {
2739        let columns = ["open", "high", "low", "close"].map(String::from);
2740        assert_eq!(columns_label(&columns, 40), "open, high, low, close");
2741        assert_eq!(columns_label(&columns, 14), "open, high +2");
2742        assert_eq!(columns_label(&columns[..1], 2), "open");
2743    }
2744
2745    /// Narrowing and ordering read the report as it is: Problems stay above Notes in
2746    /// every order, a column or a check keeps only the findings that name it, and the
2747    /// rows a finding counts or the share they are decide the order within a
2748    /// severity.
2749    #[test]
2750    fn findings_narrow_and_order_without_measuring() {
2751        let mut results = results_with(
2752            vec![
2753                profile("price", DataType::Float64),
2754                profile("region", DataType::String),
2755                profile("note", DataType::String),
2756                profile("id", DataType::Int64),
2757            ],
2758            vec![
2759                observation(ObservationKind::NonFinite, "price", 2),
2760                observation(ObservationKind::Nulls, "region", 3),
2761                observation(ObservationKind::Nulls, "note", 30),
2762                observation(ObservationKind::Whitespace, "region", 9),
2763            ],
2764        );
2765        // A share over a smaller denominator: fewer rows, higher rate.
2766        results.observations[0].evaluated_rows = 4;
2767        let report = build_report(&results);
2768        let titles = |view: &FindingsView| {
2769            view.shown(&report)
2770                .into_iter()
2771                .map(|index| reads(&report.findings[index]))
2772                .collect::<Vec<_>>()
2773        };
2774        let ranked = FindingsView::default();
2775        assert_eq!(
2776            titles(&ranked),
2777            [
2778                (Some(ObservationKind::NonFinite), None),
2779                (Some(ObservationKind::Whitespace), None),
2780                (Some(ObservationKind::Nulls), None),
2781                (None, None)
2782            ]
2783        );
2784        let rows = FindingsView {
2785            order: FindingOrder::Rows,
2786            ..FindingsView::default()
2787        };
2788        assert_eq!(
2789            titles(&rows),
2790            [
2791                (Some(ObservationKind::Whitespace), None),
2792                (Some(ObservationKind::NonFinite), None),
2793                (Some(ObservationKind::Nulls), None),
2794                (None, None)
2795            ],
2796            "most rows first, Problems still above Notes"
2797        );
2798        let rate = FindingsView {
2799            order: FindingOrder::Rate,
2800            ..FindingsView::default()
2801        };
2802        assert_eq!(
2803            titles(&rate)[0],
2804            (Some(ObservationKind::NonFinite), None),
2805            "2 of 4 beats 9 of 100"
2806        );
2807
2808        let region = FindingsView {
2809            column: Some("region".to_string()),
2810            ..FindingsView::default()
2811        };
2812        assert_eq!(
2813            titles(&region),
2814            [
2815                (Some(ObservationKind::Whitespace), None),
2816                (Some(ObservationKind::Nulls), None)
2817            ]
2818        );
2819        assert!(region.narrowed());
2820        let clean = FindingsView {
2821            column: Some("id".to_string()),
2822            ..FindingsView::default()
2823        };
2824        assert_eq!(titles(&clean), [(None, None)], "a clean column is clean");
2825        let missing = FindingsView {
2826            check: Some("Missing values"),
2827            ..FindingsView::default()
2828        };
2829        assert_eq!(titles(&missing), [(Some(ObservationKind::Nulls), None)]);
2830        assert_eq!(
2831            missing.selected(&report, 0).map(reads),
2832            Some((Some(ObservationKind::Nulls), None))
2833        );
2834        assert_eq!(
2835            check_choices(&report),
2836            [
2837                ("NaN or infinite", 1),
2838                ("Blank text", 1),
2839                ("Missing values", 1)
2840            ]
2841        );
2842        assert_eq!(
2843            column_choices(&report, &results)
2844                .into_iter()
2845                .map(|(_, count)| count)
2846                .collect::<Vec<_>>(),
2847            [1, 2, 1, 0]
2848        );
2849    }
2850
2851    /// A finding over several columns lists each column's own count, and says the
2852    /// rows with any of them are a range nobody counted, not their sum.
2853    #[test]
2854    fn grouped_findings_break_down_by_column_and_bound_the_union() {
2855        let results = results_with(
2856            vec![
2857                profile("a", DataType::String),
2858                profile("b", DataType::String),
2859            ],
2860            vec![
2861                observation(ObservationKind::Empty, "a", 6),
2862                observation(ObservationKind::Empty, "b", 6),
2863            ],
2864        );
2865        let report = build_report(&results);
2866        let finding = &report.findings[0];
2867        assert!(finding.lists_columns());
2868        assert_eq!(
2869            finding.evidence_count(&results),
2870            None,
2871            "a union nobody counted"
2872        );
2873        let (headline, evidence) = describe(finding, &results);
2874        assert_eq!(
2875            headline,
2876            "6 of 100 rows (6.0%) empty strings in each column:"
2877        );
2878        assert_eq!(evidence[0], "a     6.0%  6 rows");
2879        assert_eq!(
2880            evidence.last().unwrap(),
2881            "Rows with any of them: 6 to 12, not counted"
2882        );
2883    }
2884
2885    /// A parseable-text finding's rows are the values that stop a cast, counted
2886    /// from the profile; with none, there is nothing to open, and it says why.
2887    #[test]
2888    fn parse_failures_are_the_evidence_of_text_that_parses() {
2889        let mut text = profile("amount", DataType::String);
2890        text.null_count = 4;
2891        text.integer_parse_count = Some(95);
2892        text.decimal_parse_count = Some(95);
2893        let mut results = results_with(
2894            vec![text],
2895            vec![observation(ObservationKind::ParseableText, "amount", 95)],
2896        );
2897        results.examples = vec![crate::analysis::data_quality::FindingExamples {
2898            kind: ObservationKind::ParseableText,
2899            column: "amount".to_string(),
2900            values: vec!["\"n/a\"".to_string()],
2901        }];
2902        let report = build_report(&results);
2903        let finding = &report.findings[0];
2904        assert_eq!(finding.check(), Some("Numbers as text"));
2905        assert_eq!(finding.failures(&results), Some(1));
2906        assert_eq!(finding.evidence_count(&results), Some(1));
2907        assert!(matches!(
2908            finding.evidence(&results),
2909            Ok(EvidenceRows::Matching(_))
2910        ));
2911        let (_, evidence) = describe(finding, &results);
2912        assert!(
2913            evidence.contains(&"1 do not parse, such as \"n/a\"".to_string()),
2914            "{evidence:?}"
2915        );
2916
2917        results.columns[0].integer_parse_count = Some(96);
2918        results.columns[0].decimal_parse_count = Some(96);
2919        results.observations[0].affected_rows = 96;
2920        let report = build_report(&results);
2921        let reason = report.findings[0].evidence(&results).unwrap_err();
2922        assert!(reason.contains("every value parses"), "{reason}");
2923    }
2924
2925    /// Duplicate rows open as a group, and the count is every row with a copy.
2926    #[test]
2927    fn duplicate_rows_open_every_row_with_a_copy() {
2928        let mut results = results_with(
2929            vec![profile("id", DataType::Int64)],
2930            vec![observation(
2931                ObservationKind::DuplicateRows,
2932                "all columns",
2933                5,
2934            )],
2935        );
2936        results.identity = Some(crate::analysis::data_quality::IdentityProfile {
2937            duplicate_groups: 2,
2938            extra_rows: 3,
2939            rows_involved: 5,
2940            evaluated_rows: 100,
2941            examples: vec![crate::analysis::data_quality::DuplicateExample {
2942                copies: 3,
2943                values: vec!["7".to_string()],
2944            }],
2945        });
2946        let report = build_report(&results);
2947        let finding = &report.findings[0];
2948        assert!(matches!(
2949            finding.evidence(&results),
2950            Ok(EvidenceRows::Duplicates)
2951        ));
2952        assert_eq!(finding.evidence_count(&results), Some(5));
2953        let (headline, evidence) = describe(finding, &results);
2954        assert_eq!(headline, "2 rows repeated; 3 extra copies");
2955        assert_eq!(evidence[0], "5 of 100 rows (5.0%) have a copy");
2956        assert_eq!(evidence[2], format!("{}3  7", crate::glyphs::get().times));
2957    }
2958}