Skip to main content

datui_lib/
quality_report.rs

1//! Data Quality results read as a report: what is likely wrong, what depends on
2//! intent, and which columns have nothing to report.
3//!
4//! Derived from [`DataQualityResults`] whenever it is drawn, so the engine, the
5//! session cache and the evidence drill-in keep working on observations; a finding
6//! only groups and ranks them.
7
8use crate::data_quality::{
9    ColumnQualityProfile, DataQualityPlan, DataQualityResults, ObservationKind, QualityPrecision,
10    QualityScope, TextReading, text_reading,
11};
12use crate::numfmt;
13use polars::prelude::Expr;
14use std::collections::BTreeMap;
15
16#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Hash)]
17pub enum Severity {
18    /// Likely a defect, and one that changes results without saying so.
19    Problem,
20    /// Fine or not depending on what the column is for.
21    Note,
22    /// Nothing to report.
23    Clean,
24}
25
26impl Severity {
27    pub fn heading(self) -> &'static str {
28        match self {
29            Self::Problem => "Problems",
30            Self::Note => "Notes",
31            Self::Clean => "Clean",
32        }
33    }
34}
35
36#[derive(Debug, Clone)]
37pub struct Finding {
38    pub severity: Severity,
39    /// `None` for the clean-columns entry.
40    pub kind: Option<ObservationKind>,
41    pub title: &'static str,
42    pub columns: Vec<String>,
43    /// Indices into [`DataQualityResults::observations`].
44    pub observations: Vec<usize>,
45    /// Rows behind the finding; per column when several columns share the count.
46    pub affected_rows: usize,
47    pub evaluated_rows: usize,
48    /// One short line for the list: counts and the telling detail.
49    pub summary: String,
50    /// Grouped columns that are missing on exactly the same rows.
51    pub same_rows: bool,
52}
53
54impl Finding {
55    /// "open, high, low +7", fitted to `width` display columns.
56    pub fn columns_label(&self, width: usize) -> String {
57        columns_label(&self.columns, width)
58    }
59
60    /// Rows matching any of the grouped observations. Grouped columns missing on the
61    /// same rows match the same rows either way; otherwise "any" is what the list
62    /// promised: every row behind the finding.
63    pub fn evidence_predicate(&self, results: &DataQualityResults) -> Option<Expr> {
64        self.observations
65            .iter()
66            .map(|index| {
67                let observation = results.observations.get(*index)?;
68                match observation.kind {
69                    // The values that stop a cast: text the reading does not parse.
70                    ObservationKind::ParseableText => {
71                        let profile = results
72                            .columns
73                            .iter()
74                            .find(|profile| profile.name == observation.column)?;
75                        crate::data_quality::unparsed_text(profile)
76                    }
77                    _ => observation.evidence_predicate().or_else(|| {
78                        results
79                            .intent
80                            .as_ref()?
81                            .evidence(observation.kind, &observation.column)
82                    }),
83                }
84            })
85            .collect::<Option<Vec<_>>>()?
86            .into_iter()
87            .reduce(Expr::or)
88    }
89
90    /// The files behind an absent column or a type conflict; those stand alone.
91    pub fn evidence_scope(&self, results: &DataQualityResults) -> Option<QualityScope> {
92        match self.observations.as_slice() {
93            [index] => results.observations.get(*index)?.evidence_scope(),
94            _ => None,
95        }
96    }
97
98    /// Which rows are the finding's evidence, or why there are none to show.
99    pub fn evidence(&self, results: &DataQualityResults) -> Result<EvidenceRows, String> {
100        if let Some(scope) = self.evidence_scope(results) {
101            return Ok(EvidenceRows::Files(scope));
102        }
103        match self.kind {
104            None => Err("No rows: every column here passed".to_string()),
105            _ if results.precision == QualityPrecision::Metadata => {
106                Err("No rows: values were not read".to_string())
107            }
108            Some(ObservationKind::DuplicateRows) => Ok(EvidenceRows::Duplicates),
109            Some(ObservationKind::ParseableText) if self.failures(results) == Some(0) => {
110                Err("No rows: every value parses".to_string())
111            }
112            _ => self
113                .evidence_predicate(results)
114                .map(EvidenceRows::Matching)
115                .ok_or_else(|| "No rows: nothing to filter on".to_string()),
116        }
117    }
118
119    /// How many rows Enter shows, where one count is all of them: one observation,
120    /// the same rows in every column, or spellings of one column, which never
121    /// overlap. `None` for a union nobody counted, and for the files' rows.
122    pub fn evidence_count(&self, results: &DataQualityResults) -> Option<usize> {
123        match self.kind? {
124            ObservationKind::DuplicateRows => {
125                Some(results.identity.as_ref()?.rows_involved).filter(|rows| *rows > 0)
126            }
127            ObservationKind::ParseableText => self.failures(results),
128            // Rows beyond one per value are the count; the rows sharing a value are
129            // what opens, and always more.
130            ObservationKind::KeyLike
131            | ObservationKind::Absent
132            | ObservationKind::TypeConflict
133            | ObservationKind::DcOffset => None,
134            ObservationKind::CategoryVariants => Some(self.affected_rows),
135            // One fact about rows, named once per key column: the same rows each time.
136            ObservationKind::KeyRepeated | ObservationKind::KeyMissing => Some(self.affected_rows),
137            _ if self.observations.len() == 1 || self.same_rows => Some(self.affected_rows),
138            _ => None,
139        }
140    }
141
142    /// Non-null text in a parseable-text column that its reading does not parse.
143    pub fn failures(&self, results: &DataQualityResults) -> Option<usize> {
144        if self.kind != Some(ObservationKind::ParseableText) {
145            return None;
146        }
147        let profile = results
148            .columns
149            .iter()
150            .find(|profile| Some(&profile.name) == self.columns.first())?;
151        let (parsed, _) = text_reading(profile)?;
152        Some(profile.non_null_rows().saturating_sub(parsed))
153    }
154
155    /// The check that makes this finding, by the name the Checks list gives it.
156    /// `None` for the clean entry.
157    pub fn check(&self) -> Option<&'static str> {
158        Some(match self.kind? {
159            ObservationKind::Nulls => "Missing values",
160            ObservationKind::NonFinite => "NaN or infinite",
161            ObservationKind::DuplicateRows => "Duplicate rows",
162            ObservationKind::Empty | ObservationKind::Whitespace => "Blank text",
163            ObservationKind::CategoryVariants => "Mixed spellings",
164            ObservationKind::TypeConflict => "Type mismatch",
165            ObservationKind::Absent => "Missing in files",
166            ObservationKind::ParseableText => "Numbers as text",
167            ObservationKind::KeyLike => "Nearly unique",
168            ObservationKind::Constant => "Single value",
169            ObservationKind::UnparsedTime => "Unparsed times",
170            ObservationKind::Clipping => "Clipping",
171            ObservationKind::ZeroRuns => "Runs of zeros",
172            ObservationKind::DcOffset => "DC offset",
173            ObservationKind::KeyRepeated
174            | ObservationKind::KeyMissing
175            | ObservationKind::RequiredMissing
176            | ObservationKind::NotAllowed
177            | ObservationKind::OutOfRange
178            | ObservationKind::UnparsedNumber => INTENT_CHECK,
179        })
180    }
181
182    /// Missing values grouped across columns that go missing at different rates.
183    pub fn varied(&self) -> bool {
184        self.kind == Some(ObservationKind::Nulls)
185            && self.columns.len() > 1
186            && !self.same_rows
187            && self.severity == Severity::Note
188    }
189
190    /// A count per column that the detail lists one column a line: several columns
191    /// grouped by what they miss, not on the same rows.
192    pub fn lists_columns(&self) -> bool {
193        matches!(
194            self.kind,
195            Some(ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace)
196        ) && self.columns.len() > 1
197            && !self.same_rows
198    }
199
200    /// Each column's count and rate, one a line, worst first; then the rows with
201    /// any of them, which no check counted: at least the largest column's count and
202    /// at most their sum.
203    pub fn breakdown(&self, results: &DataQualityResults) -> Vec<String> {
204        let rows = self
205            .observations
206            .iter()
207            .filter_map(|index| results.observations.get(*index))
208            .collect::<Vec<_>>();
209        let name_width = rows
210            .iter()
211            .map(|row| crate::glyphs::display_width(&row.column))
212            .max()
213            .unwrap_or(0)
214            .min(28);
215        let plural = |count: usize| if count == 1 { "row" } else { "rows" };
216        let mut lines = rows
217            .iter()
218            .map(|row| {
219                let name = columns_label(std::slice::from_ref(&row.column), name_width);
220                let pad = name_width.saturating_sub(crate::glyphs::display_width(&name));
221                format!(
222                    "{name}{}  {:>7}  {} {}",
223                    " ".repeat(pad),
224                    percent(row.affected_rows, row.evaluated_rows),
225                    numfmt::group_chrome(row.affected_rows),
226                    plural(row.affected_rows)
227                )
228            })
229            .collect::<Vec<_>>();
230        let least = rows.iter().map(|row| row.affected_rows).max().unwrap_or(0);
231        let most = rows
232            .iter()
233            .map(|row| row.affected_rows)
234            .sum::<usize>()
235            .min(self.evaluated_rows.max(least));
236        lines.push(if least == most {
237            format!(
238                "Rows with any of them: {} {}",
239                numfmt::group_chrome(least),
240                plural(least)
241            )
242        } else {
243            format!(
244                "Rows with any of them: {} to {}, not counted",
245                numfmt::group_chrome(least),
246                numfmt::group_chrome(most)
247            )
248        });
249        lines
250    }
251}
252
253/// The rows behind a finding.
254#[derive(Debug, Clone)]
255pub enum EvidenceRows {
256    /// Rows whose values match.
257    Matching(Expr),
258    /// Every row the named files hold.
259    Files(QualityScope),
260    /// Rows equal to another in every column, copies together; see
261    /// [`crate::data_quality::duplicate_rows`].
262    Duplicates,
263}
264
265/// How the findings list is narrowed and ordered. Presentation only: it reads the
266/// report on screen, and nothing is measured again.
267#[derive(Debug, Clone, Default, PartialEq, Eq)]
268pub struct FindingsView {
269    /// Only findings that name this column.
270    pub column: Option<String>,
271    /// Only findings this check made, by [`Finding::check`].
272    pub check: Option<&'static str>,
273    pub order: FindingOrder,
274}
275
276/// The order findings are listed in within each severity.
277#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
278pub enum FindingOrder {
279    /// What costs the most trust first; see [`build_report`].
280    #[default]
281    Ranked,
282    /// Most rows affected first.
283    Rows,
284    /// Highest share of the rows checked first.
285    Rate,
286}
287
288impl FindingOrder {
289    pub fn next(self) -> Self {
290        match self {
291            Self::Ranked => Self::Rows,
292            Self::Rows => Self::Rate,
293            Self::Rate => Self::Ranked,
294        }
295    }
296
297    /// The chip that switches to this order.
298    pub fn chip(self) -> &'static str {
299        match self {
300            Self::Ranked => "Ranked",
301            Self::Rows => "By Rows",
302            Self::Rate => "By Rate",
303        }
304    }
305}
306
307impl FindingsView {
308    pub fn narrowed(&self) -> bool {
309        self.column.is_some() || self.check.is_some()
310    }
311
312    fn admits(&self, finding: &Finding) -> bool {
313        self.column
314            .as_ref()
315            .is_none_or(|column| finding.columns.contains(column))
316            && self
317                .check
318                .is_none_or(|check| finding.check() == Some(check))
319    }
320
321    /// Indices into `report.findings`, in the order listed. Severity leads in every
322    /// order, so Problems stay above Notes; within one, a tie keeps the rank.
323    pub fn shown(&self, report: &QualityReport) -> Vec<usize> {
324        let findings = &report.findings;
325        let mut shown = (0..findings.len())
326            .filter(|index| self.admits(&findings[*index]))
327            .collect::<Vec<_>>();
328        let by = |left: &usize, right: &usize| {
329            let (left, right) = (&findings[*left], &findings[*right]);
330            left.severity
331                .cmp(&right.severity)
332                .then_with(|| match self.order {
333                    FindingOrder::Ranked => std::cmp::Ordering::Equal,
334                    FindingOrder::Rows => right.affected_rows.cmp(&left.affected_rows),
335                    // Shares compared exactly: a/b against c/d as a·d against c·b.
336                    FindingOrder::Rate => (right.affected_rows as u128
337                        * left.evaluated_rows.max(1) as u128)
338                        .cmp(&(left.affected_rows as u128 * right.evaluated_rows.max(1) as u128)),
339                })
340        };
341        shown.sort_by(by);
342        shown
343    }
344
345    /// The finding at `position` in the list as shown.
346    pub fn selected<'a>(&self, report: &'a QualityReport, position: usize) -> Option<&'a Finding> {
347        self.shown(report)
348            .get(position)
349            .and_then(|index| report.findings.get(*index))
350    }
351
352    /// What narrows and orders the list, for the line above it: "price · Missing
353    /// values · by rows". Empty when the list is the report as ranked.
354    pub fn describe(&self) -> Vec<String> {
355        let mut parts = Vec::new();
356        if let Some(column) = &self.column {
357            parts.push(format!("column {column}"));
358        }
359        if let Some(check) = self.check {
360            parts.push(check.to_string());
361        }
362        match self.order {
363            FindingOrder::Ranked => {}
364            FindingOrder::Rows => parts.push("by rows".to_string()),
365            FindingOrder::Rate => parts.push("by rate".to_string()),
366        }
367        parts
368    }
369}
370
371/// The columns a report can be narrowed to, each with how many findings name it,
372/// in the results' column order.
373pub fn column_choices(
374    report: &QualityReport,
375    results: &DataQualityResults,
376) -> Vec<(String, usize)> {
377    results
378        .columns
379        .iter()
380        .map(|profile| {
381            let count = report
382                .findings
383                .iter()
384                .filter(|finding| finding.kind.is_some() && finding.columns.contains(&profile.name))
385                .count();
386            (profile.name.clone(), count)
387        })
388        .collect()
389}
390
391/// The checks that made a finding in `report`, each with how many, most important
392/// first.
393pub fn check_choices(report: &QualityReport) -> Vec<(&'static str, usize)> {
394    let mut choices: Vec<(&'static str, usize)> = Vec::new();
395    for check in report.findings.iter().filter_map(Finding::check) {
396        match choices.iter_mut().find(|(name, _)| *name == check) {
397            Some((_, count)) => *count += 1,
398            None => choices.push((check, 1)),
399        }
400    }
401    choices
402}
403
404#[derive(Debug, Clone, Default)]
405pub struct QualityReport {
406    pub findings: Vec<Finding>,
407    pub problems: usize,
408    pub notes: usize,
409    pub clean_columns: usize,
410    pub total_columns: usize,
411    /// Worst severity per column, in the results' column order.
412    pub column_status: Vec<Severity>,
413    /// Finding titles per column, in the results' column order.
414    pub column_findings: Vec<Vec<&'static str>>,
415    /// No values were read, so the absence of a finding means nothing.
416    pub metadata_only: bool,
417    /// Values were to be read and the scope had none: no check had anything to look
418    /// at, so nothing is clean either.
419    pub no_rows: bool,
420}
421
422pub fn build_report(results: &DataQualityResults) -> QualityReport {
423    let mut findings = Vec::new();
424    // Group observations that say the same thing so the list says it once.
425    let mut groups = BTreeMap::<(u8, usize, String), Vec<usize>>::new();
426    for (index, observation) in results.observations.iter().enumerate() {
427        let always_missing = observation.kind == ObservationKind::Nulls
428            && observation.evaluated_rows > 0
429            && observation.affected_rows == observation.evaluated_rows;
430        let mostly_missing = observation.affected_rows * 2 > observation.evaluated_rows;
431        // Columns missing on the very same rows are one fact about those rows; any
432        // other missing values are one finding, with each column's rate inside it.
433        let same_rows = results.shared_nulls.iter().any(|shared| {
434            shared.same_rows()
435                && shared.columns.len() > 1
436                && shared.null_rows == observation.affected_rows
437                && shared.columns.contains(&observation.column)
438        });
439        let kind = observation.kind as u8 + 1;
440        let key = match observation.kind {
441            ObservationKind::Nulls if always_missing => (0, 0, String::new()),
442            ObservationKind::Nulls if mostly_missing => (kind, usize::MAX, "mostly".to_string()),
443            ObservationKind::Nulls if same_rows => (kind, observation.affected_rows, String::new()),
444            ObservationKind::Nulls => (kind, usize::MAX, String::new()),
445            ObservationKind::Empty | ObservationKind::Whitespace => {
446                (kind, observation.affected_rows, String::new())
447            }
448            ObservationKind::Constant => (kind, 0, String::new()),
449            ObservationKind::CategoryVariants => (kind, 0, observation.column.clone()),
450            // The key is one fact about rows, named once per key column.
451            ObservationKind::KeyRepeated | ObservationKind::KeyMissing => (kind, 0, String::new()),
452            // Everything else stands alone; the index keeps it from merging.
453            _ => (u8::MAX, index, String::new()),
454        };
455        groups.entry(key).or_default().push(index);
456    }
457    for indices in groups.into_values() {
458        findings.push(finding(results, &indices));
459    }
460
461    let mut status = vec![Severity::Clean; results.columns.len()];
462    let mut titles = vec![Vec::new(); results.columns.len()];
463    let position = results
464        .columns
465        .iter()
466        .enumerate()
467        .map(|(index, profile)| (profile.name.as_str(), index))
468        .collect::<BTreeMap<_, _>>();
469    for finding in &findings {
470        for column in &finding.columns {
471            if let Some(&index) = position.get(column.as_str()) {
472                status[index] = status[index].min(finding.severity);
473                if !titles[index].contains(&finding.title) {
474                    titles[index].push(finding.title);
475                }
476            }
477        }
478    }
479
480    findings.sort_by(|left, right| {
481        left.severity
482            .cmp(&right.severity)
483            .then_with(|| rank(left).cmp(&rank(right)))
484            .then_with(|| right.affected_rows.cmp(&left.affected_rows))
485            .then_with(|| left.columns.cmp(&right.columns))
486    });
487    let problems = findings
488        .iter()
489        .filter(|finding| finding.severity == Severity::Problem)
490        .count();
491    let notes = findings.len() - problems;
492    let metadata_only = results.precision == QualityPrecision::Metadata;
493    let no_rows = !metadata_only && results.evaluated_rows == 0;
494    let clean = results
495        .columns
496        .iter()
497        .zip(&status)
498        .filter(|(_, severity)| **severity == Severity::Clean)
499        .map(|(profile, _)| profile.name.clone())
500        .collect::<Vec<_>>();
501    let clean_columns = if no_rows { 0 } else { clean.len() };
502    if !clean.is_empty() && !metadata_only && !no_rows {
503        findings.push(Finding {
504            severity: Severity::Clean,
505            kind: None,
506            title: "No findings",
507            summary: format!(
508                "{} {}",
509                numfmt::group_chrome(clean.len()),
510                if clean.len() == 1 {
511                    "column"
512                } else {
513                    "columns"
514                }
515            ),
516            columns: clean,
517            observations: Vec::new(),
518            affected_rows: 0,
519            evaluated_rows: results.evaluated_rows,
520            same_rows: false,
521        });
522    }
523    QualityReport {
524        findings,
525        problems,
526        notes,
527        clean_columns,
528        total_columns: results.columns.len(),
529        column_status: status,
530        column_findings: titles,
531        metadata_only,
532        no_rows,
533    }
534}
535
536/// Order within a severity: what costs the most rows of trust first.
537fn rank(finding: &Finding) -> u8 {
538    match finding.kind {
539        Some(ObservationKind::TypeConflict) => 0,
540        Some(ObservationKind::Absent) => 1,
541        // What the study declared comes first: a violation is a fact, not a guess.
542        Some(ObservationKind::KeyRepeated) => 2,
543        Some(ObservationKind::KeyMissing | ObservationKind::RequiredMissing) => 2,
544        Some(ObservationKind::NotAllowed | ObservationKind::OutOfRange) => 2,
545        Some(ObservationKind::UnparsedNumber) => 2,
546        Some(ObservationKind::DuplicateRows) => 2,
547        Some(ObservationKind::UnparsedTime) => 3,
548        Some(ObservationKind::Nulls) if finding.severity == Severity::Problem => 3,
549        Some(ObservationKind::Nulls) if finding.title == "Mostly missing" => 9,
550        Some(ObservationKind::NonFinite) => 4,
551        Some(ObservationKind::Clipping) => 4,
552        Some(ObservationKind::ZeroRuns) => 8,
553        Some(ObservationKind::DcOffset) => 12,
554        Some(ObservationKind::CategoryVariants) => 5,
555        Some(ObservationKind::Whitespace) => 6,
556        Some(ObservationKind::Empty) => 7,
557        Some(ObservationKind::Nulls) => 10,
558        Some(ObservationKind::KeyLike) => 11,
559        Some(ObservationKind::ParseableText) => 12,
560        Some(ObservationKind::Constant) => 13,
561        None => 20,
562    }
563}
564
565fn finding(results: &DataQualityResults, indices: &[usize]) -> Finding {
566    let mut indices = indices.to_vec();
567    // Missing values list their columns, and mixed spellings their values, worst
568    // first.
569    if matches!(
570        results.observations[indices[0]].kind,
571        ObservationKind::Nulls | ObservationKind::CategoryVariants
572    ) {
573        indices.sort_by_key(|index| std::cmp::Reverse(results.observations[*index].affected_rows));
574    }
575    let indices = indices.as_slice();
576    let first = &results.observations[indices[0]];
577    let kind = first.kind;
578    let mut columns = Vec::new();
579    for index in indices {
580        let column = &results.observations[*index].column;
581        if !columns.contains(column) {
582            columns.push(column.clone());
583        }
584    }
585    let grouped = columns.len() > 1;
586    let always_missing = kind == ObservationKind::Nulls
587        && first.evaluated_rows > 0
588        && first.affected_rows == first.evaluated_rows;
589    let mostly_missing = kind == ObservationKind::Nulls
590        && !always_missing
591        && first.affected_rows * 2 > first.evaluated_rows;
592    let severity = match kind {
593        ObservationKind::Nulls if always_missing => Severity::Problem,
594        ObservationKind::Nulls
595        | ObservationKind::Constant
596        | ObservationKind::ParseableText
597        | ObservationKind::KeyLike
598        | ObservationKind::ZeroRuns
599        | ObservationKind::DcOffset => Severity::Note,
600        ObservationKind::Empty
601        | ObservationKind::Whitespace
602        | ObservationKind::NonFinite
603        | ObservationKind::DuplicateRows
604        | ObservationKind::CategoryVariants
605        | ObservationKind::Absent
606        | ObservationKind::TypeConflict
607        | ObservationKind::UnparsedTime
608        | ObservationKind::KeyRepeated
609        | ObservationKind::KeyMissing
610        | ObservationKind::RequiredMissing
611        | ObservationKind::NotAllowed
612        | ObservationKind::OutOfRange
613        | ObservationKind::UnparsedNumber
614        | ObservationKind::Clipping => Severity::Problem,
615    };
616    let profile = results
617        .columns
618        .iter()
619        .find(|profile| profile.name == first.column);
620    let reading = profile.and_then(text_reading).map(|(_, reading)| reading);
621    let shared = results.shared_nulls.iter().find(|shared| {
622        shared.null_rows == first.affected_rows
623            && columns.iter().all(|column| shared.columns.contains(column))
624    });
625    let same_rows =
626        kind == ObservationKind::Nulls && grouped && shared.is_some_and(|s| s.same_rows());
627    let title = match kind {
628        ObservationKind::Nulls if always_missing => "Always missing",
629        ObservationKind::Nulls if mostly_missing => "Mostly missing",
630        // No row lacks one of these columns without lacking them all.
631        ObservationKind::Nulls if same_rows => "Missing together",
632        ObservationKind::Nulls => "Missing values",
633        ObservationKind::Empty => "Empty text",
634        ObservationKind::Whitespace => "Blank text",
635        ObservationKind::NonFinite => "NaN or infinite",
636        ObservationKind::Constant => "Single value",
637        ObservationKind::ParseableText => match (reading, profile) {
638            (Some(reading), Some(profile)) if reading.is_number() && is_code(profile) => {
639                "Codes as text"
640            }
641            (Some(TextReading::Datetime | TextReading::Date), _) => "Dates as text",
642            _ => "Numbers as text",
643        },
644        ObservationKind::DuplicateRows => "Duplicate rows",
645        ObservationKind::CategoryVariants => "Mixed spellings",
646        ObservationKind::Absent => "Missing in files",
647        ObservationKind::TypeConflict => "Type mismatch",
648        ObservationKind::KeyLike => "Nearly unique",
649        ObservationKind::UnparsedTime => "Unparsed times",
650        ObservationKind::KeyRepeated => "Repeated key",
651        ObservationKind::KeyMissing => "Incomplete key",
652        ObservationKind::RequiredMissing => "Required, missing",
653        ObservationKind::NotAllowed => "Not allowed",
654        ObservationKind::OutOfRange => "Out of range",
655        ObservationKind::UnparsedNumber => "Unparsed numbers",
656        ObservationKind::Clipping => "Clipping",
657        ObservationKind::ZeroRuns => "Runs of zeros",
658        ObservationKind::DcOffset => "DC offset",
659    };
660    let affected_rows = match kind {
661        // One group per normalized value; the column's cost is all of them.
662        ObservationKind::CategoryVariants => indices
663            .iter()
664            .map(|index| results.observations[*index].affected_rows)
665            .sum(),
666        _ => first.affected_rows,
667    };
668    let rows_each = |count: usize, each: &str| {
669        format!(
670            "{} {}{each} ({})",
671            numfmt::group_chrome(count),
672            if count == 1 { "row" } else { "rows" },
673            percent(count, first.evaluated_rows)
674        )
675    };
676    let rows = |count: usize| rows_each(count, "");
677    let summary = match kind {
678        ObservationKind::Nulls if always_missing => "no value in any row".to_string(),
679        ObservationKind::Nulls if same_rows => rows(first.affected_rows),
680        ObservationKind::Nulls if grouped => {
681            let fewest = results.observations[indices[indices.len() - 1]].affected_rows;
682            let (low, high) = (
683                percent(fewest, first.evaluated_rows),
684                percent(first.affected_rows, first.evaluated_rows),
685            );
686            if low == high {
687                rows_each(first.affected_rows, " each")
688            } else {
689                format!("{low} to {high} per column")
690            }
691        }
692        ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace
693            if grouped =>
694        {
695            rows_each(first.affected_rows, " each")
696        }
697        ObservationKind::Nulls
698        | ObservationKind::Empty
699        | ObservationKind::Whitespace
700        | ObservationKind::NonFinite => rows(first.affected_rows),
701        ObservationKind::UnparsedTime => format!(
702            "{} {} ({})",
703            numfmt::group_chrome(first.affected_rows),
704            if first.affected_rows == 1 {
705                "value"
706            } else {
707                "values"
708            },
709            percent(first.affected_rows, first.evaluated_rows)
710        ),
711        ObservationKind::Constant if grouped => "one value each".to_string(),
712        ObservationKind::Constant => profile
713            .and_then(|profile| profile.dominant_value.as_ref())
714            .map(|value| format!("always {}", quoted(value, 24)))
715            .unwrap_or_else(|| "one value".to_string()),
716        ObservationKind::ParseableText => match (reading, profile) {
717            (Some(_), Some(profile)) if is_code(profile) => code_shape(profile),
718            (Some(reading), Some(profile)) => format!(
719                "{} parse as {}",
720                percent(first.affected_rows, profile.non_null_rows()),
721                reading.label()
722            ),
723            _ => first.fact.clone(),
724        },
725        ObservationKind::DuplicateRows => match results.identity.as_ref() {
726            Some(identity) => format!(
727                "{} extra {}",
728                numfmt::group_chrome(identity.extra_rows),
729                if identity.extra_rows == 1 {
730                    "copy"
731                } else {
732                    "copies"
733                }
734            ),
735            None => first.fact.clone(),
736        },
737        ObservationKind::CategoryVariants => {
738            let example = results
739                .category_variants
740                .iter()
741                .find(|group| Some(&group.normalized) == first.normalized_category.as_ref())
742                .map(|group| {
743                    group
744                        .variants
745                        .iter()
746                        .take(2)
747                        .map(|(value, _)| quoted(value, 24))
748                        .collect::<Vec<_>>()
749                        .join(" vs ")
750                })
751                .unwrap_or_default();
752            // Several values: the count is the finding, and the detail lists them.
753            if indices.len() == 1 {
754                example
755            } else {
756                format!("{} values spelled more than one way", indices.len())
757            }
758        }
759        ObservationKind::KeyLike => {
760            let unique = profile
761                .and_then(ColumnQualityProfile::uniqueness_rate)
762                .map(|rate| format!(" ({:.1}% unique)", rate * 100.0))
763                .unwrap_or_default();
764            format!(
765                "{} repeated{unique}",
766                numfmt::group_chrome(first.affected_rows)
767            )
768        }
769        ObservationKind::Absent | ObservationKind::TypeConflict => first.fact.clone(),
770        // An audio measurement says its runs or its mean itself; across channels, the
771        // worst channel's.
772        ObservationKind::Clipping | ObservationKind::ZeroRuns | ObservationKind::DcOffset => {
773            first.fact.clone()
774        }
775        ObservationKind::KeyRepeated => {
776            match results.intent.as_ref().and_then(|i| i.key.as_ref()) {
777                Some(key) => format!(
778                    "{} {} repeat; {} extra {}",
779                    numfmt::group_chrome(key.groups),
780                    if key.groups == 1 { "value" } else { "values" },
781                    numfmt::group_chrome(key.extra_rows),
782                    if key.extra_rows == 1 { "row" } else { "rows" }
783                ),
784                None => first.fact.clone(),
785            }
786        }
787        ObservationKind::KeyMissing | ObservationKind::RequiredMissing => rows(first.affected_rows),
788        ObservationKind::NotAllowed
789        | ObservationKind::OutOfRange
790        | ObservationKind::UnparsedNumber => format!(
791            "{} {} ({})",
792            numfmt::group_chrome(first.affected_rows),
793            if first.affected_rows == 1 {
794                "value"
795            } else {
796                "values"
797            },
798            percent(first.affected_rows, first.evaluated_rows)
799        ),
800    };
801    Finding {
802        severity,
803        kind: Some(kind),
804        title,
805        columns,
806        observations: indices.to_vec(),
807        affected_rows,
808        evaluated_rows: first.evaluated_rows,
809        summary,
810        same_rows,
811    }
812}
813
814/// Whole numbers written with a leading zero keep it only as text, which makes them
815/// a code rather than a quantity.
816fn is_code(profile: &ColumnQualityProfile) -> bool {
817    profile.leading_zero_count.is_some_and(|count| count > 0)
818        || (profile.min_length.is_some()
819            && profile.min_length == profile.max_length
820            && profile.min_length.is_some_and(|length| length > 1))
821}
822
823fn code_shape(profile: &ColumnQualityProfile) -> String {
824    let zeros = profile.leading_zero_count.unwrap_or(0);
825    match (profile.min_length, profile.max_length) {
826        (Some(min), Some(max)) if min == max && zeros > 0 => {
827            format!("{min} digits, leading zeros")
828        }
829        (Some(min), Some(max)) if min == max => format!("{min} digits each"),
830        _ => format!("{} with a leading zero", numfmt::group_chrome(zeros)),
831    }
832}
833
834pub fn percent(count: usize, of: usize) -> String {
835    if of == 0 {
836        return "-".to_string();
837    }
838    let value = count as f64 / of as f64 * 100.0;
839    // Two places below 1% so a small share never rounds to a misleading 0.0%.
840    if value > 0.0 && value < 1.0 {
841        format!("{value:.2}%")
842    } else {
843        format!("{value:.1}%")
844    }
845}
846
847pub(crate) fn quoted(value: &str, width: usize) -> String {
848    let text = format!("{value:?}");
849    if crate::glyphs::display_width(&text) <= width {
850        text
851    } else {
852        let mut cut = String::new();
853        for ch in text.chars() {
854            if crate::glyphs::display_width(&cut) + 2 > width {
855                break;
856            }
857            cut.push(ch);
858        }
859        format!("{cut}{}", crate::glyphs::get().ellipsis)
860    }
861}
862
863/// Column names joined until `width`, then "+N" for the rest.
864pub fn columns_label(columns: &[String], width: usize) -> String {
865    let mut label = String::new();
866    for (index, column) in columns.iter().enumerate() {
867        let candidate = if label.is_empty() {
868            column.clone()
869        } else {
870            format!("{label}, {column}")
871        };
872        let rest = columns.len() - index - 1;
873        let suffix = if rest > 0 {
874            format!(" +{rest}")
875        } else {
876            String::new()
877        };
878        let needed =
879            crate::glyphs::display_width(&candidate) + crate::glyphs::display_width(&suffix);
880        if !label.is_empty() && needed > width {
881            return format!("{label} +{}", rest + 1);
882        }
883        label = candidate;
884    }
885    label
886}
887
888/// What one check found: nothing, something, nothing because there was nothing for
889/// it to look at, or nothing because this run could not look.
890#[derive(Debug, Clone, PartialEq, Eq)]
891pub enum Outcome {
892    Passed,
893    Found {
894        tier: Severity,
895        detail: String,
896    },
897    /// Nothing in the data for it: no column of its kind, or one file.
898    Skipped(&'static str),
899    /// It applies, and this run could not answer it: no values read, or a sample
900    /// where the answer needs every row.
901    Unavailable(&'static str),
902}
903
904impl Outcome {
905    pub fn ran(&self) -> bool {
906        matches!(self, Self::Passed | Self::Found { .. })
907    }
908}
909
910/// One check the run makes: what it looks for, how far it reached, what it found.
911/// The list is the answer to "checked for what?" when nothing turned up.
912#[derive(Debug, Clone)]
913pub struct Check {
914    pub name: &'static str,
915    pub looks_for: &'static str,
916    pub applies_to: String,
917    pub outcome: Outcome,
918    /// What its numbers were read from: every row, a sample, or file footers.
919    pub basis: QualityPrecision,
920}
921
922/// How many checks the collapsed list shows before "more".
923pub const CHECKS_SHOWN: usize = 6;
924
925/// Why a value check could not run on a scope with no rows.
926const NO_ROWS: &str = "no rows to check";
927
928/// Every check, most important first.
929pub fn checks(results: &DataQualityResults, report: &QualityReport) -> Vec<Check> {
930    use polars::prelude::DataType;
931    let columns = &results.columns;
932    let count = |filter: &dyn Fn(&ColumnQualityProfile) -> bool| {
933        columns.iter().filter(|profile| filter(profile)).count()
934    };
935    let text = |profile: &ColumnQualityProfile| {
936        matches!(profile.dtype, DataType::String | DataType::Categorical(..))
937    };
938    let all = columns.len();
939    let floats = count(&|profile| profile.dtype.is_float());
940    let texts = count(&text);
941    let keys = count(&|profile| profile.dtype.is_integer() || text(profile));
942    let reach = |count: usize, kind: &str| {
943        let noun = if count == 1 { "column" } else { "columns" };
944        if kind.is_empty() {
945            format!("{} {noun}", numfmt::group_chrome(count))
946        } else {
947            format!("{} {kind} {noun}", numfmt::group_chrome(count))
948        }
949    };
950    let values_read = !report.metadata_only;
951    // Columns behind the findings a check produces, by the findings' titles.
952    // Nothing to look at is known from the schema, whatever the run read.
953    let outcome = |titles: &[&str], applies: usize, none: &'static str| {
954        if applies == 0 {
955            return Outcome::Skipped(none);
956        }
957        if !values_read {
958            return Outcome::Unavailable("values not read");
959        }
960        if report.no_rows {
961            return Outcome::Unavailable(NO_ROWS);
962        }
963        found(report, titles)
964    };
965    let files = results.source_files.filter(|files| *files > 1);
966    let by_files = |titles: &[&str]| match files {
967        Some(_) => found(report, titles),
968        None => Outcome::Skipped("needs several files"),
969    };
970    // A dataset too large to read every footer is checked over the footers read.
971    let files_reach = match (files, results.footers_read) {
972        (Some(files), Some(read)) if read < files => format!(
973            "{} of {} files",
974            numfmt::group_chrome(read),
975            numfmt::group_chrome(files)
976        ),
977        (Some(files), _) => format!("{} files", numfmt::group_chrome(files)),
978        (None, _) => "files".to_string(),
979    };
980    let values = results.precision;
981    let mut checks = Vec::new();
982    // Declared, so first: the question the user asked before the ones datui asks.
983    if let Some(intent) = &results.intent {
984        checks.push(Check {
985            name: INTENT_CHECK,
986            looks_for: "values against the key and rules declared",
987            applies_to: reach(intent.declared.len(), "declared"),
988            outcome: if !intent.measured {
989                Outcome::Unavailable("values not read")
990            } else if report.no_rows {
991                Outcome::Unavailable(NO_ROWS)
992            } else {
993                found(report, &INTENT_TITLES)
994            },
995            basis: intent.precision,
996        });
997    }
998    checks.extend([
999        Check {
1000            name: "Missing values",
1001            looks_for: "nulls in any column",
1002            applies_to: reach(all, ""),
1003            outcome: outcome(
1004                &[
1005                    "Missing values",
1006                    "Missing together",
1007                    "Mostly missing",
1008                    "Always missing",
1009                ],
1010                all,
1011                "no columns",
1012            ),
1013            basis: values,
1014        },
1015        Check {
1016            name: "NaN or infinite",
1017            looks_for: "NaN or +/-infinity in float columns",
1018            applies_to: reach(floats, "float"),
1019            outcome: outcome(&["NaN or infinite"], floats, "no float columns"),
1020            basis: values,
1021        },
1022        Check {
1023            name: "Duplicate rows",
1024            looks_for: "rows identical in every column",
1025            applies_to: "whole rows".to_string(),
1026            outcome: match results.identity.as_ref() {
1027                _ if !values_read => Outcome::Unavailable("values not read"),
1028                _ if report.no_rows => Outcome::Unavailable(NO_ROWS),
1029                Some(identity) if identity.extra_rows > 0 => Outcome::Found {
1030                    tier: Severity::Problem,
1031                    detail: format!("{} extra rows", numfmt::group_chrome(identity.extra_rows)),
1032                },
1033                Some(_) => Outcome::Passed,
1034                None => Outcome::Unavailable("not measured"),
1035            },
1036            basis: values,
1037        },
1038        Check {
1039            name: "Blank text",
1040            looks_for: "text that is empty or only whitespace",
1041            applies_to: reach(texts, "text"),
1042            outcome: outcome(&["Blank text", "Empty text"], texts, "no text columns"),
1043            basis: values,
1044        },
1045        Check {
1046            name: "Mixed spellings",
1047            looks_for: "one value in several cases or spacings",
1048            applies_to: reach(texts, "text"),
1049            outcome: outcome(&["Mixed spellings"], texts, "no text columns"),
1050            basis: values,
1051        },
1052        Check {
1053            name: "Type mismatch",
1054            looks_for: "a column typed differently by some files",
1055            applies_to: files_reach.clone(),
1056            outcome: by_files(&["Type mismatch"]),
1057            basis: QualityPrecision::Metadata,
1058        },
1059        Check {
1060            name: "Missing in files",
1061            looks_for: "a column some files do not have",
1062            applies_to: files_reach,
1063            outcome: by_files(&["Missing in files"]),
1064            basis: QualityPrecision::Metadata,
1065        },
1066        Check {
1067            name: "Numbers as text",
1068            looks_for: "text that reads as numbers or dates",
1069            applies_to: reach(texts, "text"),
1070            outcome: outcome(
1071                &["Numbers as text", "Dates as text", "Codes as text"],
1072                texts,
1073                "no text columns",
1074            ),
1075            basis: values,
1076        },
1077        Check {
1078            name: "Nearly unique",
1079            looks_for: "a would-be key whose values repeat",
1080            applies_to: reach(keys, "integer/text"),
1081            outcome: if keys > 0 && values_read && results.precision != QualityPrecision::Exact {
1082                Outcome::Unavailable("needs every row checked")
1083            } else {
1084                match outcome(&["Nearly unique"], keys, "no integer or text columns") {
1085                    // A declared key's repeats replace the note on that column; the
1086                    // check found them all the same.
1087                    Outcome::Passed if declared_key_repeats(results) => Outcome::Found {
1088                        tier: Severity::Problem,
1089                        detail: "the declared key repeats".to_string(),
1090                    },
1091                    outcome => outcome,
1092                }
1093            },
1094            basis: values,
1095        },
1096        Check {
1097            name: "Single value",
1098            looks_for: "a column with one value throughout",
1099            applies_to: reach(all, ""),
1100            outcome: outcome(&["Single value"], all, "no columns"),
1101            basis: values,
1102        },
1103    ]);
1104    checks
1105}
1106
1107/// Whether a one-column declared key repeats: the case whose "Nearly unique" note
1108/// the key's own finding replaces.
1109fn declared_key_repeats(results: &DataQualityResults) -> bool {
1110    results
1111        .intent
1112        .as_ref()
1113        .and_then(|intent| intent.key.as_ref())
1114        .is_some_and(|key| key.columns.len() == 1 && key.rows_involved > 0)
1115}
1116
1117/// The check the declared intent makes, by the name the Checks list gives it.
1118pub const INTENT_CHECK: &str = "Column intent";
1119
1120/// The findings the declared intent makes.
1121pub const INTENT_TITLES: [&str; 6] = [
1122    "Repeated key",
1123    "Incomplete key",
1124    "Required, missing",
1125    "Not allowed",
1126    "Out of range",
1127    "Unparsed numbers",
1128];
1129
1130fn found(report: &QualityReport, titles: &[&str]) -> Outcome {
1131    let mut columns = Vec::new();
1132    let mut tier = Severity::Note;
1133    for finding in report
1134        .findings
1135        .iter()
1136        .filter(|finding| finding.kind.is_some() && titles.contains(&finding.title))
1137    {
1138        tier = tier.min(finding.severity);
1139        for column in &finding.columns {
1140            if !columns.contains(column) {
1141                columns.push(column.clone());
1142            }
1143        }
1144    }
1145    match columns.len() {
1146        0 => Outcome::Passed,
1147        count => Outcome::Found {
1148            tier,
1149            detail: format!(
1150                "{} {}",
1151                numfmt::group_chrome(count),
1152                if count == 1 { "column" } else { "columns" }
1153            ),
1154        },
1155    }
1156}
1157
1158/// A segment with fewer sampled rows than this is thin: only a large change in it
1159/// clears the sampling noise, and a clean one says little.
1160pub const THIN_SEGMENT_ROWS: usize = 30;
1161
1162/// How far the findings reach, beside them on every report: the checks that ran
1163/// and what they read, the ones that did not and why, the rows behind the numbers,
1164/// and what else bounds them. From what the run measured and saw; nothing here
1165/// reads.
1166#[derive(Debug, Clone, Default, PartialEq, Eq)]
1167pub struct Coverage {
1168    /// Checks that ran over every row in scope.
1169    pub exact: usize,
1170    /// Checks that ran over a sample.
1171    pub sampled: usize,
1172    /// Checks that ran over file footers.
1173    pub metadata: usize,
1174    /// Checks with nothing in the data to look at.
1175    pub skipped: usize,
1176    /// Checks that apply and this run could not answer: each reason, and the checks
1177    /// it kept from running.
1178    pub unavailable: Vec<(&'static str, Vec<&'static str>)>,
1179    /// The rows the numbers are over, with their denominator, and what the run's
1180    /// reads were seen to traverse.
1181    pub rows: Vec<String>,
1182    /// What else bounds the findings: thin segments, footers not read, time roles
1183    /// that measure nothing.
1184    pub limits: Vec<String>,
1185}
1186
1187impl Coverage {
1188    /// Checks by what they read, then the ones that did not run: "6 sampled ·
1189    /// 2 metadata · 1 skipped · 1 unavailable".
1190    pub fn checks(&self) -> Vec<String> {
1191        let unavailable = self
1192            .unavailable
1193            .iter()
1194            .map(|(_, names)| names.len())
1195            .sum::<usize>();
1196        [
1197            (self.exact, QualityPrecision::Exact.label()),
1198            (self.sampled, QualityPrecision::Sampled.label()),
1199            (self.metadata, QualityPrecision::Metadata.label()),
1200            (self.skipped, "skipped"),
1201            (unavailable, "unavailable"),
1202        ]
1203        .into_iter()
1204        .filter(|(count, _)| *count > 0)
1205        .map(|(count, label)| format!("{} {label}", numfmt::group_chrome(count)))
1206        .collect()
1207    }
1208
1209    /// Why each unavailable check did not run, then the other limits.
1210    pub fn limits(&self) -> Vec<String> {
1211        self.unavailable
1212            .iter()
1213            .map(|(reason, names)| match names.as_slice() {
1214                [name] => format!("{name}: {reason}"),
1215                _ => format!("{} checks: {reason}", names.len()),
1216            })
1217            .chain(self.limits.iter().cloned())
1218            .collect()
1219    }
1220}
1221
1222pub fn coverage(
1223    results: &DataQualityResults,
1224    checks: &[Check],
1225    plan: &DataQualityPlan,
1226) -> Coverage {
1227    let mut coverage = Coverage::default();
1228    for check in checks {
1229        match &check.outcome {
1230            Outcome::Skipped(_) => coverage.skipped += 1,
1231            Outcome::Unavailable(reason) => {
1232                match coverage
1233                    .unavailable
1234                    .iter_mut()
1235                    .find(|(known, _)| known == reason)
1236                {
1237                    Some((_, names)) => names.push(check.name),
1238                    None => coverage.unavailable.push((reason, vec![check.name])),
1239                }
1240            }
1241            Outcome::Passed | Outcome::Found { .. } => match check.basis {
1242                QualityPrecision::Exact => coverage.exact += 1,
1243                QualityPrecision::Metadata => coverage.metadata += 1,
1244                QualityPrecision::Sampled | QualityPrecision::Estimated => coverage.sampled += 1,
1245            },
1246        }
1247    }
1248
1249    let count = numfmt::group_chrome;
1250    let evaluated = results.evaluated_rows;
1251    coverage
1252        .rows
1253        .push(match (results.precision, results.total_rows) {
1254            (QualityPrecision::Metadata, _) => "none read, file metadata only".to_string(),
1255            _ if evaluated == 0 => "none: the scope has no rows".to_string(),
1256            (QualityPrecision::Exact, _) => format!("all {} read, exact", count(evaluated)),
1257            (_, Some(total)) => format!(
1258                "{} of {} sampled ({})",
1259                count(evaluated),
1260                count(total),
1261                percent(evaluated, total)
1262            ),
1263            (_, None) => format!("{} sampled, total not counted", count(evaluated)),
1264        });
1265    if let Some(each) = results.per_value {
1266        coverage
1267            .rows
1268            .push(format!("up to {} per value", count(each)));
1269    }
1270    // Only what the reads were seen to do: a read that counted nothing says nothing.
1271    if let Some(reads) = results
1272        .reads
1273        .filter(|_| results.precision != QualityPrecision::Metadata)
1274    {
1275        if reads.reads == 0 {
1276            coverage.rows.push("no source read".to_string());
1277        } else if reads.counted == reads.reads {
1278            coverage
1279                .rows
1280                .push(format!("{} traversed", count(reads.rows)));
1281        } else if reads.counted > 0 {
1282            coverage
1283                .rows
1284                .push(format!("at least {} traversed", count(reads.rows)));
1285        }
1286        if let Some(copy) = reads.copy {
1287            let bytes = crate::widgets::info::format_bytes(copy.bytes);
1288            coverage.rows.push(if copy.fetched {
1289                format!("passes read a local copy, fetched once ({bytes})")
1290            } else {
1291                format!("passes read a local copy fetched earlier ({bytes})")
1292            });
1293        }
1294    }
1295
1296    let segments = &results.segments;
1297    if matches!(
1298        results.precision,
1299        QualityPrecision::Sampled | QualityPrecision::Estimated
1300    ) && segments.len() > 1
1301    {
1302        let thin = segments
1303            .iter()
1304            .filter(|segment| segment.evaluated_rows < THIN_SEGMENT_ROWS)
1305            .count();
1306        if thin > 0 {
1307            coverage.limits.push(format!(
1308                "{} of {} segments under {THIN_SEGMENT_ROWS} sampled rows",
1309                count(thin),
1310                count(segments.len())
1311            ));
1312        }
1313    }
1314    if !results.unsampled_segments.is_empty() {
1315        coverage.limits.push(format!(
1316            "{} segments with rows, none sampled",
1317            count(results.unsampled_segments.len())
1318        ));
1319    }
1320    if let (Some(files), Some(read)) = (results.source_files, results.footers_read)
1321        && read < files
1322    {
1323        coverage.limits.push(format!(
1324            "footers of {} of {} files read",
1325            count(read),
1326            count(files)
1327        ));
1328    }
1329    if let Some(intent) = results.intent.as_ref().filter(|intent| intent.measured) {
1330        // A key with no repeat in a sample is unique among those rows, and no more.
1331        if intent.key.is_some() && intent.precision != QualityPrecision::Exact {
1332            coverage.limits.push(format!(
1333                "key repeats among {} sampled rows only",
1334                count(intent.evaluated_rows)
1335            ));
1336        }
1337    }
1338    if let Some(intent) = &results.intent
1339        && !intent.absent.is_empty()
1340    {
1341        coverage.limits.push(format!(
1342            "intent on {}: not in scope",
1343            columns_label(&intent.absent, 24)
1344        ));
1345    }
1346    if !plan.temporal_roles.is_empty() && plan.interval_pairs().is_empty() {
1347        coverage
1348            .limits
1349            .push("time roles form no interval".to_string());
1350    }
1351    coverage
1352}
1353
1354/// What a finding means for the data and what to do about it, a fragment a line.
1355/// The reader knows what a null is; this says only what is particular.
1356pub fn advice(finding: &Finding) -> Vec<String> {
1357    let lines: &[&str] = match (finding.kind, finding.title) {
1358        (Some(ObservationKind::Nulls), "Always missing") => {
1359            &["Carries nothing; check the load or a rename upstream"]
1360        }
1361        (Some(ObservationKind::Nulls), "Mostly missing") => {
1362            let rest = finding.evaluated_rows.saturating_sub(finding.affected_rows);
1363            return vec![
1364                if finding.columns.len() == 1 {
1365                    format!(
1366                        "Aggregates and joins see only {}",
1367                        percent(rest, finding.evaluated_rows)
1368                    )
1369                } else {
1370                    "Aggregates and joins see only the filled rows".to_string()
1371                },
1372                "Check: filled only for some rows, or stopped at some point".to_string(),
1373            ];
1374        }
1375        (Some(ObservationKind::Nulls), "Missing together") => {
1376            &["Likely one cause: a join with no match, or a source with gaps"]
1377        }
1378        (Some(ObservationKind::Nulls), _) => {
1379            &["Check: clustered in some files or dates (Segments, by file or window)"]
1380        }
1381        (Some(ObservationKind::Empty), _) => {
1382            &["Counted as filled; treat as null if it means missing"]
1383        }
1384        (Some(ObservationKind::Whitespace), _) => {
1385            &["Counted as filled; trim to null if it means missing"]
1386        }
1387        (Some(ObservationKind::NonFinite), _) => {
1388            &["Sums and means become NaN; check for division by zero upstream"]
1389        }
1390        (Some(ObservationKind::Constant), _) => {
1391            &["Tells no rows apart; a stuck feed if it should vary"]
1392        }
1393        (Some(ObservationKind::ParseableText), "Codes as text") => {
1394            &["Fine as text; cast only for arithmetic"]
1395        }
1396        (Some(ObservationKind::ParseableText), "Dates as text") => {
1397            &["Sorts as text; parse as a date to filter by range"]
1398        }
1399        (Some(ObservationKind::ParseableText), _) => {
1400            &["Sorts as text (\"10\" before \"9\"); cast to a number to sum"]
1401        }
1402        (Some(ObservationKind::DuplicateRows), _) => {
1403            &["Counted more than once; check for a double load or a join fan-out"]
1404        }
1405        (Some(ObservationKind::CategoryVariants), _) => {
1406            &["Group-bys and joins split them; trim and normalize case"]
1407        }
1408        (Some(ObservationKind::Absent), _) => &["Check: files written before the column existed"],
1409        (Some(ObservationKind::TypeConflict), _) => {
1410            &["Values dropped, not converted; read as text in Info, or fix the writer"]
1411        }
1412        (Some(ObservationKind::KeyLike), _) => {
1413            &["Duplicates if it is a key; expected if it is a measurement"]
1414        }
1415        (Some(ObservationKind::UnparsedTime), _) => &[
1416            "Left out of time windows and intervals, not counted as missing",
1417            "Check: another format, or values that are not times (Setup, e)",
1418        ],
1419        (Some(ObservationKind::KeyRepeated), _) => &[
1420            "A key names one row; joins on it fan out and counts double",
1421            "Check: a double load, or a key that needs another column",
1422        ],
1423        (Some(ObservationKind::KeyMissing), _) => {
1424            &["Rows with no key cannot be joined or told apart by it"]
1425        }
1426        (Some(ObservationKind::RequiredMissing), _) => {
1427            &["Check: the load, or rows the source writes without it"]
1428        }
1429        (Some(ObservationKind::NotAllowed), _) => {
1430            &["Check: a new value upstream, or the allowed list (Setup, e)"]
1431        }
1432        (Some(ObservationKind::OutOfRange), _) => {
1433            &["Check: units, placeholders such as -1 or 9999, or the range (Setup, e)"]
1434        }
1435        (Some(ObservationKind::UnparsedNumber), _) => {
1436            &["A cast makes them null; check the values or the reading (Setup, e)"]
1437        }
1438        (Some(ObservationKind::Clipping), _) => {
1439            &["Cut flat at the limit; lower the gain at the source, nothing restores it"]
1440        }
1441        (Some(ObservationKind::ZeroRuns), _) => {
1442            &["Dropouts mid-recording; digital silence if at the start or end"]
1443        }
1444        (Some(ObservationKind::DcOffset), _) => {
1445            &["A constant bias that eats headroom; a high-pass filter removes it"]
1446        }
1447        (None, _) => &[],
1448    };
1449    lines.iter().map(|line| line.to_string()).collect()
1450}
1451
1452/// The finding's numbers in a fragment, then the evidence that makes it concrete:
1453/// the values, the spellings, the files.
1454pub fn describe(finding: &Finding, results: &DataQualityResults) -> (String, Vec<String>) {
1455    let count = |value: usize| numfmt::group_chrome(value);
1456    let of = format!(
1457        "{} of {} rows ({})",
1458        count(finding.affected_rows),
1459        count(finding.evaluated_rows),
1460        percent(finding.affected_rows, finding.evaluated_rows)
1461    );
1462    let profile = |name: &str| results.columns.iter().find(|profile| profile.name == name);
1463    let observation = |index: &usize| &results.observations[*index];
1464    let grouped = finding.columns.len() > 1;
1465    let mut evidence = Vec::new();
1466    let headline = match finding.kind {
1467        None => format!(
1468            "{} {} passed every check",
1469            count(finding.columns.len()),
1470            if finding.columns.len() == 1 {
1471                "column"
1472            } else {
1473                "columns"
1474            }
1475        ),
1476        Some(ObservationKind::Nulls) if finding.severity == Severity::Problem => {
1477            format!("Null in all {} rows checked", count(finding.evaluated_rows))
1478        }
1479        Some(ObservationKind::Nulls) if finding.same_rows => {
1480            if finding.columns.len() == 2 {
1481                evidence.push("No row misses one without the other".to_string());
1482                format!("{of} null in both columns")
1483            } else {
1484                evidence.push("No row misses one of them without the others".to_string());
1485                format!("{of} null in all {} columns", finding.columns.len())
1486            }
1487        }
1488        Some(ObservationKind::Nulls) if finding.varied() => {
1489            // Each column's own rate, worst first: the list row gave only the range.
1490            evidence.extend(finding.breakdown(results));
1491            format!("Null rate in {} columns:", finding.columns.len())
1492        }
1493        Some(ObservationKind::Nulls | ObservationKind::Empty | ObservationKind::Whitespace)
1494            if finding.lists_columns() =>
1495        {
1496            evidence.extend(finding.breakdown(results));
1497            let what = match finding.kind {
1498                Some(ObservationKind::Nulls) => "null",
1499                Some(ObservationKind::Empty) => "empty strings",
1500                _ => "only spaces or tabs",
1501            };
1502            format!("{of} {what} in each column:")
1503        }
1504        Some(ObservationKind::Nulls) => format!("{of} null"),
1505        Some(ObservationKind::Empty) => format!("{of} empty strings"),
1506        Some(ObservationKind::Whitespace) => format!("{of} only spaces or tabs"),
1507        Some(ObservationKind::NonFinite) => {
1508            if let Some(profile) = profile(&finding.columns[0]) {
1509                evidence.push(format!(
1510                    "NaN {}, +inf {}, -inf {}",
1511                    count(profile.nan_count.unwrap_or(0)),
1512                    count(profile.positive_infinity_count.unwrap_or(0)),
1513                    count(profile.negative_infinity_count.unwrap_or(0))
1514                ));
1515            }
1516            format!("{of} NaN or infinite")
1517        }
1518        Some(ObservationKind::Constant) if grouped => {
1519            for column in &finding.columns {
1520                if let Some(value) = profile(column).and_then(|p| p.dominant_value.as_ref()) {
1521                    evidence.push(format!("{column}: always {}", quoted(value, 40)));
1522                }
1523            }
1524            format!(
1525                "One value per column in {} rows",
1526                count(finding.evaluated_rows)
1527            )
1528        }
1529        Some(ObservationKind::Constant) => {
1530            match profile(&finding.columns[0]).and_then(|p| p.dominant_value.as_ref()) {
1531                Some(value) => format!(
1532                    "Always {} in {} rows",
1533                    quoted(value, 40),
1534                    count(finding.evaluated_rows)
1535                ),
1536                None => format!("One value in {} rows", count(finding.evaluated_rows)),
1537            }
1538        }
1539        Some(ObservationKind::ParseableText) => {
1540            let column = &finding.columns[0];
1541            let profile = profile(column);
1542            let reading = profile.and_then(text_reading);
1543            if let (Some(profile), Some((parsed, reading))) = (profile, reading) {
1544                if let (Some(min), Some(max)) = (profile.min_length, profile.max_length) {
1545                    evidence.push(if min == max {
1546                        format!("{min} characters each")
1547                    } else {
1548                        format!("{min} to {max} characters")
1549                    });
1550                }
1551                if let Some(zeros) = profile.leading_zero_count.filter(|zeros| *zeros > 0) {
1552                    evidence.push(format!("{} with a leading zero", count(zeros)));
1553                }
1554                if let (Some(min), Some(max)) = (&profile.min, &profile.max) {
1555                    evidence.push(format!("From {} to {}", quoted(min, 24), quoted(max, 24)));
1556                }
1557                let failed = profile.non_null_rows().saturating_sub(parsed);
1558                if failed > 0 {
1559                    let examples = results.examples_of(ObservationKind::ParseableText, column);
1560                    evidence.push(if examples.is_empty() {
1561                        format!("{} do not parse", count(failed))
1562                    } else {
1563                        cut(
1564                            &format!(
1565                                "{} do not parse, such as {}",
1566                                count(failed),
1567                                examples.join(", ")
1568                            ),
1569                            EXAMPLE_WIDTH,
1570                        )
1571                    });
1572                }
1573                format!(
1574                    "{} of {} values ({}) parse as {}",
1575                    count(parsed),
1576                    count(profile.non_null_rows()),
1577                    percent(parsed, profile.non_null_rows()),
1578                    reading.label()
1579                )
1580            } else {
1581                observation(&finding.observations[0]).fact.clone()
1582            }
1583        }
1584        Some(ObservationKind::DuplicateRows) => match results.identity.as_ref() {
1585            Some(identity) => {
1586                evidence.push(format!(
1587                    "{} of {} rows ({}) have a copy",
1588                    count(identity.rows_involved),
1589                    count(identity.evaluated_rows),
1590                    percent(identity.rows_involved, identity.evaluated_rows)
1591                ));
1592                if !identity.examples.is_empty() {
1593                    evidence.push("Most copied:".to_string());
1594                }
1595                for example in &identity.examples {
1596                    evidence.push(cut(
1597                        &format!(
1598                            "{}{}  {}",
1599                            crate::glyphs::get().times,
1600                            example.copies,
1601                            example.values.join(", ")
1602                        ),
1603                        EXAMPLE_WIDTH,
1604                    ));
1605                }
1606                format!(
1607                    "{} rows repeated; {} extra {}",
1608                    count(identity.duplicate_groups),
1609                    count(identity.extra_rows),
1610                    if identity.extra_rows == 1 {
1611                        "copy"
1612                    } else {
1613                        "copies"
1614                    }
1615                )
1616            }
1617            None => observation(&finding.observations[0]).fact.clone(),
1618        },
1619        Some(ObservationKind::CategoryVariants) => {
1620            for index in &finding.observations {
1621                let normalized = observation(index).normalized_category.as_ref();
1622                let Some(group) = results
1623                    .category_variants
1624                    .iter()
1625                    .find(|group| Some(&group.normalized) == normalized)
1626                else {
1627                    continue;
1628                };
1629                let mut variants = group.variants.iter().collect::<Vec<_>>();
1630                variants.sort_by_key(|(_, rows)| std::cmp::Reverse(*rows));
1631                evidence.push(
1632                    variants
1633                        .into_iter()
1634                        .take(4)
1635                        .map(|(value, rows)| format!("{} ({})", quoted(value, 28), count(*rows)))
1636                        .collect::<Vec<_>>()
1637                        .join("  "),
1638                );
1639            }
1640            let values = finding.observations.len();
1641            format!(
1642                "{} {} spelled more than one way, {of}",
1643                count(values),
1644                if values == 1 { "value" } else { "values" }
1645            )
1646        }
1647        Some(ObservationKind::KeyLike) => {
1648            let column = &finding.columns[0];
1649            if let Some(profile) = profile(column) {
1650                if let (Some(value), Some(times)) =
1651                    (&profile.dominant_value, profile.dominant_count)
1652                {
1653                    evidence.push(format!(
1654                        "Most repeated: {} ({} times)",
1655                        quoted(value, 32),
1656                        count(times)
1657                    ));
1658                }
1659                format!(
1660                    "{} distinct in {} rows; {} repeats",
1661                    count(profile.distinct_count.unwrap_or(0)),
1662                    count(profile.non_null_rows()),
1663                    count(finding.affected_rows)
1664                )
1665            } else {
1666                observation(&finding.observations[0]).fact.clone()
1667            }
1668        }
1669        Some(ObservationKind::UnparsedTime) => {
1670            let observation = observation(&finding.observations[0]);
1671            if let Some(format) = &observation.time_format {
1672                evidence.push(format!("Read as {} for this study only", format.label()));
1673            }
1674            let examples = results.examples_of(ObservationKind::UnparsedTime, &observation.column);
1675            if !examples.is_empty() {
1676                evidence.push(cut(
1677                    &format!("Such as {}", examples.join(", ")),
1678                    EXAMPLE_WIDTH,
1679                ));
1680            }
1681            format!(
1682                "{} of {} values ({}) do not parse",
1683                count(finding.affected_rows),
1684                count(finding.evaluated_rows),
1685                percent(finding.affected_rows, finding.evaluated_rows)
1686            )
1687        }
1688        Some(
1689            kind @ (ObservationKind::KeyRepeated
1690            | ObservationKind::KeyMissing
1691            | ObservationKind::RequiredMissing
1692            | ObservationKind::NotAllowed
1693            | ObservationKind::OutOfRange
1694            | ObservationKind::UnparsedNumber),
1695        ) => describe_intent(kind, finding, results, &mut evidence),
1696        Some(
1697            kind @ (ObservationKind::Clipping
1698            | ObservationKind::ZeroRuns
1699            | ObservationKind::DcOffset),
1700        ) => {
1701            for index in &finding.observations {
1702                let observation = observation(index);
1703                evidence.push(format!("{}: {}", observation.column, observation.fact));
1704            }
1705            match kind {
1706                ObservationKind::Clipping => format!("{of} in runs at full scale"),
1707                ObservationKind::ZeroRuns => format!("{of} in runs of exact zeros"),
1708                _ => "Mean away from zero".to_string(),
1709            }
1710        }
1711        Some(ObservationKind::Absent | ObservationKind::TypeConflict) => {
1712            let observation = observation(&finding.observations[0]);
1713            // Read from the footers, so the denominator is the whole loaded source
1714            // whatever the plan's scope was.
1715            evidence.push(format!(
1716                "{} of {} rows of the loaded source ({})",
1717                count(finding.affected_rows),
1718                count(finding.evaluated_rows),
1719                percent(finding.affected_rows, finding.evaluated_rows)
1720            ));
1721            for file in observation.files.iter().take(4) {
1722                let stored = file
1723                    .stored_type
1724                    .as_ref()
1725                    .map(|dtype| format!(" as {dtype}"))
1726                    .unwrap_or_default();
1727                let examples = if file.examples.is_empty() {
1728                    String::new()
1729                } else {
1730                    format!(
1731                        ": {}",
1732                        file.examples
1733                            .iter()
1734                            .map(|value| quoted(value, 20))
1735                            .collect::<Vec<_>>()
1736                            .join(", ")
1737                    )
1738                };
1739                evidence.push(format!(
1740                    "#{} {} ({} rows){stored}{examples}",
1741                    file.number,
1742                    file.name,
1743                    count(file.rows)
1744                ));
1745            }
1746            if observation.files.len() > 4 {
1747                evidence.push(format!(
1748                    "{} more {}",
1749                    observation.files.len() - 4,
1750                    crate::glyphs::get().ellipsis
1751                ));
1752            }
1753            upper_first(&observation.fact)
1754        }
1755    };
1756    (headline, evidence)
1757}
1758
1759/// How wide a line of examples runs before it is cut: a row of many columns would
1760/// otherwise wrap over the whole detail.
1761const EXAMPLE_WIDTH: usize = 72;
1762
1763/// `text` cut to `width` display columns, the ellipsis glyph marking the cut.
1764fn cut(text: &str, width: usize) -> String {
1765    if crate::glyphs::display_width(text) <= width {
1766        return text.to_string();
1767    }
1768    let ellipsis = crate::glyphs::get().ellipsis;
1769    format!(
1770        "{}{ellipsis}",
1771        crate::glyphs::take_columns(
1772            text,
1773            width.saturating_sub(crate::glyphs::display_width(ellipsis))
1774        )
1775    )
1776}
1777
1778/// A declared rule's violation: the count against what it is out of, the rule as
1779/// declared, and what the rows in memory showed of it. A sample's numbers say so.
1780fn describe_intent(
1781    kind: ObservationKind,
1782    finding: &Finding,
1783    results: &DataQualityResults,
1784    evidence: &mut Vec<String>,
1785) -> String {
1786    let count = numfmt::group_chrome;
1787    let Some(intent) = results.intent.as_ref() else {
1788        return finding.summary.clone();
1789    };
1790    let sampled = intent.precision != QualityPrecision::Exact;
1791    let rows_word = if sampled { "sampled rows" } else { "rows" };
1792    let share = percent(finding.affected_rows, finding.evaluated_rows);
1793    let examples = |values: &[(String, usize)]| {
1794        values
1795            .iter()
1796            .map(|(value, rows)| format!("{} ({})", quoted(value, 24), count(*rows)))
1797            .collect::<Vec<_>>()
1798            .join("  ")
1799    };
1800    let column = finding.columns.first().map(String::as_str).unwrap_or("");
1801    let check = intent.column(column);
1802    match kind {
1803        ObservationKind::KeyRepeated | ObservationKind::KeyMissing => {
1804            let Some(key) = intent.key.as_ref() else {
1805                return finding.summary.clone();
1806            };
1807            evidence.push(format!("Declared key: {}", key.columns.join(", ")));
1808            if kind == ObservationKind::KeyMissing {
1809                return format!(
1810                    "{} of {} {rows_word} ({share}) have no value in part of the key",
1811                    count(key.missing),
1812                    count(intent.evaluated_rows)
1813                );
1814            }
1815            evidence.push(format!(
1816                "{} {} held by more than one row; {} rows beyond one per value",
1817                count(key.groups),
1818                if key.groups == 1 { "value" } else { "values" },
1819                count(key.extra_rows)
1820            ));
1821            if sampled {
1822                evidence.push(
1823                    "Each repeat here is one in the data; unsampled rows are not checked"
1824                        .to_string(),
1825                );
1826            }
1827            format!(
1828                "{} of {} {rows_word} ({share}) share their key with another row",
1829                count(key.rows_involved),
1830                count(intent.evaluated_rows)
1831            )
1832        }
1833        ObservationKind::RequiredMissing => {
1834            evidence.push(format!("Declared required: {column}"));
1835            format!(
1836                "{} of {} {rows_word} ({share}) have no {column}",
1837                count(finding.affected_rows),
1838                count(finding.evaluated_rows)
1839            )
1840        }
1841        ObservationKind::NotAllowed => {
1842            if let Some(check) = check {
1843                evidence.push(format!("Allowed: {}", check.intent.allowed_label(8)));
1844                if !check.outside_examples.is_empty() {
1845                    evidence.push(format!("Found: {}", examples(&check.outside_examples)));
1846                }
1847            }
1848            format!(
1849                "{} of {} values ({share}) are not allowed",
1850                count(finding.affected_rows),
1851                count(finding.evaluated_rows)
1852            )
1853        }
1854        ObservationKind::OutOfRange => {
1855            if let Some(check) = check {
1856                evidence.push(format!(
1857                    "Range: {}",
1858                    check.intent.range_label().unwrap_or_default()
1859                ));
1860                if let Some(below) = check.below.filter(|below| *below > 0) {
1861                    let lowest = check
1862                        .lowest
1863                        .as_ref()
1864                        .map(|value| format!(", lowest {value}"))
1865                        .unwrap_or_default();
1866                    evidence.push(format!("Below: {}{lowest}", count(below)));
1867                }
1868                if let Some(above) = check.above.filter(|above| *above > 0) {
1869                    let highest = check
1870                        .highest
1871                        .as_ref()
1872                        .map(|value| format!(", highest {value}"))
1873                        .unwrap_or_default();
1874                    evidence.push(format!("Above: {}{highest}", count(above)));
1875                }
1876            }
1877            format!(
1878                "{} of {} values ({share}) outside the range",
1879                count(finding.affected_rows),
1880                count(finding.evaluated_rows)
1881            )
1882        }
1883        _ => {
1884            let reading = check
1885                .and_then(|check| check.intent.number)
1886                .map_or("number", crate::quality_intent::NumberReading::label);
1887            if let Some(check) = check.filter(|check| !check.unparsed_examples.is_empty()) {
1888                evidence.push(format!("Such as: {}", examples(&check.unparsed_examples)));
1889            }
1890            evidence.push(format!("Read as a {reading} for this study only"));
1891            format!(
1892                "{} of {} values ({share}) do not read as a {reading}",
1893                count(finding.affected_rows),
1894                count(finding.evaluated_rows)
1895            )
1896        }
1897    }
1898}
1899
1900fn upper_first(text: &str) -> String {
1901    let mut chars = text.chars();
1902    match chars.next() {
1903        Some(first) => first.to_uppercase().chain(chars).collect(),
1904        None => String::new(),
1905    }
1906}
1907
1908/// Headline counts, one line: the answer before the evidence.
1909pub fn verdict(report: &QualityReport) -> String {
1910    // Footers still say which files lack a column or hold it in another type, so a
1911    // metadata run can find problems; it cannot call a column clean.
1912    if report.metadata_only {
1913        return match report.problems {
1914            0 => "Values not read: file metadata only".to_string(),
1915            1 => "1 problem in file metadata; values not read".to_string(),
1916            count => format!(
1917                "{} problems in file metadata; values not read",
1918                numfmt::group_chrome(count)
1919            ),
1920        };
1921    }
1922    // No rows is no evidence: nothing passed, and nothing is clean.
1923    if report.no_rows {
1924        return match report.problems {
1925            0 => "No rows to check".to_string(),
1926            1 => "1 problem in file metadata; no rows to check".to_string(),
1927            count => format!(
1928                "{} problems in file metadata; no rows to check",
1929                numfmt::group_chrome(count)
1930            ),
1931        };
1932    }
1933    let clean = format!(
1934        "{} of {} columns clean",
1935        numfmt::group_chrome(report.clean_columns),
1936        numfmt::group_chrome(report.total_columns)
1937    );
1938    let notes = match report.notes {
1939        0 => String::new(),
1940        1 => "1 note  ".to_string(),
1941        count => format!("{} notes  ", numfmt::group_chrome(count)),
1942    };
1943    match report.problems {
1944        0 => format!("No problems found  {notes}{clean}"),
1945        1 => format!("1 problem  {notes}{clean}"),
1946        count => format!("{} problems  {notes}{clean}", numfmt::group_chrome(count)),
1947    }
1948}
1949
1950#[cfg(test)]
1951mod tests {
1952    use super::*;
1953    use crate::data_quality::{DataQualityPlan, QualityObservation, SharedNulls};
1954    use polars::prelude::DataType;
1955
1956    fn profile(name: &str, dtype: DataType) -> ColumnQualityProfile {
1957        ColumnQualityProfile {
1958            name: name.to_string(),
1959            dtype,
1960            evaluated_rows: 100,
1961            null_count: 0,
1962            empty_count: None,
1963            whitespace_count: None,
1964            nan_count: None,
1965            positive_infinity_count: None,
1966            negative_infinity_count: None,
1967            distinct_count: None,
1968            min: None,
1969            max: None,
1970            integer_parse_count: None,
1971            decimal_parse_count: None,
1972            date_parse_count: None,
1973            datetime_parse_count: None,
1974            leading_zero_count: None,
1975            dominant_value: None,
1976            dominant_count: None,
1977            min_length: None,
1978            max_length: None,
1979        }
1980    }
1981
1982    fn observation(kind: ObservationKind, column: &str, affected: usize) -> QualityObservation {
1983        QualityObservation {
1984            kind,
1985            column: column.to_string(),
1986            affected_rows: affected,
1987            evaluated_rows: 100,
1988            fact: String::new(),
1989            normalized_category: None,
1990            files: Vec::new(),
1991            time_format: None,
1992            full_scale: None,
1993        }
1994    }
1995
1996    fn results(
1997        columns: Vec<ColumnQualityProfile>,
1998        observations: Vec<QualityObservation>,
1999    ) -> DataQualityResults {
2000        let plan = DataQualityPlan::default();
2001        let mut results =
2002            DataQualityResults::empty(Some(100), &plan, &polars::prelude::Schema::default());
2003        results.evaluated_rows = 100;
2004        results.precision = QualityPrecision::Exact;
2005        results.columns = columns;
2006        results.observations = observations;
2007        results
2008    }
2009
2010    /// Sixteen columns missing on the same rows are one fact, and it says so.
2011    #[test]
2012    fn nulls_with_one_count_become_one_finding() {
2013        let names = ["open", "high", "low", "close"];
2014        let mut columns = names
2015            .iter()
2016            .map(|name| profile(name, DataType::Float64))
2017            .collect::<Vec<_>>();
2018        columns.push(profile("ticker", DataType::String));
2019        let mut results = results(
2020            columns,
2021            names
2022                .iter()
2023                .map(|name| observation(ObservationKind::Nulls, name, 4))
2024                .collect(),
2025        );
2026        results.shared_nulls = vec![SharedNulls {
2027            columns: names.iter().map(|name| name.to_string()).collect(),
2028            null_rows: 4,
2029            rows_null_in_all: 4,
2030        }];
2031        let report = build_report(&results);
2032        assert_eq!(report.findings.len(), 2, "one note and the clean entry");
2033        let missing = &report.findings[0];
2034        assert_eq!(missing.title, "Missing together");
2035        assert_eq!(missing.severity, Severity::Note);
2036        assert_eq!(missing.columns.len(), 4);
2037        assert!(missing.same_rows);
2038        assert_eq!(missing.summary, "4 rows (4.0%)");
2039        assert_eq!(report.findings[1].severity, Severity::Clean);
2040        assert_eq!(report.findings[1].columns, vec!["ticker".to_string()]);
2041        assert_eq!(report.problems, 0);
2042        assert!(verdict(&report).starts_with("No problems found"));
2043    }
2044
2045    /// Missing values at different rates are one finding listing each column worst
2046    /// first; columns missing in most rows are their own note, ranked above it.
2047    #[test]
2048    fn missing_values_collapse_and_mostly_missing_leads() {
2049        let names = ["a", "b", "c", "d"];
2050        let results = results(
2051            names
2052                .iter()
2053                .map(|name| profile(name, DataType::Float64))
2054                .collect(),
2055            vec![
2056                observation(ObservationKind::Nulls, "a", 3),
2057                observation(ObservationKind::Nulls, "b", 40),
2058                observation(ObservationKind::Nulls, "c", 12),
2059                observation(ObservationKind::Nulls, "d", 80),
2060            ],
2061        );
2062        let report = build_report(&results);
2063        assert_eq!(report.findings.len(), 2);
2064        let mostly = &report.findings[0];
2065        assert_eq!(mostly.title, "Mostly missing");
2066        assert_eq!(mostly.columns, vec!["d".to_string()]);
2067        let missing = &report.findings[1];
2068        assert_eq!(missing.title, "Missing values");
2069        assert_eq!(missing.columns, vec!["b", "c", "a"]);
2070        assert_eq!(missing.summary, "3.0% to 40.0% per column");
2071        let (headline, evidence) = describe(missing, &results);
2072        assert_eq!(headline, "Null rate in 3 columns:");
2073        assert_eq!(evidence[0], "b    40.0%  40 rows");
2074        assert_eq!(evidence[2], "a     3.0%  3 rows");
2075    }
2076
2077    #[test]
2078    fn problems_rank_before_notes_and_mark_their_columns() {
2079        let results = results(
2080            vec![
2081                profile("price", DataType::Float64),
2082                profile("region", DataType::String),
2083            ],
2084            vec![
2085                observation(ObservationKind::Nulls, "region", 3),
2086                observation(ObservationKind::NonFinite, "price", 2),
2087            ],
2088        );
2089        let report = build_report(&results);
2090        assert_eq!(report.findings[0].title, "NaN or infinite");
2091        assert_eq!(report.findings[0].severity, Severity::Problem);
2092        assert_eq!(report.findings[1].title, "Missing values");
2093        assert_eq!(
2094            report.column_status,
2095            vec![Severity::Problem, Severity::Note]
2096        );
2097        assert_eq!(report.clean_columns, 0);
2098        assert_eq!(verdict(&report), "1 problem  1 note  0 of 2 columns clean");
2099    }
2100
2101    #[test]
2102    fn a_column_with_no_values_at_all_is_a_problem() {
2103        let results = results(
2104            vec![profile("legacy", DataType::String)],
2105            vec![observation(ObservationKind::Nulls, "legacy", 100)],
2106        );
2107        let report = build_report(&results);
2108        assert_eq!(report.findings[0].title, "Always missing");
2109        assert_eq!(report.findings[0].severity, Severity::Problem);
2110    }
2111
2112    /// Fixed-width digits with leading zeros are a code, and the report says the
2113    /// column is fine as text rather than asking for a cast.
2114    #[test]
2115    fn zero_padded_digits_read_as_codes() {
2116        let mut code = profile("industry", DataType::String);
2117        code.integer_parse_count = Some(100);
2118        code.decimal_parse_count = Some(100);
2119        code.leading_zero_count = Some(20);
2120        code.min_length = Some(4);
2121        code.max_length = Some(4);
2122        let results = results(
2123            vec![code],
2124            vec![observation(ObservationKind::ParseableText, "industry", 100)],
2125        );
2126        let finding = &build_report(&results).findings[0];
2127        assert_eq!(finding.title, "Codes as text");
2128        assert_eq!(finding.summary, "4 digits, leading zeros");
2129    }
2130
2131    /// Footers can show a problem without reading a value; nothing can be called
2132    /// clean that way.
2133    #[test]
2134    fn a_metadata_run_reports_footer_problems_and_no_clean_columns() {
2135        let mut results = results(
2136            vec![
2137                profile("fee", DataType::Float64),
2138                profile("id", DataType::Int64),
2139            ],
2140            vec![observation(ObservationKind::Absent, "fee", 20)],
2141        );
2142        results.precision = QualityPrecision::Metadata;
2143        let report = build_report(&results);
2144        assert_eq!(report.findings.len(), 1, "no clean entry");
2145        assert_eq!(report.findings[0].title, "Missing in files");
2146        assert_eq!(
2147            verdict(&report),
2148            "1 problem in file metadata; values not read"
2149        );
2150    }
2151
2152    /// The checks say what they covered, what they found, and what they could not
2153    /// look at and why, so a clean result is one the reader can trust.
2154    #[test]
2155    fn checks_report_reach_findings_and_what_did_not_run() {
2156        let mut results = results(
2157            vec![
2158                profile("price", DataType::Float64),
2159                profile("region", DataType::String),
2160                profile("id", DataType::Int64),
2161            ],
2162            vec![observation(ObservationKind::NonFinite, "price", 2)],
2163        );
2164        results.precision = QualityPrecision::Sampled;
2165        let report = build_report(&results);
2166        let list = checks(&results, &report);
2167        let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2168        assert_eq!(list[0].name, "Missing values", "most important first");
2169        assert_eq!(by_name("Missing values").outcome, Outcome::Passed);
2170        assert_eq!(by_name("Missing values").applies_to, "3 columns");
2171        assert_eq!(by_name("NaN or infinite").applies_to, "1 float column");
2172        assert_eq!(
2173            by_name("NaN or infinite").outcome,
2174            Outcome::Found {
2175                tier: Severity::Problem,
2176                detail: "1 column".to_string()
2177            }
2178        );
2179        assert_eq!(
2180            by_name("Nearly unique").outcome,
2181            Outcome::Unavailable("needs every row checked"),
2182            "a sample cannot say a column is nearly a key"
2183        );
2184        assert_eq!(
2185            by_name("Type mismatch").outcome,
2186            Outcome::Skipped("needs several files")
2187        );
2188
2189        results.precision = QualityPrecision::Metadata;
2190        results.source_files = Some(3);
2191        let report = build_report(&results);
2192        let list = checks(&results, &report);
2193        let by_name = |name: &str| list.iter().find(|check| check.name == name).unwrap();
2194        assert_eq!(
2195            by_name("Missing values").outcome,
2196            Outcome::Unavailable("values not read")
2197        );
2198        assert_eq!(by_name("Type mismatch").outcome, Outcome::Passed);
2199        assert_eq!(by_name("Type mismatch").applies_to, "3 files");
2200    }
2201
2202    /// Coverage tells a clean sample from an exhaustive run: which checks ran over
2203    /// what, which had nothing to look at, which this run could not answer and why,
2204    /// and what else bounds the result, each count beside its denominator.
2205    #[test]
2206    fn coverage_separates_checked_skipped_and_unavailable() {
2207        use crate::data_quality::{
2208            ObservedReads, SegmentQualityProfile, TemporalRole, TemporalRoleAssignment,
2209        };
2210        let columns = vec![
2211            profile("price", DataType::Float64),
2212            profile("region", DataType::String),
2213            profile("id", DataType::Int64),
2214        ];
2215        let plan = DataQualityPlan::default();
2216        let measured = |mut results: DataQualityResults| {
2217            results.identity = Some(crate::data_quality::IdentityProfile {
2218                duplicate_groups: 0,
2219                extra_rows: 0,
2220                rows_involved: 0,
2221                evaluated_rows: results.evaluated_rows,
2222                precision: results.precision,
2223                examples: Vec::new(),
2224            });
2225            results
2226        };
2227
2228        // A clean sampled run of one file, its sample streamed from 1,000 rows.
2229        let mut sampled = measured(results(columns.clone(), Vec::new()));
2230        sampled.precision = QualityPrecision::Sampled;
2231        sampled.total_rows = Some(1_000);
2232        sampled.reads = Some(ObservedReads {
2233            reads: 1,
2234            counted: 1,
2235            rows: 1_000,
2236            copy: None,
2237        });
2238        let report = build_report(&sampled);
2239        assert_eq!(report.problems + report.notes, 0, "a clean report");
2240        let found = coverage(&sampled, &checks(&sampled, &report), &plan);
2241        assert_eq!(
2242            found.checks(),
2243            ["7 sampled", "2 skipped", "1 unavailable"],
2244            "{found:?}"
2245        );
2246        assert_eq!(
2247            found.rows,
2248            ["100 of 1,000 sampled (10.0%)", "1,000 traversed"]
2249        );
2250        assert_eq!(found.limits(), ["Nearly unique: needs every row checked"]);
2251
2252        // Thin segments, footers read for only some files, roles that pair nothing.
2253        let segment = |label: &str, rows: usize| SegmentQualityProfile {
2254            label: label.to_string(),
2255            total_rows: Some(500),
2256            evaluated_rows: rows,
2257            columns: Vec::new(),
2258            null_cells: 0,
2259            null_rate: 0.0,
2260            compared_with: None,
2261            largest_change: None,
2262            change_size: None,
2263        };
2264        sampled.segments = vec![segment("a", 90), segment("b", 10), segment("c", 0)];
2265        sampled.source_files = Some(400);
2266        sampled.footers_read = Some(100);
2267        let roles = DataQualityPlan {
2268            temporal_roles: vec![TemporalRoleAssignment {
2269                role: TemporalRole::Event,
2270                column: "id".to_string(),
2271                timezone: None,
2272            }],
2273            ..plan.clone()
2274        };
2275        let report = build_report(&sampled);
2276        let list = checks(&sampled, &report);
2277        let type_mismatch = list.iter().find(|c| c.name == "Type mismatch").unwrap();
2278        assert_eq!(type_mismatch.applies_to, "100 of 400 files");
2279        assert_eq!(type_mismatch.basis, QualityPrecision::Metadata);
2280        let found = coverage(&sampled, &list, &roles);
2281        assert_eq!(found.checks(), ["7 sampled", "2 metadata", "1 unavailable"]);
2282        assert_eq!(
2283            found.limits(),
2284            [
2285                "Nearly unique: needs every row checked",
2286                "2 of 3 segments under 30 sampled rows",
2287                "footers of 100 of 400 files read",
2288                "time roles form no interval",
2289            ]
2290        );
2291
2292        // Every row read: nothing unavailable, and the passes' rows beside the total.
2293        let mut full = measured(results(columns.clone(), Vec::new()));
2294        full.reads = Some(ObservedReads {
2295            reads: 4,
2296            counted: 3,
2297            rows: 300,
2298            copy: None,
2299        });
2300        let report = build_report(&full);
2301        let found = coverage(&full, &checks(&full, &report), &plan);
2302        assert_eq!(found.checks(), ["8 exact", "2 skipped"]);
2303        assert_eq!(
2304            found.rows,
2305            ["all 100 read, exact", "at least 300 traversed"]
2306        );
2307        assert!(found.limits().is_empty());
2308
2309        // No values read: the footers are all that was checked.
2310        let mut metadata = measured(results(columns, Vec::new()));
2311        metadata.precision = QualityPrecision::Metadata;
2312        metadata.source_files = Some(3);
2313        metadata.footers_read = Some(3);
2314        metadata.reads = Some(ObservedReads::default());
2315        let report = build_report(&metadata);
2316        let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2317        assert_eq!(found.checks(), ["2 metadata", "8 unavailable"]);
2318        assert_eq!(found.rows, ["none read, file metadata only"]);
2319        assert_eq!(found.limits(), ["8 checks: values not read"]);
2320
2321        // A check with no column of its kind is skipped whatever was read: the schema
2322        // says so without the values.
2323        let floats_only = vec![profile("price", DataType::Float64)];
2324        let mut metadata = measured(results(floats_only.clone(), Vec::new()));
2325        metadata.precision = QualityPrecision::Metadata;
2326        let report = build_report(&metadata);
2327        let found = coverage(&metadata, &checks(&metadata, &report), &plan);
2328        assert_eq!(found.checks(), ["6 skipped", "4 unavailable"], "{found:?}");
2329        let mut sampled = measured(results(floats_only, Vec::new()));
2330        sampled.precision = QualityPrecision::Sampled;
2331        let report = build_report(&sampled);
2332        let list = checks(&sampled, &report);
2333        let nearly = list.iter().find(|c| c.name == "Nearly unique").unwrap();
2334        assert_eq!(
2335            nearly.outcome,
2336            Outcome::Skipped("no integer or text columns")
2337        );
2338    }
2339
2340    #[test]
2341    fn column_labels_fit_and_count_the_rest() {
2342        let columns = ["open", "high", "low", "close"].map(String::from);
2343        assert_eq!(columns_label(&columns, 40), "open, high, low, close");
2344        assert_eq!(columns_label(&columns, 14), "open, high +2");
2345        assert_eq!(columns_label(&columns[..1], 2), "open");
2346    }
2347
2348    /// Narrowing and ordering read the report as it is: Problems stay above Notes in
2349    /// every order, a column or a check keeps only the findings that name it, and the
2350    /// rows a finding counts or the share they are decide the order within a
2351    /// severity.
2352    #[test]
2353    fn findings_narrow_and_order_without_measuring() {
2354        let mut results = results(
2355            vec![
2356                profile("price", DataType::Float64),
2357                profile("region", DataType::String),
2358                profile("note", DataType::String),
2359                profile("id", DataType::Int64),
2360            ],
2361            vec![
2362                observation(ObservationKind::NonFinite, "price", 2),
2363                observation(ObservationKind::Nulls, "region", 3),
2364                observation(ObservationKind::Nulls, "note", 30),
2365                observation(ObservationKind::Whitespace, "region", 9),
2366            ],
2367        );
2368        // A share over a smaller denominator: fewer rows, higher rate.
2369        results.observations[0].evaluated_rows = 4;
2370        let report = build_report(&results);
2371        let titles = |view: &FindingsView| {
2372            view.shown(&report)
2373                .into_iter()
2374                .map(|index| report.findings[index].title)
2375                .collect::<Vec<_>>()
2376        };
2377        let ranked = FindingsView::default();
2378        assert_eq!(
2379            titles(&ranked),
2380            [
2381                "NaN or infinite",
2382                "Blank text",
2383                "Missing values",
2384                "No findings"
2385            ]
2386        );
2387        let rows = FindingsView {
2388            order: FindingOrder::Rows,
2389            ..FindingsView::default()
2390        };
2391        assert_eq!(
2392            titles(&rows),
2393            [
2394                "Blank text",
2395                "NaN or infinite",
2396                "Missing values",
2397                "No findings"
2398            ],
2399            "most rows first, Problems still above Notes"
2400        );
2401        let rate = FindingsView {
2402            order: FindingOrder::Rate,
2403            ..FindingsView::default()
2404        };
2405        assert_eq!(titles(&rate)[0], "NaN or infinite", "2 of 4 beats 9 of 100");
2406
2407        let region = FindingsView {
2408            column: Some("region".to_string()),
2409            ..FindingsView::default()
2410        };
2411        assert_eq!(titles(&region), ["Blank text", "Missing values"]);
2412        assert!(region.narrowed());
2413        let clean = FindingsView {
2414            column: Some("id".to_string()),
2415            ..FindingsView::default()
2416        };
2417        assert_eq!(titles(&clean), ["No findings"], "a clean column is clean");
2418        let missing = FindingsView {
2419            check: Some("Missing values"),
2420            ..FindingsView::default()
2421        };
2422        assert_eq!(titles(&missing), ["Missing values"]);
2423        assert_eq!(
2424            missing.selected(&report, 0).map(|finding| finding.title),
2425            Some("Missing values")
2426        );
2427        assert_eq!(
2428            check_choices(&report),
2429            [
2430                ("NaN or infinite", 1),
2431                ("Blank text", 1),
2432                ("Missing values", 1)
2433            ]
2434        );
2435        assert_eq!(
2436            column_choices(&report, &results)
2437                .into_iter()
2438                .map(|(_, count)| count)
2439                .collect::<Vec<_>>(),
2440            [1, 2, 1, 0]
2441        );
2442    }
2443
2444    /// A finding over several columns lists each column's own count, and says the
2445    /// rows with any of them are a range nobody counted, not their sum.
2446    #[test]
2447    fn grouped_findings_break_down_by_column_and_bound_the_union() {
2448        let results = results(
2449            vec![
2450                profile("a", DataType::String),
2451                profile("b", DataType::String),
2452            ],
2453            vec![
2454                observation(ObservationKind::Empty, "a", 6),
2455                observation(ObservationKind::Empty, "b", 6),
2456            ],
2457        );
2458        let report = build_report(&results);
2459        let finding = &report.findings[0];
2460        assert!(finding.lists_columns());
2461        assert_eq!(
2462            finding.evidence_count(&results),
2463            None,
2464            "a union nobody counted"
2465        );
2466        let (headline, evidence) = describe(finding, &results);
2467        assert_eq!(
2468            headline,
2469            "6 of 100 rows (6.0%) empty strings in each column:"
2470        );
2471        assert_eq!(evidence[0], "a     6.0%  6 rows");
2472        assert_eq!(
2473            evidence.last().unwrap(),
2474            "Rows with any of them: 6 to 12, not counted"
2475        );
2476    }
2477
2478    /// A parseable-text finding's rows are the values that stop a cast, counted
2479    /// from the profile; with none, there is nothing to open, and it says why.
2480    #[test]
2481    fn parse_failures_are_the_evidence_of_text_that_parses() {
2482        let mut text = profile("amount", DataType::String);
2483        text.null_count = 4;
2484        text.integer_parse_count = Some(95);
2485        text.decimal_parse_count = Some(95);
2486        let mut results = results(
2487            vec![text],
2488            vec![observation(ObservationKind::ParseableText, "amount", 95)],
2489        );
2490        results.examples = vec![crate::data_quality::FindingExamples {
2491            kind: ObservationKind::ParseableText,
2492            column: "amount".to_string(),
2493            values: vec!["\"n/a\"".to_string()],
2494        }];
2495        let report = build_report(&results);
2496        let finding = &report.findings[0];
2497        assert_eq!(finding.check(), Some("Numbers as text"));
2498        assert_eq!(finding.failures(&results), Some(1));
2499        assert_eq!(finding.evidence_count(&results), Some(1));
2500        assert!(matches!(
2501            finding.evidence(&results),
2502            Ok(EvidenceRows::Matching(_))
2503        ));
2504        let (_, evidence) = describe(finding, &results);
2505        assert!(
2506            evidence.contains(&"1 do not parse, such as \"n/a\"".to_string()),
2507            "{evidence:?}"
2508        );
2509
2510        results.columns[0].integer_parse_count = Some(96);
2511        results.columns[0].decimal_parse_count = Some(96);
2512        results.observations[0].affected_rows = 96;
2513        let report = build_report(&results);
2514        let reason = report.findings[0].evidence(&results).unwrap_err();
2515        assert!(reason.contains("every value parses"), "{reason}");
2516    }
2517
2518    /// Duplicate rows open as a group, and the count is every row with a copy.
2519    #[test]
2520    fn duplicate_rows_open_every_row_with_a_copy() {
2521        let mut results = results(
2522            vec![profile("id", DataType::Int64)],
2523            vec![observation(
2524                ObservationKind::DuplicateRows,
2525                "all columns",
2526                5,
2527            )],
2528        );
2529        results.identity = Some(crate::data_quality::IdentityProfile {
2530            duplicate_groups: 2,
2531            extra_rows: 3,
2532            rows_involved: 5,
2533            evaluated_rows: 100,
2534            precision: QualityPrecision::Exact,
2535            examples: vec![crate::data_quality::DuplicateExample {
2536                copies: 3,
2537                values: vec!["7".to_string()],
2538            }],
2539        });
2540        let report = build_report(&results);
2541        let finding = &report.findings[0];
2542        assert!(matches!(
2543            finding.evidence(&results),
2544            Ok(EvidenceRows::Duplicates)
2545        ));
2546        assert_eq!(finding.evidence_count(&results), Some(5));
2547        let (headline, evidence) = describe(finding, &results);
2548        assert_eq!(headline, "2 rows repeated; 3 extra copies");
2549        assert_eq!(evidence[0], "5 of 100 rows (5.0%) have a copy");
2550        assert_eq!(evidence[2], format!("{}3  7", crate::glyphs::get().times));
2551    }
2552}