Skip to main content

datui_lib/
data_quality.rs

1use crate::statistics::collect_lazy;
2use color_eyre::Result;
3use color_eyre::eyre::Report;
4use polars::chunked_array::cast::CastOptions;
5use polars::prelude::*;
6use std::collections::BTreeMap;
7use std::sync::Arc;
8
9// A dataset-grain sample is spread across the whole scope (see `statistics::analysis_rows`).
10const DEFAULT_SAMPLE_ROWS: usize = 10_000;
11const DEFAULT_CHUNK_ROWS: usize = 1_000_000;
12const QUALITY_WINDOW_START: &str = "__datui_quality_window_start";
13pub const QUALITY_SOURCE_FILE_COLUMN: &str = "__datui_quality_source_file";
14/// How nearly unique a column's values must be before its repeats are worth naming.
15///
16/// A key that is not quite one is the interesting case: an id that repeats twice in a
17/// million rows is a fact about the data, while a category that repeats constantly is
18/// just a category. The line has to fall somewhere, and 95% puts it where a column is
19/// clearly meant to identify a row rather than to group them.
20pub const KEY_LIKE_UNIQUENESS: f64 = 0.95;
21/// Files named per drift observation, and values read from each of them. Both are the
22/// evidence, not the measurement: the counts above them cover every file.
23const MAX_EVIDENCE_FILES: usize = 20;
24const MAX_CONFLICT_EXAMPLES: usize = 5;
25/// Values kept per finding from the rows a run read, and groups of duplicate rows:
26/// enough to recognize the problem in the detail, which opens the rest.
27pub const MAX_FINDING_EXAMPLES: usize = 3;
28/// Window widths offered for time-window grain, in the order the plan cycles them.
29pub const QUALITY_WINDOW_WIDTHS: [&str; 4] = ["1h", "1d", "1w", "1mo"];
30
31#[derive(Debug, Clone, PartialEq, Eq, Default)]
32pub enum QualityScope {
33    #[default]
34    CurrentView,
35    WholeSource,
36    FirstRows(usize),
37    ViewRows {
38        start: usize,
39        end: usize,
40    },
41    SourceFiles(Vec<usize>),
42    SourcePartition {
43        column: String,
44        value: String,
45    },
46    SourceTimeRange {
47        column: String,
48        start: String,
49        end: String,
50    },
51}
52
53impl QualityScope {
54    pub fn label(&self) -> String {
55        match self {
56            Self::CurrentView => "current view".to_string(),
57            Self::WholeSource => "whole source".to_string(),
58            Self::FirstRows(rows) => format!(
59                "first {} rows of the view",
60                crate::numfmt::group_chrome(*rows)
61            ),
62            Self::ViewRows { start, end } => format!(
63                "view rows {}-{}",
64                crate::numfmt::group_chrome(*start),
65                crate::numfmt::group_chrome(*end)
66            ),
67            Self::SourceFiles(indices) => format!(
68                "source files {}",
69                indices
70                    .iter()
71                    .map(usize::to_string)
72                    .collect::<Vec<_>>()
73                    .join(",")
74            ),
75            Self::SourcePartition { column, value } => format!("source {column}={value}"),
76            Self::SourceTimeRange { column, start, end } => {
77                format!("source {column} {start}..{end}")
78            }
79        }
80    }
81
82    pub fn uses_source(&self) -> bool {
83        matches!(
84            self,
85            Self::WholeSource
86                | Self::SourceFiles(_)
87                | Self::SourcePartition { .. }
88                | Self::SourceTimeRange { .. }
89        )
90    }
91
92    pub fn command(&self) -> String {
93        match self {
94            Self::CurrentView => "view".to_string(),
95            Self::WholeSource => "source".to_string(),
96            Self::FirstRows(rows) => format!("rows 1..{rows}"),
97            Self::ViewRows { start, end } => format!("rows {start}..{end}"),
98            Self::SourceFiles(indices) => format!(
99                "files {}",
100                indices
101                    .iter()
102                    .map(usize::to_string)
103                    .collect::<Vec<_>>()
104                    .join(",")
105            ),
106            Self::SourcePartition { column, value } => format!("partition {column}={value}"),
107            Self::SourceTimeRange { column, start, end } => format!("time {column}={start}..{end}"),
108        }
109    }
110
111    pub fn parse_command(text: &str) -> Result<Self> {
112        let value = text.trim();
113        if value == "view" {
114            return Ok(Self::CurrentView);
115        }
116        if value == "source" {
117            return Ok(Self::WholeSource);
118        }
119        if let Some(range) = value.strip_prefix("rows ") {
120            let (start, end) = range
121                .split_once("..")
122                .ok_or_else(|| color_eyre::eyre::eyre!("use rows START..END"))?;
123            let start = start.parse::<usize>()?;
124            let end = end.parse::<usize>()?;
125            if start == 0 || end < start {
126                return Err(color_eyre::eyre::eyre!(
127                    "row range must be 1-based with END >= START"
128                ));
129            }
130            if start == 1 {
131                // The same rows as FirstRows(end); use the one spelling so the
132                // scope round-trips through the editor and keeps its cache entry.
133                return Ok(Self::FirstRows(end));
134            }
135            return Ok(Self::ViewRows { start, end });
136        }
137        if let Some(files) = value.strip_prefix("files ") {
138            let indices = files
139                .split(',')
140                .map(|part| part.trim().parse::<usize>())
141                .collect::<std::result::Result<Vec<_>, _>>()?;
142            if indices.is_empty() || indices.contains(&0) {
143                return Err(color_eyre::eyre::eyre!(
144                    "use 1-based file numbers, for example files 1,3"
145                ));
146            }
147            let mut indices = indices;
148            indices.sort_unstable();
149            indices.dedup();
150            return Ok(Self::SourceFiles(indices));
151        }
152        if let Some(partition) = value.strip_prefix("partition ") {
153            let (column, value) = partition
154                .split_once('=')
155                .ok_or_else(|| color_eyre::eyre::eyre!("use partition COLUMN=VALUE"))?;
156            if column.trim().is_empty() || value.trim().is_empty() {
157                return Err(color_eyre::eyre::eyre!(
158                    "partition column and value are required"
159                ));
160            }
161            return Ok(Self::SourcePartition {
162                column: column.trim().to_string(),
163                value: value.trim().to_string(),
164            });
165        }
166        if let Some(time) = value.strip_prefix("time ") {
167            let (column, range) = time
168                .split_once('=')
169                .ok_or_else(|| color_eyre::eyre::eyre!("use time COLUMN=START..END"))?;
170            let (start, end) = range
171                .split_once("..")
172                .ok_or_else(|| color_eyre::eyre::eyre!("use time COLUMN=START..END"))?;
173            let (start, end) = (start.trim(), end.trim());
174            if column.trim().is_empty()
175                || parse_scope_time(start).is_none()
176                || parse_scope_time(end).is_none()
177                || parse_scope_time(end) <= parse_scope_time(start)
178            {
179                return Err(color_eyre::eyre::eyre!(
180                    "time range needs a column and increasing ISO dates or UTC timestamps"
181                ));
182            }
183            return Ok(Self::SourceTimeRange {
184                column: column.trim().to_string(),
185                start: start.to_string(),
186                end: end.to_string(),
187            });
188        }
189        Err(color_eyre::eyre::eyre!(
190            "use view, source, rows, files, partition, or time"
191        ))
192    }
193}
194
195pub(crate) fn parse_scope_time(text: &str) -> Option<i64> {
196    chrono::DateTime::parse_from_rfc3339(text)
197        .ok()
198        .map(|value| value.timestamp_micros())
199        .or_else(|| {
200            chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d")
201                .ok()
202                .and_then(|value| value.and_hms_opt(0, 0, 0))
203                .map(|value| value.and_utc().timestamp_micros())
204        })
205}
206
207#[derive(Debug, Clone, Default)]
208pub struct QualitySourceContext {
209    pub file_names: Vec<String>,
210    pub file_starts: Vec<usize>,
211    pub row_index_column: String,
212    /// Per file, in the order of `file_names`, its group in `drift_groups`. Empty when
213    /// the dataset's files all agree with its schema, which is nearly all of them.
214    pub file_group: Vec<u32>,
215    /// The distinct ways this dataset's files differ from its schema, as the footers
216    /// found them. Group 0 is always "nothing missing".
217    pub drift_groups: Arc<Vec<crate::schema_union::DriftGroup>>,
218    /// Per file, the type it holds each of its unreadable columns in. Empty for a file
219    /// whose types all fit, which is why it is kept beside the groups rather than in
220    /// them: the type is the only way back to the values a conflict hides.
221    pub file_omitted: Vec<Vec<(PlSmallStr, DataType)>>,
222    /// Rows in the whole loaded source, which is what closes the last file's range.
223    pub dataset_rows: usize,
224    /// How many of the source's footers were read. Below the file count on a dataset
225    /// too large to read every footer, where a file nobody looked at is indistinguishable
226    /// from one missing nothing — so a count over the files is a floor, not a total.
227    pub footers_read: usize,
228    /// How to read a column at the type a file wrote it in, for the values a type
229    /// conflict hides. `None` for a dataset whose files agree, and for a run whose
230    /// budget did not promise the extra reads.
231    pub conflict_scan: Option<QualityConflictScan>,
232}
233
234impl QualitySourceContext {
235    /// What the file at `file` is missing. Group 0 for a file that agrees with the
236    /// dataset's schema, and for a dataset whose files were never grouped.
237    fn group_of_file(&self, file: usize) -> Option<&crate::schema_union::DriftGroup> {
238        let group = *self.file_group.get(file)? as usize;
239        self.drift_groups.get(group)
240    }
241
242    /// The files this dataset is missing something from, by 1-based inventory number,
243    /// paired with what each is missing. Only files that differ have an entry.
244    fn drifting_files(&self) -> impl Iterator<Item = (usize, &crate::schema_union::DriftGroup)> {
245        (0..self.file_names.len()).filter_map(move |file| {
246            let group = self.group_of_file(file)?;
247            (!group.is_empty()).then_some((file, group))
248        })
249    }
250
251    /// Rows the file at `file` holds, from its footer.
252    fn file_rows(&self, file: usize) -> usize {
253        let Some(start) = self.file_starts.get(file) else {
254            return 0;
255        };
256        self.file_starts
257            .get(file + 1)
258            .copied()
259            .unwrap_or(self.dataset_rows)
260            .saturating_sub(*start)
261    }
262
263    /// The type the file at `file` holds `column` in, when that is not the type the
264    /// scan reads it as.
265    fn stored_type(&self, file: usize, column: &str) -> Option<&DataType> {
266        self.file_omitted
267            .get(file)?
268            .iter()
269            .find(|(name, _)| name.as_str() == column)
270            .map(|(_, dtype)| dtype)
271    }
272}
273
274/// Prepare the loaded source in the worker, keeping only a provenance index
275/// and replacing binary payloads before any value collection.
276pub fn prepare_source_quality_scan(
277    lf: LazyFrame,
278    source: Option<&QualitySourceContext>,
279) -> Result<LazyFrame> {
280    let schema = lf.clone().collect_schema()?;
281    let expressions = schema
282        .iter()
283        .filter_map(|(name, dtype)| {
284            let column = name.as_str();
285            if column == crate::schema_union::DRIFT_COLUMN
286                && !source.is_some_and(|context| context.row_index_column == column)
287            {
288                return None;
289            }
290            Some(if matches!(dtype, DataType::Binary) {
291                lit(crate::widgets::datatable::binary_stub()).alias(column)
292            } else {
293                col(column)
294            })
295        })
296        .collect::<Vec<_>>();
297    let lf = lf.select(expressions);
298    Ok(
299        if source.is_some_and(|context| context.row_index_column == "__datui_quality_row") {
300            lf.with_row_index("__datui_quality_row", None)
301        } else {
302            lf
303        },
304    )
305}
306
307/// The rows of one partition value, a list of them (`2019,2021`), or an inclusive
308/// range (`2020..2022`). Each value is read as the column's own type, so years and
309/// dates compare as numbers and dates, `1.5` is a `Decimal(10, 2)` column's `1.50`,
310/// and the predicate stays one a file's statistics can answer; `∅` is the null
311/// partition. A value that does not read as the type says so.
312fn partition_predicate(column: &str, value: &str, schema: &Schema) -> Result<Expr> {
313    let dtype = schema
314        .get(column)
315        .ok_or_else(|| color_eyre::eyre::eyre!("partition column {column:?} is unavailable"))?;
316    let read = |text: &str| {
317        crate::typed_value::parse(text, dtype)
318            .map(lit)
319            .map_err(|why| color_eyre::eyre::eyre!("{column}: {why}"))
320    };
321    if let Some((start, end)) = value.split_once("..") {
322        let (start, end) = (start.trim(), end.trim());
323        if start.is_empty() || end.is_empty() {
324            return Err(color_eyre::eyre::eyre!(
325                "a partition range needs both ends, for example year=2020..2022"
326            ));
327        }
328        return Ok(col(column)
329            .gt_eq(read(start)?)
330            .and(col(column).lt_eq(read(end)?)));
331    }
332    value
333        .split(',')
334        .map(str::trim)
335        .filter(|value| !value.is_empty())
336        .map(|value| {
337            Ok(if value == "∅" {
338                col(column).is_null()
339            } else {
340                col(column).eq(read(value)?)
341            })
342        })
343        .reduce(|all, one| Ok(all?.or(one?)))
344        .ok_or_else(|| color_eyre::eyre::eyre!("name at least one partition value"))?
345}
346
347pub fn apply_quality_scope(
348    lf: LazyFrame,
349    scope: &QualityScope,
350    source: Option<&QualitySourceContext>,
351) -> Result<LazyFrame> {
352    match scope {
353        QualityScope::CurrentView | QualityScope::WholeSource => Ok(lf),
354        QualityScope::FirstRows(rows) => Ok(lf.slice(0, (*rows).min(u32::MAX as usize) as u32)),
355        QualityScope::ViewRows { start, end } => {
356            if *start == 0 || end < start {
357                return Err(color_eyre::eyre::eyre!("invalid 1-based view row range"));
358            }
359            let offset = i64::try_from(start - 1)?;
360            let length = end
361                .saturating_sub(*start)
362                .saturating_add(1)
363                .min(u32::MAX as usize) as u32;
364            Ok(lf.slice(offset, length))
365        }
366        QualityScope::SourceFiles(indices) => {
367            let source = source
368                .ok_or_else(|| color_eyre::eyre::eyre!("source-file positions are unavailable"))?;
369            let mut predicate: Option<Expr> = None;
370            for index in indices {
371                let file = index
372                    .checked_sub(1)
373                    .ok_or_else(|| color_eyre::eyre::eyre!("source file numbers start at 1"))?;
374                let start = *source.file_starts.get(file).ok_or_else(|| {
375                    color_eyre::eyre::eyre!("source file #{index} is unavailable")
376                })?;
377                let start = u32::try_from(start)?;
378                let mut range = col(&source.row_index_column).gt_eq(lit(start));
379                if let Some(end) = source.file_starts.get(*index) {
380                    range = range.and(col(&source.row_index_column).lt(lit(u32::try_from(*end)?)));
381                }
382                predicate = Some(match predicate {
383                    Some(previous) => previous.or(range),
384                    None => range,
385                });
386            }
387            Ok(lf.filter(
388                predicate.ok_or_else(|| color_eyre::eyre::eyre!("select at least one file"))?,
389            ))
390        }
391        QualityScope::SourcePartition { column, value } => {
392            let schema = lf.clone().collect_schema()?;
393            if !schema.contains(column.as_str()) {
394                return Err(color_eyre::eyre::eyre!(
395                    "partition column {column:?} is unavailable"
396                ));
397            }
398            Ok(lf.filter(partition_predicate(column, value, &schema)?))
399        }
400        QualityScope::SourceTimeRange { column, start, end } => {
401            let schema = lf.clone().collect_schema()?;
402            let dtype = schema
403                .get(column.as_str())
404                .ok_or_else(|| color_eyre::eyre::eyre!("time column {column:?} is unavailable"))?;
405            if !matches!(dtype, DataType::Date | DataType::Datetime(..)) {
406                return Err(color_eyre::eyre::eyre!(
407                    "{column:?} is not a date or datetime column"
408                ));
409            }
410            let start = parse_scope_time(start)
411                .ok_or_else(|| color_eyre::eyre::eyre!("invalid start time"))?;
412            let end =
413                parse_scope_time(end).ok_or_else(|| color_eyre::eyre::eyre!("invalid end time"))?;
414            if end <= start {
415                return Err(color_eyre::eyre::eyre!("time end must be after start"));
416            }
417            let value = col(column).cast(DataType::Datetime(TimeUnit::Microseconds, None));
418            Ok(lf.filter(
419                value
420                    .clone()
421                    .gt_eq(lit(start).cast(DataType::Datetime(TimeUnit::Microseconds, None)))
422                    .and(value.lt(lit(end).cast(DataType::Datetime(TimeUnit::Microseconds, None)))),
423            ))
424        }
425    }
426}
427
428#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
429pub enum QualityPage {
430    /// Everything a run needs, staged until Enter runs it. Not a tab: a report's
431    /// pages are tabs, and Setup is where a report comes from.
432    #[default]
433    Setup,
434    Overview,
435    Columns,
436    Segments,
437    Trends,
438    Detail,
439    /// One segment's columns beside the segment it is compared with.
440    SegmentDetail,
441    /// Each interval in each segment: the time between two roles.
442    Intervals,
443    /// One interval in one segment, every count it took and out of what.
444    IntervalDetail,
445    TimeRoles,
446    /// Which starts and ends the intervals are, chosen from the assigned roles.
447    IntervalPairs,
448    /// One bar of one Trends line: the segments it pools, what they hold and how
449    /// much of them was read.
450    TrendDetail,
451    /// The expected windows with no rows to show, by why.
452    Gaps,
453    /// Which time windows rows are expected in, edited from Setup.
454    ExpectedWindows,
455    /// What each column must hold, declared: the key and each column's rules.
456    Intent,
457}
458
459impl QualityPage {
460    /// The report's tabs, in the order ←→ walk them. A column's detail sits under
461    /// Columns, a segment's under Segments and an interval's under Intervals.
462    pub const TABS: [Self; 5] = [
463        Self::Overview,
464        Self::Columns,
465        Self::Segments,
466        Self::Trends,
467        Self::Intervals,
468    ];
469
470    pub fn tab(self) -> Self {
471        match self {
472            Self::Detail => Self::Columns,
473            Self::SegmentDetail => Self::Segments,
474            Self::IntervalDetail => Self::Intervals,
475            Self::TrendDetail | Self::Gaps => Self::Trends,
476            Self::TimeRoles | Self::IntervalPairs | Self::ExpectedWindows | Self::Intent => {
477                Self::Setup
478            }
479            page => page,
480        }
481    }
482
483    /// Setup and its editors, which stage a run rather than show one.
484    pub fn is_setup(self) -> bool {
485        self.tab() == Self::Setup
486    }
487
488    pub fn title(self) -> &'static str {
489        match self.tab() {
490            Self::Overview => "Overview",
491            Self::Columns => "Columns",
492            Self::Segments => "Segments",
493            Self::Trends => "Trends",
494            Self::Intervals => "Intervals",
495            _ => "Setup",
496        }
497    }
498}
499
500/// What an empty page is missing, which Enter opens in Setup.
501#[derive(Debug, Clone, Copy, PartialEq, Eq)]
502pub enum QualitySetup {
503    Grain,
504    TimeRoles,
505    Intervals,
506}
507
508impl QualitySetup {
509    pub fn label(self) -> &'static str {
510        match self {
511            Self::Grain => "Set Grain",
512            Self::TimeRoles => "Time Roles",
513            Self::Intervals => "Intervals",
514        }
515    }
516}
517
518/// Whether the Trends page can draw a column's measure across segments: that
519/// needs segments in an order, and more than one of them.
520pub fn shows_trend(plan: &DataQualityPlan, results: &DataQualityResults) -> bool {
521    matches!(
522        plan.grain,
523        QualityGrain::RowChunks(_) | QualityGrain::TimeWindows { .. } | QualityGrain::Partition(_)
524    ) && results.segments.len() + results.unsampled_segments.len() > 1
525}
526
527/// The plan setting a result page needs before it has anything to show, if any.
528/// Intervals need time roles, and ask only when there are dates to assign.
529pub fn page_setup(
530    page: QualityPage,
531    plan: &DataQualityPlan,
532    results: Option<&DataQualityResults>,
533    has_time_columns: bool,
534) -> Option<QualitySetup> {
535    let results = results?;
536    match page {
537        QualityPage::Segments if plan.grain == QualityGrain::Dataset => Some(QualitySetup::Grain),
538        QualityPage::Trends if !shows_trend(plan, results) => Some(QualitySetup::Grain),
539        // Roles that make no interval want a pair chosen; otherwise, roles. Pairs
540        // that measured nothing (metadata only, text with no format) are not
541        // fixed by either, and the page says what is.
542        QualityPage::Intervals if results.temporal.is_empty() && has_time_columns => {
543            if plan.candidate_pairs().is_empty() {
544                Some(QualitySetup::TimeRoles)
545            } else if plan.interval_pairs().is_empty() {
546                Some(QualitySetup::Intervals)
547            } else {
548                None
549            }
550        }
551        _ => None,
552    }
553}
554
555#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
556pub enum QualityCompute {
557    Metadata,
558    #[default]
559    Sample,
560    Full,
561}
562
563impl QualityCompute {
564    pub fn label(self) -> &'static str {
565        match self {
566            Self::Metadata => "metadata",
567            Self::Sample => "sample",
568            Self::Full => "full",
569        }
570    }
571}
572
573#[derive(Debug, Clone, PartialEq, Eq, Default)]
574pub enum QualityGrain {
575    #[default]
576    Dataset,
577    File,
578    Partition(String),
579    RowChunks(usize),
580    TimeWindows {
581        column: String,
582        every: String,
583    },
584}
585
586impl QualityGrain {
587    /// How the rows are split, in words: "by day of date", "by year".
588    pub fn label(&self) -> String {
589        match self {
590            Self::Dataset => "whole dataset".to_string(),
591            Self::File => "by file".to_string(),
592            Self::Partition(column) => format!("by {column}"),
593            Self::RowChunks(rows) => {
594                format!("in chunks of {} rows", crate::numfmt::group_chrome(*rows))
595            }
596            Self::TimeWindows { column, every } => {
597                let unit = match every.as_str() {
598                    "1h" => "hour",
599                    "1d" => "day",
600                    "1w" => "week",
601                    "1mo" => "month",
602                    other => other,
603                };
604                // A column named for its unit would read "by day of day".
605                if column.eq_ignore_ascii_case(unit) {
606                    format!("by {unit} of the {column} column")
607                } else {
608                    format!("by {unit} of {column}")
609                }
610            }
611        }
612    }
613}
614
615#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
616pub enum QualityComparison {
617    #[default]
618    None,
619    Previous,
620    Baseline,
621}
622
623impl QualityComparison {
624    /// The comparison as a plan choice says it.
625    pub fn choice_label(self) -> &'static str {
626        match self {
627            Self::None => "none",
628            Self::Previous => "the segment before",
629            Self::Baseline => "a baseline segment (the first, or b on Segments)",
630        }
631    }
632
633    pub fn label(self) -> &'static str {
634        match self {
635            Self::None => "none",
636            Self::Previous => "previous",
637            Self::Baseline => "baseline (first)",
638        }
639    }
640}
641
642#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
643pub enum TemporalRole {
644    Event,
645    Effective,
646    PeriodEnd,
647    Created,
648    Published,
649    Received,
650    Processed,
651    ValidFrom,
652    ValidTo,
653}
654
655impl TemporalRole {
656    pub const ALL: [Self; 9] = [
657        Self::Event,
658        Self::Effective,
659        Self::PeriodEnd,
660        Self::Created,
661        Self::Published,
662        Self::Received,
663        Self::Processed,
664        Self::ValidFrom,
665        Self::ValidTo,
666    ];
667
668    pub fn label(self) -> &'static str {
669        match self {
670            Self::Event => "event",
671            Self::Effective => "effective/as-of",
672            Self::PeriodEnd => "period end",
673            Self::Created => "created",
674            Self::Published => "published",
675            Self::Received => "received",
676            Self::Processed => "processed",
677            Self::ValidFrom => "valid from",
678            Self::ValidTo => "valid to",
679        }
680    }
681}
682
683#[derive(Debug, Clone, PartialEq, Eq)]
684pub struct TemporalRoleAssignment {
685    pub role: TemporalRole,
686    pub column: String,
687    pub timezone: Option<String>,
688}
689
690/// The intervals a run measures when none are chosen, start role to end role: the
691/// pairs whose order the roles themselves state. Any other start and end is a
692/// choice under Intervals in Setup; a role in no interval measures nothing, and
693/// Setup says so before a run.
694pub const INTERVAL_PAIRS: [(TemporalRole, TemporalRole); 7] = [
695    (TemporalRole::Event, TemporalRole::Published),
696    (TemporalRole::Event, TemporalRole::Received),
697    (TemporalRole::PeriodEnd, TemporalRole::Published),
698    (TemporalRole::Published, TemporalRole::Received),
699    (TemporalRole::Received, TemporalRole::Processed),
700    (TemporalRole::Event, TemporalRole::Processed),
701    (TemporalRole::ValidFrom, TemporalRole::ValidTo),
702];
703
704/// `event to received`.
705pub fn interval_label((start, end): (TemporalRole, TemporalRole)) -> String {
706    format!("{} to {}", start.label(), end.label())
707}
708
709/// Which time puts an interval in a window, when the grain is time windows: the
710/// grain's own column, or the interval's start or end. By the end, an interval is
711/// counted on the day it finished rather than the day it began.
712#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default)]
713pub enum IntervalClock {
714    #[default]
715    Grain,
716    Start,
717    End,
718}
719
720impl IntervalClock {
721    pub const ALL: [Self; 3] = [Self::Grain, Self::Start, Self::End];
722
723    pub fn label(self) -> &'static str {
724        match self {
725            Self::Grain => "the grain's column",
726            Self::Start => "each interval's start",
727            Self::End => "each interval's end",
728        }
729    }
730}
731
732/// Whether text read as time is a date or a date with a time of day.
733#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
734pub enum TimeKind {
735    Date,
736    Datetime,
737}
738
739impl TimeKind {
740    pub fn label(self) -> &'static str {
741        match self {
742            Self::Date => "date",
743            Self::Datetime => "datetime",
744        }
745    }
746}
747
748/// The formats Setup offers for reading text as time, the unambiguous ones first.
749/// Named formats rather than inference: a run reads every row the same way, and a
750/// value the format does not read is counted, not guessed at.
751pub const TIME_FORMATS: [(TimeKind, &str); 16] = [
752    (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S"),
753    (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S"),
754    (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S%.f"),
755    (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S%.f"),
756    // ISO 8601 with an offset: `%#z` takes `Z`, `+05:00`, `-0500` and `+05`, and
757    // the values are read as instants in UTC.
758    (TimeKind::Datetime, "%Y-%m-%dT%H:%M:%S%.f%#z"),
759    (TimeKind::Datetime, "%Y-%m-%d %H:%M:%S%.f%#z"),
760    (TimeKind::Datetime, "%Y-%m-%d %H:%M"),
761    (TimeKind::Date, "%Y-%m-%d"),
762    (TimeKind::Date, "%Y%m%d"),
763    (TimeKind::Datetime, "%m/%d/%Y %H:%M:%S"),
764    (TimeKind::Datetime, "%m/%d/%Y %I:%M:%S %p"),
765    (TimeKind::Datetime, "%d/%m/%Y %H:%M:%S"),
766    (TimeKind::Datetime, "%d.%m.%Y %H:%M:%S"),
767    (TimeKind::Date, "%m/%d/%Y"),
768    (TimeKind::Date, "%d/%m/%Y"),
769    (TimeKind::Date, "%d.%m.%Y"),
770];
771
772/// A text column read as a date or datetime for one study. Grain and time roles see
773/// the parsed value; every other check sees the text as stored, so a column's own
774/// findings keep their physical meaning. A value the format does not read is counted
775/// as unparsed, never folded into the column's missing values.
776#[derive(Debug, Clone, PartialEq, Eq, Hash)]
777pub struct TimeInterpretation {
778    pub column: String,
779    pub kind: TimeKind,
780    /// A strftime format, as Polars' `str.to_datetime` takes it. With an offset
781    /// (`%z`), the values are instants in UTC; without one, times with no zone.
782    pub format: String,
783}
784
785impl TimeInterpretation {
786    /// Whether the format reads an offset, so its values are instants in UTC
787    /// rather than times with no zone.
788    pub fn zoned(&self) -> bool {
789        self.format.contains('z')
790    }
791
792    /// `datetime %Y-%m-%d %H:%M:%S`.
793    pub fn label(&self) -> String {
794        format!("{} {}", self.kind.label(), self.format)
795    }
796
797    /// The column's values as time: null where the format does not read the text.
798    pub fn expr(&self) -> Expr {
799        let options = StrptimeOptions {
800            format: Some(PlSmallStr::from(self.format.as_str())),
801            strict: false,
802            exact: true,
803            cache: true,
804        };
805        // A categorical column holds codes; its values are read as the text they name.
806        let text = col(self.column.as_str()).cast(DataType::String).str();
807        match self.kind {
808            TimeKind::Date => text.to_date(options),
809            TimeKind::Datetime => text.to_datetime(
810                Some(TimeUnit::Microseconds),
811                None,
812                options,
813                lit(PlSmallStr::from_static("raise")),
814            ),
815        }
816    }
817
818    /// Rows holding text the format does not read.
819    pub fn unparsed(&self) -> Expr {
820        col(self.column.as_str())
821            .is_not_null()
822            .and(self.expr().is_null())
823    }
824
825    /// Whether the format reads `value`, the way a run will: for the examples Setup
826    /// shows beside each format, from rows already on screen.
827    pub fn reads(&self, value: &str) -> bool {
828        match self.kind {
829            TimeKind::Date => chrono::NaiveDate::parse_from_str(value, &self.format).is_ok(),
830            // An offset format needs the offset: without one there is no instant.
831            TimeKind::Datetime if self.zoned() => {
832                chrono::DateTime::parse_from_str(value, &self.format).is_ok()
833            }
834            TimeKind::Datetime => {
835                chrono::NaiveDateTime::parse_from_str(value, &self.format).is_ok()
836            }
837        }
838    }
839}
840
841/// What a Data Quality run is doing now. The worker names each stage as it enters
842/// it, and the progress view shows the latest.
843#[derive(Debug, Clone, Copy, PartialEq, Eq)]
844pub enum QualityStage {
845    Preparing,
846    CopyingSource,
847    ReusingSample,
848    ReadingSample,
849    CountingRows,
850    CountingSegments,
851    ProfilingColumns,
852    CheckingDuplicates,
853    CheckingKey,
854    CheckingSpellings,
855    ReadingConflicts,
856    ProfilingSegments,
857    ComputingIntervals,
858    CheckingSharedNulls,
859    /// An audio file's samples, read whole for clipping, runs of zeros and DC offset.
860    CheckingSignal,
861    Assembling,
862}
863
864impl QualityStage {
865    pub fn label(self) -> &'static str {
866        match self {
867            Self::Preparing => "Preparing the plan",
868            Self::CopyingSource => "Copying the source locally",
869            Self::ReusingSample => "Reusing the retained sample",
870            Self::ReadingSample => "Reading the sample",
871            Self::CountingRows => "Counting rows",
872            Self::CountingSegments => "Counting segment rows",
873            Self::ProfilingColumns => "Profiling columns",
874            Self::CheckingDuplicates => "Checking duplicate rows",
875            Self::CheckingKey => "Checking the declared key",
876            Self::CheckingSpellings => "Checking category spellings",
877            Self::ReadingConflicts => "Reading conflicting values",
878            Self::ProfilingSegments => "Profiling segments",
879            Self::ComputingIntervals => "Computing intervals",
880            Self::CheckingSharedNulls => "Checking columns missing together",
881            Self::CheckingSignal => "Checking the signal",
882            Self::Assembling => "Assembling the report",
883        }
884    }
885}
886
887/// A stage, whether it reads the source or works on rows already read, and whether
888/// a cancel stops it partway.
889#[derive(Debug, Clone, Copy, PartialEq, Eq)]
890pub struct QualityPhase {
891    pub stage: QualityStage,
892    pub reads_source: bool,
893    /// A cancel ends this stage within a batch. A stage that is one collect Polars
894    /// cannot watch runs to its end, and the screen says so while it does.
895    pub interruptible: bool,
896}
897
898/// What a run's reads of the source were seen to do, counted as the rows went by.
899/// Bytes and requests are not counted: a Polars scan does not report them.
900#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
901pub struct ObservedReads {
902    /// Stages that read the source.
903    pub reads: usize,
904    /// Of those, the ones whose rows were counted as they came.
905    pub counted: usize,
906    /// Rows the counted reads passed through from the scope, over every pass.
907    pub rows: usize,
908    /// The local copy a full scan's passes read instead of the source, when they did.
909    pub copy: Option<CopyRead>,
910}
911
912/// A local copy of a remote source that a full scan's passes read.
913#[derive(Debug, Clone, Copy, PartialEq, Eq)]
914pub struct CopyRead {
915    pub bytes: u64,
916    pub objects: usize,
917    /// This run fetched it; otherwise an earlier run did and this one reused it.
918    pub fetched: bool,
919}
920
921/// A run's line to the screen: its stages as it enters them, the rows its reads
922/// have seen, and a stop the run checks between stages and its reads check between
923/// batches.
924#[derive(Clone, Default)]
925pub struct QualityWatch {
926    read: crate::sampling::ReadWatch,
927    report: Option<Arc<dyn Fn(QualityPhase) + Send + Sync>>,
928    /// The stage under way, and what the stages before it read.
929    last: Arc<std::sync::Mutex<(Option<QualityPhase>, ObservedReads)>>,
930    /// Set once the scope reads a local copy: its passes then read no source.
931    copy: Arc<std::sync::OnceLock<CopyRead>>,
932}
933
934impl std::fmt::Debug for QualityWatch {
935    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
936        f.debug_struct("QualityWatch")
937            .field("read", &self.read)
938            .finish_non_exhaustive()
939    }
940}
941
942impl QualityWatch {
943    /// A watch that hands each new stage to `report`.
944    pub fn new(report: impl Fn(QualityPhase) + Send + Sync + 'static) -> Self {
945        Self {
946            report: Some(Arc::new(report)),
947            ..Self::default()
948        }
949    }
950
951    pub fn cancel(&self) {
952        self.read.stop();
953    }
954
955    pub fn cancelled(&self) -> bool {
956        self.read.stopped()
957    }
958
959    /// The reads' side: the stop, and the rows the stage under way has seen.
960    pub fn read(&self) -> &crate::sampling::ReadWatch {
961        &self.read
962    }
963
964    /// What the run's reads were seen to do, the one under way included.
965    pub fn observed(&self) -> ObservedReads {
966        let Ok(last) = self.last.lock() else {
967            return ObservedReads::default();
968        };
969        let (phase, mut observed) = *last;
970        if phase.is_some_and(|phase| phase.reads_source) {
971            observed.reads += 1;
972            if let Some(rows) = self.read.rows_seen() {
973                observed.counted += 1;
974                observed.rows += rows;
975            }
976        }
977        observed.copy = self.copy.get().copied();
978        observed
979    }
980
981    /// The scope is read from a local copy from here on.
982    pub(crate) fn use_copy(&self, copy: CopyRead) {
983        let _ = self.copy.set(copy);
984    }
985
986    /// Whether a pass over the scope reads the source: not once it reads a copy.
987    fn scope_reads(&self, reads: bool) -> bool {
988        reads && self.copy.get().is_none()
989    }
990
991    /// `lf`, watched: every batch that reaches its top is counted, and once the run
992    /// is cancelled the next one fails the query. On the streaming engine a collect
993    /// of it stops within a batch rather than at its end; in memory the scope
994    /// arrives as one batch, after the read. Projections and filters pass through
995    /// to the scan as they would without it.
996    fn watched(&self, lf: &LazyFrame) -> LazyFrame {
997        let read = self.read.clone();
998        lf.clone().map(
999            move |df: DataFrame| {
1000                if read.stopped() {
1001                    return Err(PolarsError::ComputeError(crate::sampling::CANCELLED.into()));
1002                }
1003                read.saw(df.height());
1004                Ok(df)
1005            },
1006            OptFlags::PROJECTION_PUSHDOWN | OptFlags::PREDICATE_PUSHDOWN | OptFlags::STREAMING,
1007            None,
1008            Some("quality watch"),
1009        )
1010    }
1011
1012    /// Enter `stage`. Said once however often it is entered, and refused once the run
1013    /// is cancelled: between stages is where a run stops. Leaving a stage that read
1014    /// the source adds the rows it counted to what was observed.
1015    pub(crate) fn stage(
1016        &self,
1017        stage: QualityStage,
1018        reads_source: bool,
1019        interruptible: bool,
1020    ) -> Result<()> {
1021        self.read.check()?;
1022        let phase = QualityPhase {
1023            stage,
1024            reads_source,
1025            interruptible,
1026        };
1027        let mut last = self
1028            .last
1029            .lock()
1030            .map_err(|_| Report::msg("quality progress lock failed"))?;
1031        let (previous, observed) = &mut *last;
1032        if *previous != Some(phase) {
1033            // Rows on screen are the stage's own, so each read counts from zero.
1034            let seen = self.read.restart();
1035            if previous.is_some_and(|phase| phase.reads_source) {
1036                observed.reads += 1;
1037                if let Some(rows) = seen {
1038                    observed.counted += 1;
1039                    observed.rows += rows;
1040                }
1041            }
1042            *previous = Some(phase);
1043            if let Some(report) = &self.report {
1044                report(phase);
1045            }
1046        }
1047        Ok(())
1048    }
1049
1050    /// A watched collect's error, or the cancel that caused it: stopped partway, it
1051    /// fails as a stopped sampler does.
1052    fn failed(&self, error: impl Into<Report>) -> Report {
1053        if self.cancelled() {
1054            Report::msg(crate::sampling::CANCELLED)
1055        } else {
1056            error.into()
1057        }
1058    }
1059}
1060
1061#[derive(Debug, Clone, PartialEq, Eq)]
1062pub struct DataQualityPlan {
1063    pub scope: QualityScope,
1064    pub compute: QualityCompute,
1065    /// How a dataset-grain sample picks its rows, from the shared analysis sample.
1066    pub method: crate::sampling::SampleMethod,
1067    /// Rows a dataset-grain sample keeps: the shared analysis sample's size.
1068    pub dataset_rows: usize,
1069    pub sample_seed: u64,
1070    pub grain: QualityGrain,
1071    pub comparison: QualityComparison,
1072    pub baseline_segment: Option<String>,
1073    pub temporal_roles: Vec<TemporalRoleAssignment>,
1074    /// The intervals chosen under Intervals, start role to end role. `None` until
1075    /// one is chosen: the suggested pairs the assigned roles make.
1076    pub intervals: Option<Vec<(TemporalRole, TemporalRole)>>,
1077    /// Which time puts an interval in a time window.
1078    pub interval_clock: IntervalClock,
1079    pub latency_threshold_seconds: Option<i64>,
1080    /// Text columns read as time for this study, by grain and roles only.
1081    pub time_formats: Vec<TimeInterpretation>,
1082    /// The time windows rows are expected in, when stated: what makes a window
1083    /// with no rows a gap. Read from the segments a run counted, so it changes what
1084    /// the report says, never what a run reads.
1085    pub expected: Option<ExpectedWindows>,
1086    /// What the columns must hold, declared: the key and each column's rules. Read
1087    /// from the rows the run reads; it decides no rows, so it is the report's.
1088    pub intent: crate::quality_intent::DeclaredIntent,
1089}
1090
1091/// Which time windows a study expects rows in, as stated in Setup: every window of
1092/// the grain, or Monday to Friday's only, from one time and before another. Unset,
1093/// no window is called a gap: a quiet weekend is not a defect unless someone says so.
1094#[derive(Debug, Clone, PartialEq, Eq, Default)]
1095pub struct ExpectedWindows {
1096    /// Only Monday to Friday's hours or days are expected.
1097    pub weekdays: bool,
1098    /// The first expected time, as typed: a date or a UTC timestamp. `None` starts
1099    /// at the first window the run found.
1100    pub from: Option<String>,
1101    /// The time expected windows end before. `None` ends after the last window the
1102    /// run found.
1103    pub before: Option<String>,
1104}
1105
1106impl ExpectedWindows {
1107    /// Whether `every` is a width whose windows can fall on a weekend: an hour or a
1108    /// day. A week or a month always holds weekdays.
1109    pub fn weekdays_apply(every: &str) -> bool {
1110        matches!(every, "1h" | "1d")
1111    }
1112
1113    /// The cadence, in Setup's words: "every day", "weekdays".
1114    pub fn cadence_label(&self, every: &str) -> String {
1115        if self.weekdays && Self::weekdays_apply(every) {
1116            "weekdays".to_string()
1117        } else {
1118            let unit = match every {
1119                "1h" => "hour",
1120                "1d" => "day",
1121                "1w" => "week",
1122                "1mo" => "month",
1123                other => other,
1124            };
1125            format!("every {unit}")
1126        }
1127    }
1128
1129    /// The range, in Setup's words: "2024-01-01 to before 2024-04-01", or the windows
1130    /// found where a side is not stated.
1131    pub fn range_label(&self) -> String {
1132        match (self.from.as_deref(), self.before.as_deref()) {
1133            (None, None) => "first to last window found".to_string(),
1134            (Some(from), None) => format!("{from} to the last window found"),
1135            (None, Some(before)) => format!("first window found to before {before}"),
1136            (Some(from), Some(before)) => format!("{from} to before {before}"),
1137        }
1138    }
1139
1140    /// Why the typed range cannot be read, if it cannot.
1141    pub fn problem(&self) -> Option<String> {
1142        let read = |text: &Option<String>| match text.as_deref() {
1143            None => Ok(None),
1144            Some(text) => parse_scope_time(text)
1145                .map(Some)
1146                .ok_or_else(|| format!("{text} is not a date or UTC timestamp")),
1147        };
1148        match (read(&self.from), read(&self.before)) {
1149            (Err(problem), _) | (_, Err(problem)) => Some(problem),
1150            (Ok(Some(from)), Ok(Some(before))) if before <= from => {
1151                Some("Before must be after From".to_string())
1152            }
1153            _ => None,
1154        }
1155    }
1156
1157    /// The typed range in microseconds since the epoch, each side when stated and
1158    /// readable.
1159    pub fn bounds(&self) -> (Option<i64>, Option<i64>) {
1160        (
1161            self.from.as_deref().and_then(parse_scope_time),
1162            self.before.as_deref().and_then(parse_scope_time),
1163        )
1164    }
1165}
1166
1167impl Default for DataQualityPlan {
1168    fn default() -> Self {
1169        Self {
1170            scope: QualityScope::CurrentView,
1171            compute: QualityCompute::Sample,
1172            method: crate::sampling::SampleMethod::Spread,
1173            dataset_rows: DEFAULT_SAMPLE_ROWS,
1174            sample_seed: 42_891,
1175            grain: QualityGrain::Dataset,
1176            comparison: QualityComparison::None,
1177            baseline_segment: None,
1178            temporal_roles: Vec::new(),
1179            intervals: None,
1180            interval_clock: IntervalClock::Grain,
1181            latency_threshold_seconds: None,
1182            time_formats: Vec::new(),
1183            expected: None,
1184            intent: crate::quality_intent::DeclaredIntent::default(),
1185        }
1186    }
1187}
1188
1189impl DataQualityPlan {
1190    /// Only a full scan asks first. Every grain reads the shared sample and cuts it
1191    /// into segments, so no grain reads more than the sample says.
1192    pub fn requires_confirmation(&self) -> bool {
1193        self.compute == QualityCompute::Full
1194    }
1195
1196    pub fn comparison_label(&self) -> String {
1197        if self.comparison == QualityComparison::Baseline {
1198            self.baseline_segment
1199                .as_ref()
1200                .map(|label| format!("baseline: {label}"))
1201                .unwrap_or_else(|| self.comparison.label().to_string())
1202        } else {
1203            self.comparison.label().to_string()
1204        }
1205    }
1206
1207    /// The shared analysis sample this plan carries, as the Sample form shows it.
1208    pub fn sample(&self) -> crate::sampling::Sample {
1209        crate::sampling::Sample {
1210            scope: self.scope.clone(),
1211            method: self.method.clone(),
1212            rows: self.dataset_rows,
1213            seed: self.sample_seed,
1214        }
1215    }
1216
1217    /// Take `sample` as the rows this plan reads. Metadata-only stays metadata-only;
1218    /// otherwise every row is a full read and anything less a sampled one.
1219    ///
1220    /// Choosing equal rows per value of a column is choosing to look at that column's
1221    /// values side by side, and the grain is what does that. Taken only when the
1222    /// choice is new and the grain has not been set, so a grain chosen afterwards
1223    /// stays chosen.
1224    pub fn adopt_sample(&mut self, sample: &crate::sampling::Sample) {
1225        if self.scope != sample.scope {
1226            self.baseline_segment = None;
1227        }
1228        self.scope = sample.scope.clone();
1229        self.sample_seed = sample.seed;
1230        self.dataset_rows = sample.rows;
1231        if self.compute != QualityCompute::Metadata {
1232            self.compute = if sample.method == crate::sampling::SampleMethod::EveryRow {
1233                QualityCompute::Full
1234            } else {
1235                QualityCompute::Sample
1236            };
1237        }
1238        if let crate::sampling::SampleMethod::PerPartition { column } = &sample.method
1239            && self.method != sample.method
1240            && self.grain == QualityGrain::Dataset
1241        {
1242            self.grain = QualityGrain::Partition(column.clone());
1243            self.baseline_segment = None;
1244        }
1245        self.method = sample.method.clone();
1246    }
1247
1248    /// How `column` is read as time, when it is text read through a format.
1249    pub fn time_format(&self, column: &str) -> Option<&TimeInterpretation> {
1250        self.time_formats
1251            .iter()
1252            .find(|interpretation| interpretation.column == column)
1253    }
1254
1255    /// A column's values as time: parsed through its format when it has one, and as
1256    /// stored otherwise.
1257    pub fn time_value(&self, column: &str) -> Expr {
1258        self.time_format(column)
1259            .map(TimeInterpretation::expr)
1260            .unwrap_or_else(|| col(column))
1261    }
1262
1263    /// Whether `column` of `schema` can be read as time: a date or time type, or text
1264    /// with a format.
1265    pub fn reads_as_time(&self, column: &str, schema: &Schema) -> bool {
1266        self.time_format(column).is_some() || schema.get(column).is_some_and(DataType::is_temporal)
1267    }
1268
1269    /// The column a role is assigned to.
1270    pub fn role_column(&self, role: TemporalRole) -> Option<&str> {
1271        self.temporal_roles
1272            .iter()
1273            .find(|assignment| assignment.role == role)
1274            .map(|assignment| assignment.column.as_str())
1275    }
1276
1277    /// The intervals this plan measures: the chosen ones, or until one is chosen the
1278    /// suggested pairs; either way only those whose two roles are assigned.
1279    pub fn interval_pairs(&self) -> Vec<(TemporalRole, TemporalRole)> {
1280        let assigned = |(start, end): &(TemporalRole, TemporalRole)| {
1281            self.role_column(*start).is_some() && self.role_column(*end).is_some()
1282        };
1283        match &self.intervals {
1284            None => INTERVAL_PAIRS.into_iter().filter(assigned).collect(),
1285            Some(chosen) => chosen.iter().copied().filter(assigned).collect(),
1286        }
1287    }
1288
1289    /// Every start and end the assigned roles can make, the suggested pairs first,
1290    /// then the rest in role order: what Intervals in Setup lists.
1291    pub fn candidate_pairs(&self) -> Vec<(TemporalRole, TemporalRole)> {
1292        let roles = TemporalRole::ALL
1293            .into_iter()
1294            .filter(|role| self.role_column(*role).is_some())
1295            .collect::<Vec<_>>();
1296        let mut pairs = INTERVAL_PAIRS
1297            .into_iter()
1298            .filter(|(start, end)| roles.contains(start) && roles.contains(end))
1299            .collect::<Vec<_>>();
1300        for start in &roles {
1301            for end in &roles {
1302                if start != end && !pairs.contains(&(*start, *end)) {
1303                    pairs.push((*start, *end));
1304                }
1305            }
1306        }
1307        pairs
1308    }
1309
1310    /// Measure `pair`, or stop measuring it. The first choice makes the list
1311    /// explicit, starting from what was measured.
1312    pub fn toggle_interval(&mut self, pair: (TemporalRole, TemporalRole)) {
1313        let mut chosen = self.interval_pairs();
1314        match chosen.iter().position(|chosen| *chosen == pair) {
1315            Some(index) => {
1316                chosen.remove(index);
1317            }
1318            None => chosen.push(pair),
1319        }
1320        self.intervals = Some(chosen);
1321    }
1322
1323    /// Assigned roles that no measured interval uses: they measure nothing.
1324    pub fn unpaired_roles(&self) -> Vec<TemporalRole> {
1325        let pairs = self.interval_pairs();
1326        TemporalRole::ALL
1327            .into_iter()
1328            .filter(|role| {
1329                self.role_column(*role).is_some()
1330                    && !pairs
1331                        .iter()
1332                        .any(|(start, end)| start == role || end == role)
1333            })
1334            .collect()
1335    }
1336
1337    /// Whether the clock choice means anything: intervals cut into time windows.
1338    pub fn windows_intervals(&self) -> bool {
1339        matches!(self.grain, QualityGrain::TimeWindows { .. }) && !self.interval_pairs().is_empty()
1340    }
1341
1342    /// The grain an interval from `start` to `end` is cut by: the plan's, except
1343    /// that time windows go by the interval's own start or end when the clock says.
1344    pub fn interval_grain(&self, start: &str, end: &str) -> QualityGrain {
1345        match (&self.grain, self.interval_clock) {
1346            (QualityGrain::TimeWindows { every, .. }, IntervalClock::Start) => {
1347                QualityGrain::TimeWindows {
1348                    column: start.to_string(),
1349                    every: every.clone(),
1350                }
1351            }
1352            (QualityGrain::TimeWindows { every, .. }, IntervalClock::End) => {
1353                QualityGrain::TimeWindows {
1354                    column: end.to_string(),
1355                    every: every.clone(),
1356                }
1357            }
1358            (grain, _) => grain.clone(),
1359        }
1360    }
1361
1362    /// Whether `column` holds instants (a zoned type, or text read with an
1363    /// offset) rather than times with no zone; `None` when it is not read as time.
1364    pub fn zoned(&self, column: &str, schema: &Schema) -> Option<bool> {
1365        if let Some(format) = self.time_format(column) {
1366            return Some(format.zoned());
1367        }
1368        match schema.get(column)? {
1369            DataType::Datetime(_, zone) => Some(zone.is_some()),
1370            DataType::Date => Some(false),
1371            _ => None,
1372        }
1373    }
1374
1375    /// Whether `other` measures what this plan measures: they differ, if at all, in
1376    /// the windows they expect, which a report checks against the counts it already
1377    /// holds, or in what its segments are compared with, which is worked out from the
1378    /// segments it holds ([`DataQualityResults::compare_segments`]).
1379    pub fn same_measurement(&self, other: &Self) -> bool {
1380        let measured = |plan: &Self| Self {
1381            expected: None,
1382            comparison: QualityComparison::None,
1383            baseline_segment: None,
1384            ..plan.clone()
1385        };
1386        measured(self) == measured(other)
1387    }
1388
1389    /// Whether `other` compares segments differently from this plan.
1390    pub fn compares_differently(&self, other: &Self) -> bool {
1391        self.comparison != other.comparison || self.baseline_segment != other.baseline_segment
1392    }
1393
1394    /// The windows this plan expects rows in: only on a time-window grain.
1395    pub fn expected_windows(&self) -> Option<&ExpectedWindows> {
1396        matches!(self.grain, QualityGrain::TimeWindows { .. })
1397            .then_some(self.expected.as_ref())
1398            .flatten()
1399    }
1400
1401    /// The next coarser grain to offer when segments are thin: a day for an hour, a
1402    /// week for a day, a month for a week, and a larger row chunk. Partitions and files
1403    /// have none.
1404    pub fn coarser_grain(&self) -> Option<QualityGrain> {
1405        match &self.grain {
1406            QualityGrain::TimeWindows { column, every } => {
1407                let coarser = match every.as_str() {
1408                    "1h" => "1d",
1409                    "1d" => "1w",
1410                    "1w" => "1mo",
1411                    _ => return None,
1412                };
1413                Some(QualityGrain::TimeWindows {
1414                    column: column.clone(),
1415                    every: coarser.to_string(),
1416                })
1417            }
1418            QualityGrain::RowChunks(rows) if *rows < DEFAULT_CHUNK_ROWS => {
1419                Some(QualityGrain::RowChunks(DEFAULT_CHUNK_ROWS))
1420            }
1421            _ => None,
1422        }
1423    }
1424
1425    pub fn set_row_chunks(&mut self) {
1426        self.grain = QualityGrain::RowChunks(DEFAULT_CHUNK_ROWS);
1427    }
1428
1429    pub fn compact_summary(&self) -> String {
1430        format!(
1431            "scope {} -> grain {} -> compute {} -> compare {}",
1432            self.scope.label(),
1433            self.grain.label(),
1434            match self.compute {
1435                QualityCompute::Sample => format!(
1436                    "{} rows {}",
1437                    self.dataset_rows,
1438                    self.method.label().to_lowercase()
1439                ),
1440                other => other.label().to_string(),
1441            },
1442            self.comparison_label()
1443        )
1444    }
1445}
1446
1447#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1448pub enum QualityPrecision {
1449    Metadata,
1450    Sampled,
1451    Exact,
1452    Estimated,
1453}
1454
1455impl QualityPrecision {
1456    pub fn label(self) -> &'static str {
1457        match self {
1458            Self::Metadata => "metadata",
1459            Self::Sampled => "sampled",
1460            Self::Exact => "exact",
1461            Self::Estimated => "estimated",
1462        }
1463    }
1464}
1465
1466#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
1467pub enum QualityMetric {
1468    #[default]
1469    NullRate,
1470    EmptyRate,
1471    WhitespaceRate,
1472    NonFiniteRate,
1473    DistinctShare,
1474    IntegerParseShare,
1475    DecimalParseShare,
1476}
1477
1478impl QualityMetric {
1479    pub const ALL: [Self; 7] = [
1480        Self::NullRate,
1481        Self::EmptyRate,
1482        Self::WhitespaceRate,
1483        Self::NonFiniteRate,
1484        Self::DistinctShare,
1485        Self::IntegerParseShare,
1486        Self::DecimalParseShare,
1487    ];
1488
1489    pub fn label(self) -> &'static str {
1490        match self {
1491            Self::NullRate => "Null rate",
1492            Self::EmptyRate => "Empty rate",
1493            Self::WhitespaceRate => "Whitespace rate",
1494            Self::NonFiniteRate => "Non-finite rate",
1495            Self::DistinctShare => "Distinct share",
1496            Self::IntegerParseShare => "Integer parse share",
1497            Self::DecimalParseShare => "Decimal parse share",
1498        }
1499    }
1500
1501    /// The rows a rate is taken over: every row, or the rows with a value.
1502    pub fn denominator(self, column: &ColumnQualityProfile) -> usize {
1503        match self {
1504            Self::NullRate | Self::EmptyRate | Self::WhitespaceRate | Self::NonFiniteRate => {
1505                column.evaluated_rows
1506            }
1507            Self::DistinctShare | Self::IntegerParseShare | Self::DecimalParseShare => {
1508                column.non_null_rows()
1509            }
1510        }
1511    }
1512
1513    /// The measure's name in a table cell or a change: "nulls", "distinct".
1514    pub fn short_label(self) -> &'static str {
1515        match self {
1516            Self::NullRate => "nulls",
1517            Self::EmptyRate => "empty",
1518            Self::WhitespaceRate => "blank",
1519            Self::NonFiniteRate => "NaN/inf",
1520            Self::DistinctShare => "distinct",
1521            Self::IntegerParseShare => "integer parse",
1522            Self::DecimalParseShare => "decimal parse",
1523        }
1524    }
1525
1526    pub fn value(self, column: &ColumnQualityProfile) -> Option<f64> {
1527        let ratio = |numerator: usize, denominator: usize| {
1528            (denominator > 0).then(|| numerator as f64 / denominator as f64)
1529        };
1530        match self {
1531            Self::NullRate => ratio(column.null_count, column.evaluated_rows),
1532            Self::EmptyRate => ratio(column.empty_count?, column.evaluated_rows),
1533            Self::WhitespaceRate => ratio(column.whitespace_count?, column.evaluated_rows),
1534            Self::NonFiniteRate => ratio(
1535                column.nan_count?
1536                    + column.positive_infinity_count?
1537                    + column.negative_infinity_count?,
1538                column.evaluated_rows,
1539            ),
1540            Self::DistinctShare => ratio(column.distinct_count?, column.non_null_rows()),
1541            Self::IntegerParseShare => ratio(column.integer_parse_count?, column.non_null_rows()),
1542            Self::DecimalParseShare => ratio(column.decimal_parse_count?, column.non_null_rows()),
1543        }
1544    }
1545}
1546
1547#[derive(Debug, Clone)]
1548pub struct ColumnQualityProfile {
1549    pub name: String,
1550    pub dtype: DataType,
1551    pub evaluated_rows: usize,
1552    pub null_count: usize,
1553    pub empty_count: Option<usize>,
1554    pub whitespace_count: Option<usize>,
1555    pub nan_count: Option<usize>,
1556    pub positive_infinity_count: Option<usize>,
1557    pub negative_infinity_count: Option<usize>,
1558    pub distinct_count: Option<usize>,
1559    pub min: Option<String>,
1560    pub max: Option<String>,
1561    pub integer_parse_count: Option<usize>,
1562    pub decimal_parse_count: Option<usize>,
1563    pub date_parse_count: Option<usize>,
1564    pub datetime_parse_count: Option<usize>,
1565    /// Text values that parse as whole numbers and are written with a leading zero:
1566    /// the mark of a code (a ZIP, an account, an industry code) rather than a number.
1567    pub leading_zero_count: Option<usize>,
1568    pub dominant_value: Option<String>,
1569    pub dominant_count: Option<usize>,
1570    pub min_length: Option<usize>,
1571    pub max_length: Option<usize>,
1572}
1573
1574impl ColumnQualityProfile {
1575    pub fn null_rate(&self) -> f64 {
1576        rate(self.null_count, self.evaluated_rows)
1577    }
1578
1579    pub fn non_null_rows(&self) -> usize {
1580        self.evaluated_rows.saturating_sub(self.null_count)
1581    }
1582
1583    pub fn uniqueness_rate(&self) -> Option<f64> {
1584        self.distinct_count
1585            .map(|count| rate(count, self.non_null_rows()))
1586    }
1587}
1588
1589#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1590pub enum ObservationKind {
1591    Nulls,
1592    Empty,
1593    Whitespace,
1594    NonFinite,
1595    Constant,
1596    ParseableText,
1597    DuplicateRows,
1598    CategoryVariants,
1599    /// Rows whose own file has no such column. Their cells are absent, not null, and
1600    /// no measurement over values can tell the two apart.
1601    Absent,
1602    /// Rows whose file holds the column in a type the dataset's schema cannot read, so
1603    /// the column is not read from that file at all.
1604    TypeConflict,
1605    /// A column whose values are nearly unique and still repeat: the shape of a key
1606    /// that is not quite one.
1607    KeyLike,
1608    /// Text read as time that the chosen format does not read.
1609    UnparsedTime,
1610    /// Rows sharing a value of the declared key.
1611    KeyRepeated,
1612    /// Rows with no value in some part of the declared key.
1613    KeyMissing,
1614    /// A column declared required, with no value.
1615    RequiredMissing,
1616    /// Values outside a column's declared allowed set.
1617    NotAllowed,
1618    /// Values outside a column's declared range.
1619    OutOfRange,
1620    /// Text declared to read as a number that does not.
1621    UnparsedNumber,
1622    /// Audio samples in runs at full scale: the waveform cut flat at the limit.
1623    Clipping,
1624    /// Audio samples in long runs of exact zeros: dropouts, or digital silence.
1625    ZeroRuns,
1626    /// An audio channel whose mean sits away from zero.
1627    DcOffset,
1628}
1629
1630impl ObservationKind {
1631    pub fn label(self) -> &'static str {
1632        match self {
1633            Self::Nulls => "Null",
1634            Self::Empty => "Empty",
1635            Self::Whitespace => "Whitespace",
1636            Self::NonFinite => "Non-finite",
1637            Self::Constant => "Constant",
1638            Self::ParseableText => "Stored as text",
1639            Self::DuplicateRows => "Duplicate rows",
1640            Self::CategoryVariants => "Category variants",
1641            Self::Absent => "Absent",
1642            Self::TypeConflict => "Type conflict",
1643            Self::KeyLike => "Key-like",
1644            Self::UnparsedTime => "Unparsed time",
1645            Self::KeyRepeated => "Repeated key",
1646            Self::KeyMissing => "Incomplete key",
1647            Self::RequiredMissing => "Required, missing",
1648            Self::NotAllowed => "Not allowed",
1649            Self::OutOfRange => "Out of range",
1650            Self::UnparsedNumber => "Unparsed number",
1651            Self::Clipping => "Clipping",
1652            Self::ZeroRuns => "Zero runs",
1653            Self::DcOffset => "DC offset",
1654        }
1655    }
1656
1657    /// What the check divides, as the detail pane and the user guide state it.
1658    pub fn definition(self) -> &'static str {
1659        match self {
1660            Self::Nulls => "Null values / evaluated rows",
1661            Self::Empty => "Exact empty strings / evaluated rows",
1662            Self::Whitespace => "Nonempty strings that trim to empty / evaluated rows",
1663            Self::NonFinite => "NaN or positive/negative infinity / evaluated rows",
1664            Self::Constant => "One distinct non-null value in evaluated rows",
1665            Self::ParseableText => "Values parseable as a typed value, stored as text",
1666            Self::DuplicateRows => "Equal complete rows; extras = sum(group size - 1)",
1667            Self::CategoryVariants => "Distinct originals equal after trim and lowercase",
1668            Self::Absent => "Rows in files whose footer has no such column / source rows",
1669            Self::TypeConflict => {
1670                "Rows in files holding the column in an unreadable type / source rows"
1671            }
1672            Self::KeyLike => {
1673                "Non-null rows - distinct values, where distinct >= 95% of non-null rows"
1674            }
1675            Self::UnparsedTime => {
1676                "Non-null text the chosen time format does not read / non-null values"
1677            }
1678            Self::KeyRepeated => "Rows sharing a declared key value / rows checked",
1679            Self::KeyMissing => "Rows with no value in part of the declared key / rows checked",
1680            Self::RequiredMissing => "Null values in a required column / rows checked",
1681            Self::NotAllowed => "Values not in the declared set / non-null values",
1682            Self::OutOfRange => "Values below the minimum or above the maximum / values read",
1683            Self::UnparsedNumber => {
1684                "Non-null text that does not read as the number / non-null values"
1685            }
1686            Self::Clipping => "Samples in runs of 3 or more at full scale / samples",
1687            Self::ZeroRuns => {
1688                "Samples in runs of exact zeros 10 ms or longer (16 samples at least) / samples"
1689            }
1690            Self::DcOffset => "The channel's mean / full scale; noted from 1%",
1691        }
1692    }
1693}
1694
1695/// One file behind a drift observation: what it holds, and what that costs the column.
1696#[derive(Debug, Clone, PartialEq, Eq)]
1697pub struct QualityFileEvidence {
1698    /// Position in the Scope page's file inventory, which numbers files from 1.
1699    pub number: usize,
1700    pub name: String,
1701    /// Rows this file holds, from its footer.
1702    pub rows: usize,
1703    /// The type this file holds the column in, when the scan cannot read it as the
1704    /// dataset's. `None` for a file that simply has no such column.
1705    pub stored_type: Option<String>,
1706    /// The first values this file holds, read at its own type and rendered as text.
1707    /// Empty until a full run reads them.
1708    pub examples: Vec<String>,
1709}
1710
1711#[derive(Debug, Clone)]
1712pub struct QualityObservation {
1713    pub kind: ObservationKind,
1714    pub column: String,
1715    pub affected_rows: usize,
1716    pub evaluated_rows: usize,
1717    pub fact: String,
1718    pub normalized_category: Option<String>,
1719    /// The files behind an [`ObservationKind::Absent`] or
1720    /// [`ObservationKind::TypeConflict`] measurement, commonest first. Empty for every
1721    /// check measured over values rather than over footers.
1722    pub files: Vec<QualityFileEvidence>,
1723    /// The format an [`ObservationKind::UnparsedTime`] measurement read the text with.
1724    pub time_format: Option<TimeInterpretation>,
1725    /// The values at or past which an audio sample is at full scale, for an
1726    /// [`ObservationKind::Clipping`] measurement's rows.
1727    pub full_scale: Option<(f64, f64)>,
1728}
1729
1730impl QualityObservation {
1731    /// The scope that holds the rows behind this observation, when they are a set of
1732    /// files rather than a predicate over values. An absent or conflicting cell has no
1733    /// value to filter on — the rows are simply the ones the files contributed.
1734    pub fn evidence_scope(&self) -> Option<QualityScope> {
1735        if !matches!(
1736            self.kind,
1737            ObservationKind::Absent | ObservationKind::TypeConflict
1738        ) || self.files.is_empty()
1739        {
1740            return None;
1741        }
1742        Some(QualityScope::SourceFiles(
1743            self.files.iter().map(|file| file.number).collect(),
1744        ))
1745    }
1746
1747    pub fn evidence_predicate(&self) -> Option<Expr> {
1748        let value = col(&self.column);
1749        match self.kind {
1750            ObservationKind::Nulls => Some(value.is_null()),
1751            ObservationKind::Empty => Some(value.eq(lit(""))),
1752            ObservationKind::Whitespace => Some(
1753                value
1754                    .clone()
1755                    .cast(DataType::String)
1756                    .str()
1757                    .strip_chars(lit(LiteralValue::untyped_null()))
1758                    .eq(lit(""))
1759                    .and(value.neq(lit(""))),
1760            ),
1761            ObservationKind::NonFinite => Some(
1762                value
1763                    .clone()
1764                    .is_nan()
1765                    .or(value.clone().eq(lit(f64::INFINITY)))
1766                    .or(value.eq(lit(f64::NEG_INFINITY))),
1767            ),
1768            ObservationKind::Constant => Some(value.is_not_null()),
1769            ObservationKind::CategoryVariants => Some(
1770                value
1771                    .cast(DataType::String)
1772                    .str()
1773                    .strip_chars(lit(LiteralValue::untyped_null()))
1774                    .str()
1775                    .to_lowercase()
1776                    .eq(lit(self.normalized_category.clone()?)),
1777            ),
1778            // Nearly unique and still repeating: the repeats are exactly the rows
1779            // whose value is not the only one of its kind. Nulls are outside the
1780            // measurement, so they are outside its rows too.
1781            ObservationKind::KeyLike => {
1782                Some(value.clone().is_duplicated().and(value.is_not_null()))
1783            }
1784            ObservationKind::UnparsedTime => {
1785                self.time_format.as_ref().map(TimeInterpretation::unparsed)
1786            }
1787            // Every sample at full scale, in a run or not: the runs are what is
1788            // counted, and the samples around them are what a look wants.
1789            ObservationKind::Clipping => self
1790                .full_scale
1791                .map(|(low, high)| value.clone().lt_eq(lit(low)).or(value.gt_eq(lit(high)))),
1792            ObservationKind::ZeroRuns => Some(value.eq(lit(0))),
1793            // Absent and conflicting rows are named by their files, not by a predicate
1794            // over values: the column is not in those rows to be tested.
1795            // Declared rules find their rows through what the run measured them with:
1796            // see `IntentResults::evidence`.
1797            ObservationKind::ParseableText
1798            | ObservationKind::DuplicateRows
1799            | ObservationKind::Absent
1800            | ObservationKind::TypeConflict
1801            | ObservationKind::KeyRepeated
1802            | ObservationKind::KeyMissing
1803            | ObservationKind::RequiredMissing
1804            | ObservationKind::NotAllowed
1805            | ObservationKind::OutOfRange
1806            | ObservationKind::UnparsedNumber
1807            | ObservationKind::DcOffset => None,
1808        }
1809    }
1810}
1811
1812#[derive(Debug, Clone)]
1813pub struct IdentityProfile {
1814    pub duplicate_groups: usize,
1815    pub extra_rows: usize,
1816    pub rows_involved: usize,
1817    pub evaluated_rows: usize,
1818    pub precision: QualityPrecision,
1819    /// The most copied groups, from the rows the run kept. Empty after a full scan,
1820    /// which keeps no rows.
1821    pub examples: Vec<DuplicateExample>,
1822}
1823
1824/// One group of identical rows: how many there are, and the row, a value a column.
1825#[derive(Debug, Clone, PartialEq, Eq)]
1826pub struct DuplicateExample {
1827    pub copies: usize,
1828    /// Rendered for reading: text quoted, a null as `null`.
1829    pub values: Vec<String>,
1830}
1831
1832/// A few of the values behind one column's finding, from the rows the run kept.
1833#[derive(Debug, Clone, PartialEq, Eq)]
1834pub struct FindingExamples {
1835    pub kind: ObservationKind,
1836    pub column: String,
1837    /// Distinct values, first seen first, quoted.
1838    pub values: Vec<String>,
1839}
1840
1841#[derive(Debug, Clone)]
1842pub struct CategoryVariantGroup {
1843    pub column: String,
1844    pub normalized: String,
1845    pub variants: Vec<(String, usize)>,
1846    pub rows_involved: usize,
1847    pub complete: bool,
1848}
1849
1850#[derive(Debug, Clone)]
1851pub struct SegmentQualityProfile {
1852    pub label: String,
1853    pub total_rows: Option<usize>,
1854    pub evaluated_rows: usize,
1855    pub columns: Vec<ColumnQualityProfile>,
1856    pub null_cells: usize,
1857    pub null_rate: f64,
1858    pub compared_with: Option<String>,
1859    pub largest_change: Option<String>,
1860    /// How big `largest_change` is (points, or percent for a row count), to rank
1861    /// segments by; `None` when nothing clear moved.
1862    pub change_size: Option<f64>,
1863}
1864
1865/// One column's measure in a segment, and in the segment it is compared with.
1866#[derive(Debug, Clone, PartialEq)]
1867pub struct SegmentChange {
1868    pub column: String,
1869    pub metric: QualityMetric,
1870    pub before: Option<f64>,
1871    pub now: f64,
1872    /// The move is past sampling noise (always, on an exact profile) and a point
1873    /// or more.
1874    pub clear: bool,
1875}
1876
1877impl SegmentChange {
1878    /// Percentage points moved, when there is something to have moved from.
1879    pub fn change(&self) -> Option<f64> {
1880        self.before.map(|before| (self.now - before) * 100.0)
1881    }
1882}
1883
1884/// The order Segments lists its rows in: as they fall, or the clearest change
1885/// first (ties, and segments with no clear change, keep their order).
1886pub fn segment_order(results: &DataQualityResults, by_change: bool) -> Vec<usize> {
1887    let mut order = (0..results.segments.len()).collect::<Vec<_>>();
1888    if by_change {
1889        order.sort_by(|&left, &right| {
1890            let size = |index: usize| results.segments[index].change_size.unwrap_or(-1.0);
1891            size(right).total_cmp(&size(left))
1892        });
1893    }
1894    order
1895}
1896
1897/// Every column's measures in segment `index`: beside the segment it is compared
1898/// with and largest move first, or on its own worst first. A measure that is zero
1899/// on both sides says nothing and is left out.
1900pub fn segment_changes(results: &DataQualityResults, index: usize) -> Vec<SegmentChange> {
1901    let Some(segment) = results.segments.get(index) else {
1902        return Vec::new();
1903    };
1904    let compared = segment
1905        .compared_with
1906        .as_ref()
1907        .and_then(|label| results.segments.iter().find(|other| &other.label == label));
1908    let mut changes = Vec::new();
1909    for column in &segment.columns {
1910        let prior = compared.and_then(|other| other.columns.iter().find(|c| c.name == column.name));
1911        for metric in CHANGE_MEASURES {
1912            let Some(now) = metric.value(column) else {
1913                continue;
1914            };
1915            let before = prior.and_then(|prior| metric.value(prior));
1916            if now == 0.0 && before.unwrap_or(0.0) == 0.0 {
1917                continue;
1918            }
1919            let clear = match (prior, before) {
1920                (Some(prior), Some(before)) => {
1921                    (now - before).abs() * 100.0 >= MATERIAL_CHANGE_PP
1922                        && (results.precision == QualityPrecision::Exact
1923                            || beyond_noise(
1924                                now,
1925                                metric.denominator(column),
1926                                before,
1927                                metric.denominator(prior),
1928                            ))
1929                }
1930                _ => false,
1931            };
1932            changes.push(SegmentChange {
1933                column: column.name.clone(),
1934                metric,
1935                before,
1936                now,
1937                clear,
1938            });
1939        }
1940    }
1941    if compared.is_some() {
1942        // What cleared the noise first, then the rest, each largest first.
1943        changes.sort_by(|left, right| {
1944            let size = |change: &SegmentChange| change.change().unwrap_or(0.0).abs();
1945            right
1946                .clear
1947                .cmp(&left.clear)
1948                .then_with(|| size(right).total_cmp(&size(left)))
1949        });
1950    } else {
1951        changes.sort_by(|left, right| right.now.total_cmp(&left.now));
1952    }
1953    changes
1954}
1955
1956#[derive(Debug, Clone)]
1957pub struct TemporalLatencyProfile {
1958    pub segment: String,
1959    pub start_role: TemporalRole,
1960    pub end_role: TemporalRole,
1961    pub start_column: String,
1962    pub end_column: String,
1963    /// Rows in the segment.
1964    pub evaluated_rows: usize,
1965    /// Rows with both endpoints present and read: the rows a duration is taken on,
1966    /// and what negative, zero and threshold counts are out of. Not the rows less
1967    /// the missing ones, since a row can miss both.
1968    pub paired_rows: usize,
1969    pub missing_start: usize,
1970    pub missing_end: usize,
1971    /// Text the start column's format did not read; not counted as missing.
1972    pub unparsed_start: usize,
1973    pub unparsed_end: usize,
1974    /// Durations below zero: the end before the start.
1975    pub negative_count: usize,
1976    /// Durations of exactly zero: the end at the start.
1977    pub zero_count: usize,
1978    pub p50_seconds: Option<i64>,
1979    pub p90_seconds: Option<i64>,
1980    pub p95_seconds: Option<i64>,
1981    pub p99_seconds: Option<i64>,
1982    pub max_seconds: Option<i64>,
1983    /// The threshold the breaches were counted against: `duration > threshold`,
1984    /// strictly, so a duration of exactly the threshold is not a breach.
1985    pub threshold_seconds: Option<i64>,
1986    pub above_threshold_count: Option<usize>,
1987}
1988
1989impl TemporalLatencyProfile {
1990    pub fn pair(&self) -> (TemporalRole, TemporalRole) {
1991        (self.start_role, self.end_role)
1992    }
1993
1994    /// `event to received`.
1995    pub fn label(&self) -> String {
1996        interval_label(self.pair())
1997    }
1998
1999    /// A validity period, valid from to valid to: an end before the start is a
2000    /// period that is not valid, and no end is a period still open.
2001    pub fn is_validity(&self) -> bool {
2002        self.pair() == (TemporalRole::ValidFrom, TemporalRole::ValidTo)
2003    }
2004
2005    /// How many rows `fact` counts, and out of how many. `None` for a fact this
2006    /// interval does not measure: unparsed text with no format, a threshold not set.
2007    pub fn count(&self, fact: IntervalFact, plan: &DataQualityPlan) -> Option<(usize, usize)> {
2008        let rows = self.evaluated_rows;
2009        let paired = self.paired_rows;
2010        match fact {
2011            IntervalFact::MissingStart => Some((self.missing_start, rows)),
2012            IntervalFact::MissingEnd => Some((self.missing_end, rows)),
2013            IntervalFact::UnparsedStart => plan
2014                .time_format(&self.start_column)
2015                .map(|_| (self.unparsed_start, rows)),
2016            IntervalFact::UnparsedEnd => plan
2017                .time_format(&self.end_column)
2018                .map(|_| (self.unparsed_end, rows)),
2019            IntervalFact::Negative => Some((self.negative_count, paired)),
2020            IntervalFact::Zero => Some((self.zero_count, paired)),
2021            IntervalFact::OverThreshold => self.above_threshold_count.map(|count| (count, paired)),
2022        }
2023    }
2024
2025    /// Whether this interval's segment is a value its rows can be found by, rather
2026    /// than a stretch of rows or a file.
2027    pub fn segment_opens(&self, plan: &DataQualityPlan) -> bool {
2028        let grain = plan.interval_grain(&self.start_column, &self.end_column);
2029        segment_predicate(plan, &grain, &self.segment, None).is_some()
2030    }
2031
2032    /// The rows behind `fact` in this interval's segment, as a predicate over the
2033    /// scope `plan` measured: `None` when a segment cannot be told by its values
2034    /// (row chunks, files) or the fact is not measured.
2035    /// `schema` is the data's, where known: with it a partition segment compares in
2036    /// its column's type.
2037    pub fn evidence_predicate(
2038        &self,
2039        fact: IntervalFact,
2040        plan: &DataQualityPlan,
2041        schema: Option<&Schema>,
2042    ) -> Option<Expr> {
2043        self.count(fact, plan)?;
2044        let micros = || interval_micros(plan, &self.start_column, &self.end_column);
2045        let rows = match fact {
2046            IntervalFact::MissingStart => col(self.start_column.as_str()).is_null(),
2047            IntervalFact::MissingEnd => col(self.end_column.as_str()).is_null(),
2048            IntervalFact::UnparsedStart => plan.time_format(&self.start_column)?.unparsed(),
2049            IntervalFact::UnparsedEnd => plan.time_format(&self.end_column)?.unparsed(),
2050            IntervalFact::Negative => micros().lt(lit(0i64)),
2051            IntervalFact::Zero => micros().eq(lit(0i64)),
2052            IntervalFact::OverThreshold => {
2053                micros().gt(lit(self.threshold_seconds?.saturating_mul(1_000_000)))
2054            }
2055        };
2056        let grain = plan.interval_grain(&self.start_column, &self.end_column);
2057        Some(
2058            match segment_predicate(plan, &grain, &self.segment, schema)? {
2059                Some(segment) => segment.and(rows),
2060                None => rows,
2061            },
2062        )
2063    }
2064}
2065
2066/// What an interval's detail counts, each with the rows behind it.
2067#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2068pub enum IntervalFact {
2069    MissingStart,
2070    MissingEnd,
2071    UnparsedStart,
2072    UnparsedEnd,
2073    Negative,
2074    Zero,
2075    OverThreshold,
2076}
2077
2078impl IntervalFact {
2079    pub const ALL: [Self; 7] = [
2080        Self::MissingStart,
2081        Self::MissingEnd,
2082        Self::UnparsedStart,
2083        Self::UnparsedEnd,
2084        Self::Negative,
2085        Self::Zero,
2086        Self::OverThreshold,
2087    ];
2088
2089    /// The fact as its row in the detail names it. A validity period's missing end
2090    /// is an open period, and its negative duration one that ends before it starts.
2091    pub fn label(self, profile: &TemporalLatencyProfile) -> String {
2092        let validity = profile.is_validity();
2093        match self {
2094            Self::MissingStart => "Missing start".to_string(),
2095            Self::MissingEnd if validity => "Open, no end".to_string(),
2096            Self::MissingEnd => "Missing end".to_string(),
2097            Self::UnparsedStart => "Unparsed start".to_string(),
2098            Self::UnparsedEnd => "Unparsed end".to_string(),
2099            Self::Negative if validity => "Ends first".to_string(),
2100            Self::Negative => "Negative".to_string(),
2101            Self::Zero => "Zero".to_string(),
2102            Self::OverThreshold => format!(
2103                "Over {}",
2104                crate::analysis_modal::threshold_label(profile.threshold_seconds)
2105            ),
2106        }
2107    }
2108
2109    /// Short words for a list of rows: a view's label.
2110    pub fn short(self) -> &'static str {
2111        match self {
2112            Self::MissingStart => "missing start",
2113            Self::MissingEnd => "missing end",
2114            Self::UnparsedStart => "unparsed start",
2115            Self::UnparsedEnd => "unparsed end",
2116            Self::Negative => "negative",
2117            Self::Zero => "zero",
2118            Self::OverThreshold => "over threshold",
2119        }
2120    }
2121}
2122
2123/// The rows of the segment labeled `label` under `grain`, as a predicate: `Some(None)`
2124/// for the whole scope, `None` where a segment is a stretch of rows or a file and not
2125/// a value to filter on. Read back from the label, which names a partition's value
2126/// as its segment was keyed and a window's start exactly.
2127/// Each value of `column` as a segment label writes it, null where it is null. A
2128/// cast to text writes a float or a datetime differently than the label does, and
2129/// then the rows a label names would not be found.
2130fn label_text(column: &str) -> Expr {
2131    col(column).map(
2132        |values| {
2133            let text = (0..values.len())
2134                .map(|row| {
2135                    let value = values.get(row)?;
2136                    Ok((!value.is_null()).then(|| crate::exact::str_value(&value).into_owned()))
2137                })
2138                .collect::<PolarsResult<StringChunked>>()?;
2139            Ok(text.with_name(values.name().clone()).into_column())
2140        },
2141        |_, field| Ok(Field::new(field.name().clone(), DataType::String)),
2142    )
2143}
2144
2145/// The rows of a partition segment whose label writes `value`. Where the column's
2146/// type writes each value one way (text, flags, whole numbers, dates, decimals) and
2147/// `value` reads back as that very label, a plain comparison, which file statistics
2148/// can answer; otherwise, such as a float the label rounds, the label of each row.
2149fn partition_label_predicate(column: &str, value: &str, schema: Option<&Schema>) -> Expr {
2150    let writes_each_once = |dtype: &&DataType| {
2151        dtype.is_integer()
2152            || matches!(
2153                dtype,
2154                DataType::String
2155                    | DataType::Boolean
2156                    | DataType::Date
2157                    | DataType::Decimal(..)
2158                    | DataType::Categorical(..)
2159                    | DataType::Enum(..)
2160            )
2161    };
2162    let native = schema
2163        .and_then(|schema| schema.get(column))
2164        .filter(writes_each_once)
2165        .and_then(|dtype| crate::typed_value::parse(value, dtype).ok())
2166        .filter(|scalar| crate::exact::str_value(scalar.value()) == value);
2167    match native {
2168        Some(scalar) => col(column).eq(lit(scalar)),
2169        None => label_text(column).eq(lit(value.to_string())),
2170    }
2171}
2172
2173fn segment_predicate(
2174    plan: &DataQualityPlan,
2175    grain: &QualityGrain,
2176    label: &str,
2177    schema: Option<&Schema>,
2178) -> Option<Option<Expr>> {
2179    match grain {
2180        QualityGrain::Dataset => Some(None),
2181        QualityGrain::Partition(column) => {
2182            let value = label.strip_prefix(&format!("{column}="))?;
2183            Some(Some(if value == "∅" {
2184                col(column.as_str()).is_null()
2185            } else {
2186                partition_label_predicate(column, value, schema)
2187            }))
2188        }
2189        QualityGrain::TimeWindows { column, every } => {
2190            let value = plan.time_value(column);
2191            // The rows in no window: nulls, and dates past the calendar.
2192            if label == time_window_label(column, every, None) {
2193                return Some(Some(time_window_start(value, every).is_null()));
2194            }
2195            let date = |text: &str| chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d").ok();
2196            let start = match every.as_str() {
2197                "1h" => chrono::NaiveDateTime::parse_from_str(label, "%Y-%m-%d %H:%M").ok()?,
2198                "1d" => date(label)?.and_hms_opt(0, 0, 0)?,
2199                "1w" => date(label.strip_prefix("week of ")?)?.and_hms_opt(0, 0, 0)?,
2200                "1mo" => date(&format!("{label}-01"))?.and_hms_opt(0, 0, 0)?,
2201                _ => return None,
2202            };
2203            Some(Some(
2204                time_window_start(value, every).eq(lit(start.and_utc().timestamp_micros())
2205                    .cast(DataType::Datetime(TimeUnit::Microseconds, None))),
2206            ))
2207        }
2208        QualityGrain::RowChunks(_) | QualityGrain::File => None,
2209    }
2210}
2211/// Columns that are null the same number of times, and how many rows are null in all
2212/// of them at once. When the two counts agree, the columns go missing together: one
2213/// fact about some rows, not one per column.
2214#[derive(Debug, Clone, PartialEq, Eq)]
2215pub struct SharedNulls {
2216    pub columns: Vec<String>,
2217    pub null_rows: usize,
2218    pub rows_null_in_all: usize,
2219}
2220
2221impl SharedNulls {
2222    pub fn same_rows(&self) -> bool {
2223        self.rows_null_in_all == self.null_rows
2224    }
2225}
2226
2227#[derive(Debug, Clone)]
2228pub struct DataQualityResults {
2229    pub total_rows: Option<usize>,
2230    pub evaluated_rows: usize,
2231    pub precision: QualityPrecision,
2232    pub sample_seed: u64,
2233    pub columns: Vec<ColumnQualityProfile>,
2234    pub observations: Vec<QualityObservation>,
2235    pub segments: Vec<SegmentQualityProfile>,
2236    pub temporal: Vec<TemporalLatencyProfile>,
2237    pub identity: Option<IdentityProfile>,
2238    pub category_variants: Vec<CategoryVariantGroup>,
2239    pub shared_nulls: Vec<SharedNulls>,
2240    /// How many source files' footers were compared, when the scope has files to
2241    /// compare. `None` means the checks that compare files could not run.
2242    pub source_files: Option<usize>,
2243    /// Rows an equal-per-value sample kept of each value. See [`crate::sampling::PerValue`].
2244    pub per_value: Option<usize>,
2245    /// How many of `source_files` had their footers read: fewer on a dataset too
2246    /// large to read every footer, where the file checks cover only those.
2247    pub footers_read: Option<usize>,
2248    /// What the run's reads of the source were seen to do. `None` for results no
2249    /// watched run produced.
2250    pub reads: Option<ObservedReads>,
2251    /// Values behind text findings, from the rows the run kept; empty after a full
2252    /// scan.
2253    pub examples: Vec<FindingExamples>,
2254    /// Segments a sampled run counted rows in but drew none of, in segment order:
2255    /// the rows are there, the sample did not reach them. Not in `segments`, which
2256    /// profile only what was read.
2257    pub unsampled_segments: Vec<UnsampledSegment>,
2258    /// What the declared column intent found; `None` when nothing was declared.
2259    pub intent: Option<Box<crate::quality_intent::IntentResults>>,
2260    /// What the rows were read from, as the run that measured them labeled it.
2261    pub source: Option<Box<crate::quality_export::SourceIdentity>>,
2262}
2263
2264/// A segment the scope has rows in and a sample drew none of.
2265#[derive(Debug, Clone, PartialEq, Eq)]
2266pub struct UnsampledSegment {
2267    pub label: String,
2268    /// The rows the scope holds in it, by exact count.
2269    pub total_rows: usize,
2270}
2271
2272impl DataQualityResults {
2273    /// Memory the report holds, near enough to budget by: a profile per column, and
2274    /// another per column of every segment, which is what grows, and the text it
2275    /// keeps, whole in spellings and cut short in examples.
2276    pub fn estimated_bytes(&self) -> usize {
2277        let profile = |column: &ColumnQualityProfile| {
2278            std::mem::size_of::<ColumnQualityProfile>()
2279                + column.name.len()
2280                + column.min.as_ref().map_or(0, String::len)
2281                + column.max.as_ref().map_or(0, String::len)
2282                + column.dominant_value.as_ref().map_or(0, String::len)
2283        };
2284        let segments = self
2285            .segments
2286            .iter()
2287            .map(|segment| {
2288                std::mem::size_of::<SegmentQualityProfile>()
2289                    + segment.label.len()
2290                    + segment.columns.iter().map(profile).sum::<usize>()
2291            })
2292            .sum::<usize>();
2293        let observations = self
2294            .observations
2295            .iter()
2296            .map(|observation| {
2297                std::mem::size_of::<QualityObservation>()
2298                    + observation.fact.len()
2299                    + observation.column.len()
2300                    + observation.normalized_category.as_ref().map_or(0, String::len)
2301                    // A footer finding names every file it applies to.
2302                    + observation
2303                        .files
2304                        .iter()
2305                        .map(|file| {
2306                            std::mem::size_of::<QualityFileEvidence>()
2307                                + file.name.len()
2308                                + file.stored_type.as_ref().map_or(0, String::len)
2309                                + file.examples.iter().map(String::len).sum::<usize>()
2310                        })
2311                        .sum::<usize>()
2312            })
2313            .sum::<usize>();
2314        let unsampled = self
2315            .unsampled_segments
2316            .iter()
2317            .map(|segment| std::mem::size_of::<UnsampledSegment>() + segment.label.len())
2318            .sum::<usize>();
2319        let texts = |values: &[String]| {
2320            values
2321                .iter()
2322                .map(|value| std::mem::size_of::<String>() + value.len())
2323                .sum::<usize>()
2324        };
2325        // Spellings are whole values, as wide as the column's text is.
2326        let spellings = self
2327            .category_variants
2328            .iter()
2329            .map(|group| {
2330                std::mem::size_of::<CategoryVariantGroup>()
2331                    + group.column.len()
2332                    + group.normalized.len()
2333                    + group
2334                        .variants
2335                        .iter()
2336                        .map(|(variant, _)| std::mem::size_of::<(String, usize)>() + variant.len())
2337                        .sum::<usize>()
2338            })
2339            .sum::<usize>();
2340        let examples = self
2341            .examples
2342            .iter()
2343            .map(|found| std::mem::size_of::<FindingExamples>() + texts(&found.values))
2344            .sum::<usize>()
2345            + self.identity.as_ref().map_or(0, |identity| {
2346                identity
2347                    .examples
2348                    .iter()
2349                    .map(|example| std::mem::size_of::<DuplicateExample>() + texts(&example.values))
2350                    .sum()
2351            });
2352        let temporal = self
2353            .temporal
2354            .iter()
2355            .map(|latency| {
2356                std::mem::size_of::<TemporalLatencyProfile>()
2357                    + latency.segment.len()
2358                    + latency.start_column.len()
2359                    + latency.end_column.len()
2360            })
2361            .sum::<usize>();
2362        let shared = self
2363            .shared_nulls
2364            .iter()
2365            .map(|shared| std::mem::size_of::<SharedNulls>() + texts(&shared.columns))
2366            .sum::<usize>();
2367        // Declared intent keeps whole values: the extremes and the commonest misfits.
2368        let intent = self.intent.as_ref().map_or(0, |intent| {
2369            let counted = |values: &[(String, usize)]| {
2370                values
2371                    .iter()
2372                    .map(|(value, _)| std::mem::size_of::<(String, usize)>() + value.len())
2373                    .sum::<usize>()
2374            };
2375            std::mem::size_of::<crate::quality_intent::IntentResults>()
2376                + intent
2377                    .columns
2378                    .iter()
2379                    .map(|check| {
2380                        std::mem::size_of::<crate::quality_intent::ColumnCheck>()
2381                            + check.lowest.as_ref().map_or(0, String::len)
2382                            + check.highest.as_ref().map_or(0, String::len)
2383                            + counted(&check.outside_examples)
2384                            + counted(&check.unparsed_examples)
2385                    })
2386                    .sum::<usize>()
2387        });
2388        std::mem::size_of::<Self>()
2389            + self.columns.iter().map(profile).sum::<usize>()
2390            + segments
2391            + unsampled
2392            + observations
2393            + temporal
2394            + spellings
2395            + examples
2396            + shared
2397            + intent
2398    }
2399
2400    pub fn compare_segments(&mut self, plan: &DataQualityPlan) {
2401        apply_comparisons(
2402            &mut self.segments,
2403            plan.comparison,
2404            plan.baseline_segment.as_deref(),
2405            self.precision,
2406        );
2407    }
2408
2409    pub fn empty(total_rows: Option<usize>, plan: &DataQualityPlan, schema: &Schema) -> Self {
2410        Self {
2411            total_rows,
2412            evaluated_rows: 0,
2413            precision: QualityPrecision::Metadata,
2414            sample_seed: plan.sample_seed,
2415            columns: schema
2416                .iter()
2417                .map(|(name, dtype)| ColumnQualityProfile {
2418                    name: name.to_string(),
2419                    dtype: dtype.clone(),
2420                    evaluated_rows: 0,
2421                    null_count: 0,
2422                    empty_count: None,
2423                    whitespace_count: None,
2424                    nan_count: None,
2425                    positive_infinity_count: None,
2426                    negative_infinity_count: None,
2427                    distinct_count: None,
2428                    min: None,
2429                    max: None,
2430                    integer_parse_count: None,
2431                    decimal_parse_count: None,
2432                    date_parse_count: None,
2433                    datetime_parse_count: None,
2434                    leading_zero_count: None,
2435                    dominant_value: None,
2436                    dominant_count: None,
2437                    min_length: None,
2438                    max_length: None,
2439                })
2440                .collect(),
2441            observations: Vec::new(),
2442            segments: Vec::new(),
2443            temporal: Vec::new(),
2444            identity: None,
2445            category_variants: Vec::new(),
2446            shared_nulls: Vec::new(),
2447            source_files: None,
2448            per_value: None,
2449            footers_read: None,
2450            reads: None,
2451            examples: Vec::new(),
2452            unsampled_segments: Vec::new(),
2453            intent: None,
2454            source: None,
2455        }
2456    }
2457
2458    /// The kept examples of `kind` in `column`.
2459    pub fn examples_of(&self, kind: ObservationKind, column: &str) -> &[String] {
2460        self.examples
2461            .iter()
2462            .find(|examples| examples.kind == kind && examples.column == column)
2463            .map(|examples| examples.values.as_slice())
2464            .unwrap_or_default()
2465    }
2466}
2467
2468/// The rows a sampled run read, kept beside its results: an acquisition.
2469///
2470/// What decides these rows is the acquisition's identity — the dataset, the view, the
2471/// scope, the method, the size and the seed — which the caller keys it by. Everything
2472/// else a plan says (grain, comparison, time roles, text read as time, the latency
2473/// threshold) is the report's, and a run that changes only those cuts these rows
2474/// again rather than reading the source. Every column of the scope is kept, and where
2475/// each row sat, so any role, format or row-chunk grain finds what it needs here.
2476#[derive(Debug, Clone)]
2477pub struct QualitySample {
2478    df: DataFrame,
2479    /// Where each row sat in the scope, in the order of `df`.
2480    positions: Vec<IdxSize>,
2481    precision: QualityPrecision,
2482    total_rows: Option<usize>,
2483    per_value: Option<crate::sampling::PerValue>,
2484    /// Rows of the whole scope by segment key, by the grain they were counted for and
2485    /// the format its column was read through, when it is text read as time. Keyed as
2486    /// the key reads (`AnyValue::str_value`), `None` for null.
2487    counted: Vec<(SegmentKey, SegmentCounts)>,
2488    /// Grains whose count stopped at [`crate::sampling::MAX_COUNTED_KEYS`], so a run
2489    /// of one again says so rather than reading to find out.
2490    too_many: Vec<SegmentKey>,
2491}
2492
2493/// Rows by segment key, as a count read them.
2494type SegmentCounts = BTreeMap<Option<String>, usize>;
2495
2496/// What decides a segment count: the grain, and how its column was read as time.
2497type SegmentKey = (QualityGrain, Option<TimeInterpretation>);
2498
2499fn segment_key(plan: &DataQualityPlan) -> SegmentKey {
2500    let format = match &plan.grain {
2501        QualityGrain::TimeWindows { column, .. } => plan.time_format(column).cloned(),
2502        _ => None,
2503    };
2504    (plan.grain.clone(), format)
2505}
2506
2507/// How a full scan of a remote source gets its rows, as Setup says before Run: one
2508/// fetch into a local copy that every pass reads, a copy fetched earlier, or a pass
2509/// over the source for each check.
2510#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
2511pub enum CopyPlan {
2512    /// Not a full scan of a remote source read in place.
2513    #[default]
2514    NotApplicable,
2515    /// Every pass reads the source, for the reason given.
2516    Passes(NoCopy),
2517    /// The objects are fetched once into the cache directory first.
2518    Fetch { bytes: u64, objects: usize },
2519    /// A copy fetched earlier this session serves every pass.
2520    Kept { bytes: u64, objects: usize },
2521}
2522
2523/// Why a remote full scan reads the source in each pass instead of a local copy.
2524#[derive(Debug, Clone, Copy, PartialEq, Eq)]
2525pub enum NoCopy {
2526    /// `analysis.quality_local_copy` is 0.
2527    Off,
2528    /// The open did not learn every object's size.
2529    SizeUnknown,
2530    /// A copy fetched this session did not read as the source.
2531    Unusable,
2532    /// The scope reads only some of the rows or columns: its passes may read less
2533    /// than the whole objects a copy would fetch.
2534    PartOfTheSource,
2535    /// Larger than `analysis.quality_local_copy`.
2536    TooLarge { bytes: u64, limit: u64 },
2537    /// More than the cache directory has free, or its free space is unknown.
2538    NoRoom { bytes: u64, free: Option<u64> },
2539}
2540
2541/// Where a run's exact segment totals come from, as Setup says before Run.
2542#[derive(Debug, Clone, PartialEq, Eq, Default)]
2543pub enum SegmentCount {
2544    /// Nothing to count: the grain's sizes are known (files, row chunks, the whole
2545    /// scope), the run reads every row, or it reads no values.
2546    #[default]
2547    NotNeeded,
2548    /// An equal-per-value sample by the grain's column counts every value as it reads.
2549    PerValue,
2550    /// The pass that reads the sample counts the grain's key as it streams.
2551    InSamplePass,
2552    /// Counted by an earlier run of the rows being reused.
2553    Retained,
2554    /// Summed from a finer window's count of the same column, which it nests in
2555    /// exactly. Holds the finer width.
2556    RolledUp(String),
2557    /// A read of the grain's column of its own, after the sample.
2558    CountPass,
2559    /// The grain had more keys than a count holds: a coarser grain is needed.
2560    TooMany,
2561}
2562
2563impl SegmentCount {
2564    /// Whether the count reads the source in a pass of its own.
2565    pub fn reads(&self) -> bool {
2566        *self == Self::CountPass
2567    }
2568}
2569
2570/// A window width as a cadence: `1d` is daily.
2571pub fn window_cadence(every: &str) -> &str {
2572    match every {
2573        "1h" => "hourly",
2574        "1d" => "daily",
2575        "1w" => "weekly",
2576        "1mo" => "monthly",
2577        other => other,
2578    }
2579}
2580
2581/// Whether windows of width `fine` nest exactly in windows of `coarse`: every hour in
2582/// one day, every day in one week (weeks start on Monday) and one month. Windows are
2583/// cut on the stored clock with no time zone (UTC for a zoned column; see
2584/// [`time_window_start`]), where no day has 23 or 25 hours, so a sum of the finer
2585/// counts is the coarser count. A week does not nest in a month.
2586pub fn window_nests(fine: &str, coarse: &str) -> bool {
2587    matches!(
2588        (fine, coarse),
2589        ("1h", "1d" | "1w" | "1mo") | ("1d", "1w" | "1mo")
2590    )
2591}
2592
2593impl QualitySample {
2594    /// Whether `plan`'s segments need a count this sample does not hold: a partition
2595    /// or time-window grain on a sample, counted neither while sampling nor by an
2596    /// earlier run, nor summed from a finer count.
2597    pub fn needs_segment_count(&self, plan: &DataQualityPlan) -> bool {
2598        self.segment_count(plan).reads()
2599    }
2600
2601    /// Where a run of `plan` over these rows gets its segment totals.
2602    pub fn segment_count(&self, plan: &DataQualityPlan) -> SegmentCount {
2603        if self.precision != QualityPrecision::Sampled || !segments_need_count(plan) {
2604            return SegmentCount::NotNeeded;
2605        }
2606        if per_value_counts(plan, self.per_value.as_ref()) {
2607            return SegmentCount::PerValue;
2608        }
2609        let key = segment_key(plan);
2610        if self.too_many.contains(&key) {
2611            return SegmentCount::TooMany;
2612        }
2613        if self.counted.iter().any(|(counted, _)| *counted == key) {
2614            return SegmentCount::Retained;
2615        }
2616        match self.finer_count(&key) {
2617            Some(((QualityGrain::TimeWindows { every, .. }, _), _)) => {
2618                SegmentCount::RolledUp(every.clone())
2619            }
2620            _ => SegmentCount::CountPass,
2621        }
2622    }
2623
2624    /// A count of a finer window of the same column, read the same way, that `key`'s
2625    /// windows nest in exactly.
2626    fn finer_count(&self, key: &SegmentKey) -> Option<&(SegmentKey, SegmentCounts)> {
2627        let (QualityGrain::TimeWindows { column, every }, format) = key else {
2628            return None;
2629        };
2630        self.counted.iter().find(|((grain, counted_format), _)| {
2631            matches!(
2632                grain,
2633                QualityGrain::TimeWindows { column: counted, every: fine }
2634                    if counted == column && window_nests(fine, every)
2635            ) && counted_format == format
2636        })
2637    }
2638
2639    /// The rows themselves, as the sample every tool reads.
2640    pub fn df(&self) -> &DataFrame {
2641        &self.df
2642    }
2643
2644    /// Memory the rows, their positions and their counts hold, near enough to budget
2645    /// by.
2646    pub fn estimated_bytes(&self) -> usize {
2647        let counts = self
2648            .counted
2649            .iter()
2650            .flat_map(|(_, counts)| counts.keys())
2651            .map(|key| key.as_ref().map_or(0, String::len) + 64)
2652            .sum::<usize>();
2653        let per_value = self.per_value.as_ref().map_or(0, |per_value| {
2654            per_value
2655                .totals
2656                .keys()
2657                .map(|key| key.as_ref().map_or(0, String::len) + 64)
2658                .sum()
2659        });
2660        self.df.estimated_size()
2661            + self.positions.len() * std::mem::size_of::<IdxSize>()
2662            + counts
2663            + per_value
2664    }
2665
2666    /// `df`, cut from these rows, described as the sampler described them.
2667    pub fn analysis_rows(&self, df: DataFrame) -> crate::statistics::AnalysisRows {
2668        crate::statistics::AnalysisRows {
2669            sample_size: (self.precision == QualityPrecision::Sampled).then_some(df.height()),
2670            total_rows: self.total_rows.unwrap_or(df.height()),
2671            per_value: self.per_value.clone(),
2672            df,
2673        }
2674    }
2675}
2676
2677/// Whether a sampled run of `plan` counts its segments' rows: partitions and time
2678/// windows are counted for exact totals; files and row chunks are known without it.
2679pub fn segments_need_count(plan: &DataQualityPlan) -> bool {
2680    matches!(
2681        plan.grain,
2682        QualityGrain::Partition(_) | QualityGrain::TimeWindows { .. }
2683    )
2684}
2685
2686/// Whether the pass that samples `plan`'s rows also counts its segments: an
2687/// equal-per-value sample by the column the grain splits by counts every value as it
2688/// streams.
2689pub fn sampler_counts_segments(plan: &DataQualityPlan) -> bool {
2690    matches!(
2691        (&plan.grain, &plan.method),
2692        (
2693            QualityGrain::Partition(column),
2694            crate::sampling::SampleMethod::PerPartition { column: sampled },
2695        ) if column == sampled
2696    )
2697}
2698
2699/// Where a run of `plan` that reads a new sample gets its segment totals.
2700/// `may_read_blocks` is whether the sample may be seeded runs of one file, which see
2701/// too few rows to count; the head sees too few as well. Every other sample is one
2702/// streamed pass over the scope, which counts the grain's key as it goes.
2703pub fn fresh_segment_count(plan: &DataQualityPlan, may_read_blocks: bool) -> SegmentCount {
2704    if plan.compute != QualityCompute::Sample || !segments_need_count(plan) {
2705        return SegmentCount::NotNeeded;
2706    }
2707    if sampler_counts_segments(plan) {
2708        return SegmentCount::PerValue;
2709    }
2710    match plan.method {
2711        crate::sampling::SampleMethod::FirstRows => SegmentCount::CountPass,
2712        crate::sampling::SampleMethod::Spread if may_read_blocks => SegmentCount::CountPass,
2713        _ => SegmentCount::InSamplePass,
2714    }
2715}
2716
2717/// The key a partition or time-window grain splits rows by, as both the count and
2718/// the segments read it.
2719fn segment_count_key(plan: &DataQualityPlan) -> Option<Expr> {
2720    match &plan.grain {
2721        QualityGrain::Partition(column) => Some(col(column.as_str())),
2722        QualityGrain::TimeWindows { column, every } => {
2723            Some(time_window_start(plan.time_value(column), every))
2724        }
2725        _ => None,
2726    }
2727}
2728
2729/// Whether a sample's own counts are `plan`'s segment totals: the sampler counted
2730/// them, and the sample kept what it counted.
2731fn per_value_counts(plan: &DataQualityPlan, per_value: Option<&crate::sampling::PerValue>) -> bool {
2732    sampler_counts_segments(plan) && per_value.is_some()
2733}
2734
2735pub fn compute_data_quality(
2736    lf: &LazyFrame,
2737    total_rows: Option<usize>,
2738    plan: &DataQualityPlan,
2739    source: Option<&QualitySourceContext>,
2740    polars_streaming: bool,
2741) -> Result<DataQualityResults> {
2742    compute_data_quality_kept(lf, total_rows, plan, source, polars_streaming, None)
2743        .map(|(results, _)| results)
2744}
2745
2746/// [`compute_data_quality`], cutting `kept` instead of reading when it serves the
2747/// plan, and returning the sample a sampled run read so the next run can do the same.
2748pub fn compute_data_quality_kept(
2749    lf: &LazyFrame,
2750    total_rows: Option<usize>,
2751    plan: &DataQualityPlan,
2752    source: Option<&QualitySourceContext>,
2753    polars_streaming: bool,
2754    kept: Option<&QualitySample>,
2755) -> Result<(DataQualityResults, Option<QualitySample>)> {
2756    let (results, kept) = compute_data_quality_watched(
2757        lf,
2758        total_rows,
2759        plan,
2760        source,
2761        polars_streaming,
2762        kept,
2763        &QualityWatch::default(),
2764    );
2765    results.map(|results| (results, kept))
2766}
2767
2768/// [`compute_data_quality_kept`], naming each stage to `watch` as it enters it and
2769/// stopping between stages, or inside a streamed read, once `watch` is cancelled.
2770///
2771/// The sample a sampled run read comes back whether or not the run finished: a run
2772/// stopped after its read has still paid for it, and the next run can cut it.
2773pub fn compute_data_quality_watched(
2774    lf: &LazyFrame,
2775    total_rows: Option<usize>,
2776    plan: &DataQualityPlan,
2777    source: Option<&QualitySourceContext>,
2778    polars_streaming: bool,
2779    kept: Option<&QualitySample>,
2780    watch: &QualityWatch,
2781) -> (Result<DataQualityResults>, Option<QualitySample>) {
2782    let mut acquired = None;
2783    let inputs = QualityInputs {
2784        lf,
2785        total_rows,
2786        plan,
2787        source,
2788        // Without the feature `collect_lazy` is one in-memory collect whatever the
2789        // setting, with no batch boundary for a cancel to stop at (#498).
2790        polars_streaming: polars_streaming && cfg!(feature = "streaming"),
2791        watch,
2792    };
2793    let results = profile_quality(inputs, kept, &mut acquired);
2794    (results, acquired)
2795}
2796
2797/// What a run is asked to measure, and how it reports.
2798#[derive(Clone, Copy)]
2799struct QualityInputs<'a> {
2800    lf: &'a LazyFrame,
2801    total_rows: Option<usize>,
2802    plan: &'a DataQualityPlan,
2803    source: Option<&'a QualitySourceContext>,
2804    polars_streaming: bool,
2805    watch: &'a QualityWatch,
2806}
2807
2808/// The run itself. A sampled run's rows go into `acquired` the moment they are
2809/// read, so they outlive a run that stops after.
2810fn profile_quality(
2811    inputs: QualityInputs<'_>,
2812    kept: Option<&QualitySample>,
2813    acquired: &mut Option<QualitySample>,
2814) -> Result<DataQualityResults> {
2815    let QualityInputs {
2816        lf,
2817        total_rows,
2818        plan,
2819        source,
2820        polars_streaming,
2821        watch,
2822    } = inputs;
2823    watch.stage(QualityStage::Preparing, false, false)?;
2824    let collected_schema = lf.clone().collect_schema()?;
2825    let schema = visible_schema(&collected_schema, source);
2826    // What the footers already said: which files have which columns. Free at every
2827    // compute budget, including the one that reads no values at all.
2828    if plan.compute == QualityCompute::Metadata {
2829        watch.stage(QualityStage::Assembling, false, false)?;
2830        let mut results = DataQualityResults::empty(total_rows, plan, &schema);
2831        if let Some(source) = source {
2832            results.observations = drift_observations(source, None, polars_streaming, watch);
2833        }
2834        results.source_files = source.map(|source| source.file_names.len());
2835        results.footers_read = source.map(|source| source.footers_read);
2836        results.reads = Some(watch.observed());
2837        results.intent =
2838            crate::quality_intent::IntentResults::unmeasured(plan, &schema).map(Box::new);
2839        return Ok(results);
2840    }
2841    let grain_column = match &plan.grain {
2842        QualityGrain::Partition(column) | QualityGrain::TimeWindows { column, .. } => Some(column),
2843        _ => None,
2844    };
2845    if let Some(column) = grain_column
2846        && collected_schema.get(column).is_none()
2847    {
2848        return Err(Report::msg(format!(
2849            "Grain column {column} is not in scope {}; choose another grain or scope",
2850            plan.scope.label()
2851        )));
2852    }
2853    if let QualityGrain::TimeWindows { column, .. } = &plan.grain
2854        && !plan.reads_as_time(column, &collected_schema)
2855    {
2856        return Err(Report::msg(format!(
2857            "Grain column {column} is text; choose a format for it under Text as time"
2858        )));
2859    }
2860    if plan.compute == QualityCompute::Full {
2861        let total_rows = match total_rows {
2862            Some(rows) => rows,
2863            None => {
2864                // Unwatched: Parquet and IPC answer a count from their metadata, which
2865                // a watch between the count and the scan would turn into a read.
2866                watch.stage(QualityStage::CountingRows, watch.scope_reads(true), false)?;
2867                let count = collect_lazy(
2868                    crate::widgets::datatable::row_count_lf(lf),
2869                    polars_streaming,
2870                )
2871                .map_err(Report::from)?;
2872                let count_values = count
2873                    .get(0)
2874                    .ok_or_else(|| Report::msg("Data quality row count was not returned"))?;
2875                let Some(AnyValue::UInt64(rows)) = count_values.first() else {
2876                    return Err(Report::msg("Data quality row count was not UInt64"));
2877                };
2878                *rows as usize
2879            }
2880        };
2881        if total_rows == 0 && plan.scope != QualityScope::CurrentView {
2882            return Err(crate::sampling::no_rows_error(&plan.scope));
2883        }
2884        return compute_full_quality(
2885            lf,
2886            total_rows,
2887            plan,
2888            source,
2889            &schema,
2890            polars_streaming,
2891            watch,
2892        );
2893    }
2894
2895    // The shared analysis sampler, as every other tool reads: by default spread
2896    // across the whole scope, so a file sorted by date is not judged by its first
2897    // stretch. Every grain cuts its segments from this one sample, so a segmented
2898    // run reads no more than the sample says and measures the rows every tool reads.
2899    let kept = acquired.insert(match kept {
2900        Some(kept) => {
2901            watch.stage(QualityStage::ReusingSample, false, false)?;
2902            kept.clone()
2903        }
2904        None => {
2905            // The first rows are one collect; every other method streams in batches
2906            // or reads seeded runs, and stops between them. Without the streaming
2907            // engine a streamed sample's batches come after its whole read; seeded
2908            // runs still stop between runs, sooner than this promises.
2909            let interruptible = plan.method != crate::sampling::SampleMethod::FirstRows
2910                && cfg!(feature = "streaming");
2911            watch.stage(QualityStage::ReadingSample, true, interruptible)?;
2912            read_quality_sample(lf, total_rows, plan, polars_streaming, watch)?
2913        }
2914    });
2915    let profile_df = kept.df.clone();
2916    let sample_positions = kept.positions.clone();
2917    let evaluated_rows = profile_df.height();
2918    let precision = kept.precision;
2919    let total_rows = kept.total_rows;
2920
2921    // Rows chosen by the sample that match nothing are a mistake to name, not an
2922    // empty report that reads as clean.
2923    if total_rows == Some(0) && plan.scope != QualityScope::CurrentView {
2924        return Err(crate::sampling::no_rows_error(&plan.scope));
2925    }
2926    let profile_df = attach_source_file(profile_df, source)?;
2927    watch.stage(QualityStage::ProfilingColumns, false, false)?;
2928    let mut columns = profile_columns(&profile_df, &schema, polars_streaming)?;
2929    // The same Polars aggregations a full scan uses, over the rows the sample kept:
2930    // they scale to any sample the shared form asks for, where a walk over rows did
2931    // not, and a sample and a scan are measured the same way.
2932    let profile_lf = profile_df.clone().lazy();
2933    add_dominance_lazy(&profile_lf, &mut columns, polars_streaming)?;
2934    // Text read as time and the declared intent are counted over the rows in memory,
2935    // as the columns were.
2936    let mut formats = interpretation_exprs(plan, &collected_schema);
2937    formats.extend(crate::quality_intent::intent_exprs(plan, &schema));
2938    let unparsed = if formats.is_empty() {
2939        DataFrame::default()
2940    } else {
2941        collect_lazy(profile_lf.clone().select(formats), polars_streaming).map_err(Report::from)?
2942    };
2943    watch.stage(QualityStage::CheckingDuplicates, false, false)?;
2944    let identity = profile_identity_lazy(
2945        &profile_lf,
2946        &schema,
2947        evaluated_rows,
2948        precision,
2949        polars_streaming,
2950    )?;
2951    // The declared key's repeats among the rows in memory: a repeat among distinct
2952    // sampled rows is a repeat in the data, and no repeat says nothing past them.
2953    let repeats = crate::quality_intent::key_repeats(&profile_lf, plan, &schema, polars_streaming)?;
2954    let intent = crate::quality_intent::IntentResults::from_counts(
2955        plan,
2956        &schema,
2957        &unparsed,
2958        repeats,
2959        evaluated_rows,
2960        precision,
2961        Some(&profile_lf),
2962    )?
2963    .map(Box::new);
2964    watch.stage(QualityStage::CheckingSpellings, false, false)?;
2965    let category_variants = profile_category_variants_lazy(&profile_lf, &schema, polars_streaming)?;
2966    let mut observations = observations_from_profiles(&columns, precision);
2967    observations.extend(interpretation_observations(
2968        &unparsed,
2969        plan,
2970        &collected_schema,
2971    ));
2972    observations.extend(identity_observations(&identity, &category_variants));
2973    if let Some(intent) = &intent {
2974        observations.extend(intent.observations());
2975    }
2976    crate::quality_intent::supersede(&mut observations, plan);
2977    // The rows are in memory, so the detail can show a few of the values behind a
2978    // finding without reading anything again.
2979    let mut identity = identity;
2980    if identity.duplicate_groups > 0 {
2981        identity.examples = duplicate_examples(&profile_lf, &schema, polars_streaming)?;
2982    }
2983    let examples = finding_examples(&profile_lf, &columns, &observations, polars_streaming)?;
2984    // A sampled run does not promise the extra reads, so the counts come without the
2985    // values behind them.
2986    if let Some(source) = source {
2987        observations.extend(drift_observations(source, None, polars_streaming, watch));
2988    }
2989    let totals = {
2990        let mut totals = known_segment_totals(plan, total_rows, source);
2991        if precision == QualityPrecision::Sampled {
2992            totals.extend(sampled_segment_totals(
2993                lf,
2994                plan,
2995                kept,
2996                polars_streaming,
2997                watch,
2998            )?);
2999        }
3000        totals
3001    };
3002    watch.stage(QualityStage::ProfilingSegments, false, false)?;
3003    let (segments, unsampled_segments) = profile_segments(
3004        &profile_df,
3005        total_rows,
3006        plan,
3007        precision,
3008        &schema,
3009        SegmentSampleProvenance {
3010            positions: Some(sample_positions.as_slice()),
3011            totals: &totals,
3012        },
3013        polars_streaming,
3014    )?;
3015    watch.stage(QualityStage::ComputingIntervals, false, false)?;
3016    let temporal = profile_temporal(&profile_df, plan, Some(sample_positions.as_slice()))?;
3017    watch.stage(QualityStage::CheckingSharedNulls, false, false)?;
3018    let shared_nulls = profile_shared_nulls(&profile_df.lazy(), &columns, polars_streaming)?;
3019    let per_value = kept.per_value.as_ref().map(|per_value| per_value.kept);
3020    watch.stage(QualityStage::Assembling, false, false)?;
3021
3022    let results = DataQualityResults {
3023        total_rows,
3024        evaluated_rows,
3025        precision,
3026        sample_seed: plan.sample_seed,
3027        columns,
3028        observations,
3029        segments,
3030        temporal,
3031        identity: Some(identity),
3032        category_variants,
3033        shared_nulls,
3034        source_files: source.map(|source| source.file_names.len()),
3035        per_value,
3036        footers_read: source.map(|source| source.footers_read),
3037        reads: Some(watch.observed()),
3038        examples,
3039        unsampled_segments,
3040        intent,
3041        source: None,
3042    };
3043    Ok(results)
3044}
3045
3046/// Read the rows a sampled run measures, counting the grain's segments in the same
3047/// pass when the sampler streams every row.
3048fn read_quality_sample(
3049    lf: &LazyFrame,
3050    total_rows: Option<usize>,
3051    plan: &DataQualityPlan,
3052    polars_streaming: bool,
3053    watch: &QualityWatch,
3054) -> Result<QualitySample> {
3055    let sample = crate::sampling::Sample {
3056        scope: QualityScope::CurrentView,
3057        method: plan.method.clone(),
3058        rows: plan.dataset_rows,
3059        seed: plan.sample_seed,
3060    };
3061    let count = if sampler_counts_segments(plan) {
3062        None
3063    } else {
3064        segment_count_key(plan)
3065    };
3066    let sampled = crate::sampling::acquire(
3067        lf,
3068        &sample,
3069        total_rows,
3070        polars_streaming,
3071        Some(watch.read()),
3072        count.as_ref(),
3073    )?;
3074    let precision = if sampled.rows.sample_size.is_some() {
3075        QualityPrecision::Sampled
3076    } else {
3077        QualityPrecision::Exact
3078    };
3079    let mut kept = QualitySample {
3080        df: sampled.rows.df,
3081        positions: sampled.positions,
3082        precision,
3083        total_rows: Some(sampled.rows.total_rows),
3084        per_value: sampled.rows.per_value,
3085        counted: Vec::new(),
3086        too_many: Vec::new(),
3087    };
3088    match sampled.counted {
3089        Some(crate::sampling::Counted::Totals(totals)) => {
3090            kept.counted.push((segment_key(plan), totals));
3091        }
3092        Some(crate::sampling::Counted::TooMany) => kept.too_many.push(segment_key(plan)),
3093        None => {}
3094    }
3095    Ok(kept)
3096}
3097
3098/// How many rows each segment of a sampled run holds, reading only what nothing has
3099/// counted yet.
3100///
3101/// An equal-per-value sample counted every value as it streamed, so a grain by the
3102/// same column is already counted, and a streamed sample counted the grain it was read
3103/// for. A coarser window is summed from a finer window's count when it nests in it
3104/// exactly. Anything else is counted by a read of its key, once: the count is kept with
3105/// the sample for the next run.
3106fn sampled_segment_totals(
3107    lf: &LazyFrame,
3108    plan: &DataQualityPlan,
3109    kept: &mut QualitySample,
3110    polars_streaming: bool,
3111    watch: &QualityWatch,
3112) -> Result<BTreeMap<String, usize>> {
3113    if !segments_need_count(plan) {
3114        return Ok(BTreeMap::new());
3115    }
3116    let labeled = |counts: &SegmentCounts| {
3117        counts
3118            .iter()
3119            .map(|(raw, rows)| (segment_label(&plan.grain, raw.as_deref()), *rows))
3120            .collect::<BTreeMap<_, _>>()
3121    };
3122    if per_value_counts(plan, kept.per_value.as_ref())
3123        && let Some(per_value) = &kept.per_value
3124    {
3125        return Ok(labeled(&per_value.totals));
3126    }
3127    let key = segment_key(plan);
3128    if kept.too_many.contains(&key) {
3129        return Err(too_many_segments(plan));
3130    }
3131    if let Some((_, counts)) = kept.counted.iter().find(|(counted, _)| *counted == key) {
3132        return Ok(labeled(counts));
3133    }
3134    if let Some((_, finer)) = kept.finer_count(&key)
3135        && let QualityGrain::TimeWindows { every, .. } = &plan.grain
3136    {
3137        let counts = roll_up_windows(finer, every)?;
3138        let totals = labeled(&counts);
3139        kept.counted.push((key, counts));
3140        return Ok(totals);
3141    }
3142    watch.stage(QualityStage::CountingSegments, true, polars_streaming)?;
3143    let counts = counted_segment_totals(&watch.watched(lf), plan, polars_streaming)
3144        .map_err(|error| watch.failed(error))?;
3145    if counts.len() > crate::sampling::MAX_COUNTED_KEYS {
3146        kept.too_many.push(key);
3147        return Err(too_many_segments(plan));
3148    }
3149    let totals = labeled(&counts);
3150    kept.counted.push((key, counts));
3151    Ok(totals)
3152}
3153
3154fn too_many_segments(plan: &DataQualityPlan) -> Report {
3155    Report::msg(format!(
3156        "More than {} segments {}; choose a coarser grain",
3157        crate::numfmt::group_chrome(crate::sampling::MAX_COUNTED_KEYS),
3158        plan.grain.label()
3159    ))
3160}
3161
3162/// A finer window's counts summed into `every`'s windows, through the expression
3163/// that cuts every window, so the sum lands where a count by `every` would. Exact only
3164/// where [`window_nests`] says so; the caller asks it first.
3165fn roll_up_windows(finer: &SegmentCounts, every: &str) -> Result<SegmentCounts> {
3166    let mut rolled = SegmentCounts::new();
3167    let mut starts = Vec::with_capacity(finer.len());
3168    let mut rows = Vec::with_capacity(finer.len());
3169    for (raw, count) in finer {
3170        match raw {
3171            Some(raw) => {
3172                let start = chrono::NaiveDateTime::parse_from_str(raw, "%Y-%m-%d %H:%M:%S%.f")
3173                    .map_err(|_| Report::msg(format!("Window start {raw:?} is not a time")))?;
3174                starts.push(start.and_utc().timestamp_micros());
3175                rows.push(*count as u64);
3176            }
3177            // A row with no time is in no window at any width.
3178            None => *rolled.entry(None).or_default() += count,
3179        }
3180    }
3181    let finer = DataFrame::new(
3182        starts.len(),
3183        vec![
3184            Column::new("start".into(), starts)
3185                .cast(&DataType::Datetime(TimeUnit::Microseconds, None))?,
3186            Column::new("rows".into(), rows),
3187        ],
3188    )?;
3189    let coarse = finer
3190        .lazy()
3191        .select([time_window_start(col("start"), every), col("rows")])
3192        .collect()?;
3193    let (starts, rows) = (coarse.column("start")?, coarse.column("rows")?.u64()?);
3194    for (row, count) in rows.into_no_null_iter().enumerate() {
3195        let start = starts.get(row)?;
3196        let key = (!start.is_null()).then(|| crate::exact::str_value(&start).into_owned());
3197        *rolled.entry(key).or_default() += count as usize;
3198    }
3199    Ok(rolled)
3200}
3201
3202fn compute_full_quality(
3203    lf: &LazyFrame,
3204    total_rows: usize,
3205    plan: &DataQualityPlan,
3206    source: Option<&QualitySourceContext>,
3207    schema: &Schema,
3208    polars_streaming: bool,
3209    watch: &QualityWatch,
3210) -> Result<DataQualityResults> {
3211    let full_schema = lf.clone().collect_schema()?;
3212    // Every pass reads through the watch, so a cancel stops it within a batch on the
3213    // streaming engine, and the rows each pass traverses are counted.
3214    let lf = &watch.watched(lf);
3215    let failed = |error: Report| watch.failed(error);
3216    watch.stage(
3217        QualityStage::ProfilingColumns,
3218        watch.scope_reads(true),
3219        polars_streaming,
3220    )?;
3221    // Text read as time is counted in the same pass as every column's profile.
3222    let mut exprs = build_profile_exprs(schema);
3223    exprs.extend(interpretation_exprs(plan, &full_schema));
3224    // The declared intent's counts too: sums over the same rows, in the same pass.
3225    exprs.extend(crate::quality_intent::intent_exprs(plan, schema));
3226    let aggregate = collect_lazy(lf.clone().select(exprs), polars_streaming)
3227        .map_err(|error| watch.failed(error))?;
3228    let mut columns = parse_profiles(&aggregate, schema, total_rows);
3229    add_dominance_lazy(lf, &mut columns, polars_streaming).map_err(failed)?;
3230    watch.stage(
3231        QualityStage::CheckingDuplicates,
3232        watch.scope_reads(true),
3233        polars_streaming,
3234    )?;
3235    let identity = profile_identity_lazy(
3236        lf,
3237        schema,
3238        total_rows,
3239        QualityPrecision::Exact,
3240        polars_streaming,
3241    )
3242    .map_err(failed)?;
3243    let texts = schema
3244        .iter_values()
3245        .any(|dtype| matches!(dtype, DataType::String | DataType::Categorical(..)));
3246    watch.stage(
3247        QualityStage::CheckingSpellings,
3248        watch.scope_reads(texts),
3249        polars_streaming,
3250    )?;
3251    let category_variants =
3252        profile_category_variants_lazy(lf, schema, polars_streaming).map_err(failed)?;
3253    // The declared key is one grouping of its columns: a pass of its own, which
3254    // Setup counts among the passes before Run.
3255    let keyed = !plan.intent.key.is_empty();
3256    watch.stage(
3257        QualityStage::CheckingKey,
3258        watch.scope_reads(keyed),
3259        polars_streaming,
3260    )?;
3261    let repeats =
3262        crate::quality_intent::key_repeats(lf, plan, schema, polars_streaming).map_err(failed)?;
3263    let intent = crate::quality_intent::IntentResults::from_counts(
3264        plan,
3265        schema,
3266        &aggregate,
3267        repeats,
3268        total_rows,
3269        QualityPrecision::Exact,
3270        None,
3271    )?
3272    .map(Box::new);
3273    let mut observations = observations_from_profiles(&columns, QualityPrecision::Exact);
3274    observations.extend(interpretation_observations(&aggregate, plan, &full_schema));
3275    observations.extend(identity_observations(&identity, &category_variants));
3276    if let Some(intent) = &intent {
3277        observations.extend(intent.observations());
3278    }
3279    crate::quality_intent::supersede(&mut observations, plan);
3280    // Only a run that already reads every value pays for the conflicting values, and
3281    // only that run's access plan promised the read. Each file is read on its own,
3282    // so a cancel stops them between files.
3283    if let Some(source) = source {
3284        if source.conflict_scan.is_some() {
3285            watch.stage(QualityStage::ReadingConflicts, true, true)?;
3286        }
3287        observations.extend(drift_observations(
3288            source,
3289            source.conflict_scan.as_ref(),
3290            polars_streaming,
3291            watch,
3292        ));
3293    }
3294    let whole = unsegmented(plan, source);
3295    watch.stage(
3296        QualityStage::ProfilingSegments,
3297        watch.scope_reads(!whole),
3298        polars_streaming,
3299    )?;
3300    let segments = if whole {
3301        // The whole scope is one segment, and its profile is the one just measured:
3302        // reading it again would be a second pass for the same numbers.
3303        vec![whole_segment(plan, total_rows, &columns, schema.len())]
3304    } else {
3305        profile_segments_lazy(lf, total_rows, plan, source, schema, polars_streaming)
3306            .map_err(failed)?
3307    };
3308    let intervals = !resolved_intervals(plan, &full_schema).is_empty();
3309    watch.stage(
3310        QualityStage::ComputingIntervals,
3311        watch.scope_reads(intervals),
3312        polars_streaming,
3313    )?;
3314    let temporal = profile_temporal_lazy(lf, plan, source, polars_streaming).map_err(failed)?;
3315    let shared = !shared_null_groups(&columns).is_empty();
3316    watch.stage(
3317        QualityStage::CheckingSharedNulls,
3318        watch.scope_reads(shared),
3319        polars_streaming,
3320    )?;
3321    let shared_nulls = profile_shared_nulls(lf, &columns, polars_streaming).map_err(failed)?;
3322    watch.stage(QualityStage::Assembling, false, false)?;
3323    Ok(DataQualityResults {
3324        total_rows: Some(total_rows),
3325        evaluated_rows: total_rows,
3326        precision: QualityPrecision::Exact,
3327        sample_seed: plan.sample_seed,
3328        columns,
3329        observations,
3330        segments,
3331        temporal,
3332        identity: Some(identity),
3333        category_variants,
3334        shared_nulls,
3335        source_files: source.map(|source| source.file_names.len()),
3336        per_value: None,
3337        footers_read: source.map(|source| source.footers_read),
3338        reads: Some(watch.observed()),
3339        examples: Vec::new(),
3340        unsampled_segments: Vec::new(),
3341        intent,
3342        source: None,
3343    })
3344}
3345
3346/// Columns sharing a nonzero null count, by the count: the sets worth checking for
3347/// rows null in all of them.
3348fn shared_null_groups(columns: &[ColumnQualityProfile]) -> Vec<(usize, Vec<String>)> {
3349    let mut by_count = BTreeMap::<usize, Vec<String>>::new();
3350    for profile in columns.iter().filter(|profile| profile.null_count > 0) {
3351        by_count
3352            .entry(profile.null_count)
3353            .or_default()
3354            .push(profile.name.clone());
3355    }
3356    by_count
3357        .into_iter()
3358        .filter(|(_, names)| names.len() > 1)
3359        .collect()
3360}
3361
3362/// For every set of two or more columns with the same nonzero null count, how many
3363/// rows are null in all of them.
3364///
3365/// Equal counts are only a hint; this is the check. It reads just those columns, once,
3366/// and is skipped entirely when no two columns share a count.
3367fn profile_shared_nulls(
3368    lf: &LazyFrame,
3369    columns: &[ColumnQualityProfile],
3370    polars_streaming: bool,
3371) -> Result<Vec<SharedNulls>> {
3372    let groups = shared_null_groups(columns);
3373    if groups.is_empty() {
3374        return Ok(Vec::new());
3375    }
3376    let exprs = groups
3377        .iter()
3378        .enumerate()
3379        .map(|(index, (_, names))| {
3380            names
3381                .iter()
3382                .map(|name| col(name.as_str()).is_null())
3383                .reduce(Expr::and)
3384                .expect("a group has two columns")
3385                .sum()
3386                .alias(format!("__quality_shared_null_{index}"))
3387        })
3388        .collect::<Vec<_>>();
3389    let counts = collect_lazy(lf.clone().select(exprs), polars_streaming).map_err(Report::from)?;
3390    Ok(groups
3391        .into_iter()
3392        .enumerate()
3393        .map(|(index, (null_rows, columns))| SharedNulls {
3394            columns,
3395            null_rows,
3396            rows_null_in_all: usize_value(&counts, &format!("__quality_shared_null_{index}")),
3397        })
3398        .collect())
3399}
3400
3401/// The most common value of every column, in one pass. A scan per column would
3402/// re-read the whole source once per column, which on a remote dataset is the
3403/// difference between one read and sixty — and the access plan promises one.
3404fn add_dominance_lazy(
3405    lf: &LazyFrame,
3406    profiles: &mut [ColumnQualityProfile],
3407    polars_streaming: bool,
3408) -> Result<()> {
3409    if profiles.is_empty() {
3410        return Ok(());
3411    }
3412    const COUNT: &str = "__quality_value_count";
3413    let exprs = profiles
3414        .iter()
3415        .enumerate()
3416        .map(|(index, profile)| {
3417            col(&profile.name)
3418                .drop_nulls()
3419                .value_counts(true, true, COUNT, false)
3420                .first()
3421                .alias(format!("__quality_dominant_{index}"))
3422        })
3423        .collect::<Vec<_>>();
3424    let top = collect_lazy(lf.clone().select(exprs), polars_streaming).map_err(Report::from)?;
3425    for (index, profile) in profiles.iter_mut().enumerate() {
3426        let Ok(column) = top.column(&format!("__quality_dominant_{index}")) else {
3427            continue;
3428        };
3429        let Ok(fields) = column.struct_() else {
3430            continue;
3431        };
3432        let Ok(value) = fields.field_by_name(&profile.name) else {
3433            continue;
3434        };
3435        let Ok(counts) = fields.field_by_name(COUNT) else {
3436            continue;
3437        };
3438        profile.dominant_value = value
3439            .get(0)
3440            .ok()
3441            .filter(|value| !value.is_null())
3442            .map(|value| crate::exact::str_value(&value).to_string());
3443        profile.dominant_count = counts
3444            .get(0)
3445            .ok()
3446            .and_then(|value| value.try_extract::<u64>().ok())
3447            .map(|count| count as usize);
3448    }
3449    Ok(())
3450}
3451
3452fn profile_category_variants_lazy(
3453    lf: &LazyFrame,
3454    schema: &Schema,
3455    polars_streaming: bool,
3456) -> Result<Vec<CategoryVariantGroup>> {
3457    let normalized_name = "__quality_normalized";
3458    let original_name = "__quality_original";
3459    let count_name = "__quality_variant_rows";
3460    let variant_count_name = "__quality_variant_count";
3461    let mut result = Vec::new();
3462    for (name, dtype) in schema.iter() {
3463        if !matches!(dtype, DataType::String | DataType::Categorical(..)) {
3464            continue;
3465        }
3466        let original = text_expr(col(name.as_str()), dtype);
3467        let normalized = original
3468            .clone()
3469            .str()
3470            .strip_chars(lit(LiteralValue::untyped_null()))
3471            .str()
3472            .to_lowercase();
3473        let variant_count = col(original_name)
3474            .n_unique()
3475            .over([col(normalized_name)])?
3476            .alias(variant_count_name);
3477        let query = lf
3478            .clone()
3479            .filter(original.clone().is_not_null())
3480            .select([
3481                normalized.alias(normalized_name),
3482                original.alias(original_name),
3483            ])
3484            .group_by([col(normalized_name), col(original_name)])
3485            .agg([len().alias(count_name)])
3486            .with_columns([variant_count])
3487            .filter(col(variant_count_name).gt(lit(1u32)))
3488            .limit(1_001);
3489        let groups = collect_lazy(query, polars_streaming).map_err(Report::from)?;
3490        let complete = groups.height() <= 1_000;
3491        let mut by_normalized = BTreeMap::<String, Vec<(String, usize)>>::new();
3492        for row in 0..groups.height().min(1_000) {
3493            let Some(normalized) = string_value_at(&groups, normalized_name, row) else {
3494                continue;
3495            };
3496            let Some(original) = string_value_at(&groups, original_name, row) else {
3497                continue;
3498            };
3499            let count = usize_value_at(&groups, count_name, row);
3500            by_normalized
3501                .entry(normalized)
3502                .or_default()
3503                .push((original, count));
3504        }
3505        for (normalized, variants) in by_normalized {
3506            if variants.len() < 2 {
3507                continue;
3508            }
3509            let rows_involved = variants.iter().map(|(_, count)| count).sum();
3510            result.push(CategoryVariantGroup {
3511                column: name.to_string(),
3512                normalized,
3513                variants,
3514                rows_involved,
3515                complete,
3516            });
3517            if result.len() >= 100 {
3518                return Ok(result);
3519            }
3520        }
3521    }
3522    Ok(result)
3523}
3524
3525fn profile_identity_lazy(
3526    lf: &LazyFrame,
3527    schema: &Schema,
3528    total_rows: usize,
3529    precision: QualityPrecision,
3530    polars_streaming: bool,
3531) -> Result<IdentityProfile> {
3532    let keys = schema
3533        .iter_names()
3534        .map(|name| col(name.as_str()))
3535        .collect::<Vec<_>>();
3536    let duplicate_count = "__quality_duplicate_count";
3537    let grouped = lf
3538        .clone()
3539        .group_by(keys)
3540        .agg([len().alias(duplicate_count)])
3541        .filter(col(duplicate_count).gt(lit(1u32)))
3542        .select([
3543            len().alias("duplicate_groups"),
3544            (col(duplicate_count) - lit(1u32)).sum().alias("extra_rows"),
3545            col(duplicate_count).sum().alias("rows_involved"),
3546        ]);
3547    let summary = collect_lazy(grouped, polars_streaming).map_err(Report::from)?;
3548    Ok(IdentityProfile {
3549        duplicate_groups: usize_value(&summary, "duplicate_groups"),
3550        extra_rows: usize_value(&summary, "extra_rows"),
3551        rows_involved: usize_value(&summary, "rows_involved"),
3552        evaluated_rows: total_rows,
3553        precision,
3554        examples: Vec::new(),
3555    })
3556}
3557
3558const DUPLICATE_COPIES: &str = "__datui_quality_copies";
3559
3560/// Groups of rows identical in every one of `keys`, with how many copies each has,
3561/// most copies first and then first seen first: the grouping the duplicate check
3562/// counts with.
3563fn duplicate_groups(lf: LazyFrame, keys: &[PlSmallStr]) -> LazyFrame {
3564    lf.group_by_stable(keys.iter().map(|key| col(key.clone())).collect::<Vec<_>>())
3565        .agg([len().alias(DUPLICATE_COPIES)])
3566        .filter(col(DUPLICATE_COPIES).gt(lit(1u32)))
3567        .sort(
3568            [DUPLICATE_COPIES],
3569            SortMultipleOptions::default()
3570                .with_order_descending(true)
3571                .with_maintain_order(true),
3572        )
3573}
3574
3575/// The rows the duplicate check counted: every row equal to another in every one of
3576/// `keys`, copies together, most copies first. One pass, grouping as the check did.
3577///
3578/// Copies are equal in every key, so each group's key is its rows: it is repeated as
3579/// many times as it occurs rather than looked up in a second read.
3580pub fn duplicate_rows(
3581    lf: LazyFrame,
3582    keys: &[PlSmallStr],
3583    polars_streaming: bool,
3584) -> Result<DataFrame> {
3585    let groups =
3586        collect_lazy(duplicate_groups(lf, keys), polars_streaming).map_err(Report::from)?;
3587    let copies = groups
3588        .column(DUPLICATE_COPIES)?
3589        .cast(&DataType::UInt64)?
3590        .u64()?
3591        .into_no_null_iter()
3592        .collect::<Vec<_>>();
3593    let mut take = Vec::with_capacity(copies.iter().sum::<u64>() as usize);
3594    for (group, copies) in copies.into_iter().enumerate() {
3595        take.extend(std::iter::repeat_n(group as IdxSize, copies as usize));
3596    }
3597    let rows = groups.drop(DUPLICATE_COPIES)?;
3598    Ok(rows.take(&IdxCa::from_vec(PlSmallStr::EMPTY, take))?)
3599}
3600
3601/// The most copied groups of rows kept in memory, rendered for the detail.
3602fn duplicate_examples(
3603    lf: &LazyFrame,
3604    schema: &Schema,
3605    polars_streaming: bool,
3606) -> Result<Vec<DuplicateExample>> {
3607    let keys = schema.iter_names().cloned().collect::<Vec<_>>();
3608    let groups = collect_lazy(
3609        duplicate_groups(lf.clone(), &keys).limit(MAX_FINDING_EXAMPLES as IdxSize),
3610        polars_streaming,
3611    )
3612    .map_err(Report::from)?;
3613    Ok((0..groups.height())
3614        .map(|row| DuplicateExample {
3615            copies: usize_value_at(&groups, DUPLICATE_COPIES, row),
3616            values: keys
3617                .iter()
3618                .map(|key| {
3619                    groups
3620                        .column(key)
3621                        .and_then(|column| column.get(row))
3622                        .map(|value| example_text(&value))
3623                        .unwrap_or_else(|_| "null".to_string())
3624                })
3625                .collect(),
3626        })
3627        .collect())
3628}
3629
3630/// A value as the detail shows it: text quoted and cut, a null named.
3631fn example_text(value: &AnyValue<'_>) -> String {
3632    match value {
3633        AnyValue::Null => "null".to_string(),
3634        AnyValue::String(text) => crate::quality_report::quoted(text, 24),
3635        AnyValue::StringOwned(text) => crate::quality_report::quoted(text, 24),
3636        other => {
3637            let text = crate::exact::str_value(other).to_string();
3638            if crate::glyphs::display_width(&text) > 24 {
3639                format!(
3640                    "{}{}",
3641                    crate::glyphs::take_columns(&text, 23),
3642                    crate::glyphs::get().ellipsis
3643                )
3644            } else {
3645                text
3646            }
3647        }
3648    }
3649}
3650
3651/// A few distinct values behind each text finding a sample can show: text its
3652/// reading does not parse, and text its time format does not read.
3653fn finding_examples(
3654    lf: &LazyFrame,
3655    columns: &[ColumnQualityProfile],
3656    observations: &[QualityObservation],
3657    polars_streaming: bool,
3658) -> Result<Vec<FindingExamples>> {
3659    let mut examples = Vec::new();
3660    for observation in observations {
3661        let profile = columns
3662            .iter()
3663            .find(|profile| profile.name == observation.column);
3664        let failed = match observation.kind {
3665            ObservationKind::ParseableText => profile.and_then(unparsed_text),
3666            ObservationKind::UnparsedTime => observation
3667                .time_format
3668                .as_ref()
3669                .map(TimeInterpretation::unparsed),
3670            _ => None,
3671        };
3672        let (Some(failed), Some(profile)) = (failed, profile) else {
3673            continue;
3674        };
3675        // The first failures, told apart here: a few hundred bound the work, and a
3676        // value repeated that often is the example anyway.
3677        let values = text_expr(col(observation.column.as_str()), &profile.dtype)
3678            .filter(failed)
3679            .head(Some(256))
3680            .alias("values");
3681        let found =
3682            collect_lazy(lf.clone().select([values]), polars_streaming).map_err(Report::from)?;
3683        let mut values = Vec::new();
3684        for value in (0..found.height()).filter_map(|row| string_value_at(&found, "values", row)) {
3685            let value = crate::quality_report::quoted(&value, 24);
3686            if !values.contains(&value) {
3687                values.push(value);
3688            }
3689            if values.len() == MAX_FINDING_EXAMPLES {
3690                break;
3691            }
3692        }
3693        if !values.is_empty() {
3694            examples.push(FindingExamples {
3695                kind: observation.kind,
3696                column: observation.column.clone(),
3697                values,
3698            });
3699        }
3700    }
3701    Ok(examples)
3702}
3703
3704fn visible_schema(schema: &Schema, source: Option<&QualitySourceContext>) -> Schema {
3705    let mut visible = Schema::with_capacity(schema.len());
3706    for (name, dtype) in schema.iter() {
3707        if source.is_some_and(|context| name.as_str() == context.row_index_column) {
3708            continue;
3709        }
3710        visible.insert(name.clone(), dtype.clone());
3711    }
3712    visible
3713}
3714
3715fn attach_source_file(
3716    mut df: DataFrame,
3717    source: Option<&QualitySourceContext>,
3718) -> Result<DataFrame> {
3719    let Some(source) = source else {
3720        return Ok(df);
3721    };
3722    let rows = df.drop_in_place(&source.row_index_column)?;
3723    let rows = rows.u32()?;
3724    let names: Vec<Option<&str>> = rows
3725        .iter()
3726        .map(|row| {
3727            let row = row? as usize;
3728            let file = source
3729                .file_starts
3730                .partition_point(|start| *start <= row)
3731                .saturating_sub(1);
3732            source.file_names.get(file).map(String::as_str)
3733        })
3734        .collect();
3735    df.with_column(Column::new(QUALITY_SOURCE_FILE_COLUMN.into(), names))?;
3736    Ok(df)
3737}
3738
3739fn profile_columns(
3740    df: &DataFrame,
3741    schema: &Schema,
3742    polars_streaming: bool,
3743) -> Result<Vec<ColumnQualityProfile>> {
3744    let aggregate = collect_lazy(
3745        df.clone().lazy().select(build_profile_exprs(schema)),
3746        polars_streaming,
3747    )
3748    .map_err(Report::from)?;
3749    Ok(parse_profiles(&aggregate, schema, df.height()))
3750}
3751
3752fn identity_observations(
3753    identity: &IdentityProfile,
3754    variants: &[CategoryVariantGroup],
3755) -> Vec<QualityObservation> {
3756    let mut observations = Vec::new();
3757    if identity.duplicate_groups > 0 {
3758        observations.push(QualityObservation {
3759            kind: ObservationKind::DuplicateRows,
3760            column: "all columns".to_string(),
3761            affected_rows: identity.rows_involved,
3762            evaluated_rows: identity.evaluated_rows,
3763            fact: format!(
3764                "{} groups; {} extra rows ({})",
3765                identity.duplicate_groups,
3766                identity.extra_rows,
3767                identity.precision.label()
3768            ),
3769            normalized_category: None,
3770            files: Vec::new(),
3771            time_format: None,
3772            full_scale: None,
3773        });
3774    }
3775    observations.extend(variants.iter().map(|group| QualityObservation {
3776        kind: ObservationKind::CategoryVariants,
3777        column: group.column.clone(),
3778        affected_rows: group.rows_involved,
3779        evaluated_rows: identity.evaluated_rows,
3780        fact: format!(
3781            "{}{} variants normalize to {:?}",
3782            if group.complete { "" } else { "at least " },
3783            group.variants.len(),
3784            group.normalized
3785        ),
3786        normalized_category: Some(group.normalized.clone()),
3787        files: Vec::new(),
3788        time_format: None,
3789        full_scale: None,
3790    }));
3791    observations
3792}
3793
3794#[derive(Debug)]
3795struct SegmentRows {
3796    label: String,
3797    indices: Vec<u32>,
3798}
3799
3800fn segment_rows(
3801    df: &DataFrame,
3802    plan: &DataQualityPlan,
3803    sample_positions: Option<&[IdxSize]>,
3804) -> Result<Vec<SegmentRows>> {
3805    let all_rows = || SegmentRows {
3806        label: "current view".to_string(),
3807        indices: (0..df.height() as u32).collect(),
3808    };
3809    let groups = match &plan.grain {
3810        QualityGrain::Dataset => vec![all_rows()],
3811        QualityGrain::RowChunks(size) => {
3812            let size = (*size).max(1);
3813            let mut chunks = BTreeMap::<usize, Vec<u32>>::new();
3814            for row in 0..df.height() {
3815                let position = sample_positions
3816                    .and_then(|positions| positions.get(row))
3817                    .copied()
3818                    .unwrap_or(row as IdxSize) as usize;
3819                chunks.entry(position / size).or_default().push(row as u32);
3820            }
3821            chunks
3822                .into_iter()
3823                .map(|(chunk, indices)| SegmentRows {
3824                    label: format!(
3825                        "rows {}-{}",
3826                        chunk.saturating_mul(size) + 1,
3827                        (chunk + 1).saturating_mul(size)
3828                    ),
3829                    indices,
3830                })
3831                .collect()
3832        }
3833        QualityGrain::Partition(column) => group_by_value(df, column, &format!("{column}="))?,
3834        QualityGrain::TimeWindows { column, every } => {
3835            group_by_time_window(df, plan, column, every)?
3836        }
3837        QualityGrain::File => {
3838            if df.column(QUALITY_SOURCE_FILE_COLUMN).is_ok() {
3839                group_by_value(df, QUALITY_SOURCE_FILE_COLUMN, "file ")?
3840            } else {
3841                vec![SegmentRows {
3842                    label: "file mapping unavailable for this view".to_string(),
3843                    indices: (0..df.height() as u32).collect(),
3844                }]
3845            }
3846        }
3847    };
3848    Ok(groups)
3849}
3850
3851fn group_by_value(df: &DataFrame, column: &str, prefix: &str) -> Result<Vec<SegmentRows>> {
3852    let values = df.column(column)?;
3853    let mut groups: BTreeMap<String, Vec<u32>> = BTreeMap::new();
3854    let mut missing = Vec::new();
3855    for row in 0..df.height() {
3856        let value = values.get(row)?;
3857        if value.is_null() {
3858            missing.push(row as u32);
3859        } else {
3860            groups
3861                .entry(format!("{prefix}{}", crate::exact::str_value(&value)))
3862                .or_default()
3863                .push(row as u32);
3864        }
3865    }
3866    let mut result: Vec<SegmentRows> = groups
3867        .into_iter()
3868        .map(|(label, indices)| SegmentRows { label, indices })
3869        .collect();
3870    // Rows the grain could not place carry no order, so they follow the ones it
3871    // could — the same rule the scanned path applies. Sorting "∅" by codepoint
3872    // would put it before any value that outranks U+2205.
3873    if !missing.is_empty() {
3874        result.push(SegmentRows {
3875            label: format!("{prefix}∅"),
3876            indices: missing,
3877        });
3878    }
3879    Ok(result)
3880}
3881
3882/// Where a row's window starts. Both the sampled and the full-scan path bucket
3883/// through this one expression, so a week never starts on a different day
3884/// depending on how much of it was read. A date past the calendar's range falls
3885/// in no window, as a null does: truncating it overflows.
3886fn time_window_start(value: Expr, every: &str) -> Expr {
3887    value
3888        .map(
3889            |c| {
3890                Ok(
3891                    crate::exact::calendar_without_out_of_range(c.as_materialized_series())?
3892                        .map_or(c, Column::from),
3893                )
3894            },
3895            |_, field| Ok(field.clone()),
3896        )
3897        .cast(DataType::Datetime(TimeUnit::Microseconds, None))
3898        .dt()
3899        .truncate(lit(every.to_string()))
3900}
3901
3902fn group_by_time_window(
3903    df: &DataFrame,
3904    plan: &DataQualityPlan,
3905    column: &str,
3906    every: &str,
3907) -> Result<Vec<SegmentRows>> {
3908    let starts = df
3909        .clone()
3910        .lazy()
3911        .select([time_window_start(plan.time_value(column), every).alias(QUALITY_WINDOW_START)])
3912        .collect()?;
3913    let starts = starts.column(QUALITY_WINDOW_START)?;
3914    let mut groups: BTreeMap<String, Vec<u32>> = BTreeMap::new();
3915    let mut missing = Vec::new();
3916    for row in 0..df.height() {
3917        let value = starts.get(row)?;
3918        if value.is_null() {
3919            missing.push(row as u32);
3920        } else {
3921            groups
3922                .entry(crate::exact::str_value(&value).into_owned())
3923                .or_default()
3924                .push(row as u32);
3925        }
3926    }
3927    let mut result: Vec<SegmentRows> = groups
3928        .into_iter()
3929        .map(|(start, indices)| SegmentRows {
3930            label: time_window_label(column, every, Some(&start)),
3931            indices,
3932        })
3933        .collect();
3934    if !missing.is_empty() {
3935        result.push(SegmentRows {
3936            label: time_window_label(column, every, None),
3937            indices: missing,
3938        });
3939    }
3940    Ok(result)
3941}
3942
3943/// A window by where it starts, to the precision its width needs: an hour to the
3944/// minute, a day as its date, a week as the date it starts, a month as the month.
3945pub(crate) fn time_window_label(column: &str, every: &str, start: Option<&str>) -> String {
3946    let Some(start) = start else {
3947        return format!("{column} ∅");
3948    };
3949    let prefix = |length: usize| start.get(..length).unwrap_or(start).to_string();
3950    match every {
3951        "1h" => prefix(16),
3952        "1d" => prefix(10),
3953        "1w" => format!("week of {}", prefix(10)),
3954        "1mo" => prefix(7),
3955        _ => format!("{start} / {every}"),
3956    }
3957}
3958
3959fn value_epoch_micros(value: AnyValue<'_>) -> Option<i64> {
3960    // A one-row segment's column can be a scalar, whose values come back owned.
3961    match value.as_borrowed() {
3962        AnyValue::Date(days) => Some(i64::from(days) * 86_400_000_000),
3963        AnyValue::Datetime(value, TimeUnit::Nanoseconds, _) => Some(value / 1_000),
3964        AnyValue::Datetime(value, TimeUnit::Microseconds, _) => Some(value),
3965        AnyValue::Datetime(value, TimeUnit::Milliseconds, _) => Some(value * 1_000),
3966        _ => None,
3967    }
3968}
3969
3970fn take_rows(df: &DataFrame, indices: &[u32]) -> PolarsResult<DataFrame> {
3971    df.take(&UInt32Chunked::new("quality_rows".into(), indices.to_vec()))
3972}
3973
3974struct SegmentSampleProvenance<'a> {
3975    positions: Option<&'a [IdxSize]>,
3976    totals: &'a BTreeMap<String, usize>,
3977}
3978
3979/// Segment sizes known without reading them: a file's rows from its footer, when
3980/// the scope holds whole files, and a row chunk's from the scope's size. Others are
3981/// unknown on a sample, and are left unknown rather than estimated.
3982///
3983/// Partitions and time windows are counted instead: a grouped count reads only
3984/// the grain's column, a small read beside the sample's, and a day whose rows fell
3985/// by half is the first thing a daily check is for.
3986fn counted_segment_totals(
3987    lf: &LazyFrame,
3988    plan: &DataQualityPlan,
3989    polars_streaming: bool,
3990) -> Result<SegmentCounts> {
3991    const KEY: &str = "__quality_count_key";
3992    const ROWS: &str = "__quality_count_rows";
3993    let Some(key) = segment_count_key(plan) else {
3994        return Ok(SegmentCounts::new());
3995    };
3996    let counts = collect_lazy(
3997        lf.clone()
3998            .select([key.alias(KEY)])
3999            .group_by([col(KEY)])
4000            .agg([len().alias(ROWS)]),
4001        polars_streaming,
4002    )
4003    .map_err(Report::from)?;
4004    let keys = counts.column(KEY)?;
4005    let mut totals = BTreeMap::new();
4006    for row in 0..counts.height() {
4007        let raw = keys.get(row)?;
4008        // Keyed as the key reads, as a streamed count keys it: named as a segment only
4009        // when a run asks, so a finer window's count can be summed into a coarser one.
4010        let raw = (!raw.is_null()).then(|| crate::exact::str_value(&raw).into_owned());
4011        totals.insert(raw, usize_value_at(&counts, ROWS, row));
4012    }
4013    Ok(totals)
4014}
4015
4016fn known_segment_totals(
4017    plan: &DataQualityPlan,
4018    total_rows: Option<usize>,
4019    source: Option<&QualitySourceContext>,
4020) -> BTreeMap<String, usize> {
4021    let mut totals = BTreeMap::new();
4022    match &plan.grain {
4023        QualityGrain::File => {
4024            let Some(source) = source else {
4025                return totals;
4026            };
4027            let whole_files = matches!(plan.scope, QualityScope::SourceFiles(_))
4028                || total_rows == Some(source.dataset_rows);
4029            if whole_files {
4030                for (index, name) in source.file_names.iter().enumerate() {
4031                    totals.insert(format!("file {name}"), source.file_rows(index));
4032                }
4033            }
4034        }
4035        QualityGrain::RowChunks(size) => {
4036            let (Some(total), size) = (total_rows, (*size).max(1)) else {
4037                return totals;
4038            };
4039            for chunk in 0..total.div_ceil(size) {
4040                let start = chunk * size;
4041                totals.insert(
4042                    format!("rows {}-{}", start + 1, (chunk + 1).saturating_mul(size)),
4043                    size.min(total - start),
4044                );
4045            }
4046        }
4047        _ => {}
4048    }
4049    totals
4050}
4051
4052fn profile_segments(
4053    df: &DataFrame,
4054    total_rows: Option<usize>,
4055    plan: &DataQualityPlan,
4056    precision: QualityPrecision,
4057    schema: &Schema,
4058    sample: SegmentSampleProvenance<'_>,
4059    polars_streaming: bool,
4060) -> Result<(Vec<SegmentQualityProfile>, Vec<UnsampledSegment>)> {
4061    let groups = segment_rows(df, plan, sample.positions)?;
4062    // Every segment in one grouped query, keyed by the segment each row fell in.
4063    // A query per segment is thousands of them for a daily grain over years, and
4064    // each pays Polars' planning cost for a few dozen rows.
4065    let mut segment_of = vec![0u32; df.height()];
4066    for (index, group) in groups.iter().enumerate() {
4067        for row in &group.indices {
4068            segment_of[*row as usize] = index as u32;
4069        }
4070    }
4071    const SEGMENT: &str = "__quality_segment_index";
4072    let mut keyed = df.clone();
4073    keyed.with_column(Column::new(SEGMENT.into(), segment_of))?;
4074    let grouped = collect_lazy(
4075        keyed
4076            .lazy()
4077            .group_by([col(SEGMENT)])
4078            .agg(build_profile_exprs(schema)),
4079        polars_streaming,
4080    )
4081    .map_err(Report::from)?;
4082    let mut by_segment = vec![None; groups.len()];
4083    for row in 0..grouped.height() {
4084        let index = usize_value_at(&grouped, SEGMENT, row);
4085        if let Some(slot) = by_segment.get_mut(index) {
4086            *slot = Some(row);
4087        }
4088    }
4089    let mut profiles = Vec::with_capacity(groups.len());
4090    for (group, row) in groups.into_iter().zip(by_segment) {
4091        let evaluated_rows = group.indices.len();
4092        let Some(row) = row else {
4093            continue;
4094        };
4095        let columns = parse_profiles_at(&grouped, schema, evaluated_rows, row);
4096        let null_cells = columns
4097            .iter()
4098            .map(|column| column.null_count)
4099            .sum::<usize>();
4100        let denominator = evaluated_rows.saturating_mul(columns.len());
4101        let known_segment_rows = sample.totals.get(&group.label);
4102        profiles.push(SegmentQualityProfile {
4103            label: group.label,
4104            total_rows: if let Some(total) = known_segment_rows {
4105                Some(*total)
4106            } else if matches!(plan.grain, QualityGrain::Dataset) {
4107                total_rows
4108            } else if precision == QualityPrecision::Exact {
4109                Some(evaluated_rows)
4110            } else {
4111                None
4112            },
4113            evaluated_rows,
4114            columns,
4115            null_cells,
4116            null_rate: rate(null_cells, denominator),
4117            compared_with: None,
4118            largest_change: None,
4119            change_size: None,
4120        });
4121    }
4122    order_segments(&mut profiles);
4123    apply_comparisons(
4124        &mut profiles,
4125        plan.comparison,
4126        plan.baseline_segment.as_deref(),
4127        precision,
4128    );
4129    // What the count found and the sample did not: kept apart, so a segment with
4130    // rows the sample missed is never read as one with none.
4131    let drawn = profiles
4132        .iter()
4133        .map(|profile| profile.label.as_str())
4134        .collect::<std::collections::HashSet<_>>();
4135    let mut unsampled = sample
4136        .totals
4137        .iter()
4138        .filter(|(label, rows)| **rows > 0 && !drawn.contains(label.as_str()))
4139        .map(|(label, rows)| UnsampledSegment {
4140            label: label.clone(),
4141            total_rows: *rows,
4142        })
4143        .collect::<Vec<_>>();
4144    unsampled.sort_by(|left, right| segment_cmp(&left.label, &right.label));
4145    Ok((profiles, unsampled))
4146}
4147
4148/// Segments in the order their names count: year=9 before year=10, part-2 before
4149/// part-10, and the rows no segment could place (`∅`) last. "Previous" means the
4150/// segment before in this order, so it has to be the order a person would read.
4151fn order_segments(segments: &mut [SegmentQualityProfile]) {
4152    segments.sort_by(|left, right| segment_cmp(&left.label, &right.label));
4153}
4154
4155/// The order of two segments by their labels, as [`order_segments`] puts them.
4156pub(crate) fn segment_cmp(left: &str, right: &str) -> std::cmp::Ordering {
4157    left.ends_with('∅')
4158        .cmp(&right.ends_with('∅'))
4159        .then_with(|| natural_cmp(left, right))
4160}
4161
4162/// Text compared with its runs of digits compared as numbers.
4163fn natural_cmp(left: &str, right: &str) -> std::cmp::Ordering {
4164    use std::cmp::Ordering;
4165    let (mut left, mut right) = (left, right);
4166    loop {
4167        let (Some(l), Some(r)) = (left.chars().next(), right.chars().next()) else {
4168            return left.len().cmp(&right.len());
4169        };
4170        if l.is_ascii_digit() && r.is_ascii_digit() {
4171            let digits = |text: &str| {
4172                text.find(|c: char| !c.is_ascii_digit())
4173                    .unwrap_or(text.len())
4174            };
4175            let (l_end, r_end) = (digits(left), digits(right));
4176            let (l_num, r_num) = (
4177                left[..l_end].trim_start_matches('0'),
4178                right[..r_end].trim_start_matches('0'),
4179            );
4180            let order = l_num.len().cmp(&r_num.len()).then_with(|| l_num.cmp(r_num));
4181            if order != Ordering::Equal {
4182                return order;
4183            }
4184            left = &left[l_end..];
4185            right = &right[r_end..];
4186        } else {
4187            if l != r {
4188                return l.cmp(&r);
4189            }
4190            left = &left[l.len_utf8()..];
4191            right = &right[r.len_utf8()..];
4192        }
4193    }
4194}
4195
4196fn profile_segments_lazy(
4197    lf: &LazyFrame,
4198    total_rows: usize,
4199    plan: &DataQualityPlan,
4200    source: Option<&QualitySourceContext>,
4201    schema: &Schema,
4202    polars_streaming: bool,
4203) -> Result<Vec<SegmentQualityProfile>> {
4204    if unsegmented(plan, source) {
4205        let aggregate = collect_lazy(
4206            lf.clone().select(build_profile_exprs(schema)),
4207            polars_streaming,
4208        )
4209        .map_err(Report::from)?;
4210        let columns = parse_profiles(&aggregate, schema, total_rows);
4211        return Ok(vec![whole_segment(
4212            plan,
4213            total_rows,
4214            &columns,
4215            schema.len(),
4216        )]);
4217    }
4218
4219    let (grouped_lf, group) = grouped_frame(lf, plan, source)?;
4220    let mut aggregates = vec![len().alias("__quality_segment_rows")];
4221    aggregates.extend(build_profile_exprs(schema));
4222    let grouped = collect_lazy(
4223        grouped_lf
4224            .group_by([group.alias("__quality_segment")])
4225            .agg(aggregates),
4226        polars_streaming,
4227    )
4228    .map_err(Report::from)?;
4229    let mut segments = Vec::with_capacity(grouped.height());
4230    let mut unassigned = Vec::with_capacity(grouped.height());
4231    for row in 0..grouped.height() {
4232        let evaluated_rows = usize_value_at(&grouped, "__quality_segment_rows", row);
4233        let columns = parse_profiles_at(&grouped, schema, evaluated_rows, row);
4234        let null_cells = columns
4235            .iter()
4236            .map(|column| column.null_count)
4237            .sum::<usize>();
4238        let denominator = evaluated_rows.saturating_mul(schema.len());
4239        let raw_label = string_value_at(&grouped, "__quality_segment", row);
4240        unassigned.push(raw_label.is_none());
4241        segments.push(SegmentQualityProfile {
4242            label: segment_label(&plan.grain, raw_label.as_deref()),
4243            total_rows: Some(evaluated_rows),
4244            evaluated_rows,
4245            columns,
4246            null_cells,
4247            null_rate: rate(null_cells, denominator),
4248            compared_with: None,
4249            largest_change: None,
4250            change_size: None,
4251        });
4252    }
4253    // Rows the grain could not place carry no order, so they follow the ones it could.
4254    let mut ordered = unassigned.into_iter().zip(segments).collect::<Vec<_>>();
4255    ordered.sort_by(|left, right| {
4256        left.0
4257            .cmp(&right.0)
4258            .then_with(|| natural_cmp(&left.1.label, &right.1.label))
4259    });
4260    let mut segments = ordered
4261        .into_iter()
4262        .map(|(_, segment)| segment)
4263        .collect::<Vec<_>>();
4264    if matches!(plan.grain, QualityGrain::RowChunks(_)) {
4265        for segment in &mut segments {
4266            segment.label = pretty_chunk_label(&segment.label);
4267        }
4268    }
4269    apply_comparisons(
4270        &mut segments,
4271        plan.comparison,
4272        plan.baseline_segment.as_deref(),
4273        QualityPrecision::Exact,
4274    );
4275    Ok(segments)
4276}
4277
4278/// Whether a full run's grain leaves the scope whole: the dataset grain, or files
4279/// where the view has lost which file a row came from.
4280fn unsegmented(plan: &DataQualityPlan, source: Option<&QualitySourceContext>) -> bool {
4281    matches!(plan.grain, QualityGrain::Dataset)
4282        || matches!(plan.grain, QualityGrain::File) && source.is_none()
4283}
4284
4285/// The scope as its one segment, from its columns' profile.
4286fn whole_segment(
4287    plan: &DataQualityPlan,
4288    total_rows: usize,
4289    columns: &[ColumnQualityProfile],
4290    column_count: usize,
4291) -> SegmentQualityProfile {
4292    let null_cells = columns
4293        .iter()
4294        .map(|column| column.null_count)
4295        .sum::<usize>();
4296    let denominator = total_rows.saturating_mul(column_count);
4297    SegmentQualityProfile {
4298        label: if matches!(plan.grain, QualityGrain::File) {
4299            "file mapping unavailable for this view".to_string()
4300        } else {
4301            "current view".to_string()
4302        },
4303        total_rows: Some(total_rows),
4304        evaluated_rows: total_rows,
4305        columns: columns.to_vec(),
4306        null_cells,
4307        null_rate: rate(null_cells, denominator),
4308        compared_with: None,
4309        largest_change: None,
4310        change_size: None,
4311    }
4312}
4313
4314fn grouped_frame(
4315    lf: &LazyFrame,
4316    plan: &DataQualityPlan,
4317    source: Option<&QualitySourceContext>,
4318) -> Result<(LazyFrame, Expr)> {
4319    match &plan.grain {
4320        QualityGrain::Dataset => Err(color_eyre::eyre::eyre!(
4321            "dataset grain does not need grouping"
4322        )),
4323        QualityGrain::Partition(column) => Ok((lf.clone(), col(column))),
4324        QualityGrain::RowChunks(size) => {
4325            let row = "__datui_quality_row";
4326            Ok((
4327                lf.clone().with_row_index(row, None),
4328                col(row).cast(DataType::UInt64) / lit((*size).max(1) as u64),
4329            ))
4330        }
4331        QualityGrain::TimeWindows { column, every } => Ok((
4332            lf.clone(),
4333            time_window_start(plan.time_value(column), every),
4334        )),
4335        QualityGrain::File => {
4336            let source = source
4337                .ok_or_else(|| color_eyre::eyre::eyre!("source-file mapping is unavailable"))?;
4338            let mut file = lit("unknown");
4339            for (start, name) in source.file_starts.iter().zip(source.file_names.iter()) {
4340                file = when(col(&source.row_index_column).gt_eq(lit(*start as u32)))
4341                    .then(lit(name.clone()))
4342                    .otherwise(file);
4343            }
4344            Ok((lf.clone(), file))
4345        }
4346    }
4347}
4348
4349fn pretty_chunk_label(label: &str) -> String {
4350    let Some(range) = label.strip_prefix("rows ") else {
4351        return label.to_string();
4352    };
4353    let Some((start, end)) = range.split_once('-') else {
4354        return label.to_string();
4355    };
4356    let start = start.trim_start_matches('0');
4357    let end = end.trim_start_matches('0');
4358    format!(
4359        "rows {}-{}",
4360        if start.is_empty() { "0" } else { start },
4361        if end.is_empty() { "0" } else { end }
4362    )
4363}
4364
4365fn segment_label(grain: &QualityGrain, raw: Option<&str>) -> String {
4366    match grain {
4367        QualityGrain::RowChunks(size) => {
4368            let raw = raw.unwrap_or("∅");
4369            raw.parse::<usize>()
4370                .map(|chunk| {
4371                    let start = chunk.saturating_mul(*size) + 1;
4372                    let end = start.saturating_add(*size).saturating_sub(1);
4373                    format!("rows {start:012}-{end:012}")
4374                })
4375                .unwrap_or_else(|_| format!("rows {raw}"))
4376        }
4377        QualityGrain::Partition(column) => format!("{column}={}", raw.unwrap_or("∅")),
4378        QualityGrain::TimeWindows { column, every } => time_window_label(column, every, raw),
4379        QualityGrain::File => format!("file {}", raw.unwrap_or("∅")),
4380        QualityGrain::Dataset => "current view".to_string(),
4381    }
4382}
4383
4384fn apply_comparisons(
4385    segments: &mut [SegmentQualityProfile],
4386    comparison: QualityComparison,
4387    baseline_segment: Option<&str>,
4388    precision: QualityPrecision,
4389) {
4390    for segment in segments.iter_mut() {
4391        segment.compared_with = None;
4392        segment.largest_change = None;
4393        segment.change_size = None;
4394    }
4395    let baseline_index = baseline_segment
4396        .and_then(|label| segments.iter().position(|segment| segment.label == label))
4397        .or_else(|| baseline_segment.is_none().then_some(0));
4398    if comparison == QualityComparison::Baseline && baseline_index.is_none() {
4399        for segment in segments {
4400            segment.largest_change = Some("selected baseline unavailable".to_string());
4401        }
4402        return;
4403    }
4404    for index in 0..segments.len() {
4405        let compared = match comparison {
4406            QualityComparison::None => None,
4407            QualityComparison::Previous if index > 0 => Some(index - 1),
4408            QualityComparison::Baseline if Some(index) != baseline_index => baseline_index,
4409            QualityComparison::Previous | QualityComparison::Baseline => None,
4410        };
4411        if let Some(other) = compared {
4412            let change = largest_material_change(&segments[index], &segments[other], precision);
4413            segments[index].compared_with = Some(segments[other].label.clone());
4414            if let Some((what, size)) = change {
4415                segments[index].largest_change = Some(what);
4416                segments[index].change_size = Some(size);
4417            }
4418        }
4419    }
4420}
4421
4422/// How far a measurement has to move between segments before it is worth naming, in
4423/// percentage points.
4424pub(crate) const MATERIAL_CHANGE_PP: f64 = 1.0;
4425
4426/// How many standard errors apart two sampled rates must be before the difference
4427/// is named. A segment is dozens of columns and measures, and a daily grain is
4428/// thousands of segments: at three, sampling alone would name a change most days.
4429const NOISE_Z: f64 = 4.0;
4430
4431/// Whether rates `a` of `n_a` rows and `b` of `n_b` rows differ by more than two
4432/// samples of those sizes would by chance (a two-proportion z-test).
4433pub fn beyond_noise(a: f64, n_a: usize, b: f64, n_b: usize) -> bool {
4434    if n_a == 0 || n_b == 0 {
4435        return false;
4436    }
4437    let (n_a, n_b) = (n_a as f64, n_b as f64);
4438    let pooled = (a * n_a + b * n_b) / (n_a + n_b);
4439    let error = (pooled * (1.0 - pooled) * (1.0 / n_a + 1.0 / n_b)).sqrt();
4440    error > 0.0 && (a - b).abs() / error >= NOISE_Z
4441}
4442
4443/// The rates a segment is compared on. A distinct share is not one of them: it
4444/// falls as a segment grows, so two segments of different sizes differ by it
4445/// whatever their data.
4446const CHANGE_MEASURES: [QualityMetric; 4] = [
4447    QualityMetric::NullRate,
4448    QualityMetric::EmptyRate,
4449    QualityMetric::WhitespaceRate,
4450    QualityMetric::NonFiniteRate,
4451];
4452
4453/// The clearest move between two segments, and its size.
4454///
4455/// #196 asks where a column's null rate or range shifts sharply, which is a
4456/// question about the sharpest single move rather than about the average of all of
4457/// them: one column going from never-null to always-null is the finding, and a mean
4458/// over sixty columns buries it. A row count that halved or doubled comes first:
4459/// for a feed split by day it is the loudest thing that can go wrong. On a sample,
4460/// a move is named only past sampling noise; a range that moved only on an exact
4461/// profile, since a sample's minimum and maximum move with the draw.
4462fn largest_material_change(
4463    segment: &SegmentQualityProfile,
4464    baseline: &SegmentQualityProfile,
4465    precision: QualityPrecision,
4466) -> Option<(String, f64)> {
4467    if let (Some(now), Some(before)) = (segment.total_rows, baseline.total_rows)
4468        && before > 0
4469    {
4470        let ratio = now as f64 / before as f64;
4471        if !(0.5..2.0).contains(&ratio) {
4472            let percent = (ratio - 1.0) * 100.0;
4473            return Some((
4474                format!("rows {} ({percent:+.0}%)", crate::numfmt::group_chrome(now)),
4475                percent.abs(),
4476            ));
4477        }
4478    }
4479    let sampled = precision != QualityPrecision::Exact;
4480    let mut largest: Option<(f64, String)> = None;
4481    let mut range: Option<String> = None;
4482    for (index, column) in segment.columns.iter().enumerate() {
4483        // Both profiles are built by walking the same schema, so the columns line up.
4484        // A linear search per column per segment is a square over the column count,
4485        // which is paid exactly where this feature is for: thousands of file segments
4486        // over hundreds of columns.
4487        let Some(prior) = baseline
4488            .columns
4489            .get(index)
4490            .filter(|other| other.name == column.name)
4491            .or_else(|| {
4492                baseline
4493                    .columns
4494                    .iter()
4495                    .find(|other| other.name == column.name)
4496            })
4497        else {
4498            continue;
4499        };
4500        for metric in CHANGE_MEASURES {
4501            let (Some(now), Some(before)) = (metric.value(column), metric.value(prior)) else {
4502                continue;
4503            };
4504            let change = (now - before) * 100.0;
4505            if change.abs() < MATERIAL_CHANGE_PP
4506                || sampled
4507                    && !beyond_noise(
4508                        now,
4509                        metric.denominator(column),
4510                        before,
4511                        metric.denominator(prior),
4512                    )
4513            {
4514                continue;
4515            }
4516            if largest
4517                .as_ref()
4518                .is_none_or(|(most, _)| change.abs() > most.abs())
4519            {
4520                largest = Some((change, format!("{} {}", column.name, metric.short_label())));
4521            }
4522        }
4523        if !sampled && range.is_none() && (column.min != prior.min || column.max != prior.max) {
4524            range = Some(format!(
4525                "{} range {} -> {}",
4526                column.name,
4527                range_label(prior),
4528                range_label(column)
4529            ));
4530        }
4531    }
4532    match (largest, range) {
4533        (Some((change, what)), _) => Some((format!("{what} {change:+.1} pp"), change.abs())),
4534        (None, Some(moved)) => Some((moved, 0.0)),
4535        (None, None) => None,
4536    }
4537}
4538
4539fn range_label(column: &ColumnQualityProfile) -> String {
4540    match (&column.min, &column.max) {
4541        (Some(min), Some(max)) => format!("{min}..{max}"),
4542        (Some(min), None) => format!("{min}.."),
4543        (None, Some(max)) => format!("..{max}"),
4544        (None, None) => "none".to_string(),
4545    }
4546}
4547
4548/// A role's column as a run reads it: the column's own name, where its time values
4549/// are, and where the rows are flagged whose text the column's format did not read.
4550struct TimedColumn {
4551    name: String,
4552    values: String,
4553    unparsed: Option<String>,
4554}
4555
4556/// One interval a run measures: its roles, the columns they sit on, and the grain
4557/// its rows are cut by.
4558struct ResolvedInterval {
4559    start_role: TemporalRole,
4560    end_role: TemporalRole,
4561    start: String,
4562    end: String,
4563    grain: QualityGrain,
4564}
4565
4566/// The measured intervals whose two roles sit on columns the run can read as time.
4567/// A role on text with no format measures nothing: its interval is left out rather
4568/// than read as all missing.
4569fn resolved_intervals(plan: &DataQualityPlan, schema: &Schema) -> Vec<ResolvedInterval> {
4570    let usable = |role| {
4571        plan.role_column(role)
4572            .filter(|column| plan.reads_as_time(column, schema))
4573            .map(str::to_string)
4574    };
4575    plan.interval_pairs()
4576        .into_iter()
4577        .filter_map(|(start_role, end_role)| {
4578            let (start, end) = (usable(start_role)?, usable(end_role)?);
4579            let grain = plan.interval_grain(&start, &end);
4580            Some(ResolvedInterval {
4581                start_role,
4582                end_role,
4583                start,
4584                end,
4585                grain,
4586            })
4587        })
4588        .collect()
4589}
4590
4591/// The distinct grains `intervals` are cut by, in the order they first appear: one
4592/// grouping each, however many intervals share it.
4593fn interval_grains(intervals: &[ResolvedInterval]) -> Vec<QualityGrain> {
4594    let mut grains = Vec::new();
4595    for interval in intervals {
4596        if !grains.contains(&interval.grain) {
4597            grains.push(interval.grain.clone());
4598        }
4599    }
4600    grains
4601}
4602
4603/// How many groupings a run's intervals take: one per distinct grain. A full run
4604/// reads the scope once for each.
4605pub fn interval_passes(plan: &DataQualityPlan, schema: &Schema) -> usize {
4606    interval_grains(&resolved_intervals(plan, schema)).len()
4607}
4608
4609/// End minus start per row, as a duration: null where either is missing or unread.
4610/// Dates are midnight; a zoned time is its instant in UTC, and a time with no zone
4611/// is read as if it were UTC.
4612fn interval_duration(plan: &DataQualityPlan, start: &str, end: &str) -> Expr {
4613    let as_time = |column: &str| {
4614        plan.time_value(column)
4615            .cast(DataType::Datetime(TimeUnit::Microseconds, None))
4616    };
4617    as_time(end) - as_time(start)
4618}
4619
4620/// [`interval_duration`] in microseconds, the unit its counts are taken in.
4621fn interval_micros(plan: &DataQualityPlan, start: &str, end: &str) -> Expr {
4622    interval_duration(plan, start, end)
4623        .dt()
4624        .total_microseconds(false)
4625}
4626
4627fn profile_temporal(
4628    df: &DataFrame,
4629    plan: &DataQualityPlan,
4630    sample_positions: Option<&[IdxSize]>,
4631) -> Result<Vec<TemporalLatencyProfile>> {
4632    // Resolved before the rows are grouped, as the lazy path does: the default plan
4633    // assigns no roles at all, and splitting the sample into ten thousand segments to
4634    // discover that costs a DataFrame copy per segment and answers nothing.
4635    let resolved = resolved_intervals(plan, df.schema());
4636    if resolved.is_empty() {
4637        return Ok(Vec::new());
4638    }
4639    // Text read as time is parsed once, beside the text, with a flag on the rows the
4640    // format did not read, so an unread value is told apart from a missing one.
4641    let mut parsed = Vec::new();
4642    let mut timed = |column: &str| {
4643        let Some(format) = plan.time_format(column) else {
4644            return TimedColumn {
4645                name: column.to_string(),
4646                values: column.to_string(),
4647                unparsed: None,
4648            };
4649        };
4650        let values = format!("__datui_quality_time::{column}");
4651        let unparsed = format!("__datui_quality_unparsed::{column}");
4652        if !parsed
4653            .iter()
4654            .any(|(name, _): &(String, Expr)| *name == values)
4655        {
4656            parsed.push((values.clone(), format.expr()));
4657            parsed.push((unparsed.clone(), format.unparsed()));
4658        }
4659        TimedColumn {
4660            name: column.to_string(),
4661            values,
4662            unparsed: Some(unparsed),
4663        }
4664    };
4665    let intervals = resolved
4666        .iter()
4667        .map(|interval| (interval, timed(&interval.start), timed(&interval.end)))
4668        .collect::<Vec<_>>();
4669    let df = if parsed.is_empty() {
4670        df.clone()
4671    } else {
4672        df.clone()
4673            .lazy()
4674            .with_columns(
4675                parsed
4676                    .into_iter()
4677                    .map(|(name, expr)| expr.alias(name))
4678                    .collect::<Vec<_>>(),
4679            )
4680            .collect()?
4681    };
4682    // Each grain's segments are cut once, whichever intervals share it.
4683    let mut cut: Vec<(QualityGrain, Vec<(String, DataFrame)>)> = Vec::new();
4684    for grain in interval_grains(&resolved) {
4685        let grain_plan = DataQualityPlan {
4686            grain: grain.clone(),
4687            ..plan.clone()
4688        };
4689        let segments = segment_rows(&df, &grain_plan, sample_positions)?
4690            .into_iter()
4691            .map(|group| Ok((group.label, take_rows(&df, &group.indices)?)))
4692            .collect::<Result<Vec<_>>>()?;
4693        cut.push((grain, segments));
4694    }
4695    // One interval's segments together, in their order, then the next interval's.
4696    let mut profiles = Vec::new();
4697    for (interval, start, end) in &intervals {
4698        let Some((_, segments)) = cut.iter().find(|(grain, _)| *grain == interval.grain) else {
4699            continue;
4700        };
4701        for (label, segment) in segments {
4702            profiles.push(latency_profile(
4703                segment,
4704                label,
4705                (interval.start_role, start),
4706                (interval.end_role, end),
4707                plan.latency_threshold_seconds,
4708            )?);
4709        }
4710    }
4711    Ok(profiles)
4712}
4713
4714fn profile_temporal_lazy(
4715    lf: &LazyFrame,
4716    plan: &DataQualityPlan,
4717    source: Option<&QualitySourceContext>,
4718    polars_streaming: bool,
4719) -> Result<Vec<TemporalLatencyProfile>> {
4720    let schema = lf.clone().collect_schema()?;
4721    let resolved = resolved_intervals(plan, &schema);
4722    if resolved.is_empty() {
4723        return Ok(Vec::new());
4724    }
4725
4726    let unparsed = |column: &str| {
4727        plan.time_format(column)
4728            .map(|format| format.unparsed().sum())
4729            .unwrap_or_else(|| lit(0u32))
4730    };
4731    let mut profiles = (0..resolved.len()).map(|_| Vec::new()).collect::<Vec<_>>();
4732    // One collect per grain the intervals are cut by: usually one, and one more for
4733    // each clock that differs.
4734    for grain in interval_grains(&resolved) {
4735        let mut expressions = vec![len().alias("__quality_temporal_rows")];
4736        let members = resolved
4737            .iter()
4738            .enumerate()
4739            .filter(|(_, interval)| interval.grain == grain)
4740            .map(|(index, _)| index)
4741            .collect::<Vec<_>>();
4742        for index in &members {
4743            let interval = &resolved[*index];
4744            let prefix = format!("latency::{index}::");
4745            let micros = interval_micros(plan, &interval.start, &interval.end);
4746            // Seconds as the sampled path takes them: whole seconds, toward zero.
4747            let seconds = interval_duration(plan, &interval.start, &interval.end)
4748                .dt()
4749                .total_seconds(false);
4750            expressions.extend([
4751                // Missing is the stored value; text the format did not read is
4752                // counted on its own.
4753                col(interval.start.as_str())
4754                    .is_null()
4755                    .sum()
4756                    .alias(format!("{prefix}missing_start")),
4757                col(interval.end.as_str())
4758                    .is_null()
4759                    .sum()
4760                    .alias(format!("{prefix}missing_end")),
4761                unparsed(&interval.start).alias(format!("{prefix}unparsed_start")),
4762                unparsed(&interval.end).alias(format!("{prefix}unparsed_end")),
4763                micros
4764                    .clone()
4765                    .is_not_null()
4766                    .sum()
4767                    .alias(format!("{prefix}paired")),
4768                micros
4769                    .clone()
4770                    .lt(lit(0i64))
4771                    .sum()
4772                    .alias(format!("{prefix}negative")),
4773                micros
4774                    .clone()
4775                    .eq(lit(0i64))
4776                    .sum()
4777                    .alias(format!("{prefix}zero")),
4778                seconds
4779                    .clone()
4780                    .quantile(lit(0.50), QuantileMethod::Nearest)
4781                    .alias(format!("{prefix}p50")),
4782                seconds
4783                    .clone()
4784                    .quantile(lit(0.90), QuantileMethod::Nearest)
4785                    .alias(format!("{prefix}p90")),
4786                seconds
4787                    .clone()
4788                    .quantile(lit(0.95), QuantileMethod::Nearest)
4789                    .alias(format!("{prefix}p95")),
4790                seconds
4791                    .clone()
4792                    .quantile(lit(0.99), QuantileMethod::Nearest)
4793                    .alias(format!("{prefix}p99")),
4794                seconds.max().alias(format!("{prefix}max")),
4795            ]);
4796            if let Some(threshold) = plan.latency_threshold_seconds {
4797                expressions.push(
4798                    micros
4799                        .gt(lit(threshold.saturating_mul(1_000_000)))
4800                        .sum()
4801                        .alias(format!("{prefix}above")),
4802                );
4803            }
4804        }
4805
4806        let grain_plan = DataQualityPlan {
4807            grain: grain.clone(),
4808            ..plan.clone()
4809        };
4810        let ungrouped = matches!(grain, QualityGrain::Dataset)
4811            || matches!(grain, QualityGrain::File) && source.is_none();
4812        let aggregate = if ungrouped {
4813            collect_lazy(lf.clone().select(expressions), polars_streaming).map_err(Report::from)?
4814        } else {
4815            let (grouped_lf, group) = grouped_frame(lf, &grain_plan, source)?;
4816            collect_lazy(
4817                grouped_lf
4818                    .group_by([group.alias("__quality_segment")])
4819                    .agg(expressions),
4820                polars_streaming,
4821            )
4822            .map_err(Report::from)?
4823        };
4824
4825        for row in 0..aggregate.height() {
4826            let segment = if ungrouped {
4827                if matches!(grain, QualityGrain::File) {
4828                    "file mapping unavailable for this view".to_string()
4829                } else {
4830                    "current view".to_string()
4831                }
4832            } else {
4833                let raw = string_value_at(&aggregate, "__quality_segment", row);
4834                segment_label(&grain, raw.as_deref())
4835            };
4836            let evaluated_rows = usize_value_at(&aggregate, "__quality_temporal_rows", row);
4837            for index in &members {
4838                let interval = &resolved[*index];
4839                let prefix = format!("latency::{index}::");
4840                let count =
4841                    |name: &str| usize_value_at(&aggregate, &format!("{prefix}{name}"), row);
4842                let seconds =
4843                    |name: &str| optional_i64_at(&aggregate, &format!("{prefix}{name}"), row);
4844                profiles[*index].push(TemporalLatencyProfile {
4845                    segment: segment.clone(),
4846                    start_role: interval.start_role,
4847                    end_role: interval.end_role,
4848                    start_column: interval.start.clone(),
4849                    end_column: interval.end.clone(),
4850                    evaluated_rows,
4851                    paired_rows: count("paired"),
4852                    missing_start: count("missing_start"),
4853                    missing_end: count("missing_end"),
4854                    unparsed_start: count("unparsed_start"),
4855                    unparsed_end: count("unparsed_end"),
4856                    negative_count: count("negative"),
4857                    zero_count: count("zero"),
4858                    p50_seconds: seconds("p50"),
4859                    p90_seconds: seconds("p90"),
4860                    p95_seconds: seconds("p95"),
4861                    p99_seconds: seconds("p99"),
4862                    max_seconds: seconds("max"),
4863                    threshold_seconds: plan.latency_threshold_seconds,
4864                    above_threshold_count: plan.latency_threshold_seconds.map(|_| count("above")),
4865                });
4866            }
4867        }
4868    }
4869    let mut ordered = Vec::new();
4870    for (interval, mut segments) in resolved.iter().zip(profiles) {
4871        segments.sort_by(|left, right| left.segment.cmp(&right.segment));
4872        // The zero padding exists so a lexicographic sort orders chunks numerically,
4873        // and comes off once it has. Segments does the same thing in the same place;
4874        // leaving it on here had Trends and Segments name one chunk two ways.
4875        if matches!(interval.grain, QualityGrain::RowChunks(_)) {
4876            for profile in &mut segments {
4877                profile.segment = pretty_chunk_label(&profile.segment);
4878            }
4879        }
4880        ordered.extend(segments);
4881    }
4882    Ok(ordered)
4883}
4884
4885fn latency_profile(
4886    df: &DataFrame,
4887    segment: &str,
4888    (start_role, start): (TemporalRole, &TimedColumn),
4889    (end_role, end): (TemporalRole, &TimedColumn),
4890    threshold_seconds: Option<i64>,
4891) -> Result<TemporalLatencyProfile> {
4892    let starts = df.column(&start.values)?;
4893    let ends = df.column(&end.values)?;
4894    let flags = |column: &TimedColumn| {
4895        column
4896            .unparsed
4897            .as_ref()
4898            .map(|name| df.column(name))
4899            .transpose()
4900    };
4901    let (start_flags, end_flags) = (flags(start)?, flags(end)?);
4902    let unread = |flags: Option<&Column>, row: usize| -> Result<bool> {
4903        Ok(match flags {
4904            Some(flags) => flags.get(row)? == AnyValue::Boolean(true),
4905            None => false,
4906        })
4907    };
4908    let mut missing_start = 0;
4909    let mut missing_end = 0;
4910    let mut unparsed_start = 0;
4911    let mut unparsed_end = 0;
4912    let mut micros = Vec::new();
4913    for row in 0..df.height() {
4914        let start_at = value_epoch_micros(starts.get(row)?);
4915        let end_at = value_epoch_micros(ends.get(row)?);
4916        if start_at.is_none() {
4917            if unread(start_flags, row)? {
4918                unparsed_start += 1;
4919            } else {
4920                missing_start += 1;
4921            }
4922        }
4923        if end_at.is_none() {
4924            if unread(end_flags, row)? {
4925                unparsed_end += 1;
4926            } else {
4927                missing_end += 1;
4928            }
4929        }
4930        if let (Some(start_at), Some(end_at)) = (start_at, end_at) {
4931            micros.push(end_at - start_at);
4932        }
4933    }
4934    // Counted on the exact difference, so half a second early is early; the
4935    // percentiles are whole seconds.
4936    let negative_count = micros.iter().filter(|value| **value < 0).count();
4937    let zero_count = micros.iter().filter(|value| **value == 0).count();
4938    let above_threshold_count = threshold_seconds.map(|threshold| {
4939        let threshold = threshold.saturating_mul(1_000_000);
4940        micros.iter().filter(|value| **value > threshold).count()
4941    });
4942    let mut seconds = micros
4943        .iter()
4944        .map(|value| value / 1_000_000)
4945        .collect::<Vec<_>>();
4946    seconds.sort_unstable();
4947    let percentile = |percent: usize| {
4948        if seconds.is_empty() {
4949            None
4950        } else {
4951            let index = ((seconds.len() - 1) * percent + 50) / 100;
4952            seconds.get(index).copied()
4953        }
4954    };
4955    Ok(TemporalLatencyProfile {
4956        segment: segment.to_string(),
4957        start_role,
4958        end_role,
4959        start_column: start.name.clone(),
4960        end_column: end.name.clone(),
4961        evaluated_rows: df.height(),
4962        paired_rows: micros.len(),
4963        missing_start,
4964        missing_end,
4965        unparsed_start,
4966        unparsed_end,
4967        negative_count,
4968        zero_count,
4969        p50_seconds: percentile(50),
4970        p90_seconds: percentile(90),
4971        p95_seconds: percentile(95),
4972        p99_seconds: percentile(99),
4973        max_seconds: seconds.last().copied(),
4974        threshold_seconds,
4975        above_threshold_count,
4976    })
4977}
4978
4979/// Two counts per text column read as time, in whatever pass profiles the columns:
4980/// its non-null values, and those the format does not read.
4981fn interpretation_exprs(plan: &DataQualityPlan, schema: &Schema) -> Vec<Expr> {
4982    plan.time_formats
4983        .iter()
4984        .enumerate()
4985        .filter(|(_, format)| schema.get(&format.column).is_some())
4986        .flat_map(|(index, format)| {
4987            [
4988                col(format.column.as_str())
4989                    .is_not_null()
4990                    .sum()
4991                    .alias(format!("__datui_time::{index}::values")),
4992                format
4993                    .unparsed()
4994                    .sum()
4995                    .alias(format!("__datui_time::{index}::unparsed")),
4996            ]
4997        })
4998        .collect()
4999}
5000
5001/// Text the chosen format does not read, one observation per column that has any,
5002/// from the counts [`interpretation_exprs`] took.
5003fn interpretation_observations(
5004    counts: &DataFrame,
5005    plan: &DataQualityPlan,
5006    schema: &Schema,
5007) -> Vec<QualityObservation> {
5008    plan.time_formats
5009        .iter()
5010        .enumerate()
5011        .filter(|(_, format)| schema.get(&format.column).is_some())
5012        .filter_map(|(index, format)| {
5013            let values = optional_usize(counts, &format!("__datui_time::{index}::values"))?;
5014            let unparsed = optional_usize(counts, &format!("__datui_time::{index}::unparsed"))?;
5015            (unparsed > 0).then(|| QualityObservation {
5016                kind: ObservationKind::UnparsedTime,
5017                column: format.column.clone(),
5018                affected_rows: unparsed,
5019                evaluated_rows: values,
5020                fact: format!(
5021                    "{} of {} values do not read as {}",
5022                    crate::numfmt::group_chrome(unparsed),
5023                    crate::numfmt::group_chrome(values),
5024                    format.label()
5025                ),
5026                normalized_category: None,
5027                files: Vec::new(),
5028                time_format: Some(format.clone()),
5029                full_scale: None,
5030            })
5031        })
5032        .collect()
5033}
5034
5035/// A categorical column stores integer codes, not text: `.str()` rejects it and
5036/// a numeric cast would measure the codes. Read its values as strings instead.
5037fn text_expr(column: Expr, dtype: &DataType) -> Expr {
5038    if matches!(dtype, DataType::Categorical(..)) {
5039        column.cast(DataType::String)
5040    } else {
5041        column
5042    }
5043}
5044
5045fn build_profile_exprs(schema: &Schema) -> Vec<Expr> {
5046    let mut exprs = Vec::new();
5047    for (name, dtype) in schema.iter() {
5048        let column = col(name.as_str());
5049        let prefix = format!("{}::", name);
5050        exprs.push(column.clone().null_count().alias(format!("{prefix}null")));
5051        exprs.push(
5052            column
5053                .clone()
5054                .filter(column.clone().is_not_null())
5055                .n_unique()
5056                .alias(format!("{prefix}distinct")),
5057        );
5058
5059        if supports_range(dtype) {
5060            exprs.push(column.clone().min().alias(format!("{prefix}min")));
5061            exprs.push(column.clone().max().alias(format!("{prefix}max")));
5062        }
5063
5064        if matches!(dtype, DataType::String | DataType::Categorical(..)) {
5065            let text = text_expr(column.clone(), dtype);
5066            let trimmed = text
5067                .clone()
5068                .str()
5069                .strip_chars(lit(LiteralValue::untyped_null()));
5070            exprs.push(
5071                text.clone()
5072                    .eq(lit(""))
5073                    .sum()
5074                    .alias(format!("{prefix}empty")),
5075            );
5076            exprs.push(
5077                trimmed
5078                    .eq(lit(""))
5079                    .and(text.clone().neq(lit("")))
5080                    .sum()
5081                    .alias(format!("{prefix}whitespace")),
5082            );
5083            exprs.push(
5084                text.clone()
5085                    .cast(DataType::Int64)
5086                    .is_not_null()
5087                    .and(text.clone().is_not_null())
5088                    .sum()
5089                    .alias(format!("{prefix}parse_int")),
5090            );
5091            exprs.push(
5092                text.clone()
5093                    .str()
5094                    .starts_with(lit("0"))
5095                    .and(text.clone().str().len_chars().gt(lit(1u32)))
5096                    .and(text.clone().cast(DataType::Int64).is_not_null())
5097                    .sum()
5098                    .alias(format!("{prefix}leading_zero")),
5099            );
5100            for (reading, name) in [
5101                (TextReading::Decimal, "parse_decimal"),
5102                (TextReading::Date, "parse_date"),
5103                (TextReading::Datetime, "parse_datetime"),
5104            ] {
5105                exprs.push(
5106                    parses_as(text.clone(), reading)
5107                        .and(text.clone().is_not_null())
5108                        .sum()
5109                        .alias(format!("{prefix}{name}")),
5110                );
5111            }
5112            exprs.push(
5113                text.clone()
5114                    .str()
5115                    .len_chars()
5116                    .min()
5117                    .alias(format!("{prefix}min_length")),
5118            );
5119            exprs.push(
5120                text.str()
5121                    .len_chars()
5122                    .max()
5123                    .alias(format!("{prefix}max_length")),
5124            );
5125        }
5126
5127        if matches!(dtype, DataType::List(_)) {
5128            exprs.push(
5129                column
5130                    .clone()
5131                    .list()
5132                    .len()
5133                    .min()
5134                    .alias(format!("{prefix}min_length")),
5135            );
5136            exprs.push(
5137                column
5138                    .clone()
5139                    .list()
5140                    .len()
5141                    .max()
5142                    .alias(format!("{prefix}max_length")),
5143            );
5144        }
5145
5146        if dtype.is_float() {
5147            let float = column.cast(DataType::Float64);
5148            exprs.push(float.clone().is_nan().sum().alias(format!("{prefix}nan")));
5149            exprs.push(
5150                float
5151                    .clone()
5152                    .eq(lit(f64::INFINITY))
5153                    .sum()
5154                    .alias(format!("{prefix}pos_inf")),
5155            );
5156            exprs.push(
5157                float
5158                    .eq(lit(f64::NEG_INFINITY))
5159                    .sum()
5160                    .alias(format!("{prefix}neg_inf")),
5161            );
5162        }
5163    }
5164    exprs
5165}
5166
5167/// Whether text parses as `reading`: the test the profile counts with, so a count
5168/// and the rows it opens agree. A whole number is counted among the decimals, and
5169/// fails where they fail.
5170fn parses_as(text: Expr, reading: TextReading) -> Expr {
5171    // Named formats, not inference: "parses as an ISO date" has to mean the same
5172    // thing on every column, including one where nothing does.
5173    let strptime = |format: &str| StrptimeOptions {
5174        format: Some(PlSmallStr::from(format)),
5175        strict: false,
5176        exact: true,
5177        cache: true,
5178    };
5179    match reading {
5180        TextReading::WholeNumber | TextReading::Decimal => {
5181            text.cast(DataType::Float64).is_not_null()
5182        }
5183        TextReading::Date => text.str().to_date(strptime("%Y-%m-%d")).is_not_null(),
5184        TextReading::Datetime => [
5185            "%Y-%m-%d %H:%M:%S%.f",
5186            "%Y-%m-%dT%H:%M:%S%.f%#z",
5187            "%Y-%m-%dT%H:%M:%S%.f",
5188            "%Y-%m-%d %H:%M:%S",
5189            "%Y-%m-%dT%H:%M:%S%#z",
5190            "%Y-%m-%dT%H:%M:%S",
5191        ]
5192        .into_iter()
5193        .map(|format| {
5194            text.clone()
5195                .str()
5196                .to_datetime(
5197                    Some(TimeUnit::Microseconds),
5198                    None,
5199                    strptime(format),
5200                    lit(PlSmallStr::from_static("raise")),
5201                )
5202                .is_not_null()
5203        })
5204        .reduce(Expr::or)
5205        .expect("at least one datetime format"),
5206    }
5207}
5208
5209/// The rows of a parseable-text column its reading does not parse: non-null text
5210/// that stops a cast. `None` when the column has no reading.
5211pub fn unparsed_text(profile: &ColumnQualityProfile) -> Option<Expr> {
5212    let (_, reading) = text_reading(profile)?;
5213    let text = text_expr(col(profile.name.as_str()), &profile.dtype);
5214    Some(
5215        text.clone()
5216            .is_not_null()
5217            .and(parses_as(text, reading).not()),
5218    )
5219}
5220
5221fn supports_range(dtype: &DataType) -> bool {
5222    dtype.is_numeric()
5223        || dtype.is_temporal()
5224        || matches!(
5225            dtype,
5226            DataType::String | DataType::Categorical(..) | DataType::Boolean
5227        )
5228}
5229
5230fn parse_profiles(
5231    aggregate: &DataFrame,
5232    schema: &Schema,
5233    evaluated_rows: usize,
5234) -> Vec<ColumnQualityProfile> {
5235    parse_profiles_at(aggregate, schema, evaluated_rows, 0)
5236}
5237
5238fn parse_profiles_at(
5239    aggregate: &DataFrame,
5240    schema: &Schema,
5241    evaluated_rows: usize,
5242    row: usize,
5243) -> Vec<ColumnQualityProfile> {
5244    schema
5245        .iter()
5246        .map(|(name, dtype)| {
5247            let prefix = format!("{}::", name);
5248            ColumnQualityProfile {
5249                name: name.to_string(),
5250                dtype: dtype.clone(),
5251                evaluated_rows,
5252                null_count: usize_value_at(aggregate, &format!("{prefix}null"), row),
5253                empty_count: optional_usize_at(aggregate, &format!("{prefix}empty"), row),
5254                whitespace_count: optional_usize_at(aggregate, &format!("{prefix}whitespace"), row),
5255                nan_count: optional_usize_at(aggregate, &format!("{prefix}nan"), row),
5256                positive_infinity_count: optional_usize_at(
5257                    aggregate,
5258                    &format!("{prefix}pos_inf"),
5259                    row,
5260                ),
5261                negative_infinity_count: optional_usize_at(
5262                    aggregate,
5263                    &format!("{prefix}neg_inf"),
5264                    row,
5265                ),
5266                distinct_count: optional_usize_at(aggregate, &format!("{prefix}distinct"), row),
5267                min: string_value_at(aggregate, &format!("{prefix}min"), row),
5268                max: string_value_at(aggregate, &format!("{prefix}max"), row),
5269                integer_parse_count: optional_usize_at(
5270                    aggregate,
5271                    &format!("{prefix}parse_int"),
5272                    row,
5273                ),
5274                decimal_parse_count: optional_usize_at(
5275                    aggregate,
5276                    &format!("{prefix}parse_decimal"),
5277                    row,
5278                ),
5279                date_parse_count: optional_usize_at(aggregate, &format!("{prefix}parse_date"), row),
5280                datetime_parse_count: optional_usize_at(
5281                    aggregate,
5282                    &format!("{prefix}parse_datetime"),
5283                    row,
5284                ),
5285                leading_zero_count: optional_usize_at(
5286                    aggregate,
5287                    &format!("{prefix}leading_zero"),
5288                    row,
5289                ),
5290                dominant_value: None,
5291                dominant_count: None,
5292                min_length: optional_usize_at(aggregate, &format!("{prefix}min_length"), row),
5293                max_length: optional_usize_at(aggregate, &format!("{prefix}max_length"), row),
5294            }
5295        })
5296        .collect()
5297}
5298
5299/// The share of non-null text values that must parse before a text column is said
5300/// to hold numbers or dates. Below it the column is text that happens to contain a
5301/// few numbers, which is not a finding.
5302pub const TEXT_READING_SHARE: f64 = 0.95;
5303
5304/// What the values of a text column parse as, most specific first.
5305#[derive(Debug, Clone, Copy, PartialEq, Eq)]
5306pub enum TextReading {
5307    WholeNumber,
5308    Decimal,
5309    Datetime,
5310    Date,
5311}
5312
5313impl TextReading {
5314    pub fn label(self) -> &'static str {
5315        match self {
5316            Self::WholeNumber => "whole numbers",
5317            Self::Decimal => "decimal numbers",
5318            Self::Datetime => "ISO datetimes",
5319            Self::Date => "ISO dates",
5320        }
5321    }
5322
5323    pub fn is_number(self) -> bool {
5324        matches!(self, Self::WholeNumber | Self::Decimal)
5325    }
5326}
5327
5328/// The one typed reading a text column's values support, with how many parse.
5329///
5330/// A whole number also parses as a decimal and a datetime string may also parse as a
5331/// date, so the column gets one answer rather than three rows saying overlapping
5332/// things. Numbers are whole only when every number is.
5333pub fn text_reading(profile: &ColumnQualityProfile) -> Option<(usize, TextReading)> {
5334    let non_null = profile.non_null_rows();
5335    if non_null == 0 {
5336        return None;
5337    }
5338    let enough = |count: Option<usize>| {
5339        count.filter(|parsed| *parsed as f64 >= non_null as f64 * TEXT_READING_SHARE)
5340    };
5341    if let Some(parsed) = enough(profile.decimal_parse_count) {
5342        let reading = if profile.integer_parse_count == Some(parsed) {
5343            TextReading::WholeNumber
5344        } else {
5345            TextReading::Decimal
5346        };
5347        return Some((parsed, reading));
5348    }
5349    [
5350        (profile.datetime_parse_count, TextReading::Datetime),
5351        (profile.date_parse_count, TextReading::Date),
5352    ]
5353    .into_iter()
5354    .find_map(|(count, reading)| enough(count).map(|parsed| (parsed, reading)))
5355}
5356
5357fn observations_from_profiles(
5358    columns: &[ColumnQualityProfile],
5359    precision: QualityPrecision,
5360) -> Vec<QualityObservation> {
5361    let mut observations = Vec::new();
5362    for profile in columns {
5363        if profile.null_count > 0 {
5364            observations.push(observation(
5365                ObservationKind::Nulls,
5366                profile,
5367                profile.null_count,
5368                format!("{:.2}% null", profile.null_rate() * 100.0),
5369            ));
5370        }
5371        if let Some(count) = profile.empty_count.filter(|count| *count > 0) {
5372            observations.push(observation(
5373                ObservationKind::Empty,
5374                profile,
5375                count,
5376                format!("{:.2}% empty", rate(count, profile.evaluated_rows) * 100.0),
5377            ));
5378        }
5379        if let Some(count) = profile.whitespace_count.filter(|count| *count > 0) {
5380            observations.push(observation(
5381                ObservationKind::Whitespace,
5382                profile,
5383                count,
5384                format!(
5385                    "{:.2}% whitespace only",
5386                    rate(count, profile.evaluated_rows) * 100.0
5387                ),
5388            ));
5389        }
5390        let non_finite = profile.nan_count.unwrap_or(0)
5391            + profile.positive_infinity_count.unwrap_or(0)
5392            + profile.negative_infinity_count.unwrap_or(0);
5393        if non_finite > 0 {
5394            observations.push(observation(
5395                ObservationKind::NonFinite,
5396                profile,
5397                non_finite,
5398                format!("{non_finite} NaN or infinite"),
5399            ));
5400        }
5401        if profile.distinct_count == Some(1) && profile.non_null_rows() > 0 {
5402            observations.push(observation(
5403                ObservationKind::Constant,
5404                profile,
5405                profile.non_null_rows(),
5406                "one non-null value".to_string(),
5407            ));
5408        }
5409        if let Some((parsed, reading)) = text_reading(profile) {
5410            observations.push(observation(
5411                ObservationKind::ParseableText,
5412                profile,
5413                parsed,
5414                format!(
5415                    "{:.2}% parse as {}",
5416                    rate(parsed, profile.non_null_rows()) * 100.0,
5417                    reading.label()
5418                ),
5419            ));
5420        }
5421        // Near-unique and still repeating. Both numbers are already measured, so this
5422        // check costs the comparison and nothing else.
5423        //
5424        // Only on an exact profile: a distinct count does not extrapolate the way a
5425        // null rate does. An order id repeating ten times in a billion rows is unique
5426        // in every 50,000-row sample of it, and "sampled" under a claim that a column
5427        // is nearly a key does not take the claim back.
5428        //
5429        // Only where a key can live: integers and text. A float measure or a timestamp
5430        // is nearly unique by nature, and its repeats are coincidences, not duplicates.
5431        if precision == QualityPrecision::Exact
5432            && (profile.dtype.is_integer()
5433                || matches!(profile.dtype, DataType::String | DataType::Categorical(..)))
5434            && let (Some(distinct), Some(uniqueness)) =
5435                (profile.distinct_count, profile.uniqueness_rate())
5436            && (KEY_LIKE_UNIQUENESS..1.0).contains(&uniqueness)
5437        {
5438            // Rows beyond one per value, as `DuplicateRows` counts extras. Not the
5439            // rows that share a value, which is what the drill-in opens and always
5440            // more; the detail pane says which is which.
5441            let extras = profile.non_null_rows().saturating_sub(distinct);
5442            if extras > 0 {
5443                let example = match (&profile.dominant_value, profile.dominant_count) {
5444                    (Some(value), Some(count)) if count > 1 => {
5445                        format!("; {value:?} appears {count} times")
5446                    }
5447                    _ => String::new(),
5448                };
5449                observations.push(observation(
5450                    ObservationKind::KeyLike,
5451                    profile,
5452                    extras,
5453                    format!(
5454                        "{distinct} distinct over {} non-null rows ({:.4}%); {extras} rows beyond one per value{example}",
5455                        profile.non_null_rows(),
5456                        uniqueness * 100.0,
5457                    ),
5458                ));
5459            }
5460        }
5461    }
5462    observations
5463}
5464
5465/// How many one-column file reads a full run makes for the values type conflicts hide,
5466/// so the access plan can promise them before anything is read.
5467///
5468/// Over the footers rather than over a [`QualitySourceContext`]: the access plan asks
5469/// this on every frame it is open, and building a context to answer would clone a file
5470/// list per frame.
5471pub(crate) fn conflict_reads(
5472    file_group: &[u32],
5473    groups: &[crate::schema_union::DriftGroup],
5474) -> usize {
5475    let mut per_column = BTreeMap::<&str, usize>::new();
5476    for group in file_group {
5477        let Some(group) = groups.get(*group as usize) else {
5478            continue;
5479        };
5480        for column in &group.unread {
5481            *per_column.entry(column.as_str()).or_default() += 1;
5482        }
5483    }
5484    per_column
5485        .values()
5486        .map(|files| (*files).min(MAX_EVIDENCE_FILES))
5487        .sum()
5488}
5489
5490/// Reads named columns of named files at the type each file wrote, which is the only
5491/// way back to the values a type conflict hides. Given to a run that already reads
5492/// every value, so the extra read is one column of the few files that disagree.
5493#[derive(Clone)]
5494pub struct QualityConflictScan(pub crate::widgets::datatable::FileScan);
5495
5496impl std::fmt::Debug for QualityConflictScan {
5497    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
5498        f.write_str("QualityConflictScan")
5499    }
5500}
5501
5502/// Absent columns and type conflicts, from the footers datui already read.
5503///
5504/// Both are facts about which files hold which columns, so they are measured over the
5505/// whole loaded source however the run was scoped: no value in a scope can say
5506/// anything about a column its file never had, and a conflicting column is not read
5507/// into the scope at all. The detail pane says so rather than leaving the reader to
5508/// notice that these two denominators are not the others.
5509/// What one column loses to the files that disagree: every one of them counted, and
5510/// the largest few kept by name.
5511///
5512/// A dataset of 6,541 files can have a column missing from nearly all of them, so the
5513/// names are pruned as they arrive. Holding one entry per file per column is how a
5514/// measurement that costs nothing to compute ends up costing hundreds of megabytes.
5515#[derive(Default)]
5516struct DriftTally {
5517    files: usize,
5518    rows: usize,
5519    named: Vec<QualityFileEvidence>,
5520}
5521
5522impl DriftTally {
5523    fn add(&mut self, evidence: QualityFileEvidence) {
5524        self.files += 1;
5525        self.rows += evidence.rows;
5526        self.named.push(evidence);
5527        if self.named.len() > MAX_EVIDENCE_FILES * 2 {
5528            self.prune();
5529        }
5530    }
5531
5532    /// Largest first: the files that cost the column the most rows are the ones worth
5533    /// naming and worth reading values from.
5534    fn prune(&mut self) {
5535        self.named.sort_by(|left, right| {
5536            right
5537                .rows
5538                .cmp(&left.rows)
5539                .then_with(|| left.number.cmp(&right.number))
5540        });
5541        self.named.truncate(MAX_EVIDENCE_FILES);
5542    }
5543}
5544
5545fn drift_observations(
5546    source: &QualitySourceContext,
5547    conflicts: Option<&QualityConflictScan>,
5548    polars_streaming: bool,
5549    watch: &QualityWatch,
5550) -> Vec<QualityObservation> {
5551    let mut absent = BTreeMap::<String, DriftTally>::new();
5552    let mut unread = BTreeMap::<String, DriftTally>::new();
5553    for (file, group) in source.drifting_files() {
5554        let evidence = |stored_type: Option<String>| QualityFileEvidence {
5555            number: file + 1,
5556            name: source
5557                .file_names
5558                .get(file)
5559                .cloned()
5560                .unwrap_or_else(|| format!("file {}", file + 1)),
5561            rows: source.file_rows(file),
5562            stored_type,
5563            examples: Vec::new(),
5564        };
5565        for column in &group.absent {
5566            absent
5567                .entry(column.to_string())
5568                .or_default()
5569                .add(evidence(None));
5570        }
5571        for column in &group.unread {
5572            let stored = source
5573                .stored_type(file, column)
5574                .map(|dtype| dtype.to_string());
5575            unread
5576                .entry(column.to_string())
5577                .or_default()
5578                .add(evidence(stored));
5579        }
5580    }
5581
5582    let mut observations = Vec::new();
5583    for (kind, columns) in [
5584        (ObservationKind::Absent, absent),
5585        (ObservationKind::TypeConflict, unread),
5586    ] {
5587        for (column, mut tally) in columns {
5588            tally.prune();
5589            let mut files = tally.named;
5590            if kind == ObservationKind::TypeConflict
5591                && let Some(scan) = conflicts
5592            {
5593                read_conflict_examples(scan, &column, &mut files, polars_streaming, watch);
5594            }
5595            let named = if tally.files > files.len() {
5596                format!(", largest {} named", files.len())
5597            } else {
5598                String::new()
5599            };
5600            let verb = match (kind, tally.files) {
5601                (ObservationKind::Absent, 1) => "has no such column",
5602                (ObservationKind::Absent, _) => "have no such column",
5603                (_, 1) => "holds a type the scan cannot read",
5604                (_, _) => "hold a type the scan cannot read",
5605            };
5606            // A file whose footer was not read looks exactly like one missing nothing,
5607            // so on a sampled dataset the count is a floor and has to say so.
5608            let sampled = if source.footers_read < source.file_names.len() {
5609                format!(", from {} footers read", source.footers_read)
5610            } else {
5611                String::new()
5612            };
5613            observations.push(QualityObservation {
5614                kind,
5615                column,
5616                affected_rows: tally.rows,
5617                evaluated_rows: source.dataset_rows,
5618                fact: format!(
5619                    "{} of {} files {verb}{sampled}{named}",
5620                    tally.files,
5621                    source.file_names.len()
5622                ),
5623                normalized_category: None,
5624                files,
5625                time_format: None,
5626                full_scale: None,
5627            });
5628        }
5629    }
5630    observations
5631}
5632
5633/// The first values each conflicting file holds, read at that file's own type.
5634///
5635/// One scan per file, of one column, limited to the first few rows: a conflict is a
5636/// property of the file rather than of any row, so the first values it holds are as
5637/// good evidence as any and stop the read at once. A file that cannot be read this way
5638/// keeps its count and loses only its examples.
5639fn read_conflict_examples(
5640    scan: &QualityConflictScan,
5641    column: &str,
5642    files: &mut [QualityFileEvidence],
5643    polars_streaming: bool,
5644    watch: &QualityWatch,
5645) {
5646    let name = PlSmallStr::from(column);
5647    for file in files.iter_mut() {
5648        if watch.cancelled() {
5649            return;
5650        }
5651        let Ok(lf) = (scan.0)(
5652            std::slice::from_ref(&file.name),
5653            std::slice::from_ref(&name),
5654        ) else {
5655            continue;
5656        };
5657        let query = lf
5658            .select([crate::past_calendar::text_expr(
5659                col(column),
5660                CastOptions::NonStrict,
5661            )])
5662            .drop_nulls(None)
5663            .limit(MAX_CONFLICT_EXAMPLES as u32);
5664        let Ok(values) = collect_lazy(query, polars_streaming) else {
5665            continue;
5666        };
5667        file.examples = (0..values.height())
5668            .filter_map(|row| string_value_at(&values, column, row))
5669            .collect();
5670    }
5671}
5672
5673/// Read every sample of `audio` once and add what a recording's quality turns on to
5674/// `results`: clipping, runs of exact zeros, and DC offset, per channel. Only for a
5675/// full run whose scope is every frame of the file, which the caller decides: the
5676/// read is of the file, not of the view.
5677pub fn add_signal_observations(
5678    results: &mut DataQualityResults,
5679    audio: &crate::audio::AudioSource,
5680    watch: &QualityWatch,
5681) -> Result<()> {
5682    watch.stage(QualityStage::CheckingSignal, true, true)?;
5683    let reports = audio
5684        .signal_report(&|| watch.cancelled())?
5685        .ok_or_else(|| Report::msg(crate::sampling::CANCELLED))?;
5686    results
5687        .observations
5688        .extend(signal_observations(&reports, audio.header().sample_rate));
5689    Ok(())
5690}
5691
5692/// Observations from [`crate::audio::SignalReport`]s: a channel with runs at full
5693/// scale, runs of exact zeros, or a mean 1% of full scale or more from zero.
5694pub fn signal_observations(
5695    reports: &[crate::audio::SignalReport],
5696    sample_rate: f64,
5697) -> Vec<QualityObservation> {
5698    let mut observations = Vec::new();
5699    let samples = |n: u64| {
5700        format!(
5701            "{} {}",
5702            crate::numfmt::group_chrome(n as usize),
5703            if n == 1 { "sample" } else { "samples" }
5704        )
5705    };
5706    let runs = |n: u64| if n == 1 { "run" } else { "runs" };
5707    for report in reports {
5708        let evaluated = report.frames as usize;
5709        let push = |observations: &mut Vec<QualityObservation>,
5710                    kind: ObservationKind,
5711                    affected: u64,
5712                    fact: String,
5713                    full_scale: Option<(f64, f64)>| {
5714            observations.push(QualityObservation {
5715                kind,
5716                column: report.channel.clone(),
5717                affected_rows: affected as usize,
5718                evaluated_rows: evaluated,
5719                fact,
5720                normalized_category: None,
5721                files: Vec::new(),
5722                time_format: None,
5723                full_scale,
5724            });
5725        };
5726        if report.clip_runs > 0 {
5727            push(
5728                &mut observations,
5729                ObservationKind::Clipping,
5730                report.in_clip_runs,
5731                format!(
5732                    "{} {} of {}+ samples at full scale; longest {}",
5733                    crate::numfmt::group_chrome(report.clip_runs as usize),
5734                    runs(report.clip_runs),
5735                    report.clip_run_min,
5736                    samples(report.longest_clip)
5737                ),
5738                Some(report.full_scale),
5739            );
5740        }
5741        if report.zero_runs > 0 {
5742            push(
5743                &mut observations,
5744                ObservationKind::ZeroRuns,
5745                report.in_zero_runs,
5746                format!(
5747                    "{} {} of exact zeros, {}+ samples; longest {} ({})",
5748                    crate::numfmt::group_chrome(report.zero_runs as usize),
5749                    runs(report.zero_runs),
5750                    report.zero_run_min,
5751                    samples(report.longest_zeros),
5752                    crate::widgets::info::clock(report.longest_zeros as f64 / sample_rate)
5753                ),
5754                None,
5755            );
5756        }
5757        let (low, high) = report.full_scale;
5758        let half_range = (high - low) / 2.0;
5759        let share = if half_range > 0.0 {
5760            report.mean.abs() / half_range
5761        } else {
5762            0.0
5763        };
5764        if share >= DC_OFFSET_SHARE {
5765            // Integers in their own units; float and normalized to four places.
5766            let mean = if half_range > 2.0 {
5767                format!("{:+.1}", report.mean)
5768            } else {
5769                format!("{:+.4}", report.mean)
5770            };
5771            push(
5772                &mut observations,
5773                ObservationKind::DcOffset,
5774                report.frames,
5775                format!("mean {mean} ({:.1}% of full scale)", share * 100.0),
5776                None,
5777            );
5778        }
5779    }
5780    observations
5781}
5782
5783/// A channel's mean, as a share of full scale, from which it is called DC offset: 1%,
5784/// -40 dBFS, well above any dither or noise floor.
5785const DC_OFFSET_SHARE: f64 = 0.01;
5786
5787fn observation(
5788    kind: ObservationKind,
5789    profile: &ColumnQualityProfile,
5790    affected_rows: usize,
5791    fact: String,
5792) -> QualityObservation {
5793    QualityObservation {
5794        kind,
5795        column: profile.name.clone(),
5796        affected_rows,
5797        evaluated_rows: profile.evaluated_rows,
5798        fact,
5799        normalized_category: None,
5800        files: Vec::new(),
5801        time_format: None,
5802        full_scale: None,
5803    }
5804}
5805
5806fn rate(numerator: usize, denominator: usize) -> f64 {
5807    if denominator == 0 {
5808        0.0
5809    } else {
5810        numerator as f64 / denominator as f64
5811    }
5812}
5813
5814fn optional_usize(df: &DataFrame, name: &str) -> Option<usize> {
5815    optional_usize_at(df, name, 0)
5816}
5817
5818fn optional_usize_at(df: &DataFrame, name: &str, row: usize) -> Option<usize> {
5819    let value = df.column(name).ok()?.get(row).ok()?;
5820    match value {
5821        AnyValue::UInt32(value) => Some(value as usize),
5822        AnyValue::UInt64(value) => Some(value as usize),
5823        AnyValue::Int32(value) => usize::try_from(value).ok(),
5824        AnyValue::Int64(value) => usize::try_from(value).ok(),
5825        _ => None,
5826    }
5827}
5828
5829fn usize_value(df: &DataFrame, name: &str) -> usize {
5830    optional_usize(df, name).unwrap_or(0)
5831}
5832
5833fn usize_value_at(df: &DataFrame, name: &str, row: usize) -> usize {
5834    optional_usize_at(df, name, row).unwrap_or(0)
5835}
5836
5837fn optional_i64_at(df: &DataFrame, name: &str, row: usize) -> Option<i64> {
5838    let value = df.column(name).ok()?.get(row).ok()?;
5839    match value {
5840        AnyValue::Int64(value) => Some(value),
5841        AnyValue::Int32(value) => Some(i64::from(value)),
5842        AnyValue::UInt64(value) => i64::try_from(value).ok(),
5843        AnyValue::UInt32(value) => Some(i64::from(value)),
5844        AnyValue::Float64(value) if value.is_finite() => Some(value.round() as i64),
5845        AnyValue::Float32(value) if value.is_finite() => Some(value.round() as i64),
5846        _ => None,
5847    }
5848}
5849
5850fn string_value_at(df: &DataFrame, name: &str, row: usize) -> Option<String> {
5851    let value = df.column(name).ok()?.get(row).ok()?;
5852    if value.is_null() {
5853        None
5854    } else {
5855        Some(crate::exact::str_value(&value).to_string())
5856    }
5857}
5858
5859#[cfg(test)]
5860mod tests {
5861    use super::*;
5862
5863    fn fixture() -> LazyFrame {
5864        df!(
5865            "id" => &[1i64, 2, 3, 4],
5866            "amount" => &[1.0f64, f64::NAN, f64::INFINITY, 4.0],
5867            "constant" => &["x", "x", "x", "x"],
5868            "text_number" => &[Some("1"), Some("2.5"), Some("bad"), None],
5869            "dirty" => &[Some(""), Some("   "), Some("ok"), None],
5870        )
5871        .unwrap()
5872        .lazy()
5873    }
5874
5875    /// A date past the calendar's range splits like any other value: by
5876    /// partition it is its own segment, named by its stored number, and in time
5877    /// windows it falls in none, as a null does. Each segment's rows are the ones
5878    /// it counted.
5879    /// A grain names its column; one named for its unit says it is the column.
5880    #[test]
5881    fn a_grain_label_never_reads_day_of_day() {
5882        let grain = |column: &str| QualityGrain::TimeWindows {
5883            column: column.to_string(),
5884            every: "1d".to_string(),
5885        };
5886        assert_eq!(grain("date").label(), "by day of date");
5887        assert_eq!(grain("day").label(), "by day of the day column");
5888    }
5889
5890    #[test]
5891    fn dates_past_the_calendar_fall_in_segments_without_a_panic() {
5892        let edges = [i64::MIN + 1, 0, i64::MAX];
5893        let lf = DataFrame::new(
5894            3,
5895            vec![
5896                Column::new("id".into(), [1i64, 2, 3]),
5897                Series::new("t".into(), edges)
5898                    .cast(&DataType::Datetime(TimeUnit::Milliseconds, None))
5899                    .unwrap()
5900                    .into_column(),
5901                Series::new("d".into(), [i32::MIN, 0, i32::MAX])
5902                    .cast(&DataType::Date)
5903                    .unwrap()
5904                    .into_column(),
5905            ],
5906        )
5907        .unwrap()
5908        .lazy();
5909        for grain in [
5910            QualityGrain::Partition("t".into()),
5911            QualityGrain::Partition("d".into()),
5912            QualityGrain::TimeWindows {
5913                column: "t".into(),
5914                every: "1d".into(),
5915            },
5916            QualityGrain::TimeWindows {
5917                column: "d".into(),
5918                every: "1w".into(),
5919            },
5920        ] {
5921            let plan = DataQualityPlan {
5922                compute: QualityCompute::Full,
5923                grain: grain.clone(),
5924                ..DataQualityPlan::default()
5925            };
5926            let results = compute_data_quality(&lf, Some(3), &plan, None, false).unwrap();
5927            let labels: Vec<&str> = results.segments.iter().map(|s| s.label.as_str()).collect();
5928            if let QualityGrain::Partition(column) = &grain {
5929                assert!(
5930                    labels.iter().any(|l| l.starts_with(&format!("{column}=-"))
5931                        && l.contains(" since 1970-01-01")),
5932                    "{grain:?}: {labels:?}"
5933                );
5934            } else {
5935                assert_eq!(labels.len(), 2, "{grain:?}: {labels:?}");
5936            }
5937            let schema = lf.clone().collect_schema().unwrap();
5938            for segment in &results.segments {
5939                // By the column's type, and by each row's label: the same rows.
5940                for schema in [Some(schema.as_ref()), None] {
5941                    let predicate = segment_predicate(&plan, &grain, &segment.label, schema)
5942                        .unwrap()
5943                        .unwrap();
5944                    let rows = lf.clone().filter(predicate).collect().unwrap().height();
5945                    assert_eq!(
5946                        Some(rows),
5947                        segment.total_rows,
5948                        "{grain:?} {}",
5949                        segment.label
5950                    );
5951                }
5952            }
5953        }
5954    }
5955
5956    /// A partition segment's rows are the same found by the column's type as by each
5957    /// row's label, for text, whole numbers, decimals and the floats a label rounds.
5958    #[test]
5959    fn partition_segments_find_the_same_rows_by_type_as_by_label() {
5960        let lf = df!(
5961            "region" => &[Some("west"), Some("east"), None, Some("west")],
5962            "year" => &[2020i16, 2021, 2020, 2020],
5963            "price" => &["1.50", "2.00", "1.50", "3.25"],
5964            "share" => &[1.0 / 3.0, 0.333_333_3, 0.5, 1.0 / 3.0],
5965        )
5966        .unwrap()
5967        .lazy()
5968        .with_column(col("price").cast(DataType::Decimal(10, 2)));
5969        let schema = lf.clone().collect_schema().unwrap();
5970        for column in ["region", "year", "price", "share"] {
5971            let grain = QualityGrain::Partition(column.into());
5972            let plan = DataQualityPlan {
5973                compute: QualityCompute::Full,
5974                grain: grain.clone(),
5975                ..DataQualityPlan::default()
5976            };
5977            let results = compute_data_quality(&lf, Some(4), &plan, None, false).unwrap();
5978            for segment in &results.segments {
5979                for schema in [Some(schema.as_ref()), None] {
5980                    let predicate = segment_predicate(&plan, &grain, &segment.label, schema)
5981                        .unwrap()
5982                        .unwrap();
5983                    let rows = lf.clone().filter(predicate).collect().unwrap().height();
5984                    assert_eq!(Some(rows), segment.total_rows, "{}", segment.label);
5985                }
5986            }
5987        }
5988    }
5989
5990    /// A full run over the whole scope takes its one segment from the column profile
5991    /// it already measured: the numbers a second pass over the scope gave.
5992    #[test]
5993    fn a_full_whole_scope_segment_is_the_column_profile() {
5994        let plan = DataQualityPlan {
5995            compute: QualityCompute::Full,
5996            ..DataQualityPlan::default()
5997        };
5998        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
5999        let schema = fixture().collect_schema().unwrap();
6000        let read = profile_segments_lazy(&fixture(), 4, &plan, None, &schema, false).unwrap();
6001        assert_eq!(results.segments.len(), 1);
6002        let (reused, read) = (&results.segments[0], &read[0]);
6003        assert_eq!(reused.label, read.label);
6004        assert_eq!(reused.total_rows, read.total_rows);
6005        assert_eq!(reused.evaluated_rows, read.evaluated_rows);
6006        assert_eq!(reused.null_cells, read.null_cells);
6007        assert_eq!(reused.null_rate, read.null_rate);
6008        let counts = |segment: &SegmentQualityProfile| {
6009            segment
6010                .columns
6011                .iter()
6012                .map(|column| {
6013                    (
6014                        column.name.clone(),
6015                        column.null_count,
6016                        column.distinct_count,
6017                    )
6018                })
6019                .collect::<Vec<_>>()
6020        };
6021        assert_eq!(counts(reused), counts(read));
6022    }
6023
6024    #[test]
6025    fn full_profile_reports_core_counts() {
6026        let plan = DataQualityPlan {
6027            compute: QualityCompute::Full,
6028            ..DataQualityPlan::default()
6029        };
6030        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6031
6032        assert_eq!(results.precision, QualityPrecision::Exact);
6033        assert_eq!(results.evaluated_rows, 4);
6034        let dirty = results
6035            .columns
6036            .iter()
6037            .find(|profile| profile.name == "dirty")
6038            .unwrap();
6039        assert_eq!(dirty.null_count, 1);
6040        assert_eq!(dirty.distinct_count, Some(3));
6041        assert_eq!(dirty.empty_count, Some(1));
6042        assert_eq!(dirty.whitespace_count, Some(1));
6043
6044        let amount = results
6045            .columns
6046            .iter()
6047            .find(|profile| profile.name == "amount")
6048            .unwrap();
6049        assert_eq!(amount.nan_count, Some(1));
6050        assert_eq!(amount.positive_infinity_count, Some(1));
6051
6052        let constant = results
6053            .columns
6054            .iter()
6055            .find(|profile| profile.name == "constant")
6056            .unwrap();
6057        assert_eq!(constant.distinct_count, Some(1));
6058        assert!(
6059            results
6060                .observations
6061                .iter()
6062                .any(|item| item.kind == ObservationKind::Constant)
6063        );
6064    }
6065
6066    #[test]
6067    fn exact_observation_predicates_select_matching_rows() {
6068        let examples = [
6069            (ObservationKind::Nulls, "dirty", 1),
6070            (ObservationKind::Empty, "dirty", 1),
6071            (ObservationKind::Whitespace, "dirty", 1),
6072            (ObservationKind::NonFinite, "amount", 2),
6073            (ObservationKind::Constant, "constant", 4),
6074        ];
6075        for (kind, column, expected) in examples {
6076            let observation = QualityObservation {
6077                kind,
6078                column: column.to_string(),
6079                affected_rows: expected,
6080                evaluated_rows: 4,
6081                fact: String::new(),
6082                normalized_category: None,
6083                files: Vec::new(),
6084                time_format: None,
6085                full_scale: None,
6086            };
6087            let rows = fixture()
6088                .filter(observation.evidence_predicate().unwrap())
6089                .collect()
6090                .unwrap();
6091            assert_eq!(rows.height(), expected, "{}", kind.label());
6092        }
6093        let category = QualityObservation {
6094            kind: ObservationKind::CategoryVariants,
6095            column: "category".to_string(),
6096            affected_rows: 3,
6097            evaluated_rows: 3,
6098            fact: String::new(),
6099            normalized_category: Some("north".to_string()),
6100            files: Vec::new(),
6101            time_format: None,
6102            full_scale: None,
6103        };
6104        let rows = df!("category" => &["North", " north ", "NORTH"])
6105            .unwrap()
6106            .lazy()
6107            .filter(category.evidence_predicate().unwrap())
6108            .collect()
6109            .unwrap();
6110        assert_eq!(rows.height(), 3);
6111    }
6112
6113    /// Duplicate rows come back exactly as counted, copies together and the most
6114    /// copied first; the text a reading does not parse is exactly the text the
6115    /// profile left out of its count; and a run over kept rows keeps a few of each.
6116    #[test]
6117    fn duplicate_and_parse_failure_evidence_match_their_counts() {
6118        let lf = df!(
6119            "id" => &[1i64, 2, 1, 3, 2, 1, 4],
6120            "code" => &["10", "20", "10", "3x", "20", "10", "n/a"],
6121        )
6122        .unwrap()
6123        .lazy();
6124        let plan = DataQualityPlan {
6125            compute: QualityCompute::Sample,
6126            dataset_rows: 100,
6127            ..DataQualityPlan::default()
6128        };
6129        let (results, kept) =
6130            compute_data_quality_kept(&lf, Some(7), &plan, None, false, None).unwrap();
6131        assert!(kept.is_some(), "the rows the run read are kept");
6132        let identity = results.identity.as_ref().unwrap();
6133        assert_eq!((identity.duplicate_groups, identity.rows_involved), (2, 5));
6134        let rows = duplicate_rows(lf.clone(), &["id".into(), "code".into()], false).unwrap();
6135        assert_eq!(rows.height(), identity.rows_involved);
6136        let ids = rows
6137            .column("id")
6138            .unwrap()
6139            .i64()
6140            .unwrap()
6141            .into_no_null_iter()
6142            .collect::<Vec<_>>();
6143        assert_eq!(ids, [1, 1, 1, 2, 2], "copies together, most copied first");
6144        let streamed = duplicate_rows(lf.clone(), &["id".into(), "code".into()], true).unwrap();
6145        assert!(streamed.equals(&rows), "the streaming engine agrees");
6146        assert_eq!(
6147            identity.examples,
6148            [
6149                DuplicateExample {
6150                    copies: 3,
6151                    values: vec!["1".to_string(), "\"10\"".to_string()],
6152                },
6153                DuplicateExample {
6154                    copies: 2,
6155                    values: vec!["2".to_string(), "\"20\"".to_string()],
6156                },
6157            ]
6158        );
6159
6160        // Five of seven parse: below the share a finding needs, so ask the reading
6161        // of a column that clears it.
6162        let codes = df!("code" => (0..40).map(|n| n.to_string()).chain(["n/a".to_string()]).collect::<Vec<_>>())
6163            .unwrap()
6164            .lazy();
6165        let (results, _) =
6166            compute_data_quality_kept(&codes, Some(41), &plan, None, false, None).unwrap();
6167        let profile = &results.columns[0];
6168        let (parsed, reading) = text_reading(profile).unwrap();
6169        assert_eq!((parsed, reading), (40, TextReading::WholeNumber));
6170        let failed = codes
6171            .filter(unparsed_text(profile).unwrap())
6172            .collect()
6173            .unwrap();
6174        assert_eq!(failed.height(), profile.non_null_rows() - parsed);
6175        assert_eq!(
6176            results.examples_of(ObservationKind::ParseableText, "code"),
6177            ["\"n/a\""]
6178        );
6179    }
6180
6181    #[test]
6182    fn sample_is_disclosed_and_bounded() {
6183        let plan = DataQualityPlan {
6184            compute: QualityCompute::Sample,
6185            dataset_rows: 2,
6186            sample_seed: 7,
6187            ..DataQualityPlan::default()
6188        };
6189        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6190        assert_eq!(results.precision, QualityPrecision::Sampled);
6191        assert_eq!(results.total_rows, Some(4));
6192        assert_eq!(results.evaluated_rows, 2);
6193    }
6194
6195    #[test]
6196    /// The sampler counts the whole scope as it samples it, so a sampled run knows the
6197    /// total it was drawn from even when no count was cached; metadata still does not.
6198    fn a_sampled_run_knows_the_total_it_was_drawn_from() {
6199        let frame = DataFrame::new(
6200            100,
6201            vec![Column::new("id".into(), (0..100).collect::<Vec<_>>())],
6202        )
6203        .unwrap()
6204        .lazy();
6205        let plan = DataQualityPlan {
6206            dataset_rows: 10,
6207            ..DataQualityPlan::default()
6208        };
6209        let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6210        assert_eq!(results.total_rows, Some(100));
6211        assert_eq!(results.evaluated_rows, 10);
6212        assert_eq!(results.precision, QualityPrecision::Sampled);
6213        assert_eq!(results.segments[0].total_rows, Some(100));
6214
6215        let short = frame.clone().limit(8);
6216        let results = compute_data_quality(&short, None, &plan, None, false).unwrap();
6217        assert_eq!(results.total_rows, Some(8));
6218        assert_eq!(results.evaluated_rows, 8);
6219        assert_eq!(results.precision, QualityPrecision::Exact);
6220
6221        let metadata = DataQualityPlan {
6222            compute: QualityCompute::Metadata,
6223            ..plan
6224        };
6225        let results = compute_data_quality(&frame, None, &metadata, None, false).unwrap();
6226        assert_eq!(results.total_rows, None);
6227        assert_eq!(results.evaluated_rows, 0);
6228    }
6229
6230    /// A dataset-grain sample is spread across the whole scope. A table sorted by year
6231    /// whose head is all one year must not come back with a single-value year, and the
6232    /// run knows the whole table's size rather than its head's.
6233    #[test]
6234    fn a_dataset_sample_spreads_across_a_sorted_table() {
6235        let rows = 40_000;
6236        let frame = df!(
6237            "id" => (0..rows as i64).collect::<Vec<_>>(),
6238            "year" => (0..rows).map(|row| 2020 + (row * 4 / rows) as i32).collect::<Vec<_>>(),
6239        )
6240        .unwrap()
6241        .lazy();
6242        let plan = DataQualityPlan {
6243            dataset_rows: 1_000,
6244            ..DataQualityPlan::default()
6245        };
6246        let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6247        assert_eq!(results.precision, QualityPrecision::Sampled);
6248        assert_eq!(results.evaluated_rows, 1_000);
6249        assert_eq!(results.total_rows, Some(rows));
6250        let year = results
6251            .columns
6252            .iter()
6253            .find(|profile| profile.name == "year")
6254            .unwrap();
6255        assert_eq!(year.distinct_count, Some(4), "every year is in the sample");
6256        assert!(
6257            !results
6258                .observations
6259                .iter()
6260                .any(|observation| observation.kind == ObservationKind::Constant)
6261        );
6262
6263        // Seeded: the same seed draws the same rows, another seed others.
6264        let ids = |seed| {
6265            let plan = DataQualityPlan {
6266                sample_seed: seed,
6267                ..plan.clone()
6268            };
6269            let results = compute_data_quality(&frame, None, &plan, None, false).unwrap();
6270            let id = results
6271                .columns
6272                .iter()
6273                .find(|profile| profile.name == "id")
6274                .unwrap();
6275            (id.min.clone(), id.max.clone())
6276        };
6277        assert_eq!(ids(1), ids(1));
6278        assert_ne!(ids(1), ids(2));
6279    }
6280
6281    /// Partition segments are named as the directory names them and read in the
6282    /// order their values count, so "previous" is the partition before; a segment's
6283    /// drill-in puts the measure that moved most first.
6284    #[test]
6285    fn partitions_compare_with_the_one_before_in_value_order() {
6286        let years = (0..300)
6287            .map(|row| [9i64, 10, 11][row / 100])
6288            .collect::<Vec<_>>();
6289        // Year 10 loses a tenth of its prices; 11 has them all again.
6290        let price = (0..300)
6291            .map(|row| (!(100..110).contains(&row)).then_some(row as f64))
6292            .collect::<Vec<_>>();
6293        let frame = df!("year" => years, "price" => price).unwrap().lazy();
6294        let plan = DataQualityPlan {
6295            compute: QualityCompute::Full,
6296            grain: QualityGrain::Partition("year".to_string()),
6297            comparison: QualityComparison::Previous,
6298            ..DataQualityPlan::default()
6299        };
6300        let results = compute_data_quality(&frame, Some(300), &plan, None, false).unwrap();
6301        let labels = results
6302            .segments
6303            .iter()
6304            .map(|segment| segment.label.as_str())
6305            .collect::<Vec<_>>();
6306        assert_eq!(labels, ["year=9", "year=10", "year=11"]);
6307        assert_eq!(results.segments[0].compared_with, None);
6308        assert_eq!(results.segments[1].compared_with.as_deref(), Some("year=9"));
6309        assert_eq!(
6310            results.segments[1].largest_change.as_deref(),
6311            Some("price nulls +10.0 pp")
6312        );
6313        let changes = segment_changes(&results, 1);
6314        assert_eq!(changes[0].column, "price");
6315        assert_eq!(changes[0].metric, QualityMetric::NullRate);
6316        assert_eq!(changes[0].before, Some(0.0));
6317        assert!((changes[0].change().unwrap() - 10.0).abs() < 1e-9);
6318        assert!(natural_cmp("part-2", "part-10").is_lt());
6319        assert!(natural_cmp("year=2024", "year=2025").is_lt());
6320    }
6321
6322    /// A thin sample a day names a change only past sampling noise, knows each
6323    /// day's exact rows, and says when a day's rows halve; Trends pools the days and
6324    /// draws columns that go missing together once.
6325    #[test]
6326    fn a_daily_sample_names_real_changes_and_counts_every_day() {
6327        let mut day = Vec::new();
6328        let (mut switched, mut noisy, mut twin) = (Vec::new(), Vec::new(), Vec::new());
6329        for d in 0..200i32 {
6330            // Day 150 delivered 20 rows instead of 50.
6331            let rows = if d == 150 { 20 } else { 50 };
6332            for r in 0..rows {
6333                let key = d * 50 + r;
6334                day.push(d);
6335                // Filled until day 100, then never.
6336                switched.push((d < 100).then_some(1i64));
6337                // About 30% missing every day: steady, and noisy on a sample.
6338                let gap = (key * 7919) % 10 < 3;
6339                noisy.push((!gap).then_some(1i64));
6340                twin.push((!gap).then_some(2i64));
6341            }
6342        }
6343        let total = day.len();
6344        let frame = df!("day" => day, "switched" => switched, "noisy" => noisy, "twin" => twin)
6345            .unwrap()
6346            .lazy()
6347            .with_column(col("day").cast(DataType::Date));
6348        let plan = DataQualityPlan {
6349            dataset_rows: 5_000,
6350            grain: QualityGrain::TimeWindows {
6351                column: "day".to_string(),
6352                every: "1d".to_string(),
6353            },
6354            comparison: QualityComparison::Previous,
6355            ..DataQualityPlan::default()
6356        };
6357        let results = compute_data_quality(&frame, Some(total), &plan, None, false).unwrap();
6358        assert_eq!(results.precision, QualityPrecision::Sampled);
6359        assert_eq!(results.segments.len(), 200);
6360        assert_eq!(
6361            results.segments[0].total_rows,
6362            Some(50),
6363            "counted, not sampled"
6364        );
6365        assert_eq!(
6366            results.segments[100].largest_change.as_deref(),
6367            Some("switched nulls +100.0 pp")
6368        );
6369        assert_eq!(
6370            results.segments[150].largest_change.as_deref(),
6371            Some("rows 20 (-60%)")
6372        );
6373        let named = results
6374            .segments
6375            .iter()
6376            .filter_map(|segment| segment.largest_change.as_deref())
6377            .collect::<Vec<_>>();
6378        assert!(
6379            named.iter().all(|change| !change.starts_with("noisy")),
6380            "a steady rate is never named: {named:?}"
6381        );
6382        // The clearest changes first; the rest keep their order.
6383        let order = segment_order(&results, true);
6384        assert!(order[..3].contains(&100) && order[..3].contains(&150));
6385
6386        let view = crate::quality_trends::trend_view(&results, QualityMetric::NullRate, 20);
6387        let (rows, per_bar) = (view.lines, view.per_bar);
6388        assert_eq!(per_bar, 10);
6389        assert_eq!(rows[0].names, ["rows"]);
6390        assert_eq!(rows[0].bars[0], Some(50.0));
6391        assert_eq!(
6392            rows[1].names,
6393            ["sampled rows"],
6394            "the sample's reach beside it"
6395        );
6396        assert_eq!(rows[2].names, ["switched"], "the column that moved leads");
6397        assert_eq!(rows[2].bars[0], Some(0.0));
6398        assert_eq!(rows[2].bars[19], Some(1.0));
6399        assert!(
6400            rows.iter()
6401                .any(|row| row.names == ["noisy".to_string(), "twin".to_string()]),
6402            "columns missing together are one line"
6403        );
6404    }
6405
6406    /// Hundreds of segments are profiled in one grouped query, and each keeps its
6407    /// own counts: every other day here has one missing price.
6408    #[test]
6409    fn every_segment_keeps_its_own_counts() {
6410        let days = 400i32;
6411        let day = (0..days * 3).map(|row| row / 3).collect::<Vec<_>>();
6412        let price = (0..days * 3)
6413            .map(|row| (!(row % 3 == 0 && (row / 3) % 2 == 0)).then_some(f64::from(row)))
6414            .collect::<Vec<_>>();
6415        let frame = df!("day" => day, "price" => price)
6416            .unwrap()
6417            .lazy()
6418            .with_column(col("day").cast(DataType::Date));
6419        let plan = DataQualityPlan {
6420            dataset_rows: 10_000,
6421            grain: QualityGrain::TimeWindows {
6422                column: "day".to_string(),
6423                every: "1d".to_string(),
6424            },
6425            ..DataQualityPlan::default()
6426        };
6427        let results =
6428            compute_data_quality(&frame, Some(days as usize * 3), &plan, None, false).unwrap();
6429        assert_eq!(results.segments.len(), days as usize);
6430        for (index, segment) in results.segments.iter().enumerate() {
6431            assert_eq!(segment.evaluated_rows, 3, "{}", segment.label);
6432            let price = segment.columns.iter().find(|c| c.name == "price").unwrap();
6433            assert_eq!(
6434                price.null_count,
6435                usize::from(index % 2 == 0),
6436                "{}",
6437                segment.label
6438            );
6439        }
6440    }
6441
6442    /// Row chunks are cut from the shared sample by where each sampled row sat, and
6443    /// a chunk's size is known without reading it.
6444    #[test]
6445    fn row_chunks_cut_the_shared_sample_where_its_rows_sat() {
6446        let frame = DataFrame::new(
6447            100,
6448            vec![Column::new("id".into(), (0..100i64).collect::<Vec<_>>())],
6449        )
6450        .unwrap()
6451        .lazy();
6452        let plan = DataQualityPlan {
6453            dataset_rows: 30,
6454            sample_seed: 1,
6455            grain: QualityGrain::RowChunks(10),
6456            ..DataQualityPlan::default()
6457        };
6458        let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
6459        assert_eq!(results.total_rows, Some(100));
6460        assert_eq!(results.evaluated_rows, 30);
6461        assert_eq!(results.precision, QualityPrecision::Sampled);
6462        assert!(!plan.requires_confirmation(), "a sample never asks first");
6463        assert_eq!(
6464            results
6465                .segments
6466                .iter()
6467                .map(|segment| segment.evaluated_rows)
6468                .sum::<usize>(),
6469            30
6470        );
6471        for segment in &results.segments {
6472            assert_eq!(segment.total_rows, Some(10), "{}", segment.label);
6473            // Every sampled id sits inside the chunk its label names.
6474            let (start, end) = segment
6475                .label
6476                .trim_start_matches("rows ")
6477                .split_once('-')
6478                .map(|(a, b)| (a.parse::<i64>().unwrap(), b.parse::<i64>().unwrap()))
6479                .unwrap();
6480            let id = segment.columns.iter().find(|c| c.name == "id").unwrap();
6481            let min = id.min.as_deref().unwrap().parse::<i64>().unwrap() + 1;
6482            let max = id.max.as_deref().unwrap().parse::<i64>().unwrap() + 1;
6483            assert!(
6484                start <= min && max <= end,
6485                "{} holds {min}..{max}",
6486                segment.label
6487            );
6488        }
6489        let again = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
6490        assert_eq!(
6491            results
6492                .segments
6493                .iter()
6494                .map(|s| s.label.clone())
6495                .collect::<Vec<_>>(),
6496            again
6497                .segments
6498                .iter()
6499                .map(|s| s.label.clone())
6500                .collect::<Vec<_>>(),
6501            "seeded"
6502        );
6503    }
6504
6505    /// A run that changes only how the rows are cut reads nothing: it cuts the rows
6506    /// the last run kept. An equal-per-value sample counted its values as it read, so
6507    /// a grain by the same column is sized from that; another grain is counted once
6508    /// and the count kept. The frame handed to the later runs fails on any read.
6509    #[test]
6510    fn a_grain_change_cuts_the_kept_sample() {
6511        let frame = df!(
6512            "id" => (0..120i64).collect::<Vec<_>>(),
6513            "region" => (0..120)
6514                .map(|row| if row < 100 { "big" } else { "small" })
6515                .collect::<Vec<_>>(),
6516            "kind" => (0..120)
6517                .map(|row| if row % 2 == 0 { "x" } else { "y" })
6518                .collect::<Vec<_>>(),
6519        )
6520        .unwrap()
6521        .lazy();
6522        // The same schema, and an error the moment a row is read.
6523        let poisoned = frame.clone().filter(
6524            (col("id") + lit(1_000i64))
6525                .strict_cast(DataType::UInt8)
6526                .is_not_null(),
6527        );
6528        let totals = |results: &DataQualityResults| {
6529            results
6530                .segments
6531                .iter()
6532                .map(|segment| (segment.label.clone(), segment.total_rows))
6533                .collect::<Vec<_>>()
6534        };
6535        let whole = DataQualityPlan {
6536            method: crate::sampling::SampleMethod::PerPartition {
6537                column: "region".into(),
6538            },
6539            dataset_rows: 5,
6540            ..DataQualityPlan::default()
6541        };
6542        let (_, kept) = compute_data_quality_kept(&frame, None, &whole, None, false, None).unwrap();
6543        let kept = kept.unwrap();
6544
6545        let by_region = DataQualityPlan {
6546            grain: QualityGrain::Partition("region".into()),
6547            ..whole.clone()
6548        };
6549        let (results, again) =
6550            compute_data_quality_kept(&poisoned, None, &by_region, None, false, Some(&kept))
6551                .unwrap();
6552        assert_eq!(
6553            totals(&results),
6554            [
6555                ("region=big".to_string(), Some(100)),
6556                ("region=small".to_string(), Some(20))
6557            ]
6558        );
6559        assert!(again.unwrap().counted.is_empty(), "counted by the sampler");
6560        let fresh = compute_data_quality(&frame, None, &by_region, None, false).unwrap();
6561        assert_eq!(
6562            format!("{:?}", results.segments),
6563            format!("{:?}", fresh.segments),
6564            "the same as reading afresh"
6565        );
6566
6567        let by_kind = DataQualityPlan {
6568            grain: QualityGrain::Partition("kind".into()),
6569            ..whole
6570        };
6571        let (counted, kept) =
6572            compute_data_quality_kept(&frame, None, &by_kind, None, false, Some(&kept)).unwrap();
6573        let (recut, _) =
6574            compute_data_quality_kept(&poisoned, None, &by_kind, None, false, kept.as_ref())
6575                .unwrap();
6576        assert_eq!(
6577            totals(&recut),
6578            [
6579                ("kind=x".to_string(), Some(60)),
6580                ("kind=y".to_string(), Some(60))
6581            ]
6582        );
6583        assert_eq!(totals(&recut), totals(&counted));
6584    }
6585
6586    /// Segments are the shared sample's rows, split. A random sample gives each
6587    /// segment its share; equal per value gives each the same number, which is how a
6588    /// small partition is measured as well as a large one.
6589    #[test]
6590    fn segments_are_the_shared_sample_split() {
6591        let frame = df!(
6592            "id" => (0..120i32).collect::<Vec<_>>(),
6593            "region" => (0..120)
6594                .map(|row| if row < 100 { "big" } else { "small" })
6595                .collect::<Vec<_>>(),
6596        )
6597        .unwrap()
6598        .lazy();
6599        let grain = QualityGrain::Partition("region".into());
6600        let random = DataQualityPlan {
6601            dataset_rows: 24,
6602            grain: grain.clone(),
6603            ..DataQualityPlan::default()
6604        };
6605        let results = compute_data_quality(&frame, None, &random, None, false).unwrap();
6606        assert_eq!(results.total_rows, Some(120));
6607        assert_eq!(results.evaluated_rows, 24);
6608        assert_eq!(
6609            results
6610                .segments
6611                .iter()
6612                .map(|segment| segment.total_rows)
6613                .collect::<Vec<_>>(),
6614            [Some(100), Some(20)],
6615            "a partition's size is counted beside the sample, not guessed from it"
6616        );
6617
6618        let equal = DataQualityPlan {
6619            method: crate::sampling::SampleMethod::PerPartition {
6620                column: "region".into(),
6621            },
6622            dataset_rows: 5,
6623            grain,
6624            ..DataQualityPlan::default()
6625        };
6626        let results = compute_data_quality(&frame, None, &equal, None, false).unwrap();
6627        assert_eq!(results.evaluated_rows, 10);
6628        assert_eq!(
6629            results
6630                .segments
6631                .iter()
6632                .map(|segment| (segment.label.as_str(), segment.evaluated_rows))
6633                .collect::<Vec<_>>(),
6634            vec![("region=big", 5), ("region=small", 5)]
6635        );
6636    }
6637
6638    #[test]
6639    fn metadata_mode_does_not_evaluate_values() {
6640        let plan = DataQualityPlan {
6641            compute: QualityCompute::Metadata,
6642            ..DataQualityPlan::default()
6643        };
6644        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6645        assert_eq!(results.precision, QualityPrecision::Metadata);
6646        assert_eq!(results.evaluated_rows, 0);
6647        assert_eq!(results.columns.len(), 5);
6648    }
6649
6650    #[test]
6651    fn compact_summary_keeps_plan_dimensions_visible() {
6652        let mut plan = DataQualityPlan::default();
6653        plan.set_row_chunks();
6654        plan.comparison = QualityComparison::Previous;
6655        assert_eq!(
6656            plan.compact_summary(),
6657            "scope current view -> grain in chunks of 1,000,000 rows -> compute 10000 rows random -> compare previous"
6658        );
6659    }
6660
6661    #[test]
6662    fn source_projection_preserves_rows_without_binary_payloads() {
6663        let source = QualitySourceContext {
6664            file_names: vec!["one.parquet".to_string()],
6665            file_starts: vec![0],
6666            row_index_column: "__datui_quality_row".to_string(),
6667            ..QualitySourceContext::default()
6668        };
6669        let frame = df!(
6670            "value" => &[1i64, 2, 3],
6671            "blob" => &[&b"one"[..], &b"two"[..], &b"three"[..]],
6672        )
6673        .unwrap()
6674        .lazy();
6675        let prepared = prepare_source_quality_scan(frame, Some(&source)).unwrap();
6676        let collected = prepared.collect().unwrap();
6677        assert_eq!(collected.height(), 3);
6678        assert_eq!(
6679            collected
6680                .column("__datui_quality_row")
6681                .unwrap()
6682                .u32()
6683                .unwrap()
6684                .get(2),
6685            Some(2)
6686        );
6687        assert_eq!(
6688            collected.column("blob").unwrap().str().unwrap().get(0),
6689            Some(crate::widgets::datatable::binary_stub())
6690        );
6691    }
6692
6693    #[test]
6694    fn scope_commands_round_trip_and_reject_invalid_ranges() {
6695        for command in [
6696            "view",
6697            "source",
6698            "rows 2..9",
6699            "files 1,3",
6700            "partition region=west",
6701            "time event=2024-01-01..2024-02-01",
6702        ] {
6703            let scope = QualityScope::parse_command(command).unwrap();
6704            assert_eq!(
6705                QualityScope::parse_command(&scope.command()).unwrap(),
6706                scope
6707            );
6708        }
6709        assert!(QualityScope::parse_command("rows 0..10").is_err());
6710        assert!(QualityScope::parse_command("rows 10..2").is_err());
6711        assert!(QualityScope::parse_command("files 0").is_err());
6712        assert!(QualityScope::parse_command("time event=2024-03-01..2024-01-01").is_err());
6713    }
6714
6715    #[test]
6716    fn scoped_frames_select_exact_view_source_file_partition_and_time_rows() {
6717        let frame = df!(
6718            "id" => &[1i32, 2, 3, 4, 5],
6719            "region" => &["west", "east", "west", "east", "west"],
6720            "day" => &[0i32, 1, 2, 3, 4],
6721        )
6722        .unwrap()
6723        .lazy()
6724        .with_columns([col("day").cast(DataType::Date)]);
6725        let ids = |scope: QualityScope, frame: LazyFrame, source: Option<&QualitySourceContext>| {
6726            let df = apply_quality_scope(frame, &scope, source)
6727                .unwrap()
6728                .collect()
6729                .unwrap();
6730            df.column("id")
6731                .unwrap()
6732                .i32()
6733                .unwrap()
6734                .into_no_null_iter()
6735                .collect::<Vec<_>>()
6736        };
6737        assert_eq!(
6738            ids(
6739                QualityScope::ViewRows { start: 2, end: 4 },
6740                frame.clone(),
6741                None
6742            ),
6743            vec![2, 3, 4]
6744        );
6745        let source = QualitySourceContext {
6746            file_names: vec!["one".into(), "two".into(), "three".into()],
6747            file_starts: vec![0, 2, 4],
6748            row_index_column: "__row".into(),
6749            ..QualitySourceContext::default()
6750        };
6751        assert_eq!(
6752            ids(
6753                QualityScope::SourceFiles(vec![1, 3]),
6754                frame.clone().with_row_index("__row", None),
6755                Some(&source)
6756            ),
6757            vec![1, 2, 5]
6758        );
6759        assert_eq!(
6760            ids(
6761                QualityScope::SourcePartition {
6762                    column: "region".into(),
6763                    value: "west".into()
6764                },
6765                frame.clone(),
6766                None
6767            ),
6768            vec![1, 3, 5]
6769        );
6770        assert_eq!(
6771            ids(
6772                QualityScope::SourceTimeRange {
6773                    column: "day".into(),
6774                    start: "1970-01-02".into(),
6775                    end: "1970-01-04".into()
6776                },
6777                frame,
6778                None
6779            ),
6780            vec![2, 3]
6781        );
6782    }
6783
6784    /// A partition value compares in the column's own type, whatever the type, one
6785    /// value or a list of them; a value the type cannot read says so.
6786    #[test]
6787    fn partition_values_compare_in_the_columns_type() {
6788        let frame = df!(
6789            "id" => &[1i32, 2, 3, 4],
6790            "year" => &[2019i64, 2020, 2021, 2020],
6791            "share" => &[0.5f64, 0.25, 0.5, 1.0],
6792            "price" => &["1.50", "2.00", "1.50", "3.25"],
6793            "day" => &[19723i32, 19724, 19723, -800_000],
6794            "at" => &[0i64, 3_600_000_000, 0, 7_200_000_000],
6795        )
6796        .unwrap()
6797        .lazy()
6798        .with_columns([
6799            col("price").cast(DataType::Decimal(10, 2)),
6800            col("day").cast(DataType::Date),
6801            col("at").cast(DataType::Datetime(TimeUnit::Microseconds, None)),
6802        ]);
6803        let schema = frame.clone().collect_schema().unwrap();
6804        let ids = |column: &str, value: &str| -> Vec<i32> {
6805            let predicate = partition_predicate(column, value, &schema).unwrap();
6806            let df = frame.clone().filter(predicate).collect().unwrap();
6807            df.column("id")
6808                .unwrap()
6809                .i32()
6810                .unwrap()
6811                .into_no_null_iter()
6812                .collect()
6813        };
6814        assert_eq!(ids("year", "2020"), [2, 4]);
6815        assert_eq!(ids("year", "2019, 2021"), [1, 3]);
6816        assert_eq!(ids("year", "2020..2021"), [2, 3, 4]);
6817        assert_eq!(ids("share", "0.5"), [1, 3]);
6818        assert_eq!(ids("price", "1.5"), [1, 3], "1.5 is the column's 1.50");
6819        assert_eq!(ids("day", "2024-01-01"), [1, 3]);
6820        // A date past the calendar, as its label writes it.
6821        assert_eq!(ids("day", "-800000 days since 1970-01-01"), [4]);
6822        assert_eq!(ids("at", "1970-01-01 01:00"), [2]);
6823        assert_eq!(
6824            ids("at", "1970-01-01T00:00:00, 1970-01-01 02:00"),
6825            [1, 3, 4]
6826        );
6827        let error = partition_predicate("day", "2024-13-01", &schema).unwrap_err();
6828        assert_eq!(
6829            error.to_string(),
6830            "day: \"2024-13-01\" is not a date written YYYY-MM-DD"
6831        );
6832        assert!(partition_predicate("year", "2020..soon", &schema).is_err());
6833    }
6834
6835    #[test]
6836    fn time_scope_accepts_timezone_aware_datetime_bounds() {
6837        let frame = df!("id" => &[1i32, 2, 3], "ts" => &[0i64, 1_000_000, 2_000_000])
6838            .unwrap()
6839            .lazy()
6840            .with_columns([col("ts").cast(DataType::Datetime(
6841                TimeUnit::Microseconds,
6842                Some(TimeZone::UTC),
6843            ))]);
6844        let scope =
6845            QualityScope::parse_command("time ts=1970-01-01T01:00:01+01:00..1970-01-01T00:00:02Z")
6846                .unwrap();
6847        let result = apply_quality_scope(frame, &scope, None)
6848            .unwrap()
6849            .collect()
6850            .unwrap();
6851        assert_eq!(result.column("id").unwrap().i32().unwrap().get(0), Some(2));
6852        assert_eq!(result.height(), 1);
6853    }
6854
6855    #[test]
6856    fn full_profile_accepts_list_columns() {
6857        let lists = Column::new(
6858            "items".into(),
6859            &[
6860                Series::new("".into(), &[1i32, 2]),
6861                Series::new("".into(), &[3i32]),
6862            ],
6863        );
6864        let frame = DataFrame::new(2, vec![lists]).unwrap().lazy();
6865        let plan = DataQualityPlan {
6866            compute: QualityCompute::Full,
6867            ..DataQualityPlan::default()
6868        };
6869        let result = compute_data_quality(&frame, Some(2), &plan, None, false).unwrap();
6870        assert_eq!(result.columns[0].null_count, 0);
6871        assert_eq!(result.columns[0].min_length, Some(1));
6872        assert_eq!(result.columns[0].max_length, Some(2));
6873        let sampled =
6874            compute_data_quality(&frame, Some(2), &DataQualityPlan::default(), None, false)
6875                .unwrap();
6876        assert_eq!(sampled.columns[0].min_length, Some(1));
6877        assert_eq!(sampled.columns[0].max_length, Some(2));
6878    }
6879
6880    #[test]
6881    fn row_chunks_keep_denominators_and_compare_previous() {
6882        let plan = DataQualityPlan {
6883            compute: QualityCompute::Full,
6884            grain: QualityGrain::RowChunks(2),
6885            comparison: QualityComparison::Previous,
6886            ..DataQualityPlan::default()
6887        };
6888        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
6889        assert_eq!(results.segments.len(), 2);
6890        assert_eq!(results.segments[0].evaluated_rows, 2);
6891        let first_dirty = results.segments[0]
6892            .columns
6893            .iter()
6894            .find(|column| column.name == "dirty")
6895            .unwrap();
6896        let second_dirty = results.segments[1]
6897            .columns
6898            .iter()
6899            .find(|column| column.name == "dirty")
6900            .unwrap();
6901        assert_eq!(QualityMetric::EmptyRate.value(first_dirty), Some(0.5));
6902        assert_eq!(QualityMetric::EmptyRate.value(second_dirty), Some(0.0));
6903        assert_eq!(QualityMetric::NullRate.value(second_dirty), Some(0.5));
6904        assert_eq!(
6905            results.segments[1].compared_with.as_deref(),
6906            Some("rows 1-2")
6907        );
6908        assert!(results.segments[1].largest_change.is_some());
6909
6910        let mut selected = results;
6911        let mut baseline_plan = plan.clone();
6912        baseline_plan.comparison = QualityComparison::Baseline;
6913        baseline_plan.baseline_segment = Some("rows 3-4".to_string());
6914        selected.compare_segments(&baseline_plan);
6915        assert_eq!(
6916            selected.segments[0].compared_with.as_deref(),
6917            Some("rows 3-4")
6918        );
6919        assert!(selected.segments[1].compared_with.is_none());
6920    }
6921
6922    /// A comparison worked out from the segments a report holds, as a Compare edit
6923    /// does with no read, is the comparison a fresh run with that Compare makes: on a
6924    /// full scan and on a sample, against the previous segment, the first, a chosen
6925    /// one and one that is not there.
6926    #[test]
6927    fn a_comparison_from_held_segments_matches_a_fresh_run() {
6928        let rows = 2_000usize;
6929        // Regions of different sizes and null rates, so both a row count and a rate
6930        // move between them.
6931        let region = |row: usize| match row % 10 {
6932            0..=4 => "a",
6933            5..=7 => "b",
6934            8 => "c",
6935            _ => "d",
6936        };
6937        let df = df!(
6938            "id" => (0..rows as i64).collect::<Vec<_>>(),
6939            "region" => (0..rows).map(region).collect::<Vec<_>>(),
6940            "amount" => (0..rows)
6941                .map(|row| (region(row) != "c" || row % 3 != 0).then_some(row as f64))
6942                .collect::<Vec<_>>(),
6943            "note" => (0..rows)
6944                .map(|row| if region(row) == "d" { "" } else { "ok" })
6945                .collect::<Vec<_>>(),
6946        )
6947        .unwrap()
6948        .lazy();
6949        let comparisons = [
6950            (QualityComparison::Previous, None),
6951            (QualityComparison::Baseline, None),
6952            (QualityComparison::Baseline, Some("region=c")),
6953            (QualityComparison::Baseline, Some("region=z")),
6954        ];
6955        for compute in [QualityCompute::Full, QualityCompute::Sample] {
6956            let base = DataQualityPlan {
6957                compute,
6958                dataset_rows: 1_000,
6959                sample_seed: 11,
6960                grain: QualityGrain::Partition("region".into()),
6961                ..DataQualityPlan::default()
6962            };
6963            let held = compute_data_quality(&df, Some(rows), &base, None, false).unwrap();
6964            assert_eq!(held.segments.len(), 4, "{compute:?}");
6965            for (comparison, baseline) in comparisons {
6966                let plan = DataQualityPlan {
6967                    comparison,
6968                    baseline_segment: baseline.map(str::to_string),
6969                    ..base.clone()
6970                };
6971                let fresh = compute_data_quality(&df, Some(rows), &plan, None, false).unwrap();
6972                let mut derived = held.clone();
6973                derived.compare_segments(&plan);
6974                let compared = |results: &DataQualityResults| {
6975                    results
6976                        .segments
6977                        .iter()
6978                        .map(|segment| {
6979                            (
6980                                segment.label.clone(),
6981                                segment.compared_with.clone(),
6982                                segment.largest_change.clone(),
6983                                segment.change_size.map(f64::to_bits),
6984                            )
6985                        })
6986                        .collect::<Vec<_>>()
6987                };
6988                assert_eq!(
6989                    compared(&derived),
6990                    compared(&fresh),
6991                    "{compute:?} {comparison:?} {baseline:?}"
6992                );
6993                assert!(
6994                    fresh
6995                        .segments
6996                        .iter()
6997                        .any(|segment| segment.largest_change.is_some()),
6998                    "{compute:?} {comparison:?} {baseline:?}: something to compare"
6999                );
7000            }
7001        }
7002    }
7003
7004    #[test]
7005    fn temporal_roles_produce_latency_without_name_inference() {
7006        let event = Series::new(
7007            "happened_at".into(),
7008            [Some(0i64), Some(3_600_000_000), None, Some(10_800_000_000)],
7009        )
7010        .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7011        .unwrap();
7012        let received = Series::new(
7013            "landed_at".into(),
7014            [
7015                Some(3_600_000_000i64),
7016                Some(1_800_000_000),
7017                Some(7_200_000_000),
7018                None,
7019            ],
7020        )
7021        .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7022        .unwrap();
7023        let frame = DataFrame::new(4, vec![event.into(), received.into()])
7024            .unwrap()
7025            .lazy();
7026        let mut plan = DataQualityPlan {
7027            compute: QualityCompute::Full,
7028            ..DataQualityPlan::default()
7029        };
7030
7031        let without_roles = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7032        assert!(without_roles.temporal.is_empty());
7033
7034        plan.temporal_roles = vec![
7035            TemporalRoleAssignment {
7036                role: TemporalRole::Event,
7037                column: "happened_at".to_string(),
7038                timezone: None,
7039            },
7040            TemporalRoleAssignment {
7041                role: TemporalRole::Received,
7042                column: "landed_at".to_string(),
7043                timezone: None,
7044            },
7045        ];
7046        let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7047        assert_eq!(results.temporal.len(), 1);
7048        let latency = &results.temporal[0];
7049        assert_eq!(latency.missing_start, 1);
7050        assert_eq!(latency.missing_end, 1);
7051        assert_eq!(latency.negative_count, 1);
7052        assert_eq!(latency.p50_seconds, Some(3_600));
7053    }
7054
7055    /// Every file counted, only the largest few named. A column missing from thousands
7056    /// of files must not cost one struct per file to say so.
7057    #[test]
7058    fn a_column_missing_from_many_files_counts_them_all_and_names_the_largest() {
7059        use crate::schema_union::DriftGroup;
7060
7061        const FILES: usize = 25;
7062        // File `i` holds `i + 1` rows, so the largest files are the last ones.
7063        let mut file_starts = Vec::with_capacity(FILES);
7064        let mut row = 0usize;
7065        for file in 0..FILES {
7066            file_starts.push(row);
7067            row += file + 1;
7068        }
7069        let source = QualitySourceContext {
7070            file_names: (0..FILES).map(|file| format!("{file}.parquet")).collect(),
7071            file_starts,
7072            dataset_rows: row,
7073            footers_read: FILES,
7074            // Group 1 is missing `fee`; every file is in it.
7075            file_group: vec![1; FILES],
7076            drift_groups: Arc::new(vec![
7077                DriftGroup::default(),
7078                DriftGroup {
7079                    absent: vec!["fee".into()],
7080                    unread: Vec::new(),
7081                },
7082            ]),
7083            ..QualitySourceContext::default()
7084        };
7085
7086        let observations = drift_observations(&source, None, false, &QualityWatch::default());
7087        assert_eq!(observations.len(), 1);
7088        let absent = &observations[0];
7089        assert_eq!(absent.kind, ObservationKind::Absent);
7090        assert_eq!(
7091            (absent.affected_rows, absent.evaluated_rows),
7092            (row, row),
7093            "every file is counted, not only the named ones"
7094        );
7095        assert_eq!(absent.files.len(), MAX_EVIDENCE_FILES);
7096        assert_eq!(
7097            absent.files.first().map(|file| file.number),
7098            Some(FILES),
7099            "the largest file first"
7100        );
7101        assert!(
7102            absent
7103                .fact
7104                .starts_with("25 of 25 files have no such column, largest 20 named"),
7105            "{}",
7106            absent.fact
7107        );
7108        assert_eq!(
7109            absent.evidence_scope().map(|scope| match scope {
7110                QualityScope::SourceFiles(files) => files.len(),
7111                _ => 0,
7112            }),
7113            Some(MAX_EVIDENCE_FILES),
7114            "the drill-in opens the files it named"
7115        );
7116    }
7117
7118    /// A column that is nearly a key and is not quite one: the repeats are the finding,
7119    /// and a column with three values in a hundred rows is a category, not a near-miss.
7120    #[test]
7121    fn a_nearly_unique_column_that_repeats_is_reported_with_its_repeats() {
7122        let mut ids = (0..98i64).collect::<Vec<_>>();
7123        // Two values that appear twice: 98 distinct values over 100 non-null rows.
7124        ids.push(7);
7125        ids.push(11);
7126        let frame = df!(
7127            "id" => &ids,
7128            "region" => &(0..100).map(|row| ["north", "south"][row % 2]).collect::<Vec<_>>(),
7129        )
7130        .unwrap()
7131        .lazy();
7132        let plan = DataQualityPlan {
7133            compute: QualityCompute::Full,
7134            ..DataQualityPlan::default()
7135        };
7136        let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7137        let key_like = results
7138            .observations
7139            .iter()
7140            .filter(|observation| observation.kind == ObservationKind::KeyLike)
7141            .collect::<Vec<_>>();
7142        assert_eq!(
7143            key_like
7144                .iter()
7145                .map(|o| o.column.as_str())
7146                .collect::<Vec<_>>(),
7147            vec!["id"],
7148            "two values in a hundred rows is a category, not a key that slipped"
7149        );
7150        assert_eq!(
7151            (key_like[0].affected_rows, key_like[0].evaluated_rows),
7152            (2, 100),
7153            "rows beyond one per value: non-null rows minus distinct values"
7154        );
7155        assert_eq!(
7156            key_like[0].fact,
7157            "98 distinct over 100 non-null rows (98.0000%); 2 rows beyond one per value; \"7\" appears 2 times"
7158        );
7159        // The drill-in is every row whose value is not the only one of its kind, which
7160        // is four rows for two values that each appear twice — more than the count
7161        // above it, which the detail pane says in so many words.
7162        let rows = frame
7163            .clone()
7164            .filter(key_like[0].evidence_predicate().unwrap())
7165            .collect()
7166            .unwrap();
7167        assert_eq!(rows.height(), 4);
7168
7169        // A distinct count does not extrapolate: in a sample of a large dataset every
7170        // repeated id looks unique, so the claim is not made at all.
7171        let sampled = compute_data_quality(
7172            &frame,
7173            Some(1_000_000),
7174            &DataQualityPlan {
7175                compute: QualityCompute::Sample,
7176                dataset_rows: 10,
7177                ..DataQualityPlan::default()
7178            },
7179            None,
7180            false,
7181        )
7182        .unwrap();
7183        assert_eq!(sampled.precision, QualityPrecision::Sampled);
7184        assert!(
7185            !sampled
7186                .observations
7187                .iter()
7188                .any(|observation| observation.kind == ObservationKind::KeyLike),
7189            "a sampled distinct share cannot say a column is nearly a key"
7190        );
7191    }
7192
7193    /// Data Quality reads the sample every tool reads, at its full size: past 50,000
7194    /// rows too, where it used to stop, so it and Describe measure the same rows.
7195    #[test]
7196    fn a_sample_is_read_at_its_full_size() {
7197        let rows = 80_000;
7198        let frame = df!(
7199            "id" => (0..rows as i64).collect::<Vec<_>>(),
7200            "tag" => (0..rows).map(|row| ["a", "b", "c"][row % 3]).collect::<Vec<_>>(),
7201        )
7202        .unwrap()
7203        .lazy();
7204        let plan = DataQualityPlan {
7205            dataset_rows: 60_000,
7206            ..DataQualityPlan::default()
7207        };
7208        let results = compute_data_quality(&frame, Some(rows), &plan, None, false).unwrap();
7209        assert_eq!(results.evaluated_rows, 60_000);
7210        assert_eq!(results.precision, QualityPrecision::Sampled);
7211        let tag = results
7212            .columns
7213            .iter()
7214            .find(|profile| profile.name == "tag")
7215            .unwrap();
7216        assert!(
7217            tag.dominant_value.is_some(),
7218            "the most common value is measured"
7219        );
7220        assert_eq!(
7221            results
7222                .identity
7223                .as_ref()
7224                .map(|identity| identity.evaluated_rows),
7225            Some(60_000)
7226        );
7227    }
7228
7229    /// An empty page names the one setting that fills it: a grain for Segments, time
7230    /// roles for Trends when there are dates to assign, and a grain there otherwise.
7231    #[test]
7232    fn an_empty_page_names_the_setting_that_fills_it() {
7233        let mut plan = DataQualityPlan::default();
7234        let results = DataQualityResults::empty(Some(10), &plan, &Schema::default());
7235        let setup =
7236            |page, plan: &DataQualityPlan, dates| page_setup(page, plan, Some(&results), dates);
7237        assert_eq!(
7238            setup(QualityPage::Segments, &plan, false),
7239            Some(QualitySetup::Grain)
7240        );
7241        assert_eq!(
7242            setup(QualityPage::Intervals, &plan, true),
7243            Some(QualitySetup::TimeRoles)
7244        );
7245        assert_eq!(setup(QualityPage::Intervals, &plan, false), None);
7246        assert_eq!(
7247            setup(QualityPage::Trends, &plan, true),
7248            Some(QualitySetup::Grain)
7249        );
7250        assert_eq!(setup(QualityPage::Overview, &plan, true), None);
7251        // Two roles that make no interval want a pair chosen, not more roles.
7252        let mut paired = plan.clone();
7253        paired.temporal_roles = [TemporalRole::Created, TemporalRole::Processed]
7254            .map(|role| TemporalRoleAssignment {
7255                role,
7256                column: "at".to_string(),
7257                timezone: None,
7258            })
7259            .to_vec();
7260        assert_eq!(
7261            setup(QualityPage::Intervals, &paired, true),
7262            Some(QualitySetup::Intervals)
7263        );
7264        // A chosen pair that measured nothing, as on metadata only, is not fixed by
7265        // choosing pairs again: the page says what is, and Enter opens nothing.
7266        paired.toggle_interval((TemporalRole::Created, TemporalRole::Processed));
7267        assert_eq!(setup(QualityPage::Intervals, &paired, true), None);
7268        assert_eq!(
7269            page_setup(QualityPage::Segments, &plan, None, true),
7270            None,
7271            "nothing to set up before a run"
7272        );
7273        plan.grain = QualityGrain::RowChunks(5);
7274        assert_eq!(setup(QualityPage::Segments, &plan, false), None);
7275    }
7276
7277    /// A partition scope takes one value, a list, or an inclusive range compared in the
7278    /// column's own type: 9..10 includes 10, which as text would sort before 9.
7279    #[test]
7280    fn a_partition_scope_takes_a_value_a_list_or_a_range() {
7281        let frame = df!(
7282            "year" => [Some(8i64), Some(9), Some(10), Some(11), None],
7283            "id" => [1i64, 2, 3, 4, 5],
7284        )
7285        .unwrap()
7286        .lazy();
7287        let ids = |value: &str| {
7288            let scope = QualityScope::parse_command(&format!("partition year={value}")).unwrap();
7289            let rows = apply_quality_scope(frame.clone(), &scope, None)
7290                .unwrap()
7291                .collect()
7292                .unwrap();
7293            rows.column("id")
7294                .unwrap()
7295                .i64()
7296                .unwrap()
7297                .into_no_null_iter()
7298                .collect::<Vec<_>>()
7299        };
7300        assert_eq!(ids("9"), vec![2]);
7301        assert_eq!(ids("8,11"), vec![1, 4]);
7302        assert_eq!(ids("9..10"), vec![2, 3]);
7303        assert_eq!(ids("∅"), vec![5]);
7304    }
7305
7306    /// A float measure is nearly unique by nature: prices and volumes repeat by
7307    /// coincidence, and calling that a key that slipped is noise.
7308    #[test]
7309    fn a_nearly_unique_float_is_not_a_key() {
7310        let mut prices = (0..98).map(|row| row as f64 + 0.5).collect::<Vec<_>>();
7311        prices.push(7.5);
7312        prices.push(11.5);
7313        let frame = df!("price" => &prices).unwrap().lazy();
7314        let plan = DataQualityPlan {
7315            compute: QualityCompute::Full,
7316            ..DataQualityPlan::default()
7317        };
7318        let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7319        assert!(
7320            !results
7321                .observations
7322                .iter()
7323                .any(|observation| observation.kind == ObservationKind::KeyLike)
7324        );
7325    }
7326
7327    /// One reading per text column, and only when nearly every value supports it: a
7328    /// column of names with a few numeric ones is text, not numbers stored as text.
7329    #[test]
7330    fn text_is_read_as_numbers_only_when_nearly_all_of_it_parses() {
7331        let mut names = (0..97).map(|row| format!("name {row}")).collect::<Vec<_>>();
7332        names.extend(["1", "2", "3"].map(String::from));
7333        let codes = (0..100)
7334            .map(|row| format!("{:04}", row * 37))
7335            .collect::<Vec<_>>();
7336        let amounts = (0..100).map(|row| format!("{row}.25")).collect::<Vec<_>>();
7337        let frame = df!("name" => &names, "code" => &codes, "amount" => &amounts)
7338            .unwrap()
7339            .lazy();
7340        let plan = DataQualityPlan {
7341            compute: QualityCompute::Full,
7342            ..DataQualityPlan::default()
7343        };
7344        let results = compute_data_quality(&frame, Some(100), &plan, None, false).unwrap();
7345        let readings = results
7346            .observations
7347            .iter()
7348            .filter(|observation| observation.kind == ObservationKind::ParseableText)
7349            .map(|observation| (observation.column.as_str(), observation.fact.as_str()))
7350            .collect::<Vec<_>>();
7351        assert_eq!(
7352            readings,
7353            vec![
7354                ("code", "100.00% parse as whole numbers"),
7355                ("amount", "100.00% parse as decimal numbers"),
7356            ]
7357        );
7358        let code = results
7359            .columns
7360            .iter()
7361            .find(|profile| profile.name == "code")
7362            .unwrap();
7363        // 0000 and every value under 1000 keep a zero in front.
7364        assert_eq!(code.leading_zero_count, Some(28));
7365    }
7366
7367    /// Equal null counts are a hint; the shared-null check says whether the columns
7368    /// are missing on the same rows or merely as often.
7369    #[test]
7370    fn columns_missing_together_are_found_to_share_their_rows() {
7371        let missing = |rows: &[usize]| {
7372            (0..10)
7373                .map(|row| (!rows.contains(&row)).then_some(row as f64))
7374                .collect::<Vec<_>>()
7375        };
7376        let frame = df!(
7377            "open" => missing(&[2, 5]),
7378            "close" => missing(&[2, 5]),
7379            "volume" => missing(&[3, 8]),
7380            "note" => missing(&[3, 9]),
7381        )
7382        .unwrap()
7383        .lazy();
7384        for compute in [QualityCompute::Sample, QualityCompute::Full] {
7385            let plan = DataQualityPlan {
7386                compute,
7387                ..DataQualityPlan::default()
7388            };
7389            let results = compute_data_quality(&frame, Some(10), &plan, None, false).unwrap();
7390            assert_eq!(
7391                results.shared_nulls,
7392                vec![SharedNulls {
7393                    columns: ["open", "close", "volume", "note"]
7394                        .map(String::from)
7395                        .to_vec(),
7396                    null_rows: 2,
7397                    rows_null_in_all: 0,
7398                }],
7399                "{compute:?}: four columns with two nulls each share none of them all"
7400            );
7401        }
7402        let frame = df!("open" => missing(&[2, 5]), "close" => missing(&[2, 5]))
7403            .unwrap()
7404            .lazy();
7405        let results =
7406            compute_data_quality(&frame, Some(10), &DataQualityPlan::default(), None, false)
7407                .unwrap();
7408        assert!(results.shared_nulls[0].same_rows());
7409    }
7410
7411    /// The segment comparison names the sharpest single move, not the average of all
7412    /// of them: a column that goes from never-null to always-null is the finding.
7413    #[test]
7414    fn the_largest_change_names_the_column_and_measurement_that_moved() {
7415        let frame = df!(
7416            "steady" => &[1i64, 2, 3, 4],
7417            "fee" => &[Some(1.5f64), Some(2.5), None, None],
7418        )
7419        .unwrap()
7420        .lazy();
7421        let plan = DataQualityPlan {
7422            compute: QualityCompute::Full,
7423            grain: QualityGrain::RowChunks(2),
7424            comparison: QualityComparison::Previous,
7425            ..DataQualityPlan::default()
7426        };
7427        let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7428        assert_eq!(results.segments.len(), 2);
7429        assert_eq!(
7430            results.segments[0].largest_change, None,
7431            "the first chunk has nothing to compare against"
7432        );
7433        let change = results.segments[1]
7434            .largest_change
7435            .as_deref()
7436            .expect("the second chunk compares with the first");
7437        assert!(
7438            change == "fee nulls +100.0 pp",
7439            "the column and the measurement that moved: {change}"
7440        );
7441    }
7442
7443    /// With nothing over the material threshold, a range that moved is still a move.
7444    #[test]
7445    fn a_segment_whose_rates_hold_still_reports_the_range_that_moved() {
7446        let frame = df!("reading" => &[1i64, 2, 300, 400]).unwrap().lazy();
7447        let plan = DataQualityPlan {
7448            compute: QualityCompute::Full,
7449            grain: QualityGrain::RowChunks(2),
7450            comparison: QualityComparison::Previous,
7451            ..DataQualityPlan::default()
7452        };
7453        let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7454        let change = results.segments[1].largest_change.as_deref().unwrap();
7455        assert!(
7456            change.starts_with("reading range 1..2 -> 300..400"),
7457            "no rate moved, but the values did: {change}"
7458        );
7459    }
7460
7461    /// Trends and Segments name the same chunk the same way, whichever compute budget
7462    /// produced it. The padding a lexicographic sort needs is not a label.
7463    #[test]
7464    fn row_chunk_labels_agree_between_trends_and_segments_at_every_budget() {
7465        let frame = df!(
7466            "sent" => &[
7467                "2024-01-01T00:00:00", "2024-01-01T01:00:00",
7468                "2024-01-01T02:00:00", "2024-01-01T03:00:00",
7469            ],
7470            "landed" => &[
7471                "2024-01-01T01:00:00", "2024-01-01T03:00:00",
7472                "2024-01-01T04:00:00", "2024-01-01T06:00:00",
7473            ],
7474        )
7475        .unwrap()
7476        .lazy()
7477        .with_columns([
7478            col("sent")
7479                .str()
7480                .to_datetime(None, None, StrptimeOptions::default(), lit("raise")),
7481            col("landed")
7482                .str()
7483                .to_datetime(None, None, StrptimeOptions::default(), lit("raise")),
7484        ]);
7485        let roles = vec![
7486            TemporalRoleAssignment {
7487                role: TemporalRole::Published,
7488                column: "sent".to_string(),
7489                timezone: None,
7490            },
7491            TemporalRoleAssignment {
7492                role: TemporalRole::Received,
7493                column: "landed".to_string(),
7494                timezone: None,
7495            },
7496        ];
7497        for compute in [QualityCompute::Sample, QualityCompute::Full] {
7498            let plan = DataQualityPlan {
7499                compute,
7500                grain: QualityGrain::RowChunks(2),
7501                temporal_roles: roles.clone(),
7502                ..DataQualityPlan::default()
7503            };
7504            let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7505            let segments = results
7506                .segments
7507                .iter()
7508                .map(|segment| segment.label.clone())
7509                .collect::<Vec<_>>();
7510            assert_eq!(
7511                segments,
7512                vec!["rows 1-2", "rows 3-4"],
7513                "{compute:?} segments"
7514            );
7515            let mut trends = results
7516                .temporal
7517                .iter()
7518                .map(|profile| profile.segment.clone())
7519                .collect::<Vec<_>>();
7520            trends.dedup();
7521            assert_eq!(trends, segments, "{compute:?} trends");
7522        }
7523    }
7524
7525    /// No role assigned means no latency to report, and nothing worth splitting the
7526    /// rows up to discover.
7527    #[test]
7528    fn an_unassigned_plan_reports_no_latency() {
7529        let plan = DataQualityPlan {
7530            compute: QualityCompute::Sample,
7531            grain: QualityGrain::RowChunks(2),
7532            ..DataQualityPlan::default()
7533        };
7534        let results = compute_data_quality(&fixture(), Some(4), &plan, None, false).unwrap();
7535        assert!(results.temporal.is_empty());
7536    }
7537
7538    #[test]
7539    fn source_row_map_produces_file_segments_without_profiling_hidden_columns() {
7540        let frame = df!(
7541            "value" => &[1i64, 2, 3, 4],
7542            "__row" => &[0u32, 1, 2, 3],
7543        )
7544        .unwrap()
7545        .lazy();
7546        let source = QualitySourceContext {
7547            file_names: vec!["a.parquet".to_string(), "b.parquet".to_string()],
7548            file_starts: vec![0, 2],
7549            row_index_column: "__row".to_string(),
7550            ..QualitySourceContext::default()
7551        };
7552        let plan = DataQualityPlan {
7553            compute: QualityCompute::Full,
7554            grain: QualityGrain::File,
7555            ..DataQualityPlan::default()
7556        };
7557        let results = compute_data_quality(&frame, Some(4), &plan, Some(&source), false).unwrap();
7558        assert_eq!(results.columns.len(), 1);
7559        assert_eq!(results.segments.len(), 2);
7560        assert_eq!(results.segments[0].evaluated_rows, 2);
7561        assert!(results.segments[0].label.contains("a.parquet"));
7562    }
7563
7564    #[test]
7565    fn identity_and_category_groups_keep_distinct_duplicate_semantics() {
7566        let frame = df!(
7567            "id" => &[1i64, 1, 2, 3],
7568            "category" => &["North", "North", " north ", "NORTH"],
7569        )
7570        .unwrap()
7571        .lazy();
7572        let plan = DataQualityPlan {
7573            compute: QualityCompute::Full,
7574            ..DataQualityPlan::default()
7575        };
7576        let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7577        let identity = results.identity.unwrap();
7578        assert_eq!(identity.duplicate_groups, 1);
7579        assert_eq!(identity.extra_rows, 1);
7580        assert_eq!(identity.rows_involved, 2);
7581        assert_eq!(results.category_variants.len(), 1);
7582        assert_eq!(results.category_variants[0].rows_involved, 4);
7583        let category = results
7584            .columns
7585            .iter()
7586            .find(|profile| profile.name == "category")
7587            .unwrap();
7588        assert_eq!(category.dominant_value.as_deref(), Some("North"));
7589        assert_eq!(category.dominant_count, Some(2));
7590    }
7591
7592    #[test]
7593    fn sample_identity_does_not_equate_null_with_literal_text() {
7594        let frame = df!("value" => &[None, Some("<null>"), None])
7595            .unwrap()
7596            .lazy();
7597        let plan = DataQualityPlan {
7598            dataset_rows: 3,
7599            ..DataQualityPlan::default()
7600        };
7601        let results = compute_data_quality(&frame, Some(3), &plan, None, false).unwrap();
7602        let identity = results.identity.unwrap();
7603        assert_eq!(identity.duplicate_groups, 1);
7604        assert_eq!(identity.rows_involved, 2);
7605    }
7606
7607    #[test]
7608    fn partition_and_time_window_grains_create_ordered_profiles() {
7609        let timestamps = Series::new(
7610            "event_at".into(),
7611            [0i64, 86_400_000_000, 8 * 86_400_000_000],
7612        )
7613        .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7614        .unwrap();
7615        let frame = DataFrame::new(
7616            3,
7617            vec![
7618                Column::new("partition".into(), ["a", "a", "b"]),
7619                Column::new("value".into(), [Some(1i64), None, Some(3)]),
7620                timestamps.into(),
7621            ],
7622        )
7623        .unwrap()
7624        .lazy();
7625
7626        let partition_plan = DataQualityPlan {
7627            compute: QualityCompute::Full,
7628            grain: QualityGrain::Partition("partition".to_string()),
7629            ..DataQualityPlan::default()
7630        };
7631        let partitioned =
7632            compute_data_quality(&frame, Some(3), &partition_plan, None, false).unwrap();
7633        assert_eq!(partitioned.segments.len(), 2);
7634        assert_eq!(partitioned.segments[0].evaluated_rows, 2);
7635
7636        let window_plan = DataQualityPlan {
7637            compute: QualityCompute::Full,
7638            grain: QualityGrain::TimeWindows {
7639                column: "event_at".to_string(),
7640                every: "1w".to_string(),
7641            },
7642            ..DataQualityPlan::default()
7643        };
7644        let windowed = compute_data_quality(&frame, Some(3), &window_plan, None, false).unwrap();
7645        assert_eq!(windowed.segments.len(), 2);
7646        assert!(windowed.segments[0].label.starts_with("week of "));
7647    }
7648
7649    #[test]
7650    fn an_exact_run_measures_everything_a_sampled_run_does() {
7651        let frame = df!(
7652            "when" => &["2024-01-01", "2024-01-02", "not a date", "2024-03-09"],
7653            "amount" => &["1", "2.5", "3", "bad"],
7654        )
7655        .unwrap()
7656        .lazy();
7657        // Unambiguous winners, so the two paths cannot differ by tie-breaking.
7658        let dupes = df!(
7659            "label" => &[Some("x"), Some("x"), Some("x"), Some("y"), None],
7660            "n" => &[7i64, 7, 7, 7, 1],
7661        )
7662        .unwrap()
7663        .lazy();
7664        let measured = |compute| {
7665            let plan = DataQualityPlan {
7666                compute,
7667                ..DataQualityPlan::default()
7668            };
7669            let results = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
7670            results
7671                .columns
7672                .iter()
7673                .map(|column| {
7674                    (
7675                        column.name.clone(),
7676                        column.integer_parse_count,
7677                        column.decimal_parse_count,
7678                        column.date_parse_count,
7679                        column.datetime_parse_count,
7680                    )
7681                })
7682                .collect::<Vec<_>>()
7683        };
7684        let sampled = measured(QualityCompute::Sample);
7685        assert_eq!(sampled, measured(QualityCompute::Full));
7686
7687        let dominant = |compute| {
7688            let plan = DataQualityPlan {
7689                compute,
7690                ..DataQualityPlan::default()
7691            };
7692            compute_data_quality(&dupes, Some(5), &plan, None, false)
7693                .unwrap()
7694                .columns
7695                .iter()
7696                .map(|column| (column.dominant_value.clone(), column.dominant_count))
7697                .collect::<Vec<_>>()
7698        };
7699        assert_eq!(
7700            dominant(QualityCompute::Full),
7701            vec![
7702                (Some("x".to_string()), Some(3)),
7703                (Some("7".to_string()), Some(4))
7704            ]
7705        );
7706        assert_eq!(
7707            dominant(QualityCompute::Sample),
7708            dominant(QualityCompute::Full)
7709        );
7710        // The exact run must not be the quieter of the two.
7711        assert_eq!(sampled[0].3, Some(3), "three of four values are ISO dates");
7712
7713        // RFC 3339 allows fractional seconds, and so does a bare space separator.
7714        let stamps = df!("t" => &[
7715            "2024-01-01T00:00:00Z",
7716            "2024-01-01T00:00:00.500Z",
7717            "2024-01-01T00:00:00+01:00",
7718            "2024-01-01 00:00:00",
7719            "2024-01-01T00:00:00",
7720            "garbage",
7721        ])
7722        .unwrap()
7723        .lazy();
7724        for compute in [QualityCompute::Sample, QualityCompute::Full] {
7725            let plan = DataQualityPlan {
7726                compute,
7727                ..DataQualityPlan::default()
7728            };
7729            let results = compute_data_quality(&stamps, Some(6), &plan, None, false).unwrap();
7730            assert_eq!(
7731                results.columns[0].datetime_parse_count,
7732                Some(5),
7733                "{compute:?} should accept every ISO timestamp but the garbage"
7734            );
7735        }
7736        assert_eq!(sampled[1].2, Some(3), "three of four parse as decimal");
7737        assert_eq!(sampled[1].1, Some(2), "two of four parse as integer");
7738
7739        let observed = |compute| {
7740            let plan = DataQualityPlan {
7741                compute,
7742                ..DataQualityPlan::default()
7743            };
7744            let mut kinds = compute_data_quality(&frame, Some(4), &plan, None, false)
7745                .unwrap()
7746                .observations
7747                .iter()
7748                .map(|item| (item.kind, item.column.clone()))
7749                .collect::<Vec<_>>();
7750            kinds.sort_by(|left, right| {
7751                left.1
7752                    .cmp(&right.1)
7753                    .then(format!("{:?}", left.0).cmp(&format!("{:?}", right.0)))
7754            });
7755            kinds
7756        };
7757        assert_eq!(
7758            observed(QualityCompute::Sample),
7759            observed(QualityCompute::Full)
7760        );
7761    }
7762
7763    #[test]
7764    fn a_plan_naming_a_column_the_scope_lost_does_not_kill_the_run() {
7765        let frame = df!("id" => &[1i64, 2, 3]).unwrap().lazy();
7766        // A role left over from a wider scope is simply unassigned here.
7767        let plan = DataQualityPlan {
7768            compute: QualityCompute::Full,
7769            temporal_roles: vec![
7770                TemporalRoleAssignment {
7771                    role: TemporalRole::Event,
7772                    column: "gone".to_string(),
7773                    timezone: None,
7774                },
7775                TemporalRoleAssignment {
7776                    role: TemporalRole::Received,
7777                    column: "also_gone".to_string(),
7778                    timezone: None,
7779                },
7780            ],
7781            ..DataQualityPlan::default()
7782        };
7783        for compute in [QualityCompute::Sample, QualityCompute::Full] {
7784            let results = compute_data_quality(
7785                &frame,
7786                Some(3),
7787                &DataQualityPlan {
7788                    compute,
7789                    ..plan.clone()
7790                },
7791                None,
7792                false,
7793            )
7794            .unwrap_or_else(|error| panic!("{compute:?} with a stale role: {error}"));
7795            assert!(results.temporal.is_empty());
7796            assert_eq!(results.columns.len(), 1);
7797        }
7798
7799        // A grain the scope cannot satisfy is refused by name, not by a raw error.
7800        let error = compute_data_quality(
7801            &frame,
7802            Some(3),
7803            &DataQualityPlan {
7804                grain: QualityGrain::Partition("region".to_string()),
7805                ..plan
7806            },
7807            None,
7808            false,
7809        )
7810        .expect_err("a grain column that is not in scope must be refused");
7811        assert!(
7812            error.to_string().contains("region") && error.to_string().contains("not in scope"),
7813            "the refusal should name the column: {error}"
7814        );
7815    }
7816
7817    #[test]
7818    fn every_scope_survives_a_trip_through_the_editor() {
7819        for scope in [
7820            QualityScope::CurrentView,
7821            QualityScope::WholeSource,
7822            QualityScope::FirstRows(10_000),
7823            QualityScope::FirstRows(1_000_000),
7824            QualityScope::ViewRows {
7825                start: 100,
7826                end: 200,
7827            },
7828            QualityScope::SourceFiles(vec![1, 3]),
7829            QualityScope::SourcePartition {
7830                column: "region".to_string(),
7831                value: "west".to_string(),
7832            },
7833            QualityScope::SourceTimeRange {
7834                column: "event".to_string(),
7835                start: "2024-01-01".to_string(),
7836                end: "2024-02-01".to_string(),
7837            },
7838        ] {
7839            assert_eq!(
7840                QualityScope::parse_command(&scope.command()).unwrap(),
7841                scope,
7842                "{} should come back as itself",
7843                scope.command()
7844            );
7845        }
7846        // An empty or inverted range is still refused.
7847        assert!(QualityScope::parse_command("rows 1..0").is_err());
7848        assert!(QualityScope::parse_command("rows 0..5").is_err());
7849    }
7850
7851    #[test]
7852    fn metadata_mode_does_not_need_the_grain_column() {
7853        // It reads no values, so a grain left over from a wider scope is moot.
7854        let frame = df!("id" => &[1i64, 2]).unwrap().lazy();
7855        let plan = DataQualityPlan {
7856            compute: QualityCompute::Metadata,
7857            grain: QualityGrain::Partition("gone".to_string()),
7858            ..DataQualityPlan::default()
7859        };
7860        let results = compute_data_quality(&frame, Some(2), &plan, None, false).unwrap();
7861        assert_eq!(results.precision, QualityPrecision::Metadata);
7862    }
7863
7864    #[test]
7865    fn segments_come_back_in_one_order_however_much_was_read() {
7866        // ∅ sorts after ASCII but before U+6771, so ordering by the label alone
7867        // put the unplaceable rows in the middle of one path and last in the other.
7868        let frame = df!(
7869            "region" => &[Some("a"), Some("a"), Some("\u{6771}\u{4eac}"), Some("\u{6771}\u{4eac}"), None, None],
7870            "id" => &[1i64, 2, 3, 4, 5, 6],
7871        )
7872        .unwrap()
7873        .lazy();
7874        let plan = DataQualityPlan {
7875            grain: QualityGrain::Partition("region".to_string()),
7876            ..DataQualityPlan::default()
7877        };
7878        let labels = |compute| {
7879            compute_data_quality(
7880                &frame,
7881                Some(6),
7882                &DataQualityPlan {
7883                    compute,
7884                    ..plan.clone()
7885                },
7886                None,
7887                false,
7888            )
7889            .unwrap()
7890            .segments
7891            .iter()
7892            .map(|segment| segment.label.clone())
7893            .collect::<Vec<_>>()
7894        };
7895        let full = labels(QualityCompute::Full);
7896        assert_eq!(labels(QualityCompute::Sample), full);
7897        assert_eq!(full.last().unwrap(), "region=\u{2205}");
7898    }
7899
7900    #[test]
7901    fn a_categorical_column_is_profiled_rather_than_failing_the_run() {
7902        let frame = df!(
7903            "label" => &["a", "b", "a", " c "],
7904            "n" => &[1i64, 2, 3, 4],
7905        )
7906        .unwrap()
7907        .lazy()
7908        .with_columns([col("label").cast(DataType::from_categories(Categories::global()))]);
7909        for compute in [QualityCompute::Sample, QualityCompute::Full] {
7910            let plan = DataQualityPlan {
7911                compute,
7912                ..DataQualityPlan::default()
7913            };
7914            let results = compute_data_quality(&frame, Some(4), &plan, None, false)
7915                .unwrap_or_else(|error| panic!("{compute:?} on a categorical column: {error}"));
7916            let label = results
7917                .columns
7918                .iter()
7919                .find(|column| column.name == "label")
7920                .expect("the categorical column is profiled");
7921            assert_eq!(label.null_count, 0);
7922            assert_eq!(label.distinct_count, Some(3));
7923        }
7924    }
7925
7926    #[test]
7927    fn every_offered_window_width_cuts_the_scope_it_names() {
7928        // One row per day from 1970-01-01, far enough to cross a month boundary.
7929        let day = 86_400_000_000i64;
7930        let days = 40i64;
7931        let stamps = Series::new(
7932            "event_at".into(),
7933            (0..days).map(|d| d * day).collect::<Vec<_>>(),
7934        )
7935        .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7936        .unwrap();
7937        let frame = DataFrame::new(
7938            days as usize,
7939            vec![
7940                Column::new("value".into(), (0..days).collect::<Vec<_>>()),
7941                stamps.into(),
7942            ],
7943        )
7944        .unwrap()
7945        .lazy();
7946        // 1970-01-01 was a Thursday, so 40 days touch seven Monday weeks and two months.
7947        let expected = [("1h", 40), ("1d", 40), ("1w", 7), ("1mo", 2)];
7948        for (every, segments) in expected {
7949            let plan = DataQualityPlan {
7950                compute: QualityCompute::Full,
7951                grain: QualityGrain::TimeWindows {
7952                    column: "event_at".to_string(),
7953                    every: every.to_string(),
7954                },
7955                ..DataQualityPlan::default()
7956            };
7957            let results =
7958                compute_data_quality(&frame, Some(days as usize), &plan, None, false).unwrap();
7959            assert_eq!(results.segments.len(), segments, "{every} windows");
7960            assert_eq!(
7961                results
7962                    .segments
7963                    .iter()
7964                    .map(|segment| segment.evaluated_rows)
7965                    .sum::<usize>(),
7966                days as usize,
7967                "{every} windows must account for every row"
7968            );
7969            // Named by where the window starts, to the precision its width needs.
7970            let label = &results.segments[0].label;
7971            match every {
7972                "1h" => assert_eq!(label.len(), "2024-01-01 00:00".len(), "{label}"),
7973                "1d" => assert_eq!(label.len(), "2024-01-01".len(), "{label}"),
7974                "1w" => assert!(label.starts_with("week of "), "{label}"),
7975                _ => assert_eq!(label.len(), "2024-01".len(), "{label}"),
7976            }
7977        }
7978    }
7979
7980    #[test]
7981    fn rows_without_a_window_clock_are_named_and_ordered_the_same_however_much_was_read() {
7982        let timestamps = Series::new(
7983            "event_at".into(),
7984            [Some(0i64), Some(8 * 86_400_000_000), None, None],
7985        )
7986        .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
7987        .unwrap();
7988        let frame = DataFrame::new(
7989            4,
7990            vec![
7991                Column::new("value".into(), [1i64, 2, 3, 4]),
7992                timestamps.into(),
7993            ],
7994        )
7995        .unwrap()
7996        .lazy();
7997        let plan = DataQualityPlan {
7998            grain: QualityGrain::TimeWindows {
7999                column: "event_at".to_string(),
8000                every: "1w".to_string(),
8001            },
8002            ..DataQualityPlan::default()
8003        };
8004
8005        let sampled = compute_data_quality(&frame, Some(4), &plan, None, false).unwrap();
8006        let full = compute_data_quality(
8007            &frame,
8008            Some(4),
8009            &DataQualityPlan {
8010                compute: QualityCompute::Full,
8011                ..plan.clone()
8012            },
8013            None,
8014            false,
8015        )
8016        .unwrap();
8017
8018        let labels = |results: &DataQualityResults| {
8019            results
8020                .segments
8021                .iter()
8022                .map(|segment| segment.label.clone())
8023                .collect::<Vec<_>>()
8024        };
8025        assert_eq!(labels(&sampled), labels(&full));
8026        // Two dated weeks, then the rows the clock could not place.
8027        assert_eq!(labels(&full).len(), 3);
8028        assert_eq!(labels(&full)[2], "event_at ∅");
8029        assert!(labels(&full)[0].starts_with("week of "));
8030    }
8031
8032    fn text_times() -> LazyFrame {
8033        df!(
8034            "created" => [
8035                Some("2024-01-01 08:00:00"),
8036                Some("2024-01-01 09:30:00"),
8037                Some("2024-01-02 10:00:00"),
8038                Some("not a time"),
8039                None,
8040                Some("2024-01-03 12:00:00"),
8041            ],
8042            "sent" => [
8043                Some("2024-01-01 09:00:00"),
8044                Some("2024-01-01 09:00:00"),
8045                None,
8046                Some("2024-01-02 11:00:00"),
8047                Some("2024-01-02 11:00:00"),
8048                Some("2024-01-03 12:30:00"),
8049            ],
8050        )
8051        .unwrap()
8052        .lazy()
8053    }
8054
8055    fn read_as_datetime(column: &str) -> TimeInterpretation {
8056        TimeInterpretation {
8057            column: column.to_string(),
8058            kind: TimeKind::Datetime,
8059            format: "%Y-%m-%d %H:%M:%S".to_string(),
8060        }
8061    }
8062
8063    /// Text read through a format gives time windows and intervals, sampled or read
8064    /// in full, alike. A value the format does not read is its own count, not a
8065    /// missing value, and the column's own profile stays the text it is.
8066    #[test]
8067    fn text_read_as_time_windows_and_measures_intervals() {
8068        let plan = DataQualityPlan {
8069            grain: QualityGrain::TimeWindows {
8070                column: "created".to_string(),
8071                every: "1d".to_string(),
8072            },
8073            temporal_roles: vec![
8074                TemporalRoleAssignment {
8075                    role: TemporalRole::Event,
8076                    column: "created".to_string(),
8077                    timezone: None,
8078                },
8079                TemporalRoleAssignment {
8080                    role: TemporalRole::Received,
8081                    column: "sent".to_string(),
8082                    timezone: None,
8083                },
8084            ],
8085            time_formats: vec![read_as_datetime("created"), read_as_datetime("sent")],
8086            ..DataQualityPlan::default()
8087        };
8088        for compute in [QualityCompute::Sample, QualityCompute::Full] {
8089            let plan = DataQualityPlan {
8090                compute,
8091                ..plan.clone()
8092            };
8093            let results = compute_data_quality(&text_times(), Some(6), &plan, None, false)
8094                .unwrap_or_else(|error| panic!("{compute:?}: {error}"));
8095            let labels = results
8096                .segments
8097                .iter()
8098                .map(|segment| segment.label.clone())
8099                .collect::<Vec<_>>();
8100            assert_eq!(
8101                labels,
8102                ["2024-01-01", "2024-01-02", "2024-01-03", "created ∅"],
8103                "{compute:?}"
8104            );
8105            let (unparsed_start, missing_start) =
8106                results
8107                    .temporal
8108                    .iter()
8109                    .fold((0, 0), |(unparsed, missing), latency| {
8110                        (
8111                            unparsed + latency.unparsed_start,
8112                            missing + latency.missing_start,
8113                        )
8114                    });
8115            assert_eq!((unparsed_start, missing_start), (1, 1), "{compute:?}");
8116            let first_day = results
8117                .temporal
8118                .iter()
8119                .find(|latency| latency.segment == "2024-01-01")
8120                .unwrap();
8121            // 08:00 to 09:00, and 09:30 to 09:00.
8122            assert_eq!(first_day.negative_count, 1, "{compute:?}");
8123            assert_eq!(first_day.max_seconds, Some(3_600), "{compute:?}");
8124
8125            let unparsed = results
8126                .observations
8127                .iter()
8128                .find(|observation| observation.kind == ObservationKind::UnparsedTime)
8129                .unwrap_or_else(|| panic!("{compute:?}: no unparsed finding"));
8130            assert_eq!(unparsed.column, "created");
8131            assert_eq!((unparsed.affected_rows, unparsed.evaluated_rows), (1, 5));
8132            let rows = text_times()
8133                .filter(unparsed.evidence_predicate().unwrap())
8134                .collect()
8135                .unwrap();
8136            assert_eq!(rows.height(), 1, "the evidence is the unread value");
8137            let created = results
8138                .columns
8139                .iter()
8140                .find(|column| column.name == "created")
8141                .unwrap();
8142            assert_eq!(created.dtype, DataType::String, "still text to every check");
8143            assert_eq!(created.null_count, 1);
8144        }
8145    }
8146
8147    /// A role on text with no format measures no interval, and a time-window grain on
8148    /// it is refused with the remedy, rather than failing somewhere inside a read.
8149    #[test]
8150    fn text_without_a_format_is_not_read_as_time() {
8151        let plan = DataQualityPlan {
8152            temporal_roles: vec![
8153                TemporalRoleAssignment {
8154                    role: TemporalRole::Event,
8155                    column: "created".to_string(),
8156                    timezone: None,
8157                },
8158                TemporalRoleAssignment {
8159                    role: TemporalRole::Received,
8160                    column: "sent".to_string(),
8161                    timezone: None,
8162                },
8163            ],
8164            ..DataQualityPlan::default()
8165        };
8166        let results = compute_data_quality(&text_times(), Some(6), &plan, None, false).unwrap();
8167        assert!(results.temporal.is_empty());
8168        let windows = DataQualityPlan {
8169            grain: QualityGrain::TimeWindows {
8170                column: "created".to_string(),
8171                every: "1d".to_string(),
8172            },
8173            ..DataQualityPlan::default()
8174        };
8175        let error = compute_data_quality(&text_times(), Some(6), &windows, None, false)
8176            .unwrap_err()
8177            .to_string();
8178        assert!(error.contains("Text as time"), "{error}");
8179    }
8180
8181    /// The formats Setup offers read on screen what a run reads: chrono for the
8182    /// examples, Polars for the run, one answer.
8183    #[test]
8184    fn every_offered_format_reads_its_example_the_same_way_twice() {
8185        let samples = [
8186            "2024-01-31 08:15:00",
8187            "2024-01-31T08:15:00",
8188            "2024-01-31 08:15:00.250",
8189            "2024-01-31T08:15:00.5",
8190            "2024-01-31T08:15:00Z",
8191            "2024-01-31T08:15:00.250+05:00",
8192            "2024-01-31 08:15:00-0500",
8193            "2024-01-31 08:15",
8194            "2024-01-31",
8195            "20240131",
8196            "01/31/2024 08:15:00",
8197            "01/31/2024 08:15:00 AM",
8198            "31/01/2024 08:15:00",
8199            "31.01.2024 08:15:00",
8200            "01/31/2024",
8201            "31/01/2024",
8202            "31.01.2024",
8203            "not a time",
8204        ];
8205        let frame = df!("text" => samples).unwrap().lazy();
8206        for (kind, format) in TIME_FORMATS {
8207            let interpretation = TimeInterpretation {
8208                column: "text".to_string(),
8209                kind,
8210                format: format.to_string(),
8211            };
8212            let parsed = frame
8213                .clone()
8214                .select([interpretation.expr().is_not_null().alias("read")])
8215                .collect()
8216                .unwrap();
8217            let read = parsed.column("read").unwrap().bool().unwrap().clone();
8218            let mut any = false;
8219            for (index, sample) in samples.iter().enumerate() {
8220                let polars = read.get(index).unwrap_or(false);
8221                any |= polars;
8222                assert_eq!(
8223                    interpretation.reads(sample),
8224                    polars,
8225                    "{format} on {sample:?}"
8226                );
8227            }
8228            assert!(any, "{format} reads none of the examples");
8229        }
8230    }
8231
8232    /// A run names each stage once as it enters it, and says whether the stage reads
8233    /// the source; a cancelled run stops at the next stage instead of finishing.
8234    #[test]
8235    fn a_run_reports_its_stages_and_stops_when_cancelled() {
8236        let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8237        let seen = Arc::clone(&stages);
8238        let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8239        let plan = DataQualityPlan {
8240            grain: QualityGrain::Partition("constant".to_string()),
8241            ..DataQualityPlan::default()
8242        };
8243        let (results, kept) =
8244            compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &watch);
8245        results.unwrap();
8246        let stages = stages.lock().unwrap().clone();
8247        assert_eq!(stages.first().unwrap().stage, QualityStage::Preparing);
8248        assert_eq!(stages.last().unwrap().stage, QualityStage::Assembling);
8249        let read = stages
8250            .iter()
8251            .find(|phase| phase.stage == QualityStage::ReadingSample)
8252            .unwrap();
8253        assert!(read.reads_source);
8254        assert!(
8255            stages
8256                .iter()
8257                .filter(|phase| phase.stage == QualityStage::ProfilingColumns)
8258                .all(|phase| !phase.reads_source),
8259            "the sample is profiled in memory"
8260        );
8261        let distinct = stages
8262            .iter()
8263            .map(|phase| phase.stage.label())
8264            .collect::<std::collections::HashSet<_>>();
8265        assert_eq!(
8266            stages.len(),
8267            distinct.len(),
8268            "each stage said once: {stages:?}"
8269        );
8270
8271        // The same plan again reuses the rows it read, and says so.
8272        let again = Arc::new(std::sync::Mutex::new(Vec::new()));
8273        let seen = Arc::clone(&again);
8274        let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8275        compute_data_quality_watched(
8276            &fixture(),
8277            Some(4),
8278            &plan,
8279            None,
8280            false,
8281            kept.as_ref(),
8282            &watch,
8283        )
8284        .0
8285        .unwrap();
8286        let again = again.lock().unwrap().clone();
8287        assert!(again.iter().all(|phase| !phase.reads_source), "{again:?}");
8288        assert!(
8289            again
8290                .iter()
8291                .any(|phase| phase.stage == QualityStage::ReusingSample)
8292        );
8293
8294        let cancelled = QualityWatch::default();
8295        cancelled.cancel();
8296        let (results, kept) =
8297            compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &cancelled);
8298        assert_eq!(results.unwrap_err().to_string(), crate::sampling::CANCELLED);
8299        assert!(kept.is_none(), "stopped before its read, it read nothing");
8300
8301        // Stopped after its read, the run still hands its rows back: they were paid for.
8302        let late = QualityWatch::default();
8303        let stopper = late.clone();
8304        let late = QualityWatch {
8305            report: Some(Arc::new(move |phase: QualityPhase| {
8306                if phase.stage == QualityStage::ProfilingColumns {
8307                    stopper.cancel();
8308                }
8309            })),
8310            ..late
8311        };
8312        let (results, kept) =
8313            compute_data_quality_watched(&fixture(), Some(4), &plan, None, false, None, &late);
8314        assert!(results.is_err());
8315        assert!(kept.is_some(), "the sample it read comes back");
8316    }
8317
8318    /// A table that counts the rows read from it. Every pass over it runs its rows
8319    /// through the filter, whatever the pass selects, so a test counts reads rather
8320    /// than inferring them from the stages a run names.
8321    fn counting_table(rows: usize) -> (DataFrame, LazyFrame, Arc<std::sync::atomic::AtomicUsize>) {
8322        let start = chrono::NaiveDate::from_ymd_opt(2023, 12, 18)
8323            .unwrap()
8324            .and_hms_opt(0, 0, 0)
8325            .unwrap();
8326        // Every 97 minutes: a stride that lands in every hour of the day, across
8327        // week and month boundaries.
8328        let at: Vec<chrono::NaiveDateTime> = (0..rows)
8329            .map(|row| start + chrono::Duration::minutes(row as i64 * 97))
8330            .collect();
8331        let micros = |times: &[chrono::NaiveDateTime]| {
8332            times
8333                .iter()
8334                .map(|time| time.and_utc().timestamp_micros())
8335                .collect::<Vec<_>>()
8336        };
8337        let sent: Vec<chrono::NaiveDateTime> = at
8338            .iter()
8339            .enumerate()
8340            .map(|(row, time)| *time + chrono::Duration::seconds(30 + (row % 7) as i64))
8341            .collect();
8342        let datetime = DataType::Datetime(TimeUnit::Microseconds, None);
8343        let df = DataFrame::new(
8344            rows,
8345            vec![
8346                Column::new("id".into(), (0..rows as i64).collect::<Vec<_>>()),
8347                Column::new("at".into(), micros(&at))
8348                    .cast(&datetime)
8349                    .unwrap(),
8350                Column::new("sent".into(), micros(&sent))
8351                    .cast(&datetime)
8352                    .unwrap(),
8353                Column::new(
8354                    "sent_text".into(),
8355                    sent.iter()
8356                        .map(|time| time.format("%m/%d/%Y %H:%M:%S").to_string())
8357                        .collect::<Vec<_>>(),
8358                ),
8359                Column::new(
8360                    "region".into(),
8361                    (0..rows)
8362                        .map(|row| ["North", "South", "East"][row % 3])
8363                        .collect::<Vec<_>>(),
8364                ),
8365            ],
8366        )
8367        .unwrap();
8368        let read = Arc::new(std::sync::atomic::AtomicUsize::new(0));
8369        let counter = Arc::clone(&read);
8370        let lf = df.clone().lazy().filter(col("id").map(
8371            move |column| {
8372                counter.fetch_add(column.len(), std::sync::atomic::Ordering::Relaxed);
8373                Ok(column.is_not_null().into_column())
8374            },
8375            |_, field| Ok(Field::new(field.name().clone(), DataType::Boolean)),
8376        ));
8377        (df, lf, read)
8378    }
8379
8380    fn segment_totals(results: &DataQualityResults) -> BTreeMap<String, Option<usize>> {
8381        results
8382            .segments
8383            .iter()
8384            .map(|segment| (segment.label.clone(), segment.total_rows))
8385            .collect()
8386    }
8387
8388    /// Every sampled segment's total is its rows in a full scan of the plain table.
8389    fn assert_exact(results: &DataQualityResults, df: &DataFrame, plan: &DataQualityPlan) {
8390        let full = DataQualityPlan {
8391            compute: QualityCompute::Full,
8392            ..plan.clone()
8393        };
8394        let exact = segment_totals(
8395            &compute_data_quality(&df.clone().lazy(), None, &full, None, false).unwrap(),
8396        );
8397        assert!(!results.segments.is_empty());
8398        for (label, total) in segment_totals(results) {
8399            assert_eq!(total, exact[&label], "{label} of {:?}", plan.grain);
8400        }
8401        // Every segment with rows is one the sample drew or one it missed, never
8402        // both, and a missed one has its exact rows.
8403        for missed in &results.unsampled_segments {
8404            assert!(
8405                !results
8406                    .segments
8407                    .iter()
8408                    .any(|segment| segment.label == missed.label),
8409                "{} of {:?}",
8410                missed.label,
8411                plan.grain
8412            );
8413            assert_eq!(Some(missed.total_rows), exact[&missed.label]);
8414        }
8415        if results
8416            .segments
8417            .iter()
8418            .all(|segment| segment.total_rows.is_some())
8419        {
8420            assert_eq!(
8421                results.segments.len() + results.unsampled_segments.len(),
8422                exact.len(),
8423                "{:?}",
8424                plan.grain
8425            );
8426        }
8427    }
8428
8429    /// The rows each edit reads, counted at the table: the "What edits should cost"
8430    /// table of #415 for a streamed sample. The first run counts its daily segments
8431    /// in the pass that samples; roles, text read as time on a role, a coarser window
8432    /// the daily counts nest in, and row chunks read nothing; a finer window, a
8433    /// partition and a grain on newly interpreted text each read their key once; a
8434    /// new seed is a new sample. Every total is the exact one a full scan finds.
8435    #[test]
8436    fn each_edit_reads_only_what_it_needs() {
8437        let rows = 3_000;
8438        let (df, lf, read) = counting_table(rows);
8439        let daily = DataQualityPlan {
8440            dataset_rows: 300,
8441            sample_seed: 5,
8442            grain: QualityGrain::TimeWindows {
8443                column: "at".into(),
8444                every: "1d".into(),
8445            },
8446            ..DataQualityPlan::default()
8447        };
8448        let roles = vec![
8449            TemporalRoleAssignment {
8450                role: TemporalRole::Event,
8451                column: "at".into(),
8452                timezone: None,
8453            },
8454            TemporalRoleAssignment {
8455                role: TemporalRole::Received,
8456                column: "sent_text".into(),
8457                timezone: None,
8458            },
8459        ];
8460        let sent_format = TimeInterpretation {
8461            column: "sent_text".into(),
8462            kind: TimeKind::Datetime,
8463            format: "%m/%d/%Y %H:%M:%S".into(),
8464        };
8465        let window = |column: &str, every: &str| QualityGrain::TimeWindows {
8466            column: column.into(),
8467            every: every.into(),
8468        };
8469        let with = |grain: QualityGrain| DataQualityPlan {
8470            grain,
8471            temporal_roles: roles.clone(),
8472            time_formats: vec![sent_format.clone()],
8473            ..daily.clone()
8474        };
8475        let mut kept: Option<QualitySample> = None;
8476        let mut run = |plan: &DataQualityPlan, reuse: bool| {
8477            read.store(0, std::sync::atomic::Ordering::Relaxed);
8478            let (results, acquired) = compute_data_quality_kept(
8479                &lf,
8480                None,
8481                plan,
8482                None,
8483                false,
8484                if reuse { kept.as_ref() } else { None },
8485            )
8486            .unwrap();
8487            kept = acquired;
8488            (results, read.load(std::sync::atomic::Ordering::Relaxed))
8489        };
8490        let passes = |read: usize| read as f64 / rows as f64;
8491
8492        let (first, reads) = run(&daily, false);
8493        assert_eq!(passes(reads), 1.0, "sampled and counted in one pass");
8494        assert_eq!(first.precision, QualityPrecision::Sampled);
8495        assert_exact(&first, &df, &daily);
8496
8497        // A role: the rows are all here. Text read as time for a role, the same.
8498        let roled = DataQualityPlan {
8499            temporal_roles: roles.clone(),
8500            ..daily.clone()
8501        };
8502        let (_, reads) = run(&roled, true);
8503        assert_eq!(reads, 0, "a role edit reads nothing");
8504        let interpreted = with(daily.grain.clone());
8505        let (results, reads) = run(&interpreted, true);
8506        assert_eq!(reads, 0, "an interpretation edit reads nothing");
8507        assert!(!results.temporal.is_empty(), "and measures the interval");
8508
8509        // Days nest in weeks and months: summed, not read.
8510        for every in ["1w", "1mo"] {
8511            let plan = with(window("at", every));
8512            let (results, reads) = run(&plan, true);
8513            assert_eq!(reads, 0, "{every} from the daily counts");
8514            assert_exact(&results, &df, &plan);
8515        }
8516        // An hour does not come from a day, nor a region from time: one count each.
8517        for grain in [window("at", "1h"), QualityGrain::Partition("region".into())] {
8518            let plan = with(grain.clone());
8519            let (results, reads) = run(&plan, true);
8520            assert_eq!(passes(reads), 1.0, "{grain:?} is counted");
8521            assert_exact(&results, &df, &plan);
8522            let (_, reads) = run(&plan, true);
8523            assert_eq!(reads, 0, "{grain:?} is counted once");
8524        }
8525        // A grain on text read as time is a new key: counted once, through its format.
8526        let plan = with(window("sent_text", "1d"));
8527        let (results, reads) = run(&plan, true);
8528        assert_eq!(passes(reads), 1.0);
8529        assert_exact(&results, &df, &plan);
8530
8531        // Row chunks: every row's position was kept by the first read.
8532        let chunks = with(QualityGrain::RowChunks(500));
8533        let (results, reads) = run(&chunks, true);
8534        assert_eq!(reads, 0, "row chunks after a first run read nothing");
8535        // Chunks the sample missed are named as the chunks it drew are.
8536        let fine = with(QualityGrain::RowChunks(10));
8537        let (missed, _) = run(&fine, true);
8538        assert!(!missed.unsampled_segments.is_empty());
8539        assert_exact(&missed, &df, &fine);
8540        let (fresh, _) = run(&chunks, false);
8541        assert_eq!(
8542            format!("{:?}", results.segments),
8543            format!("{:?}", fresh.segments),
8544            "the chunks a chunked read cuts"
8545        );
8546
8547        // Another seed, size or scope is other rows: the caller keys the sample by
8548        // them and hands none over, and the run reads, counting in the same pass.
8549        for plan in [
8550            DataQualityPlan {
8551                sample_seed: 6,
8552                ..daily.clone()
8553            },
8554            DataQualityPlan {
8555                dataset_rows: 400,
8556                ..daily.clone()
8557            },
8558            DataQualityPlan {
8559                scope: QualityScope::FirstRows(2_000),
8560                ..daily.clone()
8561            },
8562        ] {
8563            let scoped = apply_quality_scope(lf.clone(), &plan.scope, None).unwrap();
8564            read.store(0, std::sync::atomic::Ordering::Relaxed);
8565            let (results, _) =
8566                compute_data_quality_kept(&scoped, None, &plan, None, false, None).unwrap();
8567            assert_eq!(passes(read.load(std::sync::atomic::Ordering::Relaxed)), 1.0);
8568            let scoped = apply_quality_scope(df.clone().lazy(), &plan.scope, None)
8569                .unwrap()
8570                .collect()
8571                .unwrap();
8572            assert_exact(&results, &scoped, &plan);
8573        }
8574    }
8575
8576    /// A count taken in the sampling pass is the count a full scan finds, segment by
8577    /// segment: nulls in the key are their own segment, a zoned column is cut in UTC
8578    /// as a scan cuts it, and a date column is cut as the midnight it is. The rows it
8579    /// counted then serve a coarser window with no read.
8580    #[test]
8581    fn counts_in_the_sampling_pass_match_a_full_scan() {
8582        let (df, _, _) = counting_table(3_000);
8583        let gaps = |name: &str| {
8584            when((col("id") % lit(11i64)).eq(lit(0i64)))
8585                .then(lit(NULL))
8586                .otherwise(col(name))
8587                .alias(name)
8588        };
8589        let df = df
8590            .lazy()
8591            .with_columns([gaps("at"), gaps("region")])
8592            .with_columns([
8593                col("at")
8594                    .dt()
8595                    .replace_time_zone(
8596                        TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
8597                        lit("earliest"),
8598                        NonExistent::Null,
8599                    )
8600                    .alias("zoned"),
8601                col("at").cast(DataType::Date).alias("day"),
8602            ])
8603            .collect()
8604            .unwrap();
8605        let window = |column: &str, every: &str| QualityGrain::TimeWindows {
8606            column: column.into(),
8607            every: every.into(),
8608        };
8609        for grain in [
8610            window("at", "1h"),
8611            window("zoned", "1d"),
8612            window("day", "1d"),
8613            QualityGrain::Partition("region".into()),
8614        ] {
8615            let plan = DataQualityPlan {
8616                dataset_rows: 200,
8617                sample_seed: 3,
8618                grain: grain.clone(),
8619                ..DataQualityPlan::default()
8620            };
8621            let (results, kept) =
8622                compute_data_quality_kept(&df.clone().lazy(), None, &plan, None, false, None)
8623                    .unwrap();
8624            assert_eq!(results.precision, QualityPrecision::Sampled);
8625            assert!(
8626                results
8627                    .segments
8628                    .iter()
8629                    .any(|segment| segment.label.contains('∅')),
8630                "{grain:?} has a null segment"
8631            );
8632            assert_exact(&results, &df, &plan);
8633            let kept = kept.unwrap();
8634            assert_eq!(
8635                kept.segment_count(&plan),
8636                SegmentCount::Retained,
8637                "{grain:?}"
8638            );
8639            if let QualityGrain::TimeWindows { column, .. } = &grain {
8640                let monthly = DataQualityPlan {
8641                    grain: window(column, "1mo"),
8642                    ..plan.clone()
8643                };
8644                let (results, _) = compute_data_quality_kept(
8645                    &df.clone().lazy(),
8646                    None,
8647                    &monthly,
8648                    None,
8649                    false,
8650                    Some(&kept),
8651                )
8652                .unwrap();
8653                assert_exact(&results, &df, &monthly);
8654            }
8655        }
8656    }
8657
8658    /// Hours sum into days, weeks and months, and days into weeks and months, to the
8659    /// counts a read of the coarser window gives: on a plain, a zoned and a date
8660    /// column, across month ends, week starts and a daylight saving change. A week
8661    /// is not summed into months.
8662    #[test]
8663    fn finer_windows_sum_to_coarser_ones_exactly() {
8664        let (df, _, _) = counting_table(4_000);
8665        let df = df
8666            .lazy()
8667            .with_columns([
8668                col("at")
8669                    .dt()
8670                    .replace_time_zone(
8671                        TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
8672                        lit("earliest"),
8673                        NonExistent::Null,
8674                    )
8675                    .alias("zoned"),
8676                col("at").cast(DataType::Date).alias("day"),
8677            ])
8678            .collect()
8679            .unwrap();
8680        let count = |column: &str, every: &str| {
8681            counted_segment_totals(
8682                &df.clone().lazy(),
8683                &DataQualityPlan {
8684                    grain: QualityGrain::TimeWindows {
8685                        column: column.into(),
8686                        every: every.into(),
8687                    },
8688                    ..DataQualityPlan::default()
8689                },
8690                false,
8691            )
8692            .unwrap()
8693        };
8694        let widths = ["1h", "1d", "1w", "1mo"];
8695        for column in ["at", "zoned", "day"] {
8696            for fine in widths {
8697                for coarse in widths
8698                    .into_iter()
8699                    .filter(|coarse| window_nests(fine, coarse))
8700                {
8701                    assert_eq!(
8702                        roll_up_windows(&count(column, fine), coarse).unwrap(),
8703                        count(column, coarse),
8704                        "{column}: {fine} into {coarse}"
8705                    );
8706                }
8707            }
8708        }
8709        assert!(!window_nests("1w", "1mo"));
8710        assert!(!window_nests("1d", "1h"));
8711        assert!(!window_nests("1d", "1d"));
8712    }
8713
8714    /// Setup's account of where totals come from matches what the run does: a
8715    /// retained count, a finer count summed, a count pass, or too many to count.
8716    #[test]
8717    fn a_sample_says_where_its_segment_totals_come_from() {
8718        let (_, lf, _) = counting_table(2_000);
8719        let daily = DataQualityPlan {
8720            dataset_rows: 100,
8721            grain: QualityGrain::TimeWindows {
8722                column: "at".into(),
8723                every: "1d".into(),
8724            },
8725            ..DataQualityPlan::default()
8726        };
8727        assert_eq!(
8728            fresh_segment_count(&daily, false),
8729            SegmentCount::InSamplePass
8730        );
8731        assert_eq!(fresh_segment_count(&daily, true), SegmentCount::CountPass);
8732        let head = DataQualityPlan {
8733            method: crate::sampling::SampleMethod::FirstRows,
8734            ..daily.clone()
8735        };
8736        assert_eq!(fresh_segment_count(&head, false), SegmentCount::CountPass);
8737        let (_, kept) = compute_data_quality_kept(&lf, None, &daily, None, false, None).unwrap();
8738        let mut kept = kept.unwrap();
8739        let grain = |every: &str| DataQualityPlan {
8740            grain: QualityGrain::TimeWindows {
8741                column: "at".into(),
8742                every: every.into(),
8743            },
8744            ..daily.clone()
8745        };
8746        assert_eq!(kept.segment_count(&daily), SegmentCount::Retained);
8747        assert_eq!(
8748            kept.segment_count(&grain("1w")),
8749            SegmentCount::RolledUp("1d".into())
8750        );
8751        assert_eq!(kept.segment_count(&grain("1h")), SegmentCount::CountPass);
8752        let chunks = DataQualityPlan {
8753            grain: QualityGrain::RowChunks(100),
8754            ..daily.clone()
8755        };
8756        assert_eq!(kept.segment_count(&chunks), SegmentCount::NotNeeded);
8757
8758        // A grain whose count gave up names the remedy, and the rows stay.
8759        // Each such grain is remembered, not only the last.
8760        let by_region = DataQualityPlan {
8761            grain: QualityGrain::Partition("region".into()),
8762            ..daily.clone()
8763        };
8764        kept.too_many.push(segment_key(&grain("1h")));
8765        kept.too_many.push(segment_key(&by_region));
8766        assert_eq!(kept.segment_count(&grain("1h")), SegmentCount::TooMany);
8767        assert_eq!(kept.segment_count(&by_region), SegmentCount::TooMany);
8768        let error = compute_data_quality_kept(&lf, None, &grain("1h"), None, false, Some(&kept))
8769            .unwrap_err();
8770        assert!(
8771            error.to_string().contains("choose a coarser grain"),
8772            "{error}"
8773        );
8774    }
8775
8776    /// A stop reaches into a streamed read: the sampler ends at its next batch and
8777    /// the partial rows never become a sample.
8778    #[test]
8779    fn a_stopped_stream_is_not_a_sample() {
8780        let watch = crate::sampling::ReadWatch::default();
8781        watch.stop();
8782        let sample = crate::sampling::Sample {
8783            scope: QualityScope::CurrentView,
8784            method: crate::sampling::SampleMethod::PerPartition {
8785                column: "constant".to_string(),
8786            },
8787            rows: 1,
8788            seed: 7,
8789        };
8790        let read =
8791            crate::sampling::read_rows_watched(&fixture(), &sample, None, false, Some(&watch));
8792        let Err(error) = read else {
8793            panic!("a stopped read returned rows");
8794        };
8795        assert_eq!(error.to_string(), crate::sampling::CANCELLED);
8796    }
8797
8798    /// A CSV of `rows` rows on disk: a source whose read takes many batches.
8799    fn csv_source(rows: usize) -> (tempfile::TempDir, LazyFrame) {
8800        let dir = tempfile::tempdir().unwrap();
8801        let path = dir.path().join("rows.csv");
8802        let ids = (0..rows as i64).collect::<Vec<_>>();
8803        let labels = (0..rows)
8804            .map(|row| if row % 7 == 0 { "b" } else { "a" })
8805            .collect::<Vec<_>>();
8806        let mut df = df!("id" => ids, "label" => labels).unwrap();
8807        CsvWriter::new(std::fs::File::create(&path).unwrap())
8808            .finish(&mut df)
8809            .unwrap();
8810        let lf = LazyCsvReader::new(PlRefPath::try_from_path(&path).unwrap())
8811            .finish()
8812            .unwrap();
8813        (dir, lf)
8814    }
8815
8816    /// A full run's passes stop within a batch when cancelled mid-read, rather than
8817    /// running their collect to its end, and say they can.
8818    #[cfg(feature = "streaming")]
8819    #[test]
8820    fn a_full_run_stops_inside_its_read() {
8821        const ROWS: usize = 2_000_000;
8822        let (_dir, lf) = csv_source(ROWS);
8823        let plan = DataQualityPlan {
8824            compute: QualityCompute::Full,
8825            ..DataQualityPlan::default()
8826        };
8827        let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8828        let seen = Arc::clone(&stages);
8829        let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8830        // Cancel from inside the read, as the source yields its second batch: that
8831        // batch has yet to reach the watch above it, so the read is still under way
8832        // whatever the machine's load. A thread that waited to cancel lost the race
8833        // to a read that had finished meanwhile.
8834        let stopper = watch.clone();
8835        let batches = Arc::new(std::sync::atomic::AtomicUsize::new(0));
8836        let lf = lf.map(
8837            move |df: DataFrame| {
8838                if batches.fetch_add(1, std::sync::atomic::Ordering::Relaxed) == 1 {
8839                    stopper.cancel();
8840                }
8841                Ok(df)
8842            },
8843            OptFlags::PROJECTION_PUSHDOWN | OptFlags::PREDICATE_PUSHDOWN | OptFlags::STREAMING,
8844            None,
8845            Some("cancel inside the read"),
8846        );
8847        let (results, _) =
8848            compute_data_quality_watched(&lf, Some(ROWS), &plan, None, true, None, &watch);
8849        assert_eq!(results.unwrap_err().to_string(), crate::sampling::CANCELLED);
8850        let stages = stages.lock().unwrap().clone();
8851        let last = stages.last().unwrap();
8852        assert_eq!(last.stage, QualityStage::ProfilingColumns, "{stages:?}");
8853        assert!(last.reads_source && last.interruptible);
8854        let observed = watch.observed();
8855        assert!(
8856            observed.rows < ROWS,
8857            "stopped partway through the first pass: {observed:?}"
8858        );
8859    }
8860
8861    /// Without the streaming engine every read is one collect a cancel cannot enter,
8862    /// and no stage promises otherwise (#498).
8863    #[cfg(not(feature = "streaming"))]
8864    #[test]
8865    fn without_streaming_no_read_says_it_stops_partway() {
8866        let (_dir, lf) = csv_source(1_000);
8867        for compute in [QualityCompute::Full, QualityCompute::Sample] {
8868            let plan = DataQualityPlan {
8869                compute,
8870                ..DataQualityPlan::default()
8871            };
8872            let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8873            let seen = Arc::clone(&stages);
8874            let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8875            let (results, _) =
8876                compute_data_quality_watched(&lf, Some(1_000), &plan, None, true, None, &watch);
8877            results.unwrap();
8878            let stages = stages.lock().unwrap().clone();
8879            assert!(stages.iter().any(|phase| phase.reads_source), "{stages:?}");
8880            assert!(
8881                stages.iter().all(|phase| !phase.interruptible),
8882                "{stages:?}"
8883            );
8884        }
8885    }
8886
8887    /// A file that stores a conflicting column as dates gives a date past the
8888    /// calendar as its stored number, as the table shows it, where Polars' cast to
8889    /// text panicked and lost every file's examples (#506).
8890    #[test]
8891    fn conflict_examples_give_a_date_past_the_calendar_as_its_stored_number() {
8892        let paris = TimeZone::opt_try_new(Some("Europe/Paris")).unwrap();
8893        let values = move |name: &str| match name {
8894            "d" => Series::new("n".into(), [0, i32::MAX]).cast(&DataType::Date),
8895            "ms" => Series::new("n".into(), [0, i64::MIN + 1])
8896                .cast(&DataType::Datetime(TimeUnit::Milliseconds, None)),
8897            _ => Series::new("n".into(), [0, i64::MIN + 1])
8898                .cast(&DataType::Datetime(TimeUnit::Microseconds, paris.clone())),
8899        };
8900        let scan = QualityConflictScan(Arc::new(move |files, _| {
8901            let n = values(&files[0])?;
8902            Ok(DataFrame::new_infer_height(vec![n.into()])?.lazy())
8903        }));
8904        let mut files: Vec<QualityFileEvidence> = ["d", "ms", "us_tz"]
8905            .into_iter()
8906            .enumerate()
8907            .map(|(i, name)| QualityFileEvidence {
8908                number: i + 1,
8909                name: name.to_string(),
8910                rows: 2,
8911                stored_type: None,
8912                examples: Vec::new(),
8913            })
8914            .collect();
8915        let watch = QualityWatch::new(|_| {});
8916        for streaming in [false, true] {
8917            read_conflict_examples(&scan, "n", &mut files, streaming, &watch);
8918            let examples: Vec<&[String]> = files.iter().map(|f| f.examples.as_slice()).collect();
8919            assert_eq!(
8920                examples,
8921                [
8922                    ["1970-01-01", "2147483647 days since 1970-01-01"],
8923                    [
8924                        "1970-01-01 00:00:00.000",
8925                        "-9223372036854775807 ms since 1970-01-01 UTC"
8926                    ],
8927                    [
8928                        "1970-01-01 01:00:00.000000+01:00",
8929                        "-9223372036854775807 us since 1970-01-01 UTC"
8930                    ],
8931                ]
8932            );
8933        }
8934    }
8935
8936    /// The values a type conflict hides are read a file at a time, so a cancel stops
8937    /// between files on either engine and in any build.
8938    #[test]
8939    fn the_conflict_read_stops_between_files_on_any_engine() {
8940        let lf = df!("id" => [1i64, 2, 3]).unwrap().lazy();
8941        let source = QualitySourceContext {
8942            conflict_scan: Some(QualityConflictScan(Arc::new(|_, _| {
8943                Ok(df!("id" => ["1"]).unwrap().lazy())
8944            }))),
8945            ..QualitySourceContext::default()
8946        };
8947        let plan = DataQualityPlan {
8948            compute: QualityCompute::Full,
8949            ..DataQualityPlan::default()
8950        };
8951        let stages = Arc::new(std::sync::Mutex::new(Vec::new()));
8952        let seen = Arc::clone(&stages);
8953        let watch = QualityWatch::new(move |phase| seen.lock().unwrap().push(phase));
8954        let (results, _) =
8955            compute_data_quality_watched(&lf, Some(3), &plan, Some(&source), false, None, &watch);
8956        results.unwrap();
8957        let stages = stages.lock().unwrap().clone();
8958        let conflicts = stages
8959            .iter()
8960            .find(|phase| phase.stage == QualityStage::ReadingConflicts)
8961            .unwrap_or_else(|| panic!("{stages:?}"));
8962        assert!(conflicts.interruptible, "{stages:?}");
8963    }
8964
8965    /// A finished full run counts the rows every pass traversed; a sampled run the
8966    /// rows its sampler streamed; a run that uses rows already read reads nothing.
8967    #[test]
8968    fn a_run_counts_the_rows_its_reads_traverse() {
8969        let (_dir, lf) = csv_source(10_000);
8970        let full = DataQualityPlan {
8971            compute: QualityCompute::Full,
8972            ..DataQualityPlan::default()
8973        };
8974        let watch = QualityWatch::default();
8975        let (results, _) =
8976            compute_data_quality_watched(&lf, Some(10_000), &full, None, true, None, &watch);
8977        let reads = results.unwrap().reads.unwrap();
8978        assert_eq!(reads.reads, reads.counted, "every pass counted: {reads:?}");
8979        assert!(reads.reads >= 2, "{reads:?}");
8980        assert_eq!(reads.rows % 10_000, 0, "whole passes: {reads:?}");
8981        assert!(reads.rows >= 2 * 10_000, "{reads:?}");
8982
8983        let sampled = DataQualityPlan {
8984            dataset_rows: 100,
8985            ..DataQualityPlan::default()
8986        };
8987        let (results, kept) = compute_data_quality_watched(
8988            &lf,
8989            Some(10_000),
8990            &sampled,
8991            None,
8992            true,
8993            None,
8994            &QualityWatch::default(),
8995        );
8996        assert_eq!(
8997            results.unwrap().reads,
8998            Some(ObservedReads {
8999                reads: 1,
9000                counted: 1,
9001                rows: 10_000,
9002                copy: None,
9003            }),
9004            "the sampler streamed the scope once"
9005        );
9006        let (results, _) = compute_data_quality_watched(
9007            &lf,
9008            Some(10_000),
9009            &sampled,
9010            None,
9011            true,
9012            kept.as_ref(),
9013            &QualityWatch::default(),
9014        );
9015        assert_eq!(results.unwrap().reads, Some(ObservedReads::default()));
9016    }
9017
9018    /// Wide, nearly unique text: a report keeps only bounded pieces of it (examples
9019    /// cut short, at most 100 spelling groups), and the memory budget weighs every
9020    /// piece it keeps, the spellings' full text included.
9021    #[test]
9022    fn a_report_on_wide_text_is_weighed_by_the_text_it_holds() {
9023        let wide = "x".repeat(2_000);
9024        let rows = 600;
9025        let names = (0..rows)
9026            .map(|row| {
9027                let name = format!("Vendor {:04} {wide}", row / 2);
9028                if row % 2 == 0 {
9029                    name
9030                } else {
9031                    name.to_uppercase()
9032                }
9033            })
9034            .collect::<Vec<_>>();
9035        let df = df!(
9036            "id" => (0..rows as i64).collect::<Vec<_>>(),
9037            "name" => names,
9038            "note" => (0..rows).map(|row| format!("{row} {wide}")).collect::<Vec<_>>(),
9039        )
9040        .unwrap();
9041        for compute in [QualityCompute::Sample, QualityCompute::Full] {
9042            let plan = DataQualityPlan {
9043                compute,
9044                dataset_rows: rows,
9045                ..DataQualityPlan::default()
9046            };
9047            let results =
9048                compute_data_quality(&df.clone().lazy(), None, &plan, None, false).unwrap();
9049            assert_eq!(results.category_variants.len(), 100, "{compute:?}");
9050            let spellings = results
9051                .category_variants
9052                .iter()
9053                .map(|group| {
9054                    group.column.len()
9055                        + group.normalized.len()
9056                        + group
9057                            .variants
9058                            .iter()
9059                            .map(|(variant, _)| variant.len())
9060                            .sum::<usize>()
9061                })
9062                .sum::<usize>();
9063            assert!(spellings > 100 * 3 * 2_000, "{spellings}");
9064            // Each group's finding names its spelling again.
9065            let spellings = spellings
9066                + results
9067                    .observations
9068                    .iter()
9069                    .filter_map(|observation| observation.normalized_category.as_ref())
9070                    .map(String::len)
9071                    .sum::<usize>();
9072            assert!(
9073                results.estimated_bytes() >= spellings,
9074                "{compute:?}: {} bytes budgeted for {spellings} of text",
9075                results.estimated_bytes()
9076            );
9077            for value in results.examples.iter().flat_map(|found| &found.values) {
9078                assert!(crate::glyphs::display_width(value) <= 26, "{value}");
9079            }
9080        }
9081    }
9082
9083    /// A grain finer than a report can show: past 1,000,000 keys the count the
9084    /// sampling pass takes is dropped rather than grown with the table, the run
9085    /// names the remedy, and the rows the pass read are kept, so a coarser grain
9086    /// reads nothing. A count pass of its own stops at the same limit.
9087    #[test]
9088    fn a_count_past_a_million_keys_gives_up_and_keeps_the_rows() {
9089        let rows = crate::sampling::MAX_COUNTED_KEYS + 1;
9090        let df = df!("id" => (0..rows as i64).collect::<Vec<_>>()).unwrap();
9091        let read = Arc::new(std::sync::atomic::AtomicUsize::new(0));
9092        let counter = Arc::clone(&read);
9093        let lf = df.lazy().filter(col("id").map(
9094            move |column| {
9095                counter.fetch_add(column.len(), std::sync::atomic::Ordering::Relaxed);
9096                Ok(column.is_not_null().into_column())
9097            },
9098            |_, field| Ok(Field::new(field.name().clone(), DataType::Boolean)),
9099        ));
9100        let by_id = DataQualityPlan {
9101            dataset_rows: 1_000,
9102            grain: QualityGrain::Partition("id".into()),
9103            ..DataQualityPlan::default()
9104        };
9105        let watch = QualityWatch::default();
9106        let (results, kept) =
9107            compute_data_quality_watched(&lf, None, &by_id, None, false, None, &watch);
9108        let error = results.unwrap_err().to_string();
9109        assert!(
9110            error.contains("More than 1,000,000 segments") && error.contains("coarser grain"),
9111            "{error}"
9112        );
9113        let kept = kept.expect("the rows the pass read are kept");
9114        assert_eq!(kept.df.height(), 1_000);
9115        assert_eq!(kept.segment_count(&by_id), SegmentCount::TooMany);
9116        assert!(
9117            kept.estimated_bytes() < 1_000_000,
9118            "no map of a million keys kept"
9119        );
9120
9121        read.store(0, std::sync::atomic::Ordering::Relaxed);
9122        let dataset = DataQualityPlan {
9123            grain: QualityGrain::Dataset,
9124            ..by_id.clone()
9125        };
9126        let (results, _) =
9127            compute_data_quality_kept(&lf, None, &dataset, None, false, Some(&kept)).unwrap();
9128        assert_eq!(results.evaluated_rows, 1_000);
9129        assert_eq!(read.load(std::sync::atomic::Ordering::Relaxed), 0);
9130
9131        // Rows kept without a count, a first-rows sample, count the grain in a pass
9132        // of their own, which gives up at the same limit.
9133        let head = DataQualityPlan {
9134            method: crate::sampling::SampleMethod::FirstRows,
9135            ..by_id
9136        };
9137        let (results, kept) =
9138            compute_data_quality_watched(&lf, None, &head, None, false, None, &watch);
9139        let error = results.unwrap_err().to_string();
9140        assert!(error.contains("More than 1,000,000 segments"), "{error}");
9141        let kept = kept.expect("the head is kept");
9142        assert_eq!(kept.segment_count(&head), SegmentCount::TooMany);
9143    }
9144}
9145
9146/// Intervals between time roles: what they count, and out of what.
9147#[cfg(test)]
9148mod temporal_tests {
9149    use super::*;
9150
9151    const HOUR: i64 = 3_600_000_000;
9152
9153    fn datetimes(name: &str, values: &[Option<i64>]) -> Column {
9154        Series::new(name.into(), values)
9155            .cast(&DataType::Datetime(TimeUnit::Microseconds, None))
9156            .unwrap()
9157            .into()
9158    }
9159
9160    fn roles(pairs: &[(TemporalRole, &str)]) -> Vec<TemporalRoleAssignment> {
9161        pairs
9162            .iter()
9163            .map(|(role, column)| TemporalRoleAssignment {
9164                role: *role,
9165                column: column.to_string(),
9166                timezone: None,
9167            })
9168            .collect()
9169    }
9170
9171    /// Both ways a run measures: the sample's rows in memory, and a full scan.
9172    fn both_ways(
9173        frame: &LazyFrame,
9174        rows: usize,
9175        plan: &DataQualityPlan,
9176    ) -> [DataQualityResults; 2] {
9177        [QualityCompute::Sample, QualityCompute::Full].map(|compute| {
9178            let plan = DataQualityPlan {
9179                compute,
9180                ..plan.clone()
9181            };
9182            compute_data_quality(frame, Some(rows), &plan, None, false).unwrap()
9183        })
9184    }
9185
9186    /// Eight rows: an hour exactly, an hour and a second, two hours, a missing
9187    /// start, a missing end, both missing, half a second early, and no time at all.
9188    fn delays() -> LazyFrame {
9189        let start = [
9190            Some(0),
9191            Some(0),
9192            Some(0),
9193            None,
9194            Some(0),
9195            None,
9196            Some(500_000),
9197            Some(0),
9198        ];
9199        let end = [
9200            Some(HOUR),
9201            Some(HOUR + 1_000_000),
9202            Some(2 * HOUR),
9203            Some(HOUR),
9204            None,
9205            None,
9206            Some(0),
9207            Some(0),
9208        ];
9209        DataFrame::new(8, vec![datetimes("sent", &start), datetimes("seen", &end)])
9210            .unwrap()
9211            .lazy()
9212    }
9213
9214    /// A breach is `duration > threshold`, counted out of the rows with both ends:
9215    /// an hour exactly is not over an hour, and the rows less each end's missing
9216    /// count would be the wrong denominator, since one row misses both.
9217    #[test]
9218    fn breaches_are_out_of_rows_with_both_ends() {
9219        let plan = DataQualityPlan {
9220            temporal_roles: roles(&[
9221                (TemporalRole::Event, "sent"),
9222                (TemporalRole::Received, "seen"),
9223            ]),
9224            latency_threshold_seconds: Some(3_600),
9225            ..DataQualityPlan::default()
9226        };
9227        for results in both_ways(&delays(), 8, &plan) {
9228            let [latency] = results.temporal.as_slice() else {
9229                panic!("one interval: {:?}", results.temporal);
9230            };
9231            assert_eq!(latency.evaluated_rows, 8);
9232            assert_eq!((latency.missing_start, latency.missing_end), (2, 2));
9233            assert_eq!(latency.paired_rows, 5);
9234            assert_ne!(
9235                latency.evaluated_rows - latency.missing_start - latency.missing_end,
9236                latency.paired_rows
9237            );
9238            assert_eq!(latency.threshold_seconds, Some(3_600));
9239            assert_eq!(latency.above_threshold_count, Some(2));
9240            // Half a second early is early, though it is zero whole seconds.
9241            assert_eq!(latency.negative_count, 1);
9242            assert_eq!(latency.zero_count, 1);
9243            assert_eq!(latency.max_seconds, Some(7_200));
9244            assert_eq!(
9245                latency.count(IntervalFact::OverThreshold, &plan),
9246                Some((2, 5))
9247            );
9248            assert_eq!(latency.count(IntervalFact::MissingEnd, &plan), Some((2, 8)));
9249            assert_eq!(latency.count(IntervalFact::UnparsedStart, &plan), None);
9250        }
9251    }
9252
9253    /// Any start and end can be chosen, not only the pairs the roles suggest; the
9254    /// first choice makes the list explicit, and a role in no interval is named.
9255    #[test]
9256    fn a_chosen_pair_is_measured_and_an_unpaired_role_is_named() {
9257        let mut plan = DataQualityPlan {
9258            temporal_roles: roles(&[
9259                (TemporalRole::Created, "sent"),
9260                (TemporalRole::Processed, "seen"),
9261            ]),
9262            ..DataQualityPlan::default()
9263        };
9264        assert!(plan.interval_pairs().is_empty(), "no suggested pair");
9265        assert_eq!(
9266            plan.unpaired_roles(),
9267            vec![TemporalRole::Created, TemporalRole::Processed]
9268        );
9269        assert_eq!(
9270            plan.candidate_pairs(),
9271            vec![
9272                (TemporalRole::Created, TemporalRole::Processed),
9273                (TemporalRole::Processed, TemporalRole::Created),
9274            ]
9275        );
9276        plan.toggle_interval((TemporalRole::Created, TemporalRole::Processed));
9277        assert_eq!(
9278            plan.interval_pairs(),
9279            vec![(TemporalRole::Created, TemporalRole::Processed)]
9280        );
9281        assert!(plan.unpaired_roles().is_empty());
9282        for results in both_ways(&delays(), 8, &plan) {
9283            let [latency] = results.temporal.as_slice() else {
9284                panic!("one interval: {:?}", results.temporal);
9285            };
9286            assert_eq!(latency.label(), "created to processed");
9287            assert_eq!(latency.paired_rows, 5);
9288        }
9289
9290        // A suggested pair taken away stays away.
9291        let mut plan = DataQualityPlan {
9292            temporal_roles: roles(&[
9293                (TemporalRole::Event, "sent"),
9294                (TemporalRole::Received, "seen"),
9295            ]),
9296            ..DataQualityPlan::default()
9297        };
9298        plan.toggle_interval((TemporalRole::Event, TemporalRole::Received));
9299        assert_eq!(plan.intervals, Some(Vec::new()));
9300        assert!(plan.interval_pairs().is_empty());
9301        let results = compute_data_quality(&delays(), Some(8), &plan, None, false).unwrap();
9302        assert!(results.temporal.is_empty());
9303    }
9304
9305    /// Valid from and valid to make an interval without choosing it, and read as a
9306    /// validity period: no end is open, an end first is not valid.
9307    #[test]
9308    fn a_validity_period_counts_open_and_backwards_periods() {
9309        let days = |name: &str, values: &[Option<i32>]| -> Column {
9310            Series::new(name.into(), values)
9311                .cast(&DataType::Date)
9312                .unwrap()
9313                .into()
9314        };
9315        let frame = DataFrame::new(
9316            4,
9317            vec![
9318                days(
9319                    "from",
9320                    &[Some(19_000), Some(19_000), Some(19_010), Some(19_020)],
9321                ),
9322                days("to", &[Some(19_005), None, Some(19_009), Some(19_020)]),
9323            ],
9324        )
9325        .unwrap()
9326        .lazy();
9327        let plan = DataQualityPlan {
9328            temporal_roles: roles(&[
9329                (TemporalRole::ValidFrom, "from"),
9330                (TemporalRole::ValidTo, "to"),
9331            ]),
9332            ..DataQualityPlan::default()
9333        };
9334        for results in both_ways(&frame, 4, &plan) {
9335            let [period] = results.temporal.as_slice() else {
9336                panic!("one interval: {:?}", results.temporal);
9337            };
9338            assert!(period.is_validity());
9339            assert_eq!(period.paired_rows, 3);
9340            assert_eq!(period.missing_end, 1);
9341            assert_eq!(period.negative_count, 1);
9342            assert_eq!(period.zero_count, 1);
9343            assert_eq!(IntervalFact::MissingEnd.label(period), "Open, no end");
9344            assert_eq!(IntervalFact::Negative.label(period), "Ends first");
9345        }
9346    }
9347
9348    /// Text with an offset is an instant: `10:00+05:00` is 05:00 UTC, an hour
9349    /// before a time with no zone that reads 06:00, which is taken as UTC.
9350    #[test]
9351    fn zoned_text_compares_with_naive_time_as_utc() {
9352        let frame = DataFrame::new(
9353            2,
9354            vec![
9355                Column::new(
9356                    "stamped".into(),
9357                    ["2024-01-01T10:00:00+05:00", "2024-01-01T06:00:00Z"],
9358                ),
9359                datetimes(
9360                    "logged",
9361                    &[
9362                        Some(1_704_088_800_000_000), // 2024-01-01 06:00:00
9363                        Some(1_704_088_800_000_000),
9364                    ],
9365                ),
9366            ],
9367        )
9368        .unwrap()
9369        .lazy();
9370        let offset = TimeInterpretation {
9371            column: "stamped".to_string(),
9372            kind: TimeKind::Datetime,
9373            format: "%Y-%m-%dT%H:%M:%S%.f%#z".to_string(),
9374        };
9375        assert!(offset.zoned());
9376        assert!(offset.reads("2024-01-01T10:00:00+05:00"));
9377        assert!(offset.reads("2024-01-01T06:00:00Z"));
9378        assert!(
9379            !offset.reads("2024-01-01T06:00:00"),
9380            "no offset, no instant"
9381        );
9382        let plan = DataQualityPlan {
9383            temporal_roles: roles(&[
9384                (TemporalRole::Event, "stamped"),
9385                (TemporalRole::Received, "logged"),
9386            ]),
9387            time_formats: vec![offset],
9388            ..DataQualityPlan::default()
9389        };
9390        let schema = frame.clone().collect_schema().unwrap();
9391        assert_eq!(plan.zoned("stamped", &schema), Some(true));
9392        assert_eq!(plan.zoned("logged", &schema), Some(false));
9393        for results in both_ways(&frame, 2, &plan) {
9394            let [latency] = results.temporal.as_slice() else {
9395                panic!("one interval: {:?}", results.temporal);
9396            };
9397            assert_eq!(latency.paired_rows, 2);
9398            assert_eq!(latency.max_seconds, Some(3_600));
9399            assert_eq!((latency.zero_count, latency.negative_count), (1, 0));
9400        }
9401    }
9402
9403    /// With time windows, an interval goes in the window of the grain's column, or
9404    /// of its own start or end: a delay across midnight lands on the day it ended.
9405    #[test]
9406    fn the_window_clock_puts_an_interval_on_its_start_or_end() {
9407        let late = 1_704_150_000_000_000; // 2024-01-01 23:00:00
9408        let frame = DataFrame::new(
9409            1,
9410            vec![
9411                datetimes("sent", &[Some(late)]),
9412                datetimes("seen", &[Some(late + 2 * HOUR)]),
9413            ],
9414        )
9415        .unwrap()
9416        .lazy();
9417        let mut plan = DataQualityPlan {
9418            temporal_roles: roles(&[
9419                (TemporalRole::Event, "sent"),
9420                (TemporalRole::Received, "seen"),
9421            ]),
9422            grain: QualityGrain::TimeWindows {
9423                column: "sent".to_string(),
9424                every: "1d".to_string(),
9425            },
9426            ..DataQualityPlan::default()
9427        };
9428        assert!(plan.windows_intervals());
9429        let schema = frame.clone().collect_schema().unwrap();
9430        for (clock, day) in [
9431            (IntervalClock::Grain, "2024-01-01"),
9432            (IntervalClock::Start, "2024-01-01"),
9433            (IntervalClock::End, "2024-01-02"),
9434        ] {
9435            plan.interval_clock = clock;
9436            assert_eq!(interval_passes(&plan, &schema), 1);
9437            for results in both_ways(&frame, 1, &plan) {
9438                let segments = results
9439                    .temporal
9440                    .iter()
9441                    .map(|latency| latency.segment.as_str())
9442                    .collect::<Vec<_>>();
9443                assert_eq!(segments, vec![day], "{clock:?}");
9444            }
9445        }
9446        // By their ends, intervals ending in different columns are cut twice; by
9447        // the grain, once however many there are.
9448        plan.temporal_roles
9449            .extend(roles(&[(TemporalRole::Processed, "seen2")]));
9450        let frame = frame.with_column(col("seen").alias("seen2"));
9451        let schema = frame.clone().collect_schema().unwrap();
9452        assert_eq!(plan.interval_pairs().len(), 3);
9453        assert_eq!(interval_passes(&plan, &schema), 2);
9454        plan.interval_clock = IntervalClock::Grain;
9455        assert_eq!(interval_passes(&plan, &schema), 1);
9456    }
9457
9458    /// The rows a detail opens for a fact are the rows it counted, in the segment
9459    /// it counted them in.
9460    #[test]
9461    fn a_facts_rows_are_the_rows_it_counted() {
9462        let frame = delays().with_column(
9463            when(col("sent").is_null())
9464                .then(lit("2024-01-02"))
9465                .otherwise(lit("2024-01-01"))
9466                .str()
9467                .to_date(StrptimeOptions::default())
9468                .alias("day"),
9469        );
9470        for grain in [
9471            QualityGrain::Dataset,
9472            QualityGrain::Partition("day".to_string()),
9473            QualityGrain::TimeWindows {
9474                column: "day".to_string(),
9475                every: "1d".to_string(),
9476            },
9477        ] {
9478            let plan = DataQualityPlan {
9479                temporal_roles: roles(&[
9480                    (TemporalRole::Event, "sent"),
9481                    (TemporalRole::Received, "seen"),
9482                ]),
9483                latency_threshold_seconds: Some(3_600),
9484                grain: grain.clone(),
9485                ..DataQualityPlan::default()
9486            };
9487            for results in both_ways(&frame, 8, &plan) {
9488                assert!(!results.temporal.is_empty());
9489                for latency in &results.temporal {
9490                    for fact in IntervalFact::ALL {
9491                        let Some((count, _)) = latency.count(fact, &plan) else {
9492                            assert!(latency.evidence_predicate(fact, &plan, None).is_none());
9493                            continue;
9494                        };
9495                        let predicate = latency
9496                            .evidence_predicate(fact, &plan, None)
9497                            .unwrap_or_else(|| panic!("{grain:?} {fact:?} opens nothing"));
9498                        let rows = frame.clone().filter(predicate).collect().unwrap().height();
9499                        assert_eq!(rows, count, "{grain:?} {} {fact:?}", latency.segment);
9500                    }
9501                }
9502            }
9503        }
9504        // A row chunk is a stretch of rows, not a value to filter on.
9505        let plan = DataQualityPlan {
9506            temporal_roles: roles(&[
9507                (TemporalRole::Event, "sent"),
9508                (TemporalRole::Received, "seen"),
9509            ]),
9510            grain: QualityGrain::RowChunks(4),
9511            ..DataQualityPlan::default()
9512        };
9513        let results = compute_data_quality(&frame, Some(8), &plan, None, false).unwrap();
9514        assert!(results.temporal.iter().all(|latency| {
9515            latency
9516                .evidence_predicate(IntervalFact::Negative, &plan, None)
9517                .is_none()
9518        }));
9519    }
9520
9521    /// A segment's label finds its rows whatever the partition holds: text with
9522    /// `=` and spaces, integers, booleans, floats and datetimes (the last two write
9523    /// differently cast to text), and a zoned column's windows at every width,
9524    /// across New York's spring-forward day.
9525    #[test]
9526    fn a_facts_rows_are_found_by_any_segment_label() {
9527        let spring = 1_710_054_000_000_000i64; // 2024-03-10 07:00 UTC
9528        let zoned: Column = Series::new(
9529            "zoned".into(),
9530            [0, 3, 20, -1, -10, 40, 0, 960]
9531                .map(|hours| (hours >= 0 || hours == -10).then_some(spring + hours * HOUR)),
9532        )
9533        .cast(&DataType::Datetime(
9534            TimeUnit::Nanoseconds,
9535            TimeZone::opt_try_new(Some("America/New_York")).unwrap(),
9536        ))
9537        .unwrap()
9538        .into();
9539        let mut frame = delays().collect().unwrap();
9540        for column in [
9541            Column::new(
9542                "key=part".into(),
9543                [
9544                    Some("a=b"),
9545                    Some(" x "),
9546                    Some(""),
9547                    None,
9548                    Some("é"),
9549                    Some("a=b"),
9550                    Some("1.0"),
9551                    Some(" x "),
9552                ],
9553            ),
9554            Column::new("int".into(), [1i64, 2, 3, 1, 2, 3, 1, 2]),
9555            Column::new(
9556                "float".into(),
9557                [0.1f64, 1e20, 2.5, 0.1, 1e20, 2.5, 0.1, 3.0],
9558            ),
9559            Column::new(
9560                "bool".into(),
9561                [true, false, true, false, true, false, true, false],
9562            ),
9563            datetimes("stamp", &[0, HOUR, 0, HOUR, 0, HOUR, 1, 0].map(Some)),
9564            zoned,
9565        ] {
9566            frame.with_column(column).unwrap();
9567        }
9568        let frame = frame.lazy();
9569        let grains = ["key=part", "int", "float", "bool", "stamp"]
9570            .map(|column| QualityGrain::Partition(column.to_string()))
9571            .into_iter()
9572            .chain(
9573                QUALITY_WINDOW_WIDTHS.map(|every| QualityGrain::TimeWindows {
9574                    column: "zoned".to_string(),
9575                    every: every.to_string(),
9576                }),
9577            );
9578        for grain in grains {
9579            for clock in IntervalClock::ALL {
9580                let plan = DataQualityPlan {
9581                    temporal_roles: roles(&[
9582                        (TemporalRole::Event, "sent"),
9583                        (TemporalRole::Received, "seen"),
9584                    ]),
9585                    latency_threshold_seconds: Some(3_600),
9586                    grain: grain.clone(),
9587                    interval_clock: clock,
9588                    ..DataQualityPlan::default()
9589                };
9590                for results in both_ways(&frame, 8, &plan) {
9591                    for latency in &results.temporal {
9592                        for fact in IntervalFact::ALL {
9593                            let Some((count, _)) = latency.count(fact, &plan) else {
9594                                continue;
9595                            };
9596                            let predicate = latency.evidence_predicate(fact, &plan, None).unwrap();
9597                            let rows = frame.clone().filter(predicate).collect().unwrap();
9598                            assert_eq!(
9599                                rows.height(),
9600                                count,
9601                                "{grain:?} {clock:?} {} {fact:?}",
9602                                latency.segment
9603                            );
9604                        }
9605                    }
9606                }
9607            }
9608        }
9609    }
9610}