Skip to main content

datui_lib/
quality_intent.rs

1//! Declared column intent for Data Quality: what a column must hold, said by the
2//! user, measured on the rows a run reads anyway.
3//!
4//! The profile can say a column is nearly unique; only a declaration can say it is a
5//! key, so that a repeat is a defect and not a category. Every rule is opt-in, and a
6//! violation is counted as a fact out of the rows or values checked, never folded
7//! into a score.
8
9use crate::data_quality::{
10    DataQualityPlan, ObservationKind, QualityObservation, QualityPrecision, TimeInterpretation,
11    TimeKind,
12};
13use crate::statistics::collect_lazy;
14use color_eyre::Result;
15use polars::prelude::*;
16
17/// The most values an allowed set holds.
18pub const MAX_ALLOWED_VALUES: usize = 100;
19
20/// Values outside the allowed set, or text that does not read as a number, kept as
21/// examples where the rows are in memory.
22pub const MAX_INTENT_EXAMPLES: usize = 3;
23
24const PREFIX: &str = "__datui_intent::";
25
26/// How text read as a number is read.
27#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
28pub enum NumberReading {
29    Whole,
30    Decimal,
31}
32
33impl NumberReading {
34    pub const ALL: [Self; 2] = [Self::Whole, Self::Decimal];
35
36    pub fn label(self) -> &'static str {
37        match self {
38            Self::Whole => "whole number",
39            Self::Decimal => "decimal",
40        }
41    }
42
43    fn dtype(self) -> DataType {
44        match self {
45            Self::Whole => DataType::Int64,
46            Self::Decimal => DataType::Float64,
47        }
48    }
49}
50
51/// What one column must hold. An empty intent declares nothing.
52#[derive(Debug, Clone, PartialEq, Eq, Hash, Default)]
53pub struct ColumnIntent {
54    pub column: String,
55    /// Every row has a value.
56    pub required: bool,
57    /// The values the column may hold, as typed; empty for any.
58    pub allowed: Vec<String>,
59    /// The lowest value allowed, as typed; a number, a date or a date and time.
60    pub min: Option<String>,
61    pub max: Option<String>,
62    /// Text read as a number: a value that does not read is counted, and the range
63    /// compares the number.
64    pub number: Option<NumberReading>,
65}
66
67impl ColumnIntent {
68    pub fn new(column: &str) -> Self {
69        Self {
70            column: column.to_string(),
71            ..Self::default()
72        }
73    }
74
75    pub fn is_empty(&self) -> bool {
76        !self.required
77            && self.allowed.is_empty()
78            && self.min.is_none()
79            && self.max.is_none()
80            && self.number.is_none()
81    }
82
83    /// The range in words: `0 to 100`, `at least 0`, `at most 2024-12-31`.
84    pub fn range_label(&self) -> Option<String> {
85        match (&self.min, &self.max) {
86            (Some(min), Some(max)) => Some(format!("{min} to {max}")),
87            (Some(min), None) => Some(format!("at least {min}")),
88            (None, Some(max)) => Some(format!("at most {max}")),
89            (None, None) => None,
90        }
91    }
92
93    /// The allowed set as typed, cut to `shown` values: `open, closed +3 more`.
94    pub fn allowed_label(&self, shown: usize) -> String {
95        let mut label = format_allowed(&self.allowed[..self.allowed.len().min(shown)]);
96        if self.allowed.len() > shown {
97            label.push_str(&format!(" +{} more", self.allowed.len() - shown));
98        }
99        label
100    }
101
102    /// Each rule in a few words: `required`, `one of 3`, `0 to 100`, `read as decimal`.
103    pub fn rules(&self) -> Vec<String> {
104        let mut rules = Vec::new();
105        if self.required {
106            rules.push("required".to_string());
107        }
108        if let Some(number) = self.number {
109            rules.push(format!("read as {}", number.label()));
110        }
111        if !self.allowed.is_empty() {
112            rules.push(format!(
113                "one of {}",
114                crate::numfmt::group_chrome(self.allowed.len())
115            ));
116        }
117        if let Some(range) = self.range_label() {
118            rules.push(range);
119        }
120        rules
121    }
122}
123
124/// Everything a study declares about its columns: the key, and each column's rules.
125#[derive(Debug, Clone, PartialEq, Eq, Hash, Default)]
126pub struct DeclaredIntent {
127    /// The columns whose values together name one row, in the order declared.
128    pub key: Vec<String>,
129    pub columns: Vec<ColumnIntent>,
130}
131
132impl DeclaredIntent {
133    pub fn is_empty(&self) -> bool {
134        self.key.is_empty() && self.columns.is_empty()
135    }
136
137    pub fn column(&self, name: &str) -> Option<&ColumnIntent> {
138        self.columns.iter().find(|intent| intent.column == name)
139    }
140
141    /// Declare `intent` for its column in place of what it had; an empty intent
142    /// removes the column's rules.
143    pub fn set(&mut self, intent: ColumnIntent) {
144        match self
145            .columns
146            .iter()
147            .position(|known| known.column == intent.column)
148        {
149            Some(index) if intent.is_empty() => {
150                self.columns.remove(index);
151            }
152            Some(index) => self.columns[index] = intent,
153            None if intent.is_empty() => {}
154            None => self.columns.push(intent),
155        }
156    }
157
158    /// Put `column` in the key, or take it out.
159    pub fn set_key(&mut self, column: &str, in_key: bool) {
160        let known = self.key.iter().any(|name| name == column);
161        if in_key && !known {
162            self.key.push(column.to_string());
163        } else if !in_key {
164            self.key.retain(|name| name != column);
165        }
166    }
167
168    /// The columns any rule names, the key first.
169    pub fn declared_columns(&self) -> Vec<&str> {
170        let mut names = self.key.iter().map(String::as_str).collect::<Vec<_>>();
171        for intent in &self.columns {
172            if !names.contains(&intent.column.as_str()) {
173                names.push(intent.column.as_str());
174            }
175        }
176        names
177    }
178
179    /// What Setup's row says: `key id, region ยท 2 columns with rules`.
180    pub fn summary(&self) -> String {
181        let mut parts = Vec::new();
182        if !self.key.is_empty() {
183            parts.push(format!("key {}", self.key.join(", ")));
184        }
185        match self.columns.len() {
186            0 => {}
187            1 => parts.push(format!(
188                "{}: {}",
189                self.columns[0].column,
190                self.columns[0].rules().join(", ")
191            )),
192            count => parts.push(format!("rules on {count} columns")),
193        }
194        parts.join(&format!(" {} ", crate::glyphs::get().middot))
195    }
196}
197
198/// What a column's values are, once read as declared: what a range can compare and
199/// a set can list.
200#[derive(Debug, Clone, Copy, PartialEq, Eq)]
201pub enum ValueKind {
202    Number,
203    Date,
204    Datetime,
205    Text,
206    Boolean,
207    Other,
208}
209
210impl ValueKind {
211    /// `column`'s values, read through `number` when it is text read as a number,
212    /// or `time` when it is text read as time.
213    pub fn of(
214        dtype: &DataType,
215        number: Option<NumberReading>,
216        time: Option<&TimeInterpretation>,
217    ) -> Self {
218        if let Some(time) = time {
219            return match time.kind {
220                TimeKind::Date => Self::Date,
221                TimeKind::Datetime => Self::Datetime,
222            };
223        }
224        if number.is_some() && is_text(dtype) {
225            return Self::Number;
226        }
227        match dtype {
228            DataType::Date => Self::Date,
229            DataType::Datetime(..) => Self::Datetime,
230            DataType::String | DataType::Categorical(..) => Self::Text,
231            DataType::Boolean => Self::Boolean,
232            dtype if dtype.is_primitive_numeric() || dtype.is_decimal() => Self::Number,
233            _ => Self::Other,
234        }
235    }
236
237    /// Whether a range applies: numbers and times have an order worth declaring.
238    pub fn ranges(self) -> bool {
239        matches!(self, Self::Number | Self::Date | Self::Datetime)
240    }
241
242    /// What a bound is typed as, for the form's hint.
243    pub fn bound_hint(self) -> &'static str {
244        match self {
245            Self::Number => "a number",
246            Self::Date => "a date, 2024-01-31",
247            Self::Datetime => "a date or 2024-01-31 08:00:00",
248            _ => "",
249        }
250    }
251}
252
253fn is_text(dtype: &DataType) -> bool {
254    matches!(dtype, DataType::String | DataType::Categorical(..))
255}
256
257/// Whether an allowed set applies: values compared as stored, text, whole numbers or
258/// true and false. A float or a time is a measurement, not a code.
259pub fn allows_set(dtype: &DataType) -> bool {
260    is_text(dtype) || dtype.is_integer() || matches!(dtype, DataType::Boolean)
261}
262
263/// Whether the column can be read as a number: text.
264pub fn reads_as_number(dtype: &DataType) -> bool {
265    is_text(dtype)
266}
267
268/// A range bound, as compared: a number, or a time in microseconds since the epoch,
269/// read as UTC when it has no zone, as every other time in the study is.
270#[derive(Debug, Clone, Copy, PartialEq)]
271enum Bound {
272    Number(f64),
273    Micros(i64),
274}
275
276impl Bound {
277    fn lit(self) -> Expr {
278        match self {
279            Self::Number(value) => lit(value),
280            Self::Micros(value) => lit(value),
281        }
282    }
283
284    fn value(self) -> f64 {
285        match self {
286            Self::Number(value) => value,
287            Self::Micros(value) => value as f64,
288        }
289    }
290}
291
292/// A date alone as a time's maximum takes in its whole day: `at most 2024-06-30`
293/// keeps 2024-06-30 08:00, as it reads.
294fn parse_bound(kind: ValueKind, text: &str, upper: bool) -> std::result::Result<Bound, String> {
295    let text = text.trim();
296    let midnight = |date: chrono::NaiveDate| {
297        date.and_hms_opt(0, 0, 0)
298            .map(|time| Bound::Micros(time.and_utc().timestamp_micros()))
299    };
300    let day_end = |date: chrono::NaiveDate| {
301        date.succ_opt()
302            .and_then(|next| next.and_hms_opt(0, 0, 0))
303            .map(|time| Bound::Micros(time.and_utc().timestamp_micros() - 1))
304    };
305    let parsed = match kind {
306        ValueKind::Number => text
307            .parse::<f64>()
308            .ok()
309            .filter(|value| value.is_finite())
310            .map(Bound::Number),
311        ValueKind::Date => chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d")
312            .ok()
313            .and_then(midnight),
314        ValueKind::Datetime => ["%Y-%m-%d %H:%M:%S", "%Y-%m-%dT%H:%M:%S", "%Y-%m-%d %H:%M"]
315            .into_iter()
316            .find_map(|format| chrono::NaiveDateTime::parse_from_str(text, format).ok())
317            .map(|time| Bound::Micros(time.and_utc().timestamp_micros()))
318            .or_else(|| {
319                chrono::NaiveDate::parse_from_str(text, "%Y-%m-%d")
320                    .ok()
321                    .and_then(|date| if upper { day_end(date) } else { midnight(date) })
322            }),
323        _ => None,
324    };
325    parsed.ok_or_else(|| format!("{text:?} is not {}", kind.bound_hint()))
326}
327
328/// The values an allowed set holds, from what was typed: separated by commas, outer
329/// spaces dropped, each once, in the order typed. A value in double quotes is taken
330/// as it stands, commas and spaces included, with `""` for a quote inside it.
331pub fn parse_allowed(dtype: &DataType, text: &str) -> std::result::Result<Vec<String>, String> {
332    let values = split_allowed(text)?;
333    check_allowed(dtype, &values)?;
334    Ok(values)
335}
336
337fn split_allowed(text: &str) -> std::result::Result<Vec<String>, String> {
338    let mut values: Vec<String> = Vec::new();
339    let mut push = |value: String| {
340        if !values.contains(&value) {
341            values.push(value);
342        }
343    };
344    let mut chars = text.chars().peekable();
345    loop {
346        while chars.next_if(|c| c.is_whitespace()).is_some() {}
347        if chars.next_if_eq(&'"').is_some() {
348            let mut value = String::new();
349            loop {
350                match chars.next() {
351                    Some('"') if chars.next_if_eq(&'"').is_some() => value.push('"'),
352                    Some('"') => break,
353                    Some(c) => value.push(c),
354                    None => return Err("A quoted value has no closing quote".to_string()),
355                }
356            }
357            while chars.next_if(|c| c.is_whitespace()).is_some() {}
358            match chars.next() {
359                None | Some(',') => push(value),
360                Some(_) => return Err(format!("Put a comma after {value:?}")),
361            }
362        } else {
363            let mut value = String::new();
364            for c in chars.by_ref() {
365                if c == ',' {
366                    break;
367                }
368                value.push(c);
369            }
370            let value = value.trim();
371            if !value.is_empty() {
372                push(value.to_string());
373            }
374        }
375        if chars.peek().is_none() {
376            break;
377        }
378    }
379    Ok(values)
380}
381
382/// Whether `values` can be compared with a column of `dtype`, and few enough.
383fn check_allowed(dtype: &DataType, values: &[String]) -> std::result::Result<(), String> {
384    // Read as the set compares it, so what is accepted here is what matches.
385    if dtype.is_integer() || matches!(dtype, DataType::Boolean) {
386        for value in values {
387            crate::typed_value::parse(value, dtype)?;
388        }
389    }
390    if values.len() > MAX_ALLOWED_VALUES {
391        return Err(format!("At most {MAX_ALLOWED_VALUES} allowed values"));
392    }
393    Ok(())
394}
395
396/// An allowed set as it is typed: values joined by commas, quoted where a comma, a
397/// quote or outer spaces would change how it reads back.
398pub fn format_allowed(values: &[String]) -> String {
399    values
400        .iter()
401        .map(|value| {
402            let plain = !value.is_empty()
403                && value.trim() == value
404                && !value.contains(',')
405                && !value.starts_with('"');
406            if plain {
407                value.clone()
408            } else {
409                format!("\"{}\"", value.replace('"', "\"\""))
410            }
411        })
412        .collect::<Vec<_>>()
413        .join(", ")
414}
415
416/// Why `intent` cannot be measured on a column of `dtype`, said on the form: a bound
417/// that is not a value of the column's kind, or a minimum above the maximum.
418pub fn check_intent(
419    intent: &ColumnIntent,
420    dtype: &DataType,
421    time: Option<&TimeInterpretation>,
422) -> std::result::Result<(), String> {
423    let kind = ValueKind::of(dtype, intent.number, time);
424    if !intent.allowed.is_empty() {
425        if !allows_set(dtype) {
426            return Err("Allowed values are for text, whole numbers or true/false".to_string());
427        }
428        check_allowed(dtype, &intent.allowed)?;
429    }
430    let bound = |text: &Option<String>, upper: bool| {
431        text.as_deref()
432            .map(|text| parse_bound(kind, text, upper))
433            .transpose()
434    };
435    if intent.min.is_some() || intent.max.is_some() {
436        if !kind.ranges() {
437            return Err("A range is for numbers, dates and times".to_string());
438        }
439        if let (Some(min), Some(max)) = (bound(&intent.min, false)?, bound(&intent.max, true)?)
440            && min.value() > max.value()
441        {
442            return Err("Minimum is above maximum".to_string());
443        }
444    }
445    if intent.number.is_some() && !reads_as_number(dtype) {
446        return Err("Only text is read as a number".to_string());
447    }
448    Ok(())
449}
450
451/// A date or millisecond datetime held to what microseconds since the epoch can
452/// count, so converting it to them cannot overflow, which made a date past the
453/// calendar null and never compared. One held at a limit (about 292,000 years from
454/// 1970) is still further out than any bound, which the calendar keeps within
455/// 262,143 years. Any other value as it is; a batch with none so far out costs a
456/// min and a max.
457fn within_micros(value: Expr) -> Expr {
458    value.map(
459        |c| {
460            let limit = match c.dtype() {
461                DataType::Date => i64::MAX / 86_400_000_000,
462                DataType::Datetime(TimeUnit::Milliseconds, _) => i64::MAX / 1_000,
463                _ => return Ok(c),
464            };
465            let series = c.as_materialized_series();
466            let stored = series.to_physical_repr().cast(&DataType::Int64)?;
467            let stored = stored.i64()?;
468            let fits = |v: i64| (-limit..=limit).contains(&v);
469            if [stored.min(), stored.max()].into_iter().flatten().all(fits) {
470                return Ok(c);
471            }
472            let held = stored.apply_values(|v| v.clamp(-limit, limit));
473            Ok(held
474                .into_series()
475                .cast(c.dtype())?
476                .with_name(series.name().clone())
477                .into_column())
478        },
479        |_, field| Ok(field.clone()),
480    )
481}
482
483/// One column's declared rules, as a run measures them: the declaration, the type it
484/// met, and the expressions that count it.
485struct Measured<'a> {
486    intent: &'a ColumnIntent,
487    dtype: DataType,
488    time: Option<TimeInterpretation>,
489}
490
491impl Measured<'_> {
492    fn stored(&self) -> Expr {
493        col(self.intent.column.as_str())
494    }
495
496    fn kind(&self) -> ValueKind {
497        ValueKind::of(&self.dtype, self.intent.number, self.time.as_ref())
498    }
499
500    /// The value as declared: text read as a number or a time, otherwise as stored.
501    /// Read as `str.cast` and the time format read it for the profile, so the counts
502    /// agree with "Numbers as text" and "Unparsed times".
503    fn value(&self) -> Expr {
504        if let Some(time) = &self.time {
505            return time.expr();
506        }
507        match self.intent.number {
508            Some(number) if is_text(&self.dtype) => {
509                self.stored().cast(DataType::String).cast(number.dtype())
510            }
511            _ => self.stored(),
512        }
513    }
514
515    /// What a range compares: a number as a float, a time as microseconds since the
516    /// epoch (an instant's own; a time with no zone read as UTC).
517    fn compared(&self) -> Option<Expr> {
518        let value = self.value();
519        Some(match self.kind() {
520            ValueKind::Number => value.cast(DataType::Float64),
521            ValueKind::Date => within_micros(value)
522                .cast(DataType::Datetime(TimeUnit::Microseconds, None))
523                .dt()
524                .timestamp(TimeUnit::Microseconds),
525            ValueKind::Datetime => within_micros(value).dt().timestamp(TimeUnit::Microseconds),
526            _ => return None,
527        })
528    }
529
530    fn bounds(&self) -> (Option<Bound>, Option<Bound>) {
531        let kind = self.kind();
532        let bound = |text: &Option<String>, upper: bool| {
533            text.as_deref()
534                .and_then(|text| parse_bound(kind, text, upper).ok())
535        };
536        (
537            bound(&self.intent.min, false),
538            bound(&self.intent.max, true),
539        )
540    }
541
542    fn below(&self) -> Option<Expr> {
543        Some(self.compared()?.lt(self.bounds().0?.lit()))
544    }
545
546    fn above(&self) -> Option<Expr> {
547        Some(self.compared()?.gt(self.bounds().1?.lit()))
548    }
549
550    fn out_of_range(&self) -> Option<Expr> {
551        match (self.below(), self.above()) {
552            (Some(below), Some(above)) => Some(below.or(above)),
553            (below, above) => below.or(above),
554        }
555    }
556
557    /// Stored values in the allowed set, each compared at the column's own type: text
558    /// exactly as stored, a whole number as the width it is stored at, so a `u64`
559    /// past `i64::MAX` is still itself.
560    fn in_set(&self) -> Option<Expr> {
561        if self.intent.allowed.is_empty() || !allows_set(&self.dtype) {
562            return None;
563        }
564        let stored = self.stored();
565        self.intent
566            .allowed
567            .iter()
568            .filter_map(|value| {
569                let value = crate::typed_value::parse(value, &self.dtype).ok()?;
570                Some(stored.clone().eq(lit(value)))
571            })
572            .reduce(Expr::or)
573    }
574
575    fn outside(&self) -> Option<Expr> {
576        Some(self.stored().is_not_null().and(self.in_set()?.not()))
577    }
578
579    /// Text that does not read as the declared number.
580    fn unparsed(&self) -> Option<Expr> {
581        if self.intent.number.is_none() || !is_text(&self.dtype) || self.time.is_some() {
582            return None;
583        }
584        Some(self.stored().is_not_null().and(self.value().is_null()))
585    }
586}
587
588fn measured<'a>(plan: &'a DataQualityPlan, schema: &Schema) -> Vec<Measured<'a>> {
589    plan.intent
590        .columns
591        .iter()
592        .filter_map(|intent| {
593            Some(Measured {
594                intent,
595                dtype: schema.get(&intent.column)?.clone(),
596                time: plan.time_format(&intent.column).cloned(),
597            })
598        })
599        .collect()
600}
601
602fn name(index: usize, what: &str) -> String {
603    format!("{PREFIX}{index}::{what}")
604}
605
606/// The key's columns, when every one of them is in `schema`.
607fn key_columns(plan: &DataQualityPlan, schema: &Schema) -> Option<Vec<String>> {
608    let key = &plan.intent.key;
609    (!key.is_empty() && key.iter().all(|column| schema.get(column).is_some())).then(|| key.clone())
610}
611
612fn any_null(key: &[String]) -> Expr {
613    key.iter()
614        .map(|column| col(column.as_str()).is_null())
615        .reduce(Expr::or)
616        .unwrap_or_else(|| lit(false))
617}
618
619/// Every rule's counts, as aggregations over the scope: added to the pass that
620/// profiles the columns, so they cost that pass nothing but the sums.
621pub(crate) fn intent_exprs(plan: &DataQualityPlan, schema: &Schema) -> Vec<Expr> {
622    let mut exprs = Vec::new();
623    for (index, rules) in measured(plan, schema).iter().enumerate() {
624        exprs.push(
625            rules
626                .stored()
627                .is_not_null()
628                .sum()
629                .alias(name(index, "values")),
630        );
631        if rules.intent.required {
632            exprs.push(rules.stored().is_null().sum().alias(name(index, "missing")));
633        }
634        if let Some(unparsed) = rules.unparsed() {
635            exprs.push(unparsed.sum().alias(name(index, "unparsed")));
636        }
637        if let Some(outside) = rules.outside() {
638            exprs.push(outside.sum().alias(name(index, "outside")));
639        }
640        if let Some(compared) = rules.compared().filter(|_| rules.out_of_range().is_some()) {
641            exprs.push(compared.is_not_null().sum().alias(name(index, "compared")));
642            let value = rules.value();
643            if let Some(below) = rules.below() {
644                exprs.push(below.clone().sum().alias(name(index, "below")));
645                exprs.push(
646                    value
647                        .clone()
648                        .filter(below)
649                        .min()
650                        .alias(name(index, "lowest")),
651                );
652            }
653            if let Some(above) = rules.above() {
654                exprs.push(above.clone().sum().alias(name(index, "above")));
655                exprs.push(value.filter(above).max().alias(name(index, "highest")));
656            }
657        }
658    }
659    if let Some(key) = key_columns(plan, schema) {
660        exprs.push(any_null(&key).sum().alias(format!("{PREFIX}key::missing")));
661    }
662    exprs
663}
664
665/// How often the declared key's value repeats, over the rows with every part of it:
666/// groups of rows sharing one value, the rows beyond one per value, and every row in
667/// such a group. One grouping of the key's columns alone, as duplicate rows are
668/// counted: over rows in memory it reads nothing, and over the scope it is a pass of
669/// its own, which Setup names before Run.
670pub(crate) fn key_repeats(
671    lf: &LazyFrame,
672    plan: &DataQualityPlan,
673    schema: &Schema,
674    polars_streaming: bool,
675) -> Result<Option<(usize, usize, usize)>> {
676    let Some(key) = key_columns(plan, schema) else {
677        return Ok(None);
678    };
679    const COUNT: &str = "__datui_intent_key_rows";
680    let columns = key
681        .iter()
682        .map(|name| col(name.as_str()))
683        .collect::<Vec<_>>();
684    let query = lf
685        .clone()
686        .select(columns.clone())
687        .filter(any_null(&key).not())
688        .group_by(columns)
689        .agg([len().alias(COUNT)])
690        .filter(col(COUNT).gt(lit(1u32)))
691        .select([
692            len().alias("groups"),
693            (col(COUNT) - lit(1u32)).sum().alias("extra"),
694            col(COUNT).sum().alias("involved"),
695        ]);
696    let summary = collect_lazy(query, polars_streaming)?;
697    Ok(Some((
698        count_at(&summary, "groups").unwrap_or(0),
699        count_at(&summary, "extra").unwrap_or(0),
700        count_at(&summary, "involved").unwrap_or(0),
701    )))
702}
703
704fn count_at(df: &DataFrame, name: &str) -> Option<usize> {
705    match df.column(name).ok()?.get(0).ok()? {
706        AnyValue::UInt32(value) => Some(value as usize),
707        AnyValue::UInt64(value) => Some(value as usize),
708        AnyValue::Int32(value) => usize::try_from(value).ok(),
709        AnyValue::Int64(value) => usize::try_from(value).ok(),
710        _ => None,
711    }
712}
713
714fn text_at(df: &DataFrame, name: &str) -> Option<String> {
715    let value = df.column(name).ok()?.get(0).ok()?;
716    (!value.is_null()).then(|| crate::exact::str_value(&value).into_owned())
717}
718
719/// The commonest values `rows` holds in `value`, with how many rows hold each: over
720/// rows in memory only.
721fn commonest(lf: &LazyFrame, rows: Expr, value: Expr) -> Result<Vec<(String, usize)>> {
722    const VALUE: &str = "__datui_intent_value";
723    const COUNT: &str = "__datui_intent_count";
724    let top = lf
725        .clone()
726        .filter(rows)
727        .select([value.cast(DataType::String).alias(VALUE)])
728        .group_by([col(VALUE)])
729        .agg([len().alias(COUNT)])
730        .sort_by_exprs(
731            [col(COUNT), col(VALUE)],
732            SortMultipleOptions::default().with_order_descending_multi([true, false]),
733        )
734        .limit(MAX_INTENT_EXAMPLES as IdxSize)
735        .collect()?;
736    let (values, counts) = (top.column(VALUE)?, top.column(COUNT)?);
737    Ok((0..top.height())
738        .filter_map(|row| {
739            let value = values.get(row).ok()?;
740            let count = match counts.get(row).ok()? {
741                AnyValue::UInt32(count) => count as usize,
742                AnyValue::UInt64(count) => count as usize,
743                _ => return None,
744            };
745            Some((crate::exact::str_value(&value).into_owned(), count))
746        })
747        .collect())
748}
749
750/// Whether the declared key was checked against every row in scope, or a sample's.
751#[derive(Debug, Clone, PartialEq, Eq)]
752pub struct KeyCheck {
753    pub columns: Vec<String>,
754    /// Rows with no value in some part of the key.
755    pub missing: usize,
756    /// Key values held by more than one row.
757    pub groups: usize,
758    /// Rows beyond one per key value.
759    pub extra_rows: usize,
760    /// Rows that share their key value with another row.
761    pub rows_involved: usize,
762}
763
764/// What a column's declared rules found.
765#[derive(Debug, Clone)]
766pub struct ColumnCheck {
767    pub intent: ColumnIntent,
768    /// The column's type in the scope measured.
769    pub dtype: DataType,
770    /// The format text was read as time with, under Text as time.
771    pub time: Option<TimeInterpretation>,
772    /// Rows with a value, as stored.
773    pub values: usize,
774    /// Required: rows with no value.
775    pub missing: Option<usize>,
776    /// Read as a number: values that do not read as one.
777    pub unparsed: Option<usize>,
778    /// Allowed: values outside the set.
779    pub outside: Option<usize>,
780    /// Range: values the range compared, the ones read.
781    pub compared: Option<usize>,
782    pub below: Option<usize>,
783    pub above: Option<usize>,
784    /// The lowest value below the minimum and the highest above the maximum.
785    pub lowest: Option<String>,
786    pub highest: Option<String>,
787    /// The commonest values outside the set, with their rows; from rows in memory.
788    pub outside_examples: Vec<(String, usize)>,
789    /// The commonest text that does not read as the number; from rows in memory.
790    pub unparsed_examples: Vec<(String, usize)>,
791}
792
793impl ColumnCheck {
794    fn rules(&self) -> Measured<'_> {
795        Measured {
796            intent: &self.intent,
797            dtype: self.dtype.clone(),
798            time: self.time.clone(),
799        }
800    }
801
802    /// Values outside the range, both ways.
803    pub fn out_of_range(&self) -> Option<usize> {
804        match (self.below, self.above) {
805            (None, None) => None,
806            (below, above) => Some(below.unwrap_or(0) + above.unwrap_or(0)),
807        }
808    }
809}
810
811/// What a run found of the declared intent.
812#[derive(Debug, Clone)]
813pub struct IntentResults {
814    /// False when the run read no values, so nothing declared was checked.
815    pub measured: bool,
816    pub precision: QualityPrecision,
817    /// Rows checked: the scope's on a full read, the sample's otherwise.
818    pub evaluated_rows: usize,
819    pub key: Option<KeyCheck>,
820    pub columns: Vec<ColumnCheck>,
821    /// Every column a rule names, the key's first.
822    pub declared: Vec<String>,
823    /// Declared columns the scope measured does not have; their rules did not run.
824    pub absent: Vec<String>,
825}
826
827impl IntentResults {
828    /// A run that read no values: what was declared, and that none of it ran.
829    pub(crate) fn unmeasured(plan: &DataQualityPlan, schema: &Schema) -> Option<Self> {
830        if plan.intent.is_empty() {
831            return None;
832        }
833        Some(Self {
834            measured: false,
835            precision: QualityPrecision::Metadata,
836            evaluated_rows: 0,
837            key: None,
838            columns: Vec::new(),
839            declared: declared(plan),
840            absent: absent(plan, schema),
841        })
842    }
843
844    /// The rules' counts from `counts`, the row that [`intent_exprs`] aggregated, and
845    /// the key's repeats from [`key_repeats`]. `rows` are the rows in memory, when the
846    /// run measured those: the examples come from them and from nothing else.
847    pub(crate) fn from_counts(
848        plan: &DataQualityPlan,
849        schema: &Schema,
850        counts: &DataFrame,
851        repeats: Option<(usize, usize, usize)>,
852        evaluated_rows: usize,
853        precision: QualityPrecision,
854        rows: Option<&LazyFrame>,
855    ) -> Result<Option<Self>> {
856        if plan.intent.is_empty() {
857            return Ok(None);
858        }
859        let mut columns = Vec::new();
860        for (index, rules) in measured(plan, schema).iter().enumerate() {
861            let count = |what: &str| count_at(counts, &name(index, what));
862            let examples = |predicate: Option<Expr>, value: Expr| match (rows, predicate) {
863                (Some(rows), Some(predicate)) => commonest(rows, predicate, value),
864                _ => Ok(Vec::new()),
865            };
866            let unparsed = count("unparsed");
867            let outside = count("outside");
868            columns.push(ColumnCheck {
869                intent: rules.intent.clone(),
870                dtype: rules.dtype.clone(),
871                time: rules.time.clone(),
872                values: count("values").unwrap_or(0),
873                missing: count("missing"),
874                unparsed,
875                outside,
876                compared: count("compared"),
877                below: count("below"),
878                above: count("above"),
879                lowest: text_at(counts, &name(index, "lowest")),
880                highest: text_at(counts, &name(index, "highest")),
881                outside_examples: if outside.unwrap_or(0) > 0 {
882                    examples(rules.outside(), rules.stored())?
883                } else {
884                    Vec::new()
885                },
886                unparsed_examples: if unparsed.unwrap_or(0) > 0 {
887                    examples(rules.unparsed(), rules.stored())?
888                } else {
889                    Vec::new()
890                },
891            });
892        }
893        let key = key_columns(plan, schema).map(|key| {
894            let (groups, extra_rows, rows_involved) = repeats.unwrap_or((0, 0, 0));
895            KeyCheck {
896                columns: key,
897                missing: count_at(counts, &format!("{PREFIX}key::missing")).unwrap_or(0),
898                groups,
899                extra_rows,
900                rows_involved,
901            }
902        });
903        Ok(Some(Self {
904            measured: true,
905            precision,
906            evaluated_rows,
907            key,
908            columns,
909            declared: declared(plan),
910            absent: absent(plan, schema),
911        }))
912    }
913
914    /// A rule's violations as observations, one per column; the key's once per key
915    /// column, so its finding names each.
916    pub(crate) fn observations(&self) -> Vec<QualityObservation> {
917        let mut observations = Vec::new();
918        let count = crate::numfmt::group_chrome;
919        let push = |observations: &mut Vec<QualityObservation>,
920                    kind: ObservationKind,
921                    column: &str,
922                    affected_rows: usize,
923                    evaluated_rows: usize,
924                    fact: String| {
925            observations.push(QualityObservation {
926                kind,
927                column: column.to_string(),
928                affected_rows,
929                evaluated_rows,
930                fact,
931                normalized_category: None,
932                files: Vec::new(),
933                time_format: None,
934                full_scale: None,
935            });
936        };
937        if !self.measured {
938            return observations;
939        }
940        if let Some(key) = &self.key {
941            for column in &key.columns {
942                if key.rows_involved > 0 {
943                    push(
944                        &mut observations,
945                        ObservationKind::KeyRepeated,
946                        column,
947                        key.rows_involved,
948                        self.evaluated_rows,
949                        format!(
950                            "{} key values repeat; {} extra rows",
951                            count(key.groups),
952                            count(key.extra_rows)
953                        ),
954                    );
955                }
956                if key.missing > 0 {
957                    push(
958                        &mut observations,
959                        ObservationKind::KeyMissing,
960                        column,
961                        key.missing,
962                        self.evaluated_rows,
963                        format!("{} rows have no complete key", count(key.missing)),
964                    );
965                }
966            }
967        }
968        for check in &self.columns {
969            let column = check.intent.column.as_str();
970            if let Some(missing) = check.missing.filter(|missing| *missing > 0) {
971                push(
972                    &mut observations,
973                    ObservationKind::RequiredMissing,
974                    column,
975                    missing,
976                    self.evaluated_rows,
977                    format!("{} rows with no value", count(missing)),
978                );
979            }
980            if let Some(unparsed) = check.unparsed.filter(|unparsed| *unparsed > 0) {
981                push(
982                    &mut observations,
983                    ObservationKind::UnparsedNumber,
984                    column,
985                    unparsed,
986                    check.values,
987                    format!(
988                        "{} of {} values do not read as a {}",
989                        count(unparsed),
990                        count(check.values),
991                        check.intent.number.map_or("number", NumberReading::label)
992                    ),
993                );
994            }
995            if let Some(outside) = check.outside.filter(|outside| *outside > 0) {
996                push(
997                    &mut observations,
998                    ObservationKind::NotAllowed,
999                    column,
1000                    outside,
1001                    check.values,
1002                    format!(
1003                        "{} of {} values are not one of {}",
1004                        count(outside),
1005                        count(check.values),
1006                        check.intent.allowed_label(5)
1007                    ),
1008                );
1009            }
1010            if let Some(out) = check.out_of_range().filter(|out| *out > 0) {
1011                push(
1012                    &mut observations,
1013                    ObservationKind::OutOfRange,
1014                    column,
1015                    out,
1016                    check.compared.unwrap_or(0),
1017                    format!(
1018                        "{} of {} values outside {}",
1019                        count(out),
1020                        count(check.compared.unwrap_or(0)),
1021                        check.intent.range_label().unwrap_or_default()
1022                    ),
1023                );
1024            }
1025        }
1026        observations
1027    }
1028
1029    /// The rows behind a violation, as a predicate over the rows the run read.
1030    pub fn evidence(&self, kind: ObservationKind, column: &str) -> Option<Expr> {
1031        match kind {
1032            ObservationKind::KeyRepeated => {
1033                let key = self.key.as_ref()?;
1034                let columns = key
1035                    .columns
1036                    .iter()
1037                    .map(|name| col(name.as_str()))
1038                    .collect::<Vec<_>>();
1039                let shared = len().over(columns).ok()?.gt(lit(1u32));
1040                Some(any_null(&key.columns).not().and(shared))
1041            }
1042            ObservationKind::KeyMissing => Some(any_null(&self.key.as_ref()?.columns)),
1043            _ => {
1044                let check = self
1045                    .columns
1046                    .iter()
1047                    .find(|check| check.intent.column == column)?;
1048                let rules = check.rules();
1049                match kind {
1050                    ObservationKind::RequiredMissing => Some(rules.stored().is_null()),
1051                    ObservationKind::UnparsedNumber => rules.unparsed(),
1052                    ObservationKind::NotAllowed => rules.outside(),
1053                    ObservationKind::OutOfRange => rules.out_of_range(),
1054                    _ => None,
1055                }
1056            }
1057        }
1058    }
1059
1060    /// What the column's declaration was measured with, for a finding's detail.
1061    pub fn column(&self, column: &str) -> Option<&ColumnCheck> {
1062        self.columns
1063            .iter()
1064            .find(|check| check.intent.column == column)
1065    }
1066}
1067
1068fn declared(plan: &DataQualityPlan) -> Vec<String> {
1069    plan.intent
1070        .declared_columns()
1071        .into_iter()
1072        .map(str::to_string)
1073        .collect()
1074}
1075
1076fn absent(plan: &DataQualityPlan, schema: &Schema) -> Vec<String> {
1077    plan.intent
1078        .declared_columns()
1079        .into_iter()
1080        .filter(|column| schema.get(column).is_none())
1081        .map(str::to_string)
1082        .collect()
1083}
1084
1085/// A column declared to be the key answers what "Nearly unique" can only suggest, so
1086/// the suggestion goes: the key's own count says whether it repeats.
1087pub(crate) fn supersede(observations: &mut Vec<QualityObservation>, plan: &DataQualityPlan) {
1088    if let [key] = plan.intent.key.as_slice() {
1089        observations.retain(|observation| {
1090            !(observation.kind == ObservationKind::KeyLike && &observation.column == key)
1091        });
1092    }
1093}
1094
1095#[cfg(test)]
1096mod tests {
1097    use super::*;
1098    use crate::data_quality::{DataQualityResults, QualityCompute, compute_data_quality};
1099    use crate::quality_report::{Outcome, build_report, checks, coverage, describe};
1100
1101    fn fixture() -> DataFrame {
1102        df!(
1103            "id" => &[Some(1i64), Some(2), Some(2), Some(3), Some(4), Some(5), None],
1104            "status" => &[Some("open"), Some("closed"), Some("open"), Some("void"), Some("Open"), None, Some("closed")],
1105            "amount" => &[5.0f64, 50.0, -1.0, 20.0, 200.0, 10.0, 0.0],
1106            "code" => &["1", "2", "x", "4", "5", "6", "7"],
1107        )
1108        .unwrap()
1109    }
1110
1111    fn declared() -> DeclaredIntent {
1112        DeclaredIntent {
1113            key: vec!["id".to_string()],
1114            columns: vec![
1115                ColumnIntent {
1116                    required: true,
1117                    allowed: vec!["open".to_string(), "closed".to_string()],
1118                    ..ColumnIntent::new("status")
1119                },
1120                ColumnIntent {
1121                    min: Some("0".to_string()),
1122                    max: Some("100".to_string()),
1123                    ..ColumnIntent::new("amount")
1124                },
1125                ColumnIntent {
1126                    number: Some(NumberReading::Whole),
1127                    min: Some("2".to_string()),
1128                    ..ColumnIntent::new("code")
1129                },
1130            ],
1131        }
1132    }
1133
1134    fn plan(compute: QualityCompute) -> DataQualityPlan {
1135        DataQualityPlan {
1136            compute,
1137            intent: declared(),
1138            ..DataQualityPlan::default()
1139        }
1140    }
1141
1142    fn run(df: &DataFrame, plan: &DataQualityPlan) -> DataQualityResults {
1143        compute_data_quality(&df.clone().lazy(), Some(df.height()), plan, None, false).unwrap()
1144    }
1145
1146    fn affected(results: &DataQualityResults, kind: ObservationKind, column: &str) -> usize {
1147        results
1148            .observations
1149            .iter()
1150            .find(|observation| observation.kind == kind && observation.column == column)
1151            .map_or(0, |observation| observation.affected_rows)
1152    }
1153
1154    /// Rows the evidence predicate picks out of `df`: the rows the finding counted.
1155    fn matching(
1156        df: &DataFrame,
1157        results: &DataQualityResults,
1158        kind: ObservationKind,
1159        column: &str,
1160    ) -> usize {
1161        let predicate = results
1162            .intent
1163            .as_ref()
1164            .and_then(|intent| intent.evidence(kind, column))
1165            .expect("a predicate");
1166        df.clone()
1167            .lazy()
1168            .filter(predicate)
1169            .collect()
1170            .unwrap()
1171            .height()
1172    }
1173
1174    /// Every rule on a full read, counted exactly, its rows found by its predicate.
1175    #[test]
1176    fn a_full_read_counts_every_declared_rule() {
1177        let df = fixture();
1178        let results = run(&df, &plan(QualityCompute::Full));
1179        let intent = results.intent.as_ref().expect("intent measured");
1180        assert!(intent.measured);
1181        assert_eq!(intent.precision, QualityPrecision::Exact);
1182        let key = intent.key.as_ref().unwrap();
1183        assert_eq!(
1184            (key.missing, key.groups, key.extra_rows, key.rows_involved),
1185            (1, 1, 1, 2)
1186        );
1187        let status = intent.column("status").unwrap();
1188        assert_eq!(status.missing, Some(1));
1189        assert_eq!(status.values, 6);
1190        assert_eq!(status.outside, Some(2));
1191        let amount = intent.column("amount").unwrap();
1192        assert_eq!((amount.below, amount.above), (Some(1), Some(1)));
1193        assert_eq!(amount.lowest.as_deref(), Some("-1.0"));
1194        assert_eq!(amount.highest.as_deref(), Some("200.0"));
1195        let code = intent.column("code").unwrap();
1196        assert_eq!(code.unparsed, Some(1));
1197        // The range compares what reads as a number: six of the seven.
1198        assert_eq!((code.compared, code.below), (Some(6), Some(1)));
1199        // A full read keeps no rows, so it lists no examples.
1200        assert!(status.outside_examples.is_empty());
1201
1202        for (kind, column, count) in [
1203            (ObservationKind::KeyRepeated, "id", 2),
1204            (ObservationKind::KeyMissing, "id", 1),
1205            (ObservationKind::RequiredMissing, "status", 1),
1206            (ObservationKind::NotAllowed, "status", 2),
1207            (ObservationKind::OutOfRange, "amount", 2),
1208            (ObservationKind::UnparsedNumber, "code", 1),
1209            (ObservationKind::OutOfRange, "code", 1),
1210        ] {
1211            assert_eq!(affected(&results, kind, column), count, "{kind:?} {column}");
1212            assert_eq!(
1213                matching(&df, &results, kind, column),
1214                count,
1215                "{kind:?} {column}"
1216            );
1217        }
1218
1219        // Problems, stated as facts, and the check names its reach.
1220        let report = build_report(&results);
1221        for title in [
1222            "Repeated key",
1223            "Incomplete key",
1224            "Required, missing",
1225            "Not allowed",
1226            "Out of range",
1227            "Unparsed numbers",
1228        ] {
1229            let finding = report
1230                .findings
1231                .iter()
1232                .find(|finding| finding.title == title)
1233                .unwrap_or_else(|| panic!("{title}"));
1234            assert_eq!(finding.severity, crate::quality_report::Severity::Problem);
1235        }
1236        let all = checks(&results, &report);
1237        assert_eq!(all[0].name, crate::quality_report::INTENT_CHECK);
1238        assert_eq!(all[0].basis, QualityPrecision::Exact);
1239        assert!(matches!(all[0].outcome, Outcome::Found { .. }));
1240        let repeated = report
1241            .findings
1242            .iter()
1243            .find(|finding| finding.title == "Repeated key")
1244            .unwrap();
1245        let (headline, _) = describe(repeated, &results);
1246        assert_eq!(
1247            headline,
1248            "2 of 7 rows (28.6%) share their key with another row"
1249        );
1250        // A key with no repeat on a full read leaves no limit behind.
1251        assert!(
1252            !coverage(&results, &all, &plan(QualityCompute::Full))
1253                .limits()
1254                .iter()
1255                .any(|limit| limit.contains("key"))
1256        );
1257    }
1258
1259    /// A sample counts what its rows show, says so, and keeps examples from them.
1260    #[test]
1261    fn a_sample_counts_its_rows_and_says_what_it_cannot() {
1262        let ids = (0..2_000i64).map(|row| row % 1_000).collect::<Vec<_>>();
1263        let status = (0..2_000)
1264            .map(|row| if row % 10 == 0 { "lost" } else { "open" })
1265            .collect::<Vec<_>>();
1266        let df = df!("id" => ids, "status" => status).unwrap();
1267        let plan = DataQualityPlan {
1268            dataset_rows: 400,
1269            intent: DeclaredIntent {
1270                key: vec!["id".to_string()],
1271                columns: vec![ColumnIntent {
1272                    allowed: vec!["open".to_string()],
1273                    ..ColumnIntent::new("status")
1274                }],
1275            },
1276            ..DataQualityPlan::default()
1277        };
1278        let results = run(&df, &plan);
1279        assert_eq!(results.precision, QualityPrecision::Sampled);
1280        let intent = results.intent.as_ref().unwrap();
1281        assert_eq!(intent.precision, QualityPrecision::Sampled);
1282        assert_eq!(intent.evaluated_rows, 400);
1283        let status = intent.column("status").unwrap();
1284        let outside = status.outside.unwrap();
1285        assert!(outside > 0 && outside < 400);
1286        // The examples come from the rows in memory, with their counts.
1287        assert_eq!(status.outside_examples, vec![("lost".to_string(), outside)]);
1288        // Every id repeats once in the data; the sample holds some of the pairs, and
1289        // a repeat among its distinct rows is one in the data.
1290        let key = intent.key.as_ref().unwrap();
1291        assert!(key.rows_involved <= 400);
1292        assert_eq!(key.rows_involved, key.groups * 2);
1293
1294        let report = build_report(&results);
1295        let all = checks(&results, &report);
1296        assert_eq!(all[0].basis, QualityPrecision::Sampled);
1297        let limits = coverage(&results, &all, &plan).limits();
1298        assert!(
1299            limits.contains(&"key repeats among 400 sampled rows only".to_string()),
1300            "{limits:?}"
1301        );
1302        let not_allowed = report
1303            .findings
1304            .iter()
1305            .find(|finding| finding.title == "Not allowed")
1306            .unwrap();
1307        let (_, evidence) = describe(not_allowed, &results);
1308        assert!(
1309            evidence
1310                .iter()
1311                .any(|line| line.starts_with("Found: \"lost\"")),
1312            "{evidence:?}"
1313        );
1314        if key.groups > 0 {
1315            let repeated = report
1316                .findings
1317                .iter()
1318                .find(|finding| finding.title == "Repeated key")
1319                .unwrap();
1320            let (headline, evidence) = describe(repeated, &results);
1321            assert!(headline.contains("of 400 sampled rows"), "{headline}");
1322            assert!(evidence.iter().any(|line| line.contains("not checked")));
1323        }
1324    }
1325
1326    /// No repeat in a sample is not a unique key: the check passes on the sampled
1327    /// rows and the coverage says how far that reaches.
1328    #[test]
1329    fn a_sample_without_repeats_claims_only_its_rows() {
1330        let df = df!("id" => (0..5_000i64).collect::<Vec<_>>()).unwrap();
1331        let plan = DataQualityPlan {
1332            dataset_rows: 500,
1333            intent: DeclaredIntent {
1334                key: vec!["id".to_string()],
1335                columns: Vec::new(),
1336            },
1337            ..DataQualityPlan::default()
1338        };
1339        let results = run(&df, &plan);
1340        let report = build_report(&results);
1341        assert!(
1342            !report
1343                .findings
1344                .iter()
1345                .any(|finding| finding.title == "Repeated key")
1346        );
1347        let all = checks(&results, &report);
1348        assert_eq!(all[0].outcome, Outcome::Passed);
1349        assert_eq!(all[0].basis, QualityPrecision::Sampled);
1350        assert!(
1351            coverage(&results, &all, &plan)
1352                .limits()
1353                .contains(&"key repeats among 500 sampled rows only".to_string())
1354        );
1355    }
1356
1357    /// Declaring a column the key answers what "Nearly unique" could only suggest.
1358    #[test]
1359    fn a_declared_key_replaces_the_nearly_unique_note() {
1360        let ids = (0..100i64)
1361            .map(|row| if row == 99 { 0 } else { row })
1362            .collect::<Vec<_>>();
1363        let df = df!("id" => ids).unwrap();
1364        let undeclared = run(
1365            &df,
1366            &DataQualityPlan {
1367                compute: QualityCompute::Full,
1368                ..DataQualityPlan::default()
1369            },
1370        );
1371        assert!(
1372            undeclared
1373                .observations
1374                .iter()
1375                .any(|o| o.kind == ObservationKind::KeyLike)
1376        );
1377        let declared = run(
1378            &df,
1379            &DataQualityPlan {
1380                compute: QualityCompute::Full,
1381                intent: DeclaredIntent {
1382                    key: vec!["id".to_string()],
1383                    columns: Vec::new(),
1384                },
1385                ..DataQualityPlan::default()
1386            },
1387        );
1388        assert!(
1389            !declared
1390                .observations
1391                .iter()
1392                .any(|o| o.kind == ObservationKind::KeyLike)
1393        );
1394        assert_eq!(affected(&declared, ObservationKind::KeyRepeated, "id"), 2);
1395    }
1396
1397    /// A composite key repeats only where every part does.
1398    #[test]
1399    fn a_composite_key_repeats_where_all_its_parts_do() {
1400        let df = df!(
1401            "region" => &[Some("east"), Some("east"), Some("west"), Some("west"), Some("east"), Some("east"), None],
1402            "id" => &[Some(1i64), Some(2), Some(1), Some(1), None, None, Some(1)],
1403        )
1404        .unwrap();
1405        let results = run(
1406            &df,
1407            &DataQualityPlan {
1408                compute: QualityCompute::Full,
1409                intent: DeclaredIntent {
1410                    key: vec!["region".to_string(), "id".to_string()],
1411                    columns: Vec::new(),
1412                },
1413                ..DataQualityPlan::default()
1414            },
1415        );
1416        let key = results.intent.as_ref().unwrap().key.clone().unwrap();
1417        // Rows missing a part are incomplete, never a repeat of each other.
1418        assert_eq!((key.groups, key.rows_involved, key.missing), (1, 2, 3));
1419        assert_eq!(
1420            matching(&df, &results, ObservationKind::KeyRepeated, "region"),
1421            2
1422        );
1423        assert_eq!(
1424            matching(&df, &results, ObservationKind::KeyMissing, "id"),
1425            3
1426        );
1427        // One finding names both columns.
1428        let report = build_report(&results);
1429        let repeated = report
1430            .findings
1431            .iter()
1432            .find(|f| f.title == "Repeated key")
1433            .unwrap();
1434        assert_eq!(repeated.columns, vec!["region", "id"]);
1435    }
1436
1437    /// Values not read, nothing checked: the check says so rather than passing.
1438    #[test]
1439    fn metadata_only_leaves_the_intent_unavailable() {
1440        let results = run(&fixture(), &plan(QualityCompute::Metadata));
1441        let intent = results.intent.as_ref().unwrap();
1442        assert!(!intent.measured);
1443        let report = build_report(&results);
1444        let all = checks(&results, &report);
1445        assert_eq!(all[0].outcome, Outcome::Unavailable("values not read"));
1446        assert!(intent.observations().is_empty());
1447    }
1448
1449    /// Dates and times compare as instants, text read as time through its format.
1450    #[test]
1451    fn ranges_compare_dates_and_text_read_as_time() {
1452        let df = df!(
1453            "day" => &["2024-01-01", "2024-02-15", "2023-12-31", "bad"],
1454        )
1455        .unwrap();
1456        let mut plan = DataQualityPlan {
1457            compute: QualityCompute::Full,
1458            intent: DeclaredIntent {
1459                key: Vec::new(),
1460                columns: vec![ColumnIntent {
1461                    min: Some("2024-01-01".to_string()),
1462                    max: Some("2024-01-31".to_string()),
1463                    ..ColumnIntent::new("day")
1464                }],
1465            },
1466            ..DataQualityPlan::default()
1467        };
1468        crate::analysis_modal::set_time_format(
1469            &mut plan,
1470            "day",
1471            Some((TimeKind::Date, "%Y-%m-%d")),
1472        );
1473        let results = run(&df, &plan);
1474        let day = results
1475            .intent
1476            .as_ref()
1477            .unwrap()
1478            .column("day")
1479            .unwrap()
1480            .clone();
1481        assert_eq!(
1482            (day.compared, day.below, day.above),
1483            (Some(3), Some(1), Some(1))
1484        );
1485        assert_eq!(day.lowest.as_deref(), Some("2023-12-31"));
1486        assert_eq!(
1487            matching(&df, &results, ObservationKind::OutOfRange, "day"),
1488            2
1489        );
1490    }
1491
1492    /// A date or datetime past the calendar is the furthest out of range a value can
1493    /// be, and counts so, where its conversion to microseconds overflowed and the
1494    /// rule never compared it (#518). Values in range count as they did.
1495    #[test]
1496    fn a_date_past_the_calendar_is_out_of_range() {
1497        const DAY_MS: i64 = 86_400_000;
1498        let paris = TimeZone::opt_try_new(Some("Europe/Paris")).unwrap();
1499        let df = df!(
1500            "d" => &[Some(19_737i32), Some(19_000), Some(i32::MAX), Some(i32::MIN), None],
1501            "ms" => &[
1502                Some(19_737 * DAY_MS),
1503                Some(19_000 * DAY_MS),
1504                Some(i64::MAX),
1505                Some(i64::MIN + 1),
1506                None,
1507            ],
1508            "us" => &[
1509                Some(19_737 * DAY_MS * 1000),
1510                Some(19_000 * DAY_MS * 1000),
1511                Some(i64::MAX),
1512                Some(i64::MIN + 1),
1513                None,
1514            ],
1515        )
1516        .unwrap()
1517        .lazy()
1518        .with_columns([
1519            col("d").cast(DataType::Date),
1520            col("ms").cast(DataType::Datetime(TimeUnit::Milliseconds, None)),
1521            col("us").cast(DataType::Datetime(TimeUnit::Microseconds, paris)),
1522        ])
1523        .collect()
1524        .unwrap();
1525        let plan = DataQualityPlan {
1526            compute: QualityCompute::Full,
1527            intent: DeclaredIntent {
1528                key: Vec::new(),
1529                columns: ["d", "ms", "us"]
1530                    .into_iter()
1531                    .map(|column| ColumnIntent {
1532                        min: Some("2024-01-01".to_string()),
1533                        max: Some("2024-12-31".to_string()),
1534                        ..ColumnIntent::new(column)
1535                    })
1536                    .collect(),
1537            },
1538            ..DataQualityPlan::default()
1539        };
1540        for streaming in [false, true] {
1541            let results = compute_data_quality(
1542                &df.clone().lazy(),
1543                Some(df.height()),
1544                &plan,
1545                None,
1546                streaming,
1547            )
1548            .unwrap();
1549            for (column, unit) in [("d", "days"), ("ms", "ms"), ("us", "us")] {
1550                let check = results.intent.as_ref().unwrap().column(column).unwrap();
1551                // 2024-01-15 in range; 2022-01-08 and the two past the calendar out.
1552                assert_eq!(
1553                    (check.compared, check.below, check.above),
1554                    (Some(4), Some(2), Some(1)),
1555                    "{column}"
1556                );
1557                let (low, high) = match unit {
1558                    "days" => (i64::from(i32::MIN), i64::from(i32::MAX)),
1559                    _ => (i64::MIN + 1, i64::MAX),
1560                };
1561                let since = |v: i64| match unit {
1562                    "days" => format!("{v} days since 1970-01-01"),
1563                    unit => format!("{v} {unit} since 1970-01-01 UTC"),
1564                };
1565                assert_eq!(check.lowest, Some(since(low)), "{column}");
1566                assert_eq!(check.highest, Some(since(high)), "{column}");
1567                assert_eq!(
1568                    matching(&df, &results, ObservationKind::OutOfRange, column),
1569                    3,
1570                    "{column}"
1571                );
1572            }
1573        }
1574    }
1575
1576    #[test]
1577    fn allowed_values_are_split_trimmed_and_checked_against_the_type() {
1578        assert_eq!(
1579            parse_allowed(&DataType::String, " open, closed ,,open ").unwrap(),
1580            vec!["open", "closed"]
1581        );
1582        assert!(parse_allowed(&DataType::Int64, "1, two").is_err());
1583        assert_eq!(
1584            parse_allowed(&DataType::Boolean, "true").unwrap(),
1585            vec!["true"]
1586        );
1587        let many = (0..=MAX_ALLOWED_VALUES)
1588            .map(|value| value.to_string())
1589            .collect::<Vec<_>>()
1590            .join(",");
1591        assert!(parse_allowed(&DataType::String, &many).is_err());
1592    }
1593
1594    /// An allowed set compares at the column's own type: a `u64` past `i64::MAX` is
1595    /// itself, and a category is compared with its name.
1596    #[test]
1597    fn an_allowed_set_compares_at_the_columns_type() {
1598        let mut df = df!(
1599            "code" => &[u64::MAX, 1, u64::MAX - 1, 2],
1600            "kind" => &["open", "closed", "open", "lost"],
1601        )
1602        .unwrap();
1603        df = df
1604            .lazy()
1605            .with_column(col("kind").cast(DataType::from_categories(Categories::global())))
1606            .collect()
1607            .unwrap();
1608        assert!(parse_allowed(&DataType::UInt64, &u64::MAX.to_string()).is_ok());
1609        assert!(parse_allowed(&DataType::UInt64, "-1").is_err());
1610        let plan = DataQualityPlan {
1611            compute: QualityCompute::Full,
1612            intent: DeclaredIntent {
1613                key: Vec::new(),
1614                columns: vec![
1615                    ColumnIntent {
1616                        allowed: vec![u64::MAX.to_string(), "1".into()],
1617                        ..ColumnIntent::new("code")
1618                    },
1619                    ColumnIntent {
1620                        allowed: vec!["open".into(), "closed".into()],
1621                        ..ColumnIntent::new("kind")
1622                    },
1623                ],
1624            },
1625            ..DataQualityPlan::default()
1626        };
1627        let results = run(&df, &plan);
1628        let intent = results.intent.as_ref().unwrap();
1629        assert_eq!(intent.column("code").unwrap().outside, Some(2));
1630        assert_eq!(intent.column("kind").unwrap().outside, Some(1));
1631        assert_eq!(
1632            matching(&df, &results, ObservationKind::NotAllowed, "code"),
1633            2
1634        );
1635        assert_eq!(
1636            matching(&df, &results, ObservationKind::NotAllowed, "kind"),
1637            1
1638        );
1639    }
1640
1641    /// A quoted value keeps its commas, spaces and doubled quotes, and is written
1642    /// back quoted, so reopening the form reads the same set.
1643    #[test]
1644    fn a_quoted_allowed_value_holds_commas_and_spaces() {
1645        let typed = r#""a, b", c, " open", "say ""hi""", 5" pipe"#;
1646        let values = parse_allowed(&DataType::String, typed).unwrap();
1647        assert_eq!(values, vec!["a, b", "c", " open", "say \"hi\"", "5\" pipe"]);
1648        assert_eq!(
1649            parse_allowed(&DataType::String, &format_allowed(&values)).unwrap(),
1650            values
1651        );
1652        assert_eq!(
1653            parse_allowed(&DataType::Int64, r#""1", 2"#).unwrap(),
1654            vec!["1", "2"]
1655        );
1656        assert!(parse_allowed(&DataType::String, r#""a, b"#).is_err());
1657        assert!(parse_allowed(&DataType::String, r#""a" b, c"#).is_err());
1658
1659        // Compared as stored: case and spaces count.
1660        let df = df!("label" => &["a, b", "c", " open", "open", "A, B"]).unwrap();
1661        let plan = DataQualityPlan {
1662            compute: QualityCompute::Full,
1663            intent: DeclaredIntent {
1664                key: Vec::new(),
1665                columns: vec![ColumnIntent {
1666                    allowed: values,
1667                    ..ColumnIntent::new("label")
1668                }],
1669            },
1670            ..DataQualityPlan::default()
1671        };
1672        let results = run(&df, &plan);
1673        let label = results.intent.as_ref().unwrap().column("label").unwrap();
1674        assert_eq!(label.outside, Some(2));
1675        assert_eq!(
1676            matching(&df, &results, ObservationKind::NotAllowed, "label"),
1677            2
1678        );
1679    }
1680
1681    /// A date alone as a time's maximum keeps that whole day; as its minimum, the day
1682    /// starts at midnight. Both bounds are in range.
1683    #[test]
1684    fn a_date_bounds_a_datetime_column_by_whole_days() {
1685        let at = |text: &str| {
1686            chrono::NaiveDateTime::parse_from_str(text, "%Y-%m-%d %H:%M:%S")
1687                .unwrap()
1688                .and_utc()
1689                .timestamp_micros()
1690        };
1691        let df = df!(
1692            "at" => &[
1693                at("2024-06-01 00:00:00"),
1694                at("2024-06-30 23:59:59"),
1695                at("2024-07-01 00:00:00"),
1696                at("2024-05-31 23:59:59"),
1697            ],
1698        )
1699        .unwrap()
1700        .lazy()
1701        .with_column(col("at").cast(DataType::Datetime(TimeUnit::Microseconds, None)))
1702        .collect()
1703        .unwrap();
1704        let plan = DataQualityPlan {
1705            compute: QualityCompute::Full,
1706            intent: DeclaredIntent {
1707                key: Vec::new(),
1708                columns: vec![ColumnIntent {
1709                    min: Some("2024-06-01".to_string()),
1710                    max: Some("2024-06-30".to_string()),
1711                    ..ColumnIntent::new("at")
1712                }],
1713            },
1714            ..DataQualityPlan::default()
1715        };
1716        let results = run(&df, &plan);
1717        let check = results.intent.as_ref().unwrap().column("at").unwrap();
1718        assert_eq!((check.below, check.above), (Some(1), Some(1)));
1719        // The same day as both ends is a day, not an empty range.
1720        let one_day = ColumnIntent {
1721            min: Some("2024-06-30".to_string()),
1722            max: Some("2024-06-30".to_string()),
1723            ..ColumnIntent::new("at")
1724        };
1725        assert!(
1726            check_intent(
1727                &one_day,
1728                &DataType::Datetime(TimeUnit::Microseconds, None),
1729                None
1730            )
1731            .is_ok()
1732        );
1733    }
1734
1735    #[test]
1736    fn a_range_needs_bounds_of_the_columns_kind_in_order() {
1737        let number = ColumnIntent {
1738            min: Some("10".to_string()),
1739            max: Some("2".to_string()),
1740            ..ColumnIntent::new("amount")
1741        };
1742        assert_eq!(
1743            check_intent(&number, &DataType::Float64, None),
1744            Err("Minimum is above maximum".to_string())
1745        );
1746        let date = ColumnIntent {
1747            min: Some("yesterday".to_string()),
1748            ..ColumnIntent::new("day")
1749        };
1750        assert!(check_intent(&date, &DataType::Date, None).is_err());
1751        let text = ColumnIntent {
1752            min: Some("a".to_string()),
1753            ..ColumnIntent::new("name")
1754        };
1755        assert!(check_intent(&text, &DataType::String, None).is_err());
1756        // Text read as a number takes a numeric range.
1757        let parsed = ColumnIntent {
1758            number: Some(NumberReading::Decimal),
1759            min: Some("0".to_string()),
1760            ..ColumnIntent::new("price")
1761        };
1762        assert_eq!(check_intent(&parsed, &DataType::String, None), Ok(()));
1763    }
1764
1765    #[test]
1766    fn an_empty_intent_removes_the_column() {
1767        let mut declared = DeclaredIntent::default();
1768        declared.set(ColumnIntent {
1769            required: true,
1770            ..ColumnIntent::new("id")
1771        });
1772        assert_eq!(declared.columns.len(), 1);
1773        declared.set(ColumnIntent::new("id"));
1774        assert!(declared.is_empty());
1775        declared.set_key("id", true);
1776        declared.set_key("id", true);
1777        assert_eq!(declared.key, vec!["id"]);
1778        declared.set_key("id", false);
1779        assert!(declared.is_empty());
1780    }
1781}