Skip to main content

datui_lib/
quality_export.rs

1//! A Data Quality report written to a file: the measurements on screen, the setup
2//! they were measured with, and the source they were measured on. Built from the
3//! results in memory; writing one reads nothing from the data.
4//!
5//! Two forms: versioned JSON for tools, with every measurement, and a short
6//! Markdown report for people. The JSON schema is documented in
7//! `docs/reference/data-quality.md`; a change that breaks a reader bumps
8//! [`REPORT_VERSION`].
9
10use crate::data_quality::{
11    ColumnQualityProfile, DataQualityPlan, DataQualityResults, QualityPrecision, interval_label,
12};
13use crate::quality_report::{Outcome, Severity, build_report, checks, coverage, describe, verdict};
14use serde::{Deserialize, Serialize};
15use std::path::Path;
16
17/// What the JSON says it is, so a reader can tell it from any other JSON.
18pub const REPORT_FORMAT: &str = "datui-data-quality-report";
19
20/// The JSON schema's version. Fields may be added within a version; one removed,
21/// renamed or changed in meaning is a new version.
22pub const REPORT_VERSION: u32 = 1;
23
24/// The most file names the source identity lists.
25const MAX_FILE_NAMES: usize = 100;
26
27/// Which form of the report to write.
28#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
29pub enum ReportFormat {
30    #[default]
31    Json,
32    Markdown,
33}
34
35impl ReportFormat {
36    pub const ALL: [Self; 2] = [Self::Json, Self::Markdown];
37
38    pub fn label(self) -> &'static str {
39        match self {
40            Self::Json => "JSON",
41            Self::Markdown => "Markdown",
42        }
43    }
44
45    pub fn extension(self) -> &'static str {
46        match self {
47            Self::Json => "json",
48            Self::Markdown => "md",
49        }
50    }
51
52    /// What the format holds, for the dialog's line.
53    pub fn holds(self) -> &'static str {
54        match self {
55            Self::Json => "Every measurement, versioned, for tools",
56            Self::Markdown => "The verdict, coverage, findings and setup, for people",
57        }
58    }
59}
60
61/// What a report was measured on, as far as datui can say without reading it: where
62/// the data came from, what the view did to it, and for a local file its size and
63/// modification time when the run began. No content hash: that would be a read.
64#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
65pub struct SourceIdentity {
66    /// The path or URL as opened.
67    pub location: Option<String>,
68    pub remote: bool,
69    pub format: Option<String>,
70    /// Files the dataset reads, when it reads several.
71    pub files: Option<usize>,
72    /// Their names, the first hundred.
73    pub file_names: Vec<String>,
74    /// A local file's size in bytes when the run began.
75    pub bytes: Option<u64>,
76    /// A local file's modification time when the run began, RFC 3339 in UTC.
77    pub modified: Option<String>,
78    /// What the view did to the rows: query, filters, sort, reshape. Empty for the
79    /// data as loaded.
80    pub view: Vec<String>,
81}
82
83impl SourceIdentity {
84    /// Record `location`'s size and modification time, when it is one local file.
85    /// A stat, not a read; run in the worker as the run begins.
86    pub fn stat(&mut self) {
87        let Some(location) = self.location.as_deref().filter(|_| !self.remote) else {
88            return;
89        };
90        let Ok(meta) = std::fs::metadata(location) else {
91            return;
92        };
93        if !meta.is_file() {
94            return;
95        }
96        self.bytes = Some(meta.len());
97        self.modified = meta.modified().ok().map(|time| {
98            chrono::DateTime::<chrono::Utc>::from(time)
99                .to_rfc3339_opts(chrono::SecondsFormat::Secs, true)
100        });
101    }
102
103    /// File names, cut to the first hundred.
104    pub fn with_files(mut self, names: &[String]) -> Self {
105        if names.len() > 1 {
106            self.files = Some(names.len());
107            self.file_names = names.iter().take(MAX_FILE_NAMES).cloned().collect();
108        }
109        self
110    }
111}
112
113/// The report as JSON: [`REPORT_FORMAT`], version [`REPORT_VERSION`].
114#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
115pub struct ReportFile {
116    pub format: String,
117    pub version: u32,
118    /// The datui that wrote it.
119    pub datui_version: String,
120    /// When the file was written, RFC 3339 in UTC. Not when the data was read.
121    pub exported_at: String,
122    /// Where the rows came from. `None` for results no run labeled.
123    pub source: Option<SourceIdentity>,
124    pub setup: SetupJson,
125    pub run: RunJson,
126    /// The headline: `2 problems  1 note  9 of 12 columns clean`.
127    pub verdict: String,
128    pub coverage: CoverageJson,
129    pub checks: Vec<CheckJson>,
130    pub findings: Vec<FindingJson>,
131    pub columns: Vec<ColumnJson>,
132    pub duplicates: Option<DuplicatesJson>,
133    pub segments: Vec<SegmentJson>,
134    pub intervals: Vec<IntervalJson>,
135    pub intent: Option<IntentJson>,
136    /// The expected windows checked for gaps; `None` when none are stated.
137    pub gaps: Option<GapsJson>,
138}
139
140/// The setup the report was measured with: every setting a run takes.
141#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
142pub struct SetupJson {
143    /// `current view`, `whole source`, a partition, files or a row range.
144    pub scope: String,
145    /// `sample`, `full` or `metadata`.
146    pub values: String,
147    pub sample: SampleJson,
148    /// `whole dataset`, `by file`, `by year`, `by day of date`, `in chunks of …`.
149    pub grain: String,
150    /// `none`, `previous` or `baseline`.
151    pub comparison: String,
152    pub baseline_segment: Option<String>,
153    pub time_formats: Vec<TimeFormatJson>,
154    pub time_roles: Vec<TimeRoleJson>,
155    /// Measured intervals, `start to end` by role.
156    pub intervals: Vec<String>,
157    /// What puts an interval in a time window.
158    pub window_by: String,
159    pub latency_threshold_seconds: Option<i64>,
160    pub intent: DeclaredJson,
161    /// The windows rows are expected in, on a time-window grain; `None` unstated.
162    pub expected: Option<ExpectedJson>,
163}
164
165#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
166pub struct ExpectedJson {
167    /// Only Monday to Friday's hours or days are expected.
168    pub weekdays: bool,
169    /// The first expected time and the time expected windows end before, as typed;
170    /// `None` takes the first or last window the run found.
171    pub from: Option<String>,
172    pub before: Option<String>,
173}
174
175/// The expected windows checked against the segments the run counted.
176#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
177pub struct GapsJson {
178    /// `checked`; `no_values` (file metadata only), `no_windows` (no range stated
179    /// and no window found) or `too_many` (more windows than are checked).
180    pub status: String,
181    /// The grain's column and window width: `1h`, `1d`, `1w`, `1mo`.
182    pub column: String,
183    pub every: String,
184    /// `every hour`, `weekdays`, ...
185    pub cadence: String,
186    /// Windows in the stated range, for `too_many`.
187    pub windows_in_range: Option<usize>,
188    /// The first expected window's start and where the last ends, UTC.
189    pub from: Option<String>,
190    pub before: Option<String>,
191    /// Windows expected in the range, and weekend windows left out of it.
192    pub expected: Option<usize>,
193    pub weekend: Option<usize>,
194    pub with_rows: Option<usize>,
195    pub empty: Option<usize>,
196    pub not_sampled: Option<usize>,
197    pub out_of_scope: Option<usize>,
198    /// Whether a window with no rows found is known to be empty.
199    pub counted: Option<bool>,
200    /// Consecutive gap windows of one kind, in order.
201    pub runs: Vec<GapRunJson>,
202    /// Runs past the listed ones, counted and not listed.
203    pub more_runs: usize,
204}
205
206#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
207pub struct GapRunJson {
208    /// `empty`, `not_sampled` or `out_of_scope`.
209    pub kind: String,
210    /// Where its first and last windows start, UTC.
211    pub first: String,
212    pub last: String,
213    /// The windows it covers, in calendar terms: `2024-01-06 to 2024-01-07`.
214    pub span: String,
215    pub windows: usize,
216    /// Rows the scope holds in it, for windows the sample missed, when counted.
217    pub rows: Option<usize>,
218}
219
220fn utc(time: chrono::NaiveDateTime) -> String {
221    time.format("%Y-%m-%dT%H:%M:%SZ").to_string()
222}
223
224fn gaps_json(plan: &DataQualityPlan, results: &DataQualityResults) -> Option<GapsJson> {
225    use crate::quality_trends::{GapKind, Gaps};
226    let gaps = crate::quality_trends::expected_gaps(plan, results)?;
227    let crate::data_quality::QualityGrain::TimeWindows { column, every } = &plan.grain else {
228        return None;
229    };
230    let kind = |kind: GapKind| match kind {
231        GapKind::Empty => "empty",
232        GapKind::Unsampled => "not_sampled",
233        GapKind::OutOfScope => "out_of_scope",
234    };
235    let mut json = GapsJson {
236        status: String::new(),
237        column: column.clone(),
238        every: every.clone(),
239        cadence: plan
240            .expected_windows()
241            .map(|expected| expected.cadence_label(every))
242            .unwrap_or_default(),
243        windows_in_range: None,
244        from: None,
245        before: None,
246        expected: None,
247        weekend: None,
248        with_rows: None,
249        empty: None,
250        not_sampled: None,
251        out_of_scope: None,
252        counted: None,
253        runs: Vec::new(),
254        more_runs: 0,
255    };
256    json.status = match gaps {
257        Gaps::NoValues => "no_values",
258        Gaps::NoWindows => "no_windows",
259        Gaps::TooMany { windows } => {
260            json.windows_in_range = Some(windows);
261            "too_many"
262        }
263        Gaps::Checked(check) => {
264            json.from = Some(utc(check.from));
265            json.before = Some(utc(check.before));
266            json.expected = Some(check.expected);
267            json.weekend = Some(check.weekend);
268            json.with_rows = Some(check.with_rows);
269            json.empty = Some(check.empty);
270            json.not_sampled = Some(check.unsampled);
271            json.out_of_scope = Some(check.out_of_scope);
272            json.counted = Some(check.counted);
273            json.more_runs = check.more_runs;
274            json.runs = check
275                .runs
276                .iter()
277                .map(|run| GapRunJson {
278                    kind: kind(run.kind).to_string(),
279                    first: utc(run.first),
280                    last: utc(run.last),
281                    span: crate::quality_trends::calendar_span(run.first, run.last, every),
282                    windows: run.windows,
283                    rows: run.rows,
284                })
285                .collect();
286            "checked"
287        }
288    }
289    .to_string();
290    Some(json)
291}
292
293#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
294pub struct SampleJson {
295    /// `Random`, `Equal per <column>`, `First rows` or `Every row`.
296    pub method: String,
297    /// Rows asked for: in all, or per value for an equal-per-value sample.
298    pub rows: usize,
299    /// The seed: the same seed, scope and source draw the same rows.
300    pub seed: u64,
301}
302
303#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
304pub struct TimeFormatJson {
305    pub column: String,
306    /// `date` or `datetime`.
307    pub kind: String,
308    /// A strftime format.
309    pub format: String,
310}
311
312#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
313pub struct TimeRoleJson {
314    pub role: String,
315    pub column: String,
316}
317
318/// The declared intent, as Setup holds it.
319#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
320pub struct DeclaredJson {
321    pub key: Vec<String>,
322    pub columns: Vec<ColumnIntentJson>,
323}
324
325#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
326pub struct ColumnIntentJson {
327    pub column: String,
328    pub required: bool,
329    pub allowed: Vec<String>,
330    pub min: Option<String>,
331    pub max: Option<String>,
332    /// `whole number` or `decimal` for text read as a number.
333    pub read_as: Option<String>,
334}
335
336/// What the run read and how far its numbers reach.
337#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
338pub struct RunJson {
339    /// `exact`, `sampled` or `metadata`.
340    pub precision: String,
341    /// Rows in scope, when known.
342    pub total_rows: Option<usize>,
343    /// Rows measured: the sample's, or every row in scope.
344    pub evaluated_rows: usize,
345    /// Rows an equal-per-value sample kept of each value.
346    pub per_value: Option<usize>,
347    pub source_files: Option<usize>,
348    pub footers_read: Option<usize>,
349    /// What the run's reads were seen to do; `None` when nothing watched them.
350    pub reads: Option<ReadsJson>,
351}
352
353#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
354pub struct ReadsJson {
355    /// Stages that read the source.
356    pub source_reads: usize,
357    /// Of those, the ones whose rows were counted.
358    pub counted: usize,
359    /// Rows the counted reads passed through, over every pass.
360    pub rows_traversed: usize,
361    /// The local copy of a remote source a full scan's passes read, when they did.
362    #[serde(default, skip_serializing_if = "Option::is_none")]
363    pub local_copy: Option<LocalCopyJson>,
364}
365
366#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
367pub struct LocalCopyJson {
368    pub bytes: u64,
369    pub objects: usize,
370    /// This run fetched it; false when an earlier run did.
371    pub fetched_by_this_run: bool,
372}
373
374#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
375pub struct CoverageJson {
376    pub exact: usize,
377    pub sampled: usize,
378    pub metadata: usize,
379    pub skipped: usize,
380    pub unavailable: Vec<UnavailableJson>,
381    pub rows: Vec<String>,
382    pub limits: Vec<String>,
383}
384
385#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
386pub struct UnavailableJson {
387    pub reason: String,
388    pub checks: Vec<String>,
389}
390
391#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
392pub struct CheckJson {
393    pub name: String,
394    pub looks_for: String,
395    pub applies_to: String,
396    /// `passed`, `found`, `skipped` or `unavailable`.
397    pub outcome: String,
398    /// What was found, or why it was skipped or unavailable.
399    pub detail: Option<String>,
400    /// `exact`, `sampled` or `metadata`.
401    pub basis: String,
402}
403
404#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
405pub struct FindingJson {
406    /// `problem`, `note` or `clean`.
407    pub severity: String,
408    pub title: String,
409    pub columns: Vec<String>,
410    pub affected_rows: usize,
411    pub evaluated_rows: usize,
412    pub summary: String,
413    /// The finding's numbers in a sentence, as its detail shows them.
414    pub headline: String,
415    pub evidence: Vec<String>,
416}
417
418#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
419pub struct ColumnJson {
420    pub name: String,
421    pub dtype: String,
422    pub evaluated_rows: usize,
423    pub null_count: usize,
424    pub distinct_count: Option<usize>,
425    pub empty_count: Option<usize>,
426    pub whitespace_count: Option<usize>,
427    pub nan_count: Option<usize>,
428    pub positive_infinity_count: Option<usize>,
429    pub negative_infinity_count: Option<usize>,
430    pub min: Option<String>,
431    pub max: Option<String>,
432    pub dominant_value: Option<String>,
433    pub dominant_count: Option<usize>,
434    pub min_length: Option<usize>,
435    pub max_length: Option<usize>,
436    pub integer_parse_count: Option<usize>,
437    pub decimal_parse_count: Option<usize>,
438    pub date_parse_count: Option<usize>,
439    pub datetime_parse_count: Option<usize>,
440}
441
442#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
443pub struct DuplicatesJson {
444    pub groups: usize,
445    pub extra_rows: usize,
446    pub rows_involved: usize,
447    pub evaluated_rows: usize,
448}
449
450#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
451pub struct SegmentJson {
452    pub label: String,
453    /// Rows in the segment, when counted.
454    pub total_rows: Option<usize>,
455    pub evaluated_rows: usize,
456    pub null_cells: usize,
457    pub null_rate: f64,
458    pub compared_with: Option<String>,
459    pub largest_change: Option<String>,
460}
461
462#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
463pub struct IntervalJson {
464    pub interval: String,
465    pub segment: String,
466    pub start_column: String,
467    pub end_column: String,
468    pub rows: usize,
469    /// Rows with both ends present and read: what negative, zero and over are of.
470    pub both_ends: usize,
471    pub missing_start: usize,
472    pub missing_end: usize,
473    pub unparsed_start: usize,
474    pub unparsed_end: usize,
475    pub negative: usize,
476    pub zero: usize,
477    pub p50_seconds: Option<i64>,
478    pub p90_seconds: Option<i64>,
479    pub p95_seconds: Option<i64>,
480    pub p99_seconds: Option<i64>,
481    pub max_seconds: Option<i64>,
482    /// `duration > threshold`, strictly.
483    pub threshold_seconds: Option<i64>,
484    pub over_threshold: Option<usize>,
485}
486
487/// What the declared intent found.
488#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
489pub struct IntentJson {
490    /// False when the run read no values, so nothing declared was checked.
491    pub measured: bool,
492    pub precision: String,
493    pub evaluated_rows: usize,
494    pub key: Option<KeyJson>,
495    pub columns: Vec<ColumnCheckJson>,
496    /// Declared columns the scope did not have.
497    pub absent: Vec<String>,
498}
499
500#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
501pub struct KeyJson {
502    pub columns: Vec<String>,
503    /// Rows with no value in some part of the key.
504    pub missing: usize,
505    /// Key values held by more than one row.
506    pub groups: usize,
507    pub extra_rows: usize,
508    pub rows_involved: usize,
509}
510
511#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
512pub struct ColumnCheckJson {
513    pub column: String,
514    pub dtype: String,
515    /// Rows with a value, as stored.
516    pub values: usize,
517    pub missing: Option<usize>,
518    pub unparsed: Option<usize>,
519    pub outside: Option<usize>,
520    /// Values the range compared.
521    pub compared: Option<usize>,
522    pub below: Option<usize>,
523    pub above: Option<usize>,
524    pub lowest: Option<String>,
525    pub highest: Option<String>,
526}
527
528fn precision_label(precision: QualityPrecision) -> String {
529    precision.label().to_string()
530}
531
532fn column_json(column: &ColumnQualityProfile) -> ColumnJson {
533    ColumnJson {
534        name: column.name.clone(),
535        dtype: column.dtype.to_string(),
536        evaluated_rows: column.evaluated_rows,
537        null_count: column.null_count,
538        distinct_count: column.distinct_count,
539        empty_count: column.empty_count,
540        whitespace_count: column.whitespace_count,
541        nan_count: column.nan_count,
542        positive_infinity_count: column.positive_infinity_count,
543        negative_infinity_count: column.negative_infinity_count,
544        min: column.min.clone(),
545        max: column.max.clone(),
546        dominant_value: column.dominant_value.clone(),
547        dominant_count: column.dominant_count,
548        min_length: column.min_length,
549        max_length: column.max_length,
550        integer_parse_count: column.integer_parse_count,
551        decimal_parse_count: column.decimal_parse_count,
552        date_parse_count: column.date_parse_count,
553        datetime_parse_count: column.datetime_parse_count,
554    }
555}
556
557fn setup_json(plan: &DataQualityPlan) -> SetupJson {
558    SetupJson {
559        scope: plan.scope.label(),
560        values: plan.compute.label().to_string(),
561        sample: SampleJson {
562            method: plan.method.label(),
563            rows: plan.dataset_rows,
564            seed: plan.sample_seed,
565        },
566        grain: plan.grain.label(),
567        comparison: plan.comparison.label().to_string(),
568        baseline_segment: plan.baseline_segment.clone(),
569        time_formats: plan
570            .time_formats
571            .iter()
572            .map(|format| TimeFormatJson {
573                column: format.column.clone(),
574                kind: format.kind.label().to_string(),
575                format: format.format.clone(),
576            })
577            .collect(),
578        time_roles: plan
579            .temporal_roles
580            .iter()
581            .map(|assignment| TimeRoleJson {
582                role: assignment.role.label().to_string(),
583                column: assignment.column.clone(),
584            })
585            .collect(),
586        intervals: plan
587            .interval_pairs()
588            .into_iter()
589            .map(interval_label)
590            .collect(),
591        window_by: plan.interval_clock.label().to_string(),
592        latency_threshold_seconds: plan.latency_threshold_seconds,
593        expected: plan.expected_windows().map(|expected| ExpectedJson {
594            weekdays: expected.weekdays,
595            from: expected.from.clone(),
596            before: expected.before.clone(),
597        }),
598        intent: DeclaredJson {
599            key: plan.intent.key.clone(),
600            columns: plan
601                .intent
602                .columns
603                .iter()
604                .map(|intent| ColumnIntentJson {
605                    column: intent.column.clone(),
606                    required: intent.required,
607                    allowed: intent.allowed.clone(),
608                    min: intent.min.clone(),
609                    max: intent.max.clone(),
610                    read_as: intent.number.map(|number| number.label().to_string()),
611                })
612                .collect(),
613        },
614    }
615}
616
617fn outcome_json(outcome: &Outcome) -> (String, Option<String>) {
618    match outcome {
619        Outcome::Passed => ("passed".to_string(), None),
620        Outcome::Found { detail, .. } => ("found".to_string(), Some(detail.clone())),
621        Outcome::Skipped(reason) => ("skipped".to_string(), Some(reason.to_string())),
622        Outcome::Unavailable(reason) => ("unavailable".to_string(), Some(reason.to_string())),
623    }
624}
625
626/// The report as [`ReportFile`]: built from `results`, measured with `plan` on
627/// `results.source`, at `exported_at`. Nothing here reads the data.
628pub fn report_file(
629    results: &DataQualityResults,
630    plan: &DataQualityPlan,
631    exported_at: &str,
632) -> ReportFile {
633    let report = build_report(results);
634    let all_checks = checks(results, &report);
635    let covered = coverage(results, &all_checks, plan);
636    let severity = |severity: Severity| match severity {
637        Severity::Problem => "problem",
638        Severity::Note => "note",
639        Severity::Clean => "clean",
640    };
641    ReportFile {
642        format: REPORT_FORMAT.to_string(),
643        version: REPORT_VERSION,
644        datui_version: env!("CARGO_PKG_VERSION").to_string(),
645        exported_at: exported_at.to_string(),
646        source: results.source.as_deref().cloned(),
647        setup: setup_json(plan),
648        run: RunJson {
649            precision: precision_label(results.precision),
650            total_rows: results.total_rows,
651            evaluated_rows: results.evaluated_rows,
652            per_value: results.per_value,
653            source_files: results.source_files,
654            footers_read: results.footers_read,
655            reads: results.reads.map(|reads| ReadsJson {
656                source_reads: reads.reads,
657                counted: reads.counted,
658                rows_traversed: reads.rows,
659                local_copy: reads.copy.map(|copy| LocalCopyJson {
660                    bytes: copy.bytes,
661                    objects: copy.objects,
662                    fetched_by_this_run: copy.fetched,
663                }),
664            }),
665        },
666        verdict: verdict(&report),
667        coverage: CoverageJson {
668            exact: covered.exact,
669            sampled: covered.sampled,
670            metadata: covered.metadata,
671            skipped: covered.skipped,
672            unavailable: covered
673                .unavailable
674                .iter()
675                .map(|(reason, names)| UnavailableJson {
676                    reason: reason.to_string(),
677                    checks: names.iter().map(|name| name.to_string()).collect(),
678                })
679                .collect(),
680            rows: covered.rows.clone(),
681            limits: covered.limits.clone(),
682        },
683        checks: all_checks
684            .iter()
685            .map(|check| {
686                let (outcome, detail) = outcome_json(&check.outcome);
687                CheckJson {
688                    name: check.name.to_string(),
689                    looks_for: check.looks_for.to_string(),
690                    applies_to: check.applies_to.clone(),
691                    outcome,
692                    detail,
693                    basis: precision_label(check.basis),
694                }
695            })
696            .collect(),
697        findings: report
698            .findings
699            .iter()
700            .map(|finding| {
701                let (headline, evidence) = describe(finding, results);
702                FindingJson {
703                    severity: severity(finding.severity).to_string(),
704                    title: finding.title.to_string(),
705                    columns: finding.columns.clone(),
706                    affected_rows: finding.affected_rows,
707                    evaluated_rows: finding.evaluated_rows,
708                    summary: finding.summary.clone(),
709                    headline,
710                    evidence,
711                }
712            })
713            .collect(),
714        columns: results.columns.iter().map(column_json).collect(),
715        duplicates: results.identity.as_ref().map(|identity| DuplicatesJson {
716            groups: identity.duplicate_groups,
717            extra_rows: identity.extra_rows,
718            rows_involved: identity.rows_involved,
719            evaluated_rows: identity.evaluated_rows,
720        }),
721        segments: results
722            .segments
723            .iter()
724            .map(|segment| SegmentJson {
725                label: segment.label.clone(),
726                total_rows: segment.total_rows,
727                evaluated_rows: segment.evaluated_rows,
728                null_cells: segment.null_cells,
729                null_rate: segment.null_rate,
730                compared_with: segment.compared_with.clone(),
731                largest_change: segment.largest_change.clone(),
732            })
733            .collect(),
734        intervals: results
735            .temporal
736            .iter()
737            .map(|interval| IntervalJson {
738                interval: interval.label(),
739                segment: interval.segment.clone(),
740                start_column: interval.start_column.clone(),
741                end_column: interval.end_column.clone(),
742                rows: interval.evaluated_rows,
743                both_ends: interval.paired_rows,
744                missing_start: interval.missing_start,
745                missing_end: interval.missing_end,
746                unparsed_start: interval.unparsed_start,
747                unparsed_end: interval.unparsed_end,
748                negative: interval.negative_count,
749                zero: interval.zero_count,
750                p50_seconds: interval.p50_seconds,
751                p90_seconds: interval.p90_seconds,
752                p95_seconds: interval.p95_seconds,
753                p99_seconds: interval.p99_seconds,
754                max_seconds: interval.max_seconds,
755                threshold_seconds: interval.threshold_seconds,
756                over_threshold: interval.above_threshold_count,
757            })
758            .collect(),
759        gaps: gaps_json(plan, results),
760        intent: results.intent.as_ref().map(|intent| IntentJson {
761            measured: intent.measured,
762            precision: precision_label(intent.precision),
763            evaluated_rows: intent.evaluated_rows,
764            key: intent.key.as_ref().map(|key| KeyJson {
765                columns: key.columns.clone(),
766                missing: key.missing,
767                groups: key.groups,
768                extra_rows: key.extra_rows,
769                rows_involved: key.rows_involved,
770            }),
771            columns: intent
772                .columns
773                .iter()
774                .map(|check| ColumnCheckJson {
775                    column: check.intent.column.clone(),
776                    dtype: check.dtype.to_string(),
777                    values: check.values,
778                    missing: check.missing,
779                    unparsed: check.unparsed,
780                    outside: check.outside,
781                    compared: check.compared,
782                    below: check.below,
783                    above: check.above,
784                    lowest: check.lowest.clone(),
785                    highest: check.highest.clone(),
786                })
787                .collect(),
788            absent: intent.absent.clone(),
789        }),
790    }
791}
792
793/// The JSON report, pretty-printed.
794pub fn to_json(file: &ReportFile) -> color_eyre::Result<String> {
795    Ok(serde_json::to_string_pretty(file)?)
796}
797
798/// A cell of a Markdown table: pipes escaped, line breaks flattened.
799fn cell(text: &str) -> String {
800    text.replace('|', "\\|").replace('\n', " ")
801}
802
803/// The report for people: what was measured on, the verdict and how far it reaches,
804/// the findings with their numbers and evidence, and the setup to run it again.
805pub fn to_markdown(file: &ReportFile) -> String {
806    let mut out = String::new();
807    let mut line = |text: String| {
808        out.push_str(&text);
809        out.push('\n');
810    };
811    line("# Data quality report".to_string());
812    line(String::new());
813    if let Some(source) = &file.source {
814        if let Some(location) = &source.location {
815            let mut facts = Vec::new();
816            if let Some(format) = &source.format {
817                facts.push(format.clone());
818            }
819            if let Some(files) = source.files {
820                facts.push(format!("{} files", crate::numfmt::group_chrome(files)));
821            }
822            if let Some(bytes) = source.bytes {
823                facts.push(format!(
824                    "{} bytes",
825                    crate::numfmt::group_chrome(bytes as usize)
826                ));
827            }
828            if let Some(modified) = &source.modified {
829                facts.push(format!("modified {modified}"));
830            }
831            let facts = if facts.is_empty() {
832                String::new()
833            } else {
834                format!(" ({})", facts.join(", "))
835            };
836            line(format!("- Source: `{location}`{facts}"));
837        }
838        if !source.view.is_empty() {
839            line(format!("- View: {}", source.view.join("; ")));
840        }
841    }
842    let run = &file.run;
843    let rows = match (run.precision.as_str(), run.total_rows) {
844        ("metadata", _) => "file metadata only, no values read".to_string(),
845        ("exact", _) => format!(
846            "all {} rows, exact",
847            crate::numfmt::group_chrome(run.evaluated_rows)
848        ),
849        (_, Some(total)) => format!(
850            "{} of {} rows, sampled",
851            crate::numfmt::group_chrome(run.evaluated_rows),
852            crate::numfmt::group_chrome(total)
853        ),
854        (_, None) => format!(
855            "{} rows, sampled",
856            crate::numfmt::group_chrome(run.evaluated_rows)
857        ),
858    };
859    line(format!("- Measured: {rows}"));
860    line(format!(
861        "- Written by datui {} at {}, from the report on screen; no data read",
862        file.datui_version, file.exported_at
863    ));
864    line(String::new());
865    // The verdict's parts sit two spaces apart on screen; a comma here.
866    line(format!("**{}**", file.verdict.replace("  ", ", ")));
867    line(String::new());
868
869    line("## Coverage".to_string());
870    line(String::new());
871    let coverage = &file.coverage;
872    let checks = [
873        (coverage.exact, "exact"),
874        (coverage.sampled, "sampled"),
875        (coverage.metadata, "metadata"),
876        (coverage.skipped, "skipped"),
877        (
878            coverage
879                .unavailable
880                .iter()
881                .map(|unavailable| unavailable.checks.len())
882                .sum(),
883            "unavailable",
884        ),
885    ]
886    .into_iter()
887    .filter(|(count, _)| *count > 0)
888    .map(|(count, label)| format!("{count} {label}"))
889    .collect::<Vec<_>>();
890    line(format!("- Checks: {}", checks.join(", ")));
891    if !coverage.rows.is_empty() {
892        line(format!("- Rows: {}", coverage.rows.join(", ")));
893    }
894    let limits = coverage
895        .unavailable
896        .iter()
897        .map(|unavailable| format!("{}: {}", unavailable.checks.join(", "), unavailable.reason))
898        .chain(coverage.limits.iter().cloned())
899        .collect::<Vec<_>>();
900    if !limits.is_empty() {
901        line(format!("- Limits: {}", limits.join("; ")));
902    }
903    line(String::new());
904
905    for (severity, heading) in [("problem", "Problems"), ("note", "Notes")] {
906        let findings = file
907            .findings
908            .iter()
909            .filter(|finding| finding.severity == severity)
910            .collect::<Vec<_>>();
911        if findings.is_empty() {
912            continue;
913        }
914        line(format!("## {heading}"));
915        line(String::new());
916        for finding in findings {
917            line(format!(
918                "- **{}** ({}): {}",
919                finding.title,
920                finding.columns.join(", "),
921                finding.headline
922            ));
923            for evidence in &finding.evidence {
924                line(format!("  - {}", evidence.trim()));
925            }
926        }
927        line(String::new());
928    }
929    if let Some(clean) = file
930        .findings
931        .iter()
932        .find(|finding| finding.severity == "clean")
933    {
934        line("## Clean".to_string());
935        line(String::new());
936        line(clean.columns.join(", "));
937        line(String::new());
938    }
939
940    line("## Checks".to_string());
941    line(String::new());
942    line("| Check | Covers | Read | Result |".to_string());
943    line("|---|---|---|---|".to_string());
944    for check in &file.checks {
945        let result = match &check.detail {
946            Some(detail) => format!("{}: {detail}", check.outcome),
947            None => check.outcome.clone(),
948        };
949        line(format!(
950            "| {} | {} | {} | {} |",
951            cell(&check.name),
952            cell(&check.applies_to),
953            cell(&check.basis),
954            cell(&result)
955        ));
956    }
957    line(String::new());
958
959    if !file.intervals.is_empty() {
960        line("## Intervals".to_string());
961        line(String::new());
962        line("| Interval | Segment | Both ends | p50 s | p95 s | Negative | Over |".to_string());
963        line("|---|---|---|---|---|---|---|".to_string());
964        let seconds = |value: Option<i64>| value.map_or("-".to_string(), |value| value.to_string());
965        for interval in &file.intervals {
966            line(format!(
967                "| {} | {} | {} | {} | {} | {} | {} |",
968                cell(&interval.interval),
969                cell(&interval.segment),
970                interval.both_ends,
971                seconds(interval.p50_seconds),
972                seconds(interval.p95_seconds),
973                interval.negative,
974                interval
975                    .over_threshold
976                    .map_or("-".to_string(), |over| over.to_string())
977            ));
978        }
979        line(String::new());
980    }
981
982    if let Some(gaps) = &file.gaps {
983        line("## Gaps".to_string());
984        line(String::new());
985        let expected = format!("Expected {} by {}", gaps.cadence, gaps.column);
986        match gaps.status.as_str() {
987            "no_values" => line(format!("{expected}: file metadata only counts no windows")),
988            "no_windows" => line(format!("{expected}: no window found and no range stated")),
989            "too_many" => line(format!(
990                "{expected}: {} windows in range, more than are checked",
991                crate::numfmt::group_chrome(gaps.windows_in_range.unwrap_or(0))
992            )),
993            _ => {
994                let count = |value: Option<usize>| crate::numfmt::group_chrome(value.unwrap_or(0));
995                let mut facts = vec![
996                    format!("{} with rows", count(gaps.with_rows)),
997                    format!("{} empty", count(gaps.empty)),
998                    format!("{} not sampled", count(gaps.not_sampled)),
999                    format!("{} out of scope", count(gaps.out_of_scope)),
1000                ];
1001                if gaps.weekend.unwrap_or(0) > 0 {
1002                    facts.push(format!(
1003                        "{} weekend windows not expected",
1004                        count(gaps.weekend)
1005                    ));
1006                }
1007                line(format!(
1008                    "{expected}, {} to before {}: {} windows; {}",
1009                    gaps.from.as_deref().unwrap_or(""),
1010                    gaps.before.as_deref().unwrap_or(""),
1011                    count(gaps.expected),
1012                    facts.join(", ")
1013                ));
1014                if !gaps.runs.is_empty() {
1015                    line(String::new());
1016                    line("| Windows | Gap | Count | Rows |".to_string());
1017                    line("|---|---|---|---|".to_string());
1018                    for run in &gaps.runs {
1019                        line(format!(
1020                            "| {} | {} | {} | {} |",
1021                            cell(&run.span),
1022                            run.kind.replace('_', " "),
1023                            run.windows,
1024                            run.rows.map_or("-".to_string(), |rows| rows.to_string())
1025                        ));
1026                    }
1027                    if gaps.more_runs > 0 {
1028                        line(String::new());
1029                        line(format!("{} more runs not listed", gaps.more_runs));
1030                    }
1031                }
1032            }
1033        }
1034        line(String::new());
1035    }
1036
1037    line("## Setup".to_string());
1038    line(String::new());
1039    line("| Setting | Value |".to_string());
1040    line("|---|---|".to_string());
1041    let setup = &file.setup;
1042    let none = || "none".to_string();
1043    let joined = |items: Vec<String>| {
1044        if items.is_empty() {
1045            none()
1046        } else {
1047            items.join(", ")
1048        }
1049    };
1050    let mut rows = vec![
1051        ("Scope", setup.scope.clone()),
1052        ("Values", setup.values.clone()),
1053        (
1054            "Sample",
1055            // Every row takes no size and no seed, and the first rows no seed.
1056            match setup.sample.method.as_str() {
1057                "Every row" => setup.sample.method.clone(),
1058                "First rows" => format!(
1059                    "First rows, {} rows",
1060                    crate::numfmt::group_chrome(setup.sample.rows)
1061                ),
1062                method if method.starts_with("Equal per ") => format!(
1063                    "{method}, {} rows per value, seed {}",
1064                    crate::numfmt::group_chrome(setup.sample.rows),
1065                    setup.sample.seed
1066                ),
1067                method => format!(
1068                    "{method}, {} rows, seed {}",
1069                    crate::numfmt::group_chrome(setup.sample.rows),
1070                    setup.sample.seed
1071                ),
1072            },
1073        ),
1074        (
1075            "Text as time",
1076            joined(
1077                setup
1078                    .time_formats
1079                    .iter()
1080                    .map(|format| {
1081                        format!("{} as {} `{}`", format.column, format.kind, format.format)
1082                    })
1083                    .collect(),
1084            ),
1085        ),
1086        (
1087            "Time roles",
1088            joined(
1089                setup
1090                    .time_roles
1091                    .iter()
1092                    .map(|role| format!("{} = {}", role.role, role.column))
1093                    .collect(),
1094            ),
1095        ),
1096        ("Intervals", joined(setup.intervals.clone())),
1097        ("Grain", setup.grain.clone()),
1098        (
1099            "Compare",
1100            match &setup.baseline_segment {
1101                Some(segment) => format!("{} {segment}", setup.comparison),
1102                None => setup.comparison.clone(),
1103            },
1104        ),
1105        (
1106            "Latency over",
1107            setup
1108                .latency_threshold_seconds
1109                .map_or_else(none, |seconds| format!("{seconds} s")),
1110        ),
1111        ("Window by", setup.window_by.clone()),
1112        (
1113            "Expected",
1114            setup.expected.as_ref().map_or_else(none, |expected| {
1115                let windows = crate::data_quality::ExpectedWindows {
1116                    weekdays: expected.weekdays,
1117                    from: expected.from.clone(),
1118                    before: expected.before.clone(),
1119                };
1120                let every = match file.gaps.as_ref() {
1121                    Some(gaps) => gaps.every.as_str(),
1122                    None => "",
1123                };
1124                format!(
1125                    "{}, {}",
1126                    windows.cadence_label(every),
1127                    windows.range_label()
1128                )
1129            }),
1130        ),
1131        (
1132            "Key",
1133            if setup.intent.key.is_empty() {
1134                none()
1135            } else {
1136                setup.intent.key.join(", ")
1137            },
1138        ),
1139    ];
1140    for intent in &setup.intent.columns {
1141        let mut rules = Vec::new();
1142        if intent.required {
1143            rules.push("required".to_string());
1144        }
1145        if let Some(read_as) = &intent.read_as {
1146            rules.push(format!("read as {read_as}"));
1147        }
1148        if !intent.allowed.is_empty() {
1149            rules.push(format!(
1150                "one of {}",
1151                crate::quality_intent::format_allowed(&intent.allowed)
1152            ));
1153        }
1154        match (&intent.min, &intent.max) {
1155            (Some(min), Some(max)) => rules.push(format!("{min} to {max}")),
1156            (Some(min), None) => rules.push(format!("at least {min}")),
1157            (None, Some(max)) => rules.push(format!("at most {max}")),
1158            (None, None) => {}
1159        }
1160        rows.push(("Intent", format!("{}: {}", intent.column, rules.join("; "))));
1161    }
1162    for (setting, value) in rows {
1163        line(format!("| {setting} | {} |", cell(&value)));
1164    }
1165    out
1166}
1167
1168/// The report in `format`, ready to write.
1169pub fn render(
1170    results: &DataQualityResults,
1171    plan: &DataQualityPlan,
1172    format: ReportFormat,
1173    exported_at: &str,
1174) -> color_eyre::Result<String> {
1175    let file = report_file(results, plan, exported_at);
1176    match format {
1177        ReportFormat::Json => to_json(&file),
1178        ReportFormat::Markdown => Ok(to_markdown(&file)),
1179    }
1180}
1181
1182/// Write the report to `path`, replacing a file there only under
1183/// [`Overwrite::Replace`](crate::output_file::Overwrite). The results are in
1184/// memory: this writes, and reads nothing.
1185pub fn write(
1186    path: &Path,
1187    results: &DataQualityResults,
1188    plan: &DataQualityPlan,
1189    format: ReportFormat,
1190    overwrite: crate::output_file::Overwrite,
1191) -> color_eyre::Result<()> {
1192    use std::io::Write;
1193    let now = chrono::Utc::now().to_rfc3339_opts(chrono::SecondsFormat::Secs, true);
1194    let text = render(results, plan, format, &now)?;
1195    let mut out = crate::output_file::OutputFile::create(path, overwrite)?;
1196    out.file().write_all(text.as_bytes())?;
1197    out.commit()?;
1198    Ok(())
1199}
1200
1201/// The export dialog: where to write, and which form.
1202pub struct ExportForm {
1203    pub path: crate::widgets::text_input::TextInput,
1204    pub format: ReportFormat,
1205    /// The cursor is on the format rather than the path.
1206    pub on_format: bool,
1207    /// Why Enter did not write, until the next edit.
1208    pub error: Option<String>,
1209}
1210
1211impl ExportForm {
1212    /// The dialog on `stem`'s report as JSON in the current directory.
1213    pub fn new(stem: &str, theme: &crate::config::Theme) -> Self {
1214        let mut path = crate::widgets::text_input::TextInput::new().with_theme(theme);
1215        path.suggest(format!("{stem}-quality.{}", ReportFormat::Json.extension()));
1216        path.set_focused(true);
1217        Self {
1218            path,
1219            format: ReportFormat::Json,
1220            on_format: false,
1221            error: None,
1222        }
1223    }
1224
1225    /// The next or previous form; a path ending in the old form's extension takes
1226    /// the new one's.
1227    pub fn cycle_format(&mut self) {
1228        let next = match self.format {
1229            ReportFormat::Json => ReportFormat::Markdown,
1230            ReportFormat::Markdown => ReportFormat::Json,
1231        };
1232        let old = format!(".{}", self.format.extension());
1233        let value = self.path.value().to_string();
1234        if let Some(stem) = value.strip_suffix(&old) {
1235            let renamed = format!("{stem}.{}", next.extension());
1236            if self.path.is_suggested() {
1237                self.path.suggest(renamed);
1238            } else {
1239                self.path.set_value(renamed);
1240            }
1241        }
1242        self.format = next;
1243        self.error = None;
1244    }
1245
1246    pub fn toggle_focus(&mut self) {
1247        self.on_format = !self.on_format;
1248        self.path.set_focused(!self.on_format);
1249    }
1250
1251    /// Where to write and in which form: `~` and `$VAR` expanded, a typed `.json`
1252    /// or `.md` choosing the form, and the form's extension added when the path has
1253    /// none. A blank path is refused.
1254    pub fn target(&self) -> Result<(std::path::PathBuf, ReportFormat), String> {
1255        let typed = self.path.value().trim();
1256        if typed.is_empty() {
1257            return Err("Type a path to write to".to_string());
1258        }
1259        let mut path = crate::home::expand_user_path(typed);
1260        let typed_format = path
1261            .extension()
1262            .and_then(|extension| extension.to_str())
1263            .and_then(|extension| {
1264                ReportFormat::ALL
1265                    .into_iter()
1266                    .find(|format| extension.eq_ignore_ascii_case(format.extension()))
1267            });
1268        if path.extension().is_none() {
1269            path.set_extension(self.format.extension());
1270        }
1271        Ok((path, typed_format.unwrap_or(self.format)))
1272    }
1273}
1274
1275#[cfg(test)]
1276mod tests {
1277    use super::*;
1278    use crate::data_quality::{QualityCompute, compute_data_quality};
1279    use crate::quality_intent::{ColumnIntent, DeclaredIntent};
1280    use polars::prelude::*;
1281
1282    fn measured() -> (DataQualityResults, DataQualityPlan) {
1283        let df = df!(
1284            "id" => &[1i64, 2, 2, 4],
1285            "status" => &[Some("open"), Some("void"), None, Some("open")],
1286        )
1287        .unwrap();
1288        let plan = DataQualityPlan {
1289            compute: QualityCompute::Full,
1290            intent: DeclaredIntent {
1291                key: vec!["id".to_string()],
1292                columns: vec![ColumnIntent {
1293                    required: true,
1294                    allowed: vec!["open".to_string(), "closed".to_string()],
1295                    ..ColumnIntent::new("status")
1296                }],
1297            },
1298            ..DataQualityPlan::default()
1299        };
1300        let mut results = compute_data_quality(&df.lazy(), Some(4), &plan, None, false).unwrap();
1301        results.source = Some(Box::new(SourceIdentity {
1302            location: Some("orders.parquet".to_string()),
1303            format: Some("Parquet".to_string()),
1304            view: vec!["filter: status = open".to_string()],
1305            ..SourceIdentity::default()
1306        }));
1307        (results, plan)
1308    }
1309
1310    /// The JSON reads back into the schema it was written from, with its format and
1311    /// version, the setup, the intent and its findings.
1312    #[test]
1313    fn json_round_trips_through_its_schema() {
1314        let (results, plan) = measured();
1315        let text = render(&results, &plan, ReportFormat::Json, "2026-09-30T00:00:00Z").unwrap();
1316        let file: ReportFile = serde_json::from_str(&text).unwrap();
1317        assert_eq!(file, report_file(&results, &plan, "2026-09-30T00:00:00Z"));
1318        assert_eq!(file.format, REPORT_FORMAT);
1319        assert_eq!(file.version, REPORT_VERSION);
1320        assert_eq!(file.setup.sample.seed, plan.sample_seed);
1321        assert_eq!(file.setup.values, "full");
1322        assert_eq!(file.setup.intent.key, vec!["id"]);
1323        assert_eq!(file.run.precision, "exact");
1324        assert_eq!(file.run.evaluated_rows, 4);
1325        assert_eq!(
1326            file.source.as_ref().unwrap().location.as_deref(),
1327            Some("orders.parquet")
1328        );
1329        let key = file.intent.as_ref().unwrap().key.clone().unwrap();
1330        assert_eq!((key.groups, key.rows_involved), (1, 2));
1331        assert!(
1332            file.findings
1333                .iter()
1334                .any(|finding| finding.title == "Repeated key")
1335        );
1336        assert_eq!(file.checks[0].name, "Column intent");
1337        // A generic reader finds the version where the schema says.
1338        let value: serde_json::Value = serde_json::from_str(&text).unwrap();
1339        assert_eq!(value["version"], serde_json::json!(REPORT_VERSION));
1340        assert_eq!(value["format"], serde_json::json!(REPORT_FORMAT));
1341    }
1342
1343    /// The Markdown names the source, the verdict, coverage, every finding with its
1344    /// numbers, and the setup to run it again.
1345    #[test]
1346    fn markdown_says_what_was_measured_and_how() {
1347        let (results, plan) = measured();
1348        let text = render(
1349            &results,
1350            &plan,
1351            ReportFormat::Markdown,
1352            "2026-09-30T00:00:00Z",
1353        )
1354        .unwrap();
1355        for expected in [
1356            "# Data quality report",
1357            "- Source: `orders.parquet` (Parquet)",
1358            "- View: filter: status = open",
1359            "- Measured: all 4 rows, exact",
1360            "## Coverage",
1361            "## Problems",
1362            "**Repeated key** (id): 2 of 4 rows (50.0%) share their key with another row",
1363            "**Not allowed** (status): 1 of 3 values (33.3%) are not allowed",
1364            "  - Allowed: open, closed",
1365            "| Column intent |",
1366            "| Key | id |",
1367            "| Intent | status: required; one of open, closed |",
1368            &format!("seed {}", plan.sample_seed),
1369        ] {
1370            assert!(text.contains(expected), "missing {expected:?} in\n{text}");
1371        }
1372        // Every row takes no size or seed, so the setup names none.
1373        let every_row = DataQualityPlan {
1374            method: crate::sampling::SampleMethod::EveryRow,
1375            ..plan
1376        };
1377        let text = render(
1378            &results,
1379            &every_row,
1380            ReportFormat::Markdown,
1381            "2026-09-30T00:00:00Z",
1382        )
1383        .unwrap();
1384        assert!(text.contains("| Sample | Every row |"), "{text}");
1385    }
1386
1387    /// Expected windows go into the setup, and the gaps they find into the report:
1388    /// counts by kind and each run, in JSON and in the Markdown.
1389    #[test]
1390    fn gaps_are_exported_with_their_windows() {
1391        // 2024-01-01 to 2024-01-10, with no rows on the 4th and 5th.
1392        let days = [0, 1, 2, 5, 6, 7, 8, 9];
1393        let base = chrono::NaiveDate::from_ymd_opt(2024, 1, 1).unwrap();
1394        let epoch = chrono::NaiveDate::from_ymd_opt(1970, 1, 1).unwrap();
1395        let day = days
1396            .iter()
1397            .map(|offset| (base - epoch).num_days() as i32 + offset)
1398            .collect::<Vec<_>>();
1399        let lf = df!("day" => day, "value" => [1i64, 2, 3, 4, 5, 6, 7, 8])
1400            .unwrap()
1401            .lazy()
1402            .with_column(col("day").cast(DataType::Date));
1403        let plan = DataQualityPlan {
1404            compute: QualityCompute::Full,
1405            grain: crate::data_quality::QualityGrain::TimeWindows {
1406                column: "day".to_string(),
1407                every: "1d".to_string(),
1408            },
1409            expected: Some(crate::data_quality::ExpectedWindows {
1410                weekdays: false,
1411                from: Some("2024-01-01".to_string()),
1412                before: Some("2024-01-12".to_string()),
1413            }),
1414            ..DataQualityPlan::default()
1415        };
1416        let results = compute_data_quality(&lf, None, &plan, None, false).unwrap();
1417        let file = report_file(&results, &plan, "2026-09-30T00:00:00Z");
1418        let text = to_json(&file).unwrap();
1419        assert_eq!(serde_json::from_str::<ReportFile>(&text).unwrap(), file);
1420        let expected = file.setup.expected.as_ref().unwrap();
1421        assert_eq!(expected.from.as_deref(), Some("2024-01-01"));
1422        let gaps = file.gaps.as_ref().unwrap();
1423        assert_eq!(gaps.status, "checked");
1424        assert_eq!((gaps.column.as_str(), gaps.every.as_str()), ("day", "1d"));
1425        assert_eq!(gaps.from.as_deref(), Some("2024-01-01T00:00:00Z"));
1426        assert_eq!(gaps.before.as_deref(), Some("2024-01-12T00:00:00Z"));
1427        assert_eq!(
1428            (gaps.expected, gaps.with_rows, gaps.empty, gaps.not_sampled),
1429            (Some(11), Some(8), Some(3), Some(0))
1430        );
1431        assert_eq!(gaps.counted, Some(true));
1432        let runs = gaps
1433            .runs
1434            .iter()
1435            .map(|run| (run.kind.as_str(), run.span.as_str(), run.windows))
1436            .collect::<Vec<_>>();
1437        assert_eq!(
1438            runs,
1439            [
1440                ("empty", "2024-01-04 to 2024-01-05", 2),
1441                ("empty", "2024-01-11", 1)
1442            ]
1443        );
1444
1445        let markdown = to_markdown(&file);
1446        for expected in [
1447            "## Gaps",
1448            "Expected every day by day, 2024-01-01T00:00:00Z to before 2024-01-12T00:00:00Z: 11 windows; 8 with rows, 3 empty, 0 not sampled, 0 out of scope",
1449            "| 2024-01-04 to 2024-01-05 | empty | 2 | - |",
1450            "| Expected | every day, 2024-01-01 to before 2024-01-12 |",
1451        ] {
1452            assert!(
1453                markdown.contains(expected),
1454                "missing {expected:?} in\n{markdown}"
1455            );
1456        }
1457
1458        // No windows stated: no gaps, and the setup says none.
1459        let unstated = DataQualityPlan {
1460            expected: None,
1461            ..plan
1462        };
1463        let file = report_file(&results, &unstated, "2026-09-30T00:00:00Z");
1464        assert!(file.gaps.is_none() && file.setup.expected.is_none());
1465        assert!(!to_markdown(&file).contains("## Gaps"));
1466    }
1467
1468    #[test]
1469    fn the_path_takes_the_forms_extension() {
1470        let theme =
1471            crate::config::Theme::from_config(&crate::config::AppConfig::default().theme).unwrap();
1472        let mut form = ExportForm::new("orders", &theme);
1473        assert_eq!(form.path.value(), "orders-quality.json");
1474        form.cycle_format();
1475        assert_eq!(form.format, ReportFormat::Markdown);
1476        assert_eq!(form.path.value(), "orders-quality.md");
1477        // Still the form's own suggestion, so typing replaces it.
1478        assert!(form.path.is_suggested());
1479        form.path.set_value("report");
1480        assert_eq!(
1481            form.target().unwrap(),
1482            (
1483                std::path::PathBuf::from("report.md"),
1484                ReportFormat::Markdown
1485            )
1486        );
1487        // A typed extension chooses the form.
1488        form.path.set_value("report.json");
1489        assert_eq!(form.target().unwrap().1, ReportFormat::Json);
1490        form.path.set_value("  ");
1491        assert!(form.target().is_err());
1492    }
1493}