Skip to main content

datui_lib/widgets/
analysis.rs

1use ratatui::{
2    buffer::Buffer,
3    layout::{Constraint, Direction, Layout, Rect},
4    style::{Color, Modifier, Style},
5    text::{Line, Span},
6    widgets::{
7        Bar, BarChart, BarGroup, Block, Cell, Chart, Dataset, GraphType, HighlightSpacing,
8        Paragraph, Row, StatefulWidget, Table, TableState, Widget,
9    },
10};
11
12use crate::analysis_modal::{
13    AnalysisFocus, AnalysisTool, AnalysisView, ColumnScroll, HistogramScale,
14};
15use crate::chart_data::{AxisFormat, AxisNumbers};
16use crate::config::Theme;
17use crate::distribution_fit::{FitOutcome, FitTest};
18use crate::glyphs::PlotMarks;
19use crate::numfmt::{self, NumberFormatSettings};
20use crate::render::context::RenderContext;
21use crate::statistics::{
22    AnalysisContext, AnalysisResults, CategoricalStatistics, ColumnStatistics, CorrelationMethod,
23    DistributionAnalysis, DistributionType, NumericStatistics, TemporalStatistics,
24};
25use crate::widgets::axes::{AxisSpec, PlotAxes};
26use crate::widgets::datatable::DataTableState;
27use crate::widgets::ui::Surface;
28use polars::prelude::{AnyValue, DataType};
29
30pub struct AnalysisWidgetConfig<'a> {
31    pub state: &'a DataTableState,
32    pub results: Option<&'a AnalysisResults>,
33    pub context: &'a AnalysisContext,
34    pub view: AnalysisView,
35    pub selected_tool: Option<AnalysisTool>,
36    pub selected_correlation: Option<(usize, usize)>,
37    pub correlation_method: CorrelationMethod,
38    pub focus: AnalysisFocus,
39    pub selected_theoretical_distribution: DistributionType,
40    pub histogram_scale: HistogramScale,
41    pub theme: &'a Theme,
42    pub table_cell_padding: u16,
43    /// Display-time number formatting, so counts here match the data table.
44    pub number_format: &'a NumberFormatSettings,
45    /// The shared sample the results were read with, for the header.
46    pub sample: &'a crate::sampling::Sample,
47    pub ctx: &'a RenderContext,
48}
49
50pub struct AnalysisWidget<'a> {
51    _state: &'a DataTableState,
52    results: Option<&'a AnalysisResults>,
53    _context: &'a AnalysisContext,
54    view: AnalysisView,
55    selected_tool: Option<AnalysisTool>,
56    table_state: &'a mut TableState,
57    distribution_table_state: &'a mut TableState,
58    correlation_table_state: &'a mut TableState,
59    sidebar_state: &'a mut TableState,
60    selected_correlation: Option<(usize, usize)>,
61    correlation_method: CorrelationMethod,
62    focus: AnalysisFocus,
63    selected_theoretical_distribution: DistributionType,
64    distribution_selector_state: &'a mut TableState,
65    histogram_scale: HistogramScale,
66    theme: &'a Theme,
67    table_cell_padding: u16,
68    number_format: &'a NumberFormatSettings,
69    sample: &'a crate::sampling::Sample,
70    ctx: &'a RenderContext,
71    /// The selected tool's statistic scroll; the table sets how far it goes.
72    column_scroll: &'a mut ColumnScroll,
73}
74
75impl<'a> AnalysisWidget<'a> {
76    pub fn new(
77        config: AnalysisWidgetConfig<'a>,
78        table_state: &'a mut TableState,
79        distribution_table_state: &'a mut TableState,
80        correlation_table_state: &'a mut TableState,
81        sidebar_state: &'a mut TableState,
82        distribution_selector_state: &'a mut TableState,
83        column_scroll: &'a mut ColumnScroll,
84    ) -> Self {
85        Self {
86            _state: config.state,
87            results: config.results,
88            _context: config.context,
89            view: config.view,
90            selected_tool: config.selected_tool,
91            table_state,
92            distribution_table_state,
93            correlation_table_state,
94            sidebar_state,
95            selected_correlation: config.selected_correlation,
96            correlation_method: config.correlation_method,
97            focus: config.focus,
98            selected_theoretical_distribution: config.selected_theoretical_distribution,
99            distribution_selector_state,
100            histogram_scale: config.histogram_scale,
101            theme: config.theme,
102            table_cell_padding: config.table_cell_padding,
103            number_format: config.number_format,
104            sample: config.sample,
105            ctx: config.ctx,
106            column_scroll,
107        }
108    }
109}
110
111impl<'a> Widget for AnalysisWidget<'a> {
112    fn render(self, area: Rect, buf: &mut Buffer) {
113        match self.view {
114            AnalysisView::Main => self.render_main_view(area, buf),
115            AnalysisView::DistributionDetail => self.render_distribution_detail(area, buf),
116            AnalysisView::CorrelationDetail => self.render_correlation_detail(area, buf),
117        }
118    }
119}
120
121impl<'a> AnalysisWidget<'a> {
122    fn render_main_view(self, area: Rect, buf: &mut Buffer) {
123        // The tool list never takes more than a third of the screen: the
124        // results are what the screen is for.
125        let sidebar_width = sidebar_width(area.width);
126
127        // Full-screen layout: breadcrumb, main area (no separate keybind hints line)
128        let layout = Layout::default()
129            .direction(Direction::Vertical)
130            .constraints([
131                Constraint::Length(1), // Breadcrumb
132                Constraint::Fill(1),   // Main area + sidebar
133            ])
134            .split(area);
135
136        // Breadcrumb: tool name when a tool is selected, or "Analysis" when none selected
137        let tool_name = match self.selected_tool {
138            Some(AnalysisTool::Describe) => "Describe".to_string(),
139            Some(AnalysisTool::DistributionAnalysis) => "Distribution Analysis".to_string(),
140            // The coefficient is part of the title: the cells do not say which it is.
141            Some(AnalysisTool::CorrelationMatrix) => format!(
142                "Correlation Matrix {} {}",
143                crate::glyphs::get().middot,
144                coefficient_name(self.correlation_method)
145            ),
146            Some(AnalysisTool::DataQuality) => "Data Quality".to_string(),
147            None => "Analysis".to_string(),
148        };
149
150        // What the numbers are of, stated rather than implied: a sample says how big,
151        // of how many, and of which rows, so a surprising figure can be told apart
152        // from a rare one.
153        let breadcrumb_text = match self.results {
154            Some(results) if self.selected_tool.is_some() => format!(
155                "{tool_name} {} {}",
156                crate::glyphs::get().middot,
157                self.sample
158                    .outcome(results.total_rows, results.sample_size, results.per_value)
159            ),
160            _ => tool_name,
161        };
162
163        let header_row_style = header_style(self.theme, "controls_bg", "table_header");
164        Paragraph::new(breadcrumb_text)
165            .style(header_row_style)
166            .render(layout[0], buf);
167
168        // Split main area into content area and sidebar
169        let main_layout = Layout::default()
170            .direction(Direction::Horizontal)
171            .constraints([
172                Constraint::Fill(1),               // Main content area
173                Constraint::Length(sidebar_width), // Sidebar
174            ])
175            .split(layout[1]);
176
177        // Main content area: instructions when no tool selected, else selected tool (or "Computing...")
178        match self.selected_tool {
179            None => {
180                const INSTRUCTION_LINES: u16 = 1;
181                let inner = Layout::default()
182                    .direction(Direction::Vertical)
183                    .constraints([
184                        Constraint::Min(0),
185                        Constraint::Length(INSTRUCTION_LINES),
186                        Constraint::Min(0),
187                    ])
188                    .split(main_layout[0]);
189                Paragraph::new("Pick a tool in the sidebar")
190                    .centered()
191                    .style(Style::default().fg(self.theme.get("text_primary")))
192                    .render(inner[1], buf);
193            }
194            Some(tool) => {
195                if let Some(results) = self.results {
196                    match tool {
197                        AnalysisTool::Describe => {
198                            StatisticsTable {
199                                results,
200                                focused: self.focus == AnalysisFocus::Main,
201                                theme: self.theme,
202                                table_cell_padding: self.table_cell_padding,
203                                number_format: self.number_format,
204                            }
205                            .render(
206                                main_layout[0],
207                                buf,
208                                self.table_state,
209                                self.column_scroll,
210                            );
211                        }
212                        AnalysisTool::DistributionAnalysis => {
213                            render_distribution_table(
214                                results,
215                                self.distribution_table_state,
216                                self.column_scroll,
217                                self.focus == AnalysisFocus::Main,
218                                main_layout[0],
219                                buf,
220                                self.theme,
221                            );
222                        }
223                        AnalysisTool::CorrelationMatrix => {
224                            render_correlation_matrix(
225                                results.correlation_matrix.as_ref().map(|matrix| Shown {
226                                    matrix,
227                                    method: self.correlation_method,
228                                }),
229                                self.correlation_table_state,
230                                MatrixCursor {
231                                    cell: self.selected_correlation,
232                                    focused: self.focus == AnalysisFocus::Main,
233                                },
234                                self.column_scroll,
235                                main_layout[0],
236                                buf,
237                                self.theme,
238                            );
239                        }
240                        AnalysisTool::DataQuality => {
241                            Paragraph::new("Data Quality")
242                                .centered()
243                                .render(main_layout[0], buf);
244                        }
245                    }
246                }
247                // No result yet: the Sample form fills this pane until the first run,
248                // and the progress overlay covers it during one.
249            }
250        }
251
252        // Sidebar: Tool list
253        render_sidebar(
254            main_layout[1],
255            buf,
256            self.sidebar_state,
257            self.selected_tool,
258            self.focus,
259            self.theme,
260        );
261
262        // Keybind hints are now shown on the main bottom bar (see lib.rs)
263    }
264
265    fn render_distribution_detail(self, area: Rect, buf: &mut Buffer) {
266        // Get selected distribution
267        let selected_idx = self.distribution_table_state.selected();
268        let dist_analysis: Option<&DistributionAnalysis> = self.results.and_then(|results| {
269            selected_idx.and_then(|idx| results.distribution_analyses.get(idx))
270        });
271
272        if let Some(dist) = dist_analysis {
273            // Layout: breadcrumb, main content (no keybind hints line)
274            let layout = Layout::default()
275                .direction(Direction::Vertical)
276                .constraints([
277                    Constraint::Length(1), // Breadcrumb
278                    Constraint::Fill(1),   // Main content
279                ])
280                .split(area);
281
282            // The breadcrumb carries the name alone; the control bar says Esc.
283            let title_text = format!("Distribution Analysis: {}", dist.column_name);
284            let header_row_style = header_style(self.theme, "controls_bg", "table_header");
285            Paragraph::new(title_text)
286                .style(header_row_style)
287                .render(layout[0], buf);
288
289            // The key figures over the charts, wrapped rather than cut: at 60
290            // columns they take three lines.
291            let stats = condensed_statistics_lines(
292                &condensed_statistics(dist),
293                layout[1].width,
294                self.theme,
295            );
296            let stats_height = (stats.len() as u16).clamp(1, 3);
297            let main_layout = Layout::default()
298                .direction(Direction::Vertical)
299                .constraints([Constraint::Length(stats_height), Constraint::Fill(1)])
300                .split(layout[1]);
301            Paragraph::new(stats).render(main_layout[0], buf);
302
303            // Charts on the left; the family list on the right, never narrower than
304            // its longest name and p-value.
305            let content_layout = Layout::default()
306                .direction(Direction::Horizontal)
307                .constraints([
308                    Constraint::Fill(1),
309                    Constraint::Length(SELECTOR_WIDTH.max(main_layout[1].width / 4)),
310                ])
311                .split(main_layout[1]);
312
313            // Left side: Q-Q plot and histogram with spacing
314            let charts_layout = Layout::default()
315                .direction(Direction::Vertical)
316                .constraints([
317                    Constraint::Percentage(52), // Q-Q plot (slightly reduced to make room for spacing)
318                    Constraint::Length(1),      // Vertical spacing between charts
319                    Constraint::Percentage(47), // Histogram (slightly reduced to make room for spacing)
320                ])
321                .split(content_layout[0]);
322
323            // Add padding around chart areas for better visual separation
324            let chart_padding = 1u16; // 1 character padding on all sides
325            let right_padding_extra = 1u16; // Extra padding on right side to separate from distribution box
326            let top_padding_extra = 1u16; // Extra padding at top to separate title from chart
327            let qq_plot_area = Rect::new(
328                charts_layout[0].left() + chart_padding,
329                charts_layout[0].top() + chart_padding + top_padding_extra, // Extra top padding
330                charts_layout[0]
331                    .width
332                    .saturating_sub(chart_padding) // Left padding
333                    .saturating_sub(right_padding_extra), // Extra right padding
334                charts_layout[0]
335                    .height
336                    .saturating_sub(chart_padding * 2)
337                    .saturating_sub(top_padding_extra), // Account for extra top padding
338            );
339            let histogram_area = Rect::new(
340                charts_layout[2].left() + chart_padding,
341                charts_layout[2].top() + chart_padding + top_padding_extra, // Extra top padding
342                charts_layout[2]
343                    .width
344                    .saturating_sub(chart_padding) // Left padding
345                    .saturating_sub(right_padding_extra), // Extra right padding
346                charts_layout[2]
347                    .height
348                    .saturating_sub(chart_padding * 2)
349                    .saturating_sub(top_padding_extra), // Account for extra top padding
350            );
351
352            // The value axes, both x axes and the Q-Q plot's y, read the column's
353            // numbers over the sample's range; the histogram's y reads counts.
354            let values = AxisNumbers::measure(self.number_format, &dist.column_name);
355            let counts = AxisNumbers::count(self.number_format);
356            let sorted_data = &dist.sorted_sample_values;
357            let unified_x_range = match (sorted_data.first(), sorted_data.last()) {
358                (Some(&lo), Some(&hi)) => (lo, hi),
359                _ => (0.0, 1.0),
360            };
361
362            // Both plots' y labels take one width, so the plots start in the same
363            // column: the widest Q-Q value, or the widest count the histogram could
364            // reach, the sample's size.
365            let (lo, hi) = unified_x_range;
366            let qq_format = AxisFormat::ends_and_middle([lo, hi], &values);
367            let qq_width = [lo, (lo + hi) / 2.0, hi]
368                .iter()
369                .filter_map(|&v| qq_format.label(v, 0))
370                .map(|l| l.chars().count())
371                .max()
372                .unwrap_or(1);
373            let n = sorted_data.len() as f64;
374            let count_width = AxisFormat::new(&[0.0, n], &counts)
375                .label(n, 0)
376                .map_or(1, |l| l.chars().count());
377            let shared_y_axis_label_width = (qq_width.max(count_width) as u16).max(1) + 1;
378
379            // Both plots of the selected theoretical distribution, on one x range.
380            let plot = DistributionPlotConfig {
381                dist,
382                dist_type: self.selected_theoretical_distribution,
383                area: qq_plot_area,
384                shared_y_axis_label_width,
385                theme: self.theme,
386                unified_x_range: Some(unified_x_range),
387                histogram_scale: self.histogram_scale,
388                glyphs: crate::glyphs::get(),
389                values: &values,
390                counts: &counts,
391            };
392            render_qq_plot(plot, buf);
393
394            // Check if log scale is requested but can't be used
395            // Use actual data values, not unified range (which may include theoretical bounds and padding)
396            let sorted_data = &dist.sorted_sample_values;
397            let can_use_log_scale = !sorted_data.is_empty() && sorted_data.iter().all(|&v| v > 0.0);
398            let log_scale_requested_but_unavailable =
399                matches!(self.histogram_scale, HistogramScale::Log) && !can_use_log_scale;
400
401            render_distribution_histogram(
402                DistributionPlotConfig {
403                    area: histogram_area,
404                    ..plot
405                },
406                buf,
407            );
408
409            render_distribution_selector(
410                SelectorConfig {
411                    dist,
412                    selected: self.selected_theoretical_distribution,
413                    histogram_scale: self.histogram_scale,
414                    log_scale_unavailable: log_scale_requested_but_unavailable,
415                    theme: self.theme,
416                    ctx: self.ctx,
417                },
418                self.distribution_selector_state,
419                content_layout[1],
420                buf,
421            );
422
423        // No keybind hints line - removed
424        } else {
425            Paragraph::new("No distribution selected")
426                .centered()
427                .render(area, buf);
428        }
429    }
430
431    fn render_correlation_detail(self, area: Rect, buf: &mut Buffer) {
432        let matrix = self
433            .results
434            .and_then(|results| results.correlation_matrix.as_ref());
435        let pair = self.selected_correlation.and_then(|(row, col)| {
436            matrix.and_then(|m| {
437                (row < m.columns.len() && col < m.columns.len()).then_some((row, col))
438            })
439        });
440
441        let (Some(matrix), Some((row, col))) = (matrix, pair) else {
442            Paragraph::new("No correlation pair selected")
443                .centered()
444                .render(area, buf);
445            return;
446        };
447
448        let layout = Layout::default()
449            .direction(Direction::Vertical)
450            .constraints([Constraint::Length(1), Constraint::Fill(1)])
451            .split(area);
452
453        // The breadcrumb carries the pair alone; the control bar says Esc.
454        let title_text = format!(
455            "Correlation: {} vs {}",
456            matrix.columns[row], matrix.columns[col]
457        );
458        let header_row_style = header_style(self.theme, "controls_bg", "table_header");
459        Paragraph::new(title_text)
460            .style(header_row_style)
461            .render(layout[0], buf);
462
463        let total_rows = self.results.map(|r| r.total_rows).unwrap_or(0);
464        render_correlation_pair_summary(
465            Shown {
466                matrix,
467                method: self.correlation_method,
468            },
469            (row, col),
470            total_rows,
471            layout[1],
472            buf,
473            self.theme,
474            self.number_format,
475        );
476    }
477}
478
479/// The correlation matrix as the screen shows it: by the method chosen.
480#[derive(Clone, Copy)]
481struct Shown<'a> {
482    matrix: &'a crate::statistics::CorrelationMatrix,
483    method: CorrelationMethod,
484}
485
486/// Why a matrix has no Spearman: the rows read hold more values than it ranks.
487const SPEARMAN_TOO_MANY: &str = "Too many values to rank for Spearman; s chooses a smaller sample";
488
489/// The coefficient's name and symbol, as the matrix title and the pair detail
490/// give it.
491fn coefficient_name(method: CorrelationMethod) -> String {
492    match method {
493        CorrelationMethod::Pearson => "Pearson r".to_string(),
494        CorrelationMethod::Spearman => format!("Spearman {}", crate::glyphs::get().rho),
495    }
496}
497
498/// `r` at `decimals` places, where rounding never makes it ±1 unless it is: a
499/// near-perfect 0.9996 reads 0.999 at three places, not a perfect 1.000.
500fn format_coefficient(r: f64, decimals: usize) -> String {
501    let text = format!("{r:.decimals$}");
502    if r.abs() < 1.0 && text.trim_start_matches('-').starts_with('1') {
503        let sign = if r < 0.0 { "-" } else { "" };
504        format!("{sign}0.{}", "9".repeat(decimals))
505    } else {
506        text
507    }
508}
509
510/// A short reading of a coefficient, using the same 0.05/0.3 boundaries as the
511/// matrix's colors so the word never disagrees with the color.
512fn describe_correlation(r: f64) -> &'static str {
513    let strength = r.abs();
514    if strength < 0.05 {
515        "none"
516    } else if strength < 0.3 {
517        if r > 0.0 {
518            "weak positive"
519        } else {
520            "weak negative"
521        }
522    } else if strength < 0.7 {
523        if r > 0.0 {
524            "moderate positive"
525        } else {
526            "moderate negative"
527        }
528    } else if r > 0.0 {
529        "strong positive"
530    } else {
531        "strong negative"
532    }
533}
534
535/// The body of the correlation pair detail: everything the matrix already knows
536/// about the pair. Nothing is collected here — a scatter or per-column moments
537/// would need the pair's values, which the correlation results do not carry.
538fn render_correlation_pair_summary(
539    Shown { matrix, method }: Shown,
540    (row, col): (usize, usize),
541    total_rows: usize,
542    area: Rect,
543    buf: &mut Buffer,
544    theme: &Theme,
545    number_format: &NumberFormatSettings,
546) {
547    let r = matrix.coefficient(method, row, col);
548    let pairs = matrix.sample_sizes[row][col];
549    let p_value = matrix.p_value(method, row, col);
550
551    let label_style = Style::default().fg(theme.get("text_secondary"));
552    let value_style = Style::default().fg(theme.get("text_primary"));
553
554    let mut lines: Vec<Line> = Vec::new();
555    if r.is_nan() {
556        let unranked = method == CorrelationMethod::Spearman && matrix.rank_correlations.is_none();
557        let why = if unranked {
558            SPEARMAN_TOO_MANY
559        } else if pairs < 3 {
560            "Fewer than 3 overlapping pairs"
561        } else {
562            "A column holds one value"
563        };
564        lines.push(Line::from(vec![Span::styled(why, value_style)]));
565    } else {
566        lines.push(Line::from(vec![
567            Span::styled(format!("{}: ", coefficient_name(method)), label_style),
568            Span::styled(
569                format_coefficient(r, 4),
570                Style::default().fg(get_correlation_color(r, theme)),
571            ),
572            Span::styled(format!("  ({})", describe_correlation(r)), value_style),
573        ]));
574        lines.push(Line::from(vec![
575            Span::styled(format!("{}: ", crate::glyphs::get().r_squared), label_style),
576            Span::styled(format_coefficient(r * r, 4), value_style),
577        ]));
578        if let Some(p) = p_value {
579            lines.push(Line::from(vec![
580                Span::styled("P-value: ", label_style),
581                Span::styled(format_pvalue(p), value_style),
582            ]));
583        }
584    }
585    lines.push(Line::from(vec![
586        Span::styled("Pairs used: ", label_style),
587        Span::styled(
588            format!(
589                "{} of {} rows",
590                format_count(pairs, number_format),
591                format_count(total_rows, number_format)
592            ),
593            value_style,
594        ),
595    ]));
596
597    // One character of margin, like the distribution detail's charts.
598    let inner = Rect::new(
599        area.left() + 1,
600        area.top() + 1,
601        area.width.saturating_sub(2),
602        area.height.saturating_sub(1),
603    );
604    Paragraph::new(lines).render(inner, buf);
605}
606
607/// The Describe tool's table: a row per column, a column per statistic.
608struct StatisticsTable<'a> {
609    results: &'a AnalysisResults,
610    focused: bool,
611    theme: &'a Theme,
612    table_cell_padding: u16,
613    number_format: &'a NumberFormatSettings,
614}
615
616impl StatisticsTable<'_> {
617    fn render(
618        self,
619        area: Rect,
620        buf: &mut Buffer,
621        table_state: &mut TableState,
622        columns: &mut ColumnScroll,
623    ) {
624        let StatisticsTable {
625            results,
626            focused,
627            theme,
628            table_cell_padding,
629            number_format,
630        } = self;
631        let num_columns = results.column_statistics.len();
632        if num_columns == 0 {
633            Paragraph::new("No columns to display")
634                .centered()
635                .render(area, buf);
636            return;
637        }
638
639        // Statistics to display (in order) - internal names for matching data
640        let stat_names = vec![
641            "count",
642            "null_count",
643            "mean",
644            "std",
645            "min",
646            "25%",
647            "50%",
648            "75%",
649            "max",
650        ];
651        // Display names in Title case for headers
652        let stat_display_names = vec![
653            "Count", "Nulls", "Mean", "Std", "Min", "25%", "50%", "75%", "Max",
654        ];
655        let num_stats = stat_names.len();
656
657        // Calculate column widths based on header names and content (minimal spacing)
658        // First, determine minimum width for each column based on header length
659        // Note: ratatui Table adds 1 space between columns by default, so we don't add extra padding
660        let mut min_col_widths: Vec<u16> = stat_display_names
661            .iter()
662            .map(|name| name.chars().count() as u16) // header length (no extra padding - table handles spacing)
663            .collect();
664
665        // Scan all data to find maximum width needed for each column
666        for col_stat in &results.column_statistics {
667            for (stat_idx, stat_name) in stat_names.iter().enumerate() {
668                let value_str = describe_value(col_stat, stat_name, number_format);
669                let value_len = value_str.chars().count() as u16;
670                // Ensure width is at least the header length (already initialized) AND value length
671                // This preserves header widths even if all data values are shorter
672                let header_len = stat_display_names[stat_idx].chars().count() as u16;
673                min_col_widths[stat_idx] = min_col_widths[stat_idx].max(value_len).max(header_len);
674                // must fit both header and content (no padding - table handles spacing)
675            }
676        }
677
678        // Locked column width (column name) - calculate from header text AND actual column names
679        let header_text = "Column";
680        let header_len = header_text.chars().count() as u16;
681        let max_col_name_len = results
682            .column_statistics
683            .iter()
684            .map(|cs| cs.name.chars().count() as u16)
685            .max()
686            .unwrap_or(header_len);
687        let locked_col_width = max_col_name_len.max(header_len).max(10); // min 10, must fit both header and data (no padding - table handles spacing)
688
689        let column_spacing = table_cell_padding;
690        // The rail's column comes first, then the locked names.
691        let available_width = area
692            .width
693            .saturating_sub(RAIL_WIDTH + locked_col_width)
694            .saturating_sub(column_spacing);
695        let (start_stat, end_stat) =
696            stat_window(&min_col_widths, available_width, column_spacing, columns);
697        let visible_stats: Vec<usize> = (start_stat..end_stat).collect();
698
699        if visible_stats.is_empty() {
700            return;
701        }
702
703        let mut rows = Vec::new();
704
705        let mut header_cells = vec![Cell::from("Column").style(Style::default())];
706        for &stat_idx in &visible_stats {
707            header_cells.push(Cell::from(stat_display_names[stat_idx]).style(Style::default()));
708        }
709        let header_row_style = header_style(theme, "controls_bg", "table_header");
710        let header_row = Row::new(header_cells.clone()).style(header_row_style);
711
712        for col_stat in &results.column_statistics {
713            let mut cells = vec![
714                Cell::from(col_stat.name.as_str())
715                    .style(Style::default().fg(theme.get("text_primary"))),
716            ];
717            for &stat_idx in &visible_stats {
718                let stat_name = stat_names[stat_idx];
719                let value = describe_value(col_stat, stat_name, number_format);
720
721                cells.push(Cell::from(value));
722            }
723
724            rows.push(Row::new(cells));
725        }
726
727        let mut constraints = vec![Constraint::Length(locked_col_width)];
728        for &stat_idx in &visible_stats {
729            // Use minimum width needed (ratatui will add spacing between columns)
730            constraints.push(Constraint::Length(min_col_widths[stat_idx]));
731        }
732
733        let table = Table::new(rows, constraints)
734            .header(header_row)
735            .column_spacing(table_cell_padding)
736            .row_highlight_style(cursor_style(focused, theme))
737            .highlight_symbol(cursor_rail(focused, theme))
738            .highlight_spacing(HighlightSpacing::Always);
739
740        StatefulWidget::render(table, area, buf, table_state);
741        draw_scroll_marks(
742            area,
743            buf,
744            RAIL_WIDTH + locked_col_width,
745            (start_stat, end_stat, num_stats),
746            theme,
747        );
748    }
749}
750
751/// One Describe cell. A date, time or duration column gets its range, quartiles
752/// and mean in its own format; the statistics it has no value for, and nulls, read `-`.
753fn describe_value(
754    col_stat: &ColumnStatistics,
755    stat_name: &str,
756    number_format: &NumberFormatSettings,
757) -> String {
758    let numeric = |f: fn(&NumericStatistics) -> f64| {
759        col_stat.numeric_stats.as_ref().map(|n| format_num(f(n)))
760    };
761    let temporal = |f: fn(&TemporalStatistics) -> &Option<String>| {
762        col_stat.temporal_stats.as_ref().and_then(|t| f(t).clone())
763    };
764    let categorical = |f: fn(&CategoricalStatistics) -> &Option<String>| {
765        col_stat
766            .categorical_stats
767            .as_ref()
768            .and_then(|c| f(c).clone())
769    };
770    match stat_name {
771        "count" => Some(format_count(col_stat.count, number_format)),
772        "null_count" => Some(format_count(col_stat.null_count, number_format)),
773        "mean" => numeric(|n| n.mean).or_else(|| temporal(|t| &t.mean)),
774        "std" => numeric(|n| n.std),
775        "min" => numeric(|n| n.min)
776            .or_else(|| temporal(|t| &t.min))
777            .or_else(|| categorical(|c| &c.min)),
778        "25%" => numeric(|n| n.q25).or_else(|| temporal(|t| &t.q25)),
779        "50%" => numeric(|n| n.median).or_else(|| temporal(|t| &t.median)),
780        "75%" => numeric(|n| n.q75).or_else(|| temporal(|t| &t.q75)),
781        "max" => numeric(|n| n.max)
782            .or_else(|| temporal(|t| &t.max))
783            .or_else(|| categorical(|c| &c.max)),
784        _ => None,
785    }
786    .unwrap_or_else(|| "-".to_string())
787}
788
789/// Format a row/null count, following the same grouping setting as the data
790/// table so a user who turned formatting on sees it everywhere they read
791/// numbers. Float statistics go through `format_num`, which switches to
792/// scientific notation well before grouping would apply.
793fn format_count(n: usize, settings: &NumberFormatSettings) -> String {
794    let fmt = settings.formatter_for("", &DataType::UInt64);
795    let mut scratch = String::new();
796    numfmt::format_any_value(&fmt, &AnyValue::UInt64(n as u64), &mut scratch).into_owned()
797}
798
799fn format_num(n: f64) -> String {
800    if n.is_nan() {
801        "-".to_string()
802    } else if n.abs() >= 1000.0 || (n.abs() < 0.01 && n != 0.0) {
803        format!("{:.2e}", n)
804    } else {
805        format!("{:.2}", n)
806    }
807}
808
809// Phase 6: Format p-value with special handling for very small values
810fn format_pvalue(p: f64) -> String {
811    if p < 0.001 {
812        "<0.001".to_string()
813    } else {
814        format!("{:.3}", p)
815    }
816}
817
818/// The p-value beside a column's verdict: the chosen family's, or with no clear fit the
819/// best any family managed. A bound reads as one.
820fn verdict_pvalue(dist: &DistributionAnalysis) -> String {
821    let chosen = dist.fit(dist.distribution_type).and_then(FitOutcome::test);
822    let best = || {
823        dist.fits
824            .iter()
825            .filter_map(|(_, outcome)| outcome.test())
826            .max_by(|a, b| a.p_value.total_cmp(&b.p_value))
827    };
828    match chosen.or_else(best) {
829        Some(test) => format_fit_pvalue(test),
830        None => "N/A".to_string(),
831    }
832}
833
834/// A fit test's p-value: `<0.005` when no simulated sample reached the column's
835/// statistic, since then the p-value is only a bound.
836fn format_fit_pvalue(test: &FitTest) -> String {
837    if test.at_bound() {
838        format!("<{:.3}", test.p_value)
839    } else {
840        format!("{:.3}", test.p_value)
841    }
842}
843
844/// Holds, marginal, rejected.
845fn pvalue_style(p: f64, theme: &Theme) -> Style {
846    if p >= 0.05 {
847        Style::default().fg(theme.get("distribution_normal"))
848    } else if p > 0.01 {
849        Style::default().fg(theme.get("distribution_skewed"))
850    } else {
851        Style::default().fg(theme.get("outlier_marker"))
852    }
853}
854
855/// Build header-style: bg+fg when bg_key is not Reset, else fg-only.
856pub(crate) fn header_style(theme: &Theme, bg_key: &str, fg_key: &str) -> Style {
857    let bg = theme.get(bg_key);
858    let fg = theme.get(fg_key);
859    if bg == Color::Reset {
860        Style::default().fg(fg)
861    } else {
862        Style::default().bg(bg).fg(fg)
863    }
864}
865
866fn render_distribution_table(
867    results: &AnalysisResults,
868    table_state: &mut TableState,
869    columns: &mut ColumnScroll,
870    focused: bool,
871    area: Rect,
872    buf: &mut Buffer,
873    theme: &Theme,
874) {
875    if results.distribution_analyses.is_empty() {
876        Paragraph::new("No numeric columns for distribution analysis")
877            .centered()
878            .render(area, buf);
879        return;
880    }
881
882    // Column headers for width calculation (excluding "Column" which will be locked)
883    // Phase 6: Add P-value column after Distribution
884    let column_names = [
885        "Distribution",
886        "P-value",
887        "Shapiro-Francia",
888        "SF p-value",
889        "CV",
890        "Outliers",
891        "Skewness",
892        "Kurtosis",
893    ];
894    let num_stats = column_names.len();
895
896    // Calculate column widths based on header names and content (minimal spacing)
897    // Note: ratatui Table adds 1 space between columns by default, so we don't add extra padding
898    let mut min_col_widths: Vec<u16> = column_names
899        .iter()
900        .map(|name| name.chars().count() as u16) // header length (no extra padding - table handles spacing)
901        .collect();
902
903    // Calculate column name width (for locked column)
904    let header_text = "Column";
905    let header_len = header_text.chars().count() as u16;
906    let max_col_name_len = results
907        .distribution_analyses
908        .iter()
909        .map(|da| da.column_name.chars().count() as u16)
910        .max()
911        .unwrap_or(header_len);
912    let locked_col_width = max_col_name_len.max(header_len).max(10);
913
914    // Scan all data to find maximum width needed for each column (excluding Column)
915    for dist_analysis in &results.distribution_analyses {
916        // Outlier count with percentage
917        let outlier_text = if dist_analysis.outliers.total_count > 0 {
918            format!(
919                "{} ({:.1}%)",
920                dist_analysis.outliers.total_count, dist_analysis.outliers.percentage
921            )
922        } else {
923            "0 (0.0%)".to_string()
924        };
925
926        // Shapiro-Wilk statistic and p-value formatting
927        let sw_stat_text = dist_analysis
928            .characteristics
929            .shapiro_wilk_stat
930            .map(|s| format!("{:.3}", s))
931            .unwrap_or_else(|| "N/A".to_string());
932        let sw_pvalue_text = dist_analysis
933            .characteristics
934            .shapiro_wilk_pvalue
935            .map(format_pvalue)
936            .unwrap_or_else(|| "N/A".to_string());
937
938        let pvalue_text = verdict_pvalue(dist_analysis);
939
940        // Update minimum widths based on content (skip column name)
941        let col_values = [
942            format!("{}", dist_analysis.distribution_type),
943            pvalue_text.clone(),
944            sw_stat_text.clone(),
945            sw_pvalue_text.clone(),
946            format!(
947                "{:.4}",
948                dist_analysis.characteristics.coefficient_of_variation
949            ),
950            outlier_text.clone(),
951            format_num(dist_analysis.characteristics.skewness),
952            format_num(dist_analysis.characteristics.kurtosis),
953        ];
954
955        for (idx, value) in col_values.iter().enumerate() {
956            let value_len = value.chars().count() as u16;
957            let header_len = column_names[idx].chars().count() as u16;
958            min_col_widths[idx] = min_col_widths[idx].max(value_len).max(header_len);
959        }
960    }
961
962    let column_spacing = 1u16;
963    // The rail's column comes first, then the locked names.
964    let available_width = area
965        .width
966        .saturating_sub(RAIL_WIDTH + locked_col_width)
967        .saturating_sub(column_spacing);
968    let (start_stat, end_stat) =
969        stat_window(&min_col_widths, available_width, column_spacing, columns);
970    let visible_stats: Vec<usize> = (start_stat..end_stat).collect();
971
972    if visible_stats.is_empty() {
973        return;
974    }
975
976    let mut rows = Vec::new();
977
978    let mut header_cells = vec![Cell::from("Column").style(Style::default())];
979    for &stat_idx in &visible_stats {
980        header_cells.push(Cell::from(column_names[stat_idx]).style(Style::default()));
981    }
982    let header_row_style = header_style(theme, "controls_bg", "table_header");
983    let header_row = Row::new(header_cells).style(header_row_style);
984    for dist_analysis in &results.distribution_analyses {
985        // The verdict in the colors of its p-value; no clear fit in the rejected one.
986        let type_color = match dist_analysis.distribution_type {
987            DistributionType::Unknown => theme.get("outlier_marker"),
988            DistributionType::Constant => theme.get("text_primary"),
989            _ => pvalue_style(dist_analysis.confidence, theme)
990                .fg
991                .unwrap_or_else(|| theme.get("text_primary")),
992        };
993
994        // Outlier count with percentage
995        let outlier_text = if dist_analysis.outliers.total_count > 0 {
996            format!(
997                "{} ({:.1}%)",
998                dist_analysis.outliers.total_count, dist_analysis.outliers.percentage
999            )
1000        } else {
1001            "0 (0.0%)".to_string()
1002        };
1003
1004        // Relaxed outlier color thresholds - red only for very high percentages that might indicate data errors
1005        let outlier_style = if dist_analysis.outliers.percentage > 20.0 {
1006            // Red: very high outlier percentage (>20%) - might indicate data errors
1007            Style::default().fg(theme.get("outlier_marker"))
1008        } else if dist_analysis.outliers.percentage > 5.0 {
1009            // Yellow for moderate outliers (5-20%)
1010            Style::default().fg(theme.get("distribution_skewed"))
1011        } else {
1012            // Default (white) for low outlier percentages (0-5%)
1013            Style::default()
1014        };
1015
1016        // Get skewness and kurtosis values for styling
1017        let skewness_value = dist_analysis.characteristics.skewness.abs();
1018        let kurtosis_value = dist_analysis.characteristics.kurtosis;
1019
1020        // Skewness color coding: similar to describe table
1021        let skewness_style = if skewness_value >= 3.0 {
1022            Style::default().fg(theme.get("outlier_marker"))
1023        } else if skewness_value >= 1.0 {
1024            Style::default().fg(theme.get("distribution_skewed"))
1025        } else {
1026            Style::default()
1027        };
1028
1029        // Kurtosis color coding: 3.0 is normal, high/low is notable
1030        let kurtosis_style = if (kurtosis_value - 3.0).abs() >= 3.0 {
1031            Style::default().fg(theme.get("outlier_marker"))
1032        } else if (kurtosis_value - 3.0).abs() >= 1.0 {
1033            Style::default().fg(theme.get("distribution_skewed"))
1034        } else {
1035            Style::default()
1036        };
1037
1038        let pvalue_text = verdict_pvalue(dist_analysis);
1039        let pvalue_style = pvalue_style(dist_analysis.confidence, theme);
1040
1041        // Shapiro-Wilk statistic and p-value formatting
1042        let sw_stat_text = dist_analysis
1043            .characteristics
1044            .shapiro_wilk_stat
1045            .map(|s| format!("{:.3}", s))
1046            .unwrap_or_else(|| "N/A".to_string());
1047        let sw_pvalue_text = dist_analysis
1048            .characteristics
1049            .shapiro_wilk_pvalue
1050            .map(format_pvalue)
1051            .unwrap_or_else(|| "N/A".to_string());
1052
1053        // Color coding for SW p-value: same semantics as p-value column
1054        // Green = normal (>0.05), Yellow = moderate (0.01-0.05), Red = non-normal (≤0.01)
1055        let sw_pvalue_style = dist_analysis
1056            .characteristics
1057            .shapiro_wilk_pvalue
1058            .map(|p| {
1059                if p > 0.05 {
1060                    Style::default().fg(theme.get("distribution_normal"))
1061                } else if p > 0.01 {
1062                    Style::default().fg(theme.get("distribution_skewed"))
1063                } else {
1064                    Style::default().fg(theme.get("outlier_marker"))
1065                }
1066            })
1067            .unwrap_or_default();
1068
1069        // Build row with locked column name + visible stat values
1070        // Use explicit text_primary so column names stay visible (avoids black-on-black)
1071        let mut cells = vec![
1072            Cell::from(dist_analysis.column_name.as_str())
1073                .style(Style::default().fg(theme.get("text_primary"))),
1074        ];
1075
1076        // Add visible statistic values
1077        for &stat_idx in &visible_stats {
1078            let cell = match stat_idx {
1079                0 => Cell::from(format!("{}", dist_analysis.distribution_type))
1080                    .style(Style::default().fg(type_color)),
1081                1 => Cell::from(pvalue_text.clone()).style(pvalue_style),
1082                2 => Cell::from(sw_stat_text.clone()),
1083                3 => Cell::from(sw_pvalue_text.clone()).style(sw_pvalue_style),
1084                4 => Cell::from(format!(
1085                    "{:.4}",
1086                    dist_analysis.characteristics.coefficient_of_variation
1087                ))
1088                .style(
1089                    if dist_analysis.characteristics.coefficient_of_variation > 1.0 {
1090                        Style::default().fg(theme.get("distribution_skewed")) // High variability
1091                    } else {
1092                        Style::default()
1093                    },
1094                ),
1095                5 => Cell::from(outlier_text.clone()).style(outlier_style),
1096                6 => Cell::from(format_num(dist_analysis.characteristics.skewness))
1097                    .style(skewness_style),
1098                7 => Cell::from(format_num(dist_analysis.characteristics.kurtosis))
1099                    .style(kurtosis_style),
1100                _ => Cell::from(""),
1101            };
1102            cells.push(cell);
1103        }
1104
1105        rows.push(Row::new(cells));
1106    }
1107
1108    let mut constraints = vec![Constraint::Length(locked_col_width)];
1109    for &stat_idx in &visible_stats {
1110        constraints.push(Constraint::Length(min_col_widths[stat_idx]));
1111    }
1112
1113    if visible_stats.len() == num_stats && constraints.len() > 1 {
1114        let last_idx = constraints.len() - 1;
1115        constraints[last_idx] = Constraint::Fill(1);
1116    }
1117
1118    let table = Table::new(rows, constraints)
1119        .header(header_row)
1120        .row_highlight_style(cursor_style(focused, theme))
1121        .highlight_symbol(cursor_rail(focused, theme))
1122        .highlight_spacing(HighlightSpacing::Always);
1123
1124    StatefulWidget::render(table, area, buf, table_state);
1125    draw_scroll_marks(
1126        area,
1127        buf,
1128        RAIL_WIDTH + locked_col_width,
1129        (start_stat, end_stat, num_stats),
1130        theme,
1131    );
1132}
1133
1134/// The column the cursor's rail sits in, kept whether or not the table has focus
1135/// so focus arriving moves nothing.
1136const RAIL_WIDTH: u16 = 1;
1137
1138/// The rail beside the row the cursor is on: in the accent while the table has
1139/// focus, dimmed while the tool list has it, so the row stays marked without
1140/// claiming the accent. The tint alone vanishes on a 16-color terminal whose
1141/// black is the background.
1142fn cursor_rail(focused: bool, theme: &Theme) -> Span<'static> {
1143    Span::styled(crate::glyphs::get().rail, rail_style(focused, theme))
1144}
1145
1146/// The rail's color: the accent with focus, dimmed without.
1147pub(crate) fn rail_style(focused: bool, theme: &Theme) -> Style {
1148    Style::default().fg(theme.get(if focused { "accent" } else { "dimmed" }))
1149}
1150
1151/// The row the cursor is on: the tint while the table has focus; without it,
1152/// only the dimmed rail marks it.
1153fn cursor_style(focused: bool, theme: &Theme) -> Style {
1154    if focused {
1155        theme.highlight_style()
1156    } else {
1157        Style::default()
1158    }
1159}
1160
1161/// The mark that counts statistics hidden to the right, as the data table's does.
1162fn more_mark(hidden: usize) -> String {
1163    format!(" +{hidden} {}", crate::glyphs::get().arrow_right)
1164}
1165
1166/// Which statistics fit beside the locked name column, as `start..end`, from the
1167/// scroll's offset. Sets the scroll's `max` to the first start that brings the
1168/// last statistic into view, and clamps the offset to it, so a key press past the
1169/// end does nothing and the first press back always moves. A window that leaves
1170/// statistics out to the right keeps room for the mark that counts them.
1171fn stat_window(
1172    widths: &[u16],
1173    available: u16,
1174    spacing: u16,
1175    columns: &mut ColumnScroll,
1176) -> (usize, usize) {
1177    let n = widths.len();
1178    if n == 0 {
1179        *columns = ColumnScroll::default();
1180        return (0, 0);
1181    }
1182    let fits = |from: usize, room: u16| {
1183        let mut used = 0u16;
1184        let mut count = 0usize;
1185        for width in &widths[from..] {
1186            let needed = width + if count > 0 { spacing } else { 0 };
1187            if used + needed > room {
1188                break;
1189            }
1190            used += needed;
1191            count += 1;
1192        }
1193        count.max(1)
1194    };
1195    let mark = crate::glyphs::display_width(&more_mark(n)) as u16;
1196    let shown = |from: usize| {
1197        let all = fits(from, available);
1198        if from + all >= n {
1199            all
1200        } else {
1201            fits(from, available.saturating_sub(mark))
1202        }
1203    };
1204    let max = (0..n)
1205        .find(|&from| from + shown(from) >= n)
1206        .unwrap_or(n - 1);
1207    columns.max = max;
1208    columns.offset = columns.offset.min(max);
1209    let start = columns.offset;
1210    (start, (start + shown(start)).min(n))
1211}
1212
1213/// Say that statistics are out of view: an arrow at the end of the locked
1214/// column's header when some are to the left, and the count at the right edge of
1215/// the header when some are to the right.
1216fn draw_scroll_marks(
1217    area: Rect,
1218    buf: &mut Buffer,
1219    locked_width: u16,
1220    (start, end, total): (usize, usize, usize),
1221    theme: &Theme,
1222) {
1223    if area.height == 0 || area.width == 0 {
1224        return;
1225    }
1226    let style = header_style(theme, "controls_bg", "accent").add_modifier(Modifier::BOLD);
1227    if start > 0 && locked_width > 0 && locked_width <= area.width {
1228        Paragraph::new(crate::glyphs::get().arrow_left)
1229            .style(style)
1230            .render(
1231                Rect {
1232                    x: area.x + locked_width - 1,
1233                    width: 1,
1234                    height: 1,
1235                    ..area
1236                },
1237                buf,
1238            );
1239    }
1240    if end < total {
1241        let mark = more_mark(total - end);
1242        let width = crate::glyphs::display_width(&mark) as u16;
1243        if width <= area.width {
1244            Paragraph::new(mark).style(style).render(
1245                Rect {
1246                    x: area.x + area.width - width,
1247                    width,
1248                    height: 1,
1249                    ..area
1250                },
1251                buf,
1252            );
1253        }
1254    }
1255}
1256
1257/// The matrix's cell cursor, and whether the matrix has the focus.
1258struct MatrixCursor {
1259    cell: Option<(usize, usize)>,
1260    focused: bool,
1261}
1262
1263fn render_correlation_matrix(
1264    shown: Option<Shown>,
1265    table_state: &mut TableState,
1266    cursor: MatrixCursor,
1267    columns: &mut ColumnScroll,
1268    area: Rect,
1269    buf: &mut Buffer,
1270    theme: &Theme,
1271) {
1272    let MatrixCursor {
1273        cell: selected_cell,
1274        focused,
1275    } = cursor;
1276    let (correlation_matrix, method) = match shown {
1277        Some(Shown { matrix, method }) => (matrix, method),
1278        None => {
1279            Paragraph::new("No correlation matrix available (need at least 2 numeric columns)")
1280                .centered()
1281                .render(area, buf);
1282            return;
1283        }
1284    };
1285
1286    if method == CorrelationMethod::Spearman && correlation_matrix.rank_correlations.is_none() {
1287        Paragraph::new(SPEARMAN_TOO_MANY)
1288            .centered()
1289            .render(area, buf);
1290        return;
1291    }
1292
1293    if correlation_matrix.columns.is_empty() {
1294        Paragraph::new("No numeric columns for correlation matrix")
1295            .centered()
1296            .render(area, buf);
1297        return;
1298    }
1299
1300    let n = correlation_matrix.columns.len();
1301
1302    // Calculate column widths - ensure they're wide enough for content
1303    let row_header_width = 20u16;
1304    let cell_width = 12u16; // Wide enough for "-0.999" and most names
1305    let column_spacing = 1u16; // Table widget adds 1 space between columns
1306
1307    let available_width = area
1308        .width
1309        .saturating_sub(row_header_width)
1310        .saturating_sub(column_spacing);
1311    let widths = vec![cell_width; n];
1312    let (mut start_col, mut end_col) =
1313        stat_window(&widths, available_width, column_spacing, columns);
1314    // Scroll to the selected cell, whatever moved it: a key, or a resize.
1315    if let Some((_, col)) = selected_cell {
1316        let col = col.min(n - 1);
1317        while col < start_col || (col >= end_col && columns.offset < columns.max) {
1318            columns.offset = if col < start_col {
1319                col
1320            } else {
1321                columns.offset + 1
1322            };
1323            (start_col, end_col) = stat_window(&widths, available_width, column_spacing, columns);
1324        }
1325    }
1326    let visible_cols = end_col - start_col;
1327
1328    let (selected_row, selected_col) = selected_cell.unwrap_or((n, n));
1329
1330    let header_row_style = header_style(theme, "controls_bg", "table_header");
1331    let dim_header_style = header_style(theme, "controls_bg", "table_header");
1332
1333    let mut header_cells = vec![Cell::from("")];
1334    for j in start_col..end_col {
1335        let col_name = &correlation_matrix.columns[j];
1336        let is_selected_col = selected_cell.is_some() && j == selected_col;
1337        let cell_style = if is_selected_col {
1338            dim_header_style
1339        } else {
1340            header_row_style
1341        };
1342        header_cells.push(Cell::from(col_name.as_str()).style(cell_style));
1343    }
1344
1345    let header_row = Row::new(header_cells).style(header_row_style);
1346
1347    // Data rows - only render visible rows (handled by TableState's visible_rows)
1348    // But we render all rows and let Table widget handle vertical scrolling
1349    let mut rows = Vec::new();
1350    for (i, col_name) in correlation_matrix.columns.iter().enumerate() {
1351        // Determine if this is the selected row
1352        let is_selected_row = selected_cell.is_some() && i == selected_row;
1353
1354        // Row header cell - dim highlight if selected row
1355        let row_header_style = if is_selected_row {
1356            Style::default().bg(theme.get("surface"))
1357        } else {
1358            Style::default()
1359        };
1360        let mut cells = vec![Cell::from(col_name.as_str()).style(row_header_style)];
1361
1362        for col_idx in start_col..end_col {
1363            let correlation = correlation_matrix.coefficient(method, i, col_idx);
1364            let text_color = get_correlation_color(correlation, theme);
1365
1366            let cell_text = if i == col_idx {
1367                "1.000".to_string()
1368            } else if correlation.is_nan() {
1369                "-".to_string()
1370            } else {
1371                format_coefficient(correlation, 3)
1372            };
1373
1374            let is_selected_cell =
1375                selected_cell.is_some() && i == selected_row && col_idx == selected_col;
1376            let is_in_selected_col = selected_cell.is_some() && col_idx == selected_col;
1377
1378            let cell_style = if is_selected_cell && focused {
1379                // The cell cursor, as the table draws its own.
1380                Style::default()
1381                    .fg(text_color)
1382                    .patch(theme.cell_cursor_style())
1383            } else if is_selected_cell {
1384                // The matrix without focus: its cell stays marked, quietly.
1385                Style::default()
1386                    .fg(text_color)
1387                    .patch(theme.column_cursor_style())
1388                    .add_modifier(Modifier::BOLD | Modifier::UNDERLINED)
1389            } else if is_selected_row || is_in_selected_col {
1390                // Selected row or column: dim background with colored text
1391                Style::default().fg(text_color).bg(theme.get("surface"))
1392            } else {
1393                // Normal cell: just text color
1394                Style::default().fg(text_color)
1395            };
1396
1397            cells.push(Cell::from(cell_text).style(cell_style));
1398        }
1399
1400        let row_style = if is_selected_row {
1401            Style::default().bg(theme.get("surface"))
1402        } else {
1403            Style::default()
1404        };
1405
1406        rows.push(Row::new(cells).style(row_style));
1407    }
1408
1409    // Build constraints - fixed widths to prevent clipping
1410    let mut constraints = vec![Constraint::Length(row_header_width)];
1411    for _ in 0..visible_cols {
1412        constraints.push(Constraint::Length(cell_width));
1413    }
1414
1415    let last_idx = constraints.len().saturating_sub(1);
1416    if visible_cols == n && constraints.len() > 1 {
1417        constraints[last_idx] = Constraint::Fill(1);
1418    }
1419
1420    let table = Table::new(rows, constraints)
1421        .header(header_row)
1422        .column_spacing(column_spacing);
1423
1424    StatefulWidget::render(table, area, buf, table_state);
1425    draw_scroll_marks(area, buf, row_header_width, (start_col, end_col, n), theme);
1426}
1427
1428fn get_correlation_color(correlation: f64, theme: &Theme) -> Color {
1429    let abs_corr = correlation.abs();
1430
1431    if abs_corr < 0.05 {
1432        // No correlation (close to 0) - dimmed
1433        theme.get("dimmed")
1434    } else if abs_corr < 0.3 {
1435        // Low correlation - normal text
1436        theme.get("text_primary")
1437    } else if correlation > 0.0 {
1438        // Positive correlation - keybind hints color (UI element, not chart)
1439        theme.get("chip_key")
1440    } else {
1441        // Negative correlation - error/warning color
1442        theme.get("outlier_marker")
1443    }
1444}
1445
1446/// What the family list shows: the column's fits, the family on the plots, and
1447/// the scale the histogram is drawn in.
1448struct SelectorConfig<'a> {
1449    dist: &'a DistributionAnalysis,
1450    selected: DistributionType,
1451    histogram_scale: HistogramScale,
1452    /// Log was asked for on values that cannot take it, so the histogram is linear.
1453    log_scale_unavailable: bool,
1454    theme: &'a Theme,
1455    ctx: &'a RenderContext,
1456}
1457
1458/// The families to compare with, one Surface on the right of the detail: each
1459/// family and its p-value, the one on the plots on the rail, and the scale the
1460/// histogram is drawn in on the last row.
1461fn render_distribution_selector(
1462    config: SelectorConfig,
1463    selector_state: &mut TableState,
1464    area: Rect,
1465    buf: &mut Buffer,
1466) {
1467    let SelectorConfig {
1468        dist,
1469        selected: selected_dist,
1470        histogram_scale,
1471        log_scale_unavailable,
1472        theme,
1473        ctx,
1474    } = config;
1475    // Tested families by p-value, then the ones that do not apply; the same order
1476    // the modal's ↑↓ walks.
1477    let distribution_scores: Vec<(DistributionType, Option<&FitOutcome>)> =
1478        crate::distribution_fit::listing_order(&dist.fits)
1479            .into_iter()
1480            .map(|family| (family, dist.fit(family)))
1481            .collect();
1482
1483    let selected_pos = distribution_scores
1484        .iter()
1485        .position(|(family, _)| *family == selected_dist)
1486        .unwrap_or(0);
1487    // Trust the cursor while it is on the list; place it only when it is unset or
1488    // has fallen off the end.
1489    match selector_state.selected() {
1490        Some(idx) if idx < distribution_scores.len() => {}
1491        _ => selector_state.select(Some(selected_pos)),
1492    }
1493    let selected = selector_state.selected().unwrap_or(0);
1494
1495    let content = Surface::new("Distribution").render(area, buf, ctx);
1496    if content.height < 3 || content.width < 8 {
1497        return;
1498    }
1499    let g = crate::glyphs::get();
1500    // The p-value column is as wide as its widest value, "<0.005" or "n/a".
1501    const PVALUE_WIDTH: u16 = 7;
1502    let name_width = content.width.saturating_sub(1 + PVALUE_WIDTH);
1503    let line = |rail: &str, name: &str, pvalue: &str| {
1504        format!(
1505            "{rail}{name:<w$}{pvalue:>p$}",
1506            name = crate::render::loading_view::truncate(name, name_width as usize),
1507            w = name_width as usize,
1508            p = PVALUE_WIDTH as usize,
1509        )
1510    };
1511    let row = |y: u16| Rect {
1512        y,
1513        height: 1,
1514        ..content
1515    };
1516
1517    Paragraph::new(line(" ", "Name", "P-value"))
1518        .style(Style::default().fg(ctx.text_secondary))
1519        .render(row(content.y), buf);
1520
1521    // The last row is the scale; the list scrolls in what is between, and counts
1522    // what it cannot show rather than cutting a family in half.
1523    let scale_y = content.y + content.height - 1;
1524    let list_height = (content.height - 2) as usize;
1525    let total = distribution_scores.len();
1526    let (offset, shown) = list_window(selected, total, list_height);
1527    let below = total - offset - shown;
1528    if below > 0 && shown < list_height {
1529        Paragraph::new(format!(" {} {below} more", g.ellipsis))
1530            .style(Style::default().fg(ctx.dimmed))
1531            .render(row(content.y + 1 + shown as u16), buf);
1532    }
1533    for (i, (family, outcome)) in distribution_scores
1534        .iter()
1535        .enumerate()
1536        .skip(offset)
1537        .take(shown)
1538    {
1539        let y = content.y + 1 + (i - offset) as u16;
1540        // A family that does not apply has no p-value, and says so rather than
1541        // ranking a placeholder.
1542        let (p_text, p_style) = match outcome.and_then(|outcome| outcome.test()) {
1543            Some(test) => (format_fit_pvalue(test), pvalue_style(test.p_value, theme)),
1544            None => ("n/a".to_string(), Style::default().fg(ctx.dimmed)),
1545        };
1546        let is_cursor = i == selected;
1547        let name = crate::render::loading_view::truncate(&family.to_string(), name_width as usize);
1548        let spans = vec![
1549            Span::styled(
1550                if is_cursor { g.rail } else { " " },
1551                Style::default().fg(ctx.accent),
1552            ),
1553            Span::styled(
1554                format!("{name:<w$}", w = name_width as usize),
1555                Style::default().fg(ctx.text_primary),
1556            ),
1557            Span::styled(format!("{p_text:>p$}", p = PVALUE_WIDTH as usize), p_style),
1558        ];
1559        let mut paragraph = Paragraph::new(Line::from(spans));
1560        if is_cursor {
1561            paragraph = paragraph.style(ctx.highlight_style());
1562        }
1563        paragraph.render(row(y), buf);
1564    }
1565
1566    // Log asked for on values that cannot take it falls back to linear, in the
1567    // warning color so the fallback is not mistaken for the choice.
1568    let (scale, scale_style) = match (histogram_scale, log_scale_unavailable) {
1569        (_, true) => ("Linear", Style::default().fg(ctx.warning)),
1570        (HistogramScale::Linear, false) => ("Linear", Style::default().fg(ctx.text_primary)),
1571        (HistogramScale::Log, false) => ("Log", Style::default().fg(ctx.text_primary)),
1572    };
1573    if scale_y > content.y + 1 {
1574        Paragraph::new(Line::from(vec![
1575            Span::styled(" Scale: ", Style::default().fg(ctx.label)),
1576            Span::styled(scale, scale_style),
1577        ]))
1578        .render(row(scale_y), buf);
1579    }
1580}
1581
1582/// What the Distribution detail's two plots draw: the Q-Q plot and the histogram.
1583#[derive(Clone, Copy)]
1584struct DistributionPlotConfig<'a> {
1585    dist: &'a DistributionAnalysis,
1586    dist_type: DistributionType,
1587    area: Rect,
1588    shared_y_axis_label_width: u16,
1589    theme: &'a Theme,
1590    unified_x_range: Option<(f64, f64)>,
1591    histogram_scale: HistogramScale,
1592    glyphs: &'a crate::glyphs::Glyphs,
1593    /// The column's numbers, on every value axis.
1594    values: &'a AxisNumbers,
1595    counts: &'a AxisNumbers,
1596}
1597
1598/// The family list's least width: the frame, the rail, "Exponential" and a p-value.
1599const SELECTOR_WIDTH: u16 = 24;
1600
1601/// Which of `total` items a list of `rows` shows around the cursor, as the
1602/// first and how many. While some are out of view below, the last row is kept
1603/// to count them, so the cursor never sits on it.
1604fn list_window(selected: usize, total: usize, rows: usize) -> (usize, usize) {
1605    if total <= rows || rows == 0 {
1606        return (0, total.min(rows));
1607    }
1608    let room = rows.saturating_sub(1).max(1);
1609    let offset = selected.saturating_sub(room - 1);
1610    if offset + rows >= total {
1611        (total - rows, rows)
1612    } else {
1613        (offset, room)
1614    }
1615}
1616
1617/// The tool list's width beside a result: a third of the screen, at most 32.
1618pub(crate) fn sidebar_width(width: u16) -> u16 {
1619    32u16.min(width / 3)
1620}
1621
1622/// Where a tool's result goes: under the one-line header, left of the tool list.
1623/// The Sample form fills it before a tool's first run.
1624pub(crate) fn main_pane(area: Rect) -> Rect {
1625    Rect {
1626        y: area.y + 1,
1627        height: area.height.saturating_sub(1),
1628        width: area.width.saturating_sub(sidebar_width(area.width)),
1629        ..area
1630    }
1631}
1632
1633/// The Analysis Tools list, the same beside every tool: one Surface, the
1634/// cursor carrying the rail and the tint while the list has focus, and the tool
1635/// on screen carrying the accent.
1636pub(crate) fn render_sidebar(
1637    area: Rect,
1638    buf: &mut Buffer,
1639    sidebar_state: &mut TableState,
1640    selected_tool: Option<AnalysisTool>,
1641    focus: AnalysisFocus,
1642    theme: &Theme,
1643) {
1644    let tools = [
1645        ("Describe", AnalysisTool::Describe),
1646        ("Distribution Analysis", AnalysisTool::DistributionAnalysis),
1647        ("Correlation Matrix", AnalysisTool::CorrelationMatrix),
1648        ("Data Quality", AnalysisTool::DataQuality),
1649    ];
1650    // Built here rather than passed: Data Quality draws this list too, from its
1651    // theme.
1652    let ctx =
1653        RenderContext::from_theme_and_config(theme, 0, false, NumberFormatSettings::default());
1654    let content = Surface::new("Analysis Tools").render(area, buf, &ctx);
1655    let g = crate::glyphs::get();
1656    let list_focused = focus == AnalysisFocus::Sidebar;
1657    for (idx, (name, tool)) in tools.iter().enumerate().take(content.height as usize) {
1658        let is_cursor = list_focused && sidebar_state.selected() == Some(idx);
1659        let on_screen = selected_tool == Some(*tool);
1660        // The tool on screen is bold; the accent is the cursor's alone.
1661        let name_style = if on_screen {
1662            Style::default()
1663                .fg(ctx.text_primary)
1664                .add_modifier(Modifier::BOLD)
1665        } else {
1666            Style::default().fg(ctx.text_primary)
1667        };
1668        // The tool on screen keeps a dimmed rail while its pane has the focus.
1669        let rail = if is_cursor || (on_screen && !list_focused) {
1670            g.rail
1671        } else {
1672            " "
1673        };
1674        let mut line = Paragraph::new(Line::from(vec![
1675            Span::styled(rail, rail_style(is_cursor, theme)),
1676            // Cut with a mark on a narrow screen, never silently.
1677            Span::styled(
1678                crate::render::loading_view::truncate(
1679                    name,
1680                    content.width.saturating_sub(1) as usize,
1681                ),
1682                name_style,
1683            ),
1684        ]));
1685        if is_cursor {
1686            line = line.style(ctx.highlight_style());
1687        }
1688        let row = Rect {
1689            y: content.y + idx as u16,
1690            height: 1,
1691            ..content
1692        };
1693        line.render(row, buf);
1694        crate::pointer::record(row, crate::pointer::Hit::Tool(idx));
1695    }
1696}
1697
1698fn render_distribution_histogram(config: DistributionPlotConfig, buf: &mut Buffer) {
1699    // Use BarChart widget to show histogram comparing data vs theoretical distribution
1700    // Use fixed-width bins that span both data range and theoretical distribution range
1701    let DistributionPlotConfig {
1702        dist,
1703        dist_type,
1704        area,
1705        shared_y_axis_label_width,
1706        theme,
1707        unified_x_range,
1708        histogram_scale,
1709        glyphs: g,
1710        values,
1711        counts,
1712    } = config;
1713    let sorted_data = &dist.sorted_sample_values;
1714
1715    if sorted_data.is_empty() || sorted_data.len() < 3 {
1716        Paragraph::new("Insufficient data for histogram")
1717            .centered()
1718            .render(area, buf);
1719        return;
1720    }
1721
1722    let n = sorted_data.len();
1723
1724    // Determine bin range: use percentile-based robust range (P1-P99) for all distributions
1725    // This is a best practice that gives more visual space to the bulk of data while
1726    // still showing outliers in edge bins. Matches professional tools like Observable Canvases.
1727    let data_min = sorted_data[0];
1728    let data_max = sorted_data[n - 1];
1729    let data_range = data_max - data_min;
1730
1731    if data_range <= 0.0 {
1732        // Constant data: all values are the same
1733        Paragraph::new("Constant data: all values are identical")
1734            .centered()
1735            .render(area, buf);
1736        return;
1737    }
1738
1739    // Use unified X-axis range (strict data range, no padding or extensions)
1740    // This keeps both Q-Q plot and histogram in sync and ensures log scale works correctly
1741    let (hist_min, hist_max, hist_range) = if let Some((unified_min, unified_max)) = unified_x_range
1742    {
1743        // Use unified range directly - it's already the strict data range
1744        let range = unified_max - unified_min;
1745        (unified_min, unified_max, range)
1746    } else {
1747        // Fallback: use actual data range (shouldn't happen if unified_x_range is always provided)
1748        (data_min, data_max, data_range)
1749    };
1750
1751    // Calculate dynamic number of bins based on available width
1752    // This ensures bars fill the horizontal space and look dense at all widths
1753
1754    let y_axis_gap = 1u16; // Minimal gap between labels and plot area (needed to prevent bars from extending outside)
1755    let total_y_axis_space = shared_y_axis_label_width + y_axis_gap;
1756
1757    // Calculate available width for bars - must match Chart widget's plot area exactly
1758    // Chart widget reserves space for Y-axis labels internally, using remaining width for plot
1759    // Less the axis line itself, which the plot starts after.
1760    let available_width = area.width.saturating_sub(total_y_axis_space + 1);
1761    // One blank column between neighboring bars.
1762    let gap_width = 1u16;
1763
1764    // Target bar width: aim for 6-8 pixels per bar for good density
1765    // Calculate optimal number of bins to fill available width
1766    // Formula: available_width = num_bins * bar_width + (num_bins - 1) * gap_width
1767    // Rearranging: num_bins = (available_width + gap_width) / (bar_width + gap_width)
1768    let target_bar_width = 7.0; // Target bar width in pixels
1769    let optimal_num_bins = ((available_width as f64 + gap_width as f64)
1770        / (target_bar_width + gap_width as f64)) as usize;
1771
1772    // Clamp to reasonable bounds: minimum 5 bins, maximum 60 bins
1773    // Fewer bins for very narrow displays, more bins for wide displays
1774    // Increased max to 60 to better utilize ultrawide displays
1775    let num_bins = optimal_num_bins.clamp(5, 60);
1776
1777    // Use log-scale binning if user has selected log scale and data is positive
1778    // Log-scale binning is standard practice for power law distributions and wide dynamic ranges
1779    // Check actual data values, not histogram range (which may include padding or theoretical bounds)
1780    let all_data_positive = sorted_data.iter().all(|&v| v > 0.0);
1781    // For log scale, ensure hist_min is positive (adjust if needed)
1782    let (log_hist_min, log_hist_max) =
1783        if matches!(histogram_scale, HistogramScale::Log) && all_data_positive {
1784            // Use actual data min/max for log scale to avoid issues with padding or theoretical bounds
1785            let actual_min = sorted_data[0];
1786            let actual_max = sorted_data[sorted_data.len() - 1];
1787            // Ensure minimum is positive for log scale
1788            if actual_min > 0.0 {
1789                (actual_min, actual_max)
1790            } else {
1791                // Can't use log scale if data includes 0
1792                (hist_min, hist_max)
1793            }
1794        } else {
1795            (hist_min, hist_max)
1796        };
1797    let use_log_scale = matches!(histogram_scale, HistogramScale::Log)
1798        && all_data_positive
1799        && log_hist_min > 0.0
1800        && log_hist_max > log_hist_min;
1801
1802    let (bin_boundaries, bin_width): (Vec<f64>, f64) = if use_log_scale {
1803        // Log-scale binning: bins with equal width in log space
1804        // This ensures each bin represents roughly equal multiplicative range
1805        // Use adjusted range based on actual data values
1806        let log_min = log_hist_min.ln();
1807        let log_max = log_hist_max.ln();
1808        let log_range = log_max - log_min;
1809        let log_bin_width = log_range / num_bins as f64;
1810
1811        let boundaries: Vec<f64> = (0..=num_bins)
1812            .map(|i| {
1813                let log_value = log_min + (i as f64) * log_bin_width;
1814                log_value.exp()
1815            })
1816            .collect();
1817
1818        // For log scale, calculate average bin width for use in theoretical PDF calculations
1819        // This is approximate but needed for compatibility
1820        let log_range_linear = log_hist_max - log_hist_min;
1821        let avg_bin_width = log_range_linear / num_bins as f64;
1822        (boundaries, avg_bin_width)
1823    } else {
1824        // Linear binning for all other distributions
1825        let bin_width = hist_range / num_bins as f64;
1826        let boundaries: Vec<f64> = (0..=num_bins)
1827            .map(|i| hist_min + (i as f64) * bin_width)
1828            .collect();
1829        (boundaries, bin_width)
1830    };
1831
1832    // Count data points in each bin
1833    let mut data_bin_counts = vec![0; num_bins];
1834    for &val in sorted_data {
1835        for (i, boundaries) in bin_boundaries.windows(2).enumerate().take(num_bins) {
1836            if val >= boundaries[0]
1837                && (val < boundaries[1] || (i == num_bins - 1 && val <= boundaries[1]))
1838            {
1839                data_bin_counts[i] += 1;
1840                break;
1841            }
1842        }
1843    }
1844
1845    // Expected counts from the fit every view of this family uses, by the CDF across
1846    // each bin: exact for log-scaled and whole-number bins, where a density at the
1847    // center is not. A family that does not apply draws no overlay.
1848    let fitted = dist
1849        .fit(dist_type)
1850        .and_then(|outcome| outcome.test())
1851        .map(|test| test.fitted);
1852    let theory_probs: Vec<f64> = match &fitted {
1853        Some(fitted) => bin_boundaries
1854            .windows(2)
1855            .enumerate()
1856            .map(|(i, edges)| {
1857                let upper = if i + 1 == num_bins {
1858                    fitted.cdf(edges[1])
1859                } else {
1860                    fitted.cdf_below(edges[1])
1861                };
1862                (upper - fitted.cdf_below(edges[0])).max(0.0)
1863            })
1864            .collect(),
1865        None => vec![0.0; num_bins],
1866    };
1867
1868    // Convert probabilities to expected counts
1869    let theory_bin_counts: Vec<f64> = theory_probs.iter().map(|&prob| prob * n as f64).collect();
1870
1871    // Normalize values for display (find the maximum for scaling)
1872    let max_data = data_bin_counts.iter().cloned().fold(0, usize::max);
1873    let max_theory = theory_bin_counts.iter().cloned().fold(0.0, f64::max);
1874    // Even, so the middle label is a whole count.
1875    let global_max = (max_data.max(max_theory.ceil() as usize).max(1) as f64 / 2.0).ceil() * 2.0;
1876
1877    // Use the shared label width calculated in the caller
1878    // This ensures both histogram and Q-Q plot use the same padding for alignment
1879    let y_axis_label_width = shared_y_axis_label_width;
1880
1881    // On Log the bins are equal in log space, and so is the x axis: a position is the
1882    // log of the value it stands for, and a bin's center is its geometric middle.
1883    let position = |x: f64| if use_log_scale { x.ln() } else { x };
1884    let bin_centers: Vec<f64> = (0..num_bins)
1885        .map(|i| {
1886            let (lo, hi) = (bin_boundaries[i], bin_boundaries[i + 1]);
1887            if use_log_scale {
1888                (lo * hi).sqrt()
1889            } else {
1890                (lo + hi) / 2.0
1891            }
1892        })
1893        .collect();
1894
1895    // Each bin's bar on the 0-100 scale the curve and the count labels use. No value
1896    // or label: the axes say what a bar's height and place mean.
1897    let data_bars: Vec<Bar> = data_bin_counts
1898        .iter()
1899        .map(|&data_count| {
1900            let data_height = if global_max > 0.0 {
1901                ((data_count as f64 / global_max) * 100.0) as u64
1902            } else {
1903                0
1904            };
1905            Bar::default()
1906                .value(data_height)
1907                .text_value(String::new())
1908                .style(Style::default().fg(theme.get("chart_1")))
1909        })
1910        .collect();
1911
1912    // The labels are padded to the width shared with the Q-Q plot, so both plots
1913    // start in the same column. The bars stand on a 0-100 scale; their labels read
1914    // counts.
1915    let label_width = y_axis_label_width as usize;
1916    let count_axis = AxisSpec::numbers_as([0.0, 100.0], counts, "Counts", move |v| {
1917        v * global_max / 100.0
1918    });
1919    let x_axis = if use_log_scale {
1920        AxisSpec::numbers_as([log_hist_min.ln(), log_hist_max.ln()], values, "", f64::exp)
1921    } else {
1922        AxisSpec::numbers([hist_min, hist_max], values, "")
1923    };
1924    let axes = distribution_axes(theme, x_axis, count_axis.padded(label_width), g.plot.line);
1925    let block = distribution_block(format!("Histogram vs {dist_type}"));
1926    let chart_area = block.inner(area);
1927
1928    // Exactly the overlay's plot area: bar `i` starts where bin `i` does. Shifting the
1929    // bars right to meet the overlay put the first bin's bar over the second bin and
1930    // drew the last one past the axis, onto whatever sits beside the chart.
1931    let bar_plot_area = axes.frame(chart_area).graph;
1932
1933    // Bin `i` takes the plot columns its values map to, as the labels and the curve
1934    // map them, and its bar fills them less a gap before the next bar. Bars of one
1935    // shared width stopped short of the right end by up to a bar, leaving each bar
1936    // left of the values it counts.
1937    let plot_width = bar_plot_area.width as usize;
1938    let bin_edge = |i: usize| ((2 * i * plot_width + num_bins) / (2 * num_bins)) as u16;
1939    let bar_charts: Vec<(Rect, BarChart)> = data_bars
1940        .into_iter()
1941        .enumerate()
1942        .filter_map(|(i, bar)| {
1943            let (start, end) = (bin_edge(i), bin_edge(i + 1));
1944            let span = end - start;
1945            let width = if i + 1 < num_bins && span > gap_width {
1946                span - gap_width
1947            } else {
1948                span
1949            };
1950            let rect = Rect {
1951                x: bar_plot_area.x + start,
1952                width,
1953                ..bar_plot_area
1954            };
1955            let chart = BarChart::default()
1956                .data(BarGroup::default().bars(&[bar]))
1957                // The same 0-100 scale the curve and the labels use; left to itself the
1958                // chart scales to its tallest bar and the curve no longer measures
1959                // against the bars.
1960                .max(100)
1961                .bar_set(g.plot.column_set())
1962                .bar_width(width)
1963                .bar_gap(0);
1964            (width > 0).then_some((rect, chart))
1965        })
1966        .collect();
1967
1968    // The fit's expected counts, drawn behind the bars; sampled densely enough that
1969    // braille renders it as a line.
1970    let num_samples = (available_width as usize * 15).clamp(1500, 10000);
1971
1972    let height = |count: f64| {
1973        if global_max > 0.0 {
1974            count / global_max * 100.0
1975        } else {
1976            0.0
1977        }
1978    };
1979    let theory_points: Vec<(f64, f64)> = match &fitted {
1980        // A continuous family on linear bins is drawn as its density, scaled to a bin's
1981        // count: a smooth curve rather than a staircase.
1982        Some(fitted) if !fitted.discrete() && !use_log_scale && hist_range > 0.0 => (0
1983            ..num_samples)
1984            .map(|i| {
1985                let x = hist_min + i as f64 / (num_samples - 1) as f64 * hist_range;
1986                (x, height(fitted.density(x) * bin_width * n as f64))
1987            })
1988            .filter(|(_, y)| y.is_finite())
1989            .collect(),
1990        // Counts, and log-scaled bins, by each bin's expected count at its center.
1991        Some(_) => bin_centers
1992            .iter()
1993            .zip(&theory_bin_counts)
1994            .map(|(center, count)| (position(*center), height(*count)))
1995            .collect(),
1996        None => Vec::new(),
1997    };
1998
1999    // Dense points in the line mark read as a continuous curve.
2000    let marker = g.plot.line;
2001
2002    let theory_dataset = Dataset::default()
2003        .name("") // Empty name to prevent legend from appearing
2004        .marker(marker)
2005        .graph_type(GraphType::Scatter)
2006        .style(Style::default().fg(theme.get("dimmed")))
2007        .data(&theory_points);
2008
2009    let theory_chart = Chart::new(vec![theory_dataset])
2010        .hidden_legend_constraints((Constraint::Length(0), Constraint::Length(0)));
2011
2012    // Render Chart overlay to full area (no borders)
2013    // Chart widget will automatically handle its own inner layout for x-axis labels
2014    // The bars, then the chart laid over them from a buffer of its own: its axes,
2015    // labels and curve, except where the curve crosses a bar. Drawn straight over the
2016    // bars, each braille cell of the curve replaced a block and cut a notch in the bar;
2017    // drawn under them, the bar chart's blank cells erased the curve and the axis title.
2018    for (rect, chart) in bar_charts {
2019        chart.render(rect, buf);
2020    }
2021    let mut overlay = Buffer::empty(area);
2022    block.render(area, &mut overlay);
2023    axes.render(theory_chart, chart_area, &mut overlay, g);
2024    let is_bar = |symbol: &str| g.plot.column_eighths.contains(&symbol);
2025    for y in area.top()..area.bottom() {
2026        for x in area.left()..area.right() {
2027            let cell = &overlay[(x, y)];
2028            let symbol = cell.symbol();
2029            if symbol == " " || (PlotMarks::is_mark(marker, symbol) && is_bar(buf[(x, y)].symbol()))
2030            {
2031                continue;
2032            }
2033            buf[(x, y)] = cell.clone();
2034        }
2035    }
2036}
2037
2038fn render_qq_plot(config: DistributionPlotConfig, buf: &mut Buffer) {
2039    let DistributionPlotConfig {
2040        dist,
2041        dist_type,
2042        area,
2043        shared_y_axis_label_width,
2044        theme,
2045        unified_x_range,
2046        glyphs: g,
2047        values,
2048        ..
2049    } = config;
2050    // Use Chart widget for Q-Q plot: Data quantiles vs Theoretical quantiles
2051    // Use sorted_sample_values and position-based quantiles (not just 5 percentiles)
2052    let sorted_data = &dist.sorted_sample_values;
2053
2054    if sorted_data.is_empty() || sorted_data.len() < 3 {
2055        Paragraph::new("Insufficient data for Q-Q plot (need at least 3 points)")
2056            .centered()
2057            .render(area, buf);
2058        return;
2059    }
2060
2061    // The fit's quantiles at each plotting position, computed with the fit.
2062    let Some(theoretical) = dist.qq(dist_type) else {
2063        let reason = match dist.fit(dist_type) {
2064            Some(FitOutcome::NotApplicable(reason)) => format!("{dist_type} {reason}"),
2065            _ => format!("{dist_type} was not fitted"),
2066        };
2067        Paragraph::new(reason)
2068            .centered()
2069            .wrap(ratatui::widgets::Wrap { trim: true })
2070            .render(area, buf);
2071        return;
2072    };
2073    let qq_data: Vec<(f64, f64)> = theoretical
2074        .iter()
2075        .zip(sorted_data)
2076        .map(|(t, d)| (*t, *d))
2077        .filter(|(t, _)| t.is_finite())
2078        .collect();
2079    if qq_data.len() < 3 {
2080        Paragraph::new("Insufficient data for Q-Q plot (need at least 3 points)")
2081            .centered()
2082            .render(area, buf);
2083        return;
2084    }
2085    let n = qq_data.len();
2086
2087    // Find data ranges for both axes
2088    // X-axis (Theoretical): calculated from probability percentiles via inverse CDF
2089    // Y-axis (Empirical): raw sorted sample data (preserve all values, even if "impossible")
2090    let theory_min = qq_data
2091        .iter()
2092        .map(|(t, _)| *t)
2093        .fold(f64::INFINITY, f64::min);
2094    let theory_max = qq_data
2095        .iter()
2096        .map(|(t, _)| *t)
2097        .fold(f64::NEG_INFINITY, f64::max);
2098    let theory_range = theory_max - theory_min;
2099
2100    let data_min = qq_data
2101        .iter()
2102        .map(|(_, d)| *d)
2103        .fold(f64::INFINITY, f64::min);
2104    let data_max = qq_data
2105        .iter()
2106        .map(|(_, d)| *d)
2107        .fold(f64::NEG_INFINITY, f64::max);
2108    let data_range = data_max - data_min;
2109
2110    // Only require data_range > 0 (allow plotting even if theoretical range is small/zero)
2111    // This handles cases where distribution doesn't match (e.g., negative data vs strictly positive distribution)
2112    if data_range <= 0.0 {
2113        Paragraph::new("Insufficient data range for Q-Q plot")
2114            .centered()
2115            .render(area, buf);
2116        return;
2117    }
2118
2119    // Use unified X-axis range if provided for visual alignment with histogram
2120    // Otherwise, handle case where all theoretical quantiles are the same (theory_range = 0)
2121    let (theory_min_plot, theory_max_plot) =
2122        if let Some((unified_min, unified_max)) = unified_x_range {
2123            // Use unified range to align with histogram
2124            (unified_min, unified_max)
2125        } else if theory_range <= 0.0 || !theory_min.is_finite() || !theory_max.is_finite() {
2126            // Fallback: use data range (no padding)
2127            (data_min, data_max)
2128        } else {
2129            // Use theoretical range, but clamp to data range to keep charts in sync
2130            (theory_min.max(data_min), theory_max.min(data_max))
2131        };
2132
2133    // Create robust reference line through Q1 and Q3 quartiles
2134    // This works even when domains don't overlap (e.g., negative data vs positive distribution)
2135    let q1_idx = (n as f64 * 0.25).floor() as usize;
2136    let q3_idx = (n as f64 * 0.75).floor() as usize;
2137    let q1_idx = q1_idx.min(n - 1);
2138    let q3_idx = q3_idx.min(n - 1);
2139
2140    let (theory_q1, data_q1) = if q1_idx < qq_data.len() {
2141        qq_data[q1_idx]
2142    } else {
2143        qq_data[0]
2144    };
2145    let (theory_q3, data_q3) = if q3_idx < qq_data.len() {
2146        qq_data[q3_idx]
2147    } else {
2148        qq_data[qq_data.len() - 1]
2149    };
2150
2151    // Calculate robust reference line through (theory_q1, data_q1) and (theory_q3, data_q3)
2152    // This works even when domains don't overlap (e.g., negative data vs positive distribution)
2153    let theory_diff = theory_q3 - theory_q1;
2154    let reference_line = if theory_diff.abs() > 1e-10 {
2155        // Normal case: calculate slope and extend line to cover plot range (no padding)
2156        let slope = (data_q3 - data_q1) / theory_diff;
2157        let x_start = theory_min_plot;
2158        let x_end = theory_max_plot;
2159        let y_start = slope * (x_start - theory_q1) + data_q1;
2160        let y_end = slope * (x_end - theory_q1) + data_q1;
2161        vec![(x_start, y_start), (x_end, y_end)]
2162    } else {
2163        // Degenerate case: all theoretical quantiles are the same (theory_range ≈ 0)
2164        // Use horizontal line through data median to show the mismatch (no padding)
2165        let y_median = (data_q1 + data_q3) / 2.0;
2166        vec![(theory_min_plot, y_median), (theory_max_plot, y_median)]
2167    };
2168
2169    // Create datasets
2170    // Use appropriate marker based on point density
2171    let marker = if qq_data.len() > 100 {
2172        g.plot.line
2173    } else {
2174        g.plot.point
2175    };
2176
2177    let datasets = vec![
2178        // Diagonal reference line
2179        Dataset::default()
2180            .name("") // Empty name to hide from legend
2181            .marker(marker)
2182            .style(Style::default().fg(theme.get("dimmed")))
2183            .graph_type(GraphType::Line)
2184            .data(&reference_line),
2185        // Q-Q plot data points
2186        Dataset::default()
2187            .name("") // Empty name to hide from legend
2188            .marker(marker)
2189            .style(Style::default().fg(theme.get("chart_1")))
2190            .graph_type(GraphType::Scatter)
2191            .data(&qq_data),
2192    ];
2193
2194    // Padded to the width shared with the histogram, so both plots start in the same
2195    // column.
2196    let label_width = shared_y_axis_label_width as usize;
2197    let axes = distribution_axes(
2198        theme,
2199        AxisSpec::numbers(
2200            [theory_min_plot, theory_max_plot],
2201            values,
2202            "Theoretical Values",
2203        ),
2204        AxisSpec::numbers([data_min, data_max], values, "Data Values").padded(label_width),
2205        marker,
2206    );
2207    let block = distribution_block(format!("Q-Q Plot vs {dist_type}"));
2208    let chart_area = block.inner(area);
2209    block.render(area, buf);
2210    let chart = Chart::new(datasets)
2211        .hidden_legend_constraints((Constraint::Length(0), Constraint::Length(0)));
2212    axes.render(chart, chart_area, buf, g);
2213}
2214
2215/// A Distribution plot's frame: its title centered above, a cell of air on the left.
2216fn distribution_block<'a>(title: String) -> Block<'a> {
2217    Block::default()
2218        .title(title)
2219        .title_style(ratatui::style::Style::reset())
2220        .title_alignment(ratatui::layout::Alignment::Center)
2221        .padding(ratatui::widgets::Padding::left(1))
2222}
2223
2224fn distribution_axes<'a>(
2225    theme: &Theme,
2226    x: AxisSpec<'a>,
2227    y: AxisSpec<'a>,
2228    marker: ratatui::symbols::Marker,
2229) -> PlotAxes<'a> {
2230    let secondary = Style::default().fg(theme.get("text_secondary"));
2231    PlotAxes {
2232        titles: Style::default(),
2233        ..PlotAxes::new(x, y, secondary, marker)
2234    }
2235}
2236
2237/// The detail's key figures, label and value: the fit found, Shapiro-Francia,
2238/// skew, kurtosis, median, mean, std and CV.
2239fn condensed_statistics(dist: &DistributionAnalysis) -> Vec<(&'static str, String)> {
2240    let chars = &dist.characteristics;
2241    // What the values were found to fit, before anything about the family the plots
2242    // compare them with: choosing a family below is a comparison, not a finding.
2243    let mut figures = vec![(
2244        "Fit",
2245        match dist.distribution_type {
2246            DistributionType::Unknown | DistributionType::Constant => {
2247                dist.distribution_type.to_string()
2248            }
2249            family => format!("{family} (p {})", verdict_pvalue(dist)),
2250        },
2251    )];
2252    if let (Some(sw_stat), Some(sw_p)) = (chars.shapiro_wilk_stat, chars.shapiro_wilk_pvalue) {
2253        figures.push((
2254            "SF",
2255            if sw_p < 0.001 {
2256                format!("{sw_stat:.3} (p<0.001)")
2257            } else {
2258                format!("{sw_stat:.3} (p={sw_p:.3})")
2259            },
2260        ));
2261    }
2262    figures.extend([
2263        ("Skew", format!("{:.2}", chars.skewness)),
2264        ("Kurt", format!("{:.2}", chars.kurtosis)),
2265        ("Median", format!("{:.2}", dist.percentiles.p50)),
2266        ("Mean", format!("{:.2}", chars.mean)),
2267        ("Std", format!("{:.2}", chars.std_dev)),
2268        ("CV", format!("{:.3}", chars.coefficient_of_variation)),
2269    ]);
2270    figures
2271}
2272
2273/// The figures packed into lines of `width`, never splitting a label from its
2274/// value, so a narrow terminal wraps them rather than cutting the last ones off.
2275fn condensed_statistics_lines(
2276    figures: &[(&'static str, String)],
2277    width: u16,
2278    theme: &Theme,
2279) -> Vec<Line<'static>> {
2280    let style = Style::default().fg(theme.get("text_primary"));
2281    let mut lines = Vec::new();
2282    let mut spans = Vec::new();
2283    let mut used = 0usize;
2284    for (label, value) in figures {
2285        let figure = format!("{label}: {value}");
2286        let w = crate::glyphs::display_width(&figure);
2287        if used > 0 && used + 1 + w > width as usize {
2288            lines.push(Line::from(std::mem::take(&mut spans)));
2289            used = 0;
2290        }
2291        if used > 0 {
2292            spans.push(Span::styled(" ", style));
2293            used += 1;
2294        }
2295        spans.push(Span::styled(figure, style));
2296        used += w;
2297    }
2298    if !spans.is_empty() {
2299        lines.push(Line::from(spans));
2300    }
2301    lines
2302}
2303
2304#[cfg(test)]
2305mod tests {
2306    use super::*;
2307    use crate::numfmt::NumberFormat;
2308
2309    fn settings(preset: &str, enabled: bool) -> NumberFormatSettings {
2310        NumberFormatSettings {
2311            format: NumberFormat::preset(preset).unwrap(),
2312            enabled,
2313            exclude: Vec::new(),
2314            align_numeric_right: true,
2315        }
2316    }
2317
2318    fn analysis(mean: f64, std_dev: f64, sorted: Vec<f64>) -> DistributionAnalysis {
2319        use crate::statistics::{
2320            DistributionCharacteristics, OutlierAnalysis, PercentileBreakdown,
2321        };
2322        DistributionAnalysis {
2323            column_name: "close".into(),
2324            distribution_type: DistributionType::Normal,
2325            confidence: 0.0,
2326            fit_quality: 0.0,
2327            characteristics: DistributionCharacteristics {
2328                shapiro_wilk_stat: None,
2329                shapiro_wilk_pvalue: None,
2330                skewness: 0.0,
2331                kurtosis: 3.0,
2332                mean,
2333                median: mean,
2334                std_dev,
2335                variance: std_dev * std_dev,
2336                coefficient_of_variation: std_dev / mean,
2337                mode: None,
2338            },
2339            outliers: OutlierAnalysis {
2340                total_count: 0,
2341                percentage: 0.0,
2342                iqr_count: 0,
2343                zscore_count: 0,
2344                outlier_rows: Vec::new(),
2345            },
2346            percentiles: PercentileBreakdown {
2347                p1: 0.0,
2348                p5: 0.0,
2349                p25: 0.0,
2350                p50: 0.0,
2351                p75: 0.0,
2352                p95: 0.0,
2353                p99: 0.0,
2354            },
2355            sample_size: sorted.len(),
2356            sorted_sample_values: sorted,
2357            is_sampled: false,
2358            fits: Vec::new(),
2359            qq: Vec::new(),
2360        }
2361    }
2362
2363    /// The histogram's bars sit on the axis they are drawn against: the first bin's
2364    /// bar in the first plot column, and nothing past the chart's right edge. They were
2365    /// shifted right by a bar and a half, onto the next bin and off the end.
2366    /// Busy at the low end, so the first bin has a bar, with a Normal fit.
2367    fn skewed_normal_fit() -> DistributionAnalysis {
2368        let mut values: Vec<f64> = (0..400).map(|i| 23.0 + (i % 20) as f64).collect();
2369        values.extend((0..100).map(|i| 23.0 + 3.18 * i as f64));
2370        values.sort_by(f64::total_cmp);
2371        let mut dist = analysis(100.0, 80.0, values);
2372        dist.fits = vec![(
2373            DistributionType::Normal,
2374            FitOutcome::Tested(FitTest {
2375                fitted: crate::distribution_fit::Fitted::Normal {
2376                    mean: 100.0,
2377                    sd: 80.0,
2378                },
2379                p_value: 0.005,
2380                beyond: 0,
2381                replicates: 199,
2382                tested_on: 500,
2383                aic: 0.0,
2384            }),
2385        )];
2386        dist
2387    }
2388
2389    /// One of the Distribution detail's plots, in a 60x20 area of an 80x20 buffer.
2390    fn render_distribution_plot(
2391        dist: &DistributionAnalysis,
2392        g: &crate::glyphs::Glyphs,
2393        render: fn(DistributionPlotConfig, &mut Buffer),
2394    ) -> Buffer {
2395        render_distribution_plot_in(dist, g, render, 60)
2396    }
2397
2398    fn render_distribution_plot_in(
2399        dist: &DistributionAnalysis,
2400        g: &crate::glyphs::Glyphs,
2401        render: fn(DistributionPlotConfig, &mut Buffer),
2402        width: u16,
2403    ) -> Buffer {
2404        let numbers = NumberFormatSettings::default();
2405        render_distribution_plot_as(dist, g, render, width, &numbers)
2406    }
2407
2408    /// The plot with its numbers in `numbers`.
2409    fn render_distribution_plot_as(
2410        dist: &DistributionAnalysis,
2411        g: &crate::glyphs::Glyphs,
2412        render: fn(DistributionPlotConfig, &mut Buffer),
2413        width: u16,
2414        numbers: &NumberFormatSettings,
2415    ) -> Buffer {
2416        let theme =
2417            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2418        let mut buf = Buffer::empty(Rect::new(0, 0, 80, 20));
2419        let values = AxisNumbers::measure(numbers, &dist.column_name);
2420        let counts = AxisNumbers::count(numbers);
2421        render(
2422            DistributionPlotConfig {
2423                dist,
2424                dist_type: DistributionType::Normal,
2425                area: Rect::new(0, 0, width, 20),
2426                shared_y_axis_label_width: 5,
2427                theme: &theme,
2428                unified_x_range: Some((23.0, 341.1)),
2429                histogram_scale: HistogramScale::Linear,
2430                glyphs: g,
2431                values: &values,
2432                counts: &counts,
2433            },
2434            &mut buf,
2435        );
2436        buf
2437    }
2438
2439    /// The histogram's counts group as the table groups numbers: a bin of thousands
2440    /// reads `9,600`, not `9600`.
2441    #[test]
2442    fn distribution_counts_follow_the_table_number_format() {
2443        let mut dist = skewed_normal_fit();
2444        let values = dist.sorted_sample_values.clone();
2445        dist.sorted_sample_values = values
2446            .iter()
2447            .cycle()
2448            .take(values.len() * 24)
2449            .copied()
2450            .collect();
2451        dist.sorted_sample_values.sort_by(f64::total_cmp);
2452        let labels = |numbers: &NumberFormatSettings| -> Vec<String> {
2453            let g = crate::glyphs::unicode();
2454            let buf =
2455                render_distribution_plot_as(&dist, g, render_distribution_histogram, 60, numbers);
2456            (0..20)
2457                .filter_map(|y| {
2458                    let row: String = (0..60).map(|x| buf[(x, y)].symbol()).collect();
2459                    let label = row.split_once(['│', '┤'])?.0.trim().to_string();
2460                    (!label.is_empty()).then_some(label)
2461                })
2462                .collect()
2463        };
2464        let grouped = labels(&settings("thousands", true));
2465        assert_eq!(grouped.len(), 3, "{grouped:?}");
2466        assert!(
2467            grouped[0].contains(',') && grouped[0].len() > 4,
2468            "{grouped:?}"
2469        );
2470        let plain = labels(&settings("thousands", false));
2471        assert_eq!(plain[0], grouped[0].replace(',', ""), "{plain:?}");
2472    }
2473
2474    #[test]
2475    fn histogram_bars_stay_on_their_axis() {
2476        let dist = skewed_normal_fit();
2477        for g in [crate::glyphs::unicode(), crate::glyphs::ascii()] {
2478            let buf = render_distribution_plot(&dist, g, render_distribution_histogram);
2479            let full = g.plot.column_eighths[7];
2480            let is_bar = |x: u16| (0..20).any(|y| buf[(x, y)].symbol() == full);
2481            // Labels, a space, the axis line: the plot's first column.
2482            assert!(is_bar(5 + 2), "the first bin starts at the axis");
2483            assert!(
2484                (60..80).all(|x| !is_bar(x)),
2485                "nothing is drawn past the chart"
2486            );
2487            // The fit's curve is drawn, and goes behind the bars rather than through
2488            // them: below the top of a bar, every cell is solid.
2489            let curve = |x: u16, y: u16| PlotMarks::is_mark(g.plot.line, buf[(x, y)].symbol());
2490            assert!(
2491                (0..80).any(|x| (0..20).any(|y| curve(x, y) && buf[(x, y)].symbol() != "\u{2800}")),
2492                "the curve is drawn"
2493            );
2494            for x in 0..80 {
2495                if let Some(top) = (0..20).find(|y| buf[(x, *y)].symbol() == full) {
2496                    assert!(
2497                        (top..20).all(|y| !curve(x, y)),
2498                        "a notch in the bar at column {x}"
2499                    );
2500                }
2501            }
2502        }
2503    }
2504
2505    /// The bars span the plot, first column to last, each bin's bar over the columns
2506    /// its values map to: bars of one width stopped up to a bar short of the right
2507    /// end, every bar left of the values the labels give.
2508    #[test]
2509    fn histogram_bars_span_the_plot() {
2510        let g = crate::glyphs::unicode();
2511        let theme =
2512            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2513        let numbers = NumberFormatSettings::default();
2514        // Spread evenly over the axis, linear or logarithmic, so every bin has a bar.
2515        let linear: Vec<f64> = (0..500).map(|i| 23.0 + 318.1 * i as f64 / 499.0).collect();
2516        let log: Vec<f64> = (0..500)
2517            .map(|i| 10f64.powf(4.0 * i as f64 / 499.0))
2518            .collect();
2519        for (scale, values, range) in [
2520            (HistogramScale::Linear, linear, (23.0, 341.1)),
2521            (HistogramScale::Log, log, (1.0, 10_000.0)),
2522        ] {
2523            let dist = analysis(100.0, 80.0, values);
2524            for width in [60u16, 80, 120] {
2525                // Wider than the plot, to catch a bar drawn past it.
2526                let mut buf = Buffer::empty(Rect::new(0, 0, width + 10, 20));
2527                render_distribution_histogram(
2528                    DistributionPlotConfig {
2529                        dist: &dist,
2530                        dist_type: DistributionType::Normal,
2531                        area: Rect::new(0, 0, width, 20),
2532                        shared_y_axis_label_width: 5,
2533                        theme: &theme,
2534                        unified_x_range: Some(range),
2535                        histogram_scale: scale,
2536                        glyphs: g,
2537                        values: &AxisNumbers::measure(&numbers, &dist.column_name),
2538                        counts: &AxisNumbers::count(&numbers),
2539                    },
2540                    &mut buf,
2541                );
2542                let text = rendered_text(&buf);
2543                let what = format!("{scale:?} at {width}:\n{text}");
2544                let axis_row = (0..20)
2545                    .rfind(|y| (0..width).any(|x| buf[(x, *y)].symbol() == g.plot.axis.bottom_left))
2546                    .expect(&what);
2547                let corner = (0..width)
2548                    .find(|x| buf[(*x, axis_row)].symbol() == g.plot.axis.bottom_left)
2549                    .unwrap();
2550                let (left, right) = (corner + 1, width - 1);
2551                assert!(
2552                    [g.plot.axis.horizontal, g.plot.tick_x]
2553                        .contains(&buf[(right, axis_row)].symbol()),
2554                    "{what}"
2555                );
2556                let is_bar = |x: u16| {
2557                    (0..axis_row).any(|y| g.plot.column_eighths.contains(&buf[(x, y)].symbol()))
2558                };
2559                let bars: Vec<u16> = (0..width + 10).filter(|x| is_bar(*x)).collect();
2560                assert_eq!(
2561                    bars.first(),
2562                    Some(&left),
2563                    "the first bar starts the plot\n{what}"
2564                );
2565                assert_eq!(
2566                    bars.last(),
2567                    Some(&right),
2568                    "the last bar ends the plot\n{what}"
2569                );
2570                // Each bar and the gap after it, or the last bar alone, take an even share
2571                // of the plot.
2572                let mut starts = vec![left];
2573                starts.extend(bars.windows(2).filter(|w| w[1] > w[0] + 1).map(|w| w[1]));
2574                let shares: Vec<u16> = starts
2575                    .windows(2)
2576                    .map(|w| w[1] - w[0])
2577                    .chain([right + 1 - starts[starts.len() - 1]])
2578                    .collect();
2579                let (least, most) = (shares.iter().min().unwrap(), shares.iter().max().unwrap());
2580                assert!(most - least <= 1, "{shares:?}\n{what}");
2581            }
2582        }
2583    }
2584
2585    /// On Log the bins are equal in log space and the labels sit where their values
2586    /// do: over 1 to 10,000 the middle of the axis is 100, not the linear 5,000.
2587    #[test]
2588    fn log_histogram_labels_sit_at_their_values() {
2589        let values: Vec<f64> = (0..400)
2590            .map(|i| 10f64.powf(4.0 * i as f64 / 399.0))
2591            .collect();
2592        let dist = analysis(1_000.0, 2_000.0, values);
2593        let theme =
2594            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2595        let numbers = NumberFormatSettings::default();
2596        let g = crate::glyphs::unicode();
2597        let mut buf = Buffer::empty(Rect::new(0, 0, 80, 20));
2598        render_distribution_histogram(
2599            DistributionPlotConfig {
2600                dist: &dist,
2601                dist_type: DistributionType::Normal,
2602                area: Rect::new(0, 0, 80, 20),
2603                shared_y_axis_label_width: 5,
2604                theme: &theme,
2605                unified_x_range: Some((1.0, 10_000.0)),
2606                histogram_scale: HistogramScale::Log,
2607                glyphs: g,
2608                values: &AxisNumbers::measure(&numbers, &dist.column_name),
2609                counts: &AxisNumbers::count(&numbers),
2610            },
2611            &mut buf,
2612        );
2613        let text = rendered_text(&buf);
2614        let rows: Vec<&str> = text.lines().collect();
2615        let axis = rows
2616            .iter()
2617            .rposition(|r| r.contains(g.plot.axis.bottom_left))
2618            .expect(&text);
2619        let row = rows[axis + 1];
2620        let labels: Vec<(usize, f64)> = row
2621            .split_whitespace()
2622            .map(|l| {
2623                let at = row.find(l).unwrap() + l.len() / 2;
2624                (at, l.replace(',', "").parse::<f64>().expect(&text))
2625            })
2626            .collect();
2627        assert_eq!(labels.len(), 3, "{text}");
2628        let [(left, first), (at, middle), (right, last)] = labels[..] else {
2629            unreachable!()
2630        };
2631        assert_eq!((first, middle, last), (1.0, 100.0, 10_000.0), "{text}");
2632        assert!(
2633            at.abs_diff((left + right) / 2) <= 1,
2634            "the middle label is at the axis's middle:\n{text}"
2635        );
2636    }
2637
2638    /// Under the ASCII set, both of the Distribution detail's plots draw ASCII only:
2639    /// bars, the fit's curve, the Q-Q points and every axis line.
2640    #[test]
2641    fn distribution_plots_are_ascii_under_the_ascii_set() {
2642        let g = crate::glyphs::ascii();
2643        let mut dist = skewed_normal_fit();
2644        let qq: Vec<f64> = (0..dist.sorted_sample_values.len())
2645            .map(|i| 23.0 + 318.0 * i as f64 / 499.0)
2646            .collect();
2647        dist.qq = vec![(DistributionType::Normal, qq)];
2648        for (name, render) in [
2649            (
2650                "histogram",
2651                render_distribution_histogram as fn(DistributionPlotConfig, &mut Buffer),
2652            ),
2653            ("Q-Q plot", render_qq_plot),
2654        ] {
2655            let text = rendered_text(&render_distribution_plot(&dist, g, render));
2656            assert!(text.is_ascii(), "{name}:\n{text}");
2657            assert!(
2658                text.contains('|') && text.contains("+-"),
2659                "{name} axes:\n{text}"
2660            );
2661        }
2662        let qq = rendered_text(&render_distribution_plot(&dist, g, render_qq_plot));
2663        assert!(qq.contains('*'), "the Q-Q points:\n{qq}");
2664    }
2665
2666    /// The Distribution plots follow the rule every chart does: x labels a space
2667    /// apart with both ends kept, and axis titles on rows that hold nothing else.
2668    #[test]
2669    fn distribution_axes_follow_the_chart_rule() {
2670        let mut dist = skewed_normal_fit();
2671        let qq: Vec<f64> = (0..dist.sorted_sample_values.len())
2672            .map(|i| 23.0 + 318.0 * i as f64 / 499.0)
2673            .collect();
2674        dist.qq = vec![(DistributionType::Normal, qq)];
2675        let is_number = |t: &str| t.trim_end_matches(['k', 'M']).parse::<f64>().is_ok();
2676        for width in [40, 60, 80] {
2677            for g in [crate::glyphs::ascii(), crate::glyphs::unicode()] {
2678                for (name, render, y_title, x_title) in [
2679                    (
2680                        "histogram",
2681                        render_distribution_histogram as fn(DistributionPlotConfig, &mut Buffer),
2682                        "Counts",
2683                        None,
2684                    ),
2685                    (
2686                        "Q-Q plot",
2687                        render_qq_plot,
2688                        "Data Values",
2689                        Some("Theoretical Values"),
2690                    ),
2691                ] {
2692                    let text = rendered_text(&render_distribution_plot_in(&dist, g, render, width));
2693                    let rows: Vec<&str> = text.lines().collect();
2694                    let what = format!("{name} at {width}:\n{text}");
2695                    let axis = rows
2696                        .iter()
2697                        .rposition(|r| r.contains(g.plot.axis.bottom_left))
2698                        .expect(&what);
2699                    let labels: Vec<&str> = rows[axis + 1].split_whitespace().collect();
2700                    assert!(labels.len() >= 2, "both ends: {what}");
2701                    assert!(labels.iter().all(|l| is_number(l)), "apart: {what}");
2702                    // The plot's title, then the y axis's on a row of its own.
2703                    assert_eq!(rows[1].trim(), y_title, "{what}");
2704                    if let Some(x_title) = x_title {
2705                        assert_eq!(rows[axis + 2].trim(), x_title, "{what}");
2706                    }
2707                }
2708            }
2709        }
2710    }
2711
2712    #[test]
2713    fn counts_follow_the_data_table_grouping_setting() {
2714        // A user who turned grouping on should see it wherever they read
2715        // numbers, not just in the main table.
2716        assert_eq!(
2717            format_count(3_088_269, &settings("thousands", true)),
2718            "3,088,269"
2719        );
2720        assert_eq!(
2721            format_count(3_088_269, &settings("european", true)),
2722            "3.088.269"
2723        );
2724    }
2725
2726    #[test]
2727    fn counts_are_raw_when_formatting_is_off() {
2728        assert_eq!(
2729            format_count(3_088_269, &settings("thousands", false)),
2730            "3088269"
2731        );
2732        assert_eq!(format_count(0, &settings("thousands", false)), "0");
2733    }
2734
2735    #[test]
2736    fn counts_group_uniformly_with_no_magnitude_threshold() {
2737        assert_eq!(format_count(42, &settings("thousands", true)), "42");
2738        assert_eq!(format_count(1000, &settings("thousands", true)), "1,000");
2739        assert_eq!(format_count(10_000, &settings("thousands", true)), "10,000");
2740    }
2741
2742    fn correlation_matrix(r: f64, pairs: usize) -> crate::statistics::CorrelationMatrix {
2743        crate::statistics::CorrelationMatrix {
2744            columns: vec!["price".to_string(), "volume".to_string()],
2745            correlations: vec![vec![1.0, r], vec![r, 1.0]],
2746            p_values: Some(vec![vec![0.0, 0.004], vec![0.004, 0.0]]),
2747            sample_sizes: vec![vec![0, pairs], vec![pairs, 0]],
2748            rank_correlations: Some(vec![vec![1.0, 0.5], vec![0.5, 1.0]]),
2749            rank_p_values: Some(vec![vec![0.0, 0.03], vec![0.03, 0.0]]),
2750        }
2751    }
2752
2753    fn rendered_text(buf: &Buffer) -> String {
2754        let mut text = String::new();
2755        for y in 0..buf.area.height {
2756            for x in 0..buf.area.width {
2757                text.push_str(buf[(x, y)].symbol());
2758            }
2759            text.push('\n');
2760        }
2761        text
2762    }
2763
2764    /// The furthest scroll is the first that shows the last statistic: one short
2765    /// of it leaves the last out, and nothing scrolls past it.
2766    #[test]
2767    fn the_statistics_scroll_stops_where_the_last_comes_into_view() {
2768        let widths = [5, 5, 6, 3, 10, 6, 6, 6, 6];
2769        for available in [12u16, 20, 30, 45, 80] {
2770            let mut columns = ColumnScroll {
2771                offset: usize::MAX,
2772                max: 0,
2773            };
2774            let (start, end) = stat_window(&widths, available, 2, &mut columns);
2775            assert_eq!(start, columns.max, "clamped to the furthest start");
2776            assert_eq!(end, widths.len(), "the last is in view at {available}");
2777            if columns.max > 0 {
2778                columns.offset = columns.max - 1;
2779                let (_, end) = stat_window(&widths, available, 2, &mut columns);
2780                assert!(end < widths.len(), "one short leaves it out at {available}");
2781            }
2782        }
2783        let mut columns = ColumnScroll::default();
2784        assert_eq!(stat_window(&widths, 200, 2, &mut columns), (0, 9));
2785        assert_eq!(columns.max, 0, "everything fits, so nothing scrolls");
2786    }
2787
2788    /// The family list keeps its cursor in view and, while families are out of
2789    /// view below, a row to count them: the cursor never takes that row.
2790    #[test]
2791    fn the_family_list_counts_what_is_below_the_cursor() {
2792        assert_eq!(list_window(3, 5, 8), (0, 5), "everything fits");
2793        for selected in 0..14 {
2794            let (offset, shown) = list_window(selected, 14, 12);
2795            assert!(
2796                (offset..offset + shown).contains(&selected),
2797                "{selected} is drawn"
2798            );
2799            let below = 14 - offset - shown;
2800            if below > 0 {
2801                assert_eq!(shown, 11, "a row is left to count {below} at {selected}");
2802            } else {
2803                assert_eq!(shown, 12);
2804            }
2805        }
2806        assert_eq!(list_window(11, 14, 12), (1, 11), "not the last row");
2807        assert_eq!(list_window(13, 14, 12), (2, 12), "the end needs no count");
2808        assert_eq!(list_window(4, 14, 1), (4, 1), "one row is the cursor's");
2809    }
2810
2811    /// The matrix scrolls to the selected cell however it got there, and counts
2812    /// the columns it cannot show.
2813    #[test]
2814    fn the_correlation_matrix_keeps_the_selected_column_in_view() {
2815        let names: Vec<String> = (0..6).map(|i| format!("col_{i}")).collect();
2816        let n = names.len();
2817        let matrix = crate::statistics::CorrelationMatrix {
2818            columns: names,
2819            correlations: vec![vec![0.5; n]; n],
2820            p_values: None,
2821            sample_sizes: vec![vec![10; n]; n],
2822            rank_correlations: Some(vec![vec![0.5; n]; n]),
2823            rank_p_values: None,
2824        };
2825        let results = AnalysisResults {
2826            column_statistics: vec![],
2827            total_rows: 10,
2828            sample_size: None,
2829            per_value: None,
2830            sample_seed: 0,
2831            correlation_matrix: Some(matrix),
2832            distribution_analyses: vec![],
2833        };
2834        let theme = Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2835        let area = Rect::new(0, 0, 60, 10);
2836        let mut columns = ColumnScroll::default();
2837        let mut state = TableState::default();
2838        let mut header = |selected: (usize, usize), columns: &mut ColumnScroll| {
2839            let mut buf = Buffer::empty(area);
2840            state.select(Some(selected.0));
2841            render_correlation_matrix(
2842                results.correlation_matrix.as_ref().map(|matrix| Shown {
2843                    matrix,
2844                    method: CorrelationMethod::Pearson,
2845                }),
2846                &mut state,
2847                MatrixCursor {
2848                    cell: Some(selected),
2849                    focused: true,
2850                },
2851                columns,
2852                area,
2853                &mut buf,
2854                &theme,
2855            );
2856            rendered_text(&buf).lines().next().unwrap().to_string()
2857        };
2858        let first = header((0, 0), &mut columns);
2859        assert!(
2860            first.contains("col_0") && !first.contains("col_5"),
2861            "{first:?}"
2862        );
2863        assert!(
2864            first.contains('+'),
2865            "the hidden columns are counted: {first:?}"
2866        );
2867        let last = header((0, 5), &mut columns);
2868        assert!(
2869            last.contains("col_5"),
2870            "the selected column is drawn: {last:?}"
2871        );
2872        let back = header((0, 0), &mut columns);
2873        assert!(
2874            back.contains("col_0"),
2875            "and so is the first again: {back:?}"
2876        );
2877    }
2878
2879    /// One accent on screen: the pane without focus keeps its selection marked by
2880    /// a dimmed rail, never the accent, whichever side has the focus.
2881    #[test]
2882    fn the_unfocused_selection_is_dimmed() {
2883        let theme = Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2884        let accent = theme.get("accent");
2885        let dimmed = theme.get("dimmed");
2886        let rail = crate::glyphs::get().rail;
2887        let area = Rect::new(0, 0, 30, 8);
2888        let sidebar = |focus: AnalysisFocus| {
2889            let mut buf = Buffer::empty(area);
2890            let mut state = TableState::default();
2891            state.select(Some(2));
2892            render_sidebar(
2893                area,
2894                &mut buf,
2895                &mut state,
2896                Some(AnalysisTool::Describe),
2897                focus,
2898                &theme,
2899            );
2900            buf
2901        };
2902        // The pane has the focus: the tool on screen keeps a dimmed rail.
2903        let buf = sidebar(AnalysisFocus::Main);
2904        let describe = (0..area.height)
2905            .find(|&y| row_text(&buf, y).contains("Describe"))
2906            .unwrap();
2907        let at = |buf: &Buffer, y: u16| {
2908            (0..area.width)
2909                .find(|&x| buf[(x, y)].symbol() == rail)
2910                .map(|x| buf[(x, y)].fg)
2911        };
2912        assert_eq!(at(&buf, describe), Some(dimmed));
2913        for y in 1..area.height - 1 {
2914            for x in 1..area.width - 1 {
2915                assert_ne!(
2916                    buf[(x, y)].fg,
2917                    accent,
2918                    "no accent in the list at ({x}, {y})"
2919                );
2920            }
2921        }
2922        // The list has the focus: the cursor's rail is the accent; the tool on
2923        // screen is only bold.
2924        let buf = sidebar(AnalysisFocus::Sidebar);
2925        let cursor = (0..area.height)
2926            .find(|&y| row_text(&buf, y).contains("Correlation"))
2927            .unwrap();
2928        assert_eq!(at(&buf, cursor), Some(accent));
2929        assert_eq!(at(&buf, describe), None);
2930    }
2931
2932    fn row_text(buf: &Buffer, y: u16) -> String {
2933        (0..buf.area.width).map(|x| buf[(x, y)].symbol()).collect()
2934    }
2935
2936    #[test]
2937    fn describe_shows_a_datetime_range_and_leaves_std_blank() {
2938        let theme =
2939            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2940        let results = crate::statistics::compute_describe_single_aggregation(
2941            &crate::statistics::describe_tests::temporal_frame(),
2942            &crate::statistics::describe_tests::temporal_frame()
2943                .schema()
2944                .clone(),
2945            6,
2946            None,
2947            0,
2948            false,
2949        )
2950        .unwrap();
2951        let area = Rect::new(0, 0, 220, 6);
2952        let mut buf = Buffer::empty(area);
2953        StatisticsTable {
2954            results: &results,
2955            focused: false,
2956            theme: &theme,
2957            table_cell_padding: 1,
2958            number_format: &settings("thousands", false),
2959        }
2960        .render(
2961            area,
2962            &mut buf,
2963            &mut TableState::default(),
2964            &mut crate::analysis_modal::ColumnScroll::default(),
2965        );
2966        let text = rendered_text(&buf);
2967        let mut lines = text.lines();
2968        let header = lines.next().unwrap();
2969        let pickup = lines
2970            .find(|l| l.trim_start().starts_with("pickup"))
2971            .unwrap_or_else(|| panic!("{text}"));
2972        // Each value sits under its own header.
2973        for (stat, value) in [
2974            ("Mean", "2024-12-31 22:47:55"),
2975            ("Std", "- "),
2976            ("Min", "2024-12-31 20:47:55"),
2977            ("25%", "2024-12-31 21:47:55"),
2978            ("50%", "2024-12-31 22:47:55"),
2979            ("75%", "2024-12-31 23:47:55"),
2980            ("Max", "2025-01-01 00:47:55"),
2981        ] {
2982            let x = header.find(stat).unwrap();
2983            assert!(pickup[x..].starts_with(value), "{stat}:\n{text}");
2984        }
2985    }
2986
2987    #[test]
2988    fn correlation_detail_shows_the_pair_facts_the_matrix_holds() {
2989        let theme =
2990            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
2991        let area = Rect::new(0, 0, 60, 8);
2992        let mut buf = Buffer::empty(area);
2993        render_correlation_pair_summary(
2994            Shown {
2995                matrix: &correlation_matrix(0.874, 42),
2996                method: CorrelationMethod::Pearson,
2997            },
2998            (0, 1),
2999            50,
3000            area,
3001            &mut buf,
3002            &theme,
3003            &settings("thousands", false),
3004        );
3005        let text = rendered_text(&buf);
3006        assert!(text.contains("Pearson r: 0.8740"), "{text}");
3007        assert!(text.contains("strong positive"), "{text}");
3008        let r_squared = crate::glyphs::get().r_squared;
3009        assert!(text.contains(&format!("{r_squared}: 0.7639")), "{text}");
3010        assert!(text.contains("P-value: 0.004"), "{text}");
3011        assert!(text.contains("Pairs used: 42 of 50 rows"), "{text}");
3012    }
3013
3014    #[test]
3015    fn correlation_detail_says_when_too_few_pairs_overlap() {
3016        // A pair with fewer than 3 overlapping values holds NaN in the matrix.
3017        let theme =
3018            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
3019        let area = Rect::new(0, 0, 70, 8);
3020        let mut buf = Buffer::empty(area);
3021        render_correlation_pair_summary(
3022            Shown {
3023                matrix: &correlation_matrix(f64::NAN, 2),
3024                method: CorrelationMethod::Pearson,
3025            },
3026            (0, 1),
3027            50,
3028            area,
3029            &mut buf,
3030            &theme,
3031            &settings("thousands", false),
3032        );
3033        let text = rendered_text(&buf);
3034        assert!(text.contains("Fewer than 3 overlapping pairs"), "{text}");
3035        assert!(!text.contains("Pearson r:"), "{text}");
3036    }
3037
3038    /// A matrix too large to rank says so under Spearman, and still shows Pearson.
3039    #[test]
3040    fn a_matrix_without_ranks_says_why_under_spearman() {
3041        let theme =
3042            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
3043        let mut matrix = correlation_matrix(0.874, 42);
3044        matrix.rank_correlations = None;
3045        matrix.rank_p_values = None;
3046        let area = Rect::new(0, 0, 90, 8);
3047        let mut buf = Buffer::empty(area);
3048        render_correlation_pair_summary(
3049            Shown {
3050                matrix: &matrix,
3051                method: CorrelationMethod::Spearman,
3052            },
3053            (0, 1),
3054            50,
3055            area,
3056            &mut buf,
3057            &theme,
3058            &settings("thousands", false),
3059        );
3060        assert!(rendered_text(&buf).contains(SPEARMAN_TOO_MANY));
3061        let mut buf = Buffer::empty(area);
3062        render_correlation_pair_summary(
3063            Shown {
3064                matrix: &matrix,
3065                method: CorrelationMethod::Pearson,
3066            },
3067            (0, 1),
3068            50,
3069            area,
3070            &mut buf,
3071            &theme,
3072            &settings("thousands", false),
3073        );
3074        assert!(rendered_text(&buf).contains("Pearson r: 0.8740"));
3075    }
3076
3077    #[test]
3078    fn correlation_detail_shows_the_chosen_method() {
3079        let theme =
3080            crate::config::Theme::from_config(&crate::config::ThemeConfig::default()).unwrap();
3081        let area = Rect::new(0, 0, 60, 8);
3082        let mut buf = Buffer::empty(area);
3083        render_correlation_pair_summary(
3084            Shown {
3085                matrix: &correlation_matrix(0.874, 42),
3086                method: CorrelationMethod::Spearman,
3087            },
3088            (0, 1),
3089            50,
3090            area,
3091            &mut buf,
3092            &theme,
3093            &settings("thousands", false),
3094        );
3095        let text = rendered_text(&buf);
3096        let rho = crate::glyphs::get().rho;
3097        assert!(text.contains(&format!("Spearman {rho}: 0.5000")), "{text}");
3098        assert!(text.contains("P-value: 0.03"), "{text}");
3099    }
3100
3101    /// Rounding never shows a perfect relation that is not one.
3102    #[test]
3103    fn a_coefficient_rounds_to_one_only_when_it_is_one() {
3104        assert_eq!(format_coefficient(0.9996, 3), "0.999");
3105        assert_eq!(format_coefficient(-0.9996, 3), "-0.999");
3106        assert_eq!(format_coefficient(0.99996, 4), "0.9999");
3107        assert_eq!(format_coefficient(0.9994, 3), "0.999");
3108        assert_eq!(format_coefficient(1.0, 3), "1.000");
3109        assert_eq!(format_coefficient(-1.0, 4), "-1.0000");
3110        assert_eq!(format_coefficient(0.12345, 3), "0.123");
3111        assert_eq!(format_coefficient(-0.5, 3), "-0.500");
3112    }
3113
3114    #[test]
3115    fn correlation_words_match_the_color_boundaries() {
3116        assert_eq!(describe_correlation(0.01), "none");
3117        assert_eq!(describe_correlation(0.2), "weak positive");
3118        assert_eq!(describe_correlation(-0.5), "moderate negative");
3119        assert_eq!(describe_correlation(0.9), "strong positive");
3120        assert_eq!(describe_correlation(-0.9), "strong negative");
3121    }
3122}