Skip to main content

datui_lib/
numfmt.rs

1//! Display-time number formatting: digit grouping, decimal separator, optional fixed
2//! float precision. Display only: exports, queries, filters, views and group-by keys use
3//! raw values. Runs per visible cell per frame, so it allocates nothing beyond the
4//! caller's `String`: integers are written digit by digit with inline separators,
5//! [`NumberFormat::width_i64`] measures arithmetically, and per-column decisions resolve
6//! to a [`CellFormatter`] once per column per frame.
7
8use std::borrow::Cow;
9use std::fmt::Write as _;
10
11use polars::prelude::{AnyValue, DataType};
12
13/// Digit grouping style.
14#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
15pub enum Grouping {
16    /// No grouping: `1234567`
17    #[default]
18    None,
19    /// Western grouping in threes: `1,234,567`
20    Thousands,
21    /// Indian grouping — three, then twos: `12,34,567` (lakh / crore)
22    Indian,
23}
24
25impl Grouping {
26    /// Number of separators a value with `digits` integer digits will contain.
27    fn separator_count(self, digits: usize) -> usize {
28        match self {
29            Grouping::None => 0,
30            Grouping::Thousands => digits.saturating_sub(1) / 3,
31            Grouping::Indian => {
32                if digits <= 3 {
33                    0
34                } else {
35                    1 + (digits - 4) / 2
36                }
37            }
38        }
39    }
40
41    /// Whether a separator goes before the next digit, given how many digits
42    /// have already been emitted (counting from the right).
43    #[inline]
44    fn breaks_after(self, digits_emitted: u32) -> bool {
45        match self {
46            Grouping::None => false,
47            Grouping::Thousands => digits_emitted.is_multiple_of(3),
48            Grouping::Indian => {
49                digits_emitted == 3 || (digits_emitted > 3 && digits_emitted % 2 == 1)
50            }
51        }
52    }
53}
54
55/// How numbers are rendered. Resolved from config once at load time.
56#[derive(Debug, Clone, PartialEq)]
57pub struct NumberFormat {
58    pub grouping: Grouping,
59    /// Character placed between digit groups.
60    pub group_sep: char,
61    /// Character used as the decimal point.
62    pub decimal_sep: char,
63    /// Apply grouping to float columns as well as integer columns.
64    pub floats: bool,
65    /// Fixed decimal places for floats. `None` keeps the default rendering.
66    pub float_precision: Option<u8>,
67}
68
69impl Default for NumberFormat {
70    fn default() -> Self {
71        Self::PLAIN
72    }
73}
74
75impl NumberFormat {
76    /// Renders numbers exactly as Polars would — no grouping, no precision
77    /// override. This is the default so an upgrade changes nothing.
78    pub const PLAIN: Self = Self {
79        grouping: Grouping::None,
80        group_sep: ',',
81        decimal_sep: '.',
82        floats: true,
83        float_precision: None,
84    };
85
86    /// A named preset covering common locale conventions without ICU/CLDR data (see
87    /// `docs/user-guide/configuration.md` on why formatting is explicit).
88    pub fn preset(name: &str) -> Option<Self> {
89        let base = Self::PLAIN;
90        Some(match name {
91            // 1234567
92            "none" | "plain" => base,
93            // 1,234,567.89
94            "thousands" => Self {
95                grouping: Grouping::Thousands,
96                group_sep: ',',
97                decimal_sep: '.',
98                ..base
99            },
100            // 1.234.567,89
101            "european" => Self {
102                grouping: Grouping::Thousands,
103                group_sep: '.',
104                decimal_sep: ',',
105                ..base
106            },
107            // 1 234 567.89 (ISO 31-0 / SI)
108            "si" => Self {
109                grouping: Grouping::Thousands,
110                group_sep: '\u{202f}', // narrow no-break space
111                decimal_sep: '.',
112                ..base
113            },
114            // 1'234'567.89
115            "swiss" => Self {
116                grouping: Grouping::Thousands,
117                group_sep: '\'',
118                decimal_sep: '.',
119                ..base
120            },
121            // 12,34,567.89
122            "indian" => Self {
123                grouping: Grouping::Indian,
124                group_sep: ',',
125                decimal_sep: '.',
126                ..base
127            },
128            // 1_234_567.89
129            "underscore" => Self {
130                grouping: Grouping::Thousands,
131                group_sep: '_',
132                decimal_sep: '.',
133                ..base
134            },
135            _ => return None,
136        })
137    }
138
139    /// Comma grouping with no digit threshold, used for the application's own
140    /// labels rather than the user's data.
141    pub const CHROME: Self = Self {
142        grouping: Grouping::Thousands,
143        group_sep: ',',
144        decimal_sep: '.',
145        floats: true,
146        float_precision: None,
147    };
148
149    /// Every preset name, for CLI value parsing and error messages.
150    pub const PRESET_NAMES: &'static [&'static str] = &[
151        "none",
152        "thousands",
153        "european",
154        "si",
155        "swiss",
156        "indian",
157        "underscore",
158    ];
159
160    /// True when this format would render every value exactly as Polars does,
161    /// so callers can take the zero-cost passthrough path.
162    pub fn is_noop(&self) -> bool {
163        self.grouping == Grouping::None && self.decimal_sep == '.' && self.float_precision.is_none()
164    }
165
166    /// Whether values are grouped: every value in a formatted column, without a magnitude
167    /// threshold, for uniform columns; identifier columns belong in `exclude`.
168    #[inline]
169    fn groups(&self) -> bool {
170        self.grouping != Grouping::None
171    }
172
173    /// Display width (in characters) of `v` as this format would render it,
174    /// computed arithmetically — no string is built.
175    pub fn width_i64(&self, v: i64) -> usize {
176        self.width_u64(v.unsigned_abs()) + usize::from(v < 0)
177    }
178
179    /// Display width (in characters) of `v` as this format would render it.
180    pub fn width_u64(&self, v: u64) -> usize {
181        let digits = digit_count(v);
182        let seps = if self.groups() {
183            self.grouping.separator_count(digits)
184        } else {
185            0
186        };
187        digits + seps
188    }
189
190    /// Append `v` to `out`, returning the display width in characters.
191    pub fn write_i64(&self, v: i64, out: &mut String) -> usize {
192        self.write_magnitude(v.unsigned_abs(), v < 0, out)
193    }
194
195    /// Append `v` to `out`, returning the display width in characters.
196    pub fn write_u64(&self, v: u64, out: &mut String) -> usize {
197        self.write_magnitude(v, false, out)
198    }
199
200    /// Core integer path: writes digits back-to-front into a stack buffer,
201    /// emitting separators inline, then appends the finished slice in one go.
202    fn write_magnitude(&self, mag: u64, negative: bool, out: &mut String) -> usize {
203        // u64::MAX is 20 digits; Indian grouping tops out at 9 separators, each
204        // at most 4 UTF-8 bytes; plus a sign.
205        let mut buf = [0u8; 64];
206        let mut pos = buf.len();
207        let mut width = 0usize;
208
209        let group = self.groups();
210        let mut sep_bytes = [0u8; 4];
211        let sep = self.group_sep.encode_utf8(&mut sep_bytes);
212        let sep = sep.as_bytes();
213
214        let mut n = mag;
215        let mut emitted: u32 = 0;
216        loop {
217            let d = (n % 10) as u8;
218            n /= 10;
219            pos -= 1;
220            buf[pos] = b'0' + d;
221            emitted += 1;
222            width += 1;
223            if n == 0 {
224                break;
225            }
226            if group && self.grouping.breaks_after(emitted) {
227                pos -= sep.len();
228                buf[pos..pos + sep.len()].copy_from_slice(sep);
229                width += 1;
230            }
231        }
232
233        if negative {
234            pos -= 1;
235            buf[pos] = b'-';
236            width += 1;
237        }
238
239        // Every byte written is either ASCII or a complete UTF-8 encoding of
240        // `group_sep`, so the slice is valid UTF-8 by construction.
241        debug_assert!(std::str::from_utf8(&buf[pos..]).is_ok());
242        match std::str::from_utf8(&buf[pos..]) {
243            Ok(s) => {
244                out.push_str(s);
245                width
246            }
247            // Unreachable; zero rather than `width`, since a width not matching what was pushed
248            // would corrupt column sizing silently.
249            Err(_) => 0,
250        }
251    }
252
253    /// Append `v` to `out`, returning its display width; `scratch` is a reused buffer.
254    pub fn write_f64(&self, v: f64, scratch: &mut String, out: &mut String) -> usize {
255        scratch.clear();
256        match self.float_precision {
257            Some(p) => {
258                let _ = write!(scratch, "{:.*}", p as usize, v);
259            }
260            None => {
261                let _ = write!(scratch, "{}", v);
262            }
263        }
264        self.regroup_decimal(scratch, out)
265    }
266
267    /// Group the integer part of an already-rendered decimal and apply `decimal_sep`,
268    /// returning the width. Keeps Polars' rendering (decimal places never change with
269    /// formatting); non-plain decimals (`NaN`, `inf`, scientific) pass through.
270    pub fn regroup_decimal(&self, src: &str, out: &mut String) -> usize {
271        let body = src.strip_prefix('-').unwrap_or(src);
272        let negative = body.len() != src.len();
273        let (int_part, frac_part) = match body.find('.') {
274            Some(i) => (&body[..i], Some(&body[i + 1..])),
275            None => (body, None),
276        };
277
278        let plain = !int_part.is_empty()
279            && int_part.bytes().all(|b| b.is_ascii_digit())
280            && frac_part.is_none_or(|f| f.bytes().all(|b| b.is_ascii_digit()));
281        if !plain {
282            // NaN, inf, 1e300 — not something to regroup.
283            out.push_str(src);
284            return src.chars().count();
285        }
286
287        let mut width = 0usize;
288        if negative {
289            out.push('-');
290            width += 1;
291        }
292
293        let digits = int_part.len();
294        if self.groups() {
295            for (i, ch) in int_part.chars().enumerate() {
296                // A separator precedes this digit when the digits still to come
297                // (including this one) land on a group boundary.
298                let remaining = (digits - i) as u32;
299                if i > 0 && self.grouping.breaks_after(remaining) {
300                    out.push(self.group_sep);
301                    width += 1;
302                }
303                out.push(ch);
304                width += 1;
305            }
306        } else {
307            out.push_str(int_part);
308            width += digits;
309        }
310
311        if let Some(frac) = frac_part {
312            out.push(self.decimal_sep);
313            width += 1 + frac.len();
314            out.push_str(frac);
315        }
316        width
317    }
318}
319
320/// A byte count in binary units: `512 B`, `1.2 MiB`, and whole from 100 up, `340 MiB`.
321pub fn bytes(n: u64) -> String {
322    const UNITS: [&str; 5] = ["B", "KiB", "MiB", "GiB", "TiB"];
323    let mut value = n as f64;
324    let mut unit = 0;
325    while value >= 1024.0 && unit < UNITS.len() - 1 {
326        value /= 1024.0;
327        unit += 1;
328    }
329    match unit {
330        0 => format!("{n} B"),
331        _ if value >= 100.0 => format!("{value:.0} {}", UNITS[unit]),
332        _ => format!("{value:.1} {}", UNITS[unit]),
333    }
334}
335
336/// A wait's clock, to the second while seconds matter: `12s`, `3m 05s`, `1h 02m`.
337pub fn clock(elapsed: std::time::Duration) -> String {
338    let s = elapsed.as_secs();
339    match s {
340        0..60 => format!("{s}s"),
341        60..3_600 => format!("{}m {:02}s", s / 60, s % 60),
342        _ => format!("{}h {:02}m", s / 3_600, s % 3_600 / 60),
343    }
344}
345
346/// A length of time in its largest unit, to a tenth past seconds so it fits a
347/// narrow column: `12s`, `2.5m`, `1.5h`, `-2.1d`.
348pub fn duration(seconds: i64) -> String {
349    let sign = if seconds < 0 { "-" } else { "" };
350    let s = seconds.unsigned_abs();
351    let (unit, name) = match s {
352        0..60 => return format!("{sign}{s}s"),
353        60..3_600 => (60.0, "m"),
354        3_600..86_400 => (3_600.0, "h"),
355        _ => (86_400.0, "d"),
356    };
357    format!("{sign}{:.1}{name}", s as f64 / unit)
358}
359
360/// [`duration`], or `-` for none.
361pub fn duration_or_dash(seconds: Option<i64>) -> String {
362    seconds.map_or_else(|| "-".to_string(), duration)
363}
364
365/// A share as a percentage: a tenth of a percent, two places below 1% so a small
366/// share never reads as none, and `<0.01%` below that.
367pub fn percent(share: f64) -> String {
368    let pct = share * 100.0;
369    if pct > 0.0 && pct < 0.01 {
370        "<0.01%".to_string()
371    } else if pct > 0.0 && pct < 1.0 {
372        format!("{pct:.2}%")
373    } else {
374        format!("{pct:.1}%")
375    }
376}
377
378/// [`percent`] of `count` in `of`, or a dash with nothing to take it over.
379pub fn percent_of(count: usize, of: usize) -> String {
380    if of == 0 {
381        return "-".to_string();
382    }
383    percent(count as f64 / of as f64)
384}
385
386/// Comma-group a count for datui's own chrome (footer count, info totals), regardless
387/// of `display.number_format` or `,`: turning formatting off for exact data never makes
388/// the UI harder to read.
389pub fn group_chrome(n: usize) -> String {
390    let mut out = String::new();
391    NumberFormat::CHROME.write_u64(n as u64, &mut out);
392    out
393}
394
395/// Digits in the base-10 representation of `n` (`0` counts as one digit).
396#[inline]
397fn digit_count(n: u64) -> usize {
398    n.checked_ilog10().map_or(0, |l| l as usize) + 1
399}
400
401/// Per-column formatting decision, resolved once per column per frame.
402#[derive(Debug, Clone, PartialEq)]
403pub enum CellFormatter {
404    /// Render exactly as Polars does. Zero added cost.
405    Passthrough,
406    /// Apply `NumberFormat` to numeric values.
407    Number(NumberFormat),
408}
409
410impl CellFormatter {
411    #[inline]
412    pub fn is_passthrough(&self) -> bool {
413        matches!(self, CellFormatter::Passthrough)
414    }
415}
416
417/// True for dtypes whose values are numbers we group.
418pub fn is_numeric_dtype(dtype: &DataType) -> bool {
419    matches!(
420        dtype,
421        DataType::Int8
422            | DataType::Int16
423            | DataType::Int32
424            | DataType::Int64
425            | DataType::UInt8
426            | DataType::UInt16
427            | DataType::UInt32
428            | DataType::UInt64
429            | DataType::Float32
430            | DataType::Float64
431    )
432}
433
434/// Dtypes rendered flush right; temporals excluded (fixed-width already).
435pub fn is_right_aligned_dtype(dtype: &DataType) -> bool {
436    is_numeric_dtype(dtype)
437}
438
439/// Fully resolved display-formatting settings, held in the render context.
440#[derive(Debug, Clone)]
441pub struct NumberFormatSettings {
442    /// The configured format.
443    pub format: NumberFormat,
444    /// Runtime `,` toggle. When false every column is `Passthrough`.
445    pub enabled: bool,
446    /// Columns never formatted (precompiled globs).
447    pub exclude: Vec<Glob>,
448    /// Right-align numeric columns and their headers.
449    pub align_numeric_right: bool,
450}
451
452impl Default for NumberFormatSettings {
453    fn default() -> Self {
454        Self {
455            format: NumberFormat::PLAIN,
456            enabled: true,
457            exclude: Vec::new(),
458            align_numeric_right: true,
459        }
460    }
461}
462
463impl NumberFormatSettings {
464    /// Resolve the formatter for one column. Called once per column per frame —
465    /// never per cell, so glob matching stays off the hot path.
466    pub fn formatter_for(&self, col_name: &str, dtype: &DataType) -> CellFormatter {
467        if !self.enabled || self.format.is_noop() || !is_numeric_dtype(dtype) {
468            return CellFormatter::Passthrough;
469        }
470        if self.exclude.iter().any(|g| g.matches(col_name)) {
471            return CellFormatter::Passthrough;
472        }
473        let mut fmt = self.format.clone();
474        if !fmt.floats && matches!(dtype, DataType::Float32 | DataType::Float64) {
475            // Grouping is off for floats, but a decimal separator or fixed
476            // precision may still apply.
477            fmt.grouping = Grouping::None;
478            if fmt.is_noop() {
479                return CellFormatter::Passthrough;
480            }
481        }
482        CellFormatter::Number(fmt)
483    }
484}
485
486/// Format one value for display; `Cow::Borrowed` when no formatting applies.
487pub fn format_any_value<'v>(
488    fmt: &CellFormatter,
489    value: &'v AnyValue<'v>,
490    scratch: &mut String,
491) -> Cow<'v, str> {
492    if matches!(value, AnyValue::Null) {
493        return Cow::Borrowed("");
494    }
495    let nf = match fmt {
496        CellFormatter::Passthrough => return crate::exact::str_value(value),
497        CellFormatter::Number(nf) => nf,
498    };
499    let mut out = String::new();
500    match *value {
501        AnyValue::Int8(v) => nf.write_i64(v as i64, &mut out),
502        AnyValue::Int16(v) => nf.write_i64(v as i64, &mut out),
503        AnyValue::Int32(v) => nf.write_i64(v as i64, &mut out),
504        AnyValue::Int64(v) => nf.write_i64(v, &mut out),
505        AnyValue::UInt8(v) => nf.write_u64(v as u64, &mut out),
506        AnyValue::UInt16(v) => nf.write_u64(v as u64, &mut out),
507        AnyValue::UInt32(v) => nf.write_u64(v as u64, &mut out),
508        AnyValue::UInt64(v) => nf.write_u64(v, &mut out),
509        // With no explicit precision, floats route through Polars' own
510        // rendering so toggling formatting never changes how many decimal
511        // places a value shows — only the grouping is layered on.
512        AnyValue::Float32(f) => match nf.float_precision {
513            Some(_) => nf.write_f64(f as f64, scratch, &mut out),
514            None => nf.regroup_decimal(&value.str_value(), &mut out),
515        },
516        AnyValue::Float64(f) => match nf.float_precision {
517            Some(_) => nf.write_f64(f, scratch, &mut out),
518            None => nf.regroup_decimal(&value.str_value(), &mut out),
519        },
520        _ => return crate::exact::str_value(value),
521    };
522    Cow::Owned(out)
523}
524
525/// A value's on-screen cells without building its string where possible (integers);
526/// others are rendered and measured as the table measures them.
527pub fn display_width(fmt: &CellFormatter, value: &AnyValue, scratch: &mut String) -> usize {
528    if matches!(value, AnyValue::Null) {
529        return 0;
530    }
531    if let CellFormatter::Number(nf) = fmt {
532        match *value {
533            AnyValue::Int8(v) => return nf.width_i64(v as i64),
534            AnyValue::Int16(v) => return nf.width_i64(v as i64),
535            AnyValue::Int32(v) => return nf.width_i64(v as i64),
536            AnyValue::Int64(v) => return nf.width_i64(v),
537            AnyValue::UInt8(v) => return nf.width_u64(v as u64),
538            AnyValue::UInt16(v) => return nf.width_u64(v as u64),
539            AnyValue::UInt32(v) => return nf.width_u64(v as u64),
540            AnyValue::UInt64(v) => return nf.width_u64(v),
541            _ => {}
542        }
543    }
544    crate::glyphs::cell_width(&format_any_value(fmt, value, scratch))
545}
546
547/// Minimal glob matcher: `*` (any run) and `?` (one character), all column-name
548/// matching needs.
549#[derive(Debug, Clone, PartialEq, Eq)]
550pub struct Glob {
551    pattern: String,
552    has_wildcard: bool,
553}
554
555impl Glob {
556    pub fn new(pattern: impl Into<String>) -> Self {
557        let pattern = pattern.into();
558        let has_wildcard = pattern.contains('*') || pattern.contains('?');
559        Self {
560            pattern,
561            has_wildcard,
562        }
563    }
564
565    pub fn matches(&self, name: &str) -> bool {
566        if !self.has_wildcard {
567            return self.pattern == name;
568        }
569        let p: Vec<char> = self.pattern.chars().collect();
570        let n: Vec<char> = name.chars().collect();
571        // Standard two-pointer wildcard match with backtracking on the last `*`.
572        let (mut pi, mut ni) = (0usize, 0usize);
573        let (mut star, mut mark) = (usize::MAX, 0usize);
574        while ni < n.len() {
575            // `*` before the literal comparison, or a literal `*` in the name would consume it as
576            // a wildcard (found by the `glob_match` fuzz target).
577            if pi < p.len() && p[pi] == '*' {
578                star = pi;
579                mark = ni;
580                pi += 1;
581            } else if pi < p.len() && (p[pi] == '?' || p[pi] == n[ni]) {
582                pi += 1;
583                ni += 1;
584            } else if star != usize::MAX {
585                pi = star + 1;
586                mark += 1;
587                ni = mark;
588            } else {
589                return false;
590            }
591        }
592        while pi < p.len() && p[pi] == '*' {
593            pi += 1;
594        }
595        pi == p.len()
596    }
597}
598
599/// Map a POSIX locale tag to its preset, only for an explicit `grouping = "system"`. A
600/// small static table rather than megabytes of ICU/CLDR to pick a separator.
601pub fn preset_for_locale_tag(tag: &str) -> &'static str {
602    // Strip encoding/modifier suffixes: "de_DE.UTF-8@euro" -> "de_DE"
603    let base = tag
604        .split(['.', '@'])
605        .next()
606        .unwrap_or(tag)
607        .replace('_', "-");
608    let lower = base.to_ascii_lowercase();
609    let lang = lower.split('-').next().unwrap_or(&lower);
610    let region = lower.split('-').nth(1).unwrap_or("");
611
612    // Swiss variants group with apostrophes regardless of language.
613    if region == "ch" {
614        return "swiss";
615    }
616    match lang {
617        "de" | "es" | "it" | "pt" | "nl" | "id" | "tr" | "da" | "el" | "ro" | "ca" | "vi"
618        | "sl" | "hr" | "sr" | "is" => "european",
619        "fr" | "nb" | "no" | "sv" | "fi" | "cs" | "sk" | "pl" | "ru" | "uk" | "hu" | "lv"
620        | "lt" | "et" | "bg" => "si",
621        "hi" | "bn" | "ta" | "te" | "mr" | "gu" | "kn" | "ml" | "pa" | "or" | "as" | "ne" => {
622            "indian"
623        }
624        // C / POSIX / unset and everything else: plain Western grouping.
625        _ => "thousands",
626    }
627}
628
629/// Read the environment's numeric locale, honouring POSIX precedence.
630/// Returns `None` when unset or explicitly the C/POSIX locale.
631pub fn system_locale_tag() -> Option<String> {
632    for var in ["LC_ALL", "LC_NUMERIC", "LANG"] {
633        if let Ok(v) = std::env::var(var) {
634            let v = v.trim();
635            if v.is_empty() {
636                continue;
637            }
638            if v == "C" || v == "POSIX" || v.starts_with("C.") {
639                return None;
640            }
641            return Some(v.to_string());
642        }
643    }
644    None
645}
646
647#[cfg(test)]
648mod tests;