Skip to main content

qframe/
text.rs

1//! Measuring text in terminal cells: width, truncation with an ellipsis at the end or in the
2//! middle, and word wrapping.
3
4use std::borrow::Cow;
5use std::ops::Range;
6
7use unicode_segmentation::UnicodeSegmentation;
8use unicode_width::UnicodeWidthStr;
9
10mod fuzzy;
11
12pub use fuzzy::{FuzzyMatch, fuzzy};
13
14/// The ellipsis drawn where text is cut.
15///
16/// [`truncate`] and [`truncate_middle`] always write this mark: they measure text and know
17/// nothing of the terminal. In ASCII glyph mode, where a terminal cannot show `…`, the painter
18/// ([`PaintCx::text`](crate::widget::PaintCx::text)) draws [`ASCII_ELLIPSIS`] in its cell
19/// instead, so every widget that cuts text is covered at once and the cut takes the same cell.
20pub const ELLIPSIS: &str = "…";
21
22/// The mark that stands for [`ELLIPSIS`] in ASCII glyph mode.
23///
24/// A tilde, one cell like the ellipsis it replaces, so a cut text is exactly as wide in every
25/// mode and nothing measured with [`width`] moves. It is the cut mark ASCII terminals already
26/// know from shortened file names (`PROGRA~1`, a file manager's `long~name.txt`); a period would
27/// read as the end of a sentence or, in the middle of a path, as part of the name, and `...`
28/// would take two more cells from a column that is already too narrow.
29pub const ASCII_ELLIPSIS: &str = "~";
30
31/// Display width of `text` in cells.
32#[must_use]
33pub fn width(text: &str) -> u16 {
34    let cells = if is_printable_ascii(text) { text.len() } else { text.width() };
35    u16::try_from(cells).unwrap_or(u16::MAX)
36}
37
38/// Display width of one grapheme cluster.
39#[must_use]
40pub fn grapheme_width(grapheme: &str) -> u16 {
41    width(grapheme)
42}
43
44/// Whether every character of `text` is printable ASCII, one cell and one grapheme cluster each.
45/// Most text a terminal application draws is, and it needs no Unicode tables to measure or split.
46pub(crate) fn is_printable_ascii(text: &str) -> bool {
47    text.bytes().all(|byte| matches!(byte, b' '..=b'~'))
48}
49
50/// Cells between a terminal's tab stops.
51const TAB_STOP: usize = 8;
52
53/// One line written for a terminal, as the terminal would leave it on screen: what a carriage
54/// return wrote over is gone, colour and cursor sequences are taken out, a tab becomes spaces up
55/// to the next stop of eight and no other control character is left.
56///
57/// Programs print for a terminal: a progress line redraws itself after `\r`, a build tool colours
58/// its words with escape sequences. A cell cannot hold a control character, so text from another
59/// program goes through this before it is drawn. Text with nothing to change is borrowed.
60///
61/// ```
62/// use qframe::text::printable;
63/// assert_eq!(printable("10%\r50%\r100%"), "100%");
64/// assert_eq!(printable("sent 2kB\r\r"), "sent 2kB");
65/// assert_eq!(printable("\u{1b}[1;32mok\u{1b}[0m done"), "ok done");
66/// assert_eq!(printable("a\tb"), "a       b");
67/// ```
68#[must_use]
69pub fn printable(text: &str) -> Cow<'_, str> {
70    if !text.chars().any(char::is_control) {
71        return Cow::Borrowed(text);
72    }
73    // A carriage return starts the line over; one at the very end leaves what came before it.
74    let shown = text.split('\r').rev().find(|part| !part.is_empty()).unwrap_or_default();
75    let mut out = String::with_capacity(shown.len());
76    let mut chars = shown.chars().peekable();
77    while let Some(c) = chars.next() {
78        match c {
79            // `ESC [` runs to its final byte; `ESC ]` to BEL or `ESC \`; any other escape takes
80            // its intermediate bytes and one final character, as `ESC ( B` does. A sequence cut
81            // off at the end of the line takes the rest.
82            '\u{1b}' => match chars.next() {
83                Some('[') => while chars.next().is_some_and(|c| !('@'..='~').contains(&c)) {},
84                Some(']') => {
85                    while let Some(c) = chars.next() {
86                        if c == '\u{7}' || (c == '\u{1b}' && chars.next_if_eq(&'\\').is_some()) {
87                            break;
88                        }
89                    }
90                }
91                Some(' '..='/') => {
92                    while chars.next_if(|c| (' '..='/').contains(c)).is_some() {}
93                    chars.next();
94                }
95                _ => {}
96            },
97            '\t' => {
98                let column = usize::from(width(&out));
99                out.extend(std::iter::repeat_n(' ', TAB_STOP - column % TAB_STOP));
100            }
101            c if c.is_control() => {}
102            c => out.push(c),
103        }
104    }
105    Cow::Owned(out)
106}
107
108/// `text` cut to at most `max` cells, ending in `…` when anything was removed.
109#[must_use]
110pub fn truncate(text: &str, max: u16) -> Cow<'_, str> {
111    if width(text) <= max {
112        return Cow::Borrowed(text);
113    }
114    if max == 0 {
115        return Cow::Borrowed("");
116    }
117    let budget = max - 1;
118    let mut used = 0u16;
119    let mut out = String::new();
120    for grapheme in text.graphemes(true) {
121        let w = grapheme_width(grapheme);
122        if used + w > budget {
123            break;
124        }
125        used += w;
126        out.push_str(grapheme);
127    }
128    out.push_str(ELLIPSIS);
129    Cow::Owned(out)
130}
131
132/// `text` cut to at most `max` cells by removing its middle, with `…` where the middle was.
133///
134/// For paths and other text whose start and end both matter: the head says where, the tail
135/// says what. Text that fits is returned unchanged. Otherwise the cells left after the
136/// ellipsis are shared between head and tail, the tail getting the odd one; a grapheme cluster
137/// (a wide character, a letter with its combining marks) is never split, and a cell one side
138/// cannot use goes to the other. With `max` 1 only the ellipsis remains, and with 0 nothing.
139///
140/// ```
141/// use qframe::text::{truncate_middle, width};
142///
143/// let path = "~/.config/quvyta/launcher.conf";
144/// assert_eq!(truncate_middle(path, 40), path);
145/// assert_eq!(truncate_middle(path, 20), "~/.config…ncher.conf");
146/// assert_eq!(width(&truncate_middle("~/文書/設定/launcher.conf", 12)), 12);
147/// ```
148#[must_use]
149pub fn truncate_middle(text: &str, max: u16) -> Cow<'_, str> {
150    if width(text) <= max {
151        return Cow::Borrowed(text);
152    }
153    if max == 0 {
154        return Cow::Borrowed("");
155    }
156    let budget = max - 1;
157    let graphemes: Vec<&str> = text.graphemes(true).collect();
158    let (mut head, mut head_used) = fitting(graphemes.iter(), budget / 2);
159    let (tail, tail_used) = fitting(graphemes[head..].iter().rev(), budget - head_used);
160    // A wide character at the tail's edge can leave a cell the head is able to use.
161    let (more, more_used) = fitting(graphemes[head..graphemes.len() - tail].iter(), budget - head_used - tail_used);
162    head += more;
163    head_used += more_used;
164    debug_assert!(head_used + tail_used <= budget);
165    let mut out = graphemes[..head].concat();
166    out.push_str(ELLIPSIS);
167    out.push_str(&graphemes[graphemes.len() - tail..].concat());
168    Cow::Owned(out)
169}
170
171/// How many of `graphemes`, taken in order, fit in `budget` cells, and the cells they use.
172fn fitting<'a>(graphemes: impl Iterator<Item = &'a &'a str>, budget: u16) -> (usize, u16) {
173    let mut count = 0;
174    let mut used = 0u16;
175    for grapheme in graphemes {
176        let w = grapheme_width(grapheme);
177        if used + w > budget {
178            break;
179        }
180        used += w;
181        count += 1;
182    }
183    (count, used)
184}
185
186/// Splits `text` into lines no wider than `max` cells.
187///
188/// Explicit newlines are kept, words move to the next line whole when they fit on it (with
189/// the punctuation attached to them, so a comma never starts a line), and words longer than a
190/// line are broken between grapheme clusters; the punctuation closing such a word breaks off
191/// with the character before it, so `.` or `)` never starts a line alone either. Spaces at a
192/// break are dropped; no-break spaces (U+00A0, U+202F, U+2007) are part of the word.
193///
194/// Chinese and Japanese, written without spaces, may break between any two ideographs or
195/// kana, except that closing punctuation (`。` `、` `」` `!`), the long vowel mark `ー` and
196/// small kana never start a line and opening brackets (`「` `(`) never end one. Only a run of
197/// such marks wider than the line itself breaks the rule.
198#[must_use]
199pub fn wrap(text: &str, max: u16) -> Vec<String> {
200    wrap_ranges(text, max).into_iter().map(|range| text[range].to_owned()).collect()
201}
202
203/// Like [`wrap`], but returns byte ranges into `text`, so styled text can be wrapped and drawn
204/// with its styles.
205#[must_use]
206pub fn wrap_ranges(text: &str, max: u16) -> Vec<Range<usize>> {
207    let mut lines = Vec::new();
208    if max == 0 {
209        return lines;
210    }
211    let mut paragraph_start = 0;
212    for paragraph in text.split('\n') {
213        let mut line: Option<Range<usize>> = None;
214        let mut line_width = 0u16;
215        for (offset, word, is_space) in runs(paragraph).flat_map(|(offset, run, space)| pieces(offset, run, space)) {
216            let start = paragraph_start + offset;
217            let end = start + word.len();
218            let word_width = width(word);
219            if is_space {
220                match &mut line {
221                    Some(current) if line_width + word_width <= max => {
222                        current.end = end;
223                        line_width += word_width;
224                    }
225                    Some(_) => {
226                        lines.push(trim_end(text, line.take()));
227                        line_width = 0;
228                    }
229                    None => {}
230                }
231                continue;
232            }
233            if line_width + word_width <= max {
234                line = Some(line.map_or(start..end, |current| current.start..end));
235                line_width += word_width;
236                continue;
237            }
238            if line.is_some() && word_width <= max {
239                lines.push(trim_end(text, line.take()));
240                line = Some(start..end);
241                line_width = word_width;
242                continue;
243            }
244            let graphemes: Vec<(usize, &str)> = word.grapheme_indices(true).collect();
245            // The punctuation closing the word never starts a line alone: it breaks off together
246            // with the grapheme before it, as in `ui.add(Badge::new("Paused"))` + `.`.
247            // A no-break space before that punctuation (French `vrai\u{a0}?`) goes with it too.
248            let tail =
249                graphemes.iter().rposition(|(_, g)| !is_closing_punctuation(g) && !is_no_break_space(g)).unwrap_or(0);
250            let tail_width: u16 = graphemes[tail..].iter().map(|(_, g)| grapheme_width(g)).sum();
251            for (index, (g_offset, grapheme)) in graphemes.iter().enumerate() {
252                let g_start = start + g_offset;
253                let g_end = g_start + grapheme.len();
254                let w = grapheme_width(grapheme);
255                let needed = if index == tail && tail_width <= max { tail_width } else { w };
256                if line_width + needed > max && line.is_some() {
257                    lines.push(trim_end(text, line.take()));
258                    line_width = 0;
259                }
260                line = Some(line.map_or(g_start..g_end, |current| current.start..g_end));
261                line_width += w;
262            }
263        }
264        lines.push(line.map_or(paragraph_start..paragraph_start, |current| trim_end(text, Some(current))));
265        paragraph_start += paragraph.len() + 1;
266    }
267    lines
268}
269
270/// The runs of whitespace and of everything between them, with their byte offsets and whether
271/// they are whitespace: a word keeps its punctuation (`boundaries,`, `2026.9.1`) and moves to the
272/// next line as one.
273fn runs(paragraph: &str) -> impl Iterator<Item = (usize, &str, bool)> {
274    let mut position = 0;
275    std::iter::from_fn(move || {
276        let start = position;
277        let (space, first) = char_at(paragraph, start)?;
278        position += first;
279        while let Some((_, len)) = char_at(paragraph, position).filter(|&(next, _)| next == space) {
280            position += len;
281        }
282        Some((start, &paragraph[start..position], space))
283    })
284}
285
286/// The pieces a run from [`runs`] may break between: the run itself, unless it holds Chinese or
287/// Japanese, which is written without spaces and breaks between ideographs and kana instead.
288fn pieces(offset: usize, run: &str, space: bool) -> impl Iterator<Item = (usize, &str, bool)> {
289    let mut ends = Vec::new();
290    if !space && !is_printable_ascii(run) {
291        let graphemes: Vec<(usize, &str)> = run.grapheme_indices(true).collect();
292        ends.extend(graphemes.windows(2).filter(|pair| may_break_between(pair[0].1, pair[1].1)).map(|pair| pair[1].0));
293    }
294    ends.push(run.len());
295    let mut start = 0;
296    ends.into_iter().map(move |end| {
297        let piece = (offset + start, &run[start..end], space);
298        start = end;
299        piece
300    })
301}
302
303/// Whether a line may break between two graphemes of one run: only next to Chinese or Japanese,
304/// and never before closing punctuation or after an opening bracket (the kinsoku rule).
305fn may_break_between(before: &str, after: &str) -> bool {
306    (is_cjk(before) || is_cjk(after)) && !is_closing_punctuation(after) && !is_opening_punctuation(before)
307}
308
309/// Whether `grapheme` is Chinese or Japanese text that breaks between its characters: ideographs,
310/// kana, CJK punctuation and fullwidth forms. Hangul is not: Korean separates words with spaces.
311fn is_cjk(grapheme: &str) -> bool {
312    grapheme.chars().next().is_some_and(|c| {
313        matches!(c,
314            '\u{2E80}'..='\u{2FDF}'      // radicals
315            | '\u{3000}'..='\u{30FF}'    // CJK punctuation, hiragana, katakana
316            | '\u{31C0}'..='\u{31FF}'    // strokes, katakana extensions
317            | '\u{3400}'..='\u{4DBF}'    // extension A
318            | '\u{4E00}'..='\u{9FFF}'    // unified ideographs
319            | '\u{F900}'..='\u{FAFF}'    // compatibility ideographs
320            | '\u{FE30}'..='\u{FE4F}'    // vertical and compatibility forms
321            | '\u{FF00}'..='\u{FFEF}'    // fullwidth and halfwidth forms
322            | '\u{20000}'..='\u{3FFFF}') // supplementary ideographs
323    })
324}
325
326/// Whether the character starting at byte `index` of `text` is a space a line may break at, and
327/// its length in bytes; `None` past the end. Wrapping looks at every character of a text, and
328/// ASCII, most of what is wrapped, needs no decoding.
329fn char_at(text: &str, index: usize) -> Option<(bool, usize)> {
330    let byte = *text.as_bytes().get(index)?;
331    if byte.is_ascii() {
332        return Some((char::from(byte).is_whitespace(), 1));
333    }
334    text.get(index..)?.chars().next().map(|c| (c.is_whitespace() && !is_no_break(c), c.len_utf8()))
335}
336
337/// Whether `c` is a space that binds the words on either side: French puts one before `?` and
338/// `:`, and numbers group their digits with one.
339fn is_no_break(c: char) -> bool {
340    matches!(c, '\u{A0}' | '\u{202F}' | '\u{2007}')
341}
342
343fn is_no_break_space(grapheme: &str) -> bool {
344    grapheme.chars().all(is_no_break)
345}
346
347/// Whether `grapheme` is punctuation that closes what comes before it and must not start a line.
348/// Chinese and Japanese add their own full stops, commas and brackets, the long vowel mark,
349/// iteration marks and the small kana that belong to the syllable before them.
350fn is_closing_punctuation(grapheme: &str) -> bool {
351    grapheme.chars().all(|c| {
352        matches!(
353            c,
354            '.' | ',' | ';' | ':' | '!' | '?' | ')' | ']' | '}' | '"' | '\'' | '…' | '’' | '”' | '»'
355                | '、' | '。' | '〃' | '々' | '〉' | '》' | '」' | '』' | '】' | '〕' | '〗' | '〙' | '〛' | '〞' | '〟'
356                | '〻' | '・' | 'ー' | 'ゝ' | 'ゞ' | 'ヽ' | 'ヾ' | '゛' | '゜' | '゠' | '‼' | '⁇' | '⁈' | '⁉'
357                | 'ぁ' | 'ぃ' | 'ぅ' | 'ぇ' | 'ぉ' | 'っ' | 'ゃ' | 'ゅ' | 'ょ' | 'ゎ' | 'ゕ' | 'ゖ'
358                | 'ァ' | 'ィ' | 'ゥ' | 'ェ' | 'ォ' | 'ッ' | 'ャ' | 'ュ' | 'ョ' | 'ヮ' | 'ヵ' | 'ヶ'
359                | '\u{31F0}'..='\u{31FF}'
360                | '!' | ')' | ',' | '.' | ':' | ';' | '?' | ']' | '}' | '⦆' | '。' | '」' | '、' | '・' | 'ー'
361                | 'ァ'..='ッ' | '゙' | '゚' | '%' | '〜' | '~'
362        )
363    })
364}
365
366/// Whether `grapheme` opens what comes after it and must not end a line.
367fn is_opening_punctuation(grapheme: &str) -> bool {
368    grapheme.chars().all(|c| {
369        matches!(
370            c,
371            '(' | '['
372                | '{'
373                | '‘'
374                | '“'
375                | '«'
376                | '〈'
377                | '《'
378                | '「'
379                | '『'
380                | '【'
381                | '〔'
382                | '〖'
383                | '〘'
384                | '〚'
385                | '〝'
386                | '('
387                | '['
388                | '{'
389                | '⦅'
390                | '「'
391        )
392    })
393}
394
395fn trim_end(text: &str, range: Option<Range<usize>>) -> Range<usize> {
396    let range = range.unwrap_or(0..0);
397    let trimmed = text[range.clone()].trim_end();
398    range.start..range.start + trimmed.len()
399}
400
401#[cfg(test)]
402mod tests {
403    use super::*;
404
405    #[test]
406    fn the_ascii_cut_mark_is_ascii_and_as_wide_as_the_ellipsis() {
407        assert!(is_printable_ascii(ASCII_ELLIPSIS));
408        assert_eq!(width(ASCII_ELLIPSIS), width(ELLIPSIS), "a cut text is as wide in every glyph mode");
409        assert_eq!(width(ELLIPSIS), 1);
410    }
411
412    #[test]
413    fn measures_wide_and_combining_text() {
414        assert_eq!(width("abc"), 3);
415        assert_eq!(width("çığ"), 3);
416        assert_eq!(width("界"), 2);
417        assert_eq!(width("e\u{301}"), 1);
418    }
419
420    #[test]
421    fn truncates_with_ellipsis_by_cells() {
422        assert_eq!(truncate("quvyta", 10), "quvyta");
423        assert_eq!(truncate("quvyta-framework", 8), "quvyta-…");
424        assert_eq!(truncate("界界界", 4), "界…");
425        assert_eq!(truncate("abc", 0), "");
426        assert_eq!(width(&truncate("quvyta-framework", 8)), 8);
427    }
428
429    #[test]
430    fn printable_leaves_what_a_terminal_would_show() {
431        assert!(matches!(printable("plain 防火墙"), Cow::Borrowed(_)), "nothing to change is borrowed");
432        assert_eq!(printable("\u{1b}]0;title\u{7}shown"), "shown", "a title sequence ends at the bell");
433        assert_eq!(printable("\u{1b}]8;;url\u{1b}\\link"), "link", "or at ESC backslash");
434        assert_eq!(printable("cut \u{1b}[38;2;1"), "cut ", "a sequence cut off takes the rest");
435        assert_eq!(printable("\u{1b}(Bx"), "x", "a two-character escape");
436        assert_eq!(printable("防\tx"), "防      x", "a tab counts the cells before it");
437        assert_eq!(printable("\r\r"), "", "nothing but returns leaves nothing");
438    }
439
440    #[test]
441    fn truncate_middle_returns_text_that_fits_unchanged() {
442        assert!(matches!(truncate_middle("launcher.conf", 13), Cow::Borrowed("launcher.conf")));
443        assert!(matches!(truncate_middle("", 0), Cow::Borrowed("")));
444    }
445
446    #[test]
447    fn truncate_middle_keeps_head_and_tail_of_a_path() {
448        let path = "~/.config/quvyta/launcher.conf";
449        assert_eq!(truncate_middle(path, 25), "~/.config/qu…auncher.conf");
450        assert_eq!(truncate_middle(path, 20), "~/.config…ncher.conf", "the tail gets the odd cell");
451        assert_eq!(truncate_middle(path, 5), "~/…nf");
452        for max in 0..=30 {
453            assert_eq!(width(&truncate_middle(path, max)), max, "{max}");
454        }
455    }
456
457    #[test]
458    fn truncate_middle_never_splits_wide_characters() {
459        let path = "~/文書/設定/launcher.conf";
460        assert_eq!(width(path), 25);
461        // The head cannot use its fifth cell for half of 書, so the tail takes it.
462        assert_eq!(truncate_middle(path, 12), "~/文…er.conf");
463        assert_eq!(truncate_middle("界界界界界界", 6), "界…界", "one cell stays empty rather than half a character");
464        for max in 0..=25 {
465            assert!(width(&truncate_middle(path, max)) <= max, "{max}");
466            assert!(width(&truncate_middle("界界界界界界", max)) <= max, "{max}");
467        }
468    }
469
470    #[test]
471    fn truncate_middle_keeps_combining_marks_with_their_letter() {
472        let accented = "e\u{301}e\u{301}e\u{301}e\u{301}e\u{301}";
473        assert_eq!(truncate_middle(accented, 4), "e\u{301}…e\u{301}e\u{301}");
474        assert_eq!(truncate_middle("café\u{301}s/ünïcödé\u{301}", 7), "caf…ödé\u{301}");
475    }
476
477    #[test]
478    fn truncate_middle_at_tiny_widths() {
479        assert_eq!(truncate_middle("launcher.conf", 0), "");
480        assert_eq!(truncate_middle("launcher.conf", 1), "…");
481        assert_eq!(truncate_middle("launcher.conf", 2), "…f");
482        assert_eq!(truncate_middle("文書", 2), "…", "a wide tail does not fit in one cell");
483        assert_eq!(truncate_middle("文書", 3), "…書");
484    }
485
486    #[test]
487    fn wraps_words_and_breaks_long_ones() {
488        assert_eq!(wrap("the quick brown fox", 9), vec!["the quick", "brown fox"]);
489        assert_eq!(wrap("abcdefghij", 4), vec!["abcd", "efgh", "ij"]);
490        assert_eq!(wrap("a\n\nb", 5), vec!["a", "", "b"]);
491        assert_eq!(wrap("one  two", 4), vec!["one", "two"]);
492        assert_eq!(wrap("at word boundaries, never", 18), vec!["at word", "boundaries, never"], "a comma stays");
493        assert_eq!(wrap("deploy 2026.9.1 done", 12), vec!["deploy", "2026.9.1", "done"]);
494        assert_eq!(wrap("abcdefgh.", 8), vec!["abcdefg", "h."], "a broken word keeps its full stop company");
495        assert_eq!(wrap("add abcdefghijk),", 8), vec!["add abcd", "efghij", "k),"]);
496        assert_eq!(wrap("abcdefghijk.", 4), vec!["abcd", "efgh", "ijk."]);
497        assert_eq!(wrap("........", 4), vec!["....", "...."], "all punctuation still breaks");
498        assert!(wrap("x", 0).is_empty());
499    }
500
501    #[test]
502    fn cjk_closing_punctuation_never_starts_a_line() {
503        assert_eq!(wrap("これはテストです。", 16), vec!["これはテストで", "す。"], "a full stop keeps its company");
504        assert_eq!(wrap("你好,世界", 4), vec!["你", "好,", "世界"], "an ideographic comma stays");
505        assert_eq!(
506            wrap("彼は「はい」と言った", 6),
507            vec!["彼は", "「は", "い」と", "言った"],
508            "brackets hold on to what they enclose"
509        );
510        assert_eq!(wrap("コーヒー", 4), vec!["コー", "ヒー"], "the long vowel mark stays after its kana");
511        assert_eq!(wrap("ちょっと", 6), vec!["ちょっ", "と"], "a small kana stays after the one it follows");
512        for text in ["一二三四五六七八九十、一二三。", "(全角)です!次は?", "設定を保存しました!次へ進みますか?"]
513        {
514            for max in 4..12 {
515                for line in wrap(text, max).iter().skip(1) {
516                    let first = line.graphemes(true).next().unwrap_or_default();
517                    assert!(!is_closing_punctuation(first), "{text:?} at {max}: a line starts with {first:?}");
518                }
519                for line in wrap(text, max) {
520                    let last = line.graphemes(true).next_back().unwrap_or_default();
521                    assert!(
522                        line.graphemes(true).count() == 1 || !is_opening_punctuation(last),
523                        "{text:?} at {max}: a line ends with {last:?}"
524                    );
525                }
526            }
527        }
528    }
529
530    #[test]
531    fn cjk_text_without_spaces_breaks_between_ideographs() {
532        assert_eq!(wrap("防火墙已启用", 4), vec!["防火", "墙已", "启用"]);
533        assert_eq!(wrap("状态 防火墙已启用", 10), vec!["状态 防火", "墙已启用"], "the rest of a line is filled");
534        assert_eq!(wrap("hello 你好世界", 8), vec!["hello 你", "好世界"]);
535        assert_eq!(wrap("Rust で書く", 7), vec!["Rust で", "書く"]);
536        assert_eq!(wrap("パッケージを更新", 10), vec!["パッケージ", "を更新"]);
537        assert_eq!(wrap("안녕하세요 세계", 10), vec!["안녕하세요", "세계"], "Korean words stay whole");
538    }
539
540    #[test]
541    fn no_break_spaces_belong_to_the_word() {
542        assert_eq!(wrap("Est-ce vrai\u{a0}? Oui", 11), vec!["Est-ce", "vrai\u{a0}? Oui"]);
543        assert_eq!(wrap("Attention\u{202f}: fin", 10), vec!["Attentio", "n\u{202f}: fin"]);
544        assert_eq!(wrap("total 10\u{2007}000 kr", 8), vec!["total", "10\u{2007}000", "kr"]);
545        assert_eq!(
546            wrap("Vraiment\u{a0}?", 9),
547            vec!["Vraimen", "t\u{a0}?"],
548            "a broken word keeps its space with the mark"
549        );
550        assert_eq!(wrap("a b\u{a0}c", 3), vec!["a", "b\u{a0}c"]);
551    }
552
553    /// Pins wrapping, truncation and width of text the ASCII fast paths do not take: other
554    /// whitespace, control characters, wide and combining characters, emoji sequences.
555    #[test]
556    fn unusual_text_measures_and_wraps_as_before() {
557        /// Text, width, its lines, their ranges, and the text truncated to the width.
558        type Case = (&'static str, u16, &'static [&'static str], &'static [Range<usize>], &'static str);
559        let cases: [Case; 12] = [
560            ("a\u{a0}b c\u{a0}\u{a0}dd", 3, &["a\u{a0}b", "c", "dd"], &[0..4, 5..6, 10..12], "a\u{a0}…"),
561            ("x\u{3000}y z", 2, &["x", "y", "z"], &[0..1, 4..5, 6..7], "x…"),
562            ("tab\there and\u{b}vt", 4, &["tab", "here", "and", "vt"], &[0..3, 4..8, 9..12, 13..15], "tab…"),
563            (
564                "界界 界界界 e\u{301}e\u{301}e\u{301}",
565                3,
566                &["界", "界", "界", "界", "界", "e\u{301}e\u{301}e\u{301}"],
567                &[0..3, 3..6, 7..10, 10..13, 13..16, 17..26],
568                "界…",
569            ),
570            ("  lead  and trail  ", 5, &["lead", "and", "trail", ""], &[2..6, 8..11, 12..17, 0..0], "  le…"),
571            ("😀😀 ok", 3, &["😀", "😀", "ok"], &[0..4, 4..8, 9..11], "😀…"),
572            ("a\r\nb c", 2, &["a", "b", "c"], &[0..1, 3..4, 5..6], "a…"),
573            ("über straße ünïcödé", 6, &["über", "straße", "ünïcöd", "é"], &[0..5, 6..13, 14..23, 23..25], "über …"),
574            ("x\u{85}y\u{2028}z", 1, &["x", "y", "z"], &[0..1, 3..4, 7..8], "…"),
575            ("control\u{7}bell word", 8, &["control\u{7}", "bell", "word"], &[0..8, 8..12, 13..17], "control…"),
576            (
577                "👨\u{200d}👩\u{200d}👧 family",
578                4,
579                &["👨\u{200d}👩\u{200d}👧 f", "amil", "y"],
580                &[0..20, 20..24, 24..25],
581                "👨\u{200d}👩\u{200d}👧 …",
582            ),
583            ("add abcdefghijk),", 8, &["add abcd", "efghij", "k),"], &[0..8, 8..14, 14..17], "add abc…"),
584        ];
585        for (text, max, lines, ranges, truncated) in cases {
586            assert_eq!(wrap(text, max), lines, "{text:?}");
587            assert_eq!(wrap_ranges(text, max), ranges, "{text:?}");
588            assert_eq!(truncate(text, max), truncated, "{text:?}");
589        }
590        let widths = ["\t", "\u{7}", "\u{b}", "\r\n", "\u{a0}", "~", " ", "👨\u{200d}👩", "\u{7f}", ""].map(width);
591        assert_eq!(widths, [1, 1, 1, 1, 1, 1, 1, 2, 1, 0]);
592    }
593}