Skip to main content

mdwire/
width.rs

1//! Display width — the number of cells a character takes in a monospace font.
2//!
3//! **Table columns are aligned by display width, not character count** (`SPEC.md` section 7).
4//! Hangul, Han characters, kana, fullwidth symbols, and emoji take two cells. Aligning with
5//! `chars().count()` always misaligns a table that contains Hangul — not occasionally, but by
6//! one cell for every Hangul character in a column.
7//!
8//! Unicode East Asian Width `W` and `F` count as width 2; combining characters and zero-width
9//! control characters count as width 0. The core has no dependencies, so it carries its own
10//! range tables (`AGENTS.md`). The wide ranges are few enough that hardcoding them stays
11//! maintainable.
12
13/// 폭 0 — 결합 문자, 폭 없는 공백, 변형 선택자.
14const ZERO: &[(u32, u32)] = &[
15    (0x0300, 0x036F), // 결합 발음 기호
16    (0x0483, 0x0489),
17    (0x0591, 0x05BD),
18    (0x0610, 0x061A),
19    (0x064B, 0x065F),
20    (0x0670, 0x0670),
21    (0x06D6, 0x06DC),
22    (0x0E31, 0x0E31),
23    (0x0E34, 0x0E3A),
24    (0x0E47, 0x0E4E),
25    (0x1160, 0x11FF), // 한글 중성·종성 자모(조합용). 초성에 붙어 폭을 더하지 않는다
26    (0x135D, 0x135F),
27    (0x1AB0, 0x1AFF),
28    (0x1DC0, 0x1DFF),
29    (0x200B, 0x200F), // ZWSP·ZWNJ·ZWJ·방향 표시
30    (0x2028, 0x202E),
31    (0x2060, 0x2064),
32    (0x20D0, 0x20F0),
33    (0xFE00, 0xFE0F), // 변형 선택자
34    (0xFE20, 0xFE2F),
35    (0xFEFF, 0xFEFF),
36    (0x1D165, 0x1D169),
37    (0x1D16D, 0x1D172),
38    (0xE0100, 0xE01EF),
39];
40
41/// 폭 2 — East Asian Width 가 Wide 또는 Fullwidth 인 구간.
42const WIDE: &[(u32, u32)] = &[
43    (0x1100, 0x115F), // 한글 초성 자모
44    (0x231A, 0x231B),
45    (0x2329, 0x232A),
46    (0x23E9, 0x23EC),
47    (0x23F0, 0x23F0),
48    (0x23F3, 0x23F3),
49    (0x25FD, 0x25FE),
50    (0x2614, 0x2615),
51    (0x2648, 0x2653),
52    (0x267F, 0x267F),
53    (0x2693, 0x2693),
54    (0x26A1, 0x26A1),
55    (0x26AA, 0x26AB),
56    (0x26BD, 0x26BE),
57    (0x26C4, 0x26C5),
58    (0x26CE, 0x26CE),
59    (0x26D4, 0x26D4),
60    (0x26EA, 0x26EA),
61    (0x26F2, 0x26F3),
62    (0x26F5, 0x26F5),
63    (0x26FA, 0x26FA),
64    (0x26FD, 0x26FD),
65    (0x2705, 0x2705),
66    (0x270A, 0x270B),
67    (0x2728, 0x2728),
68    (0x274C, 0x274C),
69    (0x274E, 0x274E),
70    (0x2753, 0x2755),
71    (0x2757, 0x2757),
72    (0x2795, 0x2797),
73    (0x27B0, 0x27B0),
74    (0x27BF, 0x27BF),
75    (0x2B1B, 0x2B1C),
76    (0x2B50, 0x2B50),
77    (0x2B55, 0x2B55),
78    (0x2E80, 0x2E99),
79    (0x2E9B, 0x2EF3),
80    (0x2F00, 0x2FD5),
81    (0x2FF0, 0x2FFB),
82    (0x3000, 0x303E), // 전각 공백 · CJK 구두점
83    (0x3041, 0x3096), // 히라가나
84    (0x3099, 0x30FF), // 가타카나
85    (0x3105, 0x312F),
86    (0x3131, 0x318E), // 한글 호환 자모
87    (0x3190, 0x31E3),
88    (0x31F0, 0x321E),
89    (0x3220, 0x3247),
90    (0x3250, 0x4DBF),
91    (0x4E00, 0xA48C), // 한중일 통합 한자
92    (0xA490, 0xA4C6),
93    (0xA960, 0xA97C),
94    (0xAC00, 0xD7A3), // 한글 음절 — 한국어 표의 거의 전부가 여기다
95    (0xF900, 0xFAFF),
96    (0xFE10, 0xFE19),
97    (0xFE30, 0xFE52),
98    (0xFE54, 0xFE66),
99    (0xFE68, 0xFE6B),
100    (0xFF01, 0xFF60), // 전각 영숫자·기호
101    (0xFFE0, 0xFFE6),
102    (0x16FE0, 0x16FE4),
103    (0x17000, 0x18CD5),
104    (0x1B000, 0x1B152),
105    (0x1B164, 0x1B167),
106    (0x1B170, 0x1B2FB),
107    (0x1F004, 0x1F004),
108    (0x1F0CF, 0x1F0CF),
109    (0x1F18E, 0x1F18E),
110    (0x1F191, 0x1F19A),
111    (0x1F200, 0x1F320),
112    (0x1F32D, 0x1F335),
113    (0x1F337, 0x1F37C),
114    (0x1F37E, 0x1F393),
115    (0x1F3A0, 0x1F3CA),
116    (0x1F3CF, 0x1F3D3),
117    (0x1F3E0, 0x1F3F0),
118    (0x1F3F4, 0x1F3F4),
119    (0x1F3F8, 0x1F43E),
120    (0x1F440, 0x1F440),
121    (0x1F442, 0x1F4FC),
122    (0x1F4FF, 0x1F53D),
123    (0x1F54B, 0x1F54E),
124    (0x1F550, 0x1F567),
125    (0x1F57A, 0x1F57A),
126    (0x1F595, 0x1F596),
127    (0x1F5A4, 0x1F5A4),
128    (0x1F5FB, 0x1F64F),
129    (0x1F680, 0x1F6C5),
130    (0x1F6CC, 0x1F6CC),
131    (0x1F6D0, 0x1F6D2),
132    (0x1F6D5, 0x1F6D7),
133    (0x1F6EB, 0x1F6EC),
134    (0x1F6F4, 0x1F6FC),
135    (0x1F7E0, 0x1F7EB),
136    (0x1F90C, 0x1F93A),
137    (0x1F93C, 0x1F945),
138    (0x1F947, 0x1F978),
139    (0x1F97A, 0x1F9CB),
140    (0x1F9CD, 0x1F9FF),
141    (0x1FA70, 0x1FA74),
142    (0x1FA78, 0x1FA7A),
143    (0x1FA80, 0x1FA86),
144    (0x1FA90, 0x1FAA8),
145    (0x1FAB0, 0x1FAB6),
146    (0x1FAC0, 0x1FAC2),
147    (0x1FAD0, 0x1FAD6),
148    (0x20000, 0x2FFFD),
149    (0x30000, 0x3FFFD),
150];
151
152/// The display width of one character: 0, 1, or 2.
153pub fn char_width(c: char) -> usize {
154    let cp = c as u32;
155    if cp == 0 {
156        return 0;
157    }
158    if cp < 0x20 || (0x7F..0xA0).contains(&cp) {
159        // 제어 문자. 표에 들어오면 폭 계산이 무너지므로 0 으로 센다.
160        return 0;
161    }
162    if cp < 0x300 {
163        // 라틴 상용 구간. 표 대부분이 여기라 먼저 끊는다.
164        return 1;
165    }
166    if in_ranges(ZERO, cp) {
167        return 0;
168    }
169    if in_ranges(WIDE, cp) {
170        return 2;
171    }
172    1
173}
174
175/// Whether this is a "CJK character" as the CJK-adjacent emphasis policy sees it.
176///
177/// **This is a different question from display width.** Using one function for both gets two
178/// cases wrong — halfwidth katakana (`アイウ`) has width 1 but is CJK and needs padding, while
179/// emoji have width 2 but are not CJK and need no padding.
180///
181/// The rule follows the CJK-friendly amendment to CommonMark — a character is CJK if its East
182/// Asian Width is `W`, `F`, or `H` and it is not emoji presentation, or if its script is Hangul.
183/// (<https://github.com/tats-u/markdown-cjk-friendly>)
184pub fn is_cjk(c: char) -> bool {
185    in_ranges(CJK, c as u32)
186}
187
188/// CJK 구간. 한자·가나·한글(조합형 자모 포함)·전각/반각 CJK 기호.
189/// **이모지와 기호는 일부러 뺐다** — 폭이 2여도 CJK 가 아니다.
190const CJK: &[(u32, u32)] = &[
191    (0x1100, 0x11FF), // 한글 자모(조합형). NFD 로 분해된 한글이 여기다
192    (0x2E80, 0x2EF3), // CJK 부수
193    (0x2F00, 0x2FD5), // 강희 부수
194    (0x3000, 0x303F), // CJK 구두점 — `。` `、` `「」` 가 여기다
195    (0x3041, 0x30FF), // 히라가나 · 가타카나
196    (0x3105, 0x312F),
197    (0x3131, 0x318E), // 한글 호환 자모
198    (0x3190, 0x31E3),
199    (0x31F0, 0x321E),
200    (0x3220, 0x3247),
201    (0x3250, 0x4DBF),
202    (0x4E00, 0x9FFF), // 한중일 통합 한자
203    (0xA960, 0xA97C), // 한글 자모 확장 A
204    (0xAC00, 0xD7A3), // 한글 음절
205    (0xD7B0, 0xD7FB), // 한글 자모 확장 B
206    (0xF900, 0xFAFF), // 호환 한자
207    (0xFE10, 0xFE19),
208    (0xFE30, 0xFE6B), // 세로쓰기 형태 · 전각 기호
209    (0xFF01, 0xFF60), // 전각 영숫자·기호 — `()` `,` 가 여기다
210    (0xFF61, 0xFFDC), // **반각** 가타카나·한글. 폭은 1이지만 CJK 다
211    (0xFFE0, 0xFFE6), // 전각 통화 기호 — `¥` `₩`. W/F/H 기준대로 CJK 다
212    (0x20000, 0x2FFFD),
213    (0x30000, 0x3FFFD),
214];
215
216/// The display width of a string.
217///
218/// Emoji sequences (such as family emoji joined with ZWJ) count each component separately, so
219/// the result can be wider than what is displayed. That errs toward extra room rather than a
220/// misaligned table, so v0.1 leaves it as is.
221pub fn str_width(s: &str) -> usize {
222    s.chars().map(char_width).sum()
223}
224
225fn in_ranges(table: &[(u32, u32)], cp: u32) -> bool {
226    table
227        .binary_search_by(|&(lo, hi)| {
228            if cp < lo {
229                std::cmp::Ordering::Greater
230            } else if cp > hi {
231                std::cmp::Ordering::Less
232            } else {
233                std::cmp::Ordering::Equal
234            }
235        })
236        .is_ok()
237}
238
239#[cfg(test)]
240mod tests {
241    use super::*;
242
243    #[test]
244    fn tables_are_sorted_and_disjoint() {
245        // 이진 탐색의 전제. 깨지면 조용히 틀린 폭이 나온다.
246        for table in [ZERO, WIDE] {
247            for w in table.windows(2) {
248                assert!(w[0].1 < w[1].0, "구간이 겹치거나 순서가 틀렸다: {:?}", w);
249            }
250            for &(lo, hi) in table {
251                assert!(lo <= hi);
252            }
253        }
254    }
255
256    #[test]
257    fn known_widths() {
258        for (c, w) in [
259            ('a', 1),
260            ('9', 1),
261            (' ', 1),
262            ('|', 1),
263            ('가', 2),
264            ('힣', 2),
265            ('漢', 2),
266            ('あ', 2),
267            ('ア', 2),
268            (',', 2),
269            (' ', 2), // 전각 공백
270            ('A', 2), // 전각 영문
271            ('✅', 2),
272            ('🚀', 2),
273            ('\u{200b}', 0), // ZWSP
274            ('\u{0301}', 0), // 결합 악센트
275            ('\n', 0),
276        ] {
277            assert_eq!(char_width(c), w, "{c:?} 의 폭이 {w} 가 아니다");
278        }
279    }
280
281    #[test]
282    fn korean_text_is_twice_its_char_count() {
283        let s = "한글";
284        assert_eq!(s.chars().count(), 2);
285        assert_eq!(str_width(s), 4, "문자 수로 맞추면 표가 어긋난다");
286    }
287
288    #[test]
289    fn cjk_is_not_the_same_question_as_width() {
290        // 폭 1인데 CJK — 반각 가타카나. 패딩이 필요하다
291        assert_eq!(char_width('ア'), 1);
292        assert!(is_cjk('ア'));
293        // 폭 2인데 CJK 아님 — 이모지. 패딩이 필요 없다
294        assert_eq!(char_width('🚀'), 2);
295        assert!(!is_cjk('🚀'));
296        assert!(!is_cjk('✅'));
297        // 조합형 한글(NFD). 폭 0인 중성·종성도 CJK 다
298        assert!(is_cjk('\u{1100}') && is_cjk('\u{1161}'));
299        // 전각 통화 기호도 전각(F)이라 CJK 다. 분리하면서 빠뜨리기 쉬운 자리다.
300        for c in ['¥', '₩', '£'] {
301            assert!(is_cjk(c), "{c} 가 CJK 로 안 잡힌다");
302        }
303        // CJK 구두점 — 개정안이 다루는 바로 그 글자들
304        for c in ['。', '、', ',', '「', '」', '(', ')'] {
305            assert!(is_cjk(c), "{c} 가 CJK 로 안 잡힌다");
306        }
307        // 라틴은 아니다
308        for c in ['a', '1', ' ', '.', '-'] {
309            assert!(!is_cjk(c), "{c} 가 CJK 로 잡힌다");
310        }
311    }
312
313    #[test]
314    fn cjk_table_is_sorted_and_disjoint() {
315        for w in CJK.windows(2) {
316            assert!(w[0].1 < w[1].0, "구간이 겹치거나 순서가 틀렸다: {:?}", w);
317        }
318    }
319
320    #[test]
321    fn mixed_width_adds_up() {
322        assert_eq!(str_width("환경 env"), 2 + 2 + 1 + 3);
323    }
324}