Skip to main content

codoseo_web/
serp.rs

1//! How wide a title and description render in Google's results, in pixels, so the explorer
2//! can show where a snippet gets cut. Google sets titles in Arial 20 px and descriptions in
3//! Arial 14 px; the widths come from Arial's advance widths (2048 units per em), grouped by
4//! width, with a sensible default for characters outside the table.
5
6/// Where Google cuts a title on desktop.
7pub const TITLE_LIMIT_PX: u32 = 580;
8/// Where Google cuts a description on desktop (about two lines).
9pub const DESCRIPTION_LIMIT_PX: u32 = 990;
10
11const UNITS_PER_EM: u64 = 2048;
12
13/// Width of `text` set as a result title (Arial 20 px).
14pub fn title_px(text: &str) -> u32 {
15    px(text, 20)
16}
17
18/// Width of `text` set as a result description (Arial 14 px).
19pub fn description_px(text: &str) -> u32 {
20    px(text, 14)
21}
22
23fn px(text: &str, size: u64) -> u32 {
24    let units: u64 = text.chars().map(|c| u64::from(advance(c))).sum();
25    // Rounded to the nearest pixel.
26    ((units * size + UNITS_PER_EM / 2) / UNITS_PER_EM).min(u64::from(u32::MAX)) as u32
27}
28
29/// Arial's advance width for `c`, in units of 1/2048 em.
30fn advance(c: char) -> u16 {
31    match c {
32        'i' | 'j' | 'l' => 455,
33        ' ' | '\u{a0}' | '!' | ',' | '.' | '/' | ':' | ';' | '[' | '\\' | ']' | 'f' | 't' | 'I' => {
34            569
35        }
36        '(' | ')' | '-' | '`' | 'r' => 682,
37        'c' | 'k' | 's' | 'v' | 'x' | 'y' | 'z' | 'J' => 1024,
38        '0'..='9'
39        | 'a'
40        | 'b'
41        | 'd'
42        | 'e'
43        | 'g'
44        | 'h'
45        | 'n'
46        | 'o'
47        | 'p'
48        | 'q'
49        | 'u'
50        | '#'
51        | '$'
52        | '?'
53        | '_'
54        | 'L' => 1139,
55        '+' | '<' | '=' | '>' | '~' | '×' => 1196,
56        'F' | 'T' | 'Z' => 1251,
57        'A' | 'B' | 'E' | 'K' | 'P' | 'S' | 'V' | 'X' | 'Y' | '&' => 1366,
58        'C' | 'D' | 'H' | 'N' | 'R' | 'U' | 'w' => 1479,
59        'G' | 'O' | 'Q' => 1593,
60        'M' | 'm' => 1706,
61        '%' => 1821,
62        'W' => 1933,
63        '@' => 2079,
64        '\'' => 391,
65        '|' => 532,
66        '{' | '}' => 684,
67        '"' => 727,
68        '*' => 797,
69        '^' => 961,
70        // Typographic punctuation and symbols common in titles.
71        '‘' | '’' | '‚' => 455,
72        '“' | '”' | '„' | '·' => 683,
73        '•' => 717,
74        '–' | '€' | '£' | '¥' => 1139,
75        '©' | '®' => 1509,
76        '—' | '…' | '™' => 2048,
77        // Accented Latin letters are as wide as their base letter; the narrow i forms aside,
78        // the case averages are close enough.
79        'ì' | 'í' | 'î' | 'ï' | 'ı' | 'Ì' | 'Í' | 'Î' | 'Ï' => 569,
80        c if is_wide(c) => 2048,
81        c if c.is_uppercase() => 1430,
82        _ => 1139,
83    }
84}
85
86/// CJK, Hangul, full-width forms and emoji: one em each.
87fn is_wide(c: char) -> bool {
88    matches!(
89        c as u32,
90        0x1100..=0x115F
91            | 0x2E80..=0xA4CF
92            | 0xAC00..=0xD7A3
93            | 0xF900..=0xFAFF
94            | 0xFE30..=0xFE4F
95            | 0xFF00..=0xFF60
96            | 0xFFE0..=0xFFE6
97            | 0x1F300..=0x1FAFF
98            | 0x20000..=0x3FFFD
99    )
100}
101
102/// Cuts `text` so it fits in `limit` pixels as measured by `f`, the way Google does: at a
103/// word boundary when that keeps most of the text, followed by `" …"`. Returns the text and
104/// whether it was cut. The result, ellipsis included, never measures more than `limit`.
105pub fn truncate_to_px(text: &str, limit: u32, f: impl Fn(&str) -> u32) -> (String, bool) {
106    const ELLIPSIS: &str = " …";
107    let text = text.trim();
108    if f(text) <= limit {
109        return (text.to_owned(), false);
110    }
111    let bounds: Vec<usize> = text
112        .char_indices()
113        .map(|(i, _)| i)
114        .chain(std::iter::once(text.len()))
115        .collect();
116    let cut_at = |k: usize| text[..bounds[k]].trim_end();
117    let fits = |k: usize| f(&format!("{}{ELLIPSIS}", cut_at(k))) <= limit;
118    if !fits(0) {
119        return (String::new(), true);
120    }
121    // The longest prefix that still fits; widths only grow as the prefix does.
122    let (mut lo, mut hi) = (0, bounds.len() - 1);
123    while lo < hi {
124        let mid = (lo + hi).div_ceil(2);
125        if fits(mid) {
126            lo = mid;
127        } else {
128            hi = mid - 1;
129        }
130    }
131    let mut cut = cut_at(lo);
132    if let Some(space) = cut.rfind(char::is_whitespace)
133        && space >= cut.len() * 2 / 3
134    {
135        cut = cut[..space].trim_end();
136    }
137    (format!("{cut}{ELLIPSIS}"), true)
138}
139
140#[cfg(test)]
141mod tests {
142    use super::*;
143
144    #[test]
145    fn a_typical_sixty_character_title_is_close_to_the_limit() {
146        let title = "How to Choose Running Shoes for Flat Feet - A Complete Guide";
147        assert_eq!(title.chars().count(), 60);
148        let w = title_px(title);
149        assert!((530..=630).contains(&w), "{w}");
150    }
151
152    #[test]
153    fn wide_letters_are_wider_than_narrow_ones() {
154        assert!(title_px("WWWW") > title_px("iiii"));
155        assert!(description_px("WWWW") > description_px("iiii"));
156        assert_eq!(title_px(""), 0);
157        // "W" is 1933/2048 em: about 19 px at 20 px.
158        assert_eq!(title_px("W"), 19);
159    }
160
161    #[test]
162    fn descriptions_are_set_smaller_than_titles() {
163        let s = "A meta description that is comfortably long enough to pass the check.";
164        let (t, d) = (title_px(s), description_px(s));
165        assert!(d < t);
166        let expected = t * 14 / 20;
167        assert!(d.abs_diff(expected) <= 2, "{d} vs {expected}");
168    }
169
170    #[test]
171    fn unknown_characters_get_a_width() {
172        assert!(title_px("東京") > title_px("ab"));
173        assert!(title_px("Émile") > 0);
174        assert_eq!(title_px("é"), title_px("e"));
175    }
176
177    #[test]
178    fn short_text_is_left_alone() {
179        let (out, cut) = truncate_to_px("  Short title ", TITLE_LIMIT_PX, title_px);
180        assert_eq!(out, "Short title");
181        assert!(!cut);
182    }
183
184    #[test]
185    fn truncation_never_exceeds_the_limit() {
186        let long = "The Complete Beginner's Guide to Growing Tomatoes Indoors All Year Round \
187                    With Hydroponics, Grow Lights and Very Little Space";
188        for limit in [0, 5, 20, 40, 100, 250, 400, 579, 580, 600, 990] {
189            for f in [title_px as fn(&str) -> u32, description_px] {
190                let (out, cut) = truncate_to_px(long, limit, f);
191                assert!(f(&out) <= limit, "{limit}: {out:?} is {}", f(&out));
192                if f(long) > limit {
193                    assert!(cut);
194                    assert!(out.is_empty() || out.ends_with(" …"), "{out:?}");
195                }
196            }
197        }
198        let (out, cut) = truncate_to_px(long, TITLE_LIMIT_PX, title_px);
199        assert!(cut);
200        assert!(out.starts_with("The Complete Beginner's Guide"));
201        // Cut at a word boundary, not mid-word.
202        let kept = out.trim_end_matches(" …");
203        assert!(long.starts_with(kept));
204        assert!(long[kept.len()..].starts_with(' '), "{out:?}");
205    }
206
207    #[test]
208    fn one_long_word_is_cut_mid_word() {
209        let word = "a".repeat(200);
210        let (out, cut) = truncate_to_px(&word, 300, title_px);
211        assert!(cut);
212        // "a" is about 11 px and " …" about 26 px: some 24 letters fit in 300 px.
213        assert!(out.starts_with(&"a".repeat(20)), "{out:?}");
214        assert!(title_px(&out) <= 300);
215    }
216}