Skip to main content

forme/font/
subset.rs

1//! # TrueType Font Subsetter
2//!
3//! Strips a TrueType font to only the glyphs actually used in the document.
4//! This dramatically reduces PDF size — a typical font is 50-200KB but a
5//! subset with ~100 glyphs is usually 5-15KB.
6//!
7//! The subsetter rebuilds a valid TrueType file with remapped glyph IDs
8//! (contiguous starting from 0). This is important because PDF CIDFont
9//! width arrays and content stream glyph references must use the new IDs.
10//!
11//! ## Approach
12//!
13//! 1. Collect all needed glyphs (used glyphs + composite glyph dependencies)
14//! 2. Remap old GIDs to new contiguous GIDs
15//! 3. Rebuild required TrueType tables (glyf, loca, hmtx, cmap, etc.)
16//! 4. Write a valid TrueType file with correct checksums and alignment
17
18use std::collections::{BTreeSet, HashMap};
19
20/// Result of subsetting a font.
21pub struct SubsetResult {
22    /// The subset TrueType file bytes.
23    pub ttf_data: Vec<u8>,
24    /// Maps original glyph IDs to new contiguous glyph IDs.
25    pub gid_remap: HashMap<u16, u16>,
26}
27
28/// Subset a TrueType font to only include the given glyph IDs.
29pub fn subset_ttf(
30    ttf_data: &[u8],
31    used_gids: &std::collections::HashSet<u16>,
32) -> Result<SubsetResult, String> {
33    let face = ttf_parser::Face::parse(ttf_data, 0)
34        .map_err(|e| format!("Failed to parse TTF: {:?}", e))?;
35
36    // Always include glyph 0 (.notdef)
37    let mut needed_gids: BTreeSet<u16> = BTreeSet::new();
38    needed_gids.insert(0);
39    for &gid in used_gids {
40        needed_gids.insert(gid);
41    }
42
43    // Resolve composite glyph dependencies
44    let raw_glyf = find_table(ttf_data, b"glyf").ok_or("Missing glyf table")?;
45    let raw_loca = find_table(ttf_data, b"loca").ok_or("Missing loca table")?;
46    let head = find_table(ttf_data, b"head").ok_or("Missing head table")?;
47
48    let num_glyphs = face.number_of_glyphs();
49    let loca_format = read_i16(head, 50); // indexToLocFormat at offset 50
50    let loca_offsets = parse_loca(raw_loca, loca_format, num_glyphs)?;
51
52    // Recursively collect composite glyph component GIDs
53    let initial_gids: Vec<u16> = needed_gids.iter().copied().collect();
54    for gid in initial_gids {
55        collect_composite_deps(raw_glyf, &loca_offsets, gid, &mut needed_gids);
56    }
57
58    // Build remap: old GID → new contiguous GID
59    let mut gid_remap: HashMap<u16, u16> = HashMap::new();
60    for (new_gid, &old_gid) in needed_gids.iter().enumerate() {
61        gid_remap.insert(old_gid, new_gid as u16);
62    }
63
64    let new_num_glyphs = needed_gids.len() as u16;
65
66    // Rebuild glyf table with remapped composite references
67    let (new_glyf, new_loca_offsets) =
68        rebuild_glyf(raw_glyf, &loca_offsets, &needed_gids, &gid_remap);
69
70    // Determine loca format based on glyf size
71    let new_loca_format: i16 = if new_glyf.len() > 0x1FFFE { 1 } else { 0 };
72    let new_loca = build_loca(&new_loca_offsets, new_loca_format);
73
74    // Rebuild hmtx (horizontal metrics)
75    let raw_hmtx = find_table(ttf_data, b"hmtx").ok_or("Missing hmtx table")?;
76    let raw_hhea = find_table(ttf_data, b"hhea").ok_or("Missing hhea table")?;
77    let num_h_metrics = read_u16(raw_hhea, 34) as usize;
78    let new_hmtx = rebuild_hmtx(raw_hmtx, &needed_gids, num_h_metrics);
79
80    // Build minimal cmap (Format 4)
81    // We need the original char→gid mapping — invert through the face
82    let gid_to_code = lowest_bmp_code_per_glyph(&face);
83    let mut char_to_new_gid: Vec<(u16, u16)> = Vec::new();
84    for &old_gid in &needed_gids {
85        if old_gid == 0 {
86            continue;
87        }
88        if let (Some(&code), Some(&new_gid)) = (gid_to_code.get(&old_gid), gid_remap.get(&old_gid))
89        {
90            char_to_new_gid.push((code, new_gid));
91        }
92    }
93    let new_cmap = build_cmap_format4(&char_to_new_gid);
94
95    // Copy or rebuild remaining required tables
96    let new_head = rebuild_head(head, new_loca_format);
97
98    let raw_hhea_data = raw_hhea.to_vec();
99    let new_hhea = rebuild_hhea(&raw_hhea_data, new_num_glyphs);
100
101    let new_maxp = build_maxp(new_num_glyphs);
102    let new_post = build_post_format3();
103
104    // Copy name table verbatim if present (or build minimal)
105    let new_name = find_table(ttf_data, b"name")
106        .map(|t| t.to_vec())
107        .unwrap_or_else(|| build_minimal_name(&face));
108
109    // Copy OS/2 table verbatim if present
110    let new_os2 = find_table(ttf_data, b"OS/2").map(|t| t.to_vec());
111
112    // Copy hinting tables verbatim if present
113    let cvt_data = find_table(ttf_data, b"cvt ").map(|t| t.to_vec());
114    let fpgm_data = find_table(ttf_data, b"fpgm").map(|t| t.to_vec());
115    let prep_data = find_table(ttf_data, b"prep").map(|t| t.to_vec());
116
117    // Assemble the final TrueType file
118    let mut tables: Vec<(u32, Vec<u8>)> = Vec::new();
119    tables.push((tag_u32(b"cmap"), new_cmap));
120    if let Some(cvt) = cvt_data {
121        tables.push((tag_u32(b"cvt "), cvt));
122    }
123    if let Some(fpgm) = fpgm_data {
124        tables.push((tag_u32(b"fpgm"), fpgm));
125    }
126    tables.push((tag_u32(b"glyf"), new_glyf));
127    tables.push((tag_u32(b"head"), new_head));
128    tables.push((tag_u32(b"hhea"), new_hhea));
129    tables.push((tag_u32(b"hmtx"), new_hmtx));
130    tables.push((tag_u32(b"loca"), new_loca));
131    tables.push((tag_u32(b"maxp"), new_maxp));
132    tables.push((tag_u32(b"name"), new_name));
133    if let Some(os2) = new_os2 {
134        tables.push((tag_u32(b"OS/2"), os2));
135    }
136    tables.push((tag_u32(b"post"), new_post));
137    if let Some(prep) = prep_data {
138        tables.push((tag_u32(b"prep"), prep));
139    }
140
141    // Sort tables by tag (required by TrueType spec for binary search)
142    tables.sort_by_key(|(tag, _)| *tag);
143
144    let output = write_ttf_file(&mut tables);
145
146    Ok(SubsetResult {
147        ttf_data: output,
148        gid_remap,
149    })
150}
151
152// ─── Table Locating ─────────────────────────────────────────────
153
154fn find_table<'a>(data: &'a [u8], tag: &[u8; 4]) -> Option<&'a [u8]> {
155    if data.len() < 12 {
156        return None;
157    }
158    let num_tables = read_u16(data, 4) as usize;
159    for i in 0..num_tables {
160        let offset = 12 + i * 16;
161        if offset + 16 > data.len() {
162            break;
163        }
164        if &data[offset..offset + 4] == tag {
165            let table_offset = read_u32(data, offset + 8) as usize;
166            let table_length = read_u32(data, offset + 12) as usize;
167            if table_offset + table_length <= data.len() {
168                return Some(&data[table_offset..table_offset + table_length]);
169            }
170        }
171    }
172    None
173}
174
175// ─── Loca Table Parsing ─────────────────────────────────────────
176
177/// Each glyph's lowest BMP codepoint, inverted from the font's Unicode cmap
178/// in one pass. Searching 0..=0xFFFF per glyph instead cost up to 65,536
179/// lookups per glyph: Hangul starts at U+AC00, and a glyph with no codepoint
180/// (a ligature or shaped form) scanned the whole plane, so a Korean document
181/// with 1,000 distinct syllables spent seconds here (#182). The lowest
182/// codepoint is kept so the subset's cmap is the same as before.
183fn lowest_bmp_code_per_glyph(face: &ttf_parser::Face) -> HashMap<u16, u16> {
184    let mut lowest: HashMap<u16, u16> = HashMap::new();
185    let Some(cmap) = face.tables().cmap else {
186        return lowest;
187    };
188    for subtable in cmap.subtables {
189        if !subtable.is_unicode() {
190            continue;
191        }
192        subtable.codepoints(|code| {
193            if code > 0xFFFF {
194                return;
195            }
196            // Resolve through the face, not the subtable, so a codepoint
197            // maps to the glyph the face would pick, as the scan did.
198            let Some(gid) = char::from_u32(code).and_then(|ch| face.glyph_index(ch)) else {
199                return;
200            };
201            let code = code as u16;
202            lowest
203                .entry(gid.0)
204                .and_modify(|c| *c = (*c).min(code))
205                .or_insert(code);
206        });
207    }
208    lowest
209}
210
211fn parse_loca(data: &[u8], format: i16, num_glyphs: u16) -> Result<Vec<u32>, String> {
212    let count = num_glyphs as usize + 1; // loca has numGlyphs + 1 entries
213    let mut offsets = Vec::with_capacity(count);
214
215    if format == 0 {
216        // Short format: offsets are u16, multiply by 2
217        for i in 0..count {
218            let pos = i * 2;
219            if pos + 2 > data.len() {
220                offsets.push(*offsets.last().unwrap_or(&0));
221            } else {
222                offsets.push(read_u16(data, pos) as u32 * 2);
223            }
224        }
225    } else {
226        // Long format: offsets are u32
227        for i in 0..count {
228            let pos = i * 4;
229            if pos + 4 > data.len() {
230                offsets.push(*offsets.last().unwrap_or(&0));
231            } else {
232                offsets.push(read_u32(data, pos));
233            }
234        }
235    }
236
237    Ok(offsets)
238}
239
240// ─── Composite Glyph Dependency Collection ──────────────────────
241
242fn collect_composite_deps(glyf: &[u8], loca_offsets: &[u32], gid: u16, needed: &mut BTreeSet<u16>) {
243    let idx = gid as usize;
244    if idx + 1 >= loca_offsets.len() {
245        return;
246    }
247
248    let start = loca_offsets[idx] as usize;
249    let end = loca_offsets[idx + 1] as usize;
250    if start >= end || start + 10 > glyf.len() {
251        return;
252    }
253
254    let num_contours = read_i16(glyf, start);
255    if num_contours >= 0 {
256        return;
257    } // Simple glyph, no deps
258
259    // Composite glyph — walk component records
260    let mut pos = start + 10; // skip header (numContours + bbox)
261
262    loop {
263        if pos + 4 > glyf.len() {
264            break;
265        }
266        let flags = read_u16(glyf, pos);
267        let component_gid = read_u16(glyf, pos + 2);
268        pos += 4;
269
270        if needed.insert(component_gid) {
271            // Recursively collect deps for newly discovered component
272            collect_composite_deps(glyf, loca_offsets, component_gid, needed);
273        }
274
275        // Determine how many bytes of arguments follow
276        if flags & 0x0001 != 0 {
277            // ARG_1_AND_2_ARE_WORDS: 2 × i16
278            pos += 4;
279        } else {
280            // 2 × i8
281            pos += 2;
282        }
283
284        // Transform matrix components
285        if flags & 0x0008 != 0 {
286            // WE_HAVE_A_SCALE: 1 × F2Dot14
287            pos += 2;
288        } else if flags & 0x0040 != 0 {
289            // WE_HAVE_AN_X_AND_Y_SCALE: 2 × F2Dot14
290            pos += 4;
291        } else if flags & 0x0080 != 0 {
292            // WE_HAVE_A_TWO_BY_TWO: 4 × F2Dot14
293            pos += 8;
294        }
295
296        if flags & 0x0020 == 0 {
297            // MORE_COMPONENTS flag not set — done
298            break;
299        }
300    }
301}
302
303// ─── Table Rebuilding ───────────────────────────────────────────
304
305fn rebuild_glyf(
306    glyf: &[u8],
307    loca_offsets: &[u32],
308    needed_gids: &BTreeSet<u16>,
309    gid_remap: &HashMap<u16, u16>,
310) -> (Vec<u8>, Vec<u32>) {
311    let mut new_glyf: Vec<u8> = Vec::new();
312    let mut new_offsets: Vec<u32> = Vec::new();
313
314    for &old_gid in needed_gids {
315        new_offsets.push(new_glyf.len() as u32);
316
317        let idx = old_gid as usize;
318        if idx + 1 >= loca_offsets.len() {
319            continue;
320        }
321
322        let start = loca_offsets[idx] as usize;
323        let end = loca_offsets[idx + 1] as usize;
324        if start >= end || start >= glyf.len() {
325            // Empty glyph
326            continue;
327        }
328
329        let glyph_data = &glyf[start..end.min(glyf.len())];
330        let mut new_glyph = glyph_data.to_vec();
331
332        // If composite, rewrite component GID references
333        if glyph_data.len() >= 2 {
334            let num_contours = read_i16(glyph_data, 0);
335            if num_contours < 0 {
336                rewrite_composite_gids(&mut new_glyph, gid_remap);
337            }
338        }
339
340        new_glyf.extend_from_slice(&new_glyph);
341
342        // Pad to 4-byte boundary (required for loca to work correctly)
343        while !new_glyf.len().is_multiple_of(4) {
344            new_glyf.push(0);
345        }
346    }
347
348    // Final offset (marks end of last glyph)
349    new_offsets.push(new_glyf.len() as u32);
350
351    (new_glyf, new_offsets)
352}
353
354fn rewrite_composite_gids(glyph_data: &mut [u8], gid_remap: &HashMap<u16, u16>) {
355    let mut pos = 10; // skip header
356
357    loop {
358        if pos + 4 > glyph_data.len() {
359            break;
360        }
361        let flags = read_u16(glyph_data, pos);
362        let old_gid = read_u16(glyph_data, pos + 2);
363
364        // Rewrite the component GID
365        if let Some(&new_gid) = gid_remap.get(&old_gid) {
366            write_u16(glyph_data, pos + 2, new_gid);
367        }
368
369        pos += 4;
370
371        if flags & 0x0001 != 0 {
372            pos += 4;
373        } else {
374            pos += 2;
375        }
376        if flags & 0x0008 != 0 {
377            pos += 2;
378        } else if flags & 0x0040 != 0 {
379            pos += 4;
380        } else if flags & 0x0080 != 0 {
381            pos += 8;
382        }
383
384        if flags & 0x0020 == 0 {
385            break;
386        }
387    }
388}
389
390fn build_loca(offsets: &[u32], format: i16) -> Vec<u8> {
391    let mut data = Vec::new();
392    if format == 0 {
393        for &offset in offsets {
394            let short = (offset / 2) as u16;
395            data.extend_from_slice(&short.to_be_bytes());
396        }
397    } else {
398        for &offset in offsets {
399            data.extend_from_slice(&offset.to_be_bytes());
400        }
401    }
402    data
403}
404
405fn rebuild_hmtx(hmtx: &[u8], needed_gids: &BTreeSet<u16>, num_h_metrics: usize) -> Vec<u8> {
406    let mut data = Vec::new();
407
408    for &old_gid in needed_gids {
409        let idx = old_gid as usize;
410        if idx < num_h_metrics {
411            // Full metric: advance_width (u16) + lsb (i16)
412            let offset = idx * 4;
413            if offset + 4 <= hmtx.len() {
414                data.extend_from_slice(&hmtx[offset..offset + 4]);
415            } else {
416                data.extend_from_slice(&[0, 0, 0, 0]);
417            }
418        } else {
419            // Only lsb — use last advance width + per-glyph lsb
420            let last_aw_offset = (num_h_metrics - 1) * 4;
421            let advance_width = if last_aw_offset + 2 <= hmtx.len() {
422                &hmtx[last_aw_offset..last_aw_offset + 2]
423            } else {
424                &[0, 0]
425            };
426            let lsb_offset = num_h_metrics * 4 + (idx - num_h_metrics) * 2;
427            let lsb = if lsb_offset + 2 <= hmtx.len() {
428                &hmtx[lsb_offset..lsb_offset + 2]
429            } else {
430                &[0, 0]
431            };
432            data.extend_from_slice(advance_width);
433            data.extend_from_slice(lsb);
434        }
435    }
436
437    data
438}
439
440fn build_cmap_format4(char_to_gid: &[(u16, u16)]) -> Vec<u8> {
441    // Build a cmap table with a single Format 4 subtable
442    // Platform 3 (Windows), Encoding 1 (Unicode BMP)
443    let mut sorted = char_to_gid.to_vec();
444    sorted.sort_by_key(|(ch, _)| *ch);
445
446    // Build segments — each segment is a contiguous run of codepoints
447    let mut segments: Vec<(u16, u16, Vec<u16>)> = Vec::new(); // (start, end, gids)
448
449    for &(ch, gid) in &sorted {
450        if let Some(last) = segments.last_mut() {
451            if ch == last.1 + 1 {
452                last.1 = ch;
453                last.2.push(gid);
454                continue;
455            }
456        }
457        segments.push((ch, ch, vec![gid]));
458    }
459
460    // Add sentinel segment (0xFFFF)
461    segments.push((0xFFFF, 0xFFFF, vec![0]));
462
463    let seg_count = segments.len() as u16;
464    let seg_count_x2 = seg_count * 2;
465    // Compute search parameters per TrueType spec
466    let entry_selector = if seg_count > 0 {
467        (seg_count as f64).log2().floor() as u16
468    } else {
469        0
470    };
471    let search_range = (1u16 << entry_selector) * 2;
472    let range_shift = seg_count_x2.saturating_sub(search_range);
473
474    // Use glyphIdArray for all segments (idRangeOffset pointing to array)
475    // Simpler: use idDelta for single-glyph segments, idRangeOffset for others
476    // Simplest approach: use glyphIdArray for everything
477
478    let mut glyph_id_array: Vec<u16> = Vec::new();
479    let mut end_codes: Vec<u16> = Vec::new();
480    let mut start_codes: Vec<u16> = Vec::new();
481    let mut id_deltas: Vec<i16> = Vec::new();
482    let mut id_range_offsets: Vec<u16> = Vec::new();
483
484    for (i, (start, end, gids)) in segments.iter().enumerate() {
485        start_codes.push(*start);
486        end_codes.push(*end);
487
488        if *start == 0xFFFF {
489            // Sentinel
490            id_deltas.push(1);
491            id_range_offsets.push(0);
492        } else if gids.len() == 1 {
493            // Single char — use idDelta
494            let delta = gids[0] as i32 - *start as i32;
495            id_deltas.push(delta as i16);
496            id_range_offsets.push(0);
497        } else {
498            // Range — use idRangeOffset into glyphIdArray
499            id_deltas.push(0);
500            // Offset from current position in idRangeOffset array to glyphIdArray
501            let remaining_offsets = (segments.len() - i) as u16;
502            let offset = (remaining_offsets + glyph_id_array.len() as u16) * 2;
503            id_range_offsets.push(offset);
504            glyph_id_array.extend_from_slice(gids);
505        }
506    }
507
508    // Build the subtable
509    let subtable_len = 14 + seg_count as usize * 8 + glyph_id_array.len() * 2;
510    let mut subtable: Vec<u8> = Vec::new();
511    subtable.extend_from_slice(&4u16.to_be_bytes()); // format
512    subtable.extend_from_slice(&(subtable_len as u16).to_be_bytes()); // length
513    subtable.extend_from_slice(&0u16.to_be_bytes()); // language
514    subtable.extend_from_slice(&seg_count_x2.to_be_bytes());
515    subtable.extend_from_slice(&search_range.to_be_bytes());
516    subtable.extend_from_slice(&entry_selector.to_be_bytes());
517    subtable.extend_from_slice(&range_shift.to_be_bytes());
518
519    for &ec in &end_codes {
520        subtable.extend_from_slice(&ec.to_be_bytes());
521    }
522    subtable.extend_from_slice(&0u16.to_be_bytes()); // reservedPad
523
524    for &sc in &start_codes {
525        subtable.extend_from_slice(&sc.to_be_bytes());
526    }
527    for &d in &id_deltas {
528        subtable.extend_from_slice(&d.to_be_bytes());
529    }
530    for &r in &id_range_offsets {
531        subtable.extend_from_slice(&r.to_be_bytes());
532    }
533    for &g in &glyph_id_array {
534        subtable.extend_from_slice(&g.to_be_bytes());
535    }
536
537    // Build cmap header
538    let mut cmap: Vec<u8> = Vec::new();
539    cmap.extend_from_slice(&0u16.to_be_bytes()); // version
540    cmap.extend_from_slice(&1u16.to_be_bytes()); // numTables
541                                                 // Encoding record: platform 3 (Windows), encoding 1 (Unicode BMP)
542    cmap.extend_from_slice(&3u16.to_be_bytes()); // platformID
543    cmap.extend_from_slice(&1u16.to_be_bytes()); // encodingID
544    cmap.extend_from_slice(&12u32.to_be_bytes()); // offset to subtable
545    cmap.extend_from_slice(&subtable);
546
547    cmap
548}
549
550fn rebuild_head(head: &[u8], new_loca_format: i16) -> Vec<u8> {
551    let mut new_head = head.to_vec();
552    // Zero out checkSumAdjustment (offset 8, 4 bytes) — will be fixed later
553    write_u32(&mut new_head, 8, 0);
554    // Update indexToLocFormat (offset 50)
555    write_i16(&mut new_head, 50, new_loca_format);
556    new_head
557}
558
559fn rebuild_hhea(hhea: &[u8], new_num_glyphs: u16) -> Vec<u8> {
560    let mut new_hhea = hhea.to_vec();
561    // Pad to 36 bytes if needed (minimum hhea size)
562    while new_hhea.len() < 36 {
563        new_hhea.push(0);
564    }
565    // Update numberOfHMetrics (offset 34) — all glyphs get full metrics
566    write_u16(&mut new_hhea, 34, new_num_glyphs);
567    new_hhea
568}
569
570fn build_maxp(num_glyphs: u16) -> Vec<u8> {
571    let mut data = vec![0u8; 32];
572    // Version 1.0
573    write_u32(&mut data, 0, 0x00010000);
574    // numGlyphs
575    write_u16(&mut data, 4, num_glyphs);
576    // Fill remaining fields with reasonable defaults
577    write_u16(&mut data, 6, 256); // maxPoints
578    write_u16(&mut data, 8, 64); // maxContours
579    write_u16(&mut data, 10, 256); // maxCompositePoints
580    write_u16(&mut data, 12, 64); // maxCompositeContours
581    write_u16(&mut data, 14, 1); // maxZones
582    write_u16(&mut data, 16, 0); // maxTwilightPoints
583    write_u16(&mut data, 18, 64); // maxStorage
584    write_u16(&mut data, 20, 64); // maxFunctionDefs
585    write_u16(&mut data, 22, 64); // maxInstructionDefs
586    write_u16(&mut data, 24, 64); // maxStackElements
587    write_u16(&mut data, 26, 0); // maxSizeOfInstructions
588    write_u16(&mut data, 28, 64); // maxComponentElements
589    write_u16(&mut data, 30, 2); // maxComponentDepth
590    data
591}
592
593fn build_post_format3() -> Vec<u8> {
594    // Format 3.0 — no glyph names (smallest possible)
595    let mut data = vec![0u8; 32];
596    write_u32(&mut data, 0, 0x00030000); // version 3.0
597                                         // italicAngle, underlinePosition, underlineThickness, isFixedPitch — all 0
598    data
599}
600
601fn build_minimal_name(face: &ttf_parser::Face) -> Vec<u8> {
602    // Build a minimal name table with just the font family name
603    let family = face
604        .names()
605        .into_iter()
606        .find(|n| n.name_id == ttf_parser::name_id::FULL_NAME)
607        .and_then(|n| n.to_string())
608        .unwrap_or_else(|| "SubsetFont".to_string());
609
610    let name_bytes: Vec<u8> = family
611        .encode_utf16()
612        .flat_map(|c| c.to_be_bytes())
613        .collect();
614
615    let mut data = Vec::new();
616    // Name table header
617    data.extend_from_slice(&0u16.to_be_bytes()); // format
618    data.extend_from_slice(&1u16.to_be_bytes()); // count
619    let string_offset = 6 + 12; // header (6) + 1 record (12)
620    data.extend_from_slice(&(string_offset as u16).to_be_bytes()); // stringOffset
621
622    // Name record: platformID=3, encodingID=1, languageID=0x0409, nameID=4 (fullName)
623    data.extend_from_slice(&3u16.to_be_bytes()); // platformID
624    data.extend_from_slice(&1u16.to_be_bytes()); // encodingID
625    data.extend_from_slice(&0x0409u16.to_be_bytes()); // languageID
626    data.extend_from_slice(&4u16.to_be_bytes()); // nameID (full name)
627    data.extend_from_slice(&(name_bytes.len() as u16).to_be_bytes()); // length
628    data.extend_from_slice(&0u16.to_be_bytes()); // offset
629
630    // String data
631    data.extend_from_slice(&name_bytes);
632
633    data
634}
635
636// ─── TrueType File Writer ───────────────────────────────────────
637
638fn write_ttf_file(tables: &mut [(u32, Vec<u8>)]) -> Vec<u8> {
639    let num_tables = tables.len() as u16;
640    let entry_selector = if num_tables > 0 {
641        (num_tables as f64).log2().floor() as u16
642    } else {
643        0
644    };
645    let search_range = (1u16 << entry_selector) * 16;
646    let range_shift = (num_tables * 16).saturating_sub(search_range);
647
648    // Offset table (12 bytes)
649    let mut output: Vec<u8> = Vec::new();
650    output.extend_from_slice(&0x00010000u32.to_be_bytes()); // sfVersion (TrueType)
651    output.extend_from_slice(&num_tables.to_be_bytes());
652    output.extend_from_slice(&search_range.to_be_bytes());
653    output.extend_from_slice(&entry_selector.to_be_bytes());
654    output.extend_from_slice(&range_shift.to_be_bytes());
655
656    // Calculate table offsets
657    let dir_size = 12 + num_tables as usize * 16;
658    let mut table_offset = dir_size;
659
660    // Pad each table to 4-byte boundary
661    for (_, data) in tables.iter_mut() {
662        while data.len() % 4 != 0 {
663            data.push(0);
664        }
665    }
666
667    // Table directory
668    for (tag, data) in tables.iter() {
669        output.extend_from_slice(&tag.to_be_bytes());
670        let checksum = calc_table_checksum(data);
671        output.extend_from_slice(&checksum.to_be_bytes());
672        output.extend_from_slice(&(table_offset as u32).to_be_bytes());
673        output.extend_from_slice(&(data.len() as u32).to_be_bytes());
674        table_offset += data.len();
675    }
676
677    // Table data
678    for (_, data) in tables.iter() {
679        output.extend_from_slice(data);
680    }
681
682    // Fix head checkSumAdjustment
683    fix_head_checksum(&mut output, tables);
684
685    output
686}
687
688fn calc_table_checksum(data: &[u8]) -> u32 {
689    let mut sum: u32 = 0;
690    let mut i = 0;
691    while i + 4 <= data.len() {
692        sum = sum.wrapping_add(read_u32(data, i));
693        i += 4;
694    }
695    // Handle remaining bytes
696    if i < data.len() {
697        let mut last = [0u8; 4];
698        for (j, &b) in data[i..].iter().enumerate() {
699            last[j] = b;
700        }
701        sum = sum.wrapping_add(u32::from_be_bytes(last));
702    }
703    sum
704}
705
706fn fix_head_checksum(output: &mut [u8], tables: &[(u32, Vec<u8>)]) {
707    // Find the head table offset in the directory
708    let num_tables = read_u16(output, 4) as usize;
709    let head_tag = tag_u32(b"head");
710
711    for i in 0..num_tables {
712        let dir_offset = 12 + i * 16;
713        let tag = read_u32(output, dir_offset);
714        if tag == head_tag {
715            let table_offset = read_u32(output, dir_offset + 8) as usize;
716
717            // Calculate file checksum
718            let file_checksum = calc_table_checksum(output);
719            let adjustment = 0xB1B0AFBAu32.wrapping_sub(file_checksum);
720
721            // Write checkSumAdjustment at head table offset + 8
722            if table_offset + 12 <= output.len() {
723                write_u32(output, table_offset + 8, adjustment);
724            }
725
726            // Update the head table checksum in the directory
727            // First, find the head table data to recalculate
728            let head_data_len = read_u32(output, dir_offset + 12) as usize;
729            if table_offset + head_data_len <= output.len() {
730                let checksum =
731                    calc_table_checksum(&output[table_offset..table_offset + head_data_len]);
732                write_u32(output, dir_offset + 4, checksum);
733            }
734
735            break;
736        }
737    }
738
739    // Suppress unused variable warning
740    let _ = tables;
741}
742
743// ─── Byte Helpers ───────────────────────────────────────────────
744
745fn read_u16(data: &[u8], offset: usize) -> u16 {
746    u16::from_be_bytes([data[offset], data[offset + 1]])
747}
748
749fn read_i16(data: &[u8], offset: usize) -> i16 {
750    i16::from_be_bytes([data[offset], data[offset + 1]])
751}
752
753fn read_u32(data: &[u8], offset: usize) -> u32 {
754    u32::from_be_bytes([
755        data[offset],
756        data[offset + 1],
757        data[offset + 2],
758        data[offset + 3],
759    ])
760}
761
762fn write_u16(data: &mut [u8], offset: usize, val: u16) {
763    let bytes = val.to_be_bytes();
764    data[offset] = bytes[0];
765    data[offset + 1] = bytes[1];
766}
767
768fn write_i16(data: &mut [u8], offset: usize, val: i16) {
769    let bytes = val.to_be_bytes();
770    data[offset] = bytes[0];
771    data[offset + 1] = bytes[1];
772}
773
774fn write_u32(data: &mut [u8], offset: usize, val: u32) {
775    let bytes = val.to_be_bytes();
776    data[offset] = bytes[0];
777    data[offset + 1] = bytes[1];
778    data[offset + 2] = bytes[2];
779    data[offset + 3] = bytes[3];
780}
781
782fn tag_u32(tag: &[u8; 4]) -> u32 {
783    u32::from_be_bytes(*tag)
784}
785
786// ─── Tests ──────────────────────────────────────────────────────
787
788#[cfg(test)]
789mod tests {
790    use super::*;
791
792    /// The one-pass inversion must pick the glyph-to-codepoint mapping the
793    /// old per-glyph scan picked (the lowest BMP codepoint the face maps to
794    /// each glyph), so subsets keep the same cmap. The reference is that
795    /// scan, run once over the plane rather than once per glyph.
796    #[test]
797    fn test_lowest_bmp_code_per_glyph_matches_the_full_scan() {
798        let fonts: [(&str, &[u8]); 6] = [
799            (
800                "Noto Sans",
801                include_bytes!("../../fonts/NotoSans-Regular.ttf"),
802            ),
803            (
804                "Noto Emoji",
805                include_bytes!("../../tests/fixtures/fonts/NotoEmoji-Regular.ttf"),
806            ),
807            (
808                "Noto Naskh Arabic",
809                include_bytes!("../../tests/fixtures/fonts/NotoNaskhArabic-Regular.ttf"),
810            ),
811            (
812                "Noto Sans Devanagari",
813                include_bytes!("../../tests/fixtures/fonts/NotoSansDevanagari-Regular.ttf"),
814            ),
815            (
816                "Noto Sans Hebrew",
817                include_bytes!("../../tests/fixtures/fonts/NotoSansHebrew-Regular.ttf"),
818            ),
819            (
820                "Liberation Sans",
821                include_bytes!("../../../packages/fonts-standard/fonts/LiberationSans-Regular.ttf"),
822            ),
823        ];
824        for (name, data) in fonts {
825            let face = ttf_parser::Face::parse(data, 0).unwrap();
826            let mut expected: HashMap<u16, u16> = HashMap::new();
827            for code in 0u32..=0xFFFF {
828                if let Some(gid) = char::from_u32(code).and_then(|ch| face.glyph_index(ch)) {
829                    expected.entry(gid.0).or_insert(code as u16);
830                }
831            }
832            // .notdef is never given a cmap entry (subset_ttf skips gid 0).
833            // A format 4 table's required final segment maps U+FFFF to it,
834            // which the scan sees and the codepoint walk does not.
835            expected.remove(&0);
836            let mut got = lowest_bmp_code_per_glyph(&face);
837            got.remove(&0);
838            assert!(
839                !expected.is_empty(),
840                "{name}: the reference found no mappings"
841            );
842            let mut diff: Vec<(u16, Option<u16>, Option<u16>)> = expected
843                .keys()
844                .chain(got.keys())
845                .copied()
846                .collect::<BTreeSet<u16>>()
847                .into_iter()
848                .filter(|gid| got.get(gid) != expected.get(gid))
849                .map(|gid| (gid, got.get(&gid).copied(), expected.get(&gid).copied()))
850                .collect();
851            diff.truncate(10);
852            assert!(diff.is_empty(), "{name}: (gid, got, scan) {diff:?}");
853        }
854    }
855
856    #[test]
857    fn test_tag_u32() {
858        assert_eq!(tag_u32(b"glyf"), 0x676C7966);
859        assert_eq!(tag_u32(b"head"), 0x68656164);
860    }
861
862    #[test]
863    fn test_calc_table_checksum() {
864        // Known checksum for "ABCD" (0x41424344)
865        let data = b"ABCD";
866        assert_eq!(calc_table_checksum(data), 0x41424344);
867    }
868
869    #[test]
870    fn test_build_post_format3() {
871        let data = build_post_format3();
872        assert_eq!(data.len(), 32);
873        assert_eq!(read_u32(&data, 0), 0x00030000);
874    }
875
876    #[test]
877    fn test_build_maxp() {
878        let data = build_maxp(42);
879        assert_eq!(read_u32(&data, 0), 0x00010000);
880        assert_eq!(read_u16(&data, 4), 42);
881    }
882
883    #[test]
884    fn test_build_loca_short() {
885        let offsets = vec![0, 100, 200, 300];
886        let data = build_loca(&offsets, 0);
887        // Short format: each offset / 2 stored as u16
888        assert_eq!(data.len(), 8); // 4 entries × 2 bytes
889        assert_eq!(read_u16(&data, 0), 0);
890        assert_eq!(read_u16(&data, 2), 50);
891        assert_eq!(read_u16(&data, 4), 100);
892        assert_eq!(read_u16(&data, 6), 150);
893    }
894
895    #[test]
896    fn test_build_loca_long() {
897        let offsets = vec![0, 100, 200, 300];
898        let data = build_loca(&offsets, 1);
899        assert_eq!(data.len(), 16); // 4 entries × 4 bytes
900        assert_eq!(read_u32(&data, 0), 0);
901        assert_eq!(read_u32(&data, 4), 100);
902        assert_eq!(read_u32(&data, 8), 200);
903        assert_eq!(read_u32(&data, 12), 300);
904    }
905
906    #[test]
907    fn test_cmap_format4_single_char() {
908        let entries = vec![(65u16, 1u16)]; // 'A' → gid 1
909        let cmap = build_cmap_format4(&entries);
910
911        // Should be a valid cmap table
912        assert_eq!(read_u16(&cmap, 0), 0); // version
913        assert_eq!(read_u16(&cmap, 2), 1); // numTables
914
915        // Encoding record
916        assert_eq!(read_u16(&cmap, 4), 3); // platformID = Windows
917        assert_eq!(read_u16(&cmap, 6), 1); // encodingID = Unicode BMP
918
919        // Subtable format should be 4
920        let subtable_offset = read_u32(&cmap, 8) as usize;
921        assert_eq!(read_u16(&cmap, subtable_offset), 4);
922    }
923}