Skip to main content

hyprforge_mime/
magic.rs

1//! What a file is, decided by its **contents**.
2//!
3//! The other half of the question [`crate::globs`] answers by name. A
4//! file with no extension, or one whose extension lies, can still be
5//! recognised by what is in it: `%PDF-` at offset 0, `\x89PNG` at offset
6//! 0, `solid ` for an ASCII STL.
7//!
8//! # The format
9//!
10//! `/usr/share/mime/magic` is binary, and deliberately so — it is read
11//! at every content lookup on the system. After the header
12//! `MIME-Magic\0\n` it is a sequence of sections:
13//!
14//! ```text
15//! [<priority>:<mime type>]\n
16//! [indent]>offset=LLvalue[&mask][~word][+range]\n
17//! ```
18//!
19//! `LL` is two bytes, big-endian, giving the length of `value` — which
20//! is raw bytes and may itself contain a newline. That single detail is
21//! why this is parsed byte by byte and not with `lines()`: splitting on
22//! `\n` cuts rules in half, quietly, and the ones it cuts are the
23//! interesting ones.
24//!
25//! # Nesting
26//!
27//! A rule indented one deeper than the line above is a *further*
28//! condition on it. A rule matches when its own pattern matches and,
29//! if it has children, at least one child matches too. That is what
30//! makes `>0=\x00\x05<?xml` plus `1>0=...DocBook...` mean "an XML file,
31//! and specifically a DocBook one" rather than two unrelated claims.
32//!
33//! # Why this exists when `globs2` already answered
34//!
35//! Most of the time the name is right and this is not consulted at all
36//! — see [`crate::lookup`] for the order. It earns its place on the
37//! files that have no name to go on: something saved as `download`,
38//! a file being examined before it is renamed, `stdin`.
39
40use std::io::Read;
41
42/// How many bytes are read from a file to match against. The database's
43/// deepest rule on this machine sits well inside this; the value is the
44/// one `xdgmime` uses, so a file this cannot recognise is one the rest
45/// of the desktop cannot recognise either.
46const SNIFF_BYTES: usize = 16 * 1024;
47
48/// One condition: these bytes, at this offset.
49#[derive(Debug, Clone, PartialEq, Eq)]
50struct Rule {
51    /// How deep this rule is nested under the one before it.
52    indent: u32,
53    /// Where the value starts.
54    offset: usize,
55    value: Vec<u8>,
56    /// Compared as `data & mask == value & mask` when present.
57    mask: Option<Vec<u8>>,
58    /// How many bytes past `offset` to keep looking.
59    range: usize,
60}
61
62impl Rule {
63    fn matches(&self, data: &[u8]) -> bool {
64        // `..=`: a range of 0 still tests the offset itself, which is
65        // the common case and the one an exclusive range would skip.
66        (self.offset..=self.offset + self.range).any(|start| {
67            let Some(window) = data.get(start..start + self.value.len()) else { return false };
68            match &self.mask {
69                None => window == self.value,
70                Some(mask) => window
71                    .iter()
72                    .zip(&self.value)
73                    .zip(mask)
74                    .all(|((byte, value), mask)| byte & mask == value & mask),
75            }
76        })
77    }
78}
79
80/// Every rule for one type, with the priority the database gives it.
81#[derive(Debug, Clone, PartialEq, Eq)]
82struct Section {
83    /// 0-100. 80 and above is "strong enough to overrule a filename" —
84    /// see [`Magic::lookup`].
85    priority: u32,
86    mime: String,
87    rules: Vec<Rule>,
88}
89
90impl Section {
91    fn matches(&self, data: &[u8]) -> bool {
92        matches_level(&self.rules, data, 0)
93    }
94}
95
96/// A rule at `indent` matches when its own pattern matches and, if it
97/// has children, one of them matches too — see the module doc.
98fn matches_level(rules: &[Rule], data: &[u8], indent: u32) -> bool {
99    let mut i = 0;
100    while i < rules.len() {
101        if rules[i].indent != indent {
102            i += 1;
103            continue;
104        }
105        if rules[i].matches(data) {
106            let children_start = i + 1;
107            let children_end = rules[children_start..]
108                .iter()
109                .position(|rule| rule.indent <= indent)
110                .map(|offset| children_start + offset)
111                .unwrap_or(rules.len());
112            if children_start == children_end {
113                return true;
114            }
115            if matches_level(&rules[children_start..children_end], data, indent + 1) {
116                return true;
117            }
118        }
119        i += 1;
120    }
121    false
122}
123
124/// The content rules, in the order the database gives them.
125#[derive(Debug, Clone, Default, PartialEq, Eq)]
126pub struct Magic {
127    sections: Vec<Section>,
128}
129
130/// What a content match found, and how much it should be trusted.
131#[derive(Debug, Clone, PartialEq, Eq)]
132pub struct Match {
133    pub mime: String,
134    pub priority: u32,
135}
136
137impl Magic {
138    /// Parses one `magic` file.
139    ///
140    /// A file that does not start with the header is not this format
141    /// and is ignored rather than guessed at. Anything unparseable
142    /// after that ends the read: the format is length-prefixed, so once
143    /// a length is misread there is no way back to a known position —
144    /// and a wrong guess about where the next rule starts is worse than
145    /// stopping with what was understood.
146    pub fn parse(bytes: &[u8]) -> Magic {
147        let mut magic = Magic::default();
148        let Some(mut rest) = bytes.strip_prefix(b"MIME-Magic\0\n") else {
149            return magic;
150        };
151        while !rest.is_empty() {
152            if rest[0] != b'[' {
153                break;
154            }
155            let Some(close) = rest.iter().position(|b| *b == b']') else { break };
156            let header = &rest[1..close];
157            rest = &rest[close + 1..];
158            if rest.first() == Some(&b'\n') {
159                rest = &rest[1..];
160            }
161            let Some((priority, mime)) = split_header(header) else { break };
162            let mut section = Section { priority, mime, rules: Vec::new() };
163            while !rest.is_empty() && rest[0] != b'[' {
164                match parse_rule(rest) {
165                    Some((rule, tail)) => {
166                        section.rules.push(rule);
167                        rest = tail;
168                    }
169                    None => {
170                        rest = &[];
171                        break;
172                    }
173                }
174            }
175            if !section.rules.is_empty() {
176                magic.sections.push(section);
177            }
178        }
179        magic
180    }
181
182    /// Reads every `magic` file under the given directories. Later
183    /// directories add their sections after the earlier ones, which is
184    /// the order the priority sort then works over.
185    pub fn load_from(dirs: &[std::path::PathBuf]) -> Magic {
186        let mut magic = Magic::default();
187        for dir in dirs {
188            if let Ok(bytes) = std::fs::read(dir.join("mime").join("magic")) {
189                magic.sections.extend(Magic::parse(&bytes).sections);
190            }
191        }
192        // Highest priority first, so the first match is the best one.
193        // A stable sort, so two rules of equal priority keep the
194        // database's own order — which is how `application/xml` stays
195        // behind the specific XML formats that share its opening bytes.
196        magic.sections.sort_by_key(|section| std::cmp::Reverse(section.priority));
197        magic
198    }
199
200    /// The best content match for these bytes.
201    pub fn of_data(&self, data: &[u8]) -> Option<Match> {
202        self.sections
203            .iter()
204            .find(|section| section.matches(data))
205            .map(|section| Match { mime: section.mime.clone(), priority: section.priority })
206    }
207
208    /// The best content match for a file, reading only its first
209    /// `SNIFF_BYTES`.
210    pub fn of_file(&self, path: &std::path::Path) -> Option<Match> {
211        self.of_data(&head(path)?)
212    }
213
214    /// Every content match, best first — `mimetype --all`.
215    pub fn all_of_data(&self, data: &[u8]) -> Vec<Match> {
216        self.sections
217            .iter()
218            .filter(|section| section.matches(data))
219            .map(|section| Match { mime: section.mime.clone(), priority: section.priority })
220            .collect()
221    }
222
223    pub fn is_empty(&self) -> bool {
224        self.sections.is_empty()
225    }
226}
227
228/// The first `SNIFF_BYTES` of a file, for content matching.
229///
230/// **The one way to read a file for typing.** Bounded on purpose: this
231/// is asked about files a person just pointed at, which can be a 40GB
232/// disk image on a slow mount, and no magic rule in the database
233/// reaches anywhere near that far in.
234///
235/// It exists as its own function because the bound was written once and
236/// then bypassed three times: `Lookup::of_file`, `Lookup::all_of_file`
237/// and the `mimetype` command each read the *whole* file instead, so
238/// nothing a person actually double-clicked was bounded at all. An
239/// 800MB file cost 766MB of resident memory to answer "what is this".
240/// That is the failure the project's own rule about testing the
241/// resource rather than the result is named after, and the tests here
242/// had asked only what type came back.
243pub fn head(path: &std::path::Path) -> Option<Vec<u8>> {
244    let mut file = std::fs::File::open(path).ok()?;
245    let mut buffer = vec![0u8; SNIFF_BYTES];
246    let read = read_up_to(&mut file, &mut buffer)?;
247    buffer.truncate(read);
248    Some(buffer)
249}
250
251/// Fills `buffer` as far as the file goes, tolerating short reads.
252///
253/// `read` is allowed to return fewer bytes than asked for without being
254/// at the end of the file, and a single call would then sniff a
255/// fraction of what it meant to.
256fn read_up_to(file: &mut std::fs::File, buffer: &mut [u8]) -> Option<usize> {
257    let mut filled = 0;
258    while filled < buffer.len() {
259        match file.read(&mut buffer[filled..]) {
260            Ok(0) => break,
261            Ok(n) => filled += n,
262            Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
263            Err(_) => return None,
264        }
265    }
266    Some(filled)
267}
268
269/// `90:application/pdf` from inside the brackets.
270fn split_header(header: &[u8]) -> Option<(u32, String)> {
271    let header = std::str::from_utf8(header).ok()?;
272    let (priority, mime) = header.split_once(':')?;
273    Some((priority.trim().parse().ok()?, mime.trim().to_string()))
274}
275
276/// One rule line, and whatever follows it.
277///
278/// Returns `None` when the line cannot be read, which stops the parse —
279/// see [`Magic::parse`] for why that is better than skipping ahead.
280fn parse_rule(bytes: &[u8]) -> Option<(Rule, &[u8])> {
281    // [indent]>
282    let gt = bytes.iter().position(|b| *b == b'>')?;
283    let indent: u32 = match gt {
284        0 => 0,
285        _ => std::str::from_utf8(&bytes[..gt]).ok()?.trim().parse().ok()?,
286    };
287    let rest = &bytes[gt + 1..];
288
289    // offset=
290    let eq = rest.iter().position(|b| *b == b'=')?;
291    let offset: usize = std::str::from_utf8(&rest[..eq]).ok()?.trim().parse().ok()?;
292    let rest = &rest[eq + 1..];
293
294    // Two bytes of length, big-endian, then that many bytes of value —
295    // which may contain newlines, so the length is the only safe guide.
296    let length = usize::from(u16::from_be_bytes([*rest.first()?, *rest.get(1)?]));
297    let value = rest.get(2..2 + length)?.to_vec();
298    let mut rest = &rest[2 + length..];
299
300    let mut mask = None;
301    if rest.first() == Some(&b'&') {
302        mask = Some(rest.get(1..1 + length)?.to_vec());
303        rest = &rest[1 + length..];
304    }
305    let mut word_size = 1usize;
306    if rest.first() == Some(&b'~') {
307        let (number, tail) = take_number(&rest[1..])?;
308        word_size = number.max(1);
309        rest = tail;
310    }
311    let mut range = 0usize;
312    if rest.first() == Some(&b'+') {
313        let (number, tail) = take_number(&rest[1..])?;
314        range = number;
315        rest = tail;
316    }
317    // The line ends here; anything else on it is not something this
318    // understands, so stop rather than guess.
319    match rest.first() {
320        Some(b'\n') => rest = &rest[1..],
321        None => {}
322        Some(_) => return None,
323    }
324
325    let (value, mask) = swap_words(value, mask, word_size);
326    Some((Rule { indent, offset, value, mask, range }, rest))
327}
328
329fn take_number(bytes: &[u8]) -> Option<(usize, &[u8])> {
330    let end = bytes.iter().position(|b| !b.is_ascii_digit()).unwrap_or(bytes.len());
331    let number = std::str::from_utf8(&bytes[..end]).ok()?.parse().ok()?;
332    Some((number, &bytes[end..]))
333}
334
335/// Byte-swaps a value stored for a different word size.
336///
337/// The database writes multi-byte words big-endian and marks them with
338/// `~2` or `~4`; on a little-endian machine the bytes in the file are
339/// the other way round. Swapping the *pattern* once here beats swapping
340/// the file's bytes at every comparison.
341fn swap_words(value: Vec<u8>, mask: Option<Vec<u8>>, word_size: usize) -> (Vec<u8>, Option<Vec<u8>>) {
342    if word_size <= 1 || !cfg!(target_endian = "little") || !value.len().is_multiple_of(word_size) {
343        return (value, mask);
344    }
345    let swap = |bytes: Vec<u8>| -> Vec<u8> {
346        bytes.chunks(word_size).flat_map(|word| word.iter().rev().copied()).collect()
347    };
348    (swap(value), mask.map(swap))
349}
350
351#[cfg(test)]
352mod tests {
353    use super::*;
354
355    /// Builds a magic file the way the real one is written: a two-byte
356    /// big-endian length before every value.
357    /// `indent, offset, value, mask, range` — one rule line, as the
358    /// file spells it.
359    type TestRule<'a> = (u32, usize, &'a [u8], Option<&'a [u8]>, usize);
360
361    fn magic_file(sections: &[(u32, &str, &[TestRule<'_>])]) -> Vec<u8> {
362        let mut out = b"MIME-Magic\0\n".to_vec();
363        for (priority, mime, rules) in sections {
364            out.extend(format!("[{priority}:{mime}]\n").into_bytes());
365            for (indent, offset, value, mask, range) in *rules {
366                if *indent > 0 {
367                    out.extend(indent.to_string().into_bytes());
368                }
369                out.extend(format!(">{offset}=").into_bytes());
370                out.extend((value.len() as u16).to_be_bytes());
371                out.extend(*value);
372                if let Some(mask) = mask {
373                    out.push(b'&');
374                    out.extend(*mask);
375                }
376                if *range > 0 {
377                    out.extend(format!("+{range}").into_bytes());
378                }
379                out.push(b'\n');
380            }
381        }
382        out
383    }
384
385    fn pdf_and_png() -> Magic {
386        Magic::parse(&magic_file(&[
387            (50, "application/pdf", &[(0, 0, b"%PDF-".as_slice(), None, 0)]),
388            (50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
389        ]))
390    }
391
392    #[test]
393    fn a_files_first_bytes_name_its_type() {
394        let magic = pdf_and_png();
395        assert_eq!(magic.of_data(b"%PDF-1.7 ...").unwrap().mime, "application/pdf");
396        assert_eq!(magic.of_data(b"\x89PNG\r\n\x1a\n").unwrap().mime, "image/png");
397        assert_eq!(magic.of_data(b"hello"), None);
398        assert_eq!(magic.of_data(b""), None, "nothing to go on");
399    }
400
401    /// The detail that makes this a byte parser: a value may contain a
402    /// newline, and splitting the file into lines would cut the rule in
403    /// half and lose every rule after it.
404    #[test]
405    fn a_value_containing_a_newline_is_read_whole() {
406        let magic = Magic::parse(&magic_file(&[
407            (50, "text/x-two-lines", &[(0, 0, b"first\nsecond".as_slice(), None, 0)]),
408            (50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
409        ]));
410        assert_eq!(magic.of_data(b"first\nsecond and more").unwrap().mime, "text/x-two-lines");
411        assert_eq!(
412            magic.of_data(b"\x89PNG").unwrap().mime,
413            "image/png",
414            "the section after it survived the parse"
415        );
416    }
417
418    #[test]
419    fn a_rule_can_look_within_a_range_rather_than_at_one_offset() {
420        let magic = Magic::parse(&magic_file(&[(
421            50,
422            "text/x-somewhere",
423            &[(0, 0, b"needle".as_slice(), None, 20)],
424        )]));
425        assert!(magic.of_data(b"..........needle").is_some(), "found later in the range");
426        assert!(magic.of_data(b"needle").is_some(), "and at the offset itself");
427        assert!(
428            magic.of_data(b"..............................needle").is_none(),
429            "past the range"
430        );
431    }
432
433    #[test]
434    fn a_mask_ignores_the_bits_it_clears() {
435        let magic = Magic::parse(&magic_file(&[(
436            50,
437            "x/masked",
438            &[(0, 0, b"\xf0".as_slice(), Some(b"\xf0".as_slice()), 0)],
439        )]));
440        assert!(magic.of_data(b"\xff").is_some(), "high nibble matches, low ignored");
441        assert!(magic.of_data(b"\xf0").is_some());
442        assert!(magic.of_data(b"\x0f").is_none());
443    }
444
445    /// Nesting is "and also": the XML rule plus a DocBook rule under it
446    /// means DocBook, not two separate claims.
447    #[test]
448    fn a_nested_rule_is_a_further_condition_on_its_parent() {
449        let magic = Magic::parse(&magic_file(&[(
450            50,
451            "application/docbook+xml",
452            &[
453                (0, 0, b"<?xml".as_slice(), None, 0),
454                (1, 0, b"-//OASIS//DTD DocBook".as_slice(), None, 200),
455            ],
456        )]));
457        assert!(
458            magic.of_data(b"<?xml version='1.0'?><!DOCTYPE book PUBLIC \"-//OASIS//DTD DocBook XML\">").is_some()
459        );
460        assert!(magic.of_data(b"<?xml version='1.0'?><html/>").is_none(), "XML, but not DocBook");
461    }
462
463    /// Several children are alternatives: any one of them is enough.
464    #[test]
465    fn any_one_child_satisfies_its_parent() {
466        let magic = Magic::parse(&magic_file(&[(
467            50,
468            "x/either",
469            &[
470                (0, 0, b"HEAD".as_slice(), None, 0),
471                (1, 4, b"one".as_slice(), None, 0),
472                (1, 4, b"two".as_slice(), None, 0),
473            ],
474        )]));
475        assert!(magic.of_data(b"HEADone").is_some());
476        assert!(magic.of_data(b"HEADtwo").is_some());
477        assert!(magic.of_data(b"HEADthree").is_none());
478    }
479
480    #[test]
481    fn the_strongest_rule_is_the_one_reported() {
482        let magic = Magic::load_from_parsed(vec![
483            Magic::parse(&magic_file(&[(20, "x/weak", &[(0, 0, b"AB".as_slice(), None, 0)])])),
484            Magic::parse(&magic_file(&[(90, "x/strong", &[(0, 0, b"AB".as_slice(), None, 0)])])),
485        ]);
486        let found = magic.of_data(b"ABCD").unwrap();
487        assert_eq!(found.mime, "x/strong");
488        assert_eq!(found.priority, 90);
489        let all: Vec<String> = magic.all_of_data(b"ABCD").into_iter().map(|m| m.mime).collect();
490        assert_eq!(all, ["x/strong", "x/weak"], "and --all lists both, best first");
491    }
492
493    #[test]
494    fn a_file_that_is_not_this_format_is_ignored_rather_than_guessed_at() {
495        assert!(Magic::parse(b"not a magic file at all").is_empty());
496        assert!(Magic::parse(b"").is_empty());
497    }
498
499    /// A truncated file costs the rules after the damage, not the ones
500    /// already understood.
501    #[test]
502    fn a_truncated_file_keeps_what_was_read() {
503        let mut bytes = magic_file(&[
504            (50, "application/pdf", &[(0, 0, b"%PDF-".as_slice(), None, 0)]),
505            (50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
506        ]);
507        bytes.truncate(bytes.len() - 3);
508        let magic = Magic::parse(&bytes);
509        assert_eq!(magic.of_data(b"%PDF-1.7").unwrap().mime, "application/pdf");
510    }
511
512    #[test]
513    fn a_word_sized_value_is_swapped_once_rather_than_per_comparison() {
514        let (value, mask) = swap_words(vec![0x12, 0x34, 0x56, 0x78], Some(vec![0xff, 0x00, 0xff, 0x00]), 2);
515        if cfg!(target_endian = "little") {
516            assert_eq!(value, vec![0x34, 0x12, 0x78, 0x56]);
517            assert_eq!(mask, Some(vec![0x00, 0xff, 0x00, 0xff]));
518        } else {
519            assert_eq!(value, vec![0x12, 0x34, 0x56, 0x78], "big-endian needs no swap");
520        }
521    }
522
523    /// The bound is the property, not an implementation detail: a file
524    /// far larger than the sniff window costs the window, not the file.
525    #[test]
526    fn a_files_contents_are_read_from_disk_and_bounded() {
527        let dir = tempfile::tempdir().unwrap();
528        let path = dir.path().join("anonymous");
529        let mut contents = b"%PDF-1.7\n".to_vec();
530        contents.extend(std::iter::repeat_n(b'x', SNIFF_BYTES * 4));
531        std::fs::write(&path, &contents).unwrap();
532
533        let read = head(&path).expect("readable");
534        assert_eq!(read.len(), SNIFF_BYTES, "a big file costs the window, not its size");
535        assert_eq!(pdf_and_png().of_file(&path).unwrap().mime, "application/pdf");
536        assert_eq!(pdf_and_png().of_file(&dir.path().join("missing")), None);
537        assert_eq!(head(&dir.path().join("missing")), None);
538    }
539
540    /// A file smaller than the window reads as itself, not as a
541    /// window-sized buffer of trailing zeroes — which would make every
542    /// short file look like padded binary.
543    #[test]
544    fn a_small_file_reads_as_exactly_itself() {
545        let dir = tempfile::tempdir().unwrap();
546        let path = dir.path().join("small");
547        std::fs::write(&path, b"hello").unwrap();
548        assert_eq!(head(&path).unwrap(), b"hello");
549    }
550
551    impl Magic {
552        /// Test-only: the merge-and-sort half of [`Magic::load_from`],
553        /// without writing files to disk to exercise it.
554        fn load_from_parsed(parts: Vec<Magic>) -> Magic {
555            let mut magic = Magic::default();
556            for part in parts {
557                magic.sections.extend(part.sections);
558            }
559            magic.sections.sort_by_key(|section| std::cmp::Reverse(section.priority));
560            magic
561        }
562    }
563}