Skip to main content

hyprforge_mime/
globs.rs

1//! What type a file is, decided by its **name**.
2//!
3//! The shared MIME database ships `globs2`, a plain list of
4//! `weight:type:pattern` lines that every desktop already agrees on.
5//! This reads it; it does not invent rules of its own.
6//!
7//! # Why by name, and not by content
8//!
9//! Content sniffing gets the interesting cases wrong in a way that is
10//! invisible until someone double-clicks. `file --mime-type` — which is
11//! what `xdg-open` falls back to on a desktop it does not recognise,
12//! Hyprland included — reads bytes and never the name, so a `.stl` is
13//! `application/octet-stream`, a `.3mf` is `application/zip` (it is one)
14//! and a `.blend` is `application/zstd` (it is that too). Those are all
15//! true statements about the bytes and useless statements about the
16//! file: nothing has a default application for them, and a lookup that
17//! finds nothing sends the file to a web browser.
18//!
19//! The name knows. `part.3mf` is a 3MF model that happens to be stored
20//! as a zip, and that is exactly what `globs2` says.
21//!
22//! Content sniffing answers the cases a name cannot — a file with no
23//! extension at all — and lives in [`crate::magic`]. Which of the two
24//! wins, and when, is [`crate::lookup`]: a name beats a weak content
25//! guess, and a strong one beats a name.
26
27use std::path::Path;
28
29/// One pattern from `globs2`.
30#[derive(Debug, Clone, PartialEq, Eq)]
31struct Glob {
32    /// The database's own weight. Higher wins.
33    weight: u32,
34    mime: String,
35    pattern: String,
36    /// The pattern's characters, and its length in them.
37    ///
38    /// Kept beside the pattern because matching walks characters, and
39    /// collecting them per rule per lookup is the whole cost of a
40    /// lookup: 1541 rules on this machine, ~40,000 on a full one, twice
41    /// each. Parsing happens once at load; matching happens on every
42    /// file anyone opens.
43    chars: Vec<char>,
44    /// `cs` in the flags field: this pattern only matches with the
45    /// letters exactly as written.
46    case_sensitive: bool,
47}
48
49impl Glob {
50    fn new(weight: u32, mime: &str, pattern: &str, case_sensitive: bool) -> Glob {
51        Glob {
52            weight,
53            mime: mime.to_string(),
54            chars: pattern.chars().collect(),
55            pattern: pattern.to_string(),
56            case_sensitive,
57        }
58    }
59}
60
61/// The filename rules, loaded from every `globs2` on the system.
62#[derive(Debug, Clone, Default, PartialEq, Eq)]
63pub struct Globs {
64    globs: Vec<Glob>,
65}
66
67impl Globs {
68    /// Parses one `globs2` file's contents.
69    ///
70    /// Lines that cannot be read are skipped rather than failing the
71    /// file: this is generated data, and one unfamiliar line in it must
72    /// not cost every rule after it. `__NOGLOBS__` — the database's way
73    /// of saying a type has had its inherited patterns removed — carries
74    /// no pattern and is skipped with the rest.
75    pub fn parse(text: &str) -> Globs {
76        let mut globs = Vec::new();
77        for line in text.lines() {
78            let line = line.trim();
79            if line.is_empty() || line.starts_with('#') {
80                continue;
81            }
82            // weight:type:pattern[:flags]
83            let mut fields = line.splitn(4, ':');
84            let (Some(weight), Some(mime), Some(pattern)) =
85                (fields.next(), fields.next(), fields.next())
86            else {
87                continue;
88            };
89            let Ok(weight) = weight.parse::<u32>() else { continue };
90            if pattern.is_empty() || pattern == "__NOGLOBS__" {
91                continue;
92            }
93            let case_sensitive =
94                fields.next().is_some_and(|flags| flags.split(':').any(|flag| flag == "cs"));
95            globs.push(Glob::new(weight, mime, pattern, case_sensitive));
96        }
97        Globs { globs }
98    }
99
100    /// The older `globs` format: `type:pattern`, with no weight and no
101    /// flags. Still shipped beside `globs2`, and the only one some
102    /// hand-made databases have.
103    ///
104    /// Everything in it is weight 50, which is what the format means
105    /// and what `update-mime-database` writes into `globs2` for a rule
106    /// that did not ask for anything else.
107    pub fn parse_legacy(text: &str) -> Globs {
108        let mut globs = Vec::new();
109        for line in text.lines() {
110            let line = line.trim();
111            if line.is_empty() || line.starts_with('#') {
112                continue;
113            }
114            let Some((mime, pattern)) = line.split_once(':') else { continue };
115            if pattern.is_empty() {
116                continue;
117            }
118            globs.push(Glob::new(50, mime, pattern, false));
119        }
120        Globs { globs }
121    }
122
123    /// Every `globs2` under `dirs` (each a `XDG_DATA_DIRS` entry),
124    /// merged. Earlier directories are more specific and are consulted
125    /// first when two rules are otherwise equal.
126    ///
127    /// A directory with no `globs2` falls back to its `globs` — the
128    /// older file, which is what a hand-made database is most likely to
129    /// have. Never both from the same directory: they say the same
130    /// thing, and reading both would double every rule in it.
131    pub fn load_from(dirs: &[std::path::PathBuf]) -> Globs {
132        let mut globs = Vec::new();
133        for dir in dirs {
134            let mime = dir.join("mime");
135            if let Ok(text) = std::fs::read_to_string(mime.join("globs2")) {
136                globs.extend(Globs::parse(&text).globs);
137            } else if let Ok(text) = std::fs::read_to_string(mime.join("globs")) {
138                globs.extend(Globs::parse_legacy(&text).globs);
139            }
140        }
141        Globs { globs }
142    }
143
144    /// The type of a file with this name, or `None` when no rule
145    /// matches.
146    ///
147    /// `None` rather than `application/octet-stream`: "no rule knows
148    /// this name" and "this is a stream of bytes" are different answers,
149    /// and a caller offering to open the file needs to tell them apart —
150    /// the first deserves "nothing here knows what this is", the second
151    /// would be a claim the database never made. This is the same
152    /// distinction as a missing config file versus an unparseable one.
153    ///
154    /// When several rules match, the database's own precedence applies:
155    /// the highest weight wins, then the longest pattern — so
156    /// `archive.tar.gz` is a compressed tar rather than a gzip file,
157    /// because `*.tar.gz` is longer than `*.gz`. A case-sensitive rule
158    /// only matches the spelling it gives.
159    ///
160    /// **Case is tried exactly first, then folded.** `a.c` and `a.C`
161    /// are different files: the database has `*.c` for C and `*.C` for
162    /// C++, and a lookup that lowercases everything up front makes one
163    /// of them unreachable. So the name is matched as written, and only
164    /// if nothing matches is it tried again in lower case — which is
165    /// what makes `script.PL` still a perl script.
166    ///
167    /// **Beyond that the specification says the result is undefined**,
168    /// and implementations do differ. Two tie-breaks, in this order,
169    /// each chosen against what the rest of this machine answers:
170    ///
171    /// 1. A registered type beats an `x-` one. `*.obj` is claimed by
172    ///    `model/obj`, `application/x-coff` and `application/x-tgif` at
173    ///    the same weight; `x-` means unregistered, and a Wavefront
174    ///    model is the better reading of a `.obj` than a COFF object
175    ///    file. `gio` answers `model/obj` here too.
176    /// 2. Then the first rule wins, which is what the reference
177    ///    implementation does (it skips a second rule for an extension
178    ///    it has already seen). `*.json` is claimed by
179    ///    `application/json` and `application/schema+json`; the first
180    ///    is the answer everything else on the machine gives.
181    pub fn type_of(&self, path: &Path) -> Option<&str> {
182        let name = path.file_name()?.to_str()?;
183        // Collected once for the whole sweep rather than once per rule
184        // — see [`Glob::chars`].
185        let chars: Vec<char> = name.chars().collect();
186        self.best_match(&chars).or_else(|| {
187            let lowered: Vec<char> = name.to_lowercase().chars().collect();
188            (lowered != chars).then(|| self.best_match(&lowered)).flatten()
189        })
190    }
191
192    /// The best rule for a name, taken exactly as given.
193    fn best_match(&self, name: &[char]) -> Option<&str> {
194        self.globs
195            .iter()
196            .filter(|glob| matches_at(&glob.chars, name))
197            // Folded rather than `max_by_key`, which keeps the *last* of
198            // several equal maxima — the opposite of what is wanted.
199            .fold(None, |best: Option<&Glob>, glob| match best {
200                Some(best) if rank(best) >= rank(glob) => Some(best),
201                _ => Some(glob),
202            })
203            .map(|glob| glob.mime.as_str())
204    }
205
206    /// Every type whose pattern matches this name, best first.
207    ///
208    /// For `mimetype --all`, and for a caller that would rather see an
209    /// ambiguity than have it resolved: `photo.jpg` matches one rule,
210    /// but `archive.tar.gz` matches both `*.tar.gz` and `*.gz` and a
211    /// person may want to know that.
212    pub fn all_matches(&self, path: &Path) -> Vec<&str> {
213        let Some(name) = path.file_name().and_then(|n| n.to_str()) else { return Vec::new() };
214        let exact: Vec<char> = name.chars().collect();
215        let lowered: Vec<char> = name.to_lowercase().chars().collect();
216        let mut matched: Vec<&Glob> = self
217            .globs
218            .iter()
219            .filter(|glob| {
220                matches_at(&glob.chars, &exact)
221                    || (!glob.case_sensitive && matches_at(&glob.chars, &lowered))
222            })
223            .collect();
224        matched.sort_by_key(|glob| std::cmp::Reverse((glob.weight, glob.pattern.len())));
225        let mut out: Vec<&str> = Vec::new();
226        for glob in matched {
227            if !out.contains(&glob.mime.as_str()) {
228                out.push(&glob.mime);
229            }
230        }
231        out
232    }
233
234    /// Every type some rule can name a file after, in load order and
235    /// with repeats — one type usually has several patterns.
236    ///
237    /// This is as close to "the types this machine knows about" as the
238    /// database gets. It is deliberately the *glob* file and not the
239    /// union of everything mentioned anywhere: a type no filename can
240    /// produce is not one somebody will go looking for by name, and
241    /// `subclasses` alone names hundreds of them.
242    pub fn mimes(&self) -> impl Iterator<Item = &str> {
243        self.globs.iter().map(|glob| glob.mime.as_str())
244    }
245
246    /// Whether anything at all was loaded. An empty database is not an
247    /// error — a machine may genuinely have no `shared-mime-info` — but
248    /// a caller that shows a chooser wants to say so rather than
249    /// reporting that every file is of unknown type.
250    pub fn is_empty(&self) -> bool {
251        self.globs.is_empty()
252    }
253
254}
255
256/// How good a rule is: the database's weight, then the length of the
257/// pattern, then whether the type is a registered one rather than an
258/// `x-` name. See [`Globs::type_of`] for where each comes from.
259fn rank(glob: &Glob) -> (u32, usize, bool) {
260    let registered = !glob.mime.split('/').nth(1).is_some_and(|sub| sub.starts_with("x-"));
261    (glob.weight, glob.pattern.len(), registered)
262}
263
264/// `fnmatch` over the subset `globs2` actually uses: `*`, `?`, and
265/// `[abc]` / `[0-9]` character classes.
266///
267/// Written out rather than pulled in as a dependency because the subset
268/// is this small and the crate is a leaf on purpose. Recursion is on the
269/// pattern, so a pathological pattern costs pattern length, not file
270/// length. Both sides arrive as characters already — see [`Glob::chars`].
271fn matches_at(pattern: &[char], name: &[char]) -> bool {
272    match pattern.first() {
273        None => name.is_empty(),
274        Some('*') => {
275            // Every split of the remaining name, shortest first.
276            (0..=name.len()).any(|skip| matches_at(&pattern[1..], &name[skip..]))
277        }
278        Some('?') => !name.is_empty() && matches_at(&pattern[1..], &name[1..]),
279        Some('[') => {
280            let Some(close) = pattern.iter().position(|c| *c == ']') else {
281                // An unclosed class is a literal bracket, which is what
282                // a shell does with it too.
283                return name.first() == Some(&'[') && matches_at(&pattern[1..], &name[1..]);
284            };
285            let Some(c) = name.first() else { return false };
286            let (negated, set) = match pattern.get(1) {
287                Some('!') => (true, &pattern[2..close]),
288                _ => (false, &pattern[1..close]),
289            };
290            if in_class(set, *c) == negated {
291                return false;
292            }
293            matches_at(&pattern[close + 1..], &name[1..])
294        }
295        Some(literal) => name.first() == Some(literal) && matches_at(&pattern[1..], &name[1..]),
296    }
297}
298
299/// Whether `c` is in a character class body, ranges included.
300fn in_class(set: &[char], c: char) -> bool {
301    let mut i = 0;
302    while i < set.len() {
303        if set.get(i + 1) == Some(&'-') {
304            if let Some(end) = set.get(i + 2) {
305                if set[i] <= c && c <= *end {
306                    return true;
307                }
308                i += 3;
309                continue;
310            }
311        }
312        if set[i] == c {
313            return true;
314        }
315        i += 1;
316    }
317    false
318}
319
320#[cfg(test)]
321mod tests {
322    use super::*;
323    use std::path::PathBuf;
324
325    fn db() -> Globs {
326        Globs::parse(
327            "# generated, do not edit\n\
328             50:model/stl:*.stl\n\
329             50:model/3mf:*.3mf\n\
330             50:application/zip:*.zip\n\
331             50:image/png:*.png\n\
332             40:image/apng:*.png\n\
333             50:application/gzip:*.gz\n\
334             50:application/x-compressed-tar:*.tar.gz\n\
335             50:text/x-makefile:Makefile:cs\n\
336             60:application/x-sharedlib:*.so.[0-9]*\n\
337             0:application/x-modrinth-modpack+zip:__NOGLOBS__\n",
338        )
339    }
340
341    /// The case this whole module exists for: a 3MF really is a zip, and
342    /// the name is what says it is a model.
343    #[test]
344    fn a_model_stored_as_a_zip_is_still_a_model() {
345        assert_eq!(db().type_of(&PathBuf::from("/d/part.3mf")), Some("model/3mf"));
346        assert_eq!(db().type_of(&PathBuf::from("/d/thing.stl")), Some("model/stl"));
347        assert_eq!(db().type_of(&PathBuf::from("/d/archive.zip")), Some("application/zip"));
348    }
349
350    /// Two rules, same weight, same pattern. The first one in the
351    /// database wins — `*.json` really is claimed twice, and picking
352    /// the second one made every `.json` on the machine a JSON
353    /// *schema*.
354    #[test]
355    fn the_first_of_two_equal_rules_wins() {
356        let globs = Globs::parse("50:application/json:*.json\n50:application/schema+json:*.json\n");
357        assert_eq!(globs.type_of(&PathBuf::from("package.json")), Some("application/json"));
358
359        // And the other way round, to prove it is the order and not the
360        // name that decides.
361        let reversed = Globs::parse("50:application/schema+json:*.json\n50:application/json:*.json\n");
362        assert_eq!(reversed.type_of(&PathBuf::from("package.json")), Some("application/schema+json"));
363    }
364
365    #[test]
366    fn the_heavier_rule_wins_and_then_the_longer_one() {
367        assert_eq!(db().type_of(&PathBuf::from("a.png")), Some("image/png"), "50 beats 40");
368        assert_eq!(
369            db().type_of(&PathBuf::from("backup.tar.gz")),
370            Some("application/x-compressed-tar"),
371            "*.tar.gz is longer than *.gz"
372        );
373    }
374
375    /// A name nothing knows is `None`, never octet-stream — see
376    /// [`Globs::type_of`].
377    #[test]
378    fn an_unknown_name_is_unknown_rather_than_a_stream_of_bytes() {
379        assert_eq!(db().type_of(&PathBuf::from("/d/mystery.qqq")), None);
380        assert_eq!(db().type_of(&PathBuf::from("/d/no-extension")), None);
381    }
382
383    /// Both rules are reported, best first, rather than one being
384    /// chosen for the caller.
385    #[test]
386    fn every_matching_rule_can_be_listed() {
387        assert_eq!(
388            db().all_matches(&PathBuf::from("backup.tar.gz")),
389            ["application/x-compressed-tar", "application/gzip"]
390        );
391        assert_eq!(db().all_matches(&PathBuf::from("a.png")), ["image/png", "image/apng"]);
392        assert!(db().all_matches(&PathBuf::from("mystery.qqq")).is_empty());
393    }
394
395    #[test]
396    fn matching_ignores_case_unless_the_rule_asks_otherwise() {
397        assert_eq!(db().type_of(&PathBuf::from("SHOUTING.STL")), Some("model/stl"));
398        assert_eq!(db().type_of(&PathBuf::from("Makefile")), Some("text/x-makefile"));
399        assert_eq!(db().type_of(&PathBuf::from("makefile")), None, "cs means exactly that");
400    }
401
402    /// `a.c` and `a.C` are different files, and the database says so:
403    /// one is C, the other C++. Lowercasing before matching makes the
404    /// second unreachable.
405    #[test]
406    fn an_exact_case_match_is_preferred_to_a_folded_one() {
407        let globs = Globs::parse("50:text/x-c++src:*.C\n50:text/x-csrc:*.c\n");
408        assert_eq!(globs.type_of(&PathBuf::from("main.c")), Some("text/x-csrc"));
409        assert_eq!(globs.type_of(&PathBuf::from("main.C")), Some("text/x-c++src"));
410        // Nothing matches `README.TXT` exactly, so it is folded and
411        // found — the rule that keeps `script.PL` a perl script.
412        let upper = Globs::parse("50:text/plain:*.txt\n");
413        assert_eq!(upper.type_of(&PathBuf::from("README.TXT")), Some("text/plain"));
414    }
415
416    /// Where the specification gives up — same weight, same pattern —
417    /// a registered type beats an `x-` one.
418    #[test]
419    fn a_registered_type_beats_an_unregistered_one_on_a_tie() {
420        let globs = Globs::parse(
421            "50:application/x-coff:*.obj\n50:application/x-tgif:*.obj\n50:model/obj:*.obj\n",
422        );
423        assert_eq!(globs.type_of(&PathBuf::from("bracket.obj")), Some("model/obj"));
424    }
425
426    #[test]
427    fn a_character_class_matches_a_range() {
428        assert_eq!(db().type_of(&PathBuf::from("libc.so.6")), Some("application/x-sharedlib"));
429        assert_eq!(db().type_of(&PathBuf::from("libc.so.x")), None);
430    }
431
432    /// Generated data with something unfamiliar in it must cost that
433    /// line only — the same rule the config parsers follow.
434    #[test]
435    fn an_unreadable_line_costs_that_line_alone() {
436        let globs = Globs::parse("nonsense\n50:text/plain:*.txt\nalso:nonsense\n\n");
437        assert_eq!(globs.type_of(&PathBuf::from("a.txt")), Some("text/plain"));
438        assert!(!globs.is_empty());
439    }
440
441    #[test]
442    fn the_older_globs_format_is_read_when_that_is_all_there_is() {
443        let globs = Globs::parse_legacy(
444            "# a test file\napplication/x-perl:*.pl\napplication/x-compressed-tar:*.tar.gz\n\
445             application/x-gzip:*.gz\ntext/x-makefile:[Mm]akefile\n",
446        );
447        assert_eq!(globs.type_of(&PathBuf::from("script.pl")), Some("application/x-perl"));
448        assert_eq!(
449            globs.type_of(&PathBuf::from("script.tar.gz")),
450            Some("application/x-compressed-tar"),
451            "the longer pattern still wins at equal weight"
452        );
453        assert_eq!(globs.type_of(&PathBuf::from("Makefile")), Some("text/x-makefile"));
454    }
455
456    #[test]
457    fn a_machine_with_no_database_is_empty_rather_than_wrong() {
458        let globs = Globs::load_from(&[PathBuf::from("/nonexistent-xyz")]);
459        assert!(globs.is_empty());
460        assert_eq!(globs.type_of(&PathBuf::from("a.png")), None);
461    }
462}