use std::io::Read;
const SNIFF_BYTES: usize = 16 * 1024;
#[derive(Debug, Clone, PartialEq, Eq)]
struct Rule {
indent: u32,
offset: usize,
value: Vec<u8>,
mask: Option<Vec<u8>>,
range: usize,
}
impl Rule {
fn matches(&self, data: &[u8]) -> bool {
(self.offset..=self.offset + self.range).any(|start| {
let Some(window) = data.get(start..start + self.value.len()) else { return false };
match &self.mask {
None => window == self.value,
Some(mask) => window
.iter()
.zip(&self.value)
.zip(mask)
.all(|((byte, value), mask)| byte & mask == value & mask),
}
})
}
}
#[derive(Debug, Clone, PartialEq, Eq)]
struct Section {
priority: u32,
mime: String,
rules: Vec<Rule>,
}
impl Section {
fn matches(&self, data: &[u8]) -> bool {
matches_level(&self.rules, data, 0)
}
}
fn matches_level(rules: &[Rule], data: &[u8], indent: u32) -> bool {
let mut i = 0;
while i < rules.len() {
if rules[i].indent != indent {
i += 1;
continue;
}
if rules[i].matches(data) {
let children_start = i + 1;
let children_end = rules[children_start..]
.iter()
.position(|rule| rule.indent <= indent)
.map(|offset| children_start + offset)
.unwrap_or(rules.len());
if children_start == children_end {
return true;
}
if matches_level(&rules[children_start..children_end], data, indent + 1) {
return true;
}
}
i += 1;
}
false
}
#[derive(Debug, Clone, Default, PartialEq, Eq)]
pub struct Magic {
sections: Vec<Section>,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Match {
pub mime: String,
pub priority: u32,
}
impl Magic {
pub fn parse(bytes: &[u8]) -> Magic {
let mut magic = Magic::default();
let Some(mut rest) = bytes.strip_prefix(b"MIME-Magic\0\n") else {
return magic;
};
while !rest.is_empty() {
if rest[0] != b'[' {
break;
}
let Some(close) = rest.iter().position(|b| *b == b']') else { break };
let header = &rest[1..close];
rest = &rest[close + 1..];
if rest.first() == Some(&b'\n') {
rest = &rest[1..];
}
let Some((priority, mime)) = split_header(header) else { break };
let mut section = Section { priority, mime, rules: Vec::new() };
while !rest.is_empty() && rest[0] != b'[' {
match parse_rule(rest) {
Some((rule, tail)) => {
section.rules.push(rule);
rest = tail;
}
None => {
rest = &[];
break;
}
}
}
if !section.rules.is_empty() {
magic.sections.push(section);
}
}
magic
}
pub fn load_from(dirs: &[std::path::PathBuf]) -> Magic {
let mut magic = Magic::default();
for dir in dirs {
if let Ok(bytes) = std::fs::read(dir.join("mime").join("magic")) {
magic.sections.extend(Magic::parse(&bytes).sections);
}
}
magic.sections.sort_by_key(|section| std::cmp::Reverse(section.priority));
magic
}
pub fn of_data(&self, data: &[u8]) -> Option<Match> {
self.sections
.iter()
.find(|section| section.matches(data))
.map(|section| Match { mime: section.mime.clone(), priority: section.priority })
}
pub fn of_file(&self, path: &std::path::Path) -> Option<Match> {
self.of_data(&head(path)?)
}
pub fn all_of_data(&self, data: &[u8]) -> Vec<Match> {
self.sections
.iter()
.filter(|section| section.matches(data))
.map(|section| Match { mime: section.mime.clone(), priority: section.priority })
.collect()
}
pub fn is_empty(&self) -> bool {
self.sections.is_empty()
}
}
pub fn head(path: &std::path::Path) -> Option<Vec<u8>> {
let mut file = std::fs::File::open(path).ok()?;
let mut buffer = vec![0u8; SNIFF_BYTES];
let read = read_up_to(&mut file, &mut buffer)?;
buffer.truncate(read);
Some(buffer)
}
fn read_up_to(file: &mut std::fs::File, buffer: &mut [u8]) -> Option<usize> {
let mut filled = 0;
while filled < buffer.len() {
match file.read(&mut buffer[filled..]) {
Ok(0) => break,
Ok(n) => filled += n,
Err(e) if e.kind() == std::io::ErrorKind::Interrupted => continue,
Err(_) => return None,
}
}
Some(filled)
}
fn split_header(header: &[u8]) -> Option<(u32, String)> {
let header = std::str::from_utf8(header).ok()?;
let (priority, mime) = header.split_once(':')?;
Some((priority.trim().parse().ok()?, mime.trim().to_string()))
}
fn parse_rule(bytes: &[u8]) -> Option<(Rule, &[u8])> {
let gt = bytes.iter().position(|b| *b == b'>')?;
let indent: u32 = match gt {
0 => 0,
_ => std::str::from_utf8(&bytes[..gt]).ok()?.trim().parse().ok()?,
};
let rest = &bytes[gt + 1..];
let eq = rest.iter().position(|b| *b == b'=')?;
let offset: usize = std::str::from_utf8(&rest[..eq]).ok()?.trim().parse().ok()?;
let rest = &rest[eq + 1..];
let length = usize::from(u16::from_be_bytes([*rest.first()?, *rest.get(1)?]));
let value = rest.get(2..2 + length)?.to_vec();
let mut rest = &rest[2 + length..];
let mut mask = None;
if rest.first() == Some(&b'&') {
mask = Some(rest.get(1..1 + length)?.to_vec());
rest = &rest[1 + length..];
}
let mut word_size = 1usize;
if rest.first() == Some(&b'~') {
let (number, tail) = take_number(&rest[1..])?;
word_size = number.max(1);
rest = tail;
}
let mut range = 0usize;
if rest.first() == Some(&b'+') {
let (number, tail) = take_number(&rest[1..])?;
range = number;
rest = tail;
}
match rest.first() {
Some(b'\n') => rest = &rest[1..],
None => {}
Some(_) => return None,
}
let (value, mask) = swap_words(value, mask, word_size);
Some((Rule { indent, offset, value, mask, range }, rest))
}
fn take_number(bytes: &[u8]) -> Option<(usize, &[u8])> {
let end = bytes.iter().position(|b| !b.is_ascii_digit()).unwrap_or(bytes.len());
let number = std::str::from_utf8(&bytes[..end]).ok()?.parse().ok()?;
Some((number, &bytes[end..]))
}
fn swap_words(value: Vec<u8>, mask: Option<Vec<u8>>, word_size: usize) -> (Vec<u8>, Option<Vec<u8>>) {
if word_size <= 1 || !cfg!(target_endian = "little") || !value.len().is_multiple_of(word_size) {
return (value, mask);
}
let swap = |bytes: Vec<u8>| -> Vec<u8> {
bytes.chunks(word_size).flat_map(|word| word.iter().rev().copied()).collect()
};
(swap(value), mask.map(swap))
}
#[cfg(test)]
mod tests {
use super::*;
type TestRule<'a> = (u32, usize, &'a [u8], Option<&'a [u8]>, usize);
fn magic_file(sections: &[(u32, &str, &[TestRule<'_>])]) -> Vec<u8> {
let mut out = b"MIME-Magic\0\n".to_vec();
for (priority, mime, rules) in sections {
out.extend(format!("[{priority}:{mime}]\n").into_bytes());
for (indent, offset, value, mask, range) in *rules {
if *indent > 0 {
out.extend(indent.to_string().into_bytes());
}
out.extend(format!(">{offset}=").into_bytes());
out.extend((value.len() as u16).to_be_bytes());
out.extend(*value);
if let Some(mask) = mask {
out.push(b'&');
out.extend(*mask);
}
if *range > 0 {
out.extend(format!("+{range}").into_bytes());
}
out.push(b'\n');
}
}
out
}
fn pdf_and_png() -> Magic {
Magic::parse(&magic_file(&[
(50, "application/pdf", &[(0, 0, b"%PDF-".as_slice(), None, 0)]),
(50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
]))
}
#[test]
fn a_files_first_bytes_name_its_type() {
let magic = pdf_and_png();
assert_eq!(magic.of_data(b"%PDF-1.7 ...").unwrap().mime, "application/pdf");
assert_eq!(magic.of_data(b"\x89PNG\r\n\x1a\n").unwrap().mime, "image/png");
assert_eq!(magic.of_data(b"hello"), None);
assert_eq!(magic.of_data(b""), None, "nothing to go on");
}
#[test]
fn a_value_containing_a_newline_is_read_whole() {
let magic = Magic::parse(&magic_file(&[
(50, "text/x-two-lines", &[(0, 0, b"first\nsecond".as_slice(), None, 0)]),
(50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
]));
assert_eq!(magic.of_data(b"first\nsecond and more").unwrap().mime, "text/x-two-lines");
assert_eq!(
magic.of_data(b"\x89PNG").unwrap().mime,
"image/png",
"the section after it survived the parse"
);
}
#[test]
fn a_rule_can_look_within_a_range_rather_than_at_one_offset() {
let magic = Magic::parse(&magic_file(&[(
50,
"text/x-somewhere",
&[(0, 0, b"needle".as_slice(), None, 20)],
)]));
assert!(magic.of_data(b"..........needle").is_some(), "found later in the range");
assert!(magic.of_data(b"needle").is_some(), "and at the offset itself");
assert!(
magic.of_data(b"..............................needle").is_none(),
"past the range"
);
}
#[test]
fn a_mask_ignores_the_bits_it_clears() {
let magic = Magic::parse(&magic_file(&[(
50,
"x/masked",
&[(0, 0, b"\xf0".as_slice(), Some(b"\xf0".as_slice()), 0)],
)]));
assert!(magic.of_data(b"\xff").is_some(), "high nibble matches, low ignored");
assert!(magic.of_data(b"\xf0").is_some());
assert!(magic.of_data(b"\x0f").is_none());
}
#[test]
fn a_nested_rule_is_a_further_condition_on_its_parent() {
let magic = Magic::parse(&magic_file(&[(
50,
"application/docbook+xml",
&[
(0, 0, b"<?xml".as_slice(), None, 0),
(1, 0, b"-//OASIS//DTD DocBook".as_slice(), None, 200),
],
)]));
assert!(
magic.of_data(b"<?xml version='1.0'?><!DOCTYPE book PUBLIC \"-//OASIS//DTD DocBook XML\">").is_some()
);
assert!(magic.of_data(b"<?xml version='1.0'?><html/>").is_none(), "XML, but not DocBook");
}
#[test]
fn any_one_child_satisfies_its_parent() {
let magic = Magic::parse(&magic_file(&[(
50,
"x/either",
&[
(0, 0, b"HEAD".as_slice(), None, 0),
(1, 4, b"one".as_slice(), None, 0),
(1, 4, b"two".as_slice(), None, 0),
],
)]));
assert!(magic.of_data(b"HEADone").is_some());
assert!(magic.of_data(b"HEADtwo").is_some());
assert!(magic.of_data(b"HEADthree").is_none());
}
#[test]
fn the_strongest_rule_is_the_one_reported() {
let magic = Magic::load_from_parsed(vec![
Magic::parse(&magic_file(&[(20, "x/weak", &[(0, 0, b"AB".as_slice(), None, 0)])])),
Magic::parse(&magic_file(&[(90, "x/strong", &[(0, 0, b"AB".as_slice(), None, 0)])])),
]);
let found = magic.of_data(b"ABCD").unwrap();
assert_eq!(found.mime, "x/strong");
assert_eq!(found.priority, 90);
let all: Vec<String> = magic.all_of_data(b"ABCD").into_iter().map(|m| m.mime).collect();
assert_eq!(all, ["x/strong", "x/weak"], "and --all lists both, best first");
}
#[test]
fn a_file_that_is_not_this_format_is_ignored_rather_than_guessed_at() {
assert!(Magic::parse(b"not a magic file at all").is_empty());
assert!(Magic::parse(b"").is_empty());
}
#[test]
fn a_truncated_file_keeps_what_was_read() {
let mut bytes = magic_file(&[
(50, "application/pdf", &[(0, 0, b"%PDF-".as_slice(), None, 0)]),
(50, "image/png", &[(0, 0, b"\x89PNG".as_slice(), None, 0)]),
]);
bytes.truncate(bytes.len() - 3);
let magic = Magic::parse(&bytes);
assert_eq!(magic.of_data(b"%PDF-1.7").unwrap().mime, "application/pdf");
}
#[test]
fn a_word_sized_value_is_swapped_once_rather_than_per_comparison() {
let (value, mask) = swap_words(vec![0x12, 0x34, 0x56, 0x78], Some(vec![0xff, 0x00, 0xff, 0x00]), 2);
if cfg!(target_endian = "little") {
assert_eq!(value, vec![0x34, 0x12, 0x78, 0x56]);
assert_eq!(mask, Some(vec![0x00, 0xff, 0x00, 0xff]));
} else {
assert_eq!(value, vec![0x12, 0x34, 0x56, 0x78], "big-endian needs no swap");
}
}
#[test]
fn a_files_contents_are_read_from_disk_and_bounded() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("anonymous");
let mut contents = b"%PDF-1.7\n".to_vec();
contents.extend(std::iter::repeat_n(b'x', SNIFF_BYTES * 4));
std::fs::write(&path, &contents).unwrap();
let read = head(&path).expect("readable");
assert_eq!(read.len(), SNIFF_BYTES, "a big file costs the window, not its size");
assert_eq!(pdf_and_png().of_file(&path).unwrap().mime, "application/pdf");
assert_eq!(pdf_and_png().of_file(&dir.path().join("missing")), None);
assert_eq!(head(&dir.path().join("missing")), None);
}
#[test]
fn a_small_file_reads_as_exactly_itself() {
let dir = tempfile::tempdir().unwrap();
let path = dir.path().join("small");
std::fs::write(&path, b"hello").unwrap();
assert_eq!(head(&path).unwrap(), b"hello");
}
impl Magic {
fn load_from_parsed(parts: Vec<Magic>) -> Magic {
let mut magic = Magic::default();
for part in parts {
magic.sections.extend(part.sections);
}
magic.sections.sort_by_key(|section| std::cmp::Reverse(section.priority));
magic
}
}
}