Skip to main content

kernel_abi_tools/
filter.rs

1//! Classic-BPF seccomp filters built directly from the data, for the local
2//! runner: no container, no libseccomp, and no root.
3//!
4//! The program makes the same decisions as [`crate::seccomp_profile`]:
5//!
6//! 1. Any architecture other than x86_64 (e.g. `int 0x80` i386 calls) and any
7//!    x32 syscall number returns `ENOSYS`.
8//! 2. `madvise` returns `EINVAL` unless the low 32 bits of the advice (the
9//!    kernel reads an `int`) are a value the target kernel knows.
10//! 3. Syscalls the target provides are allowed; everything else, including
11//!    numbers this crate's data does not know, returns `ENOSYS`.
12//!
13//! Installing a filter needs no privilege once the process has set
14//! `PR_SET_NO_NEW_PRIVS`. Like the profiles, the filter is a testing aid and
15//! not a security boundary.
16
17use crate::{madvise_accepts, madvise_advice, present_syscalls, Target, EINVAL, ENOSYS};
18
19/// `AUDIT_ARCH_X86_64`: `EM_X86_64 | __AUDIT_ARCH_64BIT | __AUDIT_ARCH_LE`.
20pub const AUDIT_ARCH_X86_64: u32 = 0xc000_003e;
21/// Syscall numbers at or above this bit belong to the x32 ABI.
22pub const X32_SYSCALL_BIT: u32 = 0x4000_0000;
23/// x86_64 `madvise` syscall number.
24pub const SYS_MADVISE: u32 = 28;
25
26/// `SECCOMP_RET_ALLOW`.
27pub const RET_ALLOW: u32 = 0x7fff_0000;
28/// `SECCOMP_RET_ERRNO`; the errno goes in the low 16 bits.
29pub const RET_ERRNO: u32 = 0x0005_0000;
30
31// Opcodes used by the program (linux/bpf_common.h).
32const LD_W_ABS: u16 = 0x20; // BPF_LD | BPF_W | BPF_ABS
33const JEQ_K: u16 = 0x15; // BPF_JMP | BPF_JEQ | BPF_K
34const JGT_K: u16 = 0x25; // BPF_JMP | BPF_JGT | BPF_K
35const JGE_K: u16 = 0x35; // BPF_JMP | BPF_JGE | BPF_K
36const RET_K: u16 = 0x06; // BPF_RET | BPF_K
37
38// Offsets into `struct seccomp_data` (x86_64 is little-endian, so the low
39// word of an argument comes first).
40const OFF_NR: u32 = 0;
41const OFF_ARCH: u32 = 4;
42const OFF_ARG2_LO: u32 = 16 + 2 * 8;
43
44/// The kernel's `BPF_MAXINSNS`.
45pub const MAX_INSNS: usize = 4096;
46
47/// One `struct sock_filter` instruction.
48#[repr(C)]
49#[derive(Clone, Copy, Debug, PartialEq, Eq)]
50pub struct Insn {
51    pub code: u16,
52    pub jt: u8,
53    pub jf: u8,
54    pub k: u32,
55}
56
57const fn stmt(code: u16, k: u32) -> Insn {
58    Insn {
59        code,
60        jt: 0,
61        jf: 0,
62        k,
63    }
64}
65
66const fn jump(code: u16, k: u32, jt: u8, jf: u8) -> Insn {
67    Insn { code, jt, jf, k }
68}
69
70/// Builds the filter program for `target`.
71///
72/// Allowed syscall numbers are merged into contiguous ranges; each range is
73/// three instructions with only short forward jumps, so the program needs no
74/// jump longer than the 255 instructions BPF allows.
75pub fn program(target: &Target) -> Vec<Insn> {
76    let enosys = RET_ERRNO | ENOSYS as u32;
77    let einval = RET_ERRNO | EINVAL as u32;
78    let mut p = vec![
79        stmt(LD_W_ABS, OFF_ARCH),
80        jump(JEQ_K, AUDIT_ARCH_X86_64, 1, 0),
81        stmt(RET_K, enosys),
82        stmt(LD_W_ABS, OFF_NR),
83        jump(JGE_K, X32_SYSCALL_BIT, 0, 1),
84        stmt(RET_K, enosys),
85    ];
86
87    let mut nrs: Vec<u32> = present_syscalls(target).iter().map(|s| s.nr).collect();
88    nrs.sort_unstable();
89    nrs.dedup();
90    if nrs.binary_search(&SYS_MADVISE).is_ok() {
91        let accepted: Vec<u32> = madvise_advice()
92            .iter()
93            .map(|a| a.value)
94            .filter(|&v| madvise_accepts(target.kernel, v))
95            .collect();
96        // Not madvise: skip the advice block (the load, one test per value,
97        // EINVAL, and ALLOW) and land on the reload of the number.
98        let skip = u8::try_from(accepted.len() + 3).expect("madvise block fits a BPF jump");
99        p.push(jump(JEQ_K, SYS_MADVISE, 0, skip));
100        p.push(stmt(LD_W_ABS, OFF_ARG2_LO));
101        let n = accepted.len();
102        for (i, v) in accepted.iter().enumerate() {
103            // Jump over the remaining tests and the EINVAL to the ALLOW.
104            let to_allow = u8::try_from(n - i).expect("madvise block fits a BPF jump");
105            p.push(jump(JEQ_K, *v, to_allow, 0));
106        }
107        p.push(stmt(RET_K, einval));
108        p.push(stmt(RET_K, RET_ALLOW));
109        // The accumulator holds the advice now; reload the number.
110        p.push(stmt(LD_W_ABS, OFF_NR));
111    }
112
113    for (lo, hi) in ranges(&nrs) {
114        p.push(jump(JGE_K, lo, 0, 2));
115        p.push(jump(JGT_K, hi, 1, 0));
116        p.push(stmt(RET_K, RET_ALLOW));
117    }
118    p.push(stmt(RET_K, enosys));
119    assert!(p.len() <= MAX_INSNS, "filter exceeds BPF_MAXINSNS");
120    p
121}
122
123fn ranges(sorted: &[u32]) -> Vec<(u32, u32)> {
124    let mut out: Vec<(u32, u32)> = Vec::new();
125    for &n in sorted {
126        match out.last_mut() {
127            Some((_, hi)) if *hi + 1 == n => *hi = n,
128            _ => out.push((n, n)),
129        }
130    }
131    out
132}
133
134/// Runs `prog` on one syscall the way the kernel does and returns the
135/// `SECCOMP_RET_*` value. Supports only the opcodes [`program`] emits.
136pub fn evaluate(prog: &[Insn], arch: u32, nr: u32, args: [u64; 6]) -> Result<u32, String> {
137    let word = |off: u32| -> Result<u32, String> {
138        match off {
139            OFF_NR => Ok(nr),
140            OFF_ARCH => Ok(arch),
141            16..=63 if off.is_multiple_of(4) => {
142                let arg = args[(off as usize - 16) / 8];
143                Ok(if off.is_multiple_of(8) {
144                    arg as u32
145                } else {
146                    (arg >> 32) as u32
147                })
148            }
149            _ => Err(format!("load from unsupported offset {off}")),
150        }
151    };
152    let mut acc = 0u32;
153    let mut pc = 0usize;
154    while let Some(insn) = prog.get(pc) {
155        pc += 1;
156        let taken = match insn.code {
157            LD_W_ABS => {
158                acc = word(insn.k)?;
159                continue;
160            }
161            RET_K => return Ok(insn.k),
162            JEQ_K => acc == insn.k,
163            JGT_K => acc > insn.k,
164            JGE_K => acc >= insn.k,
165            other => return Err(format!("unsupported opcode {other:#x}")),
166        };
167        pc += usize::from(if taken { insn.jt } else { insn.jf });
168    }
169    Err("program fell off the end".to_string())
170}
171
172/// Installs filters in the calling thread (and what it executes).
173#[cfg(target_os = "linux")]
174pub mod install {
175    use super::Insn;
176    use std::io;
177
178    const PR_SET_NO_NEW_PRIVS: i32 = 38;
179    const PR_SET_SECCOMP: i32 = 22;
180    const SECCOMP_MODE_FILTER: u64 = 2;
181
182    /// `struct sock_fprog`.
183    #[repr(C)]
184    pub struct Prog<'a> {
185        len: u16,
186        filter: *const Insn,
187        _insns: std::marker::PhantomData<&'a [Insn]>,
188    }
189
190    impl<'a> Prog<'a> {
191        pub fn new(insns: &'a [Insn]) -> Self {
192            Prog {
193                len: u16::try_from(insns.len()).expect("filter fits sock_fprog"),
194                filter: insns.as_ptr(),
195                _insns: std::marker::PhantomData,
196            }
197        }
198    }
199
200    extern "C" {
201        fn prctl(option: i32, arg2: u64, arg3: u64, arg4: u64, arg5: u64) -> i32;
202    }
203
204    /// Sets `no_new_privs` and installs `prog` for the calling thread. It
205    /// cannot be removed and is inherited across `fork` and `execve`.
206    ///
207    /// Only calls `prctl`, so it is async-signal-safe and may run between
208    /// `fork` and `exec` (for example from `CommandExt::pre_exec`).
209    pub fn apply(prog: &Prog<'_>) -> io::Result<()> {
210        // SAFETY: plain prctl calls; `prog` outlives them and points at
211        // `len` valid instructions.
212        unsafe {
213            if prctl(PR_SET_NO_NEW_PRIVS, 1, 0, 0, 0) != 0 {
214                return Err(io::Error::last_os_error());
215            }
216            let ptr = prog as *const Prog<'_> as u64;
217            if prctl(PR_SET_SECCOMP, SECCOMP_MODE_FILTER, ptr, 0, 0) != 0 {
218                return Err(io::Error::last_os_error());
219            }
220        }
221        Ok(())
222    }
223}
224
225#[cfg(test)]
226mod tests {
227    use super::*;
228    use crate::{syscalls, table_ceiling, KernelVersion};
229
230    fn kv(s: &str) -> KernelVersion {
231        KernelVersion::parse(s).unwrap()
232    }
233
234    fn eval(prog: &[Insn], nr: u32, a2: u64) -> u32 {
235        evaluate(prog, AUDIT_ARCH_X86_64, nr, [0, 0, a2, 0, 0, 0]).unwrap()
236    }
237
238    /// The filter must agree with the data (and so with the JSON profile)
239    /// for every known syscall, unknown numbers, and every advice value.
240    fn assert_matches_data(target: &Target) {
241        let prog = program(target);
242        let present: Vec<u32> = present_syscalls(target).iter().map(|s| s.nr).collect();
243        let max_nr = syscalls().iter().map(|s| s.nr).max().unwrap();
244        for nr in 0..=max_nr + 64 {
245            if nr == SYS_MADVISE {
246                continue;
247            }
248            let want = if present.contains(&nr) {
249                RET_ALLOW
250            } else {
251                RET_ERRNO | ENOSYS as u32
252            };
253            assert_eq!(eval(&prog, nr, 0), want, "{}: nr {nr}", target.label);
254        }
255        let max_advice = madvise_advice().iter().map(|a| a.value).max().unwrap();
256        for v in 0..=max_advice + 8 {
257            let want = if madvise_accepts(target.kernel, v) {
258                RET_ALLOW
259            } else {
260                RET_ERRNO | EINVAL as u32
261            };
262            assert_eq!(eval(&prog, SYS_MADVISE, v as u64), want, "advice {v}");
263            // Only the low 32 bits count, as with SCMP_CMP_MASKED_EQ.
264            let high = (0xffff_ffffu64 << 32) | v as u64;
265            assert_eq!(
266                eval(&prog, SYS_MADVISE, high),
267                want,
268                "advice {v} + high bits"
269            );
270        }
271    }
272
273    #[test]
274    fn agrees_with_the_data_for_generic_kernels() {
275        for k in crate::GENERIC_KERNELS {
276            assert_matches_data(&Target::kernel(kv(k)));
277        }
278        assert_matches_data(&Target::kernel(table_ceiling()));
279    }
280
281    #[test]
282    fn agrees_with_the_data_for_distro_backports() {
283        for d in crate::distros() {
284            assert_matches_data(&Target::distro(&d));
285        }
286    }
287
288    #[test]
289    fn blocks_other_architectures_and_x32() {
290        let prog = program(&Target::kernel(table_ceiling()));
291        let enosys = RET_ERRNO | ENOSYS as u32;
292        const AUDIT_ARCH_I386: u32 = 0x4000_0003;
293        assert_eq!(evaluate(&prog, AUDIT_ARCH_I386, 0, [0; 6]), Ok(enosys));
294        assert_eq!(eval(&prog, X32_SYSCALL_BIT, 0), enosys);
295        assert_eq!(eval(&prog, X32_SYSCALL_BIT | 1, 0), enosys);
296        assert_eq!(eval(&prog, u32::MAX, 0), enosys);
297        assert_eq!(eval(&prog, 0, 0), RET_ALLOW); // read
298    }
299
300    #[test]
301    fn stays_well_under_the_instruction_limit() {
302        let len = program(&Target::kernel(kv("3.10"))).len();
303        assert!(len < 512, "{len} instructions");
304    }
305
306    #[test]
307    fn instruction_layout_matches_sock_filter() {
308        assert_eq!(std::mem::size_of::<Insn>(), 8);
309    }
310}