use std::collections::BTreeMap;
use anyhow::{Context, Result, anyhow};
use seccompiler::{
BpfProgram, SeccompAction, SeccompCmpArgLen, SeccompCmpOp, SeccompCondition, SeccompFilter, SeccompRule, TargetArch,
};
use super::policy::{SECCOMP_PROFILE_VERSION, SeccompProfile};
const SYS_UMOUNT: Option<i64> = Some(libc::SYS_umount2);
#[cfg(target_arch = "x86_64")]
const SYS_IOPL: Option<i64> = Some(libc::SYS_iopl);
#[cfg(not(target_arch = "x86_64"))]
const SYS_IOPL: Option<i64> = None;
#[cfg(target_arch = "x86_64")]
const SYS_IOPERM: Option<i64> = Some(libc::SYS_ioperm);
#[cfg(not(target_arch = "x86_64"))]
const SYS_IOPERM: Option<i64> = None;
const NAMESPACE_CLONE_FLAGS: &[libc::c_int] = &[
libc::CLONE_NEWNS,
libc::CLONE_NEWCGROUP,
libc::CLONE_NEWUTS,
libc::CLONE_NEWIPC,
libc::CLONE_NEWUSER,
libc::CLONE_NEWPID,
libc::CLONE_NEWNET,
];
#[cfg(target_arch = "x86_64")]
const X32_SYSCALL_BIT: i64 = 0x4000_0000;
const TIOCSTI_REQUEST: u64 = 0x5412;
const TIOCSCTTY_REQUEST: u64 = 0x540E;
pub fn apply_seccomp_filter(profile: &SeccompProfile) -> Result<()> {
if profile.log_only() {
tracing::debug!("seccomp profile is log_only; installing no filter");
return Ok(());
}
tracing::debug!(
version = SECCOMP_PROFILE_VERSION,
blocked = profile.blocked_syscalls().len(),
"installing seccomp blocklist"
);
set_no_new_privs()?;
let arch = target_arch().ok_or_else(|| anyhow!("seccomp filtering is unsupported on this architecture"))?;
let primary = SeccompFilter::new(
primary_rules(profile)?,
SeccompAction::Allow,
SeccompAction::Errno(u32::try_from(libc::EPERM).context("EPERM conversion")?),
arch,
)
.map_err(|error| anyhow!("seccomp filter construction failed: {error}"))?;
install(primary)?;
if !profile.allow_namespaces() {
let mut clone3 = BTreeMap::new();
let _ = clone3.insert(libc::SYS_clone3, vec![SeccompRule::new(vec![])?]);
#[cfg(target_arch = "x86_64")]
{
let _ = clone3.insert(libc::SYS_clone3 | X32_SYSCALL_BIT, vec![SeccompRule::new(vec![])?]);
}
let filter = SeccompFilter::new(
clone3,
SeccompAction::Allow,
SeccompAction::Errno(u32::try_from(libc::ENOSYS).context("ENOSYS conversion")?),
arch,
)
.map_err(|error| anyhow!("seccomp clone3 filter construction failed: {error}"))?;
install(filter)?;
}
Ok(())
}
fn install(filter: SeccompFilter) -> Result<()> {
let program = BpfProgram::try_from(filter).map_err(|error| anyhow!("seccomp BPF compilation failed: {error}"))?;
seccompiler::apply_filter(&program).map_err(|error| anyhow!("failed to install seccomp filter: {error}"))
}
fn insert_rule(rules: &mut BTreeMap<i64, Vec<SeccompRule>>, nr: i64, rule: SeccompRule) {
#[cfg(target_arch = "x86_64")]
{
rules.entry(nr).or_default().push(rule.clone());
rules.entry(nr | X32_SYSCALL_BIT).or_default().push(rule);
}
#[cfg(not(target_arch = "x86_64"))]
{
rules.entry(nr).or_default().push(rule);
}
}
fn primary_rules(profile: &SeccompProfile) -> Result<BTreeMap<i64, Vec<SeccompRule>>> {
let mut rules: BTreeMap<i64, Vec<SeccompRule>> = BTreeMap::new();
for name in profile.blocked_syscalls() {
match syscall_number(name) {
Some(nr) => insert_rule(&mut rules, nr, SeccompRule::new(vec![])?),
None => tracing::debug!(syscall = %name, "syscall absent on this architecture; not blocking"),
}
}
if !profile.allow_namespaces() {
for flag in NAMESPACE_CLONE_FLAGS {
let mask =
u64::from(u32::try_from(*flag).map_err(|error| anyhow!("CLONE flag conversion failed: {error}"))?);
insert_rule(
&mut rules,
libc::SYS_clone,
SeccompRule::new(vec![SeccompCondition::new(
0,
SeccompCmpArgLen::Dword,
SeccompCmpOp::MaskedEq(mask),
mask,
)?])?,
);
}
}
if !profile.allow_network_sockets() {
for domain in [libc::AF_INET, libc::AF_INET6] {
insert_rule(
&mut rules,
libc::SYS_socket,
SeccompRule::new(vec![SeccompCondition::new(
0,
SeccompCmpArgLen::Dword,
SeccompCmpOp::Eq,
u64::try_from(domain).map_err(|error| anyhow!("socket domain conversion failed: {error}"))?,
)?])?,
);
}
}
for request in [TIOCSTI_REQUEST, TIOCSCTTY_REQUEST] {
insert_rule(
&mut rules,
libc::SYS_ioctl,
SeccompRule::new(vec![SeccompCondition::new(
1,
SeccompCmpArgLen::Dword,
SeccompCmpOp::Eq,
request,
)?])?,
);
}
Ok(rules)
}
fn set_no_new_privs() -> Result<()> {
nix::sys::prctl::set_no_new_privs().map_err(|error| anyhow!("PR_SET_NO_NEW_PRIVS failed: {error}"))
}
fn target_arch() -> Option<TargetArch> {
#[cfg(target_arch = "x86_64")]
{
Some(TargetArch::x86_64)
}
#[cfg(target_arch = "aarch64")]
{
Some(TargetArch::aarch64)
}
#[cfg(target_arch = "riscv64")]
{
Some(TargetArch::riscv64)
}
#[cfg(not(any(target_arch = "x86_64", target_arch = "aarch64", target_arch = "riscv64")))]
{
None
}
}
fn syscall_number(name: &str) -> Option<i64> {
match name {
"ptrace" => Some(libc::SYS_ptrace),
"kcmp" => Some(libc::SYS_kcmp),
"pidfd_getfd" => Some(libc::SYS_pidfd_getfd),
"process_madvise" => Some(libc::SYS_process_madvise),
"process_mrelease" => Some(libc::SYS_process_mrelease),
"mount" => Some(libc::SYS_mount),
"umount" => SYS_UMOUNT,
"umount2" => Some(libc::SYS_umount2),
"open_by_handle_at" => Some(libc::SYS_open_by_handle_at),
"name_to_handle_at" => Some(libc::SYS_name_to_handle_at),
"init_module" => Some(libc::SYS_init_module),
"finit_module" => Some(libc::SYS_finit_module),
"delete_module" => Some(libc::SYS_delete_module),
"kexec_load" => Some(libc::SYS_kexec_load),
"kexec_file_load" => Some(libc::SYS_kexec_file_load),
"bpf" => Some(libc::SYS_bpf),
"perf_event_open" => Some(libc::SYS_perf_event_open),
"userfaultfd" => Some(libc::SYS_userfaultfd),
"io_uring_setup" => Some(libc::SYS_io_uring_setup),
"io_uring_enter" => Some(libc::SYS_io_uring_enter),
"io_uring_register" => Some(libc::SYS_io_uring_register),
"process_vm_readv" => Some(libc::SYS_process_vm_readv),
"process_vm_writev" => Some(libc::SYS_process_vm_writev),
"reboot" => Some(libc::SYS_reboot),
"swapon" => Some(libc::SYS_swapon),
"swapoff" => Some(libc::SYS_swapoff),
"settimeofday" => Some(libc::SYS_settimeofday),
"clock_settime" => Some(libc::SYS_clock_settime),
"adjtimex" => Some(libc::SYS_adjtimex),
"add_key" => Some(libc::SYS_add_key),
"request_key" => Some(libc::SYS_request_key),
"keyctl" => Some(libc::SYS_keyctl),
"ioperm" => SYS_IOPERM,
"iopl" => SYS_IOPL,
"acct" => Some(libc::SYS_acct),
"quotactl" => Some(libc::SYS_quotactl),
"unshare" => Some(libc::SYS_unshare),
"setns" => Some(libc::SYS_setns),
"personality" => Some(libc::SYS_personality),
"clone" => Some(libc::SYS_clone),
"clone3" => Some(libc::SYS_clone3),
_ => None,
}
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn every_default_blocked_syscall_maps_to_a_number() {
for name in super::super::policy::BLOCKED_SYSCALLS {
let x86_only = matches!(*name, "iopl" | "ioperm");
if x86_only && !cfg!(target_arch = "x86_64") {
continue;
}
assert!(syscall_number(name).is_some(), "BLOCKED_SYSCALLS entry {name:?} has no number mapping");
}
}
#[test]
fn blocked_syscalls_have_no_duplicates() {
use std::collections::HashSet;
let mut seen = HashSet::new();
for name in super::super::policy::BLOCKED_SYSCALLS {
assert!(seen.insert(*name), "duplicate BLOCKED_SYSCALLS entry {name:?}");
}
}
#[test]
fn strict_profile_blocks_escalation_primitives() {
let profile = SeccompProfile::strict();
for must_block in [
"ptrace",
"kcmp",
"pidfd_getfd",
"bpf",
"perf_event_open",
"userfaultfd",
"io_uring_setup",
"io_uring_enter",
"io_uring_register",
"process_vm_readv",
"process_vm_writev",
"process_madvise",
"mount",
"open_by_handle_at",
"name_to_handle_at",
"unshare",
"setns",
] {
assert!(profile.blocked_syscalls().iter().any(|s| s == must_block), "strict profile lost {must_block}");
assert!(syscall_number(must_block).is_some(), "no number mapping for {must_block}");
}
}
#[test]
fn primary_rules_block_everything_requested() {
let profile = SeccompProfile::strict();
let rules = primary_rules(&profile).unwrap();
for name in super::super::policy::BLOCKED_SYSCALLS {
let Some(nr) = syscall_number(name) else { continue };
assert!(rules.contains_key(&nr), "syscall {name} missing from primary rules");
}
assert!(rules.contains_key(&libc::SYS_clone), "namespace-flag clone rules required");
assert!(rules.contains_key(&libc::SYS_socket), "socket domain rules required");
assert!(rules.contains_key(&libc::SYS_ioctl), "TIOCSTI/TIOCSCTTY ioctl rules required");
assert!(!rules.contains_key(&libc::SYS_clone3));
}
#[test]
fn ioctl_rules_target_terminal_injection_only() {
let profile = SeccompProfile::strict();
let rules = primary_rules(&profile).unwrap();
let ioctl_rules = rules.get(&libc::SYS_ioctl).expect("ioctl rules required");
assert_eq!(ioctl_rules.len(), 2, "expected exactly TIOCSTI + TIOCSCTTY rules, got {ioctl_rules:?}");
}
#[cfg(target_arch = "x86_64")]
#[test]
fn primary_rules_cover_x32_aliased_numbers() {
let profile = SeccompProfile::strict();
let rules = primary_rules(&profile).unwrap();
for name in super::super::policy::BLOCKED_SYSCALLS {
let Some(nr) = syscall_number(name) else { continue };
assert!(
rules.contains_key(&(nr | X32_SYSCALL_BIT)),
"x32 alias for {name} (nr {nr}) missing; __X32_SYSCALL_BIT bypass possible"
);
}
for (label, nr) in [
("clone", libc::SYS_clone),
("socket", libc::SYS_socket),
("ioctl", libc::SYS_ioctl),
] {
assert!(
rules.contains_key(&(nr | X32_SYSCALL_BIT)),
"x32 alias for filtered {label} (nr {nr}) missing; bypass possible"
);
}
}
#[test]
fn permissive_profile_keeps_network_sockets() {
let profile = SeccompProfile::permissive();
let rules = primary_rules(&profile).unwrap();
assert!(!rules.contains_key(&libc::SYS_socket), "allow_network_sockets must not block socket()");
assert!(rules.contains_key(&libc::SYS_clone));
assert!(rules.contains_key(&libc::SYS_ioctl));
}
}