use std::collections::BTreeMap;
use std::path::Component;
use std::path::Path;
use std::path::PathBuf;
use chrono::DateTime;
use chrono::Utc;
use detcore_model::config::MountInfoRootRewrite;
use detcore_model::procfs::MOUNT_PEER_PREFIXES;
use detcore_model::procfs::MountInfoRow;
use detcore_model::procfs::mount_ids_are_ordered_subset;
pub(crate) use detcore_model::procfs::parse_mountinfo;
use serde::Deserialize;
use serde::Serialize;
use crate::syscalls::DETERMINISTIC_PIPE_CAPACITY_BYTES;
#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
enum ProcfsKind {
Stat,
Status,
ThreadStat,
ThreadStatus,
ProcessStat,
Statm,
ProcessStatus,
TimerSlack(Option<i32>),
SystemStat,
Cpuinfo,
Diskstats,
Loadavg,
ProcessIo,
Uptime,
Meminfo,
BlockStat,
NodeMeminfo,
NodeNumastat,
HwmonInput,
ScalingCurFreq,
Sockstat,
PtyNr,
SelfSched,
Fdinfo,
AioNr,
AioMaxNr,
PipeMaxSize,
NumaMaps,
SmapsRollup,
ArchStatus,
CpuidleCounter,
Smaps,
Maps,
KeyUsers,
Pressure,
Buddyinfo,
Schedstat,
SelfSchedstat,
SoftnetStat,
FileNr,
FileMax,
Zoneinfo,
InodeNr,
InodeState,
Protocols,
BtrfsBytesReserved,
BtrfsBytesPinned,
Rtc,
DentryState,
Mountinfo,
RandomUuid,
Swaps,
Locks,
ThpCounter,
NodeVmstat,
CppcFeedback,
UnixSockets,
InetSockets,
BtrfsCommitStats,
SysfsRtcDate,
SysfsRtcTime,
SysfsRtcEpoch,
NetlinkSockets,
IrqPerCpuCount,
InterruptCounters,
Modules,
ModuleRefcnt(String),
UeventSeqnum,
BtrfsBytesMayUse,
BlockInflight,
Vmstat,
}
fn is_cppc_feedback_path(path: &Path) -> bool {
let mut components = path.iter().rev();
let Some("feedback_ctrs") = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some("acpi_cppc") = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(cpu) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(cpu_number) = cpu.strip_prefix("cpu") else {
return false;
};
!cpu_number.is_empty() && cpu_number.bytes().all(|byte| byte.is_ascii_digit())
}
const THP_COUNTERS: &[&str] = &[
"anon_fault_alloc",
"anon_fault_fallback",
"anon_fault_fallback_charge",
"nr_anon",
"nr_anon_partially_mapped",
"shmem_alloc",
"shmem_fallback",
"shmem_fallback_charge",
"split",
"split_deferred",
"split_failed",
"swpin",
"swpin_fallback",
"swpin_fallback_charge",
"swpout",
"swpout_fallback",
"zswpout",
];
fn is_thp_counter_path(path: &Path) -> bool {
let mut components = path.iter().rev();
let Some(counter) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(stats) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(size_dir) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(size_kb) = size_dir
.strip_prefix("hugepages-")
.and_then(|value| value.strip_suffix("kB"))
else {
return false;
};
stats == "stats"
&& !size_kb.is_empty()
&& size_kb.bytes().all(|byte| byte.is_ascii_digit())
&& THP_COUNTERS.contains(&counter)
}
fn is_btrfs_bytes_reserved_path(path: &Path) -> bool {
is_btrfs_allocation_gauge_path(path, "bytes_reserved")
}
fn is_btrfs_bytes_pinned_path(path: &Path) -> bool {
is_btrfs_allocation_gauge_path(path, "bytes_pinned")
}
fn is_btrfs_allocation_gauge_path(path: &Path, gauge: &str) -> bool {
let Ok(relative) = path.strip_prefix("/sys/fs/btrfs") else {
return false;
};
let mut components = relative.iter();
let (Some(uuid), Some("allocation"), Some(class), Some(candidate_gauge), None) = (
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next(),
) else {
return false;
};
candidate_gauge == gauge
&& is_btrfs_uuid(uuid)
&& matches!(class, "data" | "metadata" | "system")
}
fn is_btrfs_bytes_may_use_path(path: &Path) -> bool {
let Ok(relative) = path.strip_prefix("/sys/fs/btrfs") else {
return false;
};
let mut components = relative.iter();
let (Some(uuid), Some("allocation"), Some(class), Some("bytes_may_use"), None) = (
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next().and_then(|part| part.to_str()),
components.next(),
) else {
return false;
};
is_btrfs_uuid(uuid) && matches!(class, "data" | "metadata" | "system")
}
fn is_btrfs_uuid(value: &str) -> bool {
value.len() == 36
&& value.bytes().enumerate().all(|(index, byte)| {
if matches!(index, 8 | 13 | 18 | 23) {
byte == b'-'
} else {
byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte)
}
})
}
fn is_cpuidle_counter_path(path: &Path) -> bool {
let mut components = path.iter().rev();
let Some(counter) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(state) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(cpuidle) = components.next().and_then(|part| part.to_str()) else {
return false;
};
let Some(state_index) = state.strip_prefix("state") else {
return false;
};
cpuidle == "cpuidle"
&& !state_index.is_empty()
&& state_index.bytes().all(|byte| byte.is_ascii_digit())
&& matches!(counter, "time" | "usage" | "above" | "below" | "rejected")
}
fn module_refcnt_name(path: &Path) -> Option<String> {
let relative = path.strip_prefix("/sys/module").ok()?;
let mut components = relative.iter();
let module = components.next().and_then(|part| part.to_str())?;
if module.is_empty() || components.next().and_then(|part| part.to_str()) != Some("refcnt") {
return None;
}
components.next().is_none().then(|| module.to_owned())
}
fn read_host_modules() -> String {
std::fs::read_to_string("/proc/modules").unwrap_or_default()
}
fn deterministic_module_holder_count(modules: &str, module: &str) -> u64 {
modules
.lines()
.find_map(|line| {
let mut fields = line.split_whitespace();
(fields.next()? == module).then_some(fields)
})
.and_then(|mut fields| {
fields.next()?; fields.next()?; let holders = fields.next()?;
Some(if holders == "-" {
0
} else {
holders.split(',').filter(|h| !h.is_empty()).count() as u64
})
})
.unwrap_or(0)
}
fn sysfs_rtc_kind(path: &Path) -> Option<ProcfsKind> {
let relative = path.strip_prefix("/sys/class/rtc").ok()?;
let mut components = relative.iter();
let rtc = components.next()?.to_str()?;
let leaf = components.next()?.to_str()?;
if components.next().is_some() {
return None;
}
let rtc_index = rtc.strip_prefix("rtc")?;
if rtc_index.is_empty() || !rtc_index.bytes().all(|byte| byte.is_ascii_digit()) {
return None;
}
match leaf {
"date" => Some(ProcfsKind::SysfsRtcDate),
"time" => Some(ProcfsKind::SysfsRtcTime),
"since_epoch" => Some(ProcfsKind::SysfsRtcEpoch),
_ => None,
}
}
#[derive(Clone, Debug, Eq, PartialEq, Serialize, Deserialize)]
pub(crate) struct ProcfsFile {
kind: ProcfsKind,
target_fd: Option<i32>,
bound_thread_identity: Option<(i32, i32, i32)>,
#[serde(default)]
timer_slack_identity: Option<(u64, u64)>,
contents: Option<Vec<u8>>,
offset: usize,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub(crate) struct TimerSlackReadPreview {
pub(crate) bytes: Vec<u8>,
snapshot: Vec<u8>,
offset: usize,
}
#[derive(Clone, Debug, Eq, PartialEq)]
pub(crate) struct MountInfoSnapshot {
pub(crate) rows: Vec<MountInfoRow>,
pub(crate) mount_ids: BTreeMap<u64, u64>,
pub(crate) devices: BTreeMap<u64, u64>,
pub(crate) peer_groups: BTreeMap<u64, u64>,
pub(crate) root_rewrites: BTreeMap<u64, Vec<u8>>,
pub(crate) root_prefix_rewrites: Vec<(Vec<u8>, Vec<u8>)>,
pub(crate) mountpoint_prefix_rewrites: Vec<(Vec<u8>, Vec<u8>)>,
}
impl MountInfoSnapshot {
pub(crate) fn new(
rows: Vec<MountInfoRow>,
mount_id_order: &[u64],
virtualize_metadata: bool,
mut devices: BTreeMap<u64, u64>,
rewrites: BTreeMap<u64, MountInfoRootRewrite>,
) -> Option<Self> {
let visible_mount_order = rows.iter().map(|row| row.raw_mount_id).collect::<Vec<_>>();
let visible_mounts = visible_mount_order
.iter()
.copied()
.collect::<std::collections::BTreeSet<_>>();
let supplied_mounts = mount_id_order
.iter()
.copied()
.collect::<std::collections::BTreeSet<_>>();
let rewrites_are_known = if mount_id_order.is_empty() {
rewrites
.keys()
.all(|mount_id| visible_mounts.contains(mount_id))
} else {
rewrites
.keys()
.all(|mount_id| supplied_mounts.contains(mount_id))
};
if visible_mounts.len() != rows.len()
|| !rewrites_are_known
|| rewrites.values().any(|rewrite| {
rewrite.deterministic_root.first() != Some(&b'/')
|| rewrite
.deterministic_root
.iter()
.any(|byte| byte.is_ascii_control() || *byte == b' ')
|| match (&rewrite.raw_root_prefix, &rewrite.deterministic_root_prefix) {
(None, None) => false,
(Some(raw), Some(deterministic)) => {
raw.first() != Some(&b'/')
|| deterministic.first() != Some(&b'/')
|| raw
.iter()
.any(|byte| *byte == b' ' || byte.is_ascii_control())
|| deterministic
.iter()
.any(|byte| *byte == b' ' || byte.is_ascii_control())
}
_ => true,
}
|| match (
&rewrite.raw_mountpoint_prefix,
&rewrite.deterministic_mountpoint_prefix,
) {
(None, None) => false,
(Some(raw), Some(deterministic)) => {
raw.first() != Some(&b'/')
|| deterministic.first() != Some(&b'/')
|| raw
.iter()
.any(|byte| *byte == b' ' || byte.is_ascii_control())
|| deterministic
.iter()
.any(|byte| *byte == b' ' || byte.is_ascii_control())
}
_ => true,
}
})
{
return None;
}
let raw_devices = rows
.iter()
.map(|row| row.raw_device)
.collect::<std::collections::BTreeSet<_>>();
if virtualize_metadata {
if devices.len() != raw_devices.len()
|| !raw_devices.iter().all(|raw| devices.contains_key(raw))
{
return None;
}
} else {
if !devices.is_empty() {
return None;
}
devices = raw_devices.into_iter().map(|raw| (raw, raw)).collect();
}
let expected_mount_ids = rows
.iter()
.flat_map(|row| [row.raw_mount_id, row.raw_parent_id])
.collect::<std::collections::BTreeSet<_>>();
let mut derived_order = Vec::with_capacity(expected_mount_ids.len());
let mut seen = std::collections::BTreeSet::new();
for row in &rows {
if seen.insert(row.raw_mount_id) {
derived_order.push(row.raw_mount_id);
}
}
for row in &rows {
if seen.insert(row.raw_parent_id) {
derived_order.push(row.raw_parent_id);
}
}
let mount_ids = if mount_id_order.is_empty() {
derived_order
.into_iter()
.enumerate()
.map(|(index, raw)| (raw, index as u64 + 1))
.collect()
} else {
if supplied_mounts.len() != mount_id_order.len()
|| !mount_ids_are_ordered_subset(&visible_mount_order, mount_id_order)
|| !expected_mount_ids
.iter()
.all(|raw| supplied_mounts.contains(raw))
{
return None;
}
mount_id_order
.iter()
.copied()
.enumerate()
.filter(|(_, raw)| expected_mount_ids.contains(raw))
.map(|(index, raw)| (raw, index as u64 + 1))
.collect()
};
let mut peer_groups = BTreeMap::new();
for row in &rows {
for &raw in &row.raw_peer_groups {
let next = peer_groups.len() as u64 + 1;
peer_groups.entry(raw).or_insert(next);
}
}
let root_rewrites = rewrites
.iter()
.filter(|(raw_mount_id, _)| visible_mounts.contains(raw_mount_id))
.map(|(raw_mount_id, rewrite)| (*raw_mount_id, rewrite.deterministic_root.clone()))
.collect();
let mut root_prefix_rewrites = rewrites
.values()
.cloned()
.filter_map(|rewrite| {
rewrite
.raw_root_prefix
.zip(rewrite.deterministic_root_prefix)
})
.collect::<Vec<_>>();
root_prefix_rewrites.sort_by_key(|(raw, _)| std::cmp::Reverse(raw.len()));
let mut mountpoint_prefix_rewrites = rewrites
.into_values()
.filter_map(|rewrite| {
rewrite
.raw_mountpoint_prefix
.zip(rewrite.deterministic_mountpoint_prefix)
})
.collect::<Vec<_>>();
mountpoint_prefix_rewrites.sort_by_key(|(raw, _)| std::cmp::Reverse(raw.len()));
Some(Self {
rows,
mount_ids,
devices,
peer_groups,
root_rewrites,
root_prefix_rewrites,
mountpoint_prefix_rewrites,
})
}
#[cfg(test)]
pub(crate) fn canonical_mount_id(&self, raw_mount_id: u64) -> Option<u64> {
self.mount_ids.get(&raw_mount_id).copied()
}
pub(crate) fn raw_mount_id_order(&self) -> Vec<u64> {
let mut ordered = self
.mount_ids
.iter()
.map(|(raw, canonical)| (*canonical, *raw))
.collect::<Vec<_>>();
ordered.sort_unstable();
ordered.into_iter().map(|(_, raw)| raw).collect()
}
}
#[derive(Clone, Debug, Default, Eq, PartialEq)]
pub(crate) struct ProcfsSnapshotContext {
pub(crate) virtual_uptime_seconds: u64,
pub(crate) virtual_boot_time_seconds: Option<i64>,
pub(crate) virtual_realtime_seconds: i64,
pub(crate) virtual_memory_kb: u64,
pub(crate) virtual_pid: i32,
pub(crate) virtual_ppid: i32,
pub(crate) virtual_pty_count: usize,
pub(crate) fdinfo_identity: Option<(u64, i32, u64, u64)>,
pub(crate) mapping_identities: BTreeMap<(u64, u64), (u64, u64)>,
pub(crate) mountinfo: Option<MountInfoSnapshot>,
pub(crate) random_uuid: Option<[u8; 16]>,
}
impl ProcfsFile {
pub(crate) fn from_path(path: &Path) -> Option<Self> {
let path = normalize_observed_path(path)?;
let path_text = path.to_str()?;
let target_fd = parse_fdinfo_target(path_text);
let kind = match path_text {
"/proc/self/stat" => ProcfsKind::Stat,
"/proc/self/status" => ProcfsKind::Status,
"/proc/thread-self/stat" => ProcfsKind::ThreadStat,
"/proc/thread-self/status" => ProcfsKind::ThreadStatus,
"/proc/self/statm" | "/proc/thread-self/statm" => ProcfsKind::Statm,
other if is_process_file_path(other, "stat") => ProcfsKind::ProcessStat,
other if is_process_file_path(other, "statm") => ProcfsKind::Statm,
other if is_process_file_path(other, "status") => ProcfsKind::ProcessStatus,
other if let Some(target) = parse_timer_slack_target(other) => {
ProcfsKind::TimerSlack(target)
}
"/proc/cpuinfo" => ProcfsKind::Cpuinfo,
"/proc/diskstats" => ProcfsKind::Diskstats,
"/proc/loadavg" => ProcfsKind::Loadavg,
"/proc/uptime" => ProcfsKind::Uptime,
"/proc/stat" => ProcfsKind::SystemStat,
"/proc/meminfo" => ProcfsKind::Meminfo,
"/proc/sys/fs/inode-nr" => ProcfsKind::InodeNr,
"/proc/sys/fs/inode-state" => ProcfsKind::InodeState,
"/proc/sys/fs/dentry-state" => ProcfsKind::DentryState,
other if is_process_file_path(other, "mountinfo") => ProcfsKind::Mountinfo,
"/proc/sys/kernel/random/uuid" => ProcfsKind::RandomUuid,
"/proc/sys/fs/aio-nr" => ProcfsKind::AioNr,
"/proc/sys/fs/aio-max-nr" => ProcfsKind::AioMaxNr,
"/proc/sys/fs/pipe-max-size" => ProcfsKind::PipeMaxSize,
"/proc/sys/kernel/pty/nr" => ProcfsKind::PtyNr,
"/proc/net/sockstat" => ProcfsKind::Sockstat,
"/proc/vmstat" => ProcfsKind::Vmstat,
"/sys/kernel/uevent_seqnum" => ProcfsKind::UeventSeqnum,
other if is_node_vmstat_path(other) => ProcfsKind::NodeVmstat,
other if is_process_file_path(other, "sched") => ProcfsKind::SelfSched,
_ if target_fd.is_some() => ProcfsKind::Fdinfo,
other if is_process_file_path(other, "numa_maps") => ProcfsKind::NumaMaps,
other if is_process_file_path(other, "smaps_rollup") => ProcfsKind::SmapsRollup,
other if is_process_file_path(other, "arch_status") => ProcfsKind::ArchStatus,
"/proc/swaps" => ProcfsKind::Swaps,
other if is_process_file_path(other, "smaps") => ProcfsKind::Smaps,
other if is_process_file_path(other, "maps") => ProcfsKind::Maps,
"/proc/key-users" => ProcfsKind::KeyUsers,
"/proc/pressure/cpu" | "/proc/pressure/io" | "/proc/pressure/memory" => {
ProcfsKind::Pressure
}
"/proc/buddyinfo" => ProcfsKind::Buddyinfo,
"/proc/schedstat" => ProcfsKind::Schedstat,
other if is_process_schedstat_path(other) => ProcfsKind::SelfSchedstat,
"/proc/net/softnet_stat" => ProcfsKind::SoftnetStat,
"/proc/sys/fs/file-nr" => ProcfsKind::FileNr,
"/proc/sys/fs/file-max" => ProcfsKind::FileMax,
"/proc/zoneinfo" => ProcfsKind::Zoneinfo,
other if is_numa_node_file(other, "meminfo") => ProcfsKind::NodeMeminfo,
other if is_numa_node_file(other, "numastat") => ProcfsKind::NodeNumastat,
other if is_hwmon_input_file(other) => ProcfsKind::HwmonInput,
"/proc/net/protocols" => ProcfsKind::Protocols,
"/proc/locks" => ProcfsKind::Locks,
other if is_btrfs_bytes_reserved_path(Path::new(other)) => {
ProcfsKind::BtrfsBytesReserved
}
"/proc/driver/rtc" => ProcfsKind::Rtc,
"/proc/net/netlink" => ProcfsKind::NetlinkSockets,
_ if is_btrfs_commit_stats_path(&path) => ProcfsKind::BtrfsCommitStats,
"/proc/net/unix" => ProcfsKind::UnixSockets,
"/proc/net/tcp" | "/proc/net/tcp6" | "/proc/net/udp" | "/proc/net/udp6" => {
ProcfsKind::InetSockets
}
"/proc/interrupts" | "/proc/softirqs" => ProcfsKind::InterruptCounters,
"/proc/modules" => ProcfsKind::Modules,
other if let Some(module) = module_refcnt_name(Path::new(other)) => {
ProcfsKind::ModuleRefcnt(module)
}
other if is_cpufreq_policy_value_path(Path::new(other)) => ProcfsKind::ScalingCurFreq,
other if is_cppc_feedback_path(Path::new(other)) => ProcfsKind::CppcFeedback,
other if is_process_io_path(other) => ProcfsKind::ProcessIo,
other if is_block_stat_path(other) => ProcfsKind::BlockStat,
other if is_btrfs_bytes_pinned_path(Path::new(other)) => ProcfsKind::BtrfsBytesPinned,
other if is_cpuidle_counter_path(Path::new(other)) => ProcfsKind::CpuidleCounter,
other if is_thp_counter_path(Path::new(other)) => ProcfsKind::ThpCounter,
_ if is_block_inflight_path(&path) => ProcfsKind::BlockInflight,
_ if is_btrfs_bytes_may_use_path(&path) => ProcfsKind::BtrfsBytesMayUse,
_ if is_irq_per_cpu_count_path(&path) => ProcfsKind::IrqPerCpuCount,
_ => sysfs_rtc_kind(&path)?,
};
Some(Self {
kind,
target_fd,
bound_thread_identity: None,
timer_slack_identity: None,
contents: None,
offset: 0,
})
}
pub(crate) fn needs_mapping_identities(&self) -> bool {
matches!(self.kind, ProcfsKind::Maps | ProcfsKind::Smaps)
}
pub(crate) fn needs_mountinfo_identities(&self) -> bool {
self.kind == ProcfsKind::Mountinfo
}
pub(crate) fn needs_snapshot(&self) -> bool {
!matches!(self.kind, ProcfsKind::TimerSlack(_)) && self.contents.is_none()
}
pub(crate) fn needs_bound_thread_identity(&self) -> bool {
matches!(
self.kind,
ProcfsKind::ThreadStat | ProcfsKind::ThreadStatus | ProcfsKind::TimerSlack(None)
)
}
pub(crate) fn bind_thread_identity(&mut self, tgid: i32, tid: i32, ppid: i32) {
assert!(
self.needs_bound_thread_identity(),
"only thread-self procfs files bind an opener identity"
);
match &mut self.kind {
ProcfsKind::TimerSlack(target @ None) => *target = Some(tgid),
ProcfsKind::ThreadStat | ProcfsKind::ThreadStatus => {
self.bound_thread_identity = Some((tgid, tid, ppid));
}
_ => unreachable!("non-binding procfs kind requested an opener identity"),
}
}
pub(crate) fn needs_random_uuid(&self) -> bool {
self.kind == ProcfsKind::RandomUuid
}
pub(crate) fn needs_boot_time(&self) -> bool {
self.kind == ProcfsKind::SystemStat
}
pub(crate) fn initialize(&mut self, contents: Vec<u8>, context: ProcfsSnapshotContext) {
let ProcfsSnapshotContext {
virtual_uptime_seconds,
virtual_boot_time_seconds,
virtual_realtime_seconds,
virtual_memory_kb,
virtual_pid,
virtual_ppid,
virtual_pty_count,
fdinfo_identity,
random_uuid,
mapping_identities,
mountinfo,
} = context;
let mapping_identities = &mapping_identities;
self.contents = Some(match &self.kind {
ProcfsKind::Stat => sanitize_stat(&contents, Some((virtual_pid, virtual_ppid))),
ProcfsKind::Status => {
sanitize_status(&contents, Some((virtual_pid, virtual_pid, virtual_ppid)))
}
ProcfsKind::ThreadStat => {
let (_, tid, ppid) = self
.bound_thread_identity
.expect("thread-self stat was not bound when opened");
sanitize_stat(&contents, Some((tid, ppid)))
}
ProcfsKind::ThreadStatus => {
let (tgid, tid, ppid) = self
.bound_thread_identity
.expect("thread-self status was not bound when opened");
sanitize_status(&contents, Some((tgid, tid, ppid)))
}
ProcfsKind::ProcessStat => sanitize_stat(&contents, None),
ProcfsKind::Statm => sanitize_statm(&contents),
ProcfsKind::ProcessStatus => sanitize_status(&contents, None),
ProcfsKind::TimerSlack(_) => {
unreachable!("timer-slack procfs content is generated from ThreadState")
}
ProcfsKind::SystemStat => sanitize_system_stat(
&contents,
virtual_uptime_seconds,
virtual_boot_time_seconds.expect("system stat snapshot omitted its boot time"),
),
ProcfsKind::Cpuinfo => sanitize_cpuinfo(&contents),
ProcfsKind::Diskstats => sanitize_diskstats(&contents),
ProcfsKind::Loadavg => sanitize_loadavg(&contents),
ProcfsKind::ProcessIo => sanitize_process_io(&contents),
ProcfsKind::Uptime => sanitize_uptime(&contents, virtual_uptime_seconds),
ProcfsKind::Meminfo => sanitize_meminfo(&contents, virtual_memory_kb),
ProcfsKind::BlockStat => sanitize_block_stat(&contents),
ProcfsKind::NodeMeminfo => sanitize_node_meminfo(&contents),
ProcfsKind::NodeNumastat => sanitize_node_numastat(&contents),
ProcfsKind::HwmonInput => sanitize_numeric_scalar(&contents),
ProcfsKind::ScalingCurFreq => sanitize_scaling_cur_freq(&contents),
ProcfsKind::Sockstat => sanitize_sockstat(&contents),
ProcfsKind::Vmstat => sanitize_vmstat(&contents),
ProcfsKind::UeventSeqnum => sanitize_uevent_seqnum(&contents),
ProcfsKind::BtrfsBytesMayUse => sanitize_btrfs_bytes_may_use(&contents),
ProcfsKind::BlockInflight => sanitize_block_inflight(&contents),
ProcfsKind::IrqPerCpuCount => sanitize_irq_per_cpu_count(&contents),
ProcfsKind::PtyNr => sanitize_pty_nr(&contents, virtual_pty_count),
ProcfsKind::SelfSched => sanitize_self_sched(&contents),
ProcfsKind::Fdinfo => sanitize_fdinfo(&contents, fdinfo_identity),
ProcfsKind::AioNr => sanitize_aio_nr(&contents),
ProcfsKind::AioMaxNr => sanitize_aio_nr(&contents),
ProcfsKind::PipeMaxSize => sanitize_pipe_max_size(&contents),
ProcfsKind::NumaMaps => sanitize_numa_maps(&contents),
ProcfsKind::SmapsRollup => sanitize_smaps_rollup(&contents),
ProcfsKind::ArchStatus => sanitize_arch_status(&contents),
ProcfsKind::Swaps => sanitize_swaps(&contents),
ProcfsKind::CpuidleCounter => sanitize_cpuidle_counter(&contents),
ProcfsKind::Smaps => sanitize_smaps(&contents, mapping_identities),
ProcfsKind::Maps => sanitize_maps(&contents, mapping_identities),
ProcfsKind::KeyUsers => sanitize_key_users(&contents),
ProcfsKind::Pressure => sanitize_pressure(&contents),
ProcfsKind::Buddyinfo => sanitize_buddyinfo(&contents),
ProcfsKind::Schedstat => sanitize_schedstat(&contents),
ProcfsKind::SelfSchedstat => sanitize_self_schedstat(&contents),
ProcfsKind::SoftnetStat => sanitize_softnet_stat(&contents),
ProcfsKind::FileNr => sanitize_file_nr(&contents),
ProcfsKind::FileMax => sanitize_file_max(&contents),
ProcfsKind::Zoneinfo => sanitize_zoneinfo(&contents),
ProcfsKind::InodeNr => sanitize_inode_nr(&contents),
ProcfsKind::InodeState => sanitize_inode_state(&contents),
ProcfsKind::Protocols => sanitize_protocols(&contents),
ProcfsKind::BtrfsBytesReserved => sanitize_btrfs_bytes_reserved(&contents),
ProcfsKind::BtrfsBytesPinned => sanitize_btrfs_bytes_pinned(&contents),
ProcfsKind::Rtc => sanitize_rtc(&contents, virtual_realtime_seconds),
ProcfsKind::DentryState => sanitize_dentry_state(&contents),
ProcfsKind::NetlinkSockets => sanitize_netlink_sockets(&contents),
ProcfsKind::Locks => sanitize_locks(&contents),
ProcfsKind::NodeVmstat => sanitize_node_vmstat(&contents),
ProcfsKind::CppcFeedback => sanitize_cppc_feedback(&contents),
ProcfsKind::UnixSockets => sanitize_unix_sockets(&contents),
ProcfsKind::InetSockets => sanitize_inet_sockets(&contents),
ProcfsKind::BtrfsCommitStats => sanitize_btrfs_commit_stats(&contents),
ProcfsKind::SysfsRtcDate | ProcfsKind::SysfsRtcTime | ProcfsKind::SysfsRtcEpoch => {
sanitize_sysfs_rtc_attribute(&contents, self.kind.clone(), virtual_realtime_seconds)
}
ProcfsKind::ThpCounter => sanitize_thp_counter(&contents),
ProcfsKind::InterruptCounters => sanitize_interrupt_counters(&contents),
ProcfsKind::Modules => sanitize_modules(&contents),
ProcfsKind::ModuleRefcnt(module) => {
sanitize_module_refcnt(&contents, module.as_str(), &read_host_modules())
}
ProcfsKind::Mountinfo => sanitize_mountinfo(
&contents,
mountinfo
.as_ref()
.expect("mountinfo identities were not prepared"),
),
ProcfsKind::RandomUuid => sanitize_random_uuid(
&contents,
random_uuid.expect("random UUID snapshot omitted deterministic bytes"),
),
});
}
pub(crate) fn take(&mut self, maximum: usize) -> Option<Vec<u8>> {
let bytes = self.take_at(self.offset, maximum)?;
self.offset = self.offset.saturating_add(bytes.len());
Some(bytes)
}
pub(crate) fn timer_slack_target(&self) -> Option<i32> {
match self.kind {
ProcfsKind::TimerSlack(Some(target)) => Some(target),
ProcfsKind::TimerSlack(None) => {
unreachable!("/proc/self/timerslack_ns was not bound at open time")
}
_ => None,
}
}
pub(crate) fn bind_timer_slack_identity(&mut self, device: u64, inode: u64) {
self.timer_slack_target()
.expect("only timer-slack procfs files have task identities");
self.timer_slack_identity = Some((device, inode));
}
pub(crate) fn timer_slack_binding(&self) -> Option<(i32, u64, u64)> {
let target = self.timer_slack_target()?;
let (device, inode) = self
.timer_slack_identity
.expect("timer-slack procfs file lacks open-time inode identity");
Some((target, device, inode))
}
pub(crate) fn preview_timer_slack(
&self,
value: u64,
maximum: usize,
) -> Option<TimerSlackReadPreview> {
self.timer_slack_target()?;
let snapshot = self
.contents
.clone()
.unwrap_or_else(|| format!("{value}\n").into_bytes());
let start = self.offset.min(snapshot.len());
let end = start.saturating_add(maximum).min(snapshot.len());
Some(TimerSlackReadPreview {
bytes: snapshot[start..end].to_vec(),
snapshot,
offset: self.offset,
})
}
pub(crate) fn commit_timer_slack_read(
&mut self,
preview: &TimerSlackReadPreview,
copied: usize,
) {
assert!(copied <= preview.bytes.len());
if copied == 0 {
return;
}
assert_eq!(
self.offset, preview.offset,
"timer-slack cursor changed between preview and commit"
);
match &self.contents {
Some(contents) => assert_eq!(
contents, &preview.snapshot,
"timer-slack snapshot changed between preview and commit"
),
None => self.contents = Some(preview.snapshot.clone()),
}
self.offset = self.offset.saturating_add(copied);
}
pub(crate) fn take_timer_slack_at(
&self,
value: u64,
offset: usize,
maximum: usize,
) -> Option<Vec<u8>> {
self.timer_slack_target()?;
let contents = format!("{value}\n").into_bytes();
let start = offset.min(contents.len());
let end = start.saturating_add(maximum).min(contents.len());
Some(contents[start..end].to_vec())
}
pub(crate) fn take_at(&self, offset: usize, maximum: usize) -> Option<Vec<u8>> {
let contents = self.contents.as_ref()?;
let start = offset.min(contents.len());
let end = start.saturating_add(maximum).min(contents.len());
Some(contents[start..end].to_vec())
}
pub(crate) fn position(&self) -> (usize, Option<usize>) {
(self.offset, self.contents.as_ref().map(Vec::len))
}
pub(crate) fn set_offset(&mut self, offset: usize) {
self.offset = offset;
if matches!(self.kind, ProcfsKind::TimerSlack(_)) && offset == 0 {
self.contents = None;
}
}
pub(crate) fn target_fd(&self) -> Option<i32> {
self.target_fd
}
}
fn normalize_observed_path(path: &Path) -> Option<PathBuf> {
let mut normalized = PathBuf::new();
for component in path.components() {
match component {
Component::Prefix(_) => return None,
Component::RootDir => normalized.push(Path::new("/")),
Component::CurDir => {}
Component::ParentDir => {
if !normalized.pop() {
return None;
}
}
Component::Normal(part) => normalized.push(part),
}
}
Some(normalized)
}
fn is_process_file_path(path: &str, filename: &str) -> bool {
let Some(relative) = path.strip_prefix("/proc/") else {
return false;
};
let components = relative.split('/').collect::<Vec<_>>();
match components.as_slice() {
[task, candidate] => is_proc_task_name(task) && *candidate == filename,
[process, "task", thread, candidate] => {
is_proc_process_name(process) && is_numeric_id(thread) && *candidate == filename
}
_ => false,
}
}
fn parse_timer_slack_target(path: &str) -> Option<Option<i32>> {
let relative = path.strip_prefix("/proc/")?;
let components = relative.split('/').collect::<Vec<_>>();
let [task, "timerslack_ns"] = components.as_slice() else {
return None;
};
if *task == "self" {
return Some(None);
}
let target = task.parse::<i32>().ok().filter(|target| *target > 0)?;
Some(Some(target))
}
fn parse_fdinfo_target(path: &str) -> Option<i32> {
let relative = path.strip_prefix("/proc/")?;
let components = relative.split('/').collect::<Vec<_>>();
let fd = match components.as_slice() {
[task, "fdinfo", fd] if is_proc_task_name(task) => *fd,
[process, "task", thread, "fdinfo", fd]
if is_proc_process_name(process) && is_numeric_id(thread) =>
{
*fd
}
_ => return None,
};
fd.parse().ok()
}
const VIRTUAL_CPU_FREQUENCY_KHZ: u64 = 1_000_000;
fn is_cpufreq_policy_value_path(path: &Path) -> bool {
let Ok(relative) = path.strip_prefix("/sys/devices/system/cpu") else {
return false;
};
let components = relative
.iter()
.filter_map(|component| component.to_str())
.collect::<Vec<_>>();
let attribute = match components.as_slice() {
[cpu, "cpufreq", attribute]
if cpu.strip_prefix("cpu").is_some_and(|id| {
!id.is_empty() && id.bytes().all(|byte| byte.is_ascii_digit())
}) =>
{
*attribute
}
["cpufreq", policy, attribute]
if policy.strip_prefix("policy").is_some_and(|id| {
!id.is_empty() && id.bytes().all(|byte| byte.is_ascii_digit())
}) =>
{
*attribute
}
_ => return false,
};
matches!(
attribute,
"scaling_cur_freq"
| "cpuinfo_cur_freq"
| "cpuinfo_avg_freq"
| "cpuinfo_min_freq"
| "cpuinfo_max_freq"
| "scaling_min_freq"
| "scaling_max_freq"
)
}
fn is_process_io_path(path: &str) -> bool {
path == "/proc/self/io"
|| path
.strip_prefix("/proc/")
.and_then(|path| path.strip_suffix("/io"))
.is_some_and(|pid| !pid.is_empty() && pid.bytes().all(|byte| byte.is_ascii_digit()))
}
fn is_process_schedstat_path(path: &str) -> bool {
let Some(relative) = path.strip_prefix("/proc/") else {
return false;
};
let components = relative.split('/').collect::<Vec<_>>();
match components.as_slice() {
[task, "schedstat"] => is_proc_task_name(task),
[process, "task", thread, "schedstat"] => {
is_proc_process_name(process) && is_numeric_id(thread)
}
_ => false,
}
}
fn is_proc_task_name(name: &str) -> bool {
matches!(name, "self" | "thread-self") || is_numeric_id(name)
}
fn is_proc_process_name(name: &str) -> bool {
name == "self" || is_numeric_id(name)
}
fn is_numeric_id(value: &str) -> bool {
!value.is_empty() && value.bytes().all(|byte| byte.is_ascii_digit())
}
fn is_block_stat_path(path: &str) -> bool {
path.strip_prefix("/sys/block/")
.and_then(|path| path.strip_suffix("/stat"))
.is_some_and(|device| !device.is_empty() && !device.contains('/'))
}
fn is_numa_node_file(path: &str, filename: &str) -> bool {
path.strip_prefix("/sys/devices/system/node/node")
.and_then(|path| path.split_once('/'))
.is_some_and(|(node, leaf)| {
!node.is_empty() && node.bytes().all(|byte| byte.is_ascii_digit()) && leaf == filename
})
}
fn is_hwmon_input_file(path: &str) -> bool {
path.strip_prefix("/sys/class/hwmon/hwmon")
.and_then(|path| path.split_once('/'))
.is_some_and(|(instance, attribute)| {
!instance.is_empty()
&& instance.bytes().all(|byte| byte.is_ascii_digit())
&& !attribute.contains('/')
&& attribute.ends_with("_input")
})
}
fn sanitize_stat(contents: &[u8], virtual_identity: Option<(i32, i32)>) -> Vec<u8> {
const VOLATILE_FIELDS: &[usize] = &[
10, 11, 12, 13, 14, 15, 16, 17, 21, 22, 23, 24, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35, 36,
37, 39, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51,
];
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let Some(comm_start) = text.find(" (") else {
return contents.to_vec();
};
let Some(comm_end) = text.rfind(") ") else {
return contents.to_vec();
};
let comm = &text[comm_start..=comm_end];
let mut fields = text[comm_end + 2..]
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
if fields.len() < 50 {
return contents.to_vec();
}
let pid = if let Some((virtual_pid, virtual_ppid)) = virtual_identity {
fields[4 - 3] = virtual_ppid.to_string();
fields[5 - 3] = "0".to_owned();
fields[6 - 3] = "0".to_owned();
virtual_pid.to_string()
} else {
fields[0] = "S".to_owned();
text[..comm_start].to_owned()
};
for field in VOLATILE_FIELDS {
fields[*field - 3] = "0".to_owned();
}
format!("{pid}{comm} {}\n", fields.join(" ")).into_bytes()
}
fn sanitize_statm(contents: &[u8]) -> Vec<u8> {
let fields = contents
.split(|byte| byte.is_ascii_whitespace())
.filter(|field| !field.is_empty())
.collect::<Vec<_>>();
if fields.len() != 7
|| fields
.iter()
.any(|field| !field.iter().all(u8::is_ascii_digit))
{
return contents.to_vec();
}
b"0 0 0 0 0 0 0\n".to_vec()
}
fn sanitize_status(contents: &[u8], virtual_identity: Option<(i32, i32, i32)>) -> Vec<u8> {
const STATE: &[u8] = b"State:";
const TGID: &[u8] = b"Tgid:";
const PID: &[u8] = b"Pid:";
const PPID: &[u8] = b"PPid:";
const TRACER_PID: &[u8] = b"TracerPid:";
const NS_TGID: &[u8] = b"NStgid:";
const NS_PID: &[u8] = b"NSpid:";
const NS_PGID: &[u8] = b"NSpgid:";
const NS_SID: &[u8] = b"NSsid:";
const SIGQ: &[u8] = b"SigQ:";
const CPUS_ALLOWED: &[u8] = b"Cpus_allowed:";
const CPUS_ALLOWED_LIST: &[u8] = b"Cpus_allowed_list:";
const VOLUNTARY: &[u8] = b"voluntary_ctxt_switches:";
const NONVOLUNTARY: &[u8] = b"nonvoluntary_ctxt_switches:";
const MEMORY_FIELDS: &[&[u8]] = &[
b"VmPeak",
b"VmSize",
b"VmLck",
b"VmPin",
b"VmHWM",
b"VmRSS",
b"RssAnon",
b"RssFile",
b"RssShmem",
b"VmData",
b"VmStk",
b"VmExe",
b"VmLib",
b"VmPTE",
b"VmSwap",
b"HugetlbPages",
];
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
if body.starts_with(STATE) {
normalized.extend_from_slice(b"State:\tS (sleeping)");
} else if let Some((virtual_tgid, _, _)) = virtual_identity
&& (body.starts_with(TGID) || body.starts_with(NS_TGID))
{
let label = body.split(|byte| *byte == b':').next().unwrap_or_default();
normalized.extend_from_slice(label);
normalized.extend_from_slice(format!(":\t{virtual_tgid}").as_bytes());
} else if let Some((_, virtual_pid, _)) = virtual_identity
&& (body.starts_with(PID) || body.starts_with(NS_PID))
{
let label = body.split(|byte| *byte == b':').next().unwrap_or_default();
normalized.extend_from_slice(label);
normalized.extend_from_slice(format!(":\t{virtual_pid}").as_bytes());
} else if let Some((_, _, virtual_ppid)) = virtual_identity
&& body.starts_with(PPID)
{
normalized.extend_from_slice(PPID);
normalized.extend_from_slice(format!("\t{virtual_ppid}").as_bytes());
} else if body.starts_with(TRACER_PID) {
normalized.extend_from_slice(TRACER_PID);
normalized.extend_from_slice(if virtual_identity.is_some() {
b"\t1"
} else {
b"\t0"
});
} else if body.starts_with(NS_PGID) || body.starts_with(NS_SID) {
let label = body.split(|byte| *byte == b':').next().unwrap_or_default();
normalized.extend_from_slice(label);
normalized.extend_from_slice(b":\t0");
} else if body.starts_with(SIGQ) {
normalized.extend_from_slice(SIGQ);
normalized.extend_from_slice(b"\t0/0");
} else if body.starts_with(CPUS_ALLOWED) {
normalized.extend_from_slice(CPUS_ALLOWED);
normalized.extend_from_slice(b"\t00000000,00000000,00000000,00000001");
} else if body.starts_with(CPUS_ALLOWED_LIST) {
normalized.extend_from_slice(CPUS_ALLOWED_LIST);
normalized.extend_from_slice(b"\t0");
} else if body.starts_with(VOLUNTARY) {
normalized.extend_from_slice(VOLUNTARY);
normalized.extend_from_slice(b"\t0");
} else if body.starts_with(NONVOLUNTARY) {
normalized.extend_from_slice(NONVOLUNTARY);
normalized.extend_from_slice(b"\t0");
} else if let Some(name_end) = body.iter().position(|byte| *byte == b':')
&& MEMORY_FIELDS.contains(&&body[..name_end])
{
normalized.extend_from_slice(&body[..name_end]);
normalized.extend_from_slice(b":\t0 kB");
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_cpuinfo(contents: &[u8]) -> Vec<u8> {
const CPU_MHZ: &[u8] = b"cpu MHz";
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
if body.starts_with(CPU_MHZ) {
normalized.extend_from_slice(b"cpu MHz\t\t: 1000.000");
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_diskstats(contents: &[u8]) -> Vec<u8> {
sanitize_numeric_lines(contents, 3)
}
fn sanitize_block_stat(contents: &[u8]) -> Vec<u8> {
sanitize_numeric_lines(contents, 0)
}
fn sanitize_numeric_lines(contents: &[u8], stable_fields: usize) -> Vec<u8> {
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let fields = body
.split(|byte| byte.is_ascii_whitespace())
.filter(|field| !field.is_empty())
.collect::<Vec<_>>();
let counters = fields.get(stable_fields..).unwrap_or_default();
if !counters.is_empty()
&& counters
.iter()
.all(|field| field.iter().all(u8::is_ascii_digit))
{
for (index, field) in fields.iter().take(stable_fields).enumerate() {
if index > 0 {
normalized.push(b' ');
}
normalized.extend_from_slice(field);
}
for (index, _) in counters.iter().enumerate() {
if !normalized.is_empty() && normalized.last() != Some(&b'\n') {
normalized.push(b' ');
}
let value = match index {
0 | 4 => 1,
2 | 6 => 8,
_ => 0,
};
normalized.extend_from_slice(value.to_string().as_bytes());
}
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_node_numastat(contents: &[u8]) -> Vec<u8> {
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let fields = body
.split(|byte| byte.is_ascii_whitespace())
.filter(|field| !field.is_empty())
.collect::<Vec<_>>();
if fields.len() == 2 && fields[1].iter().all(u8::is_ascii_digit) {
normalized.extend_from_slice(fields[0]);
normalized.extend_from_slice(b" 0");
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_process_io(contents: &[u8]) -> Vec<u8> {
const COUNTERS: &[&[u8]] = &[
b"rchar",
b"wchar",
b"syscr",
b"syscw",
b"read_bytes",
b"write_bytes",
b"cancelled_write_bytes",
];
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let name_end = body.iter().position(|byte| *byte == b':');
let name = name_end.map_or(body, |end| &body[..end]);
if COUNTERS.contains(&name) {
normalized.extend_from_slice(name);
normalized.extend_from_slice(b": 0");
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_node_meminfo(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = String::with_capacity(text.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let Some((label, value)) = body.split_once(':') else {
normalized.push_str(body);
if has_newline {
normalized.push('\n');
}
continue;
};
let field = label.split_whitespace().last().unwrap_or_default();
let mut value_fields = value.split_whitespace();
let numeric = value_fields.next();
let unit = value_fields.next();
if numeric.is_some_and(|value| value.bytes().all(|byte| byte.is_ascii_digit()))
&& value_fields.next().is_none()
{
let synthetic = match field {
"MemTotal" | "MemFree" => 1_048_576,
_ => 0,
};
normalized.push_str(label);
normalized.push_str(": ");
normalized.push_str(&synthetic.to_string());
if let Some(unit) = unit {
normalized.push(' ');
normalized.push_str(unit);
}
} else {
normalized.push_str(body);
}
if has_newline {
normalized.push('\n');
}
}
normalized.into_bytes()
}
fn sanitize_numeric_scalar(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
if text.trim().parse::<i64>().is_err() {
return contents.to_vec();
}
if text.ends_with('\n') {
b"0\n".to_vec()
} else {
b"0".to_vec()
}
}
fn sanitize_scaling_cur_freq(contents: &[u8]) -> Vec<u8> {
let Ok(value) = std::str::from_utf8(contents) else {
return Vec::new();
};
if value.trim().parse::<u64>().is_err() {
return Vec::new();
}
format!("{VIRTUAL_CPU_FREQUENCY_KHZ}\n").into_bytes()
}
fn sanitize_btrfs_bytes_reserved(contents: &[u8]) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let value = contents.strip_suffix(b"\n").unwrap_or(contents);
let valid_value = std::str::from_utf8(value)
.ok()
.and_then(|value| value.parse::<u64>().ok())
.is_some();
if !valid_value {
return contents.to_vec();
}
if has_newline {
b"0\n".to_vec()
} else {
b"0".to_vec()
}
}
fn sanitize_btrfs_bytes_pinned(contents: &[u8]) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let value = contents.strip_suffix(b"\n").unwrap_or(contents);
if value.is_empty() || !value.iter().all(u8::is_ascii_digit) {
return contents.to_vec();
}
if has_newline {
b"0\n".to_vec()
} else {
b"0".to_vec()
}
}
fn sanitize_cpuidle_counter(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0\n".to_vec()
}
}
fn sanitize_thp_counter(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0\n".to_vec()
}
}
fn sanitize_cppc_feedback(contents: &[u8]) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let body = contents.strip_suffix(b"\n").unwrap_or(contents);
let Ok(text) = std::str::from_utf8(body) else {
return contents.to_vec();
};
let mut fields = text.split_whitespace();
let (Some(reference), Some(delivered), None) = (fields.next(), fields.next(), fields.next())
else {
return contents.to_vec();
};
let valid_reference = reference
.strip_prefix("ref:")
.is_some_and(|value| value.parse::<u64>().is_ok());
let valid_delivered = delivered
.strip_prefix("del:")
.is_some_and(|value| value.parse::<u64>().is_ok());
if !valid_reference || !valid_delivered {
return contents.to_vec();
}
let mut normalized = b"ref:0 del:0".to_vec();
if has_newline {
normalized.push(b'\n');
}
normalized
}
fn sanitize_sysfs_rtc_attribute(
contents: &[u8],
kind: ProcfsKind,
virtual_realtime_seconds: i64,
) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let value = contents.strip_suffix(b"\n").unwrap_or(contents);
let valid = match kind {
ProcfsKind::SysfsRtcDate => matches_digit_separated(value, 10, &[(4, b'-'), (7, b'-')]),
ProcfsKind::SysfsRtcTime => matches_digit_separated(value, 8, &[(2, b':'), (5, b':')]),
ProcfsKind::SysfsRtcEpoch => !value.is_empty() && value.iter().all(u8::is_ascii_digit),
_ => return contents.to_vec(),
};
if !valid {
return contents.to_vec();
}
let Some(now) = DateTime::<Utc>::from_timestamp(virtual_realtime_seconds, 0) else {
return contents.to_vec();
};
let fixed = match kind {
ProcfsKind::SysfsRtcDate => now.format("%Y-%m-%d").to_string(),
ProcfsKind::SysfsRtcTime => now.format("%H:%M:%S").to_string(),
ProcfsKind::SysfsRtcEpoch => virtual_realtime_seconds.to_string(),
_ => unreachable!("validated sysfs RTC kind changed"),
};
let mut normalized = fixed.into_bytes();
if has_newline {
normalized.push(b'\n');
}
normalized
}
fn matches_digit_separated(value: &[u8], expected_len: usize, separators: &[(usize, u8)]) -> bool {
value.len() == expected_len
&& value.iter().enumerate().all(|(index, byte)| {
separators
.iter()
.find_map(|(position, separator)| (*position == index).then_some(*separator))
.map_or_else(|| byte.is_ascii_digit(), |separator| *byte == separator)
})
}
fn sanitize_btrfs_bytes_may_use(contents: &[u8]) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let value = contents.strip_suffix(b"\n").unwrap_or(contents);
if value.is_empty() || !value.iter().all(u8::is_ascii_digit) {
return contents.to_vec();
}
if has_newline {
b"0\n".to_vec()
} else {
b"0".to_vec()
}
}
fn sanitize_loadavg(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0.00 0.00 0.00 1/1 1\n".to_vec()
}
}
fn sanitize_uptime(contents: &[u8], virtual_uptime_seconds: u64) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
format!("{virtual_uptime_seconds}.00 0.00\n").into_bytes()
}
}
fn sanitize_system_stat(
contents: &[u8],
virtual_uptime_seconds: u64,
virtual_boot_time_seconds: i64,
) -> Vec<u8> {
const VOLATILE_FIELDS: &[&[u8]] = &[
b"intr",
b"ctxt",
b"processes",
b"procs_running",
b"procs_blocked",
b"softirq",
];
let cpu_count = contents
.split(|byte| *byte == b'\n')
.filter_map(|line| line.split(|byte| byte.is_ascii_whitespace()).next())
.filter(|name| name.starts_with(b"cpu") && *name != b"cpu")
.count() as u64;
let per_cpu_idle_ticks = virtual_uptime_seconds.saturating_mul(100);
let counters = sanitize_named_counters(
contents,
|name| name.starts_with(b"cpu") || VOLATILE_FIELDS.contains(&name),
|name, index| {
if index == 0 && name == b"cpu" {
per_cpu_idle_ticks.saturating_mul(cpu_count)
} else if index == 0 && name.starts_with(b"cpu") {
per_cpu_idle_ticks
} else {
0
}
},
);
let mut normalized = Vec::with_capacity(counters.len());
for line in counters.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
if body.starts_with(b"btime ") {
normalized.extend_from_slice(format!("btime {virtual_boot_time_seconds}").as_bytes());
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_named_counters(
contents: &[u8],
should_normalize: impl Fn(&[u8]) -> bool,
counter_value: impl Fn(&[u8], usize) -> u64,
) -> Vec<u8> {
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let mut fields = body
.split(|byte| byte.is_ascii_whitespace())
.filter(|field| !field.is_empty());
let name = fields.next().unwrap_or_default();
if should_normalize(name) {
normalized.extend_from_slice(name);
for (index, _) in fields.enumerate() {
normalized.push(b' ');
normalized.extend_from_slice(counter_value(name, index).to_string().as_bytes());
}
} else {
normalized.extend_from_slice(body);
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_meminfo(contents: &[u8], virtual_memory_kb: u64) -> Vec<u8> {
if contents.is_empty() {
return Vec::new();
}
format!(
"MemTotal: {virtual_memory_kb} kB\n\
MemFree: {virtual_memory_kb} kB\n\
MemAvailable: {virtual_memory_kb} kB\n\
Buffers: 0 kB\n\
Cached: 0 kB\n\
SwapCached: 0 kB\n\
Active: 0 kB\n\
Inactive: 0 kB\n\
Shmem: 0 kB\n\
SReclaimable: 0 kB\n\
SwapTotal: 0 kB\n\
SwapFree: 0 kB\n"
)
.into_bytes()
}
fn sanitize_key_users(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body.split_whitespace().collect::<Vec<_>>();
let [uid, usage, key_counts, key_quota, byte_quota] = fields.as_slice() else {
return contents.to_vec();
};
let Some(uid) = uid
.strip_suffix(':')
.filter(|uid| uid.parse::<u32>().is_ok())
else {
return contents.to_vec();
};
if usage.parse::<u64>().is_err() || parse_key_user_pair(key_counts).is_none() {
return contents.to_vec();
}
let Some((_, max_keys)) = parse_key_user_pair(key_quota) else {
return contents.to_vec();
};
let Some((_, max_bytes)) = parse_key_user_pair(byte_quota) else {
return contents.to_vec();
};
normalized.extend_from_slice(format!("{uid}: 0 0/0 0/{max_keys} 0/{max_bytes}").as_bytes());
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn parse_key_user_pair(field: &str) -> Option<(u64, u64)> {
let (current, maximum) = field.split_once('/')?;
Some((current.parse().ok()?, maximum.parse().ok()?))
}
fn sanitize_buddyinfo(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body.split_whitespace().collect::<Vec<_>>();
let is_buddy_row = fields.len() >= 5
&& fields[0] == "Node"
&& fields[1].ends_with(',')
&& fields[2] == "zone"
&& fields[4..].iter().all(|field| field.parse::<u64>().is_ok());
if is_buddy_row {
normalized.extend_from_slice(fields[..4].join(" ").as_bytes());
for _ in &fields[4..] {
normalized.extend_from_slice(b" 0");
}
} else {
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
const VIRTUAL_FILE_MAX: u64 = i64::MAX as u64;
fn sanitize_file_nr(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
format!("0\t0\t{VIRTUAL_FILE_MAX}\n").into_bytes()
}
}
fn sanitize_file_max(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
format!("{VIRTUAL_FILE_MAX}\n").into_bytes()
}
}
fn sanitize_inode_nr(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0\t0\n".to_vec()
}
}
fn sanitize_inode_state(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0\t0\t0\t0\t0\t0\t0\n".to_vec()
}
}
fn sanitize_dentry_state(contents: &[u8]) -> Vec<u8> {
if contents.is_empty() {
Vec::new()
} else {
b"0\t0\t45\t0\t0\t0\n".to_vec()
}
}
fn sanitize_swaps(contents: &[u8]) -> Vec<u8> {
const HEADER: [&str; 5] = ["Filename", "Type", "Size", "Used", "Priority"];
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut lines = text.split_inclusive('\n');
let Some(header) = lines.next() else {
return Vec::new();
};
if !header.split_whitespace().eq(HEADER) {
return contents.to_vec();
}
let mut normalized = Vec::with_capacity(contents.len());
normalized.extend_from_slice(header.as_bytes());
for line in lines {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body.split_whitespace().collect::<Vec<_>>();
let [filename, swap_type, size, used, priority] = fields.as_slice() else {
return contents.to_vec();
};
if size.parse::<u64>().is_err()
|| used.parse::<u64>().is_err()
|| priority.parse::<i32>().is_err()
{
return contents.to_vec();
}
normalized.extend_from_slice(
format!("{filename}\t{swap_type}\t{size}\t0\t{priority}").as_bytes(),
);
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_pipe_max_size(contents: &[u8]) -> Vec<u8> {
let Ok(value) = std::str::from_utf8(contents) else {
return Vec::new();
};
value.trim().parse::<u64>().ok().map_or_else(Vec::new, |_| {
format!("{DETERMINISTIC_PIPE_CAPACITY_BYTES}\n").into_bytes()
})
}
fn sanitize_aio_nr(contents: &[u8]) -> Vec<u8> {
let Ok(value) = std::str::from_utf8(contents) else {
return Vec::new();
};
value
.trim()
.parse::<u64>()
.ok()
.map_or_else(Vec::new, |_| b"0\n".to_vec())
}
fn sanitize_pty_nr(contents: &[u8], virtual_count: usize) -> Vec<u8> {
let Ok(value) = std::str::from_utf8(contents) else {
return Vec::new();
};
if value.trim().parse::<u64>().is_err() {
return Vec::new();
}
format!("{virtual_count}\n").into_bytes()
}
fn is_node_vmstat_path(path: &str) -> bool {
let relative = path
.strip_prefix("/sys/devices/system/node/")
.unwrap_or(path);
let Some(node) = relative.strip_suffix("/vmstat") else {
return false;
};
let Some(index) = node.strip_prefix("node") else {
return false;
};
!index.is_empty() && index.bytes().all(|byte| byte.is_ascii_digit())
}
fn sanitize_node_vmstat(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let mut fields = body.split_whitespace();
let (Some(name), Some(value), None) = (fields.next(), fields.next(), fields.next()) else {
return contents.to_vec();
};
if value.parse::<u64>().is_err() {
return contents.to_vec();
}
normalized.extend_from_slice(name.as_bytes());
normalized.extend_from_slice(b" 0");
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_uevent_seqnum(contents: &[u8]) -> Vec<u8> {
let Some(value) = contents.strip_suffix(b"\n") else {
return contents.to_vec();
};
if value.is_empty() || !value.iter().all(u8::is_ascii_digit) {
contents.to_vec()
} else {
b"0\n".to_vec()
}
}
fn sanitize_sockstat(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let mut fields = body
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
match fields.first().map(String::as_str) {
Some("TCP:") => {
let inuse = sockstat_field(&fields, "inuse").unwrap_or_else(|| "0".to_owned());
replace_sockstat_field(&mut fields, "orphan", "0");
replace_sockstat_field(&mut fields, "alloc", &inuse);
replace_sockstat_field(&mut fields, "mem", "0");
}
Some("UDP:") => replace_sockstat_field(&mut fields, "mem", "0"),
_ => {}
}
normalized.extend_from_slice(fields.join(" ").as_bytes());
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_unix_sockets(contents: &[u8]) -> Vec<u8> {
const HEADER: [&str; 8] = [
"Num", "RefCount", "Protocol", "Flags", "Type", "St", "Inode", "Path",
];
const ZERO_NUM: &str = "0000000000000000:";
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let Some(body) = text.strip_suffix('\n') else {
return contents.to_vec();
};
let mut lines = body.split('\n');
let Some(header) = lines.next() else {
return contents.to_vec();
};
if !header.split_whitespace().eq(HEADER) {
return contents.to_vec();
}
let mut rows = Vec::new();
for line in lines {
let Some((mut fields, path)) = unix_socket_fields(line) else {
return contents.to_vec();
};
let Some(num) = fields[0].strip_suffix(':') else {
return contents.to_vec();
};
if !is_fixed_lower_hex(num, 16)
|| !is_fixed_lower_hex(fields[1], 8)
|| !is_fixed_lower_hex(fields[2], 8)
|| !is_fixed_lower_hex(fields[3], 8)
|| !is_fixed_lower_hex(fields[4], 4)
|| !is_fixed_lower_hex(fields[5], 2)
|| !is_decimal(fields[6])
{
return contents.to_vec();
}
fields[0] = ZERO_NUM;
fields[6] = "0";
let mut row = fields.join(" ");
if !path.is_empty() {
row.push(' ');
row.push_str(path);
}
rows.push(row);
}
rows.sort_unstable();
let mut normalized = String::with_capacity(text.len());
normalized.push_str(header);
normalized.push('\n');
for row in rows {
normalized.push_str(&row);
normalized.push('\n');
}
normalized.into_bytes()
}
fn unix_socket_fields(line: &str) -> Option<(Vec<&str>, &str)> {
const FIELD_COUNT: usize = 7;
let mut fields = Vec::with_capacity(FIELD_COUNT);
let mut remainder = line;
while fields.len() < FIELD_COUNT {
remainder = remainder.trim_start_matches(|character: char| character.is_ascii_whitespace());
if remainder.is_empty() {
return None;
}
let end = remainder
.find(|character: char| character.is_ascii_whitespace())
.unwrap_or(remainder.len());
fields.push(&remainder[..end]);
remainder = &remainder[end..];
}
Some((
fields,
remainder.trim_start_matches(|character: char| character.is_ascii_whitespace()),
))
}
fn is_fixed_lower_hex(field: &str, width: usize) -> bool {
field.len() == width
&& field
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte))
}
fn is_decimal(field: &str) -> bool {
!field.is_empty()
&& field.bytes().all(|byte| byte.is_ascii_digit())
&& field.parse::<u64>().is_ok()
}
fn sanitize_netlink_sockets(contents: &[u8]) -> Vec<u8> {
const HEADER: [&str; 10] = [
"sk", "Eth", "Pid", "Groups", "Rmem", "Wmem", "Dump", "Locks", "Drops", "Inode",
];
const ZERO_POINTER: &str = "0000000000000000";
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let Some(body) = text.strip_suffix('\n') else {
return contents.to_vec();
};
let mut lines = body.split('\n');
let Some(header) = lines.next() else {
return contents.to_vec();
};
if !header.split_whitespace().eq(HEADER) {
return contents.to_vec();
}
let mut rows = Vec::new();
for line in lines {
let mut fields = line.split_whitespace().collect::<Vec<_>>();
if fields.len() != HEADER.len()
|| !is_lower_hex(fields[0], 16)
|| !is_lower_hex(fields[3], 8)
{
return contents.to_vec();
}
let Some(key) = [
parse_decimal(fields[1]),
parse_decimal(fields[2]),
parse_lower_hex(fields[3]),
parse_decimal(fields[4]),
parse_decimal(fields[5]),
parse_decimal(fields[6]),
parse_decimal(fields[7]),
parse_decimal(fields[8]),
]
.into_iter()
.collect::<Option<Vec<_>>>() else {
return contents.to_vec();
};
if parse_decimal(fields[9]).is_none() {
return contents.to_vec();
}
fields[0] = ZERO_POINTER;
fields[9] = "0";
rows.push((key, fields.join(" ")));
}
rows.sort_unstable();
let mut normalized = String::with_capacity(text.len());
normalized.push_str(header);
normalized.push('\n');
for (_, row) in rows {
normalized.push_str(&row);
normalized.push('\n');
}
normalized.into_bytes()
}
fn sanitize_inet_sockets(contents: &[u8]) -> Vec<u8> {
const INODE_FIELD: usize = 9;
const POINTER_FIELD: usize = 11;
const MIN_FIELDS: usize = 13;
const ZERO_POINTER: &str = "0000000000000000";
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let Some(body) = text.strip_suffix('\n') else {
return contents.to_vec();
};
let mut lines = body.split('\n');
let Some(header) = lines.next() else {
return contents.to_vec();
};
let header_fields = header.split_whitespace().collect::<Vec<_>>();
if header_fields.first() != Some(&"sl") || !header_fields.contains(&"inode") {
return contents.to_vec();
}
let mut rows = Vec::new();
for line in lines {
let mut fields = line.split_whitespace().collect::<Vec<_>>();
if fields.len() < MIN_FIELDS {
return contents.to_vec();
}
if parse_decimal(fields[INODE_FIELD]).is_none() || !is_lower_hex(fields[POINTER_FIELD], 16)
{
return contents.to_vec();
}
fields[INODE_FIELD] = "0";
fields[POINTER_FIELD] = ZERO_POINTER;
rows.push(fields[1..].join(" "));
}
rows.sort_unstable();
let mut normalized = String::with_capacity(text.len());
normalized.push_str(header);
normalized.push('\n');
for (index, row) in rows.iter().enumerate() {
normalized.push_str(&format!("{index}: {row}\n"));
}
normalized.into_bytes()
}
fn is_lower_hex(field: &str, width: usize) -> bool {
field.len() == width
&& field
.bytes()
.all(|byte| byte.is_ascii_digit() || (b'a'..=b'f').contains(&byte))
}
fn parse_lower_hex(field: &str) -> Option<u64> {
u64::from_str_radix(field, 16).ok()
}
fn parse_decimal(field: &str) -> Option<u64> {
if field.is_empty() || !field.bytes().all(|byte| byte.is_ascii_digit()) {
return None;
}
field.parse().ok()
}
fn is_irq_per_cpu_count_path(path: &Path) -> bool {
if path.file_name().and_then(|leaf| leaf.to_str()) != Some("per_cpu_count") {
return false;
}
let Some(irq_directory) = path.parent() else {
return false;
};
let Some(irq) = irq_directory.file_name().and_then(|name| name.to_str()) else {
return false;
};
!irq.is_empty()
&& irq.bytes().all(|byte| byte.is_ascii_digit())
&& irq_directory.parent() == Some(Path::new("/sys/kernel/irq"))
}
fn sanitize_irq_per_cpu_count(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let has_newline = text.ends_with('\n');
let body = text.strip_suffix('\n').unwrap_or(text);
if body.contains('\n') {
return contents.to_vec();
}
let fields = body.split(',').collect::<Vec<_>>();
if fields.is_empty()
|| fields
.iter()
.any(|field| field.is_empty() || !field.bytes().all(|byte| byte.is_ascii_digit()))
{
return contents.to_vec();
}
let mut normalized = vec!["0"; fields.len()].join(",").into_bytes();
if has_newline {
normalized.push(b'\n');
}
normalized
}
fn is_block_inflight_path(path: &Path) -> bool {
path.file_name().and_then(|leaf| leaf.to_str()) == Some("inflight")
&& path.parent().and_then(Path::parent) == Some(Path::new("/sys/block"))
}
fn sanitize_block_inflight(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let has_newline = text.ends_with('\n');
let body = text.strip_suffix('\n').unwrap_or(text);
if body.contains('\n') {
return contents.to_vec();
}
let mut fields = body.split_whitespace();
let valid = matches!(
(fields.next(), fields.next(), fields.next()),
(Some(reads), Some(writes), None)
if reads.parse::<u64>().is_ok() && writes.parse::<u64>().is_ok()
);
if !valid {
return contents.to_vec();
}
if has_newline {
b"0 0\n".to_vec()
} else {
b"0 0".to_vec()
}
}
fn sanitize_vmstat(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
let mut row_count = 0;
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body.split_whitespace().collect::<Vec<_>>();
if fields.len() != 2 || fields[0].is_empty() || fields[1].parse::<u64>().is_err() {
return contents.to_vec();
}
normalized.extend_from_slice(fields[0].as_bytes());
normalized.extend_from_slice(b" 0");
if has_newline {
normalized.push(b'\n');
}
row_count += 1;
}
if row_count == 0 {
contents.to_vec()
} else {
normalized
}
}
fn sockstat_field(fields: &[String], name: &str) -> Option<String> {
let index = fields.iter().position(|field| field == name)?;
fields.get(index + 1).cloned()
}
fn replace_sockstat_field(fields: &mut [String], name: &str, value: &str) {
let Some(index) = fields.iter().position(|field| field == name) else {
return;
};
let Some(field_value) = fields.get_mut(index + 1) else {
return;
};
*field_value = value.to_owned();
}
fn sanitize_self_sched(contents: &[u8]) -> Vec<u8> {
const FLOAT_FIELDS: &[&str] = &["se.exec_start", "se.vruntime", "se.sum_exec_runtime"];
const INTEGER_FIELDS: &[&str] = &[
"se.nr_migrations",
"nr_switches",
"nr_voluntary_switches",
"nr_involuntary_switches",
"se.avg.load_sum",
"se.avg.runnable_sum",
"se.avg.util_sum",
"se.avg.load_avg",
"se.avg.runnable_avg",
"se.avg.util_avg",
"se.avg.last_update_time",
"se.avg.util_est",
"clock-delta",
"mm->numa_scan_seq",
"numa_pages_migrated",
"total_numa_faults",
];
const STABLE_INTEGER_FIELDS: &[&str] = &[
"se.load.weight",
"policy",
"prio",
"se.slice",
"ext.enabled",
"numa_preferred_nid",
"uclamp.min",
"uclamp.max",
"effective uclamp.min",
"effective uclamp.max",
];
const UCLAMP_FIELDS: &[(&str, &str)] = &[
("uclamp.min", "0"),
("uclamp.max", "1024"),
("effective uclamp.min", "0"),
("effective uclamp.max", "1024"),
];
let Ok(text) = std::str::from_utf8(contents) else {
return Vec::new();
};
let mut normalized = Vec::with_capacity(contents.len());
let mut core_fields_seen = [false; 3];
let mut header_seen = false;
for (line_index, line) in text.split_inclusive('\n').enumerate() {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
if line_index == 0 {
let Some((name, details)) = body.rsplit_once(" (") else {
return Vec::new();
};
let Some(details) = details.strip_suffix(')') else {
return Vec::new();
};
let Some((pid, threads)) = details.split_once(", #threads: ") else {
return Vec::new();
};
if pid.parse::<i32>().is_err() || threads.parse::<u64>().is_err() {
return Vec::new();
}
normalized.extend_from_slice(format!("{name} (0, #threads: 1)").as_bytes());
if has_newline {
normalized.push(b'\n');
}
header_seen = true;
continue;
}
let Some((left, right)) = body.split_once(':') else {
if !body.is_empty() && body.bytes().all(|byte| byte == b'-') {
normalized.extend_from_slice(body.as_bytes());
} else if body.starts_with("current_node=") {
let fields = body
.replace(',', "")
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
if fields.len() != 2
|| !fields.iter().all(|field| {
field.split_once('=').is_some_and(|(name, value)| {
matches!(name, "current_node" | "numa_group_id")
&& value.parse::<i64>().is_ok()
})
})
{
return Vec::new();
}
normalized.extend_from_slice(b"current_node=0, numa_group_id=0");
} else if body.starts_with("numa_faults ") {
let fields = body.split_whitespace().skip(1).collect::<Vec<_>>();
let expected = [
"node",
"task_private",
"task_shared",
"group_private",
"group_shared",
];
if fields.len() != expected.len()
|| !fields.iter().zip(expected).all(|(field, expected)| {
field.split_once('=').is_some_and(|(name, value)| {
name == expected && value.parse::<u64>().is_ok()
})
})
{
return Vec::new();
}
normalized.extend_from_slice(
b"numa_faults node=0 task_private=0 task_shared=0 group_private=0 group_shared=0",
);
} else {
return Vec::new();
}
if has_newline {
normalized.push(b'\n');
}
continue;
};
let label = left.trim();
let replacement = if let Some(index) = FLOAT_FIELDS.iter().position(|field| *field == label)
{
let Ok(value) = right.trim().parse::<f64>() else {
return Vec::new();
};
if !value.is_finite() || value.is_sign_negative() {
return Vec::new();
}
core_fields_seen[index] = true;
Some("0.000000")
} else if let Some((_, replacement)) =
UCLAMP_FIELDS.iter().find(|(field, _)| *field == label)
{
if right.trim().parse::<u128>().is_err() {
return Vec::new();
}
Some(*replacement)
} else if INTEGER_FIELDS.contains(&label) {
if right.trim().parse::<u128>().is_err() {
return Vec::new();
}
Some("0")
} else if STABLE_INTEGER_FIELDS.contains(&label) {
if right.trim().parse::<i128>().is_err() {
return Vec::new();
}
None
} else if right.trim().parse::<u128>().is_ok() {
Some("0")
} else {
return Vec::new();
};
if let Some(value) = replacement {
normalized.extend_from_slice(left.as_bytes());
normalized.extend_from_slice(b": ");
normalized.extend_from_slice(value.as_bytes());
} else {
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
if header_seen && core_fields_seen.iter().all(|seen| *seen) {
normalized
} else {
Vec::new()
}
}
fn sanitize_fdinfo(contents: &[u8], identity: Option<(u64, i32, u64, u64)>) -> Vec<u8> {
let Some((virtual_inode, logical_flags, virtual_open_file, virtual_mount_id)) = identity else {
return Vec::new();
};
let Ok(text) = std::str::from_utf8(contents) else {
return Vec::new();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
if body.starts_with("mnt_id:") {
normalized.extend_from_slice(format!("mnt_id:\t{virtual_mount_id}").as_bytes());
} else if body.starts_with("ino:") {
normalized.extend_from_slice(format!("ino:\t{virtual_inode}").as_bytes());
} else if body.starts_with("flags:") {
normalized.extend_from_slice(format!("flags:\t{logical_flags:07o}").as_bytes());
} else if body.starts_with("eventfd-id:") {
normalized.extend_from_slice(format!("eventfd-id: {virtual_open_file}").as_bytes());
} else if body.starts_with("Pid:") {
normalized.extend_from_slice(b"Pid:\t1");
} else if body.starts_with("NSpid:") {
normalized.extend_from_slice(b"NSpid:\t1");
} else if body.starts_with("tfd:")
|| body.starts_with("inotify ")
|| body.starts_with("lock:")
{
return Vec::new();
} else {
let allowed = body.split_once(':').is_some_and(|(label, _)| {
matches!(
label,
"pos"
| "eventfd-count"
| "eventfd-semaphore"
| "sigmask"
| "clockid"
| "ticks"
| "settime flags"
| "it_value"
| "it_interval"
| "seals"
| "scm_fds"
)
});
if !allowed {
return Vec::new();
}
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_numa_maps(contents: &[u8]) -> Vec<u8> {
fn page_accounting_value(field: &str) -> Option<&str> {
let (name, value) = field.split_once('=')?;
let fixed_counter = matches!(
name,
"active" | "anon" | "dirty" | "mapped" | "mapmax" | "swapcache" | "writeback"
);
let node_counter = name
.strip_prefix('N')
.is_some_and(|node| !node.is_empty() && node.bytes().all(|byte| byte.is_ascii_digit()));
(fixed_counter || node_counter).then_some(value)
}
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
let mut row_count = 0;
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body.split_whitespace().collect::<Vec<_>>();
if fields.len() < 2 || u64::from_str_radix(fields[0], 16).is_err() {
return contents.to_vec();
}
let mut kept = Vec::with_capacity(fields.len());
for field in fields {
if let Some(value) = page_accounting_value(field) {
if value.parse::<u64>().is_err() {
return contents.to_vec();
}
} else {
kept.push(field);
}
}
normalized.extend_from_slice(kept.join(" ").as_bytes());
if has_newline {
normalized.push(b'\n');
}
row_count += 1;
}
if row_count == 0 {
contents.to_vec()
} else {
normalized
}
}
const SMAPS_ACCOUNTING_FIELDS: &[&str] = &[
"Rss",
"Pss",
"Pss_Dirty",
"Pss_Anon",
"Pss_File",
"Pss_Shmem",
"Shared_Clean",
"Shared_Dirty",
"Private_Clean",
"Private_Dirty",
"Referenced",
"Anonymous",
"KSM",
"LazyFree",
"AnonHugePages",
"ShmemPmdMapped",
"FilePmdMapped",
"Shared_Hugetlb",
"Private_Hugetlb",
"Swap",
"SwapPss",
"Locked",
];
fn sanitize_smaps_rollup(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let accounting_label = body.split_once(':').and_then(|(label, value)| {
let mut fields = value.split_whitespace();
let amount = fields.next()?;
(amount.parse::<u64>().is_ok()
&& fields.next() == Some("kB")
&& fields.next().is_none())
.then_some(label)
});
if let Some(label) =
accounting_label.filter(|label| SMAPS_ACCOUNTING_FIELDS.contains(label))
{
normalized.extend_from_slice(label.as_bytes());
normalized.extend_from_slice(b":\t0 kB");
} else {
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_arch_status(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let elapsed = body
.strip_prefix("AVX512_elapsed_ms:")
.map(str::trim)
.and_then(|value| value.parse::<u64>().ok());
if elapsed.is_some() {
normalized.extend_from_slice(b"AVX512_elapsed_ms:\t0");
} else {
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_smaps(contents: &[u8], table: &BTreeMap<(u64, u64), (u64, u64)>) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
let mut mapping_count = 0;
let mut accounting_count = 0;
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
if is_smaps_mapping_header(body) {
mapping_count += 1;
normalized.extend_from_slice(rewrite_mapping_header(body, table).as_bytes());
} else {
if mapping_count == 0 {
return contents.to_vec();
}
let Some((label, value)) = body.split_once(':') else {
return contents.to_vec();
};
if SMAPS_ACCOUNTING_FIELDS.contains(&label) {
if !is_smaps_kilobyte_value(value) {
return contents.to_vec();
}
normalized.extend_from_slice(label.as_bytes());
normalized.extend_from_slice(b":\t0 kB");
accounting_count += 1;
} else {
let valid_static_field = match label {
"Size" | "KernelPageSize" | "MMUPageSize" => is_smaps_kilobyte_value(value),
"THPeligible" | "ProtectionKey" => is_smaps_integer_value(value),
"VmFlags" => value
.split_whitespace()
.all(|flag| flag.bytes().all(|byte| byte.is_ascii_alphanumeric())),
_ => false,
};
if !valid_static_field {
return contents.to_vec();
}
normalized.extend_from_slice(body.as_bytes());
}
}
if has_newline {
normalized.push(b'\n');
}
}
if mapping_count == 0 || accounting_count == 0 {
contents.to_vec()
} else {
normalized
}
}
pub(crate) fn mapping_header_identity(line: &str) -> Option<(u64, u64)> {
let mut fields = line.split_whitespace();
let _range = fields.next()?;
let _perms = fields.next()?;
let _offset = fields.next()?;
let (major, minor) = fields.next()?.split_once(':')?;
let major = u32::from_str_radix(major, 16).ok()?;
let minor = u32::from_str_radix(minor, 16).ok()?;
let inode: u64 = fields.next()?.parse().ok()?;
if inode == 0 {
return None;
}
Some((libc::makedev(major, minor), inode))
}
fn rewrite_mapping_header(line: &str, table: &BTreeMap<(u64, u64), (u64, u64)>) -> String {
let Some(raw) = mapping_header_identity(line) else {
return line.to_string();
};
let Some((det_dev, det_inode)) = table.get(&raw).copied() else {
return line.to_string();
};
let mut out = String::with_capacity(line.len());
let mut remaining = line;
for _ in 0..3 {
let field_end = match remaining.find(char::is_whitespace) {
Some(index) => index,
None => return line.to_string(),
};
let gap_end = remaining[field_end..]
.find(|c: char| !c.is_whitespace())
.map_or(remaining.len(), |offset| field_end + offset);
out.push_str(&remaining[..gap_end]);
remaining = &remaining[gap_end..];
}
let dev_end = match remaining.find(char::is_whitespace) {
Some(index) => index,
None => return line.to_string(),
};
let dev_gap = remaining[dev_end..]
.find(|c: char| !c.is_whitespace())
.map_or(remaining.len(), |offset| dev_end + offset);
out.push_str(&format!(
"{:02x}:{:02x}",
libc::major(det_dev),
libc::minor(det_dev)
));
out.push_str(&remaining[dev_end..dev_gap]);
remaining = &remaining[dev_gap..];
let inode_end = remaining
.find(char::is_whitespace)
.unwrap_or(remaining.len());
let tail = &remaining[inode_end..];
let pad_len = tail
.find(|c: char| !c.is_whitespace())
.unwrap_or(tail.len());
let name_column = (line.len() - remaining.len()) + inode_end + pad_len;
out.push_str(&det_inode.to_string());
let name = &tail[pad_len..];
if name.is_empty() {
return out;
}
if out.len() < name_column {
out.extend(std::iter::repeat_n(' ', name_column - out.len()));
} else {
out.push(' ');
}
out.push_str(name);
out
}
fn sanitize_maps(contents: &[u8], table: &BTreeMap<(u64, u64), (u64, u64)>) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut out = String::with_capacity(text.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
out.push_str(&rewrite_mapping_header(body, table));
if has_newline {
out.push('\n');
}
}
out.into_bytes()
}
fn is_smaps_mapping_header(line: &str) -> bool {
let mut fields = line.split_whitespace();
let Some((start, end)) = fields.next().and_then(|range| range.split_once('-')) else {
return false;
};
let (Ok(start), Ok(end)) = (u64::from_str_radix(start, 16), u64::from_str_radix(end, 16))
else {
return false;
};
let Some(permissions) = fields.next() else {
return false;
};
let permissions = permissions.as_bytes();
let valid_permissions = permissions.len() == 4
&& matches!(permissions[0], b'r' | b'-')
&& matches!(permissions[1], b'w' | b'-')
&& matches!(permissions[2], b'x' | b'-')
&& matches!(permissions[3], b'p' | b's');
let valid_offset = fields
.next()
.is_some_and(|offset| u64::from_str_radix(offset, 16).is_ok());
let valid_device = fields.next().is_some_and(|device| {
device.split_once(':').is_some_and(|(major, minor)| {
u64::from_str_radix(major, 16).is_ok() && u64::from_str_radix(minor, 16).is_ok()
})
});
let valid_inode = fields
.next()
.is_some_and(|inode| inode.parse::<u64>().is_ok());
start < end && valid_permissions && valid_offset && valid_device && valid_inode
}
fn is_smaps_kilobyte_value(value: &str) -> bool {
let mut fields = value.split_whitespace();
matches!(
(fields.next(), fields.next(), fields.next()),
(Some(number), Some("kB"), None) if number.parse::<u64>().is_ok()
)
}
fn is_smaps_integer_value(value: &str) -> bool {
let mut fields = value.split_whitespace();
matches!(
(fields.next(), fields.next()),
(Some(number), None) if number.parse::<u64>().is_ok()
)
}
fn sanitize_pressure(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let fields = body
.split_whitespace()
.map(|field| {
let Some((name, _)) = field.split_once('=') else {
return field.to_owned();
};
match name {
"avg10" | "avg60" | "avg300" => format!("{name}=0.00"),
"total" => "total=0".to_owned(),
_ => field.to_owned(),
}
})
.collect::<Vec<_>>();
normalized.extend_from_slice(fields.join(" ").as_bytes());
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_zoneinfo(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let trimmed = body.trim_start();
if trimmed.starts_with("Node ") || trimmed.starts_with("cpu: ") {
normalized.extend_from_slice(body.as_bytes());
} else {
normalized.extend_from_slice(zero_decimal_runs(body).as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_interrupt_counters(contents: &[u8]) -> Vec<u8> {
let mut normalized = Vec::with_capacity(contents.len());
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let Some(colon) = body.iter().position(|byte| *byte == b':') else {
normalized.extend_from_slice(line);
continue;
};
let fields = body[colon + 1..]
.split(u8::is_ascii_whitespace)
.filter(|field| !field.is_empty())
.collect::<Vec<_>>();
let counter_count = fields
.iter()
.take_while(|field| field.iter().all(u8::is_ascii_digit))
.count();
if counter_count == 0 {
normalized.extend_from_slice(line);
continue;
}
normalized.extend_from_slice(&body[..=colon]);
for (index, field) in fields.iter().enumerate() {
normalized.push(b' ');
normalized.extend_from_slice(if index < counter_count { b"0" } else { field });
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_modules(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let mut fields = body
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
if fields.len() >= 4 {
let holders = if fields[3] == "-" {
0
} else {
fields[3]
.split(',')
.filter(|holder| !holder.is_empty())
.count()
};
fields[2] = holders.to_string();
}
normalized.extend_from_slice(fields.join(" ").as_bytes());
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_module_refcnt(contents: &[u8], module: &str, modules: &str) -> Vec<u8> {
let has_newline = contents.ends_with(b"\n");
let value = contents.strip_suffix(b"\n").unwrap_or(contents);
if value.is_empty() || !value.iter().all(u8::is_ascii_digit) {
return contents.to_vec();
}
let count = deterministic_module_holder_count(modules, module);
let mut normalized = count.to_string().into_bytes();
if has_newline {
normalized.push(b'\n');
}
normalized
}
fn sanitize_schedstat(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let mut fields = body
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
let preserved_fields = match fields.first().map(String::as_str) {
Some("timestamp") => 1,
Some(label) if numbered_label(label, "cpu") => 1,
Some(label) if numbered_label(label, "domain") => 3,
_ => fields.len(),
};
for field in fields.iter_mut().skip(preserved_fields) {
if field.parse::<u128>().is_ok() {
*field = "0".to_owned();
}
}
normalized.extend_from_slice(fields.join(" ").as_bytes());
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn decimal_bytes(field: &[u8]) -> Option<u64> {
parse_decimal(std::str::from_utf8(field).ok()?)
}
fn mount_peer_group(field: &[u8]) -> Result<Option<(&'static [u8], u64)>, ()> {
for prefix in MOUNT_PEER_PREFIXES {
if let Some(raw) = field.strip_prefix(prefix) {
return decimal_bytes(raw).map(|id| Some((prefix, id))).ok_or(());
}
}
Ok(None)
}
fn sanitize_mountinfo(contents: &[u8], snapshot: &MountInfoSnapshot) -> Vec<u8> {
fn rewrite_root_prefix(field: &[u8], rewrites: &[(Vec<u8>, Vec<u8>)]) -> Option<Vec<u8>> {
for (raw, deterministic) in rewrites {
let Some(suffix) = field.strip_prefix(raw.as_slice()) else {
continue;
};
if suffix.is_empty() || suffix.first() == Some(&b'/') {
let mut rewritten = deterministic.clone();
rewritten.extend_from_slice(suffix);
return Some(rewritten);
}
}
None
}
let mut normalized = Vec::with_capacity(contents.len());
let mut rows = snapshot.rows.iter();
let mut rewritten_roots = 0;
for line in contents.split_inclusive(|byte| *byte == b'\n') {
let has_newline = line.last() == Some(&b'\n');
let body = line.strip_suffix(b"\n").unwrap_or(line);
let row = rows.next().expect("validated mountinfo row disappeared");
let fields = body.split(|byte| *byte == b' ').collect::<Vec<_>>();
let separator = fields
.iter()
.position(|field| *field == b"-")
.expect("validated mountinfo separator disappeared");
for (index, field) in fields.iter().enumerate() {
if index > 0 {
normalized.push(b' ');
}
match index {
0 | 1 => {
let raw = if index == 0 {
row.raw_mount_id
} else {
row.raw_parent_id
};
normalized.extend_from_slice(snapshot.mount_ids[&raw].to_string().as_bytes());
}
2 => {
let device = snapshot.devices[&row.raw_device];
normalized.extend_from_slice(
format!("{}:{}", libc::major(device), libc::minor(device)).as_bytes(),
);
}
3 if snapshot.root_rewrites.contains_key(&row.raw_mount_id) => {
normalized.extend_from_slice(&snapshot.root_rewrites[&row.raw_mount_id]);
rewritten_roots += 1;
}
3 => match rewrite_root_prefix(field, &snapshot.root_prefix_rewrites) {
Some(rewritten) => normalized.extend_from_slice(&rewritten),
None => normalized.extend_from_slice(field),
},
4 => match rewrite_root_prefix(field, &snapshot.mountpoint_prefix_rewrites) {
Some(rewritten) => normalized.extend_from_slice(&rewritten),
None => normalized.extend_from_slice(field),
},
6.. if index < separator => match mount_peer_group(field)
.expect("validated mountinfo peer group became malformed")
{
Some((prefix, raw)) => {
normalized.extend_from_slice(prefix);
normalized
.extend_from_slice(snapshot.peer_groups[&raw].to_string().as_bytes());
}
None => normalized.extend_from_slice(field),
},
_ => normalized.extend_from_slice(field),
}
}
if has_newline {
normalized.push(b'\n');
}
}
assert!(
rows.next().is_none(),
"validated mountinfo row count changed"
);
assert_eq!(
rewritten_roots,
snapshot.root_rewrites.len(),
"a proven mountinfo root rewrite did not identify exactly one row"
);
normalized
}
fn sanitize_random_uuid(contents: &[u8], mut random: [u8; 16]) -> Vec<u8> {
const HYPHENS: &[usize] = &[8, 13, 18, 23];
if contents.len() != 37 || contents[36] != b'\n' {
return contents.to_vec();
}
for (index, byte) in contents[..36].iter().copied().enumerate() {
if HYPHENS.contains(&index) {
if byte != b'-' {
return contents.to_vec();
}
} else if !byte.is_ascii_digit() && !(b'a'..=b'f').contains(&byte) {
return contents.to_vec();
}
}
if contents[14] != b'4' || !matches!(contents[19], b'8' | b'9' | b'a' | b'b') {
return contents.to_vec();
}
random[6] = (random[6] & 0x0f) | 0x40;
random[8] = (random[8] & 0x3f) | 0x80;
format!(
"{:02x}{:02x}{:02x}{:02x}-{:02x}{:02x}-{:02x}{:02x}-{:02x}{:02x}-{:02x}{:02x}{:02x}{:02x}{:02x}{:02x}\n",
random[0],
random[1],
random[2],
random[3],
random[4],
random[5],
random[6],
random[7],
random[8],
random[9],
random[10],
random[11],
random[12],
random[13],
random[14],
random[15]
)
.into_bytes()
}
fn numbered_label(label: &str, prefix: &str) -> bool {
label.strip_prefix(prefix).is_some_and(|suffix| {
!suffix.is_empty() && suffix.bytes().all(|byte| byte.is_ascii_digit())
})
}
fn sanitize_self_schedstat(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let fields = text.split_whitespace().collect::<Vec<_>>();
if fields.len() != 3 || fields.iter().any(|field| field.parse::<u64>().is_err()) {
return contents.to_vec();
}
if text.ends_with('\n') {
b"0 0 0\n".to_vec()
} else {
b"0 0 0".to_vec()
}
}
fn sanitize_softnet_stat(_contents: &[u8]) -> Vec<u8> {
const VIRTUAL_SOFTNET_STAT: &[u8] =
b"00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000\n";
VIRTUAL_SOFTNET_STAT.to_vec()
}
fn zero_decimal_runs(text: &str) -> String {
let mut normalized = String::with_capacity(text.len());
let mut in_digits = false;
for character in text.chars() {
if character.is_ascii_digit() {
if !in_digits {
normalized.push('0');
in_digits = true;
}
} else {
in_digits = false;
normalized.push(character);
}
}
normalized
}
fn sanitize_protocols(contents: &[u8]) -> Vec<u8> {
const HEADER: &[&str] = &[
"protocol", "size", "sockets", "memory", "press", "maxhdr", "slab", "module",
];
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let mut lines = text.lines();
let Some(header) = lines.next() else {
return contents.to_vec();
};
let header_fields = header.split_whitespace().collect::<Vec<_>>();
if !header_fields.starts_with(HEADER) {
return contents.to_vec();
}
let mut normalized = Vec::new();
normalized.push(header.to_owned());
for line in lines {
let mut fields = line
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
if fields.len() != header_fields.len()
|| fields[1].parse::<u64>().is_err()
|| fields[2].parse::<u64>().is_err()
|| (fields[3] != "-1" && fields[3].parse::<u64>().is_err())
{
return contents.to_vec();
}
fields[2] = "0".to_owned();
if fields[3] != "-1" {
fields[3] = "0".to_owned();
}
normalized.push(fields.join(" "));
}
if normalized.len() == 1 {
return contents.to_vec();
}
let mut output = normalized.join("\n").into_bytes();
if text.ends_with('\n') {
output.push(b'\n');
}
output
}
fn sanitize_rtc(contents: &[u8], virtual_realtime_seconds: i64) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let Some(now) = DateTime::<Utc>::from_timestamp(virtual_realtime_seconds, 0) else {
return contents.to_vec();
};
let rtc_time = now.format("%H:%M:%S").to_string();
let rtc_date = now.format("%Y-%m-%d").to_string();
let mut normalized = Vec::with_capacity(contents.len());
for line in text.split_inclusive('\n') {
let has_newline = line.ends_with('\n');
let body = line.strip_suffix('\n').unwrap_or(line);
let replacement = body.split_once(':').and_then(|(key, _)| {
let value = match key.trim() {
"rtc_time" => rtc_time.as_str(),
"rtc_date" => rtc_date.as_str(),
"alrm_time" => "00:00:00",
"alrm_date" => rtc_date.as_str(),
"alarm_IRQ"
| "alrm_pending"
| "update IRQ enabled"
| "periodic IRQ enabled"
| "periodic_IRQ"
| "update_IRQ" => "no",
"periodic IRQ frequency" | "max user IRQ frequency" | "periodic_freq" => "0",
_ => return None,
};
Some(format!("{key}: {value}"))
});
if let Some(replacement) = replacement {
normalized.extend_from_slice(replacement.as_bytes());
} else {
normalized.extend_from_slice(body.as_bytes());
}
if has_newline {
normalized.push(b'\n');
}
}
normalized
}
fn sanitize_locks(contents: &[u8]) -> Vec<u8> {
let Ok(text) = std::str::from_utf8(contents) else {
return Vec::new();
};
if text.is_empty() {
return Vec::new();
}
#[derive(Clone)]
struct LockRow {
sequence: String,
waiter: bool,
fields: Vec<String>,
owner_index: usize,
object_index: usize,
}
let mut rows: Vec<LockRow> = Vec::new();
for line in text.lines() {
let fields = line
.split_whitespace()
.map(str::to_owned)
.collect::<Vec<_>>();
let waiter = fields.get(1).is_some_and(|field| field == "->");
let details = usize::from(waiter) + 1;
let owner_index = details + 3;
let object_index = details + 4;
let Some(sequence) = fields
.first()
.and_then(|field| field.strip_suffix(':'))
.filter(|field| !field.is_empty() && field.bytes().all(|byte| byte.is_ascii_digit()))
else {
return Vec::new();
};
if fields.len() != details + 7
|| fields[owner_index].parse::<i64>().is_err()
|| fields[object_index].split(':').count() != 3
{
return Vec::new();
}
rows.push(LockRow {
sequence: sequence.to_owned(),
waiter,
fields,
owner_index,
object_index,
});
}
let mut object_signatures: BTreeMap<String, Vec<String>> = BTreeMap::new();
for row in &rows {
let mut signature = row.fields.clone();
signature[0] = if row.waiter { "waiter" } else { "holder" }.to_owned();
signature[row.owner_index] = "owner".to_owned();
signature[row.object_index] = "object".to_owned();
object_signatures
.entry(row.fields[row.object_index].clone())
.or_default()
.push(signature.join(" "));
}
let mut objects = object_signatures.into_iter().collect::<Vec<_>>();
for (_, signatures) in &mut objects {
signatures.sort_unstable();
}
objects.sort_by(|left, right| left.1.cmp(&right.1).then_with(|| left.0.cmp(&right.0)));
let object_ids = objects
.into_iter()
.enumerate()
.map(|(index, (object, _))| (object, index + 1))
.collect::<BTreeMap<_, _>>();
for row in &mut rows {
let object_id = object_ids[&row.fields[row.object_index]];
row.fields[row.object_index] = format!("00:00:{object_id}");
}
let mut owner_signatures: BTreeMap<i64, Vec<String>> = BTreeMap::new();
for row in &rows {
let owner = row.fields[row.owner_index]
.parse::<i64>()
.expect("owner was validated above");
if owner == -1 {
continue;
}
let mut signature = row.fields.clone();
signature[0] = if row.waiter { "waiter" } else { "holder" }.to_owned();
signature[row.owner_index] = "owner".to_owned();
owner_signatures
.entry(owner)
.or_default()
.push(signature.join(" "));
}
let mut owners = owner_signatures.into_iter().collect::<Vec<_>>();
for (_, signatures) in &mut owners {
signatures.sort_unstable();
}
owners.sort_by(|left, right| left.1.cmp(&right.1).then_with(|| left.0.cmp(&right.0)));
let owner_ids = owners
.into_iter()
.enumerate()
.map(|(index, (owner, _))| (owner, index + 1))
.collect::<BTreeMap<_, _>>();
for row in &mut rows {
let owner = row.fields[row.owner_index]
.parse::<i64>()
.expect("owner was validated above");
if owner != -1 {
row.fields[row.owner_index] = owner_ids[&owner].to_string();
}
}
let mut groups: BTreeMap<String, Vec<LockRow>> = BTreeMap::new();
for row in rows {
groups.entry(row.sequence.clone()).or_default().push(row);
}
let mut groups = groups.into_values().collect::<Vec<_>>();
for rows in &mut groups {
rows.sort_by(|left, right| {
left.waiter
.cmp(&right.waiter)
.then_with(|| left.fields.cmp(&right.fields))
});
}
groups.sort_by_key(|rows| {
rows.iter()
.map(|row| {
let mut fields = row.fields.clone();
fields[0] = "sequence:".to_owned();
fields.join(" ")
})
.collect::<Vec<_>>()
});
let normalized_rows = groups
.into_iter()
.enumerate()
.flat_map(|(sequence, rows)| {
rows.into_iter().map(move |mut row| {
row.fields[0] = format!("{}:", sequence + 1);
row.fields.join(" ")
})
})
.collect::<Vec<_>>();
let mut normalized = normalized_rows.join("\n").into_bytes();
if text.ends_with('\n') {
normalized.push(b'\n');
}
normalized
}
fn is_btrfs_commit_stats_path(path: &Path) -> bool {
if path.file_name().and_then(|leaf| leaf.to_str()) != Some("commit_stats") {
return false;
}
let Some(filesystem_directory) = path.parent() else {
return false;
};
let Some(uuid) = filesystem_directory
.file_name()
.and_then(|name| name.to_str())
else {
return false;
};
is_lowercase_uuid(uuid) && filesystem_directory.parent() == Some(Path::new("/sys/fs/btrfs"))
}
fn is_lowercase_uuid(value: &str) -> bool {
value.len() == 36
&& value.bytes().enumerate().all(|(index, byte)| {
if matches!(index, 8 | 13 | 18 | 23) {
byte == b'-'
} else {
byte.is_ascii_digit() || matches!(byte, b'a'..=b'f')
}
})
}
fn sanitize_btrfs_commit_stats(contents: &[u8]) -> Vec<u8> {
const LABELS: &[&str] = &[
"commits",
"cur_commit_ms",
"last_commit_ms",
"max_commit_ms",
"total_commit_ms",
];
let Ok(text) = std::str::from_utf8(contents) else {
return contents.to_vec();
};
let has_newline = text.ends_with('\n');
let body = text.strip_suffix('\n').unwrap_or(text);
let lines = body.split('\n').collect::<Vec<_>>();
if lines.len() != LABELS.len() {
return contents.to_vec();
}
for (line, expected_label) in lines.iter().zip(LABELS) {
let mut fields = line.split_whitespace();
let (Some(label), Some(value), None) = (fields.next(), fields.next(), fields.next()) else {
return contents.to_vec();
};
if label != *expected_label || value.parse::<u64>().is_err() {
return contents.to_vec();
}
}
let mut normalized = LABELS
.iter()
.map(|label| format!("{label} 0"))
.collect::<Vec<_>>()
.join("\n")
.into_bytes();
if has_newline {
normalized.push(b'\n');
}
normalized
}
#[cfg(test)]
mod tests {
use super::*;
#[test]
fn recognizes_only_normalized_procfs_paths() {
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/stat"))
.unwrap()
.kind,
ProcfsKind::Stat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/status"))
.unwrap()
.kind,
ProcfsKind::Status
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/thread-self/stat"))
.unwrap()
.kind,
ProcfsKind::ThreadStat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/thread-self/status"))
.unwrap()
.kind,
ProcfsKind::ThreadStatus
);
for (path, kind) in [
("/proc/123/stat", ProcfsKind::ProcessStat),
("/proc/self/statm", ProcfsKind::Statm),
("/proc/123/statm", ProcfsKind::Statm),
("/proc/123/status", ProcfsKind::ProcessStatus),
("/proc/stat", ProcfsKind::SystemStat),
] {
assert_eq!(ProcfsFile::from_path(Path::new(path)).unwrap().kind, kind);
}
assert!(ProcfsFile::from_path(Path::new("/proc/not-a-pid/stat")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/cpuinfo"))
.unwrap()
.kind,
ProcfsKind::Cpuinfo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/diskstats"))
.unwrap()
.kind,
ProcfsKind::Diskstats
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/loadavg"))
.unwrap()
.kind,
ProcfsKind::Loadavg
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/uptime"))
.unwrap()
.kind,
ProcfsKind::Uptime
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/meminfo"))
.unwrap()
.kind,
ProcfsKind::Meminfo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/net/sockstat"))
.unwrap()
.kind,
ProcfsKind::Sockstat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/vmstat"))
.unwrap()
.kind,
ProcfsKind::Vmstat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/block/md0/inflight"))
.unwrap()
.kind,
ProcfsKind::BlockInflight
);
assert!(ProcfsFile::from_path(Path::new("/sys/block/md0/size")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/class/block/md0/inflight")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/kernel/uevent_seqnum"))
.unwrap()
.kind,
ProcfsKind::UeventSeqnum
);
assert!(ProcfsFile::from_path(Path::new("/sys/kernel/uevent_helper")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/kernel/irq/254/per_cpu_count"))
.unwrap()
.kind,
ProcfsKind::IrqPerCpuCount
);
assert!(ProcfsFile::from_path(Path::new("/sys/kernel/irq/irq254/per_cpu_count")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/kernel/irq/254/actions")).is_none());
assert!(
ProcfsFile::from_path(Path::new("/sys/kernel/irq/254/device/per_cpu_count")).is_none()
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/kernel/pty/nr"))
.unwrap()
.kind,
ProcfsKind::PtyNr
);
for path in [
"/proc/self/sched",
"/proc/thread-self/sched",
"/proc/123/sched",
"/proc/self/task/456/sched",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::SelfSched
);
}
for path in [
"/proc/self/fdinfo/17",
"/proc/thread-self/fdinfo/17",
"/proc/123/fdinfo/17",
"/proc/self/task/456/fdinfo/17",
] {
let procfs = ProcfsFile::from_path(Path::new(path)).unwrap();
assert_eq!(procfs.kind, ProcfsKind::Fdinfo);
assert_eq!(procfs.target_fd(), Some(17));
}
assert!(ProcfsFile::from_path(Path::new("/proc/self/fdinfo/")).is_none());
assert!(ProcfsFile::from_path(Path::new("/proc/self/fdinfo/stdin")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/aio-nr"))
.unwrap()
.kind,
ProcfsKind::AioNr
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/aio-max-nr"))
.unwrap()
.kind,
ProcfsKind::AioMaxNr
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/numa_maps"))
.unwrap()
.kind,
ProcfsKind::NumaMaps
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/smaps_rollup"))
.unwrap()
.kind,
ProcfsKind::SmapsRollup
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/arch_status"))
.unwrap()
.kind,
ProcfsKind::ArchStatus
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/swaps"))
.unwrap()
.kind,
ProcfsKind::Swaps
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/key-users"))
.unwrap()
.kind,
ProcfsKind::KeyUsers
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/123/io"))
.unwrap()
.kind,
ProcfsKind::ProcessIo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/block/nvme0n1/stat"))
.unwrap()
.kind,
ProcfsKind::BlockStat
);
assert!(ProcfsFile::from_path(Path::new("/proc/not-a-pid/io")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/block/nvme0n1/size")).is_none());
for path in [
"/proc/pressure/cpu",
"/proc/pressure/io",
"/proc/pressure/memory",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::Pressure
);
}
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/buddyinfo"))
.unwrap()
.kind,
ProcfsKind::Buddyinfo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/schedstat"))
.unwrap()
.kind,
ProcfsKind::Schedstat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/net/softnet_stat"))
.unwrap()
.kind,
ProcfsKind::SoftnetStat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/file-nr"))
.unwrap()
.kind,
ProcfsKind::FileNr
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/file-max"))
.unwrap()
.kind,
ProcfsKind::FileMax
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/zoneinfo"))
.unwrap()
.kind,
ProcfsKind::Zoneinfo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/inode-nr"))
.unwrap()
.kind,
ProcfsKind::InodeNr
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/inode-state"))
.unwrap()
.kind,
ProcfsKind::InodeState
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/net/protocols"))
.unwrap()
.kind,
ProcfsKind::Protocols
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/locks"))
.unwrap()
.kind,
ProcfsKind::Locks
);
for alias in ["/proc/./locks", "/proc/self/../locks"] {
assert_eq!(
ProcfsFile::from_path(Path::new(alias)).unwrap().kind,
ProcfsKind::Locks
);
}
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/smaps"))
.unwrap()
.kind,
ProcfsKind::Smaps
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/driver/rtc"))
.unwrap()
.kind,
ProcfsKind::Rtc
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/sys/fs/dentry-state"))
.unwrap()
.kind,
ProcfsKind::DentryState
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/net/netlink"))
.unwrap()
.kind,
ProcfsKind::NetlinkSockets
);
assert!(ProcfsFile::from_path(Path::new("/proc/net/packet")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new(
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/commit_stats"
))
.unwrap()
.kind,
ProcfsKind::BtrfsCommitStats
);
assert!(
ProcfsFile::from_path(Path::new(
"/sys/fs/btrfs/004B7924-9DF8-4EC2-AEA0-D9775554E1BA/commit_stats"
))
.is_none()
);
assert!(
ProcfsFile::from_path(Path::new(
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/generation"
))
.is_none()
);
assert!(
ProcfsFile::from_path(Path::new(
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/nested/commit_stats"
))
.is_none()
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/net/unix"))
.unwrap()
.kind,
ProcfsKind::UnixSockets
);
assert!(ProcfsFile::from_path(Path::new("/proc/net/packet")).is_none());
for path in [
"/proc/self/schedstat",
"/proc/thread-self/schedstat",
"/proc/123/schedstat",
"/proc/self/task/456/schedstat",
"/proc/123/task/456/schedstat",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::SelfSchedstat
);
}
assert!(ProcfsFile::from_path(Path::new("/proc/self/task/nope/schedstat")).is_none());
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/maps"))
.unwrap()
.kind,
ProcfsKind::Maps
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/smaps"))
.unwrap()
.kind,
ProcfsKind::Smaps
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/self/numa_maps"))
.unwrap()
.kind,
ProcfsKind::NumaMaps
);
for path in [
"/proc/self/maps",
"/proc/123/maps",
"/proc/123/task/456/maps",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::Maps
);
}
}
#[test]
fn maps_rewrites_both_the_device_and_the_inode() {
let raw =
b"7f0000000000-7f0000001000 r-xp 00000000 08:02 1234567 /lib/libc.so.6\n" as &[u8];
let table = BTreeMap::from([(
(libc::makedev(0x08, 0x02), 1_234_567u64),
(libc::makedev(0x00, 0x2a), 99u64),
)]);
let out = String::from_utf8(sanitize_maps(raw, &table)).unwrap();
assert!(out.contains(" 00:2a 99 "), "both columns rewritten: {out}");
assert!(!out.contains("08:02"), "raw device must not survive: {out}");
assert!(
!out.contains("1234567"),
"raw inode must not survive: {out}"
);
assert!(
out.ends_with("/lib/libc.so.6\n"),
"pathname preserved: {out}"
);
}
#[test]
fn maps_padding_does_not_republish_the_host_inode_width() {
const HEAD: &str = "70f80000-71000000 r-xs 00000000 00:06 ";
const NAME: &str = "/memfd:liteinst2-trampoline (deleted)";
const PATHNAME_COLUMN: usize = 73;
let kernel_line = |inode: u64| {
let prefix = format!("{HEAD}{inode}");
let pad = PATHNAME_COLUMN - prefix.len();
format!("{prefix}{}{NAME}\n", " ".repeat(pad))
};
let table = |inode: u64| {
BTreeMap::from([(
(libc::makedev(0x00, 0x06), inode),
(libc::makedev(0x00, 0x06), 8u64),
)])
};
let four_digit = sanitize_maps(kernel_line(9_999).as_bytes(), &table(9_999));
let five_digit = sanitize_maps(kernel_line(10_000).as_bytes(), &table(10_000));
assert_eq!(
String::from_utf8(four_digit.clone()).unwrap(),
String::from_utf8(five_digit).unwrap(),
"the host inode's digit count must not survive as line length",
);
let out = String::from_utf8(four_digit).unwrap();
let body = out.strip_suffix('\n').unwrap();
assert_eq!(
body.find('/'),
Some(PATHNAME_COLUMN),
"pathname column preserved: {out:?}",
);
assert!(body.ends_with(NAME), "pathname preserved: {out:?}");
assert_eq!(
body.split_whitespace().nth(4),
Some("8"),
"raw inode must not survive in the inode column: {out:?}",
);
}
#[test]
fn maps_leaves_anonymous_mappings_alone() {
let raw = b"7ffd00000000-7ffd00021000 rw-p 00000000 00:00 0 [stack]\n" as &[u8];
assert_eq!(
mapping_header_identity(std::str::from_utf8(raw).unwrap().trim_end()),
None
);
assert_eq!(sanitize_maps(raw, &BTreeMap::new()), raw.to_vec());
}
#[test]
fn maps_never_invents_an_identity_for_an_untabled_mapping() {
let raw =
b"7f0000000000-7f0000001000 r-xp 00000000 08:02 1234567 /lib/libc.so.6\n" as &[u8];
assert_eq!(sanitize_maps(raw, &BTreeMap::new()), raw.to_vec());
}
#[test]
fn smaps_header_is_rewritten_like_maps() {
let raw = b"7f0000000000-7f0000001000 r-xp 00000000 08:02 1234567 /lib/libc.so.6\n\
Size: 4 kB\n\
Rss: 4 kB\n" as &[u8];
let table = BTreeMap::from([(
(libc::makedev(0x08, 0x02), 1_234_567u64),
(libc::makedev(0x00, 0x2a), 99u64),
)]);
let out = String::from_utf8(sanitize_smaps(raw, &table)).unwrap();
assert!(out.contains(" 00:2a 99 "), "smaps header rewritten: {out}");
assert!(
!out.contains("1234567"),
"raw inode must not survive smaps: {out}"
);
}
#[test]
fn timer_slack_recognizes_only_linux_top_level_spellings() {
let mut self_file = ProcfsFile::from_path(Path::new("/proc/self/timerslack_ns")).unwrap();
assert_eq!(self_file.kind, ProcfsKind::TimerSlack(None));
assert!(self_file.needs_bound_thread_identity());
self_file.bind_thread_identity(101, 202, 303);
assert_eq!(self_file.timer_slack_target(), Some(101));
let tid_file = ProcfsFile::from_path(Path::new("/proc/202/timerslack_ns")).unwrap();
assert_eq!(tid_file.kind, ProcfsKind::TimerSlack(Some(202)));
assert!(!tid_file.needs_bound_thread_identity());
for path in [
"/proc/thread-self/timerslack_ns",
"/proc/self/task/202/timerslack_ns",
"/proc/101/task/202/timerslack_ns",
"/proc/0/timerslack_ns",
"/proc/-1/timerslack_ns",
"/proc/not-a-tid/timerslack_ns",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none(), "{path}");
}
}
#[test]
fn timer_slack_snapshot_rewinds_and_positioned_reads_are_fresh() {
let mut procfs = ProcfsFile::from_path(Path::new("/proc/202/timerslack_ns")).unwrap();
let failed = procfs.preview_timer_slack(111, 2).unwrap();
assert_eq!(failed.bytes, b"11");
assert_eq!(procfs.position(), (0, None));
let first = procfs.preview_timer_slack(222, 2).unwrap();
assert_eq!(first.bytes, b"22");
procfs.commit_timer_slack_read(&first, first.bytes.len());
assert_eq!(
procfs.preview_timer_slack(999, usize::MAX).unwrap().bytes,
b"2\n",
"a partial sequential read retains its original scalar"
);
procfs.set_offset(0);
let rewound = procfs.preview_timer_slack(333, usize::MAX).unwrap();
assert_eq!(rewound.bytes, b"333\n");
procfs.commit_timer_slack_read(&rewound, rewound.bytes.len());
assert_eq!(procfs.take_timer_slack_at(333, 1, 2).unwrap(), b"33");
assert_eq!(
procfs.preview_timer_slack(444, usize::MAX).unwrap().bytes,
b"",
"positioned reads do not change the shared cursor"
);
}
#[test]
fn recognizes_coherent_cpufreq_policy_paths() {
for path in [
"/sys/devices/system/cpu/cpu3/cpufreq/cpuinfo_cur_freq",
"/sys/devices/system/cpu/cpu3/cpufreq/cpuinfo_avg_freq",
"/sys/devices/system/cpu/cpu3/cpufreq/scaling_min_freq",
"/sys/devices/system/cpu/cpufreq/policy3/cpuinfo_avg_freq",
"/sys/devices/system/cpu/cpufreq/policy3/cpuinfo_max_freq",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::ScalingCurFreq
);
}
assert!(
ProcfsFile::from_path(Path::new("/tmp/cpufreq/policy3/cpuinfo_avg_freq")).is_none()
);
}
#[test]
fn recognizes_mount_and_random_uuid_paths() {
for (path, kind) in [
("/proc/self/mountinfo", ProcfsKind::Mountinfo),
("/proc/thread-self/mountinfo", ProcfsKind::Mountinfo),
("/proc/37/mountinfo", ProcfsKind::Mountinfo),
("/proc/self/task/38/mountinfo", ProcfsKind::Mountinfo),
("/proc/37/task/38/mountinfo", ProcfsKind::Mountinfo),
("/proc/sys/kernel/random/uuid", ProcfsKind::RandomUuid),
] {
assert_eq!(ProcfsFile::from_path(Path::new(path)).unwrap().kind, kind);
}
for path in [
"/proc/task/mountinfo",
"/proc/37/task/self/mountinfo",
"/proc/37/task/38/mountinfo/extra",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
fn mountinfo_snapshot(contents: &[u8], devices: BTreeMap<u64, u64>) -> MountInfoSnapshot {
mountinfo_snapshot_with_policy(contents, true, devices, BTreeMap::new())
}
fn mountinfo_snapshot_with_rewrites(
contents: &[u8],
devices: BTreeMap<u64, u64>,
root_rewrites: BTreeMap<u64, Vec<u8>>,
) -> MountInfoSnapshot {
let rewrites = root_rewrites
.into_iter()
.map(|(raw_mount_id, deterministic_root)| {
(
raw_mount_id,
MountInfoRootRewrite {
raw_mount_id,
deterministic_root,
raw_root_prefix: None,
deterministic_root_prefix: None,
raw_mountpoint_prefix: None,
deterministic_mountpoint_prefix: None,
},
)
})
.collect();
mountinfo_snapshot_with_policy(contents, true, devices, rewrites)
}
fn mountinfo_snapshot_with_policy(
contents: &[u8],
virtualize_metadata: bool,
devices: BTreeMap<u64, u64>,
root_rewrites: BTreeMap<u64, MountInfoRootRewrite>,
) -> MountInfoSnapshot {
let rows = parse_mountinfo(contents).unwrap();
MountInfoSnapshot::new(rows, &[], virtualize_metadata, devices, root_rewrites).unwrap()
}
#[test]
fn only_proven_private_mount_roots_are_guest_stable() {
let input = b"37 29 0:31 /tmpvol/.tmpAb12Z9 /tmp rw - btrfs /dev/md0 rw\n38 29 0:31 /host/data /data ro - btrfs /dev/md0 ro\n39 29 0:31 /tmp/.tmp654321 /etc/group ro - btrfs /dev/md0 ro\n40 29 0:31 /arbitrary/host/tmpdir/.tmpxZruR5 /run/nscd ro - btrfs /dev/md0 rw\n41 29 0:31 /another/place/.tmp123abc /etc/group ro - btrfs /dev/md0 rw\n42 29 0:31 /stacked/.tmpABC123 /tmp rw - btrfs /dev/md0 rw\n";
let device = libc::makedev(0, 31);
let snapshot = mountinfo_snapshot_with_rewrites(
input,
BTreeMap::from([(device, device)]),
BTreeMap::from([
(37, b"/tmpvol/.hermit/tmp".to_vec()),
(39, b"/tmpvol/.hermit/etc/group".to_vec()),
(40, b"/tmpvol/.hermit/run/nscd".to_vec()),
]),
);
assert_eq!(
sanitize_mountinfo(input, &snapshot),
b"1 7 0:31 /tmpvol/.hermit/tmp /tmp rw - btrfs /dev/md0 rw\n2 7 0:31 /host/data /data ro - btrfs /dev/md0 ro\n3 7 0:31 /tmpvol/.hermit/etc/group /etc/group ro - btrfs /dev/md0 ro\n4 7 0:31 /tmpvol/.hermit/run/nscd /run/nscd ro - btrfs /dev/md0 rw\n5 7 0:31 /another/place/.tmp123abc /etc/group ro - btrfs /dev/md0 rw\n6 7 0:31 /stacked/.tmpABC123 /tmp rw - btrfs /dev/md0 rw\n"
);
}
#[test]
fn proven_private_tmp_canonicalizes_its_root_and_descendant_mountpoints_only() {
let input = b"37 29 0:31 /host/.tmpABC123 /tmp rw - tmpfs tmpfs rw\n\
38 37 0:32 / /host/.tmpABC123/user-mount rw - tmpfs tmpfs rw\n\
39 29 0:33 / /host/.tmpABC123-lookalike rw - tmpfs tmpfs rw\n\
40 37 0:34 /host/.tmpABC123/child /data rw - tmpfs tmpfs rw\n";
let devices = BTreeMap::from([
(libc::makedev(0, 31), 1),
(libc::makedev(0, 32), 2),
(libc::makedev(0, 33), 3),
(libc::makedev(0, 34), 4),
]);
let rewrite = MountInfoRootRewrite {
raw_mount_id: 37,
deterministic_root: b"/tmpvol/.hermit/tmp".to_vec(),
raw_root_prefix: Some(b"/host/.tmpABC123".to_vec()),
deterministic_root_prefix: Some(b"/tmpvol/.hermit/tmp".to_vec()),
raw_mountpoint_prefix: Some(b"/host/.tmpABC123".to_vec()),
deterministic_mountpoint_prefix: Some(b"/tmp".to_vec()),
};
let snapshot =
mountinfo_snapshot_with_policy(input, true, devices, BTreeMap::from([(37, rewrite)]));
assert_eq!(
sanitize_mountinfo(input, &snapshot),
b"1 5 0:1 /tmpvol/.hermit/tmp /tmp rw - tmpfs tmpfs rw\n\
2 1 0:2 / /tmp/user-mount rw - tmpfs tmpfs rw\n\
3 5 0:3 / /host/.tmpABC123-lookalike rw - tmpfs tmpfs rw\n\
4 1 0:4 /tmpvol/.hermit/tmp/child /data rw - tmpfs tmpfs rw\n"
);
}
#[test]
fn unrelated_mount_roots_are_preserved() {
let device = libc::makedev(0, 31);
for (root, mountpoint) in [
("/tmpvol/build/not-a-tempfile", "/tmp"),
("/tmpvol/build/.tmpAb12Z!", "/tmp"),
("/tmpvol/build/.tmpAb12Z9/child", "/tmp"),
("/tmpvol/build/.tmpAb12Z9", "/data"),
] {
let input = format!("37 29 0:31 {root} {mountpoint} rw - btrfs /dev/md0 rw\n");
let snapshot = mountinfo_snapshot(input.as_bytes(), BTreeMap::from([(device, device)]));
assert_eq!(
sanitize_mountinfo(input.as_bytes(), &snapshot),
format!("1 2 0:31 {root} {mountpoint} rw - btrfs /dev/md0 rw\n").as_bytes()
);
}
}
#[test]
fn mountinfo_identities_are_stable_with_a_forward_parent_reference() {
let first = b"20 10 259:5 / /child rw shared:1 - ext4 /dev/a rw\n10 1 8:1 / / rw - ext4 /dev/b rw\n";
let second = b"90 80 0:44 / /child rw shared:99 - ext4 /dev/a rw\n80 7 0:33 / / rw - ext4 /dev/b rw\n";
let first_snapshot = mountinfo_snapshot(
first,
BTreeMap::from([
(libc::makedev(259, 5), libc::makedev(0, 1)),
(libc::makedev(8, 1), libc::makedev(0, 2)),
]),
);
let second_snapshot = mountinfo_snapshot(
second,
BTreeMap::from([
(libc::makedev(0, 44), libc::makedev(0, 1)),
(libc::makedev(0, 33), libc::makedev(0, 2)),
]),
);
let expected =
b"1 2 0:1 / /child rw shared:1 - ext4 /dev/a rw\n2 3 0:2 / / rw - ext4 /dev/b rw\n";
assert_eq!(sanitize_mountinfo(first, &first_snapshot), expected);
assert_eq!(sanitize_mountinfo(second, &second_snapshot), expected);
}
#[test]
fn mountinfo_device_column_follows_the_metadata_virtualization_policy() {
let input = b"20 10 259:5 / /child rw - ext4 /dev/a rw\n";
let raw_device = libc::makedev(259, 5);
let deterministic_device = libc::makedev(0, 7);
let virtualized = mountinfo_snapshot_with_policy(
input,
true,
BTreeMap::from([(raw_device, deterministic_device)]),
BTreeMap::new(),
);
assert_eq!(
sanitize_mountinfo(input, &virtualized),
b"1 2 0:7 / /child rw - ext4 /dev/a rw\n"
);
let native = mountinfo_snapshot_with_policy(input, false, BTreeMap::new(), BTreeMap::new());
assert_eq!(
sanitize_mountinfo(input, &native),
b"1 2 259:5 / /child rw - ext4 /dev/a rw\n"
);
}
#[test]
fn mountinfo_topology_preserves_self_and_unknown_parents() {
let input = b"10 10 8:1 / / rw shared:44 master:55 propagate_from:66 unbindable - ext4 /dev/a rw\n20 999 8:2 / /child rw - ext4 /dev/b rw\n";
let snapshot = mountinfo_snapshot(
input,
BTreeMap::from([
(libc::makedev(8, 1), libc::makedev(0, 1)),
(libc::makedev(8, 2), libc::makedev(0, 2)),
]),
);
assert_eq!(
sanitize_mountinfo(input, &snapshot),
b"1 1 0:1 / / rw shared:1 master:2 propagate_from:3 unbindable - ext4 /dev/a rw\n2 3 0:2 / /child rw - ext4 /dev/b rw\n"
);
}
#[test]
fn mountinfo_parser_accepts_non_utf8_and_literal_control_bytes() {
let input = b"20 10 259:5 /\xff\x0broot /child rw shared:44 - ext4 /dev/\xfe rw\n";
let snapshot = mountinfo_snapshot(
input,
BTreeMap::from([(libc::makedev(259, 5), libc::makedev(0, 1))]),
);
assert_eq!(
sanitize_mountinfo(input, &snapshot),
b"1 2 0:1 /\xff\x0broot /child rw shared:1 - ext4 /dev/\xfe rw\n"
);
}
#[test]
fn one_malformed_mountinfo_row_rejects_the_snapshot() {
for input in [
b"20 10 259:5 / /child rw - ext4 /dev/a rw\n37 29 0:31 / / rw\n".as_slice(),
b"20 10 259:5 / /child rw - - ext4 /dev/a rw\n".as_slice(),
b"mount 10 259:5 / /child rw - ext4 /dev/a rw\n".as_slice(),
b"20 10 4294967296:5 / /child rw - ext4 /dev/a rw\n".as_slice(),
b"20 10 259:5 / /child rw shared:not-a-number - ext4 /dev/a rw\n".as_slice(),
b"20 10 259:5 / /child rw future_peer:77 - ext4 /dev/a rw\n".as_slice(),
] {
assert!(
parse_mountinfo(input).is_none(),
"malformed mountinfo was accepted: {}",
String::from_utf8_lossy(input)
);
}
}
#[test]
fn empty_mountinfo_snapshot_renders_empty_for_process_and_task_aliases() {
let rows = parse_mountinfo(b"").expect("empty mountinfo is a valid empty snapshot");
let snapshot = MountInfoSnapshot::new(rows, &[], false, BTreeMap::new(), BTreeMap::new())
.expect("empty mountinfo identity snapshot");
for path in ["/proc/37/mountinfo", "/proc/37/task/38/mountinfo"] {
let mut file = ProcfsFile::from_path(Path::new(path))
.unwrap_or_else(|| panic!("{path} must classify as mountinfo"));
file.initialize(
Vec::new(),
ProcfsSnapshotContext {
mountinfo: Some(snapshot.clone()),
..ProcfsSnapshotContext::default()
},
);
assert_eq!(file.take(usize::MAX), Some(Vec::new()), "{path}");
}
}
#[test]
fn duplicate_mount_id_or_unknown_proven_id_rejects_the_snapshot() {
let duplicate = b"20 10 8:1 / /a rw - ext4 /dev/a rw\n20 10 8:1 / /b rw - ext4 /dev/a rw\n";
assert!(parse_mountinfo(duplicate).is_none());
let ordinary = b"20 10 8:1 / /a rw - ext4 /dev/a rw\n";
let rows = parse_mountinfo(ordinary).unwrap();
assert!(
MountInfoSnapshot::new(
rows,
&[],
false,
BTreeMap::new(),
BTreeMap::from([(
99,
MountInfoRootRewrite {
raw_mount_id: 99,
deterministic_root: b"/deterministic".to_vec(),
raw_root_prefix: None,
deterministic_root_prefix: None,
raw_mountpoint_prefix: None,
deterministic_mountpoint_prefix: None,
},
)]),
)
.is_none()
);
let rows = parse_mountinfo(ordinary).unwrap();
assert!(
MountInfoSnapshot::new(
rows,
&[],
false,
BTreeMap::new(),
BTreeMap::from([(
20,
MountInfoRootRewrite {
raw_mount_id: 20,
deterministic_root: b"/invalid root".to_vec(),
raw_root_prefix: None,
deterministic_root_prefix: None,
raw_mountpoint_prefix: None,
deterministic_mountpoint_prefix: None,
},
)]),
)
.is_none()
);
}
#[test]
fn mountinfo_snapshot_accepts_only_known_ordered_subsets_and_matching_device_policy() {
let input = b"20 10 8:1 / /a rw - ext4 /dev/a rw\n30 20 8:2 / /b rw - ext4 /dev/b rw\n";
let rows = parse_mountinfo(input).unwrap();
let first = libc::makedev(8, 1);
assert!(
MountInfoSnapshot::new(
rows.clone(),
&[],
true,
BTreeMap::from([(first, libc::makedev(0, 1))]),
BTreeMap::new(),
)
.is_none()
);
assert!(
MountInfoSnapshot::new(
rows.clone(),
&[],
false,
BTreeMap::from([(first, first)]),
BTreeMap::new(),
)
.is_none()
);
assert!(
MountInfoSnapshot::new(
rows.clone(),
&[20, 30],
false,
BTreeMap::new(),
BTreeMap::new(),
)
.is_none(),
"recorded mount identity order omitted parent 10"
);
assert!(
MountInfoSnapshot::new(
rows,
&[20, 30, 10, 10],
false,
BTreeMap::new(),
BTreeMap::new(),
)
.is_none(),
"recorded mount identity order contained a duplicate"
);
let rows = parse_mountinfo(input).unwrap();
assert!(
MountInfoSnapshot::new(
rows.clone(),
&[30, 20, 10],
false,
BTreeMap::new(),
BTreeMap::new(),
)
.is_none(),
"visible mount rows must retain their producer-captured order"
);
let snapshot = MountInfoSnapshot::new(
rows,
&[5, 20, 25, 30, 10, 999],
false,
BTreeMap::new(),
BTreeMap::from([(
5,
MountInfoRootRewrite {
raw_mount_id: 5,
deterministic_root: b"/invisible-but-captured".to_vec(),
raw_root_prefix: None,
deterministic_root_prefix: None,
raw_mountpoint_prefix: None,
deterministic_mountpoint_prefix: None,
},
)]),
)
.expect("a chroot-visible subset should retain captured canonical positions");
assert_eq!(snapshot.canonical_mount_id(20), Some(2));
assert_eq!(snapshot.canonical_mount_id(30), Some(4));
assert_eq!(snapshot.canonical_mount_id(10), Some(5));
assert_eq!(snapshot.raw_mount_id_order(), [20, 30, 10]);
assert!(!snapshot.root_rewrites.contains_key(&5));
}
#[test]
fn random_uuid_uses_deterministic_v4_bytes_or_fails_open() {
let kernel_uuid = b"24e63f35-232a-43e2-8799-b151e9833f45\n";
assert_eq!(
sanitize_random_uuid(kernel_uuid, [0; 16]),
b"00000000-0000-4000-8000-000000000000\n"
);
assert_eq!(
sanitize_random_uuid(kernel_uuid, [0xff; 16]),
b"ffffffff-ffff-4fff-bfff-ffffffffffff\n"
);
for malformed in [
b"24e63f35-232a-1e32-8799-b151e9833f45\n".as_slice(),
b"24e63f35-232a-4e32-c799-b151e9833f45\n",
b"24E63F35-232A-4E32-8799-B151E9833F45\n",
b"24e63f35-232a-4e32-8799-b151e9833f45",
b"not-a-uuid\n",
] {
assert_eq!(sanitize_random_uuid(malformed, [0; 16]), malformed);
}
}
#[test]
fn recognizes_only_btrfs_reserved_byte_gauges() {
const UUID: &str = "004b7924-9df8-4ec2-aea0-d9775554e1ba";
for class in ["data", "metadata", "system"] {
let path = format!("/sys/fs/btrfs/{UUID}/allocation/{class}/bytes_reserved");
assert_eq!(
ProcfsFile::from_path(Path::new(&path)).unwrap().kind,
ProcfsKind::BtrfsBytesReserved
);
}
for path in [
"/sys/fs/btrfs/004B7924-9df8-4ec2-aea0-d9775554e1ba/allocation/data/bytes_reserved",
"/sys/fs/btrfs/not-a-uuid/allocation/data/bytes_reserved",
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/allocation/global/bytes_reserved",
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/allocation/data/bytes_used",
"/sys/fs/btrfs/004b7924-9df8-4ec2-aea0-d9775554e1ba/allocation/data/nested/bytes_reserved",
"/tmp/004b7924-9df8-4ec2-aea0-d9775554e1ba/allocation/data/bytes_reserved",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none(), "{path}");
}
}
#[test]
fn recognizes_only_numa_node_vmstat_paths() {
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/devices/system/node/node0/vmstat"))
.unwrap()
.kind,
ProcfsKind::NodeVmstat
);
assert_eq!(
ProcfsFile::from_path(Path::new("node12/vmstat"))
.unwrap()
.kind,
ProcfsKind::NodeVmstat
);
assert!(ProcfsFile::from_path(Path::new("node/vmstat")).is_none());
assert!(ProcfsFile::from_path(Path::new("node0/meminfo")).is_none());
}
#[test]
fn node_vmstat_preserves_fields_and_zeros_accounting() {
let contents = b"nr_free_pages 3859604\nnuma_hit 197122344154\nnr_writeback 38\n";
assert_eq!(
sanitize_node_vmstat(contents),
b"nr_free_pages 0\nnuma_hit 0\nnr_writeback 0\n"
);
assert!(sanitize_node_vmstat(b"").is_empty());
assert_eq!(
sanitize_node_vmstat(b"nr_free_pages unknown\n"),
b"nr_free_pages unknown\n"
);
assert_eq!(
sanitize_node_vmstat(b"nr_free_pages 1 extra\n"),
b"nr_free_pages 1 extra\n"
);
}
#[test]
fn recognizes_numa_and_hwmon_accounting_paths() {
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/devices/system/node/node0/numastat"))
.unwrap()
.kind,
ProcfsKind::NodeNumastat
);
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/devices/system/node/node12/meminfo"))
.unwrap()
.kind,
ProcfsKind::NodeMeminfo
);
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/class/hwmon/hwmon4/power1_input"))
.unwrap()
.kind,
ProcfsKind::HwmonInput
);
assert!(
ProcfsFile::from_path(Path::new("/sys/devices/system/node/nodeX/numastat")).is_none()
);
assert!(ProcfsFile::from_path(Path::new("/sys/class/hwmon/hwmon4/power1_cap")).is_none());
}
#[test]
fn scaling_cur_freq_is_fixed() {
assert_eq!(sanitize_scaling_cur_freq(b"2483951\n"), b"1000000\n");
assert!(sanitize_scaling_cur_freq(b"").is_empty());
}
#[test]
fn swaps_preserves_configuration_and_zeros_usage() {
let contents = b"Filename\tType\tSize\tUsed\tPriority\n\
/dev/nvme1n1p3 partition 2000892 0 5\n\
/data/swapvol/swapfile file 134217724 69308912 -2\n";
assert_eq!(
sanitize_swaps(contents),
b"Filename\tType\tSize\tUsed\tPriority\n\
/dev/nvme1n1p3\tpartition\t2000892\t0\t5\n\
/data/swapvol/swapfile\tfile\t134217724\t0\t-2\n"
);
assert!(sanitize_swaps(b"").is_empty());
assert_eq!(
sanitize_swaps(b"Filename Type Size Used Priority\n/swap file 12 3 1 extra\n"),
b"Filename Type Size Used Priority\n/swap file 12 3 1 extra\n"
);
assert_eq!(
sanitize_swaps(b"Filename Type Size Used Priority\n/swap file 12 unknown\n"),
b"Filename Type Size Used Priority\n/swap file 12 unknown\n"
);
}
#[test]
fn key_users_preserves_uids_and_quota_limits() {
let contents = b"0: 15 15/15 15/1000000 2499/25000000\n\
1000: 2 1/2 1/200 77/20000\n";
assert_eq!(
sanitize_key_users(contents),
b"0: 0 0/0 0/1000000 0/25000000\n\
1000: 0 0/0 0/200 0/20000\n"
);
assert!(sanitize_key_users(b"").is_empty());
assert_eq!(
sanitize_key_users(b"0: 15 15/15 invalid 2499/25000000\n"),
b"0: 15 15/15 invalid 2499/25000000\n"
);
assert_eq!(
sanitize_key_users(b"0: 15 15/15 15/1000000 2499/25000000 extra\n"),
b"0: 15 15/15 15/1000000 2499/25000000 extra\n"
);
}
#[test]
fn softnet_stat_synthesizes_one_virtual_cpu_row() {
let input = b"00908c97 00000002 0000009e 00000004 00000005 00000006 00000007 00000008 00000009 0000000a 0000000b 0000000c 00000003 0000000e 0000000f\n\
00123456 00000012 00000034 00000056 00000078 0000009a 000000bc 000000de 000000f0 00000011 00000022 00000033 00000007 00000044 00000055\n";
let expected = b"00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000 00000000\n";
assert_eq!(sanitize_softnet_stat(input), expected);
assert_eq!(sanitize_softnet_stat(b"not a softnet row\n"), expected);
}
#[test]
fn btrfs_reserved_bytes_are_fixed() {
assert_eq!(sanitize_btrfs_bytes_reserved(b"164425728\n"), b"0\n");
assert_eq!(sanitize_btrfs_bytes_reserved(b"164425728"), b"0");
assert_eq!(
sanitize_btrfs_bytes_reserved(b"18446744073709551615\n"),
b"0\n"
);
for malformed in [
b"".as_slice(),
b"18446744073709551616\n",
b"-1\n",
b"123 456\n",
b"unknown\n",
] {
assert_eq!(sanitize_btrfs_bytes_reserved(malformed), malformed);
}
}
#[test]
fn recognizes_only_btrfs_pinned_space_gauges() {
const UUID: &str = "63152d54-3f28-408a-80a2-46e53b5c0bda";
for class in ["data", "metadata", "system"] {
let path = format!("/sys/fs/btrfs/{UUID}/allocation/{class}/bytes_pinned");
assert_eq!(
ProcfsFile::from_path(Path::new(&path)).unwrap().kind,
ProcfsKind::BtrfsBytesPinned
);
}
let reserved_path = format!("/sys/fs/btrfs/{UUID}/allocation/data/bytes_reserved");
assert_eq!(
ProcfsFile::from_path(Path::new(&reserved_path))
.unwrap()
.kind,
ProcfsKind::BtrfsBytesReserved
);
for path in [
"/sys/fs/btrfs/63152D54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_pinned",
"/sys/fs/btrfs/not-a-uuid/allocation/data/bytes_pinned",
"/sys/fs/btrfs/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/global/bytes_pinned",
"/sys/fs/btrfs/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_pinned/extra",
"/tmp/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_pinned",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn btrfs_pinned_space_gauge_is_fixed() {
assert_eq!(sanitize_btrfs_bytes_pinned(b"66535424\n"), b"0\n");
assert_eq!(sanitize_btrfs_bytes_pinned(b"66535424"), b"0");
for malformed in [
b"".as_slice(),
b"-1\n",
b"123 456\n",
b"123\n\n",
b"unknown\n",
] {
assert_eq!(sanitize_btrfs_bytes_pinned(malformed), malformed);
}
}
#[test]
fn recognizes_only_dynamic_cpuidle_counters() {
for path in [
"cpu0/cpuidle/state0/time",
"/sys/devices/system/cpu/cpu3/cpuidle/state12/usage",
"cpu0/cpuidle/state0/above",
"cpu0/cpuidle/state0/below",
"cpu0/cpuidle/state0/rejected",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::CpuidleCounter
);
}
for path in [
"cpu0/cpuidle/state0/name",
"cpu0/cpuidle/state0/latency",
"cpu0/cpuidle/state0/residency",
"cpu0/cpuidle/state/usage",
"cpu0/cpuidle/statex/time",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn cpuidle_counter_is_fixed() {
assert_eq!(sanitize_cpuidle_counter(b"42496983978\n"), b"0\n");
assert!(sanitize_cpuidle_counter(b"").is_empty());
}
#[test]
fn recognizes_only_per_size_thp_counters() {
for counter in THP_COUNTERS {
let path = format!("hugepages-2048kB/stats/{counter}");
assert_eq!(
ProcfsFile::from_path(Path::new(&path)).unwrap().kind,
ProcfsKind::ThpCounter
);
}
for path in [
"hugepages-2048kB/enabled",
"hugepages-2048kB/shmem_enabled",
"hugepages-2048kB/stats/unknown",
"hugepages-kB/stats/nr_anon",
"hugepages-2MB/stats/nr_anon",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn thp_counter_is_fixed() {
assert_eq!(sanitize_thp_counter(b"37515411\n"), b"0\n");
assert!(sanitize_thp_counter(b"").is_empty());
}
#[test]
fn recognizes_only_numbered_cpu_cppc_feedback() {
for path in [
"cpu0/acpi_cppc/feedback_ctrs",
"/sys/devices/system/cpu/cpu315/acpi_cppc/feedback_ctrs",
] {
assert_eq!(
ProcfsFile::from_path(Path::new(path)).unwrap().kind,
ProcfsKind::CppcFeedback
);
}
for path in [
"cpu/acpi_cppc/feedback_ctrs",
"gpu0/acpi_cppc/feedback_ctrs",
"cpu0/acpi_cppc/highest_perf",
"cpu0/feedback_ctrs",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn cppc_feedback_counters_are_fixed() {
assert_eq!(
sanitize_cppc_feedback(b"ref:222494767542210 del:411574706774324\n"),
b"ref:0 del:0\n"
);
assert_eq!(sanitize_cppc_feedback(b"ref:123 del:456"), b"ref:0 del:0");
for malformed in [
b"ref:abc del:456\n".as_slice(),
b"ref:123 delivered:456\n",
b"ref:123\n",
b"ref:123 del:456 extra\n",
] {
assert_eq!(sanitize_cppc_feedback(malformed), malformed);
}
}
#[test]
fn recognizes_only_numbered_sysfs_rtc_clock_attributes() {
for (path, kind) in [
("/sys/class/rtc/rtc0/date", ProcfsKind::SysfsRtcDate),
("/sys/class/rtc/rtc12/time", ProcfsKind::SysfsRtcTime),
(
"/sys/class/rtc/rtc315/since_epoch",
ProcfsKind::SysfsRtcEpoch,
),
] {
assert_eq!(ProcfsFile::from_path(Path::new(path)).unwrap().kind, kind);
}
for path in [
"/sys/class/rtc/rtc/date",
"/sys/class/rtc/rtcX/time",
"/sys/class/rtc/rtc0/hctosys",
"/sys/class/rtc/rtc0/device/time",
"/tmp/rtc0/since_epoch",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn sysfs_rtc_uses_the_virtual_realtime() {
assert_eq!(
sanitize_sysfs_rtc_attribute(b"2026-07-27\n", ProcfsKind::SysfsRtcDate, 1_767_225_600,),
b"2026-01-01\n"
);
assert_eq!(
sanitize_sysfs_rtc_attribute(b"12:24:03\n", ProcfsKind::SysfsRtcTime, 1_767_229_261,),
b"01:01:01\n"
);
assert_eq!(
sanitize_sysfs_rtc_attribute(b"1785155071", ProcfsKind::SysfsRtcEpoch, 1_735_689_600,),
b"1735689600"
);
for (malformed, kind) in [
(b"2026/07/27\n".as_slice(), ProcfsKind::SysfsRtcDate),
(b"12:24\n".as_slice(), ProcfsKind::SysfsRtcTime),
(b"-1\n".as_slice(), ProcfsKind::SysfsRtcEpoch),
(b"\n".as_slice(), ProcfsKind::SysfsRtcEpoch),
] {
assert_eq!(
sanitize_sysfs_rtc_attribute(malformed, kind, 1_767_225_600),
malformed
);
}
}
#[test]
fn uevent_seqnum_is_fixed_after_strict_validation() {
assert_eq!(sanitize_uevent_seqnum(b"1282733\n"), b"0\n");
assert_eq!(sanitize_uevent_seqnum(b"1282733"), b"1282733");
assert_eq!(sanitize_uevent_seqnum(b"unknown\n"), b"unknown\n");
assert_eq!(sanitize_uevent_seqnum(b"\n"), b"\n");
}
#[test]
fn recognizes_only_btrfs_reservation_gauges() {
const UUID: &str = "63152d54-3f28-408a-80a2-46e53b5c0bda";
for class in ["data", "metadata", "system"] {
let path = format!("/sys/fs/btrfs/{UUID}/allocation/{class}/bytes_may_use");
assert_eq!(
ProcfsFile::from_path(Path::new(&path)).unwrap().kind,
ProcfsKind::BtrfsBytesMayUse
);
}
for path in [
"/sys/fs/btrfs/63152D54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_may_use",
"/sys/fs/btrfs/not-a-uuid/allocation/data/bytes_may_use",
"/sys/fs/btrfs/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/global/bytes_may_use",
"/sys/fs/btrfs/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_used",
"/tmp/63152d54-3f28-408a-80a2-46e53b5c0bda/allocation/data/bytes_may_use",
] {
assert!(ProcfsFile::from_path(Path::new(path)).is_none());
}
}
#[test]
fn btrfs_reservation_gauge_is_fixed() {
assert_eq!(sanitize_btrfs_bytes_may_use(b"58974208\n"), b"0\n");
assert_eq!(sanitize_btrfs_bytes_may_use(b"58974208"), b"0");
for malformed in [b"".as_slice(), b"-1\n", b"123 456\n", b"unknown\n"] {
assert_eq!(sanitize_btrfs_bytes_may_use(malformed), malformed);
}
}
#[test]
fn numa_and_hwmon_values_are_synthetic() {
assert_eq!(
sanitize_node_numastat(b"numa_hit 123\nnuma_miss 7\n"),
b"numa_hit 0\nnuma_miss 0\n"
);
assert_eq!(
sanitize_node_meminfo(
b"Node 0 MemTotal: 791462432 kB\nNode 0 MemFree: 24654068 kB\nNode 0 Active: 100 kB\nNode 0 HugePages_Total: 2\n"
),
b"Node 0 MemTotal: 1048576 kB\nNode 0 MemFree: 1048576 kB\nNode 0 Active: 0 kB\nNode 0 HugePages_Total: 0\n"
);
assert_eq!(sanitize_numeric_scalar(b"193723000\n"), b"0\n");
assert_eq!(sanitize_numeric_scalar(b"-12000"), b"0");
assert_eq!(
sanitize_numeric_scalar(b"not-a-number\n"),
b"not-a-number\n"
);
}
#[test]
fn stat_normalizes_runtime_counters() {
let input = b"3 (name with spaces) R 1 0 0 0 -1 0 89 0 1 2 3 4 5 6 20 0 1 7 520343512 2879488 123 18446744073709551615 100 200 300 0 0 0 0 3145728 0 0 0 0 17 114 0 0 9 10 11 400 500 600 700 800 900 1000 0\n";
let output = String::from_utf8(sanitize_stat(input, Some((3, 1)))).unwrap();
let comm_end = output.rfind(") ").unwrap();
let fields = output[comm_end + 2..]
.split_whitespace()
.collect::<Vec<_>>();
for field in [
10, 11, 12, 13, 14, 15, 16, 17, 21, 22, 23, 24, 26, 27, 28, 29, 30, 31, 32, 33, 34, 35,
36, 37, 39, 42, 43, 44, 45, 46, 47, 48, 49, 50, 51,
] {
assert_eq!(fields[field - 3], "0", "field {field} was not normalized");
}
assert!(output.starts_with("3 (name with spaces) R 1 0 0 "));
}
#[test]
fn status_normalizes_affinity_and_context_switches() {
let input = b"Name:\tcat\nTgid:\t1234\nPid:\t1234\nPPid:\t1200\nTracerPid:\t0\nNStgid:\t1234\nNSpid:\t1234\nNSpgid:\t1200\nNSsid:\t1190\nSigQ:\t426/2042342\nVmHWM:\t1572 kB\nVmRSS:\t1568 kB\nRssFile:\t1452 kB\nCpus_allowed:\tffffffff,ffffffff\nCpus_allowed_list:\t0-63\nvoluntary_ctxt_switches:\t120\nnonvoluntary_ctxt_switches:\t3\n";
assert_eq!(
sanitize_status(input, Some((3, 3, 1))),
b"Name:\tcat\nTgid:\t3\nPid:\t3\nPPid:\t1\nTracerPid:\t1\nNStgid:\t3\nNSpid:\t3\nNSpgid:\t0\nNSsid:\t0\nSigQ:\t0/0\nVmHWM:\t0 kB\nVmRSS:\t0 kB\nRssFile:\t0 kB\nCpus_allowed:\t00000000,00000000,00000000,00000001\nCpus_allowed_list:\t0\nvoluntary_ctxt_switches:\t0\nnonvoluntary_ctxt_switches:\t0\n"
);
}
#[test]
fn thread_status_uses_the_identity_bound_when_opened() {
let mut file = ProcfsFile::from_path(Path::new("/proc/thread-self/status")).unwrap();
assert!(file.needs_bound_thread_identity());
file.bind_thread_identity(3, 4, 1);
file.initialize(
b"Tgid:\t1234\nPid:\t1235\nPPid:\t1200\nNStgid:\t1234\nNSpid:\t1235\n".to_vec(),
ProcfsSnapshotContext {
virtual_pid: 99,
virtual_ppid: 98,
..ProcfsSnapshotContext::default()
},
);
assert_eq!(
file.take(usize::MAX).unwrap(),
b"Tgid:\t3\nPid:\t4\nPPid:\t1\nNStgid:\t3\nNSpid:\t4\n"
);
}
#[test]
fn process_accounting_preserves_identity_and_hides_live_state() {
let stat = b"42 (worker) R 1 7 8 0 -1 0 89 0 1 2 3 4 5 6 20 0 1 7 520343512 2879488 123 18446744073709551615 100 200 300 0 0 0 0 3145728 0 0 0 0 17 114 0 0 9 10 11 400 500 600 700 800 900 1000 0\n";
let output = String::from_utf8(sanitize_stat(stat, None)).unwrap();
assert!(output.starts_with("42 (worker) S 1 7 8 "));
let comm_end = output.rfind(") ").unwrap();
let fields = output[comm_end + 2..]
.split_whitespace()
.collect::<Vec<_>>();
assert_eq!(fields[23 - 3], "0");
assert_eq!(fields[24 - 3], "0");
assert_eq!(
sanitize_statm(b"62203 7952 5707 4033 0 3255 0\n"),
b"0 0 0 0 0 0 0\n"
);
let status = b"Name:\thermit\nState:\tR (running)\nPid:\t1\nPPid:\t0\nVmSize:\t249000 kB\nVmRSS:\t30000 kB\n";
assert_eq!(
sanitize_status(status, None),
b"Name:\thermit\nState:\tS (sleeping)\nPid:\t1\nPPid:\t0\nVmSize:\t0 kB\nVmRSS:\t0 kB\n"
);
}
#[test]
fn system_stat_uses_virtual_uptime_and_boot_time() {
assert_eq!(
sanitize_system_stat(
b"cpu 1 2 3 4 5 6 7 8 9 10\ncpu0 1 2 3 4 5 6 7 8 9 10\nintr 9 8 7\nbtime 1234\nprocesses 55\n",
120,
1_767_225_480,
),
b"cpu 12000 0 0 0 0 0 0 0 0 0\ncpu0 12000 0 0 0 0 0 0 0 0 0\nintr 0 0 0\nbtime 1767225480\nprocesses 0\n"
);
}
#[test]
fn only_system_stat_needs_the_boot_time() {
for (path, needs) in [
("/proc/stat", true),
("/proc/uptime", false),
("/proc/meminfo", false),
("/proc/self/stat", false),
("/proc/loadavg", false),
] {
let file = ProcfsFile::from_path(Path::new(path)).unwrap();
assert_eq!(file.needs_boot_time(), needs, "{path}");
}
let mut uptime = ProcfsFile::from_path(Path::new("/proc/uptime")).unwrap();
uptime.initialize(
b"1.00 2.00\n".to_vec(),
ProcfsSnapshotContext {
virtual_uptime_seconds: 1 << 63,
virtual_boot_time_seconds: None,
..ProcfsSnapshotContext::default()
},
);
assert_eq!(
uptime.take(usize::MAX).unwrap(),
b"9223372036854775808.00 0.00\n"
);
}
#[test]
fn cpuinfo_normalizes_frequency() {
let input = b"processor\t: 0\ncpu MHz\t\t: 2994.183\ncache size\t: 1024 KB\n";
assert_eq!(
sanitize_cpuinfo(input),
b"processor\t: 0\ncpu MHz\t\t: 1000.000\ncache size\t: 1024 KB\n"
);
}
#[test]
fn io_accounting_counters_use_synthetic_values() {
assert_eq!(
sanitize_diskstats(b"259 0 nvme0n1 100 2 300 4 500 6 700 8 9 10 11\n"),
b"259 0 nvme0n1 1 0 8 0 1 0 8 0 0 0 0\n"
);
assert_eq!(
sanitize_block_stat(b"100 2 300 4 500 6 700 8 9 10 11\n"),
b"1 0 8 0 1 0 8 0 0 0 0\n"
);
assert_eq!(
sanitize_process_io(
b"rchar: 100\nwchar: 200\nsyscr: 3\nsyscw: 4\nread_bytes: 5\nwrite_bytes: 6\ncancelled_write_bytes: 7\n"
),
b"rchar: 0\nwchar: 0\nsyscr: 0\nsyscw: 0\nread_bytes: 0\nwrite_bytes: 0\ncancelled_write_bytes: 0\n"
);
}
#[test]
fn loadavg_and_uptime_use_virtual_values() {
assert_eq!(
sanitize_loadavg(b"344.01 369.71 375.04 526/107858 512196\n"),
b"0.00 0.00 0.00 1/1 1\n"
);
assert_eq!(
sanitize_uptime(b"156980.56 37990755.08\n", 120),
b"120.00 0.00\n"
);
}
#[test]
fn inode_nr_hides_host_global_allocation_counters() {
assert_eq!(sanitize_inode_nr(b"13929543\t1109179\n"), b"0\t0\n");
assert_eq!(
sanitize_inode_state(b"13929543\t1109179\t0\t0\t0\t0\t0\n"),
b"0\t0\t0\t0\t0\t0\t0\n"
);
assert!(sanitize_inode_nr(b"").is_empty());
assert!(sanitize_inode_state(b"").is_empty());
}
#[test]
fn buddyinfo_preserves_topology_and_zeros_free_lists() {
let contents = b"Node 0, zone DMA 0 1 2 3\n\
Node 1, zone Normal 42 17 5 1\n\
malformed buddy row\n";
assert_eq!(
sanitize_buddyinfo(contents),
b"Node 0, zone DMA 0 0 0 0\n\
Node 1, zone Normal 0 0 0 0\n\
malformed buddy row\n"
);
}
#[test]
fn file_nr_hides_host_global_allocations() {
assert_eq!(
sanitize_file_nr(b"245853\t0\t9223372036854775807\n"),
b"0\t0\t9223372036854775807\n"
);
assert_eq!(sanitize_file_max(b"1048576\n"), b"9223372036854775807\n");
assert!(sanitize_file_nr(b"").is_empty());
assert!(sanitize_file_max(b"").is_empty());
}
#[test]
fn dentry_state_hides_host_global_cache_counters() {
assert_eq!(
sanitize_dentry_state(b"1888773\t1374220\t45\t0\t212904\t0\n"),
b"0\t0\t45\t0\t0\t0\n"
);
assert!(sanitize_dentry_state(b"").is_empty());
}
#[test]
fn pipe_max_size_reports_the_enforced_ceiling_not_the_host() {
let expected = format!("{DETERMINISTIC_PIPE_CAPACITY_BYTES}\n").into_bytes();
assert_eq!(sanitize_pipe_max_size(b"1048576\n"), expected);
assert_eq!(sanitize_pipe_max_size(b"65536\n"), expected);
assert_eq!(sanitize_pipe_max_size(b"4096\n"), expected);
}
#[test]
fn pipe_max_size_invents_nothing_from_unparsable_contents() {
assert!(sanitize_pipe_max_size(b"").is_empty());
assert!(sanitize_pipe_max_size(b"not a number\n").is_empty());
assert!(sanitize_pipe_max_size(&[0xff, 0xfe]).is_empty());
}
#[test]
fn aio_nr_hides_host_global_reservations() {
assert_eq!(sanitize_aio_nr(b"3040\n"), b"0\n");
assert!(sanitize_aio_nr(b"").is_empty());
}
#[test]
fn pty_nr_hides_host_global_allocations() {
assert_eq!(sanitize_pty_nr(b"107\n", 2), b"2\n");
assert!(sanitize_pty_nr(b"", 2).is_empty());
}
#[test]
fn sockstat_hides_host_global_allocation_and_memory_counters() {
let contents = b"sockets: used 41\n\
TCP: inuse 3 orphan 2 tw 7 alloc 100 mem 200\n\
UDP: inuse 4 mem 99\n\
RAW: inuse 5\n";
assert_eq!(
sanitize_sockstat(contents),
b"sockets: used 41\n\
TCP: inuse 3 orphan 0 tw 7 alloc 3 mem 0\n\
UDP: inuse 4 mem 0\n\
RAW: inuse 5\n"
);
}
#[test]
fn recognizes_interrupt_and_module_accounting_paths() {
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/interrupts"))
.unwrap()
.kind,
ProcfsKind::InterruptCounters
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/softirqs"))
.unwrap()
.kind,
ProcfsKind::InterruptCounters
);
assert_eq!(
ProcfsFile::from_path(Path::new("/proc/modules"))
.unwrap()
.kind,
ProcfsKind::Modules
);
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/module/ipmi_si/refcnt"))
.unwrap()
.kind,
ProcfsKind::ModuleRefcnt("ipmi_si".to_owned())
);
assert!(ProcfsFile::from_path(Path::new("/proc/devices")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/module/ipmi_si/coresize")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/module/refcnt")).is_none());
assert!(ProcfsFile::from_path(Path::new("/sys/module/ipmi_si/holders/refcnt")).is_none());
}
#[test]
fn interrupt_and_module_accounting_is_synthetic() {
let interrupts = b" CPU0 CPU1\n 9: 123 4 IR-PCI-MSI 0-edge acpi\nNMI: 8 9 Non-maskable interrupts\nERR: 5\n";
assert_eq!(
sanitize_interrupt_counters(interrupts),
b" CPU0 CPU1\n 9: 0 0 IR-PCI-MSI 0-edge acpi\nNMI: 0 0 Non-maskable interrupts\nERR: 0\n"
);
assert_eq!(
sanitize_interrupt_counters(b" 9: 9 99 IR-PCI-MSI 0-edge acpi\nERR: 999999\n"),
sanitize_interrupt_counters(b" 9: 123456789 4 IR-PCI-MSI 0-edge acpi\nERR: 5\n"),
"native counter width and padding must not affect the snapshot"
);
let modules = b"kvm_amd 212992 95 - Live 0x0\nkvm 1200128 1 kvm_amd, Live 0x0\nllc 20480 2 bridge,stp, Live 0x0\n";
assert_eq!(
sanitize_modules(modules),
b"kvm_amd 212992 0 - Live 0x0\nkvm 1200128 1 kvm_amd, Live 0x0\nllc 20480 2 bridge,stp, Live 0x0\n"
);
let source = std::str::from_utf8(modules).unwrap();
assert_eq!(sanitize_module_refcnt(b"7\n", "kvm_amd", source), b"0\n");
assert_eq!(sanitize_module_refcnt(b"12", "kvm_amd", source), b"0");
assert_eq!(sanitize_module_refcnt(b"7\n", "kvm", source), b"1\n");
assert_eq!(sanitize_module_refcnt(b"9", "llc", source), b"2");
assert_eq!(
sanitize_module_refcnt(b"not-a-count\n", "kvm", source),
b"not-a-count\n"
);
assert!(sanitize_module_refcnt(b"", "kvm", source).is_empty());
assert_eq!(sanitize_module_refcnt(b"7\n", "absent", source), b"0\n");
assert_eq!(sanitize_module_refcnt(b"7\n", "kvm", ""), b"0\n");
}
#[test]
fn pressure_hides_host_stall_averages_and_totals() {
let contents = b"some avg10=12.34 avg60=23.45 avg300=34.56 total=123456\n\
full avg10=1.23 avg60=2.34 avg300=3.45 total=654321\n";
assert_eq!(
sanitize_pressure(contents),
b"some avg10=0.00 avg60=0.00 avg300=0.00 total=0\n\
full avg10=0.00 avg60=0.00 avg300=0.00 total=0\n"
);
}
#[test]
fn rtc_uses_the_virtual_clock_and_hides_alarm_state() {
let contents = b"rtc_time\t: 10:07:42\n\
rtc_date\t: 2026-07-27\n\
alrm_time\t: 00:50:12\n\
alrm_date\t: 2026-07-28\n\
alarm_IRQ\t: yes\n\
periodic IRQ enabled\t: yes\n\
periodic IRQ frequency\t: 1024\n\
24hr\t\t: yes\n";
assert_eq!(
sanitize_rtc(contents, 978_307_199),
b"rtc_time\t: 23:59:59\n\
rtc_date\t: 2000-12-31\n\
alrm_time\t: 00:00:00\n\
alrm_date\t: 2000-12-31\n\
alarm_IRQ\t: no\n\
periodic IRQ enabled\t: no\n\
periodic IRQ frequency\t: 0\n\
24hr\t\t: yes\n"
);
}
#[test]
fn protocols_hides_live_socket_and_memory_counters() {
let contents = b"protocol size sockets memory press maxhdr slab module cl co\n\
TCP 2304 17 12309 no 256 yes kernel y y\n\
RAW 1008 4 -1 NI 0 yes kernel y y\n";
assert_eq!(
sanitize_protocols(contents),
b"protocol size sockets memory press maxhdr slab module cl co\n\
TCP 2304 0 0 no 256 yes kernel y y\n\
RAW 1008 0 -1 NI 0 yes kernel y y\n"
);
}
#[test]
fn schedstat_hides_host_scheduler_accounting() {
let contents = b"version 17\n\
timestamp 4671819092\n\
cpu0 1 2 3 4 5 6 129488086714063 39956207684532 545893933\n\
domain0 SMT 00000003 1 2 3\n";
assert_eq!(
sanitize_schedstat(contents),
b"version 17\n\
timestamp 0\n\
cpu0 0 0 0 0 0 0 0 0 0\n\
domain0 SMT 00000003 0 0 0\n"
);
}
#[test]
fn self_schedstat_hides_host_scheduler_accounting() {
assert_eq!(
sanitize_self_schedstat(b"3029609 1559338 150\n"),
b"0 0 0\n"
);
assert_eq!(sanitize_self_schedstat(b"3029609 1559338 150"), b"0 0 0");
}
#[test]
fn self_schedstat_leaves_unknown_formats_untouched() {
let extra_field = b"3029609 1559338 150 4\n";
assert_eq!(sanitize_self_schedstat(extra_field), extra_field);
let invalid_counter = b"3029609 waiting 150\n";
assert_eq!(sanitize_self_schedstat(invalid_counter), invalid_counter);
}
#[test]
fn zoneinfo_hides_host_memory_accounting() {
let contents = b"Node 3, zone DMA32\n\
pages free 2816\n\
nr_inactive_anon 39937459\n\
protection: (0, 2117, 772897)\n\
cpu: 7\n\
count: 12\n";
assert_eq!(
sanitize_zoneinfo(contents),
b"Node 3, zone DMA32\n\
pages free 0\n\
nr_inactive_anon 0\n\
protection: (0, 0, 0)\n\
cpu: 7\n\
count: 0\n"
);
}
#[test]
fn protocols_leaves_unknown_formats_untouched() {
let missing_column = b"protocol size sockets press\nTCP 2304 17 no\n";
assert_eq!(sanitize_protocols(missing_column), missing_column);
let invalid_counter = b"protocol size sockets memory press maxhdr slab module\n\
TCP 2304 many 12309 no 256 yes kernel\n";
assert_eq!(sanitize_protocols(invalid_counter), invalid_counter);
}
#[test]
fn smaps_hides_host_memory_accounting_but_preserves_mapping_metadata() {
let contents = b"71000000-71001000 r-xp 00000000 00:00 0\n\
Size: 4 kB\n\
KernelPageSize: 4 kB\n\
MMUPageSize: 4 kB\n\
Rss: 4 kB\n\
Pss: 3 kB\n\
Shared_Clean: 4 kB\n\
Private_Dirty: 0 kB\n\
THPeligible: 0\n\
ProtectionKey: 0\n\
VmFlags: rd ex mr mw me ac\n";
assert_eq!(
sanitize_smaps(contents, &BTreeMap::new()),
b"71000000-71001000 r-xp 00000000 00:00 0\n\
Size: 4 kB\n\
KernelPageSize: 4 kB\n\
MMUPageSize: 4 kB\n\
Rss:\t0 kB\n\
Pss:\t0 kB\n\
Shared_Clean:\t0 kB\n\
Private_Dirty:\t0 kB\n\
THPeligible: 0\n\
ProtectionKey: 0\n\
VmFlags: rd ex mr mw me ac\n"
);
}
#[test]
fn smaps_hides_page_cache_dirty_state() {
let dirty = b"00400000-00401000 r--p 00000000 00:2a 99 /guest\n\
Size: 4 kB\n\
Rss: 4 kB\n\
Private_Dirty: 4 kB\n\
VmFlags: rd mr mw me\n";
let clean = b"00400000-00401000 r--p 00000000 00:2a 99 /guest\n\
Size: 4 kB\n\
Rss: 4 kB\n\
Private_Dirty: 0 kB\n\
VmFlags: rd mr mw me\n";
let expected = b"00400000-00401000 r--p 00000000 00:2a 99 /guest\n\
Size: 4 kB\n\
Rss:\t0 kB\n\
Private_Dirty:\t0 kB\n\
VmFlags: rd mr mw me\n";
assert_eq!(sanitize_smaps(dirty, &BTreeMap::new()), expected);
assert_eq!(sanitize_smaps(clean, &BTreeMap::new()), expected);
assert_eq!(
sanitize_smaps_rollup(b"Private_Dirty: 12 kB\n"),
b"Private_Dirty:\t0 kB\n"
);
}
const KERNEL_SMAPS_ACCOUNTING_LABELS: &[&str] = &[
"Rss",
"Pss",
"Pss_Dirty",
"Pss_Anon",
"Pss_File",
"Pss_Shmem",
"Shared_Clean",
"Shared_Dirty",
"Private_Clean",
"Private_Dirty",
"Referenced",
"Anonymous",
"KSM",
"LazyFree",
"AnonHugePages",
"ShmemPmdMapped",
"FilePmdMapped",
"Shared_Hugetlb",
"Private_Hugetlb",
"Swap",
"SwapPss",
"Locked",
];
#[test]
fn smaps_hides_every_host_page_accounting_field() {
let block = |seed: u64| {
let mut block = String::from(
"00400000-00600000 rw-p 00000000 00:00 0\nSize: 2048 kB\n",
);
for (index, label) in KERNEL_SMAPS_ACCOUNTING_LABELS.iter().enumerate() {
let value = seed * (index as u64 + 1);
block.push_str(&format!("{:<16}{value:>8} kB\n", format!("{label}:")));
}
block.push_str("THPeligible: 1\nVmFlags: rd wr mr mw me ac\n");
block
};
let idle = block(0);
let busy = block(4);
let mut expected =
String::from("00400000-00600000 rw-p 00000000 00:00 0\nSize: 2048 kB\n");
for label in KERNEL_SMAPS_ACCOUNTING_LABELS {
expected.push_str(&format!("{label}:\t0 kB\n"));
}
expected.push_str("THPeligible: 1\nVmFlags: rd wr mr mw me ac\n");
for input in [&idle, &busy] {
let smaps = sanitize_smaps(input.as_bytes(), &BTreeMap::new());
assert_eq!(String::from_utf8(smaps).unwrap(), expected);
}
let rollup = |block: &str| {
let block = block.replace("00400000-00600000 rw-p", "00400000-7ffffffff000 ---p");
let body: String = block
.lines()
.filter(|line| {
!line.starts_with("Size:")
&& !line.starts_with("THPeligible:")
&& !line.starts_with("VmFlags:")
})
.map(|line| format!("{line}\n"))
.collect();
String::from_utf8(sanitize_smaps_rollup(body.as_bytes())).unwrap()
};
let idle_rollup = rollup(&idle);
assert_eq!(idle_rollup, rollup(&busy));
for label in KERNEL_SMAPS_ACCOUNTING_LABELS {
assert!(
idle_rollup.contains(&format!("\n{label}:\t0 kB\n")),
"rollup left {label} unsanitized: {idle_rollup}"
);
}
}
#[test]
fn smaps_leaves_unknown_or_malformed_formats_untouched() {
let invalid_counter = b"71000000-71001000 r-xp 00000000 00:00 0\nPss: many kB\n";
assert_eq!(
sanitize_smaps(invalid_counter, &BTreeMap::new()),
invalid_counter
);
let unknown_field = b"71000000-71001000 r-xp 00000000 00:00 0\nMystery: 1 kB\n";
assert_eq!(
sanitize_smaps(unknown_field, &BTreeMap::new()),
unknown_field
);
let invalid_header = b"not-a-range r-xp 00000000 00:00 0\nPss: 3 kB\n";
assert_eq!(
sanitize_smaps(invalid_header, &BTreeMap::new()),
invalid_header
);
}
#[test]
fn arch_status_uses_logical_avx512_elapsed_time() {
let contents = b"AVX512_elapsed_ms:\t48\n\
x86_Thread_features:\t\tshstk\n\
x86_Thread_features_locked:\t\n";
assert_eq!(
sanitize_arch_status(contents),
b"AVX512_elapsed_ms:\t0\n\
x86_Thread_features:\t\tshstk\n\
x86_Thread_features_locked:\t\n"
);
assert_eq!(
sanitize_arch_status(b"AVX512_elapsed_ms:\tunknown\n"),
b"AVX512_elapsed_ms:\tunknown\n"
);
}
#[test]
fn smaps_rollup_hides_physical_page_accounting() {
let contents = b"71000000-7ffffffff000 ---p 00000000 00:00 0 [rollup]\n\
Rss: 2216 kB\n\
Pss: 311 kB\n\
Pss_File: 131 kB\n\
Locked: 12 kB\n\
Swap: 8 kB\n\
THPeligible: 0\n";
assert_eq!(
sanitize_smaps_rollup(contents),
b"71000000-7ffffffff000 ---p 00000000 00:00 0 [rollup]\n\
Rss:\t0 kB\n\
Pss:\t0 kB\n\
Pss_File:\t0 kB\n\
Locked:\t0 kB\n\
Swap:\t0 kB\n\
THPeligible: 0\n"
);
}
#[test]
fn numa_maps_hides_host_page_accounting() {
let contents = b"71000000 default heap anon=1 dirty=1 active=0 N12=1 kernelpagesize_kB=4\n\
7ffff7c00000 default file=/usr/lib64/libc.so.6 mapped=41 mapmax=443 N0=41 swapcache=2 writeback=3 kernelpagesize_kB=4\n";
assert_eq!(
sanitize_numa_maps(contents),
b"71000000 default heap kernelpagesize_kB=4\n\
7ffff7c00000 default file=/usr/lib64/libc.so.6 kernelpagesize_kB=4\n"
);
}
#[test]
fn numa_maps_leaves_unknown_formats_untouched() {
let invalid_address = b"address default anon=1\n";
assert_eq!(sanitize_numa_maps(invalid_address), invalid_address);
let invalid_counter = b"71000000 default active=recent N0=1\n";
assert_eq!(sanitize_numa_maps(invalid_counter), invalid_counter);
}
#[test]
fn fdinfo_hides_backing_identity_only() {
let contents = b"pos:\t1\n\
flags:\t0100002\n\
mnt_id:\t16368\n\
ino:\t47761541\n\
eventfd-count: 0000000000000007\n\
scm_fds: 2\n";
assert_eq!(
sanitize_fdinfo(contents, Some((9007, 0o100002, 42, 7))),
b"pos:\t1\n\
flags:\t0100002\n\
mnt_id:\t7\n\
ino:\t9007\n\
eventfd-count: 0000000000000007\n\
scm_fds: 2\n"
);
assert!(
sanitize_fdinfo(
b"pos:\t1\nflags:\t02\nmnt_id:\t7\nino:\t8\nunknown-field: 9\n",
Some((9007, 2, 42, 7))
)
.is_empty(),
"an unknown fdinfo field must continue to refuse normalization"
);
}
#[test]
fn fdinfo_mount_ids_match_distinct_mountinfo_rows_and_repeat() {
let mountinfo = b"37 29 8:1 / / rw - ext4 /dev/a rw\n\
48 37 0:22 / /proc rw - proc proc rw\n";
let snapshot = mountinfo_snapshot(
mountinfo,
BTreeMap::from([(libc::makedev(8, 1), 1), (libc::makedev(0, 22), 2)]),
);
let root_mount = snapshot.canonical_mount_id(37).unwrap();
let proc_mount = snapshot.canonical_mount_id(48).unwrap();
assert_ne!(root_mount, proc_mount);
let root_fdinfo = b"pos:\t0\nflags:\t0100000\nmnt_id:\t37\nino:\t10\n";
let proc_fdinfo = b"pos:\t0\nflags:\t0100000\nmnt_id:\t48\nino:\t20\n";
let root = sanitize_fdinfo(root_fdinfo, Some((100, 0o100000, 1, root_mount)));
let proc = sanitize_fdinfo(proc_fdinfo, Some((200, 0o100000, 2, proc_mount)));
assert!(
root.windows(b"mnt_id:\t1".len())
.any(|w| w == b"mnt_id:\t1")
);
assert!(
proc.windows(b"mnt_id:\t2".len())
.any(|w| w == b"mnt_id:\t2")
);
assert_eq!(
root,
sanitize_fdinfo(root_fdinfo, Some((100, 0o100000, 1, root_mount)))
);
assert_eq!(
proc,
sanitize_fdinfo(proc_fdinfo, Some((200, 0o100000, 2, proc_mount)))
);
}
#[test]
fn fdinfo_mount_id_parser_refuses_missing_duplicate_or_malformed_ids() {
assert_eq!(
detcore_model::procfs::parse_fdinfo_mount_id(b"mnt_id:\t37\n"),
Some(37)
);
assert_eq!(
detcore_model::procfs::parse_fdinfo_mount_id(b"mnt_id:\t0\n"),
Some(0)
);
for contents in [
b"pos:\t0\n".as_slice(),
b"mnt_id:\t37\nmnt_id:\t38\n".as_slice(),
b"mnt_id:\t-1\n".as_slice(),
b"mnt_id:\t37 trailing\n".as_slice(),
b"mnt_id:\t18446744073709551616\n".as_slice(),
] {
assert_eq!(detcore_model::procfs::parse_fdinfo_mount_id(contents), None);
}
let snapshot = mountinfo_snapshot(
b"37 29 8:1 / / rw - ext4 /dev/a rw\n",
BTreeMap::from([(libc::makedev(8, 1), 1)]),
);
assert_eq!(snapshot.canonical_mount_id(38), None);
}
#[test]
fn self_sched_hides_host_scheduler_accounting() {
let contents = b"cat (3, #threads: 7)\n\
se.exec_start : 377650149.445644\n\
se.vruntime : 133948666.432951\n\
se.sum_exec_runtime : 3.637972\n\
nr_switches : 149\n\
se.avg.load_avg : 749\n\
uclamp.min : 128\n\
uclamp.max : 768\n\
effective uclamp.min : 256\n\
effective uclamp.max : 512\n\
policy : 0\n\
uclamp.max : 1024\n\
new.kernel.counter : 55\n\
current_node=7, numa_group_id=91\n\
numa_faults node=7 task_private=8 task_shared=9 group_private=10 group_shared=11\n";
assert_eq!(
sanitize_self_sched(contents),
b"cat (0, #threads: 1)\n\
se.exec_start : 0.000000\n\
se.vruntime : 0.000000\n\
se.sum_exec_runtime : 0.000000\n\
nr_switches : 0\n\
se.avg.load_avg : 0\n\
uclamp.min : 0\n\
uclamp.max : 1024\n\
effective uclamp.min : 0\n\
effective uclamp.max : 1024\n\
policy : 0\n\
uclamp.max : 1024\n\
new.kernel.counter : 0\n\
current_node=0, numa_group_id=0\n\
numa_faults node=0 task_private=0 task_shared=0 group_private=0 group_shared=0\n"
);
}
#[test]
fn self_sched_fails_closed_on_unknown_formats() {
let missing_core_field = b"se.exec_start : 1.0\nse.vruntime : 2.0\n";
assert!(sanitize_self_sched(missing_core_field).is_empty());
let invalid_counter = b"se.exec_start : NaN\n\
se.vruntime : 2.0\n\
se.sum_exec_runtime : 3.0\n";
assert!(sanitize_self_sched(invalid_counter).is_empty());
let negative_counter = b"se.exec_start : 1.0\n\
se.vruntime : 2.0\n\
se.sum_exec_runtime : 3.0\n\
nr_switches : -1\n";
assert!(sanitize_self_sched(negative_counter).is_empty());
}
#[test]
fn locks_virtualize_identities_but_preserve_equivalences() {
let contents = b"74: POSIX ADVISORY WRITE 480 08:02:1111 0 EOF\n\
74: -> POSIX ADVISORY WRITE 481 08:02:1111 0 EOF\n\
9: OFDLCK ADVISORY WRITE 480 08:02:2222 0 EOF\n\
21: FLOCK ADVISORY WRITE 999 08:02:1111 0 EOF\n";
let out = String::from_utf8(sanitize_locks(contents)).unwrap();
let lines: Vec<&str> = out.lines().collect();
assert!(!out.contains("74"), "raw sequence leaked: {out}");
assert!(!out.contains("480") && !out.contains("481") && !out.contains("999"));
assert!(!out.contains("1111") && !out.contains("2222"));
let holder = lines
.iter()
.position(|l| l.contains("POSIX") && !l.contains("->"));
let waiter = lines.iter().position(|l| l.contains("->"));
let (holder, waiter) = (holder.unwrap(), waiter.unwrap());
assert_eq!(waiter, holder + 1, "waiter must immediately follow holder");
let seq_of = |l: &str| l.split(':').next().unwrap().to_owned();
assert_eq!(seq_of(lines[holder]), seq_of(lines[waiter]));
let obj_of = |l: &str| l.split_whitespace().nth_back(2).unwrap().to_owned();
let flock = lines.iter().find(|l| l.contains("FLOCK")).unwrap();
let ofd = lines.iter().find(|l| l.contains("OFDLCK")).unwrap();
assert_eq!(
obj_of(lines[holder]),
obj_of(flock),
"same object must match"
);
assert_ne!(
obj_of(lines[holder]),
obj_of(ofd),
"distinct objects must differ"
);
let pid_of = |l: &str| l.split_whitespace().nth_back(3).unwrap().to_owned();
assert_eq!(pid_of(lines[holder]), pid_of(ofd), "same owner must match");
assert_ne!(pid_of(lines[holder]), pid_of(lines[waiter]));
assert_ne!(pid_of(lines[holder]), pid_of(flock));
let renumbered_and_reordered = b"800: FLOCK ADVISORY WRITE 9000 00:fe:9001 0 EOF\n\
700: OFDLCK ADVISORY WRITE 8000 00:fe:9002 0 EOF\n\
900: -> POSIX ADVISORY WRITE 7000 00:fe:9001 0 EOF\n\
900: POSIX ADVISORY WRITE 8000 00:fe:9001 0 EOF\n";
assert_eq!(
sanitize_locks(contents),
sanitize_locks(renumbered_and_reordered),
"raw ID renumbering and row order changed the virtual graph"
);
assert!(sanitize_locks(b"malformed row\n").is_empty());
assert!(
sanitize_locks(b"74: POSIX ADVISORY WRITE 480 08:02:1111 0 EOF\nmalformed\n")
.is_empty()
);
assert!(sanitize_locks(&[0xff, 0xfe]).is_empty());
assert!(sanitize_locks(b"").is_empty());
}
#[test]
fn unix_sockets_hide_kernel_identities_and_sort_semantic_rows() {
let contents = b"Num RefCount Protocol Flags Type St Inode Path\n\
00000000fedcba98: 00000003 00000000 00010000 0002 01 12346 /run/socket two\n\
000000001234abcd: 00000002 00000000 00000000 0001 03 12345\n";
assert_eq!(
sanitize_unix_sockets(contents),
b"Num RefCount Protocol Flags Type St Inode Path\n\
0000000000000000: 00000002 00000000 00000000 0001 03 0\n\
0000000000000000: 00000003 00000000 00010000 0002 01 0 /run/socket two\n"
);
}
#[test]
fn unix_sockets_fail_open_on_unknown_schemas() {
for malformed in [
b"".as_slice(),
b"Num RefCount Protocol Flags Type St Inode Path",
b"Num RefCount Protocol Flags Type St Inode Path Extra\n",
b"Num RefCount Protocol Flags Type St Inode Path\n1234: 00000003 00000000 00000000 0001 03 12345\n",
b"Num RefCount Protocol Flags Type St Inode Path\n000000001234abcd: 0000000G 00000000 00000000 0001 03 12345\n",
b"Num RefCount Protocol Flags Type St Inode Path\n000000001234abcd: 00000003 00000000 00000000 0001 03 inode\n",
b"Num RefCount Protocol Flags Type St Inode Path\n000000001234abcd: 00000003 00000000 00000000 0001 03 18446744073709551616\n",
b"Num RefCount Protocol Flags Type St Inode Path\n000000001234abcd: 00000003 00000000 00000000 0001\n",
] {
assert_eq!(sanitize_unix_sockets(malformed), malformed);
}
}
#[test]
fn btrfs_commit_stats_hides_host_commit_telemetry() {
let contents = b"commits 14545\n\
cur_commit_ms 3\n\
last_commit_ms 211\n\
max_commit_ms 1977796\n\
total_commit_ms 11281713\n";
assert_eq!(
sanitize_btrfs_commit_stats(contents),
b"commits 0\n\
cur_commit_ms 0\n\
last_commit_ms 0\n\
max_commit_ms 0\n\
total_commit_ms 0\n"
);
}
#[test]
fn btrfs_commit_stats_leaves_unknown_formats_untouched() {
let wrong_order = b"cur_commit_ms 3\ncommits 14545\n";
assert_eq!(sanitize_btrfs_commit_stats(wrong_order), wrong_order);
let invalid_value = b"commits many\ncur_commit_ms 3\nlast_commit_ms 211\nmax_commit_ms 7\ntotal_commit_ms 9\n";
assert_eq!(sanitize_btrfs_commit_stats(invalid_value), invalid_value);
let extra_field = b"commits 1 transactions\ncur_commit_ms 3\nlast_commit_ms 2\nmax_commit_ms 7\ntotal_commit_ms 9\n";
assert_eq!(sanitize_btrfs_commit_stats(extra_field), extra_field);
}
#[test]
fn netlink_sockets_hide_kernel_identities_and_sort_semantic_rows() {
let contents = b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n\
00000000fedcba98 10 20 00000002 3 4 5 6 7 12346\n\
000000001234abcd 4 10 00000001 0 0 0 2 0 12345\n";
assert_eq!(
sanitize_netlink_sockets(contents),
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n\
0000000000000000 4 10 00000001 0 0 0 2 0 0\n\
0000000000000000 10 20 00000002 3 4 5 6 7 0\n"
);
}
#[test]
fn inet_sockets_zero_both_host_identities_and_canonicalize_row_order() {
let contents = b" sl local_address rem_address st tx_queue rx_queue tr tm->when retrnsmt uid timeout inode\n 0: 0100007F:8001 00000000:0000 0A 00000000:00000000 00:00000000 00000000 0 0 3066787053 1 00000000703938cf 100 0 0 10 0\n 1: 0100007F:8000 00000000:0000 0A 00000000:00000000 00:00000000 00000000 0 0 3066787052 1 0000000042a27170 100 0 0 10 0\n"
.as_slice();
let swapped = b" sl local_address rem_address st tx_queue rx_queue tr tm->when retrnsmt uid timeout inode\n 0: 0100007F:8000 00000000:0000 0A 00000000:00000000 00:00000000 00000000 0 0 3071433840 1 000000005b1c887b 100 0 0 10 0\n 1: 0100007F:8001 00000000:0000 0A 00000000:00000000 00:00000000 00000000 0 0 3071433841 1 00000000ab7b588a 100 0 0 10 0\n"
.as_slice();
let first = sanitize_inet_sockets(contents);
assert_eq!(
first,
sanitize_inet_sockets(swapped),
"two runs differing only in host identity and row order must agree"
);
let text = String::from_utf8(first).unwrap();
for row in text.lines().skip(1) {
let fields = row.split_whitespace().collect::<Vec<_>>();
assert_eq!(fields[9], "0", "inode must be zeroed: {row}");
assert_eq!(
fields[11], "0000000000000000",
"pointer must be zeroed: {row}"
);
}
let indices = text
.lines()
.skip(1)
.map(|row| row.split_whitespace().next().unwrap().to_string())
.collect::<Vec<_>>();
assert_eq!(indices, vec!["0:", "1:"]);
assert!(text.lines().nth(1).unwrap().contains(":8000"));
}
#[test]
fn inet_sockets_fail_open_on_unknown_schemas() {
for malformed in [
b"".as_slice(),
b" sl local_address rem_address st inode",
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n",
b" sl local_address inode\n 0: 0100007F:8000 12345\n",
b" sl a b c d e f g h inode j pointer\n 0: 1 2 3 4 5 6 7 8 zero 1 0000000042a27170\n",
] {
assert_eq!(sanitize_inet_sockets(malformed), malformed);
}
}
#[test]
fn netlink_sockets_fail_open_on_unknown_schemas() {
for malformed in [
b"".as_slice(),
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode",
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode Extra\n",
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n1234 4 10 00000001 0 0 0 2 0 12345\n",
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n000000001234abcd 4 10 00000001 0 0 0 2 zero 12345\n",
b"sk Eth Pid Groups Rmem Wmem Dump Locks Drops Inode\n000000001234abcd 4 10 00000001 0 0 0 2 0 12345 extra\n",
] {
assert_eq!(sanitize_netlink_sockets(malformed), malformed);
}
}
#[test]
fn irq_per_cpu_count_hides_host_interrupt_totals() {
assert_eq!(
sanitize_irq_per_cpu_count(b"0,17,0,983421,0\n"),
b"0,0,0,0,0\n"
);
assert_eq!(sanitize_irq_per_cpu_count(b"42,0"), b"0,0");
}
#[test]
fn irq_per_cpu_count_leaves_unknown_formats_untouched() {
for contents in [
b"".as_slice(),
b"1,,2\n".as_slice(),
b"1,2,three\n".as_slice(),
b"1,2\n3,4\n".as_slice(),
] {
assert_eq!(sanitize_irq_per_cpu_count(contents), contents);
}
}
#[test]
fn block_inflight_hides_host_queue_depths() {
assert_eq!(sanitize_block_inflight(b" 24 3\n"), b"0 0\n");
assert_eq!(sanitize_block_inflight(b"0 7"), b"0 0");
}
#[test]
fn block_inflight_leaves_unknown_formats_untouched() {
let malformed = b"reads writes\n";
assert_eq!(sanitize_block_inflight(malformed), malformed);
let extra_field = b"1 2 3\n";
assert_eq!(sanitize_block_inflight(extra_field), extra_field);
let extra_row = b"1 2\n3 4\n";
assert_eq!(sanitize_block_inflight(extra_row), extra_row);
}
#[test]
fn vmstat_hides_host_vm_accounting() {
let contents = b"nr_free_pages 4587515\npgfault 175926829665\noom_kill 30\n";
assert_eq!(
sanitize_vmstat(contents),
b"nr_free_pages 0\npgfault 0\noom_kill 0\n"
);
assert_eq!(
sanitize_vmstat(b"nr_free_pages 4587515"),
b"nr_free_pages 0"
);
}
#[test]
fn vmstat_leaves_unknown_formats_untouched() {
let extra_field = b"nr_free_pages 4587515 pages\n";
assert_eq!(sanitize_vmstat(extra_field), extra_field);
let invalid_counter = b"nr_free_pages many\n";
assert_eq!(sanitize_vmstat(invalid_counter), invalid_counter);
}
#[test]
fn meminfo_uses_configured_guest_memory() {
assert_eq!(
sanitize_meminfo(b"MemTotal: 791462432 kB\n", 1_048_576),
b"MemTotal: 1048576 kB\n\
MemFree: 1048576 kB\n\
MemAvailable: 1048576 kB\n\
Buffers: 0 kB\n\
Cached: 0 kB\n\
SwapCached: 0 kB\n\
Active: 0 kB\n\
Inactive: 0 kB\n\
Shmem: 0 kB\n\
SReclaimable: 0 kB\n\
SwapTotal: 0 kB\n\
SwapFree: 0 kB\n"
);
assert!(sanitize_meminfo(b"", 1_048_576).is_empty());
}
#[test]
fn snapshot_supports_partial_reads() {
let mut file = ProcfsFile::from_path(Path::new("/proc/self/status")).unwrap();
file.initialize(
b"voluntary_ctxt_switches:\t12\n".to_vec(),
ProcfsSnapshotContext {
virtual_uptime_seconds: 120,
virtual_pid: 3,
virtual_ppid: 1,
..ProcfsSnapshotContext::default()
},
);
assert_eq!(file.take(5).unwrap(), b"volun");
assert_eq!(file.take(128).unwrap(), b"tary_ctxt_switches:\t0\n");
assert!(file.take(1).unwrap().is_empty());
}
#[test]
fn snapshot_supports_positional_reads_and_rewinds() {
let mut file = ProcfsFile::from_path(Path::new("/proc/sys/fs/file-nr")).unwrap();
file.initialize(
b"245853\t0\t1048576\n".to_vec(),
ProcfsSnapshotContext {
virtual_pid: 1,
..ProcfsSnapshotContext::default()
},
);
assert_eq!(file.take(2).unwrap(), b"0\t");
assert_eq!(file.take_at(4, 1).unwrap(), b"9");
assert_eq!(file.position().0, 2, "pread must not move the cursor");
file.set_offset(0);
assert_eq!(file.take(128).unwrap(), b"0\t0\t9223372036854775807\n");
}
#[test]
fn module_refcnt_classification_is_two_sided_over_path_shapes() {
let accepted = [
"/sys/module/kvm/refcnt",
"/sys/module/./kvm/refcnt",
"/sys/module/kvm/../kvm/refcnt",
"/sys/module/nvme/refcnt",
];
let mut accepted_count = 0;
for path in accepted {
let kind = ProcfsFile::from_path(Path::new(path))
.unwrap_or_else(|| panic!("{path} names a module refcnt object"))
.kind;
let expected = if path.contains("nvme") { "nvme" } else { "kvm" };
assert_eq!(
kind,
ProcfsKind::ModuleRefcnt(expected.to_owned()),
"{path} must classify as {expected}'s refcnt"
);
accepted_count += 1;
}
assert_eq!(
accepted_count, 4,
"all four accepted spellings were checked"
);
let refused = [
"refcnt",
"kvm/refcnt",
"sys/module/kvm/refcnt",
"/sys/module/kvm/coresize",
"/sys/module/refcnt",
"/sys/module/kvm/holders/refcnt",
"/sys/module//refcnt",
"/sys/modulefoo/kvm/refcnt",
];
let mut refused_count = 0;
for path in refused {
assert!(
!matches!(
ProcfsFile::from_path(Path::new(path)).map(|file| file.kind),
Some(ProcfsKind::ModuleRefcnt(_))
),
"{path} must not be classified as a module refcnt"
);
refused_count += 1;
}
assert_eq!(refused_count, 8, "all eight refused shapes were checked");
}
#[test]
fn sysfs_refcnt_agrees_with_proc_modules_use_count() {
let modules = "kvm_amd 212992 95 - Live 0x0\n\
kvm 1200128 1 kvm_amd, Live 0x0\n\
llc 20480 2 bridge,stp, Live 0x0\n";
let normalized = sanitize_modules(modules.as_bytes());
let normalized = std::str::from_utf8(&normalized).unwrap();
let mut compared = 0;
for line in normalized.lines() {
let mut fields = line.split_whitespace();
let module = fields.next().unwrap();
fields.next().unwrap();
let proc_use_count: u64 = fields.next().unwrap().parse().unwrap();
let sysfs = sanitize_module_refcnt(b"999\n", module, modules);
let sysfs: u64 = std::str::from_utf8(&sysfs).unwrap().trim().parse().unwrap();
assert_eq!(
sysfs, proc_use_count,
"{module}: /proc/modules says {proc_use_count} but sysfs says {sysfs}"
);
compared += 1;
}
assert_eq!(compared, 3, "all three modules were cross-checked");
assert_eq!(sanitize_module_refcnt(b"1\n", "kvm", modules), b"1\n");
}
#[test]
fn spelling_defined_kinds_classify_without_resolution() {
let spelling_defined = [
("/proc/self/stat", ProcfsKind::Stat),
("/proc/self/status", ProcfsKind::Status),
("/proc/thread-self/stat", ProcfsKind::ThreadStat),
("/proc/thread-self/status", ProcfsKind::ThreadStatus),
("/proc/self/statm", ProcfsKind::Statm),
("/proc/thread-self/statm", ProcfsKind::Statm),
("/proc/self/mountinfo", ProcfsKind::Mountinfo),
];
let mut guarded = 0;
for (path, expected) in spelling_defined {
let kind = ProcfsFile::from_path(Path::new(path))
.unwrap_or_else(|| panic!("{path} must classify from its spelling alone"))
.kind;
assert_eq!(kind, expected, "{path} must keep its spelling-defined kind");
guarded += 1;
}
assert_eq!(guarded, 7, "all seven spelling-defined kinds were guarded");
let numeric = ProcfsFile::from_path(Path::new("/proc/1234/stat"))
.expect("numeric process stat still classifies")
.kind;
assert_ne!(
numeric,
ProcfsKind::Stat,
"/proc/<pid>/stat is a different kind from /proc/self/stat; \
resolving the latter into the former is the regression this guards"
);
let mut unresolved = 0;
for path in ["refcnt", "kvm/refcnt", "sys/module/kvm/refcnt"] {
assert!(
ProcfsFile::from_path(Path::new(path)).is_none(),
"{path} must not classify lexically; only resolution can"
);
unresolved += 1;
}
assert_eq!(unresolved, 3, "all three relative spellings were checked");
assert_eq!(
ProcfsFile::from_path(Path::new("/sys/module/kvm/refcnt"))
.expect("the resolved form classifies")
.kind,
ProcfsKind::ModuleRefcnt("kvm".to_owned()),
"resolution turns the unclassifiable spelling into the real kind"
);
}
}