Skip to main content

rlmctl_core/
cgroup.rs

1use common::{CpuLimit, Error, IoLimit, Limit, MemoryLimit, Result};
2use std::fs;
3use std::path::{Path, PathBuf};
4use std::process::Command;
5
6const CGROUP_ROOT: &str = "/sys/fs/cgroup";
7
8/// Name of the shared, controller-free leaf cgroup that every `cleanup_cgroup`
9/// (i.e. every `rlm unlimit`, `remove_limit`/`remove_application_limit`, and
10/// `RuleAction::TeardownEmpty`) moves released processes into so the cgroup
11/// being torn down can be emptied and removed. It is a grab-bag of unrelated
12/// processes the user has explicitly released from rlm's control, so nothing
13/// in rlm may treat it as an actionable target (see `status.rs` and
14/// `guard/resolve.rs`).
15pub const UNLIMIT_CGROUP_NAME: &str = "unlimit";
16
17/// Sanitize cgroup name to prevent path traversal attacks.
18/// Only allows alphanumeric characters, dashes, and underscores.
19fn sanitize_cgroup_name(name: &str) -> Result<&str> {
20    // Reject empty names
21    if name.is_empty() {
22        return Err(Error::InvalidArgs("cgroup name cannot be empty".into()));
23    }
24
25    // Reject path traversal attempts
26    if name.contains('/') || name.contains('\\') || name.contains("..") {
27        return Err(Error::InvalidArgs(
28            "cgroup name contains invalid characters".into(),
29        ));
30    }
31
32    // Validate characters: alphanumeric, dash, underscore only
33    if !name
34        .chars()
35        .all(|c| c.is_alphanumeric() || c == '-' || c == '_')
36    {
37        return Err(Error::InvalidArgs(
38            "cgroup name must contain only alphanumeric characters, dashes, or underscores".into(),
39        ));
40    }
41
42    Ok(name)
43}
44
45/// Error for a process that already sits in another rlm cgroup. The hint
46/// always names the cgroup: `rlm unlimit --cgroup` works for every rlm
47/// cgroup, while `rlm unlimit --pid` refuses any cgroup other than `pid-N`.
48fn already_limited(pid: u32, cgroup: &str) -> Error {
49    Error::InvalidArgs(format!(
50        "process {pid} is already limited in cgroup '{cgroup}'; run rlm unlimit --cgroup {cgroup} first"
51    ))
52}
53
54/// Refuse to limit init (PID 1). Constraining PID 1 (systemd/init) can wedge or
55/// freeze the entire system, the opposite of what this tool is for.
56fn reject_critical_pid(pid: u32) -> Result<()> {
57    if pid <= 1 {
58        return Err(Error::InvalidArgs(format!(
59            "refusing to limit PID {pid} (init/system critical)"
60        )));
61    }
62    Ok(())
63}
64
65/// Result of [`CgroupManager::prepare_cgroup`].
66#[derive(Debug, Clone, PartialEq, Eq)]
67pub struct Prepared {
68    /// Absolute path of the cgroup directory.
69    pub path: PathBuf,
70    /// Whether this call created the directory (false when it already existed).
71    pub created: bool,
72    /// Non-fatal problems, such as I/O limits the kernel refused.
73    pub warnings: Vec<String>,
74}
75
76/// One `io.max` line per device, e.g. `8:0 rbps=5242880 wbps=1048576`.
77/// The kernel parses one device per write, so each line is written on its own.
78pub fn io_max_lines(devices: &[(u32, u32)], limit: IoLimit) -> Vec<String> {
79    devices
80        .iter()
81        .map(|(major, minor)| {
82            let mut line = format!("{major}:{minor}");
83            if let Some(rbps) = limit.read_bps {
84                line.push_str(&format!(" rbps={rbps}"));
85            }
86            if let Some(wbps) = limit.write_bps {
87                line.push_str(&format!(" wbps={wbps}"));
88            }
89            line
90        })
91        .collect()
92}
93
94pub struct CgroupManager {
95    base_path: PathBuf,
96}
97
98impl CgroupManager {
99    pub fn new() -> Result<Self> {
100        // Verify cgroups v2 is available
101        let controllers_path = PathBuf::from(CGROUP_ROOT).join("cgroup.controllers");
102        if !controllers_path.exists() {
103            return Err(Error::CgroupsV2NotAvailable(PathBuf::from(CGROUP_ROOT)));
104        }
105
106        Ok(Self {
107            base_path: Self::default_base_path(),
108        })
109    }
110
111    /// A manager rooted at `base_path`, with no checks. For tests and
112    /// diagnostics.
113    pub fn at(base_path: PathBuf) -> Self {
114        Self { base_path }
115    }
116
117    /// rlm's base cgroup: `user@UID.service/rlm` when the user's systemd
118    /// service cgroup exists, else `/sys/fs/cgroup/rlm`.
119    pub fn default_base_path() -> PathBuf {
120        // Determine our real UID from the kernel via /proc/self/status — NOT from
121        // the `$UID` environment variable, which is caller-controllable and must
122        // not be allowed to steer which cgroup path we operate on. Parsing as u32
123        // also guarantees the value can't inject path components.
124        let uid = fs::read_to_string("/proc/self/status").ok().and_then(|s| {
125            s.lines()
126                .find(|l| l.starts_with("Uid:"))
127                .and_then(|l| l.split_whitespace().nth(1))
128                .and_then(|u| u.parse::<u32>().ok())
129        });
130
131        // Try the user's systemd scope (for non-root with cgroup delegation).
132        if let Some(uid) = uid {
133            let user_slice = PathBuf::from(CGROUP_ROOT).join(format!(
134                "user.slice/user-{uid}.slice/user@{uid}.service/rlm"
135            ));
136
137            if let Some(parent) = user_slice.parent() {
138                if parent.exists() {
139                    return user_slice;
140                }
141            }
142        }
143
144        // Fallback: try directly under cgroup root (requires root or delegation)
145        PathBuf::from(CGROUP_ROOT).join("rlm")
146    }
147
148    /// Get the base path (for testing/status)
149    pub fn base_path(&self) -> &Path {
150        &self.base_path
151    }
152
153    /// Create a cgroup for a process and set limits BEFORE adding the process.
154    /// If a memory or CPU limit cannot be set, a cgroup this call created is
155    /// removed again; one that already existed is left alone, so a failed
156    /// limit write never empties a cgroup that holds processes.
157    pub fn prepare_cgroup(&self, name: &str, limit: &Limit) -> Result<Prepared> {
158        // Sanitize name to prevent path traversal
159        let safe_name = sanitize_cgroup_name(name)?;
160        let path = self.base_path.join(safe_name);
161        let created = self.create_cgroup(&path)?;
162        match self.set_limits(&path, limit) {
163            Ok(warnings) => Ok(Prepared {
164                path,
165                created,
166                warnings,
167            }),
168            Err(e) => {
169                if created {
170                    let _ = self.cleanup_cgroup(safe_name);
171                }
172                Err(e)
173            }
174        }
175    }
176
177    /// Set limits on an existing cgroup. Memory and CPU failures are errors;
178    /// I/O problems come back as warnings.
179    fn set_limits(&self, cgroup_path: &Path, limit: &Limit) -> Result<Vec<String>> {
180        if let Some(mem) = &limit.memory {
181            self.set_memory_limit(cgroup_path, *mem)?;
182        }
183
184        if let Some(cpu) = &limit.cpu {
185            self.set_cpu_limit(cgroup_path, *cpu)?;
186        }
187
188        let mut warnings = Vec::new();
189        if let Some(io) = &limit.io {
190            if !io.is_empty() {
191                warnings = self.set_io_limit(cgroup_path, *io);
192            }
193        }
194
195        Ok(warnings)
196    }
197
198    /// Build a [`Command`] that places the spawned child into `cgroup_path`
199    /// *before* it execs the target program, so resource limits apply from the
200    /// process's very first instruction.
201    ///
202    /// Without this, a process that allocates aggressively at startup could blow
203    /// past the limit during the window between spawn and being added to the
204    /// cgroup — exactly the freeze scenario this tool exists to prevent.
205    ///
206    /// Writing "0" to `cgroup.procs` from the post-fork, pre-exec child moves it
207    /// into the cgroup. The file is opened in the parent so the closure performs
208    /// only an async-signal-safe `write` to an already-open fd (no allocation, no
209    /// locks). Placement is best-effort: on failure the process still launches,
210    /// so callers should still call [`add_to_cgroup`](Self::add_to_cgroup) after
211    /// spawn as a fallback. Add command arguments to the returned `Command`.
212    pub fn placement_command(&self, cgroup_path: &Path, program: &str) -> Command {
213        use std::os::unix::process::CommandExt;
214
215        let mut cmd = Command::new(program);
216        if let Ok(file) = fs::OpenOptions::new()
217            .write(true)
218            .open(cgroup_path.join("cgroup.procs"))
219        {
220            // SAFETY: the closure only writes a fixed byte slice to an already-open
221            // file descriptor — an async-signal-safe operation that allocates
222            // nothing and takes no locks. Errors are ignored (best-effort).
223            unsafe {
224                cmd.pre_exec(move || {
225                    use std::io::Write;
226                    let _ = (&file).write_all(b"0");
227                    Ok(())
228                });
229            }
230        }
231        cmd
232    }
233
234    /// Add a process to an existing cgroup
235    pub fn add_to_cgroup(&self, cgroup_path: &Path, pid: u32) -> Result<()> {
236        self.add_process(cgroup_path, pid)?;
237        tracing::info!(pid, ?cgroup_path, "added process to cgroup");
238        Ok(())
239    }
240
241    /// Find if a PID is already in an rlm-managed cgroup. The `unlimit`
242    /// bucket is not a managed cgroup, so released processes can be limited
243    /// again.
244    pub fn find_cgroup_for_pid(&self, pid: u32) -> Option<String> {
245        let entries = fs::read_dir(&self.base_path).ok()?;
246
247        for entry in entries.flatten() {
248            let path = entry.path();
249            if !path.is_dir() || entry.file_name() == UNLIMIT_CGROUP_NAME {
250                continue;
251            }
252
253            let procs_file = path.join("cgroup.procs");
254            if let Ok(content) = fs::read_to_string(&procs_file) {
255                for line in content.lines() {
256                    if line.trim().parse::<u32>().ok() == Some(pid) {
257                        return path.file_name()?.to_str().map(String::from);
258                    }
259                }
260            }
261        }
262        None
263    }
264
265    /// Apply resource limits to a process (creates cgroup and adds process).
266    /// Returns non-fatal warnings.
267    pub fn apply_limit(&self, pid: u32, limit: &Limit) -> Result<Vec<String>> {
268        reject_critical_pid(pid)?;
269
270        // Check if process is already managed
271        if let Some(existing_cgroup) = self.find_cgroup_for_pid(pid) {
272            // If it's in a pid-{pid} cgroup, update the limits
273            if existing_cgroup == format!("pid-{pid}") {
274                let cgroup_path = self.base_path.join(&existing_cgroup);
275                let warnings = self.set_limits(&cgroup_path, limit)?;
276                tracing::info!(pid, "updated existing limits");
277                return Ok(warnings);
278            }
279            // Process is in a shared, run-* or gtk-* cgroup. `unlimit --pid`
280            // refuses those, so the hint names the whole cgroup.
281            return Err(already_limited(pid, &existing_cgroup));
282        }
283
284        let Prepared {
285            path: cgroup_path,
286            warnings,
287            ..
288        } = self.prepare_cgroup(&format!("pid-{pid}"), limit)?;
289
290        // Try to add process - if it fails because process doesn't exist,
291        // clean up the cgroup and return appropriate error
292        if let Err(e) = self.add_process(&cgroup_path, pid) {
293            // Remove the cgroup again; it holds nothing we put there.
294            let _ = self.remove_if_empty(&format!("pid-{pid}"));
295            // Check if process exists to give better error message
296            if !PathBuf::from(format!("/proc/{pid}")).exists() {
297                return Err(Error::ProcessNotFound(pid));
298            }
299            return Err(e);
300        }
301
302        tracing::info!(pid, ?cgroup_path, "applied limits");
303        Ok(warnings)
304    }
305
306    /// Apply resource limits to multiple processes (all share the same limit pool)
307    /// All processes are added to a single cgroup, so they share the resource limits.
308    /// For example, if you limit 10 processes to 4GB memory, they share 4GB total, not 4GB each.
309    /// Returns non-fatal warnings, including PIDs that could not be added.
310    pub fn apply_limit_to_multiple(
311        &self,
312        pids: &[u32],
313        limit: &Limit,
314        cgroup_name: &str,
315    ) -> Result<Vec<String>> {
316        if pids.is_empty() {
317            return Err(Error::InvalidArgs("no processes specified".into()));
318        }
319
320        for pid in pids {
321            reject_critical_pid(*pid)?;
322        }
323
324        // Sanitize cgroup name
325        let safe_name = sanitize_cgroup_name(cgroup_name)?;
326
327        // Check if any process is already managed
328        for pid in pids {
329            if let Some(existing_cgroup) = self.find_cgroup_for_pid(*pid) {
330                // Allow if it's already in the same cgroup we're creating
331                if existing_cgroup != safe_name {
332                    return Err(already_limited(*pid, &existing_cgroup));
333                }
334            }
335        }
336
337        // Create cgroup and set limits
338        let Prepared {
339            path: cgroup_path,
340            created,
341            mut warnings,
342        } = self.prepare_cgroup(safe_name, limit)?;
343
344        // Add all processes to the cgroup
345        let mut failed_pids = Vec::new();
346        for pid in pids {
347            if let Err(e) = self.add_process(&cgroup_path, *pid) {
348                tracing::warn!(pid, error = %e, "failed to add process to cgroup");
349                failed_pids.push(*pid);
350            } else {
351                tracing::info!(pid, ?cgroup_path, "added process to shared cgroup");
352            }
353        }
354
355        // If all processes failed, remove a cgroup this call created.
356        if failed_pids.len() == pids.len() {
357            if created {
358                let _ = self.cleanup_cgroup(safe_name);
359            }
360            return Err(Error::InvalidArgs(
361                "failed to add any processes to cgroup".into(),
362            ));
363        }
364
365        // If some failed, report it but continue
366        if !failed_pids.is_empty() {
367            warnings.push(format!(
368                "could not add {} of {} processes: {:?}",
369                failed_pids.len(),
370                pids.len(),
371                failed_pids
372            ));
373        }
374
375        Ok(warnings)
376    }
377
378    /// Remove limits from a process
379    pub fn remove_limit(&self, pid: u32) -> Result<()> {
380        self.cleanup_cgroup(&format!("pid-{pid}"))
381    }
382
383    /// Remove limits from an application cgroup (removes all processes in the cgroup)
384    pub fn remove_application_limit(&self, cgroup_name: &str) -> Result<()> {
385        self.cleanup_cgroup(cgroup_name)
386    }
387
388    /// Clean up a cgroup by name (moves processes out and deletes cgroup)
389    pub fn cleanup_cgroup(&self, name: &str) -> Result<()> {
390        // Sanitize name to prevent path traversal
391        let safe_name = sanitize_cgroup_name(name)?;
392        let cgroup_path = self.base_path.join(safe_name);
393
394        if !cgroup_path.exists() {
395            return Ok(());
396        }
397
398        // Move any processes out to the controller-free "unlimit" cgroup so this
399        // cgroup becomes empty and can be removed.
400        if let Ok(content) = fs::read_to_string(cgroup_path.join("cgroup.procs")) {
401            let pids: Vec<u32> = content
402                .lines()
403                .filter_map(|l| l.trim().parse().ok())
404                .collect();
405
406            if !pids.is_empty() {
407                // Create/use an "unlimit" leaf cgroup (no controllers = no limits)
408                let unlimit_path = self.base_path.join(UNLIMIT_CGROUP_NAME);
409                let _ = fs::create_dir(&unlimit_path);
410                let unlimit_procs = unlimit_path.join("cgroup.procs");
411
412                for pid in pids {
413                    if fs::write(&unlimit_procs, pid.to_string()).is_ok() {
414                        tracing::debug!(pid, "moved process to unlimit cgroup");
415                    }
416                }
417            }
418        }
419
420        // Try to remove the (now hopefully empty) cgroup.
421        for _ in 0..3 {
422            match fs::remove_dir(&cgroup_path) {
423                Ok(()) => {
424                    tracing::info!(?cgroup_path, "removed cgroup");
425                    return Ok(());
426                }
427                Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(()),
428                Err(_) => std::thread::sleep(std::time::Duration::from_millis(50)),
429            }
430        }
431
432        // Removal failed. If processes are still inside (couldn't be moved out),
433        // reset the limits in place so the caller's "remove limits" intent is
434        // still satisfied — report success but warn that the cgroup lingers.
435        let still_has_procs = fs::read_to_string(cgroup_path.join("cgroup.procs"))
436            .map(|c| c.lines().any(|l| !l.trim().is_empty()))
437            .unwrap_or(false);
438
439        if still_has_procs {
440            // Defensive: if this is a frozen guard cgroup we couldn't empty, at
441            // least unfreeze it so its tasks are never stuck paused.
442            let _ = fs::write(cgroup_path.join("cgroup.freeze"), "0");
443            let _ = fs::write(cgroup_path.join("memory.high"), "max");
444            let _ = fs::write(cgroup_path.join("memory.max"), "max");
445            let _ = fs::write(cgroup_path.join("memory.swap.max"), "max");
446            let _ = fs::write(cgroup_path.join("cpu.max"), "max");
447            let _ = fs::write(cgroup_path.join("io.max"), "");
448            tracing::warn!(
449                ?cgroup_path,
450                "could not remove cgroup (still has live processes); limits reset in place"
451            );
452            return Ok(());
453        }
454
455        // Empty but still not removable — a genuine failure the caller should see.
456        Err(Error::Cgroup(format!(
457            "failed to remove cgroup '{safe_name}'"
458        )))
459    }
460
461    /// Whether the named child cgroup (or a descendant) holds a process, from
462    /// its `cgroup.events`. `None` if unreadable.
463    pub fn is_populated(&self, name: &str) -> Option<bool> {
464        let content = fs::read_to_string(self.base_path.join(name).join("cgroup.events")).ok()?;
465        crate::guard::cgfs::parse_populated(&content)
466    }
467
468    /// The `oom_kill` count from the named cgroup's `memory.events`.
469    pub fn oom_kills(&self, name: &str) -> Option<u64> {
470        let content = fs::read_to_string(self.base_path.join(name).join("memory.events")).ok()?;
471        crate::guard::cgfs::parse_events_field(&content, "oom_kill")
472    }
473
474    /// The named cgroup's `memory.max` in bytes; `None` for `max` or unreadable.
475    pub fn memory_max(&self, name: &str) -> Option<u64> {
476        let content = fs::read_to_string(self.base_path.join(name).join("memory.max")).ok()?;
477        content.trim().parse().ok()
478    }
479
480    /// Remove the named cgroup only if it holds no process. `Ok(false)` when
481    /// it is populated; a cgroup that is already gone counts as removed.
482    /// Never moves processes.
483    pub fn remove_if_empty(&self, name: &str) -> Result<bool> {
484        let safe_name = sanitize_cgroup_name(name)?;
485        if self.is_populated(safe_name) == Some(true) {
486            return Ok(false);
487        }
488        let path = self.base_path.join(safe_name);
489        let mut last_err = None;
490        for _ in 0..3 {
491            match fs::remove_dir(&path) {
492                Ok(()) => {
493                    tracing::info!(?path, "removed empty cgroup");
494                    return Ok(true);
495                }
496                Err(e) if e.kind() == std::io::ErrorKind::NotFound => return Ok(true),
497                Err(e) => {
498                    last_err = Some(e);
499                    std::thread::sleep(std::time::Duration::from_millis(50));
500                }
501            }
502        }
503        Err(Error::Cgroup(format!(
504            "failed to remove cgroup '{safe_name}': {}",
505            last_err.map(|e| e.to_string()).unwrap_or_default()
506        )))
507    }
508
509    /// Whether a child cgroup with this name currently exists.
510    pub fn cgroup_exists(&self, name: &str) -> bool {
511        self.base_path.join(name).is_dir()
512    }
513
514    /// PIDs currently in the named child cgroup (empty if it doesn't exist).
515    pub fn pids_in_cgroup(&self, name: &str) -> Vec<u32> {
516        let procs = self.base_path.join(name).join("cgroup.procs");
517        match fs::read_to_string(procs) {
518            Ok(content) => content
519                .lines()
520                .filter_map(|l| l.trim().parse::<u32>().ok())
521                .collect(),
522            Err(_) => Vec::new(),
523        }
524    }
525
526    /// Startup recovery: thaw and clean up every leftover `guard-<pid>`
527    /// cgroup so no process is left frozen after a prior crash.
528    ///
529    /// Legacy: pre-act-in-place `rlm-guard` builds moved a target process into
530    /// its own `guard-<pid>` cgroup to freeze/cap it; the guard now acts in
531    /// place on the process's existing cgroup (journal-backed, see
532    /// `guard/effector.rs`) and never creates `guard-<pid>` cgroups itself.
533    /// This sweep only exists to clean up leftovers from an upgrade across
534    /// that change and can be removed after one release.
535    pub fn sweep_guard_leftovers(&self) -> Result<()> {
536        let Ok(entries) = fs::read_dir(&self.base_path) else {
537            return Ok(());
538        };
539        for entry in entries.flatten() {
540            let Some(name) = entry.file_name().to_str().map(str::to_string) else {
541                continue;
542            };
543            if !name.starts_with("guard-") {
544                continue;
545            }
546            // Always thaw first: a frozen task can't be migrated out, and we
547            // must never leave a process stuck frozen even if the teardown
548            // below fails.
549            let _ = fs::write(entry.path().join("cgroup.freeze"), "0");
550            let _ = self.cleanup_cgroup(&name);
551        }
552        Ok(())
553    }
554
555    /// Create the cgroup directory. `Ok(true)` only when this call created it.
556    fn create_cgroup(&self, path: &Path) -> Result<bool> {
557        // Ensure base path exists (create_dir_all is idempotent, avoids TOCTOU)
558        if let Err(e) = fs::create_dir_all(&self.base_path) {
559            if e.kind() == std::io::ErrorKind::PermissionDenied {
560                return Err(Error::PermissionDenied {
561                    path: self.base_path.clone(),
562                });
563            } else if e.kind() != std::io::ErrorKind::AlreadyExists {
564                return Err(e.into());
565            }
566        }
567
568        // Enable controllers in base cgroup for child cgroups
569        self.enable_controllers(&self.base_path)?;
570
571        // Create cgroup directory (handle AlreadyExists to avoid TOCTOU)
572        match fs::create_dir(path) {
573            Ok(()) => Ok(true),
574            Err(e) if e.kind() == std::io::ErrorKind::AlreadyExists => Ok(false),
575            Err(e) if e.kind() == std::io::ErrorKind::PermissionDenied => {
576                Err(Error::PermissionDenied {
577                    path: path.to_path_buf(),
578                })
579            }
580            Err(e) => Err(e.into()),
581        }
582    }
583
584    fn enable_controllers(&self, path: &Path) -> Result<()> {
585        let subtree_control = path.join("cgroup.subtree_control");
586
587        // Read available controllers first
588        let controllers_file = path.join("cgroup.controllers");
589        let available = fs::read_to_string(&controllers_file).unwrap_or_default();
590
591        // Only enable controllers that are available
592        let mut to_enable = Vec::new();
593        for controller in ["memory", "cpu", "io"] {
594            if available.contains(controller) {
595                to_enable.push(format!("+{controller}"));
596            }
597        }
598
599        if to_enable.is_empty() {
600            return Err(Error::Cgroup(
601                "no controllers available - run as root or configure cgroup delegation".into(),
602            ));
603        }
604
605        fs::write(&subtree_control, to_enable.join(" ")).map_err(|e| {
606            if e.kind() == std::io::ErrorKind::PermissionDenied {
607                Error::Cgroup(
608                    "cannot enable cgroup controllers - run as root or configure systemd cgroup delegation".into()
609                )
610            } else {
611                Error::Cgroup(format!("failed to enable controllers: {e}"))
612            }
613        })?;
614
615        Ok(())
616    }
617
618    fn set_memory_limit(&self, cgroup_path: &Path, limit: MemoryLimit) -> Result<()> {
619        let bytes = limit.bytes();
620
621        // memory.high (~90%): soft limit that triggers reclaim/throttling before
622        // the hard cap, giving the process a chance to free memory gracefully
623        // instead of being killed outright. Best-effort.
624        let high = bytes / 100 * 90;
625        if high > 0 {
626            let _ = fs::write(cgroup_path.join("memory.high"), high.to_string());
627        }
628
629        // memory.max: hard cap. Process is OOM-killed if it exceeds this.
630        let memory_max = cgroup_path.join("memory.max");
631        fs::write(&memory_max, bytes.to_string())
632            .map_err(|e| Error::Cgroup(format!("failed to set memory.max: {e}")))?;
633
634        // memory.swap.max=0: prevent the limited process from spilling to swap, so
635        // memory.max is a true RAM ceiling rather than an invitation to thrash.
636        // Best-effort: absent on kernels without swap accounting.
637        let _ = fs::write(cgroup_path.join("memory.swap.max"), "0");
638
639        Ok(())
640    }
641
642    fn set_cpu_limit(&self, cgroup_path: &Path, limit: CpuLimit) -> Result<()> {
643        // cpu.max format: "$QUOTA $PERIOD" (in microseconds)
644        // e.g., "50000 100000" = 50% of one CPU
645        // For multi-core: 200% = 200000 quota with 100000 period
646        let period: u64 = 100_000; // 100ms
647        let quota = u64::from(limit.percent())
648            .checked_mul(period)
649            .map(|v| v / 100)
650            .ok_or_else(|| Error::InvalidCpu("CPU percentage too large".into()))?;
651
652        let cpu_max = cgroup_path.join("cpu.max");
653        fs::write(&cpu_max, format!("{quota} {period}"))
654            .map_err(|e| Error::Cgroup(format!("failed to set cpu.max: {e}")))?;
655        Ok(())
656    }
657
658    fn add_process(&self, cgroup_path: &Path, pid: u32) -> Result<()> {
659        let procs = cgroup_path.join("cgroup.procs");
660        fs::write(&procs, pid.to_string())
661            .map_err(|e| Error::Cgroup(format!("failed to add process {pid}: {e}")))?;
662        Ok(())
663    }
664
665    /// Write `io.max` one device per write (the kernel rejects a multi-line
666    /// write with EINVAL). Never fails: problems come back as warnings, since
667    /// memory and CPU limits still apply without I/O throttling.
668    fn set_io_limit(&self, cgroup_path: &Path, limit: IoLimit) -> Vec<String> {
669        let devices = match Self::get_real_block_devices() {
670            Ok(d) => d,
671            Err(e) => {
672                return vec![format!(
673                    "I/O limits were not applied (could not list block devices: {e}); memory and CPU limits still apply"
674                )]
675            }
676        };
677        if devices.is_empty() {
678            return vec![
679                "I/O limits were not applied (no eligible block devices found); memory and CPU limits still apply"
680                    .to_string(),
681            ];
682        }
683
684        let io_max = cgroup_path.join("io.max");
685        let mut warnings = Vec::new();
686        let mut first_err = None;
687        let mut applied = 0;
688        for line in io_max_lines(&devices, limit) {
689            match fs::write(&io_max, &line) {
690                Ok(()) => applied += 1,
691                Err(e) => {
692                    let dev = line.split_whitespace().next().unwrap_or_default();
693                    warnings.push(format!("I/O limit not applied to device {dev}: {e}"));
694                    first_err.get_or_insert(e);
695                }
696            }
697        }
698        if applied == 0 {
699            if let Some(e) = first_err {
700                warnings = vec![format!(
701                    "I/O limits were not applied to any device ({e}); memory and CPU limits still apply"
702                )];
703            }
704        }
705        for w in &warnings {
706            tracing::warn!("{w}");
707        }
708        warnings
709    }
710
711    /// Get block devices eligible for I/O throttling.
712    ///
713    /// Note: device-mapper (`dm-*`) devices are intentionally included — on the
714    /// very common LVM and LUKS-encrypted-root setups, filesystem I/O is issued
715    /// to a dm device, so excluding them would silently disable I/O limiting.
716    /// Only purely virtual/pseudo devices are skipped.
717    fn get_real_block_devices() -> Result<Vec<(u32, u32)>> {
718        let mut devices = Vec::new();
719
720        let sys_block = Path::new("/sys/block");
721        if !sys_block.exists() {
722            return Ok(devices);
723        }
724
725        for entry in fs::read_dir(sys_block)? {
726            let entry = entry?;
727            let name = entry.file_name();
728            let name_str = name.to_string_lossy();
729
730            // Skip virtual/pseudo devices that never carry real filesystem I/O.
731            if name_str.starts_with("loop")
732                || name_str.starts_with("ram")
733                || name_str.starts_with("nbd")
734                || name_str.starts_with("zram")
735            {
736                continue;
737            }
738
739            let dev_file = entry.path().join("dev");
740            if let Ok(content) = fs::read_to_string(&dev_file) {
741                if let Some((major, minor)) = content.trim().split_once(':') {
742                    if let (Ok(major), Ok(minor)) = (major.parse(), minor.parse()) {
743                        devices.push((major, minor));
744                    }
745                }
746            }
747        }
748
749        Ok(devices)
750    }
751}
752
753#[cfg(test)]
754mod tests {
755    use super::*;
756
757    #[test]
758    fn rejects_init_and_kernel_pids() {
759        assert!(reject_critical_pid(0).is_err()); // kernel/swapper
760        assert!(reject_critical_pid(1).is_err()); // init/systemd
761    }
762
763    #[test]
764    fn already_limited_hint_names_the_cgroup() {
765        let msg = already_limited(1234, "app-firefox").to_string();
766        assert!(
767            msg.contains("rlm unlimit --cgroup app-firefox"),
768            "hint must name the shared cgroup: {msg}"
769        );
770        assert!(!msg.contains("--pid"), "--pid does not work here: {msg}");
771        let run = already_limited(99, "run-7").to_string();
772        assert!(run.contains("rlm unlimit --cgroup run-7"), "{run}");
773    }
774
775    #[test]
776    fn allows_normal_pids() {
777        assert!(reject_critical_pid(2).is_ok());
778        assert!(reject_critical_pid(1234).is_ok());
779    }
780
781    #[test]
782    fn sanitize_rejects_traversal_and_separators() {
783        assert!(sanitize_cgroup_name("../etc").is_err());
784        assert!(sanitize_cgroup_name("a/b").is_err());
785        assert!(sanitize_cgroup_name("a\\b").is_err());
786        assert!(sanitize_cgroup_name("").is_err());
787        assert!(sanitize_cgroup_name("bad name").is_err()); // space
788    }
789
790    #[test]
791    fn sanitize_accepts_valid_names() {
792        assert_eq!(sanitize_cgroup_name("pid-1234").unwrap(), "pid-1234");
793        assert_eq!(sanitize_cgroup_name("app_firefox").unwrap(), "app_firefox");
794        assert_eq!(sanitize_cgroup_name("run-42-99").unwrap(), "run-42-99");
795    }
796
797    #[test]
798    fn io_max_is_one_line_per_device_without_newlines() {
799        let l = io_max_lines(
800            &[(8, 0), (259, 0)],
801            IoLimit {
802                read_bps: Some(5_242_880),
803                write_bps: Some(1_048_576),
804            },
805        );
806        assert_eq!(
807            l,
808            vec![
809                "8:0 rbps=5242880 wbps=1048576".to_string(),
810                "259:0 rbps=5242880 wbps=1048576".to_string()
811            ]
812        );
813    }
814
815    #[test]
816    fn released_processes_can_be_limited_again() {
817        let dir = tempfile::tempdir().unwrap();
818        for (name, procs) in [("unlimit", "4242\n"), ("pid-7", "7\n")] {
819            std::fs::create_dir(dir.path().join(name)).unwrap();
820            std::fs::write(dir.path().join(name).join("cgroup.procs"), procs).unwrap();
821        }
822        let m = CgroupManager::at(dir.path().to_path_buf());
823        assert_eq!(
824            m.find_cgroup_for_pid(4242),
825            None,
826            "the unlimit bucket is not a managed cgroup"
827        );
828        assert_eq!(m.find_cgroup_for_pid(7).as_deref(), Some("pid-7"));
829    }
830
831    #[test]
832    fn populated_cgroup_is_not_removed() {
833        let dir = tempfile::tempdir().unwrap();
834        let cg = dir.path().join("run-1-2");
835        std::fs::create_dir(&cg).unwrap();
836        std::fs::write(cg.join("cgroup.events"), "populated 1\nfrozen 0\n").unwrap();
837        let m = CgroupManager::at(dir.path().to_path_buf());
838        assert_eq!(m.is_populated("run-1-2"), Some(true));
839        assert!(!m.remove_if_empty("run-1-2").unwrap());
840        assert!(cg.exists());
841    }
842
843    #[test]
844    fn oom_kills_and_memory_max_are_read() {
845        let dir = tempfile::tempdir().unwrap();
846        let cg = dir.path().join("run-1-2");
847        std::fs::create_dir(&cg).unwrap();
848        std::fs::write(
849            cg.join("memory.events"),
850            "low 0\nhigh 3\nmax 10\noom 1\noom_kill 2\n",
851        )
852        .unwrap();
853        std::fs::write(cg.join("memory.max"), "157286400\n").unwrap();
854        let m = CgroupManager::at(dir.path().to_path_buf());
855        assert_eq!(m.oom_kills("run-1-2"), Some(2));
856        assert_eq!(m.memory_max("run-1-2"), Some(157_286_400));
857    }
858}