Skip to main content

rlmctl_core/guard/
effector.rs

1//! Executes [`Action`]s against real cgroups, acting in place on the cgroup a
2//! process already lives in (a systemd unit's scope/service, or an existing
3//! rlm rule cgroup) rather than moving it into an ephemeral `guard-<pid>`
4//! cgroup. Every action is best-effort and logged; a failure must never
5//! panic or otherwise crash the daemon loop. `apply` may return `Err` so the
6//! caller can log it. Desktop notifications are not sent from here: the
7//! daemon loop hands each applied action to `guard::notify`.
8//!
9//! # Write-ahead journal
10//! Freeze/Cap always `journal.append` (which fsyncs) *before* touching the
11//! cgroup, so a crash between the two still leaves a durable record that
12//! startup recovery (`sweep_leftovers`) can replay. `Journal` is internally
13//! mutex-serialized (see `journal.rs`), and the daemon calls into this
14//! `Effector` from a single thread (the tick loop in `rlm-guard`'s `main`),
15//! so `Effector` just holds a `&Journal` and relies on the daemon's
16//! single-threaded call discipline plus the journal's own internal locking
17//! for safety; it adds no locking of its own.
18//!
19//! # Mechanism: systemd unit vs. raw cgroupfs
20//! When a target resolved to a systemd unit (`Mechanism::Unit`), we prefer
21//! the D-Bus call (`FreezeUnit`/`ThawUnit`) with a hard
22//! 2s timeout (`systemd::SystemdUser`'s own enforced deadline); on `Err` (bus
23//! unavailable, call failed, or timed out) we fall back to the raw cgroupfs
24//! primitives in `cgfs`. Raw-mechanism targets (rlm's own rule cgroups) skip
25//! the D-Bus attempt entirely. Thawing is mechanism-independent: we always
26//! perform the raw `cgroup.freeze` write first, unconditionally, and only
27//! then attempt `ThawUnit` best-effort so systemd's view matches. This
28//! guarantees a cgroup is never left frozen because of a stale unit or a
29//! slow bus, and tolerates a missing cgroup (the raw write simply errors and
30//! we move on). On shutdown and startup replay every raw thaw and restore
31//! finishes before the first D-Bus call. Caps are the exception: they never
32//! go through systemd (see below).
33//!
34//! # Restoring `memory.high`
35//! The kernel truncates `memory.high` writes to page multiples, so the byte
36//! count we cap to is page-aligned *before* we journal/write it
37//! ([`page_align_down`]); as a second line of defense, [`Effector::cap`]
38//! reads `memory.high` back after writing and self-corrects the journal if
39//! reality still differs.
40//! Caps and restores are raw `memory.high` writes only, never systemd's
41//! `SetUnitProperties(MemoryHigh)`: a runtime property leaves a `/run`
42//! drop-in that outlives the cap and masks the unit's own configured
43//! `MemoryHigh` until reboot, and restoring through systemd wrote `infinity`
44//! over it. If systemd later re-applies the unit's value, that only lifts our
45//! cap early, and the journal's `our_high` check then skips the restore (the
46//! safe direction).
47//!
48//! If more than one journal entry ever coexists for the same
49//! cgroup (a leak from an incomplete prior removal), every restore path
50//! treats them as one chain: liveness is judged against the chain's `Cap`
51//! entry specifically (only a `Cap` has a `memory.high` to restore; a
52//! `Freeze` entry does not, so judging against `entries.last()` when it
53//! happens to be a newer `Freeze` would wrongly look like "nothing to
54//! restore" and strand `memory.high` at the guard's value forever), and the
55//! value restored is always the *oldest* `Cap` entry's `prev_high`, the
56//! true pre-intervention value, not an intermediate entry's `prev_high`
57//! (which is just our own previous `our_high`). See [`restore_target`].
58//! Liveness, however, is judged against the *newest* `Cap` entry, not the
59//! oldest: entries are strictly appended, so a later Cap's write always
60//! supersedes an earlier one's on disk, and `should_restore`'s string
61//! comparison must match what's actually there. See [`restore_decision`].
62
63use super::cgfs;
64use super::journal::{should_restore, Journal, JournalAction, JournalEntry};
65use super::resolve::{Mechanism, Resolution};
66use super::sampler::parse_meminfo;
67use super::systemd::SystemdUser;
68use super::types::Action;
69use crate::CgroupManager;
70use common::Result;
71use std::time::Duration;
72
73/// Floor for any soft cap. A cap below this is effectively a freeze for a
74/// desktop app, so small cgroups are never squeezed further than this.
75pub const MIN_CAP_BYTES: u64 = 256 * 1024 * 1024;
76
77/// What a successful [`Effector::apply`] did, beyond "it worked".
78#[derive(Debug, Clone, Copy, PartialEq, Eq)]
79pub struct Applied {
80    /// The `memory.high` a `Cap` wrote, in bytes. `None` for other actions.
81    pub cap_bytes: Option<u64>,
82}
83
84/// Hard deadline for every D-Bus call on the freeze-guard's storm path. On
85/// `Err` (including a timeout) callers fall back to raw cgroupfs writes.
86const DBUS_TIMEOUT: Duration = Duration::from_secs(2);
87
88/// Applies guard [`Action`]s in place on the resolved target cgroup, via the
89/// systemd-unit D-Bus path (with raw-cgroup fallback) and a write-ahead
90/// [`Journal`] for crash-safe restore.
91pub struct Effector<'a> {
92    /// Kept only for the legacy `guard-<pid>` sweep during the upgrade path
93    /// (see [`Effector::sweep_leftovers`]); acting in place no longer uses it.
94    manager: &'a CgroupManager,
95    journal: &'a Journal,
96    systemd: Option<&'a SystemdUser>,
97}
98
99impl<'a> Effector<'a> {
100    pub fn new(
101        manager: &'a CgroupManager,
102        journal: &'a Journal,
103        systemd: Option<&'a SystemdUser>,
104    ) -> Self {
105        Self {
106            manager,
107            journal,
108            systemd,
109        }
110    }
111
112    /// Apply a single action. Best-effort: returns `Err` only so the caller can
113    /// log it. On success, [`Applied::cap_bytes`] holds the `memory.high` a
114    /// `Cap` wrote.
115    pub fn apply(&self, action: &Action) -> Result<Applied> {
116        let done = Applied { cap_bytes: None };
117        match action {
118            Action::Freeze { res, name } => self.freeze(res, name).map(|()| done),
119            Action::Thaw { res } => self.thaw(res).map(|()| done),
120            Action::Cap { res, name } => self.cap(res, name).map(|bytes| Applied {
121                cap_bytes: Some(bytes),
122            }),
123            Action::LiftCap { res } => self.lift_cap(res).map(|()| done),
124        }
125    }
126
127    fn freeze(&self, res: &Resolution, name: &str) -> Result<()> {
128        // The inode is the guard that lets a later restore tell "this is
129        // still the same cgroup" from "this cgroup was torn down and
130        // recreated" (`should_restore`). `unwrap_or(0)` used to substitute a
131        // poison sentinel here: 0 is never a real inode, so the guard could
132        // never match again and the entry became permanently unrestorable,
133        // yet it was still journaled-and-acted-on, then later cleared and
134        // logged as if it were an intentional skip (Promoted Minor A). Fail
135        // closed instead: refuse to freeze at all rather than act with a
136        // record we can never safely restore from.
137        let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
138            tracing::warn!(
139                cgroup = %res.cgroup, name,
140                "cannot read cgroup inode; refusing to freeze (would be unrestorable)"
141            );
142            return Err(common::Error::Cgroup(format!(
143                "cannot read inode for {}; refusing to freeze",
144                res.cgroup
145            )));
146        };
147        let entry = JournalEntry {
148            cgroup: res.cgroup.clone(),
149            inode,
150            unit: res.unit.clone(),
151            action: JournalAction::Freeze,
152            prev_high: None,
153            our_high: None,
154        };
155        // Write-ahead: the entry must be durable (journal.append fsyncs)
156        // before we ever touch the cgroup, so a crash between the two still
157        // leaves a record startup recovery can act on.
158        self.journal.append(&entry)?;
159
160        tracing::info!(cgroup = %res.cgroup, name, "freezing cgroup");
161        if res.mechanism == Mechanism::Unit {
162            if let (Some(unit), Some(systemd)) = (&res.unit, self.systemd) {
163                match systemd.freeze_unit(unit, DBUS_TIMEOUT) {
164                    Ok(()) => return Ok(()),
165                    Err(e) => tracing::warn!(
166                        cgroup = %res.cgroup, unit, error = %e,
167                        "FreezeUnit failed; falling back to raw cgroup.freeze"
168                    ),
169                }
170            }
171        }
172        cgfs::write_freeze(&res.cgroup, true)
173    }
174
175    fn thaw(&self, res: &Resolution) -> Result<()> {
176        tracing::info!(cgroup = %res.cgroup, "thawing cgroup");
177        // Gather any coexisting entries for this cgroup first: a `Thaw`
178        // action reverses a Frozen intervention, but if a Cap entry ever
179        // leaked alongside it (Important #3, Task 6 review) we must still
180        // restore its memory.high before wiping the journal record below,
181        // not just drop it silently.
182        let entries = self.entries_for(&res.cgroup);
183        let result = self.thaw_raw(&res.cgroup, res.unit.as_deref());
184        if let Err(e) = &result {
185            tracing::debug!(cgroup = %res.cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
186        }
187        self.restore_high_if_any(&res.cgroup, &entries);
188        self.journal.remove(&res.cgroup)?;
189        result
190    }
191
192    /// Returns the `memory.high` value written, in bytes.
193    fn cap(&self, res: &Resolution, name: &str) -> Result<u64> {
194        // See `freeze`'s matching comment (Promoted Minor A): a `0` inode
195        // sentinel here would make this Cap permanently unrestorable while
196        // looking like a real guard, so fail closed instead of
197        // journal-and-act with an unrestorable record.
198        let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
199            tracing::warn!(
200                cgroup = %res.cgroup, name,
201                "cannot read cgroup inode; refusing to cap (would be unrestorable)"
202            );
203            return Err(common::Error::Cgroup(format!(
204                "cannot read inode for {}; refusing to cap",
205                res.cgroup
206            )));
207        };
208        let prev_high = cgfs::read_high(&res.cgroup);
209        // Unknown swap total counts as 0: assuming anon is pinned only makes
210        // the cap gentler.
211        let swap_total_kb = std::fs::read_to_string("/proc/meminfo")
212            .ok()
213            .and_then(|m| parse_meminfo(&m))
214            .map_or(0, |m| m.swap_total_kb);
215        let anon_ok = cgfs::anon_reclaimable(&res.cgroup, swap_total_kb);
216        let Some(target) = cap_target(
217            cgfs::current_bytes(&res.cgroup),
218            cgfs::file_bytes(&res.cgroup),
219            anon_ok,
220        ) else {
221            tracing::warn!(
222                cgroup = %res.cgroup, name,
223                "cannot read memory.current; refusing to cap"
224            );
225            return Err(common::Error::Cgroup(
226                "cannot read memory.current; refusing to cap".into(),
227            ));
228        };
229        // The kernel truncates `memory.high` writes to page multiples, so we
230        // must journal/write the value it will actually store, not the raw
231        // target, or `should_restore`'s string-equality check can never
232        // pass again and the cap becomes permanent.
233        let our_bytes = page_align_down(target, page_size());
234        // Plain decimal bytes, no separators/whitespace: this must be
235        // exactly what a later `cgfs::read_high` (which only trims
236        // whitespace off the raw file contents) returns. See
237        // `our_high_string_is_plain_decimal_no_separators` below, and
238        // `reconcile_our_high` for the belt-and-braces check.
239        if !cap_tightens(prev_high.as_deref(), our_bytes) {
240            tracing::info!(
241                cgroup = %res.cgroup, name, prev_high = ?prev_high, our_bytes,
242                "existing memory.high already at or below the cap; not capping"
243            );
244            return Err(common::Error::Cgroup(
245                "existing memory.high already at or below the cap; refusing to cap".into(),
246            ));
247        }
248        let our_high = our_bytes.to_string();
249
250        let entry = JournalEntry {
251            cgroup: res.cgroup.clone(),
252            inode,
253            unit: res.unit.clone(),
254            action: JournalAction::Cap,
255            prev_high,
256            our_high: Some(our_high.clone()),
257        };
258        self.journal.append(&entry)?;
259
260        tracing::info!(
261            cgroup = %res.cgroup, name, our_high = %our_high, anon_reclaimable = anon_ok,
262            "soft-capping cgroup"
263        );
264        cgfs::write_high(&res.cgroup, &our_high)?;
265        self.reconcile_our_high(&entry);
266        Ok(our_bytes)
267    }
268
269    /// After a successful `Cap` write, read `memory.high` back; if what's
270    /// actually on disk differs from what we journaled (page truncation we
271    /// didn't fully pre-empt), correct the journal to match reality. Otherwise
272    /// `should_restore`'s string-equality check can never pass again and
273    /// the cap becomes permanent (Task 6 review, Critical #1). Rebuilds
274    /// only this cgroup's entries, preserving any others that might coexist
275    /// (Important #3), with `written`'s `our_high` corrected to the
276    /// read-back value. Uses `Journal::replace` (a single atomic rewrite),
277    /// not a separate `remove` then `append`: the latter has a window where
278    /// the cgroup has no journal record at all, so a crash/SIGKILL/append
279    /// failure right there would leave a permanent, unrecoverable cap,
280    /// exactly the invariant the journal exists to prevent (Task 6 review,
281    /// fix round 2).
282    fn reconcile_our_high(&self, written: &JournalEntry) {
283        let cgroup: &str = &written.cgroup;
284        let Some(actual) = cgfs::read_high(cgroup) else {
285            return;
286        };
287        if written.our_high.as_deref() == Some(actual.as_str()) {
288            return;
289        }
290        tracing::warn!(
291            cgroup = %cgroup, journaled = ?written.our_high, actual = %actual,
292            "memory.high on disk differs from what we journaled; correcting journal entry"
293        );
294        let mut entries = self.entries_for(cgroup);
295        let Some(pos) = entries.iter().rposition(|e| e == written) else {
296            // Already removed/replaced by something else (e.g. a concurrent
297            // Thaw/LiftCap); nothing left to correct.
298            return;
299        };
300        entries[pos].our_high = Some(actual);
301        if let Err(e) = self.journal.replace(cgroup, &entries) {
302            tracing::warn!(cgroup = %cgroup, error = %e, "failed to correct journal entry (atomic replace)");
303        }
304    }
305
306    fn lift_cap(&self, res: &Resolution) -> Result<()> {
307        tracing::info!(cgroup = %res.cgroup, "lifting cap");
308        let entries = self.entries_for(&res.cgroup);
309        // Mechanism-independent thaw always runs, regardless of whether any
310        // journal record exists: a dead-cgroup prune (carry-forward
311        // finding, Task 5 review) must never leave a target frozen just
312        // because we lost the journal entry.
313        let _ = self.thaw_raw(&res.cgroup, res.unit.as_deref());
314        self.restore_high_if_any(&res.cgroup, &entries);
315        self.journal.remove(&res.cgroup)
316    }
317
318    /// All journal entries for one cgroup, in the order `Journal::entries()`
319    /// returns them, oldest-first, since entries are strictly appended.
320    fn entries_for(&self, cgroup: &str) -> Vec<JournalEntry> {
321        self.journal
322            .entries()
323            .into_iter()
324            .filter(|e| e.cgroup == cgroup)
325            .collect()
326    }
327
328    /// Startup recovery: legacy `guard-<pid>` sweep (kept for one release as
329    /// an upgrade path from pre-act-in-place rlm), then replay every live
330    /// journal entry.
331    pub fn sweep_leftovers(&self) -> Result<()> {
332        if let Err(e) = self.manager.sweep_guard_leftovers() {
333            tracing::warn!(error = %e, "legacy guard-<pid> sweep failed (non-fatal)");
334        }
335        self.replay_and_clear()
336    }
337
338    /// Graceful shutdown: undo every live journal entry, then clear the
339    /// journal. Same replay as `sweep_leftovers`, minus the legacy sweep.
340    pub fn undo_all(&self) -> Result<()> {
341        self.replay_and_clear()
342    }
343
344    fn replay_and_clear(&self) -> Result<()> {
345        // Group by cgroup first (entries for the same cgroup aren't
346        // necessarily contiguous in the file) so each cgroup's chain is
347        // replayed as one unit (see `restore_high_if_any`) rather than
348        // entry-by-entry, which would mis-restore whenever more than one
349        // entry coexists for a cgroup (Important #3, Task 6 review).
350        let mut by_cgroup: std::collections::HashMap<String, Vec<JournalEntry>> =
351            std::collections::HashMap::new();
352        for e in self.journal.entries() {
353            by_cgroup.entry(e.cgroup.clone()).or_default().push(e);
354        }
355        // Pass 1: every raw thaw and memory.high restore, then clear the
356        // journal. No D-Bus call happens before this is done, so a slow
357        // session bus cannot push the undo past systemd's stop timeout.
358        for (cgroup, entries) in &by_cgroup {
359            if let Err(e) = cgfs::write_freeze(cgroup, false) {
360                tracing::debug!(cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
361            }
362            self.restore_high_if_any(cgroup, entries);
363        }
364        let cleared = self.journal.clear();
365        // Pass 2, best effort: tell systemd so its view of the units
366        // matches the kernel's. The processes already run again.
367        if let Some(systemd) = self.systemd {
368            for (cgroup, entries) in &by_cgroup {
369                if let Some(unit) = entries.last().and_then(|e| e.unit.as_deref()) {
370                    if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
371                        tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
372                    }
373                }
374            }
375        }
376        cleared
377    }
378
379    /// Restore `memory.high` for one cgroup's journal `entries` (oldest-first),
380    /// if the chain is still live. Called *after* the caller has already
381    /// performed the mechanism-independent thaw (see module docs: no path
382    /// here ever means "leave frozen"). All the decision logic is delegated
383    /// to the pure [`restore_decision`]: liveness is judged against the
384    /// *newest* `Cap` entry (whose write is what's actually on disk right
385    /// now), while the value restored is [`restore_target`]'s *oldest*-entry
386    /// `prev_high` (the true pre-intervention value). Judging liveness
387    /// against the oldest Cap instead (NEW-1 regression) leaves a stacked
388    /// `[Cap, Cap]` chain's on-disk value permanently un-restorable, since
389    /// the oldest entry's `our_high` never matches what a later Cap actually
390    /// wrote. Judging liveness against `entries.last()` has the same failure
391    /// for a `[Cap, Freeze]` chain (Promoted Minor B): the newest entry is
392    /// the `Freeze`, which has no `memory.high` of its own. The restore is a
393    /// raw `memory.high` write only; systemd's `MemoryHigh` property is never
394    /// touched (see module docs).
395    fn restore_high_if_any(&self, cgroup: &str, entries: &[JournalEntry]) {
396        let inode = cgfs::dir_inode(cgroup);
397        let high = cgfs::read_high(cgroup);
398        match restore_decision(entries, inode, high.as_deref()) {
399            RestoreStep::ThawAndRestoreHigh { to } => {
400                if let Err(err) = cgfs::write_high(cgroup, &to) {
401                    tracing::warn!(cgroup, error = %err, "failed to restore memory.high");
402                }
403            }
404            // No Cap entry anywhere in the chain (a Freeze-only chain,
405            // possibly with leaked duplicates): there is no memory.high to
406            // restore. The unconditional thaw already ran in the caller, so
407            // there's nothing left to do here.
408            RestoreStep::ThawOnly => {}
409            RestoreStep::SkipRemove => {
410                tracing::warn!(
411                    cgroup,
412                    "not restoring memory.high (cgroup recreated or value changed since our write)"
413                );
414            }
415        }
416    }
417
418    /// Mechanism-independent thaw: an *unconditional* raw `cgroup.freeze`
419    /// write first, so a cgroup is never left frozen and a slow bus cannot
420    /// delay it, then a best-effort `ThawUnit` (if we have a unit and a bus)
421    /// so systemd's view of the unit matches. Tolerates a missing cgroup:
422    /// the raw write then simply returns `Err`, which every caller here
423    /// treats as non-fatal.
424    fn thaw_raw(&self, cgroup: &str, unit: Option<&str>) -> Result<()> {
425        let result = cgfs::write_freeze(cgroup, false);
426        if let (Some(unit), Some(systemd)) = (unit, self.systemd) {
427            if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
428                tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
429            }
430        }
431        result
432    }
433}
434
435/// What to do with one journal entry's `memory.high` at restore time. Never
436/// speaks to freeze/thaw; that is unconditional and already handled by the
437/// caller (`Effector::thaw_raw`) before `restore_step` is even consulted, so
438/// no variant here can ever mean "leave it frozen".
439#[derive(Debug, PartialEq, Eq)]
440pub enum RestoreStep {
441    /// A `Freeze` entry whose guard still holds: nothing to restore, thaw
442    /// (already done by the caller) was the whole job.
443    ThawOnly,
444    /// A `Cap` entry whose guard still holds: restore `memory.high` to `to`
445    /// (the entry's `prev_high`, "max" if it was never recorded).
446    ThawAndRestoreHigh { to: String },
447    /// The guard no longer holds (cgroup recreated, or someone else changed
448    /// `memory.high` since our write): don't touch `memory.high`, just let
449    /// the caller remove the journal entry.
450    SkipRemove,
451}
452
453/// Pure: what to do for one journal entry at restore time, delegating the
454/// safety check to [`should_restore`].
455pub fn restore_step(e: &JournalEntry, inode: Option<u64>, high: Option<&str>) -> RestoreStep {
456    if !should_restore(e, inode, high) {
457        return RestoreStep::SkipRemove;
458    }
459    match e.action {
460        JournalAction::Freeze => RestoreStep::ThawOnly,
461        JournalAction::Cap => RestoreStep::ThawAndRestoreHigh {
462            to: e.prev_high.clone().unwrap_or_else(|| "max".into()),
463        },
464    }
465}
466
467/// Size a soft cap that slows an app down without stalling it.
468///
469/// memory.high applies to everything charged to the cgroup, page cache
470/// included, so the cap is sized from memory.current: it never asks the
471/// kernel to reclaim more than 10% of it. When anon memory cannot go to swap,
472/// only file pages can be reclaimed, so the cap never asks for more than 80%
473/// of them. Returns `None` when memory.current is unreadable: no cap is safer
474/// than a guessed one.
475pub fn cap_target(current: Option<u64>, file: Option<u64>, anon_reclaimable: bool) -> Option<u64> {
476    let current = current?;
477    let mut target = current / 10 * 9;
478    if !anon_reclaimable {
479        let file_floor = current.saturating_sub(file.unwrap_or(0).saturating_mul(8) / 10);
480        target = target.max(file_floor);
481    }
482    Some(target.max(MIN_CAP_BYTES))
483}
484
485/// Pure: whether writing `target` would tighten the existing `memory.high`.
486/// A numeric `prev_high` at or below `target` means the cap would loosen (or
487/// not change) the limit already in place, so the caller must not write it.
488/// `"max"`, or anything else that is not a byte count, is no limit at all.
489pub fn cap_tightens(prev_high: Option<&str>, target: u64) -> bool {
490    match prev_high.and_then(|s| s.trim().parse::<u64>().ok()) {
491        Some(prev) => prev > target,
492        None => true,
493    }
494}
495
496/// Pure: the full restore decision for one cgroup's journal `entries`
497/// (oldest-first), given the cgroup's current inode and on-disk
498/// `memory.high` (both already read by the caller, no IO here). Answers two
499/// orthogonal questions with two different entries on purpose:
500///
501/// - **Liveness** (is the guard we wrote still intact?) is judged against
502///   the *newest* `Cap` entry in the chain. Entries are strictly appended,
503///   so a later Cap's write always supersedes an earlier one's on disk, so
504///   `should_restore`'s string-equality check must compare against the
505///   entry whose `our_high` is what's actually there right now.
506/// - **Value** (what do we restore to?) is [`restore_target`]'s *oldest*
507///   `Cap` entry's `prev_high`, the true pre-intervention value, not an
508///   intermediate entry's `prev_high` (which is just our own previous
509///   `our_high` from an earlier cap in the same chain).
510///
511/// Judging both against the oldest Cap (NEW-1 regression) makes a stacked
512/// `[Cap, Cap]` chain's liveness check compare a stale `our_high` against
513/// the newer write on disk, so it never matches and the chain is treated as
514/// dead, stranding `memory.high` at the guard's value forever. Judging
515/// liveness against `entries.last()` fails the same way for `[Cap, Freeze]`
516/// (Promoted Minor B): the newest entry is the `Freeze`, which has no
517/// `memory.high` of its own. Returns [`RestoreStep::ThawOnly`] when there's
518/// no `Cap` entry anywhere in the chain (a `Freeze`-only chain): nothing to
519/// restore beyond the caller's unconditional thaw.
520pub fn restore_decision(
521    entries: &[JournalEntry],
522    inode: Option<u64>,
523    high: Option<&str>,
524) -> RestoreStep {
525    let Some(newest_cap) = entries
526        .iter()
527        .rev()
528        .find(|e| e.action == JournalAction::Cap)
529    else {
530        return RestoreStep::ThawOnly;
531    };
532    match restore_step(newest_cap, inode, high) {
533        RestoreStep::ThawAndRestoreHigh { .. } => match restore_target(entries) {
534            Some(to) => RestoreStep::ThawAndRestoreHigh { to },
535            // Unreachable: `newest_cap` being a Cap guarantees `restore_target`
536            // (which only needs *any* Cap entry) also finds one.
537            None => RestoreStep::ThawOnly,
538        },
539        other => other,
540    }
541}
542
543/// Pure: given all journal entries for one cgroup, oldest-first, select the
544/// `memory.high` value to restore, if the chain contains a `Cap` entry. The
545/// OLDEST `Cap` entry's `prev_high` is the true pre-intervention value; a
546/// later entry's `prev_high` is just our own previous `our_high` from an
547/// earlier cap in the same chain, not the original (Task 6 review,
548/// Important #3). Returns `None` if there's no `Cap` entry (e.g. a
549/// `Freeze`-only chain).
550pub fn restore_target(entries: &[JournalEntry]) -> Option<String> {
551    entries
552        .iter()
553        .find(|e| e.action == JournalAction::Cap)
554        .map(|e| e.prev_high.clone().unwrap_or_else(|| "max".into()))
555}
556
557/// Runtime page size in bytes, via `sysconf(_SC_PAGESIZE)`. Falls back to
558/// 4096 (by far the most common value) only if the syscall ever returns
559/// something nonsensical, a defensive fallback, not the source of truth,
560/// since the whole point is to match whatever the kernel actually enforces.
561fn page_size() -> u64 {
562    // SAFETY: sysconf(_SC_PAGESIZE) reads a static system parameter; no
563    // pointers involved, no side effects.
564    let p = unsafe { libc::sysconf(libc::_SC_PAGESIZE) };
565    if p > 0 {
566        p as u64
567    } else {
568        4096
569    }
570}
571
572/// Pure: round `bytes` down to a multiple of `page` (a no-op if `page` is 0).
573/// The kernel truncates `memory.high` writes to page multiples (Task 6
574/// review, Critical #1), so we must journal/write the value it will
575/// actually store, not the pre-truncation target; otherwise
576/// `should_restore`'s string-equality check can never pass again and a cap
577/// becomes permanent.
578fn page_align_down(bytes: u64, page: u64) -> u64 {
579    bytes.checked_div(page).map_or(bytes, |q| q * page)
580}
581
582#[cfg(test)]
583mod tests {
584    use super::super::resolve::{Coverage, Verdict};
585    use super::*;
586
587    const MIB: u64 = 1024 * 1024;
588    const GIB: u64 = 1024 * MIB;
589
590    #[test]
591    fn cap_never_demands_more_than_ten_percent_of_current() {
592        // Incident: Chrome scope with 1.5 GiB charged, mostly page cache, ~150 MiB anon.
593        let current = GIB * 3 / 2;
594        let cap = cap_target(Some(current), Some(GIB * 135 / 100), true).unwrap();
595        assert!(
596            cap >= current / 10 * 9,
597            "cap {cap} is below 90% of {current}"
598        );
599    }
600
601    #[test]
602    fn swapless_cap_only_asks_for_reclaimable_file_pages() {
603        assert_eq!(
604            cap_target(Some(GIB), Some(0), false),
605            Some(GIB),
606            "nothing reclaimable: demand nothing"
607        );
608        assert_eq!(
609            cap_target(Some(GIB), Some(512 * MIB), false),
610            Some(GIB / 10 * 9)
611        );
612        assert_eq!(
613            cap_target(Some(GIB), None, false),
614            Some(GIB),
615            "unknown file size: demand nothing"
616        );
617    }
618
619    #[test]
620    fn cap_has_a_256_mib_floor() {
621        assert_eq!(
622            cap_target(Some(100 * MIB), Some(0), true),
623            Some(MIN_CAP_BYTES)
624        );
625    }
626
627    #[test]
628    fn cap_never_loosens_an_existing_memory_high() {
629        let floor = cap_target(Some(100 * MIB), Some(0), true).unwrap();
630        assert_eq!(floor, MIN_CAP_BYTES);
631        let prev = (200 * MIB).to_string();
632        assert!(
633            !cap_tightens(Some(&prev), floor),
634            "200 MiB unit limit must not be raised to the 256 MiB floor"
635        );
636        assert!(
637            !cap_tightens(Some(&floor.to_string()), floor),
638            "equal: no-op"
639        );
640        assert!(cap_tightens(Some("max"), floor));
641        assert!(cap_tightens(None, floor));
642        let prev = (2 * GIB).to_string();
643        assert!(cap_tightens(Some(&prev), GIB));
644    }
645
646    #[test]
647    fn unreadable_current_refuses_to_cap() {
648        assert_eq!(cap_target(None, Some(1), true), None);
649    }
650
651    /// Carry-forward (Task 3 review): `our_high` must be a plain decimal
652    /// string with no grouping/whitespace, since `should_restore` compares
653    /// it by string equality against `cgfs::read_high` (which only trims).
654    #[test]
655    fn our_high_string_is_plain_decimal_no_separators() {
656        let bytes = cap_target(Some(12_345_678_900), None, true).unwrap();
657        let s = bytes.to_string();
658        assert!(
659            s.chars().all(|c| c.is_ascii_digit()),
660            "our_high must be plain digits, got {s:?}"
661        );
662        assert_eq!(s, format!("{bytes}"), "no formatting beyond plain decimal");
663    }
664
665    /// Task 6 review, Critical #1: the kernel truncates `memory.high`
666    /// writes to page multiples (the reviewer's own observed example:
667    /// writing 900_000_000 with a 4096-byte page reads back 899_997_696).
668    #[test]
669    fn page_align_down_rounds_to_page_multiple() {
670        assert_eq!(page_align_down(900_000_000, 4096), 899_997_696);
671        assert_eq!(
672            page_align_down(4096, 4096),
673            4096,
674            "already-aligned is a no-op"
675        );
676        assert_eq!(page_align_down(100, 0), 100, "page=0 guard is a no-op");
677    }
678
679    /// Task 6 review, Important #3: when duplicate/stacked `Cap` entries end
680    /// up coexisting for one cgroup (a leak from an incomplete prior
681    /// removal), the restore target must be the OLDEST entry's `prev_high`
682    /// (the true pre-intervention value). The newer entry's `prev_high` is
683    /// deliberately a different-looking value here to prove we don't
684    /// chain-follow it.
685    #[test]
686    fn restore_target_uses_oldest_caps_prev_high() {
687        let oldest = entry_cap("/x", 42, "max", "A");
688        let newest = entry_cap("/x", 42, "A-prime", "B");
689        assert_eq!(
690            restore_target(&[oldest, newest]),
691            Some("max".into()),
692            "must use the oldest entry's prev_high, not the newest's"
693        );
694    }
695
696    #[test]
697    fn restore_target_none_for_freeze_only_chain() {
698        assert_eq!(restore_target(&[entry_freeze("/x", 42)]), None);
699    }
700
701    /// Regression test for NEW-1: `restore_decision`'s liveness check must be
702    /// judged against the NEWEST `Cap` entry (whose `our_high` is what's
703    /// actually on disk), while the value restored stays the OLDEST `Cap`
704    /// entry's `prev_high`. A stacked `[Cap(prev="max", our="A"),
705    /// Cap(prev="A", our="B")]` chain with disk `memory.high == "B"` must
706    /// restore to `"max"`, not `SkipRemove`, which the old
707    /// `entries.iter().find()` (oldest-Cap-for-everything) produced: it
708    /// compared the oldest entry's `our_high` ("A") against disk ("B"),
709    /// never matched, and permanently stranded `memory.high`. Also covers
710    /// `[Cap, Freeze]` (Promoted Minor B's shape) and a bare `[Cap]` chain in
711    /// the same test so a future re-swap of either selection can't slip by.
712    #[test]
713    fn restore_decision_liveness_uses_newest_cap_value_uses_oldest() {
714        // [Cap, Cap]: disk holds the NEWEST Cap's our_high ("B"). Liveness
715        // must be checked against "B", not the oldest entry's "A".
716        let oldest = entry_cap("/x", 42, "max", "A");
717        let newest = entry_cap("/x", 42, "A", "B");
718        assert_eq!(
719            restore_decision(&[oldest.clone(), newest.clone()], Some(42), Some("B")),
720            RestoreStep::ThawAndRestoreHigh { to: "max".into() },
721            "must judge liveness against the newest Cap's our_high (matches disk \"B\"), \
722             but restore the oldest Cap's prev_high (\"max\")"
723        );
724        // Sanity: the stale intermediate value "A" is no longer live on disk,
725        // so checking against it (the old bug) would report SkipRemove.
726        assert_eq!(
727            restore_step(&oldest, Some(42), Some("B")),
728            RestoreStep::SkipRemove,
729            "confirms the bug this guards against: judging liveness against the oldest \
730             entry's our_high against the newer on-disk value mismatches"
731        );
732
733        // [Cap, Freeze]: newest entry has no memory.high of its own; must
734        // still fall through to the Cap for both liveness and value.
735        let cap = entry_cap("/x", 42, "max", "1000");
736        let frz = entry_freeze("/x", 42);
737        assert_eq!(
738            restore_decision(&[cap, frz], Some(42), Some("1000")),
739            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
740        );
741
742        // [Cap] alone: baseline single-entry behavior is unchanged.
743        let cap_only = entry_cap("/x", 42, "max", "1000");
744        assert_eq!(
745            restore_decision(&[cap_only], Some(42), Some("1000")),
746            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
747        );
748
749        // Freeze-only chain: nothing to restore.
750        assert_eq!(
751            restore_decision(&[entry_freeze("/x", 42)], Some(42), None),
752            RestoreStep::ThawOnly
753        );
754    }
755
756    fn entry_cap(cg: &str, inode: u64, prev: &str, our: &str) -> JournalEntry {
757        JournalEntry {
758            cgroup: cg.into(),
759            inode,
760            unit: None,
761            action: JournalAction::Cap,
762            prev_high: Some(prev.into()),
763            our_high: Some(our.into()),
764        }
765    }
766
767    fn entry_freeze(cg: &str, inode: u64) -> JournalEntry {
768        JournalEntry {
769            cgroup: cg.into(),
770            inode,
771            unit: None,
772            action: JournalAction::Freeze,
773            prev_high: None,
774            our_high: None,
775        }
776    }
777
778    #[test]
779    fn restore_step_matrix() {
780        let cap = entry_cap("/x", 42, "max", "1000");
781        assert_eq!(
782            restore_step(&cap, Some(42), Some("1000")),
783            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
784        );
785        assert_eq!(
786            restore_step(&cap, Some(43), Some("1000")),
787            RestoreStep::SkipRemove
788        );
789        assert_eq!(
790            restore_step(&cap, Some(42), Some("777")),
791            RestoreStep::SkipRemove
792        );
793        let frz = entry_freeze("/x", 42);
794        assert_eq!(
795            restore_step(&frz, Some(42), None),
796            RestoreStep::ThawOnly,
797            "alive freeze entry: nothing to restore beyond the unconditional thaw"
798        );
799        assert_eq!(
800            restore_step(&frz, None, None),
801            RestoreStep::SkipRemove,
802            "dead cgroup: restore_step only decides memory.high, never freeze; the \
803             unconditional thaw in Effector::thaw_raw already ran before this is consulted, \
804             so a still-frozen dead cgroup is never left behind (carry-forward: Task 5 review)"
805        );
806    }
807
808    /// Poll `cgfs::read_frozen` until it matches `want` or `timeout` elapses.
809    /// Writing `cgroup.freeze` only *requests* a state change; `cgroup.events`'s
810    /// `frozen` field (what `read_frozen` reads) only flips once the kernel has
811    /// actually quiesced every task in the cgroup, which can lag the write by a
812    /// few milliseconds under load, so a bare immediate read is flaky.
813    fn wait_for_frozen(cgroup: &str, want: bool, timeout: Duration) -> Option<bool> {
814        let deadline = std::time::Instant::now() + timeout;
815        loop {
816            let got = cgfs::read_frozen(cgroup);
817            if got == Some(want) || std::time::Instant::now() >= deadline {
818                return got;
819            }
820            std::thread::sleep(Duration::from_millis(20));
821        }
822    }
823
824    fn test_resolution(cgroup: String) -> Resolution {
825        Resolution {
826            cgroup,
827            unit: None,
828            verdict: Verdict::Freeze,
829            coverage: Coverage::Full,
830            mechanism: Mechanism::Raw,
831        }
832    }
833
834    /// Integration smoke test: freeze a real `sleep` via its (raw, rlm-created)
835    /// cgroup through the journal-backed Effector, confirm it's paused via
836    /// `cgroup.freeze` and journaled, then thaw and confirm the journal entry
837    /// is gone. Only works under cgroup v2 delegation, so it's `#[ignore]`d.
838    #[test]
839    #[ignore = "requires cgroup v2 delegation; run manually"]
840    fn freeze_thaw_real_process_raw_cgroup() {
841        use common::Limit;
842        use std::process::Command;
843
844        let manager = CgroupManager::new().expect("create CgroupManager");
845        let journal_dir = tempfile::tempdir().unwrap();
846        let journal =
847            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
848        let effector = Effector::new(&manager, &journal, None);
849
850        let abs_path = manager
851            .prepare_cgroup("test-freeze-thaw", &Limit::default())
852            .expect("create test cgroup")
853            .path;
854        let cgroup = format!(
855            "/{}",
856            abs_path
857                .strip_prefix("/sys/fs/cgroup")
858                .expect("cgroup under /sys/fs/cgroup")
859                .display()
860        );
861
862        let mut child = Command::new("sleep")
863            .arg("30")
864            .spawn()
865            .expect("spawn sleep");
866        let pid = child.id();
867        manager
868            .add_to_cgroup(&abs_path, pid)
869            .expect("add sleep to test cgroup");
870
871        let res = test_resolution(cgroup.clone());
872
873        effector
874            .apply(&Action::Freeze {
875                res: res.clone(),
876                name: "sleep".into(),
877            })
878            .expect("freeze");
879
880        assert_eq!(
881            wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
882            Some(true),
883            "cgroup should be frozen"
884        );
885        assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
886
887        effector
888            .apply(&Action::Thaw { res: res.clone() })
889            .expect("thaw");
890
891        assert_eq!(
892            wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
893            Some(false),
894            "cgroup should be thawed"
895        );
896        assert!(
897            journal.entries().is_empty(),
898            "journal entry should be removed after thaw"
899        );
900
901        let _ = child.kill();
902        let _ = child.wait();
903        let _ = manager.cleanup_cgroup("test-freeze-thaw");
904    }
905
906    /// Act-in-place integration test: freeze/thaw a real transient systemd
907    /// `--user --scope` unit through the Unit mechanism (D-Bus, with raw
908    /// fallback), confirming the journal-backed round trip end to end.
909    /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
910    #[test]
911    #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
912    fn freeze_thaw_real_transient_scope() {
913        use std::process::Command;
914
915        let uid_out = Command::new("id").arg("-u").output().expect("id -u");
916        let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
917            .trim()
918            .parse()
919            .expect("parse uid");
920
921        let unit_base = format!("rlm-e2e-{}", std::process::id());
922        let unit = format!("{unit_base}.scope");
923        let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
924
925        let mut child = Command::new("systemd-run")
926            .args([
927                "--user",
928                "--scope",
929                "--slice=app.slice",
930                &format!("--unit={unit_base}"),
931                "--",
932                "sleep",
933                "30",
934            ])
935            .spawn()
936            .expect("spawn systemd-run --scope");
937
938        // Give systemd a moment to register the transient scope's cgroup.
939        std::thread::sleep(Duration::from_millis(300));
940
941        let manager = CgroupManager::new().expect("create CgroupManager");
942        let journal_dir = tempfile::tempdir().unwrap();
943        let journal =
944            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
945        let systemd = SystemdUser::connect();
946        let effector = Effector::new(&manager, &journal, systemd.as_ref());
947
948        let res = Resolution {
949            cgroup: cgroup.clone(),
950            unit: Some(unit),
951            verdict: Verdict::Freeze,
952            coverage: Coverage::Full,
953            mechanism: Mechanism::Unit,
954        };
955
956        effector
957            .apply(&Action::Freeze {
958                res: res.clone(),
959                name: "sleep".into(),
960            })
961            .expect("freeze");
962        assert_eq!(
963            wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
964            Some(true),
965            "scope cgroup should be frozen"
966        );
967        assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
968
969        effector.apply(&Action::Thaw { res }).expect("thaw");
970        assert_eq!(
971            wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
972            Some(false),
973            "scope cgroup should be thawed"
974        );
975        assert!(
976            journal.entries().is_empty(),
977            "journal entry should be removed after thaw"
978        );
979
980        let _ = child.kill();
981        let _ = child.wait();
982        let _ = Command::new("systemctl")
983            .args(["--user", "stop", &format!("{unit_base}.scope")])
984            .status();
985    }
986
987    /// Regression test for Task 6 review Critical #1: cap a real cgroup and
988    /// assert `cgfs::read_high` matches the journaled `our_high` exactly,
989    /// i.e. the value we wrote never got silently page-truncated out from
990    /// under the journal. The target process holds enough random anon
991    /// memory (well above the 256 MiB `MIN_CAP_BYTES` floor, a round power
992    /// of two that would pass even without the fix and prove nothing) that
993    /// the sized cap is very unlikely to already sit on a page boundary. Requires cgroup v2
994    /// delegation, so it's `#[ignore]`d.
995    #[test]
996    #[ignore = "requires cgroup v2 delegation; run manually"]
997    fn cap_page_aligns_and_journal_matches_on_disk_value() {
998        use common::Limit;
999        use std::process::Command;
1000
1001        let manager = CgroupManager::new().expect("create CgroupManager");
1002        let journal_dir = tempfile::tempdir().unwrap();
1003        let journal =
1004            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1005        let effector = Effector::new(&manager, &journal, None);
1006
1007        let abs_path = manager
1008            .prepare_cgroup("test-cap-align", &Limit::default())
1009            .expect("create test cgroup")
1010            .path;
1011        let cgroup = format!(
1012            "/{}",
1013            abs_path
1014                .strip_prefix("/sys/fs/cgroup")
1015                .expect("cgroup under /sys/fs/cgroup")
1016                .display()
1017        );
1018
1019        // Hold ~400MB of anon memory (above the 256 MiB floor) whose sized
1020        // cap is very unlikely to land on a page boundary, then sleep.
1021        let mut child = Command::new("bash")
1022            .arg("-c")
1023            .arg("a=$(head -c 300000000 /dev/urandom | base64 -w0); sleep 30")
1024            .spawn()
1025            .expect("spawn memory-holding process");
1026        let pid = child.id();
1027        manager
1028            .add_to_cgroup(&abs_path, pid)
1029            .expect("add process to test cgroup");
1030        // Wait until the shell has actually built up the allocation, or the
1031        // cap would sit on the 256 MiB floor and prove nothing.
1032        let want = 300 * 1024 * 1024;
1033        let deadline = std::time::Instant::now() + Duration::from_secs(10);
1034        loop {
1035            let cur = cgfs::current_bytes(&cgroup).unwrap_or(0);
1036            if cur >= want {
1037                break;
1038            }
1039            if std::time::Instant::now() >= deadline {
1040                let _ = child.kill();
1041                let _ = child.wait();
1042                let _ = manager.cleanup_cgroup("test-cap-align");
1043                panic!("memory.current reached only {cur} bytes, need {want}, after 10s");
1044            }
1045            std::thread::sleep(Duration::from_millis(100));
1046        }
1047
1048        let res = test_resolution(cgroup.clone());
1049        effector
1050            .apply(&Action::Cap {
1051                res: res.clone(),
1052                name: "mem-hog".into(),
1053            })
1054            .expect("cap");
1055
1056        let entries = journal.entries();
1057        assert_eq!(entries.len(), 1, "cap should be journaled");
1058        let journaled = entries[0].our_high.clone().expect("our_high recorded");
1059
1060        let on_disk = cgfs::read_high(&cgroup).expect("read memory.high");
1061        assert_eq!(
1062            on_disk, journaled,
1063            "on-disk memory.high must match the journaled our_high exactly"
1064        );
1065
1066        let bytes: u64 = journaled.parse().expect("our_high is a plain decimal");
1067        assert_eq!(bytes % page_size(), 0, "written value must be page-aligned");
1068
1069        effector.apply(&Action::LiftCap { res }).expect("lift cap");
1070        assert!(
1071            journal.entries().is_empty(),
1072            "journal entry removed after lift"
1073        );
1074
1075        let _ = child.kill();
1076        let _ = child.wait();
1077        let _ = manager.cleanup_cgroup("test-cap-align");
1078    }
1079
1080    /// Cap and lift go through raw `memory.high` writes only, so the unit's
1081    /// own `MemoryHigh` property (as systemd reports it) must be the same
1082    /// before the cap and after the lift, and the raw `memory.high` must be
1083    /// back at its pre-cap value. A runtime `SetUnitProperties` would leave
1084    /// a `/run` drop-in behind that changes the reported property.
1085    /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
1086    #[test]
1087    #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
1088    fn lift_cap_leaves_unit_memory_high_property_untouched() {
1089        use std::process::Command;
1090
1091        let unit_memory_high = |unit: &str| -> String {
1092            let out = Command::new("systemctl")
1093                .args(["--user", "show", "-p", "MemoryHigh", "--value", unit])
1094                .output()
1095                .expect("systemctl --user show");
1096            String::from_utf8_lossy(&out.stdout).trim().to_string()
1097        };
1098
1099        let uid_out = Command::new("id").arg("-u").output().expect("id -u");
1100        let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
1101            .trim()
1102            .parse()
1103            .expect("parse uid");
1104
1105        let unit_base = format!("rlm-e2e-cap-{}", std::process::id());
1106        let unit = format!("{unit_base}.scope");
1107        let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
1108
1109        let mut child = Command::new("systemd-run")
1110            .args([
1111                "--user",
1112                "--scope",
1113                "--slice=app.slice",
1114                &format!("--unit={unit_base}"),
1115                // A genuine prior MemoryHigh, distinct from both "max" and
1116                // whatever Cap computes, so the restored value is
1117                // unambiguous.
1118                "--property=MemoryHigh=1500M",
1119                "--",
1120                "sleep",
1121                "30",
1122            ])
1123            .spawn()
1124            .expect("spawn systemd-run --scope with MemoryHigh set");
1125
1126        std::thread::sleep(Duration::from_millis(300));
1127
1128        let manager = CgroupManager::new().expect("create CgroupManager");
1129        let journal_dir = tempfile::tempdir().unwrap();
1130        let journal =
1131            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1132        let systemd = SystemdUser::connect();
1133        let effector = Effector::new(&manager, &journal, systemd.as_ref());
1134
1135        let original_high =
1136            cgfs::read_high(&cgroup).expect("systemd-run set an initial memory.high");
1137        assert_ne!(
1138            original_high, "max",
1139            "test needs a concrete prior MemoryHigh to distinguish from systemd's clear-to-max"
1140        );
1141        let property_before = unit_memory_high(&unit);
1142
1143        let res = Resolution {
1144            cgroup: cgroup.clone(),
1145            unit: Some(unit.clone()),
1146            verdict: Verdict::CapOnly,
1147            coverage: Coverage::Full,
1148            mechanism: Mechanism::Unit,
1149        };
1150
1151        effector
1152            .apply(&Action::Cap {
1153                res: res.clone(),
1154                name: "sleep".into(),
1155            })
1156            .expect("cap");
1157        assert_ne!(
1158            cgfs::read_high(&cgroup),
1159            Some(original_high.clone()),
1160            "cap should have changed memory.high"
1161        );
1162
1163        effector.apply(&Action::LiftCap { res }).expect("lift cap");
1164        assert_eq!(
1165            unit_memory_high(&unit),
1166            property_before,
1167            "cap/lift must not change the unit's MemoryHigh property"
1168        );
1169        assert_eq!(
1170            cgfs::read_high(&cgroup),
1171            Some(original_high),
1172            "lift must restore the raw pre-cap memory.high"
1173        );
1174        assert!(
1175            journal.entries().is_empty(),
1176            "journal entry removed after lift"
1177        );
1178
1179        let _ = child.kill();
1180        let _ = child.wait();
1181        let _ = Command::new("systemctl")
1182            .args(["--user", "stop", &unit])
1183            .status();
1184    }
1185
1186    /// Regression test for Promoted Minor B: a leaked `[Cap, Freeze]` chain
1187    /// (Cap journaled first, then the same cgroup frozen later without the
1188    /// Cap entry ever being cleared) must still restore the Cap's
1189    /// `prev_high` on thaw. Before the fix, liveness was judged against
1190    /// `entries.last()` (the Freeze entry, which has no `memory.high` of
1191    /// its own), so `restore_step` reported "nothing to restore" and
1192    /// `memory.high` stayed pinned at the guard's Cap value forever once the
1193    /// whole chain was cleared. Requires cgroup v2 delegation, so it's
1194    /// `#[ignore]`d.
1195    #[test]
1196    #[ignore = "requires cgroup v2 delegation; run manually"]
1197    fn chain_restores_cap_value_when_newest_entry_is_freeze() {
1198        use common::Limit;
1199        use std::process::Command;
1200
1201        let manager = CgroupManager::new().expect("create CgroupManager");
1202        let journal_dir = tempfile::tempdir().unwrap();
1203        let journal =
1204            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1205        let effector = Effector::new(&manager, &journal, None);
1206
1207        let abs_path = manager
1208            .prepare_cgroup("test-chain-restore", &Limit::default())
1209            .expect("create test cgroup")
1210            .path;
1211        let cgroup = format!(
1212            "/{}",
1213            abs_path
1214                .strip_prefix("/sys/fs/cgroup")
1215                .expect("cgroup under /sys/fs/cgroup")
1216                .display()
1217        );
1218
1219        let mut child = Command::new("sleep")
1220            .arg("30")
1221            .spawn()
1222            .expect("spawn sleep");
1223        let pid = child.id();
1224        manager
1225            .add_to_cgroup(&abs_path, pid)
1226            .expect("add sleep to test cgroup");
1227
1228        let original_high = cgfs::read_high(&cgroup).expect("read initial memory.high");
1229        let res = test_resolution(cgroup.clone());
1230
1231        // Cap first (oldest entry)...
1232        effector
1233            .apply(&Action::Cap {
1234                res: res.clone(),
1235                name: "sleep".into(),
1236            })
1237            .expect("cap");
1238        assert_ne!(
1239            cgfs::read_high(&cgroup),
1240            Some(original_high.clone()),
1241            "cap should have changed memory.high"
1242        );
1243
1244        // ...then freeze the same cgroup without ever clearing the Cap
1245        // entry. This is the leaked-chain scenario: two coexisting entries,
1246        // Freeze newest.
1247        effector
1248            .apply(&Action::Freeze {
1249                res: res.clone(),
1250                name: "sleep".into(),
1251            })
1252            .expect("freeze");
1253
1254        let entries = journal.entries();
1255        assert_eq!(entries.len(), 2, "both Cap and Freeze entries coexist");
1256        assert_eq!(entries[0].action, JournalAction::Cap, "Cap is oldest");
1257        assert_eq!(entries[1].action, JournalAction::Freeze, "Freeze is newest");
1258
1259        // Thaw (the action that would naturally follow a Freeze) must still
1260        // restore the Cap's prev_high, not skip restoration just because the
1261        // newest entry is a Freeze with no memory.high of its own.
1262        effector.apply(&Action::Thaw { res }).expect("thaw");
1263        assert_eq!(
1264            cgfs::read_high(&cgroup),
1265            Some(original_high),
1266            "thaw must restore the chain's Cap value even though Freeze is the newest entry"
1267        );
1268        assert!(
1269            journal.entries().is_empty(),
1270            "both chain entries removed after thaw"
1271        );
1272
1273        let _ = child.kill();
1274        let _ = child.wait();
1275        let _ = manager.cleanup_cgroup("test-chain-restore");
1276    }
1277}