Skip to main content

rlmctl_core/guard/
effector.rs

1//! Executes [`Action`]s against real cgroups, acting in place on the cgroup a
2//! process already lives in (a systemd unit's scope/service, or an existing
3//! rlm rule cgroup) rather than moving it into an ephemeral `guard-<pid>`
4//! cgroup. Every action is best-effort and logged; a failure must never
5//! panic or otherwise crash the daemon loop. `apply` may return `Err` so the
6//! caller can log it, but a missing `notify-send` (or any other notification
7//! failure) is never treated as an error.
8//!
9//! # Write-ahead journal
10//! Freeze/Cap always `journal.append` (which fsyncs) *before* touching the
11//! cgroup, so a crash between the two still leaves a durable record that
12//! startup recovery (`sweep_leftovers`) can replay. `Journal` is internally
13//! mutex-serialized (see `journal.rs`), and the daemon calls into this
14//! `Effector` from a single thread (the tick loop in `rlm-guard`'s `main`),
15//! so `Effector` just holds a `&Journal` and relies on the daemon's
16//! single-threaded call discipline plus the journal's own internal locking
17//! for safety — it adds no locking of its own.
18//!
19//! # Mechanism: systemd unit vs. raw cgroupfs
20//! When a target resolved to a systemd unit (`Mechanism::Unit`), we prefer
21//! the D-Bus call (`FreezeUnit`/`ThawUnit`) with a hard
22//! 2s timeout (`systemd::SystemdUser`'s own enforced deadline); on `Err` (bus
23//! unavailable, call failed, or timed out) we fall back to the raw cgroupfs
24//! primitives in `cgfs`. Raw-mechanism targets (rlm's own rule cgroups) skip
25//! the D-Bus attempt entirely. Thawing is mechanism-independent: we always
26//! perform the raw `cgroup.freeze` write first, unconditionally, and only
27//! then attempt `ThawUnit` best-effort so systemd's view matches. This
28//! guarantees a cgroup is never left frozen because of a stale unit or a
29//! slow bus, and tolerates a missing cgroup (the raw write simply errors and
30//! we move on). On shutdown and startup replay every raw thaw and restore
31//! finishes before the first D-Bus call. Caps are the exception: they never
32//! go through systemd (see below).
33//!
34//! # Restoring `memory.high`
35//! The kernel truncates `memory.high` writes to page multiples, so the byte
36//! count we cap to is page-aligned *before* we journal/write it
37//! ([`page_align_down`]); as a second line of defense, [`Effector::cap`]
38//! reads `memory.high` back after writing and self-corrects the journal if
39//! reality still differs.
40//! Caps and restores are raw `memory.high` writes only, never systemd's
41//! `SetUnitProperties(MemoryHigh)`: a runtime property leaves a `/run`
42//! drop-in that outlives the cap and masks the unit's own configured
43//! `MemoryHigh` until reboot, and restoring through systemd wrote `infinity`
44//! over it. If systemd later re-applies the unit's value, that only lifts our
45//! cap early, and the journal's `our_high` check then skips the restore (the
46//! safe direction).
47//!
48//! If more than one journal entry ever coexists for the same
49//! cgroup (a leak from an incomplete prior removal), every restore path
50//! treats them as one chain: liveness is judged against the chain's `Cap`
51//! entry specifically (only a `Cap` has a `memory.high` to restore — a
52//! `Freeze` entry does not, so judging against `entries.last()` when it
53//! happens to be a newer `Freeze` would wrongly look like "nothing to
54//! restore" and strand `memory.high` at the guard's value forever), and the
55//! value restored is always the *oldest* `Cap` entry's `prev_high` — the
56//! true pre-intervention value, not an intermediate entry's `prev_high`
57//! (which is just our own previous `our_high`). See [`restore_target`].
58//! Liveness, however, is judged against the *newest* `Cap` entry, not the
59//! oldest: entries are strictly appended, so a later Cap's write always
60//! supersedes an earlier one's on disk, and `should_restore`'s string
61//! comparison must match what's actually there. See [`restore_decision`].
62
63use super::cgfs;
64use super::journal::{should_restore, Journal, JournalAction, JournalEntry};
65use super::resolve::{Mechanism, Resolution};
66use super::sampler::parse_meminfo;
67use super::systemd::SystemdUser;
68use super::types::Action;
69use crate::CgroupManager;
70use common::Result;
71use std::process::Command;
72use std::time::Duration;
73
74/// Floor for any soft cap. A cap below this is effectively a freeze for a
75/// desktop app, so small cgroups are never squeezed further than this.
76pub const MIN_CAP_BYTES: u64 = 256 * 1024 * 1024;
77
78/// Hard deadline for every D-Bus call on the freeze-guard's storm path. On
79/// `Err` (including a timeout) callers fall back to raw cgroupfs writes.
80const DBUS_TIMEOUT: Duration = Duration::from_secs(2);
81
82/// Applies guard [`Action`]s in place on the resolved target cgroup, via the
83/// systemd-unit D-Bus path (with raw-cgroup fallback) and a write-ahead
84/// [`Journal`] for crash-safe restore.
85pub struct Effector<'a> {
86    /// Kept only for the legacy `guard-<pid>` sweep during the upgrade path
87    /// (see [`Effector::sweep_leftovers`]); acting in place no longer uses it.
88    manager: &'a CgroupManager,
89    journal: &'a Journal,
90    systemd: Option<&'a SystemdUser>,
91}
92
93impl<'a> Effector<'a> {
94    pub fn new(
95        manager: &'a CgroupManager,
96        journal: &'a Journal,
97        systemd: Option<&'a SystemdUser>,
98    ) -> Self {
99        Self {
100            manager,
101            journal,
102            systemd,
103        }
104    }
105
106    /// Apply a single action. Best-effort: returns `Err` only so the caller can
107    /// log it (a [`Action::Notify`] always returns `Ok`).
108    pub fn apply(&self, action: &Action) -> Result<()> {
109        match action {
110            Action::Freeze { res, name } => self.freeze(res, name),
111            Action::Thaw { res } => self.thaw(res),
112            Action::Cap { res, name } => self.cap(res, name),
113            Action::LiftCap { res } => self.lift_cap(res),
114            Action::Notify { message } => {
115                notify(message);
116                // Notification is always best-effort and never fails the caller.
117                Ok(())
118            }
119        }
120    }
121
122    fn freeze(&self, res: &Resolution, name: &str) -> Result<()> {
123        // The inode is the guard that lets a later restore tell "this is
124        // still the same cgroup" from "this cgroup was torn down and
125        // recreated" (`should_restore`). `unwrap_or(0)` used to substitute a
126        // poison sentinel here: 0 is never a real inode, so the guard could
127        // never match again and the entry became permanently unrestorable —
128        // yet it was still journaled-and-acted-on, then later cleared and
129        // logged as if it were an intentional skip (Promoted Minor A). Fail
130        // closed instead: refuse to freeze at all rather than act with a
131        // record we can never safely restore from.
132        let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
133            tracing::warn!(
134                cgroup = %res.cgroup, name,
135                "cannot read cgroup inode; refusing to freeze (would be unrestorable)"
136            );
137            return Err(common::Error::Cgroup(format!(
138                "cannot read inode for {}; refusing to freeze",
139                res.cgroup
140            )));
141        };
142        let entry = JournalEntry {
143            cgroup: res.cgroup.clone(),
144            inode,
145            unit: res.unit.clone(),
146            action: JournalAction::Freeze,
147            prev_high: None,
148            our_high: None,
149        };
150        // Write-ahead: the entry must be durable (journal.append fsyncs)
151        // before we ever touch the cgroup, so a crash between the two still
152        // leaves a record startup recovery can act on.
153        self.journal.append(&entry)?;
154
155        tracing::info!(cgroup = %res.cgroup, name, "freezing cgroup");
156        if res.mechanism == Mechanism::Unit {
157            if let (Some(unit), Some(systemd)) = (&res.unit, self.systemd) {
158                match systemd.freeze_unit(unit, DBUS_TIMEOUT) {
159                    Ok(()) => return Ok(()),
160                    Err(e) => tracing::warn!(
161                        cgroup = %res.cgroup, unit, error = %e,
162                        "FreezeUnit failed; falling back to raw cgroup.freeze"
163                    ),
164                }
165            }
166        }
167        cgfs::write_freeze(&res.cgroup, true)
168    }
169
170    fn thaw(&self, res: &Resolution) -> Result<()> {
171        tracing::info!(cgroup = %res.cgroup, "thawing cgroup");
172        // Gather any coexisting entries for this cgroup first: a `Thaw`
173        // action reverses a Frozen intervention, but if a Cap entry ever
174        // leaked alongside it (Important #3, Task 6 review) we must still
175        // restore its memory.high before wiping the journal record below,
176        // not just drop it silently.
177        let entries = self.entries_for(&res.cgroup);
178        let result = self.thaw_raw(&res.cgroup, res.unit.as_deref());
179        if let Err(e) = &result {
180            tracing::debug!(cgroup = %res.cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
181        }
182        self.restore_high_if_any(&res.cgroup, &entries);
183        self.journal.remove(&res.cgroup)?;
184        result
185    }
186
187    fn cap(&self, res: &Resolution, name: &str) -> Result<()> {
188        // See `freeze`'s matching comment (Promoted Minor A): a `0` inode
189        // sentinel here would make this Cap permanently unrestorable while
190        // looking like a real guard, so fail closed instead of
191        // journal-and-act with an unrestorable record.
192        let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
193            tracing::warn!(
194                cgroup = %res.cgroup, name,
195                "cannot read cgroup inode; refusing to cap (would be unrestorable)"
196            );
197            return Err(common::Error::Cgroup(format!(
198                "cannot read inode for {}; refusing to cap",
199                res.cgroup
200            )));
201        };
202        let prev_high = cgfs::read_high(&res.cgroup);
203        // Unknown swap total counts as 0: assuming anon is pinned only makes
204        // the cap gentler.
205        let swap_total_kb = std::fs::read_to_string("/proc/meminfo")
206            .ok()
207            .and_then(|m| parse_meminfo(&m))
208            .map_or(0, |m| m.swap_total_kb);
209        let anon_ok = cgfs::anon_reclaimable(&res.cgroup, swap_total_kb);
210        let Some(target) = cap_target(
211            cgfs::current_bytes(&res.cgroup),
212            cgfs::file_bytes(&res.cgroup),
213            anon_ok,
214        ) else {
215            tracing::warn!(
216                cgroup = %res.cgroup, name,
217                "cannot read memory.current; refusing to cap"
218            );
219            return Err(common::Error::Cgroup(
220                "cannot read memory.current; refusing to cap".into(),
221            ));
222        };
223        // The kernel truncates `memory.high` writes to page multiples, so we
224        // must journal/write the value it will actually store, not the raw
225        // target, or `should_restore`'s string-equality check can never
226        // pass again and the cap becomes permanent.
227        let our_bytes = page_align_down(target, page_size());
228        // Plain decimal bytes, no separators/whitespace: this must be
229        // exactly what a later `cgfs::read_high` (which only trims
230        // whitespace off the raw file contents) returns. See
231        // `our_high_string_is_plain_decimal_no_separators` below, and
232        // `reconcile_our_high` for the belt-and-braces check.
233        if !cap_tightens(prev_high.as_deref(), our_bytes) {
234            tracing::info!(
235                cgroup = %res.cgroup, name, prev_high = ?prev_high, our_bytes,
236                "existing memory.high already at or below the cap; not capping"
237            );
238            return Err(common::Error::Cgroup(
239                "existing memory.high already at or below the cap; refusing to cap".into(),
240            ));
241        }
242        let our_high = our_bytes.to_string();
243
244        let entry = JournalEntry {
245            cgroup: res.cgroup.clone(),
246            inode,
247            unit: res.unit.clone(),
248            action: JournalAction::Cap,
249            prev_high,
250            our_high: Some(our_high.clone()),
251        };
252        self.journal.append(&entry)?;
253
254        tracing::info!(
255            cgroup = %res.cgroup, name, our_high = %our_high, anon_reclaimable = anon_ok,
256            "soft-capping cgroup"
257        );
258        let result = cgfs::write_high(&res.cgroup, &our_high);
259
260        if result.is_ok() {
261            self.reconcile_our_high(&entry);
262        }
263        result
264    }
265
266    /// After a successful `Cap` write, read `memory.high` back; if what's
267    /// actually on disk differs from what we journaled (page truncation we
268    /// didn't fully pre-empt), correct the journal to match reality. Otherwise
269    /// `should_restore`'s string-equality check can never pass again and
270    /// the cap becomes permanent (Task 6 review, Critical #1). Rebuilds
271    /// only this cgroup's entries, preserving any others that might coexist
272    /// (Important #3), with `written`'s `our_high` corrected to the
273    /// read-back value. Uses `Journal::replace` (a single atomic rewrite),
274    /// not a separate `remove` then `append`: the latter has a window where
275    /// the cgroup has no journal record at all, so a crash/SIGKILL/append
276    /// failure right there would leave a permanent, unrecoverable cap —
277    /// exactly the invariant the journal exists to prevent (Task 6 review,
278    /// fix round 2).
279    fn reconcile_our_high(&self, written: &JournalEntry) {
280        let cgroup: &str = &written.cgroup;
281        let Some(actual) = cgfs::read_high(cgroup) else {
282            return;
283        };
284        if written.our_high.as_deref() == Some(actual.as_str()) {
285            return;
286        }
287        tracing::warn!(
288            cgroup = %cgroup, journaled = ?written.our_high, actual = %actual,
289            "memory.high on disk differs from what we journaled; correcting journal entry"
290        );
291        let mut entries = self.entries_for(cgroup);
292        let Some(pos) = entries.iter().rposition(|e| e == written) else {
293            // Already removed/replaced by something else (e.g. a concurrent
294            // Thaw/LiftCap) — nothing left to correct.
295            return;
296        };
297        entries[pos].our_high = Some(actual);
298        if let Err(e) = self.journal.replace(cgroup, &entries) {
299            tracing::warn!(cgroup = %cgroup, error = %e, "failed to correct journal entry (atomic replace)");
300        }
301    }
302
303    fn lift_cap(&self, res: &Resolution) -> Result<()> {
304        tracing::info!(cgroup = %res.cgroup, "lifting cap");
305        let entries = self.entries_for(&res.cgroup);
306        // Mechanism-independent thaw always runs, regardless of whether any
307        // journal record exists — a dead-cgroup prune (carry-forward
308        // finding, Task 5 review) must never leave a target frozen just
309        // because we lost the journal entry.
310        let _ = self.thaw_raw(&res.cgroup, res.unit.as_deref());
311        self.restore_high_if_any(&res.cgroup, &entries);
312        self.journal.remove(&res.cgroup)
313    }
314
315    /// All journal entries for one cgroup, in the order `Journal::entries()`
316    /// returns them — oldest-first, since entries are strictly appended.
317    fn entries_for(&self, cgroup: &str) -> Vec<JournalEntry> {
318        self.journal
319            .entries()
320            .into_iter()
321            .filter(|e| e.cgroup == cgroup)
322            .collect()
323    }
324
325    /// Startup recovery: legacy `guard-<pid>` sweep (kept for one release as
326    /// an upgrade path from pre-act-in-place rlm), then replay every live
327    /// journal entry.
328    pub fn sweep_leftovers(&self) -> Result<()> {
329        if let Err(e) = self.manager.sweep_guard_leftovers() {
330            tracing::warn!(error = %e, "legacy guard-<pid> sweep failed (non-fatal)");
331        }
332        self.replay_and_clear()
333    }
334
335    /// Graceful shutdown: undo every live journal entry, then clear the
336    /// journal. Same replay as `sweep_leftovers`, minus the legacy sweep.
337    pub fn undo_all(&self) -> Result<()> {
338        self.replay_and_clear()
339    }
340
341    fn replay_and_clear(&self) -> Result<()> {
342        // Group by cgroup first (entries for the same cgroup aren't
343        // necessarily contiguous in the file) so each cgroup's chain is
344        // replayed as one unit — see `restore_high_if_any` — rather than
345        // entry-by-entry, which would mis-restore whenever more than one
346        // entry coexists for a cgroup (Important #3, Task 6 review).
347        let mut by_cgroup: std::collections::HashMap<String, Vec<JournalEntry>> =
348            std::collections::HashMap::new();
349        for e in self.journal.entries() {
350            by_cgroup.entry(e.cgroup.clone()).or_default().push(e);
351        }
352        // Pass 1: every raw thaw and memory.high restore, then clear the
353        // journal. No D-Bus call happens before this is done, so a slow
354        // session bus cannot push the undo past systemd's stop timeout.
355        for (cgroup, entries) in &by_cgroup {
356            if let Err(e) = cgfs::write_freeze(cgroup, false) {
357                tracing::debug!(cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
358            }
359            self.restore_high_if_any(cgroup, entries);
360        }
361        let cleared = self.journal.clear();
362        // Pass 2, best effort: tell systemd so its view of the units
363        // matches the kernel's. The processes already run again.
364        if let Some(systemd) = self.systemd {
365            for (cgroup, entries) in &by_cgroup {
366                if let Some(unit) = entries.last().and_then(|e| e.unit.as_deref()) {
367                    if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
368                        tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
369                    }
370                }
371            }
372        }
373        cleared
374    }
375
376    /// Restore `memory.high` for one cgroup's journal `entries` (oldest-first),
377    /// if the chain is still live — called *after* the caller has already
378    /// performed the mechanism-independent thaw (see module docs: no path
379    /// here ever means "leave frozen"). All the decision logic is delegated
380    /// to the pure [`restore_decision`]: liveness is judged against the
381    /// *newest* `Cap` entry (whose write is what's actually on disk right
382    /// now), while the value restored is [`restore_target`]'s *oldest*-entry
383    /// `prev_high` (the true pre-intervention value). Judging liveness
384    /// against the oldest Cap instead (NEW-1 regression) leaves a stacked
385    /// `[Cap, Cap]` chain's on-disk value permanently un-restorable, since
386    /// the oldest entry's `our_high` never matches what a later Cap actually
387    /// wrote. Judging liveness against `entries.last()` has the same failure
388    /// for a `[Cap, Freeze]` chain (Promoted Minor B): the newest entry is
389    /// the `Freeze`, which has no `memory.high` of its own. The restore is a
390    /// raw `memory.high` write only; systemd's `MemoryHigh` property is never
391    /// touched (see module docs).
392    fn restore_high_if_any(&self, cgroup: &str, entries: &[JournalEntry]) {
393        let inode = cgfs::dir_inode(cgroup);
394        let high = cgfs::read_high(cgroup);
395        match restore_decision(entries, inode, high.as_deref()) {
396            RestoreStep::ThawAndRestoreHigh { to } => {
397                if let Err(err) = cgfs::write_high(cgroup, &to) {
398                    tracing::warn!(cgroup, error = %err, "failed to restore memory.high");
399                }
400            }
401            // No Cap entry anywhere in the chain (a Freeze-only chain,
402            // possibly with leaked duplicates): there is no memory.high to
403            // restore. The unconditional thaw already ran in the caller, so
404            // there's nothing left to do here.
405            RestoreStep::ThawOnly => {}
406            RestoreStep::SkipRemove => {
407                tracing::warn!(
408                    cgroup,
409                    "not restoring memory.high (cgroup recreated or value changed since our write)"
410                );
411            }
412        }
413    }
414
415    /// Mechanism-independent thaw: an *unconditional* raw `cgroup.freeze`
416    /// write first, so a cgroup is never left frozen and a slow bus cannot
417    /// delay it, then a best-effort `ThawUnit` (if we have a unit and a bus)
418    /// so systemd's view of the unit matches. Tolerates a missing cgroup:
419    /// the raw write then simply returns `Err`, which every caller here
420    /// treats as non-fatal.
421    fn thaw_raw(&self, cgroup: &str, unit: Option<&str>) -> Result<()> {
422        let result = cgfs::write_freeze(cgroup, false);
423        if let (Some(unit), Some(systemd)) = (unit, self.systemd) {
424            if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
425                tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
426            }
427        }
428        result
429    }
430}
431
432/// What to do with one journal entry's `memory.high` at restore time. Never
433/// speaks to freeze/thaw — that is unconditional and already handled by the
434/// caller (`Effector::thaw_raw`) before `restore_step` is even consulted, so
435/// no variant here can ever mean "leave it frozen".
436#[derive(Debug, PartialEq, Eq)]
437pub enum RestoreStep {
438    /// A `Freeze` entry whose guard still holds: nothing to restore, thaw
439    /// (already done by the caller) was the whole job.
440    ThawOnly,
441    /// A `Cap` entry whose guard still holds: restore `memory.high` to `to`
442    /// (the entry's `prev_high`, "max" if it was never recorded).
443    ThawAndRestoreHigh { to: String },
444    /// The guard no longer holds (cgroup recreated, or someone else changed
445    /// `memory.high` since our write): don't touch `memory.high`, just let
446    /// the caller remove the journal entry.
447    SkipRemove,
448}
449
450/// Pure: what to do for one journal entry at restore time, delegating the
451/// safety check to [`should_restore`].
452pub fn restore_step(e: &JournalEntry, inode: Option<u64>, high: Option<&str>) -> RestoreStep {
453    if !should_restore(e, inode, high) {
454        return RestoreStep::SkipRemove;
455    }
456    match e.action {
457        JournalAction::Freeze => RestoreStep::ThawOnly,
458        JournalAction::Cap => RestoreStep::ThawAndRestoreHigh {
459            to: e.prev_high.clone().unwrap_or_else(|| "max".into()),
460        },
461    }
462}
463
464/// Size a soft cap that slows an app down without stalling it.
465///
466/// memory.high applies to everything charged to the cgroup, page cache
467/// included, so the cap is sized from memory.current: it never asks the
468/// kernel to reclaim more than 10% of it. When anon memory cannot go to swap,
469/// only file pages can be reclaimed, so the cap never asks for more than 80%
470/// of them. Returns `None` when memory.current is unreadable: no cap is safer
471/// than a guessed one.
472pub fn cap_target(current: Option<u64>, file: Option<u64>, anon_reclaimable: bool) -> Option<u64> {
473    let current = current?;
474    let mut target = current / 10 * 9;
475    if !anon_reclaimable {
476        let file_floor = current.saturating_sub(file.unwrap_or(0).saturating_mul(8) / 10);
477        target = target.max(file_floor);
478    }
479    Some(target.max(MIN_CAP_BYTES))
480}
481
482/// Pure: whether writing `target` would tighten the existing `memory.high`.
483/// A numeric `prev_high` at or below `target` means the cap would loosen (or
484/// not change) the limit already in place, so the caller must not write it.
485/// `"max"`, or anything else that is not a byte count, is no limit at all.
486pub fn cap_tightens(prev_high: Option<&str>, target: u64) -> bool {
487    match prev_high.and_then(|s| s.trim().parse::<u64>().ok()) {
488        Some(prev) => prev > target,
489        None => true,
490    }
491}
492
493/// Pure: the full restore decision for one cgroup's journal `entries`
494/// (oldest-first), given the cgroup's current inode and on-disk
495/// `memory.high` (both already read by the caller — no IO here). Answers two
496/// orthogonal questions with two different entries on purpose:
497///
498/// - **Liveness** (is the guard we wrote still intact?) is judged against
499///   the *newest* `Cap` entry in the chain. Entries are strictly appended,
500///   so a later Cap's write always supersedes an earlier one's on disk —
501///   `should_restore`'s string-equality check must compare against the
502///   entry whose `our_high` is what's actually there right now.
503/// - **Value** (what do we restore to?) is [`restore_target`]'s *oldest*
504///   `Cap` entry's `prev_high` — the true pre-intervention value, not an
505///   intermediate entry's `prev_high` (which is just our own previous
506///   `our_high` from an earlier cap in the same chain).
507///
508/// Judging both against the oldest Cap (NEW-1 regression) makes a stacked
509/// `[Cap, Cap]` chain's liveness check compare a stale `our_high` against
510/// the newer write on disk, so it never matches and the chain is treated as
511/// dead — stranding `memory.high` at the guard's value forever. Judging
512/// liveness against `entries.last()` fails the same way for `[Cap, Freeze]`
513/// (Promoted Minor B): the newest entry is the `Freeze`, which has no
514/// `memory.high` of its own. Returns [`RestoreStep::ThawOnly`] when there's
515/// no `Cap` entry anywhere in the chain (a `Freeze`-only chain) — nothing to
516/// restore beyond the caller's unconditional thaw.
517pub fn restore_decision(
518    entries: &[JournalEntry],
519    inode: Option<u64>,
520    high: Option<&str>,
521) -> RestoreStep {
522    let Some(newest_cap) = entries
523        .iter()
524        .rev()
525        .find(|e| e.action == JournalAction::Cap)
526    else {
527        return RestoreStep::ThawOnly;
528    };
529    match restore_step(newest_cap, inode, high) {
530        RestoreStep::ThawAndRestoreHigh { .. } => match restore_target(entries) {
531            Some(to) => RestoreStep::ThawAndRestoreHigh { to },
532            // Unreachable: `newest_cap` being a Cap guarantees `restore_target`
533            // (which only needs *any* Cap entry) also finds one.
534            None => RestoreStep::ThawOnly,
535        },
536        other => other,
537    }
538}
539
540/// Pure: given all journal entries for one cgroup, oldest-first, select the
541/// `memory.high` value to restore, if the chain contains a `Cap` entry. The
542/// OLDEST `Cap` entry's `prev_high` is the true pre-intervention value — a
543/// later entry's `prev_high` is just our own previous `our_high` from an
544/// earlier cap in the same chain, not the original (Task 6 review,
545/// Important #3). Returns `None` if there's no `Cap` entry (e.g. a
546/// `Freeze`-only chain).
547pub fn restore_target(entries: &[JournalEntry]) -> Option<String> {
548    entries
549        .iter()
550        .find(|e| e.action == JournalAction::Cap)
551        .map(|e| e.prev_high.clone().unwrap_or_else(|| "max".into()))
552}
553
554/// Runtime page size in bytes, via `sysconf(_SC_PAGESIZE)`. Falls back to
555/// 4096 (by far the most common value) only if the syscall ever returns
556/// something nonsensical — a defensive fallback, not the source of truth,
557/// since the whole point is to match whatever the kernel actually enforces.
558fn page_size() -> u64 {
559    // SAFETY: sysconf(_SC_PAGESIZE) reads a static system parameter; no
560    // pointers involved, no side effects.
561    let p = unsafe { libc::sysconf(libc::_SC_PAGESIZE) };
562    if p > 0 {
563        p as u64
564    } else {
565        4096
566    }
567}
568
569/// Pure: round `bytes` down to a multiple of `page` (a no-op if `page` is 0).
570/// The kernel truncates `memory.high` writes to page multiples (Task 6
571/// review, Critical #1), so we must journal/write the value it will
572/// actually store, not the pre-truncation target — otherwise
573/// `should_restore`'s string-equality check can never pass again and a cap
574/// becomes permanent.
575fn page_align_down(bytes: u64, page: u64) -> u64 {
576    bytes.checked_div(page).map_or(bytes, |q| q * page)
577}
578
579/// Best-effort desktop notification via `notify-send`. Silently does nothing if
580/// the binary is missing or the spawn fails — notifications must never break the
581/// guard.
582fn notify(message: &str) {
583    match Command::new("notify-send")
584        .arg("rlm-guard")
585        .arg(message)
586        .spawn()
587    {
588        Ok(mut child) => {
589            // Reap asynchronously so we don't block; ignore any wait error.
590            std::thread::spawn(move || {
591                let _ = child.wait();
592            });
593        }
594        Err(e) => {
595            tracing::debug!(error = %e, "notify-send unavailable; skipping notification");
596        }
597    }
598}
599
600#[cfg(test)]
601mod tests {
602    use super::super::resolve::{Coverage, Verdict};
603    use super::*;
604
605    const MIB: u64 = 1024 * 1024;
606    const GIB: u64 = 1024 * MIB;
607
608    #[test]
609    fn cap_never_demands_more_than_ten_percent_of_current() {
610        // Incident: Chrome scope with 1.5 GiB charged, mostly page cache, ~150 MiB anon.
611        let current = GIB * 3 / 2;
612        let cap = cap_target(Some(current), Some(GIB * 135 / 100), true).unwrap();
613        assert!(
614            cap >= current / 10 * 9,
615            "cap {cap} is below 90% of {current}"
616        );
617    }
618
619    #[test]
620    fn swapless_cap_only_asks_for_reclaimable_file_pages() {
621        assert_eq!(
622            cap_target(Some(GIB), Some(0), false),
623            Some(GIB),
624            "nothing reclaimable: demand nothing"
625        );
626        assert_eq!(
627            cap_target(Some(GIB), Some(512 * MIB), false),
628            Some(GIB / 10 * 9)
629        );
630        assert_eq!(
631            cap_target(Some(GIB), None, false),
632            Some(GIB),
633            "unknown file size: demand nothing"
634        );
635    }
636
637    #[test]
638    fn cap_has_a_256_mib_floor() {
639        assert_eq!(
640            cap_target(Some(100 * MIB), Some(0), true),
641            Some(MIN_CAP_BYTES)
642        );
643    }
644
645    #[test]
646    fn cap_never_loosens_an_existing_memory_high() {
647        let floor = cap_target(Some(100 * MIB), Some(0), true).unwrap();
648        assert_eq!(floor, MIN_CAP_BYTES);
649        let prev = (200 * MIB).to_string();
650        assert!(
651            !cap_tightens(Some(&prev), floor),
652            "200 MiB unit limit must not be raised to the 256 MiB floor"
653        );
654        assert!(
655            !cap_tightens(Some(&floor.to_string()), floor),
656            "equal: no-op"
657        );
658        assert!(cap_tightens(Some("max"), floor));
659        assert!(cap_tightens(None, floor));
660        let prev = (2 * GIB).to_string();
661        assert!(cap_tightens(Some(&prev), GIB));
662    }
663
664    #[test]
665    fn unreadable_current_refuses_to_cap() {
666        assert_eq!(cap_target(None, Some(1), true), None);
667    }
668
669    /// Carry-forward (Task 3 review): `our_high` must be a plain decimal
670    /// string with no grouping/whitespace, since `should_restore` compares
671    /// it by string equality against `cgfs::read_high` (which only trims).
672    #[test]
673    fn our_high_string_is_plain_decimal_no_separators() {
674        let bytes = cap_target(Some(12_345_678_900), None, true).unwrap();
675        let s = bytes.to_string();
676        assert!(
677            s.chars().all(|c| c.is_ascii_digit()),
678            "our_high must be plain digits, got {s:?}"
679        );
680        assert_eq!(s, format!("{bytes}"), "no formatting beyond plain decimal");
681    }
682
683    /// Task 6 review, Critical #1: the kernel truncates `memory.high`
684    /// writes to page multiples (the reviewer's own observed example:
685    /// writing 900_000_000 with a 4096-byte page reads back 899_997_696).
686    #[test]
687    fn page_align_down_rounds_to_page_multiple() {
688        assert_eq!(page_align_down(900_000_000, 4096), 899_997_696);
689        assert_eq!(
690            page_align_down(4096, 4096),
691            4096,
692            "already-aligned is a no-op"
693        );
694        assert_eq!(page_align_down(100, 0), 100, "page=0 guard is a no-op");
695    }
696
697    /// Task 6 review, Important #3: when duplicate/stacked `Cap` entries end
698    /// up coexisting for one cgroup (a leak from an incomplete prior
699    /// removal), the restore target must be the OLDEST entry's `prev_high`
700    /// — the true pre-intervention value. The newer entry's `prev_high` is
701    /// deliberately a different-looking value here to prove we don't
702    /// chain-follow it.
703    #[test]
704    fn restore_target_uses_oldest_caps_prev_high() {
705        let oldest = entry_cap("/x", 42, "max", "A");
706        let newest = entry_cap("/x", 42, "A-prime", "B");
707        assert_eq!(
708            restore_target(&[oldest, newest]),
709            Some("max".into()),
710            "must use the oldest entry's prev_high, not the newest's"
711        );
712    }
713
714    #[test]
715    fn restore_target_none_for_freeze_only_chain() {
716        assert_eq!(restore_target(&[entry_freeze("/x", 42)]), None);
717    }
718
719    /// Regression test for NEW-1: `restore_decision`'s liveness check must be
720    /// judged against the NEWEST `Cap` entry (whose `our_high` is what's
721    /// actually on disk), while the value restored stays the OLDEST `Cap`
722    /// entry's `prev_high`. A stacked `[Cap(prev="max", our="A"),
723    /// Cap(prev="A", our="B")]` chain with disk `memory.high == "B"` must
724    /// restore to `"max"` — not `SkipRemove`, which the old
725    /// `entries.iter().find()` (oldest-Cap-for-everything) produced: it
726    /// compared the oldest entry's `our_high` ("A") against disk ("B"),
727    /// never matched, and permanently stranded `memory.high`. Also covers
728    /// `[Cap, Freeze]` (Promoted Minor B's shape) and a bare `[Cap]` chain in
729    /// the same test so a future re-swap of either selection can't slip by.
730    #[test]
731    fn restore_decision_liveness_uses_newest_cap_value_uses_oldest() {
732        // [Cap, Cap]: disk holds the NEWEST Cap's our_high ("B"). Liveness
733        // must be checked against "B", not the oldest entry's "A".
734        let oldest = entry_cap("/x", 42, "max", "A");
735        let newest = entry_cap("/x", 42, "A", "B");
736        assert_eq!(
737            restore_decision(&[oldest.clone(), newest.clone()], Some(42), Some("B")),
738            RestoreStep::ThawAndRestoreHigh { to: "max".into() },
739            "must judge liveness against the newest Cap's our_high (matches disk \"B\"), \
740             but restore the oldest Cap's prev_high (\"max\")"
741        );
742        // Sanity: the stale intermediate value "A" is no longer live on disk,
743        // so checking against it (the old bug) would report SkipRemove.
744        assert_eq!(
745            restore_step(&oldest, Some(42), Some("B")),
746            RestoreStep::SkipRemove,
747            "confirms the bug this guards against: judging liveness against the oldest \
748             entry's our_high against the newer on-disk value mismatches"
749        );
750
751        // [Cap, Freeze]: newest entry has no memory.high of its own; must
752        // still fall through to the Cap for both liveness and value.
753        let cap = entry_cap("/x", 42, "max", "1000");
754        let frz = entry_freeze("/x", 42);
755        assert_eq!(
756            restore_decision(&[cap, frz], Some(42), Some("1000")),
757            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
758        );
759
760        // [Cap] alone: baseline single-entry behavior is unchanged.
761        let cap_only = entry_cap("/x", 42, "max", "1000");
762        assert_eq!(
763            restore_decision(&[cap_only], Some(42), Some("1000")),
764            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
765        );
766
767        // Freeze-only chain: nothing to restore.
768        assert_eq!(
769            restore_decision(&[entry_freeze("/x", 42)], Some(42), None),
770            RestoreStep::ThawOnly
771        );
772    }
773
774    fn entry_cap(cg: &str, inode: u64, prev: &str, our: &str) -> JournalEntry {
775        JournalEntry {
776            cgroup: cg.into(),
777            inode,
778            unit: None,
779            action: JournalAction::Cap,
780            prev_high: Some(prev.into()),
781            our_high: Some(our.into()),
782        }
783    }
784
785    fn entry_freeze(cg: &str, inode: u64) -> JournalEntry {
786        JournalEntry {
787            cgroup: cg.into(),
788            inode,
789            unit: None,
790            action: JournalAction::Freeze,
791            prev_high: None,
792            our_high: None,
793        }
794    }
795
796    #[test]
797    fn restore_step_matrix() {
798        let cap = entry_cap("/x", 42, "max", "1000");
799        assert_eq!(
800            restore_step(&cap, Some(42), Some("1000")),
801            RestoreStep::ThawAndRestoreHigh { to: "max".into() }
802        );
803        assert_eq!(
804            restore_step(&cap, Some(43), Some("1000")),
805            RestoreStep::SkipRemove
806        );
807        assert_eq!(
808            restore_step(&cap, Some(42), Some("777")),
809            RestoreStep::SkipRemove
810        );
811        let frz = entry_freeze("/x", 42);
812        assert_eq!(
813            restore_step(&frz, Some(42), None),
814            RestoreStep::ThawOnly,
815            "alive freeze entry: nothing to restore beyond the unconditional thaw"
816        );
817        assert_eq!(
818            restore_step(&frz, None, None),
819            RestoreStep::SkipRemove,
820            "dead cgroup: restore_step only decides memory.high, never freeze — the \
821             unconditional thaw in Effector::thaw_raw already ran before this is consulted, \
822             so a still-frozen dead cgroup is never left behind (carry-forward: Task 5 review)"
823        );
824    }
825
826    /// Poll `cgfs::read_frozen` until it matches `want` or `timeout` elapses.
827    /// Writing `cgroup.freeze` only *requests* a state change; `cgroup.events`'s
828    /// `frozen` field (what `read_frozen` reads) only flips once the kernel has
829    /// actually quiesced every task in the cgroup, which can lag the write by a
830    /// few milliseconds under load — so a bare immediate read is flaky.
831    fn wait_for_frozen(cgroup: &str, want: bool, timeout: Duration) -> Option<bool> {
832        let deadline = std::time::Instant::now() + timeout;
833        loop {
834            let got = cgfs::read_frozen(cgroup);
835            if got == Some(want) || std::time::Instant::now() >= deadline {
836                return got;
837            }
838            std::thread::sleep(Duration::from_millis(20));
839        }
840    }
841
842    fn test_resolution(cgroup: String) -> Resolution {
843        Resolution {
844            cgroup,
845            unit: None,
846            verdict: Verdict::Freeze,
847            coverage: Coverage::Full,
848            mechanism: Mechanism::Raw,
849        }
850    }
851
852    /// Integration smoke test: freeze a real `sleep` via its (raw, rlm-created)
853    /// cgroup through the journal-backed Effector, confirm it's paused via
854    /// `cgroup.freeze` and journaled, then thaw and confirm the journal entry
855    /// is gone. Only works under cgroup v2 delegation, so it's `#[ignore]`d.
856    #[test]
857    #[ignore = "requires cgroup v2 delegation; run manually"]
858    fn freeze_thaw_real_process_raw_cgroup() {
859        use common::Limit;
860        use std::process::Command;
861
862        let manager = CgroupManager::new().expect("create CgroupManager");
863        let journal_dir = tempfile::tempdir().unwrap();
864        let journal =
865            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
866        let effector = Effector::new(&manager, &journal, None);
867
868        let abs_path = manager
869            .prepare_cgroup("test-freeze-thaw", &Limit::default())
870            .expect("create test cgroup")
871            .path;
872        let cgroup = format!(
873            "/{}",
874            abs_path
875                .strip_prefix("/sys/fs/cgroup")
876                .expect("cgroup under /sys/fs/cgroup")
877                .display()
878        );
879
880        let mut child = Command::new("sleep")
881            .arg("30")
882            .spawn()
883            .expect("spawn sleep");
884        let pid = child.id();
885        manager
886            .add_to_cgroup(&abs_path, pid)
887            .expect("add sleep to test cgroup");
888
889        let res = test_resolution(cgroup.clone());
890
891        effector
892            .apply(&Action::Freeze {
893                res: res.clone(),
894                name: "sleep".into(),
895            })
896            .expect("freeze");
897
898        assert_eq!(
899            wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
900            Some(true),
901            "cgroup should be frozen"
902        );
903        assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
904
905        effector
906            .apply(&Action::Thaw { res: res.clone() })
907            .expect("thaw");
908
909        assert_eq!(
910            wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
911            Some(false),
912            "cgroup should be thawed"
913        );
914        assert!(
915            journal.entries().is_empty(),
916            "journal entry should be removed after thaw"
917        );
918
919        let _ = child.kill();
920        let _ = child.wait();
921        let _ = manager.cleanup_cgroup("test-freeze-thaw");
922    }
923
924    /// Act-in-place integration test: freeze/thaw a real transient systemd
925    /// `--user --scope` unit through the Unit mechanism (D-Bus, with raw
926    /// fallback), confirming the journal-backed round trip end to end.
927    /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
928    #[test]
929    #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
930    fn freeze_thaw_real_transient_scope() {
931        use std::process::Command;
932
933        let uid_out = Command::new("id").arg("-u").output().expect("id -u");
934        let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
935            .trim()
936            .parse()
937            .expect("parse uid");
938
939        let unit_base = format!("rlm-e2e-{}", std::process::id());
940        let unit = format!("{unit_base}.scope");
941        let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
942
943        let mut child = Command::new("systemd-run")
944            .args([
945                "--user",
946                "--scope",
947                "--slice=app.slice",
948                &format!("--unit={unit_base}"),
949                "--",
950                "sleep",
951                "30",
952            ])
953            .spawn()
954            .expect("spawn systemd-run --scope");
955
956        // Give systemd a moment to register the transient scope's cgroup.
957        std::thread::sleep(Duration::from_millis(300));
958
959        let manager = CgroupManager::new().expect("create CgroupManager");
960        let journal_dir = tempfile::tempdir().unwrap();
961        let journal =
962            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
963        let systemd = SystemdUser::connect();
964        let effector = Effector::new(&manager, &journal, systemd.as_ref());
965
966        let res = Resolution {
967            cgroup: cgroup.clone(),
968            unit: Some(unit),
969            verdict: Verdict::Freeze,
970            coverage: Coverage::Full,
971            mechanism: Mechanism::Unit,
972        };
973
974        effector
975            .apply(&Action::Freeze {
976                res: res.clone(),
977                name: "sleep".into(),
978            })
979            .expect("freeze");
980        assert_eq!(
981            wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
982            Some(true),
983            "scope cgroup should be frozen"
984        );
985        assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
986
987        effector.apply(&Action::Thaw { res }).expect("thaw");
988        assert_eq!(
989            wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
990            Some(false),
991            "scope cgroup should be thawed"
992        );
993        assert!(
994            journal.entries().is_empty(),
995            "journal entry should be removed after thaw"
996        );
997
998        let _ = child.kill();
999        let _ = child.wait();
1000        let _ = Command::new("systemctl")
1001            .args(["--user", "stop", &format!("{unit_base}.scope")])
1002            .status();
1003    }
1004
1005    /// Regression test for Task 6 review Critical #1: cap a real cgroup and
1006    /// assert `cgfs::read_high` matches the journaled `our_high` exactly —
1007    /// i.e. the value we wrote never got silently page-truncated out from
1008    /// under the journal. The target process holds enough random anon
1009    /// memory (well above the 256 MiB `MIN_CAP_BYTES` floor, a round power
1010    /// of two that would pass even without the fix and prove nothing) that
1011    /// the sized cap is very unlikely to already sit on a page boundary. Requires cgroup v2
1012    /// delegation, so it's `#[ignore]`d.
1013    #[test]
1014    #[ignore = "requires cgroup v2 delegation; run manually"]
1015    fn cap_page_aligns_and_journal_matches_on_disk_value() {
1016        use common::Limit;
1017        use std::process::Command;
1018
1019        let manager = CgroupManager::new().expect("create CgroupManager");
1020        let journal_dir = tempfile::tempdir().unwrap();
1021        let journal =
1022            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1023        let effector = Effector::new(&manager, &journal, None);
1024
1025        let abs_path = manager
1026            .prepare_cgroup("test-cap-align", &Limit::default())
1027            .expect("create test cgroup")
1028            .path;
1029        let cgroup = format!(
1030            "/{}",
1031            abs_path
1032                .strip_prefix("/sys/fs/cgroup")
1033                .expect("cgroup under /sys/fs/cgroup")
1034                .display()
1035        );
1036
1037        // Hold ~400MB of anon memory (above the 256 MiB floor) whose sized
1038        // cap is very unlikely to land on a page boundary, then sleep.
1039        let mut child = Command::new("bash")
1040            .arg("-c")
1041            .arg("a=$(head -c 300000000 /dev/urandom | base64 -w0); sleep 30")
1042            .spawn()
1043            .expect("spawn memory-holding process");
1044        let pid = child.id();
1045        manager
1046            .add_to_cgroup(&abs_path, pid)
1047            .expect("add process to test cgroup");
1048        // Wait until the shell has actually built up the allocation, or the
1049        // cap would sit on the 256 MiB floor and prove nothing.
1050        let want = 300 * 1024 * 1024;
1051        let deadline = std::time::Instant::now() + Duration::from_secs(10);
1052        loop {
1053            let cur = cgfs::current_bytes(&cgroup).unwrap_or(0);
1054            if cur >= want {
1055                break;
1056            }
1057            if std::time::Instant::now() >= deadline {
1058                let _ = child.kill();
1059                let _ = child.wait();
1060                let _ = manager.cleanup_cgroup("test-cap-align");
1061                panic!("memory.current reached only {cur} bytes, need {want}, after 10s");
1062            }
1063            std::thread::sleep(Duration::from_millis(100));
1064        }
1065
1066        let res = test_resolution(cgroup.clone());
1067        effector
1068            .apply(&Action::Cap {
1069                res: res.clone(),
1070                name: "mem-hog".into(),
1071            })
1072            .expect("cap");
1073
1074        let entries = journal.entries();
1075        assert_eq!(entries.len(), 1, "cap should be journaled");
1076        let journaled = entries[0].our_high.clone().expect("our_high recorded");
1077
1078        let on_disk = cgfs::read_high(&cgroup).expect("read memory.high");
1079        assert_eq!(
1080            on_disk, journaled,
1081            "on-disk memory.high must match the journaled our_high exactly"
1082        );
1083
1084        let bytes: u64 = journaled.parse().expect("our_high is a plain decimal");
1085        assert_eq!(bytes % page_size(), 0, "written value must be page-aligned");
1086
1087        effector.apply(&Action::LiftCap { res }).expect("lift cap");
1088        assert!(
1089            journal.entries().is_empty(),
1090            "journal entry removed after lift"
1091        );
1092
1093        let _ = child.kill();
1094        let _ = child.wait();
1095        let _ = manager.cleanup_cgroup("test-cap-align");
1096    }
1097
1098    /// Cap and lift go through raw `memory.high` writes only, so the unit's
1099    /// own `MemoryHigh` property (as systemd reports it) must be the same
1100    /// before the cap and after the lift, and the raw `memory.high` must be
1101    /// back at its pre-cap value. A runtime `SetUnitProperties` would leave
1102    /// a `/run` drop-in behind that changes the reported property.
1103    /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
1104    #[test]
1105    #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
1106    fn lift_cap_leaves_unit_memory_high_property_untouched() {
1107        use std::process::Command;
1108
1109        let unit_memory_high = |unit: &str| -> String {
1110            let out = Command::new("systemctl")
1111                .args(["--user", "show", "-p", "MemoryHigh", "--value", unit])
1112                .output()
1113                .expect("systemctl --user show");
1114            String::from_utf8_lossy(&out.stdout).trim().to_string()
1115        };
1116
1117        let uid_out = Command::new("id").arg("-u").output().expect("id -u");
1118        let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
1119            .trim()
1120            .parse()
1121            .expect("parse uid");
1122
1123        let unit_base = format!("rlm-e2e-cap-{}", std::process::id());
1124        let unit = format!("{unit_base}.scope");
1125        let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
1126
1127        let mut child = Command::new("systemd-run")
1128            .args([
1129                "--user",
1130                "--scope",
1131                "--slice=app.slice",
1132                &format!("--unit={unit_base}"),
1133                // A genuine prior MemoryHigh, distinct from both "max" and
1134                // whatever Cap computes, so the restored value is
1135                // unambiguous.
1136                "--property=MemoryHigh=1500M",
1137                "--",
1138                "sleep",
1139                "30",
1140            ])
1141            .spawn()
1142            .expect("spawn systemd-run --scope with MemoryHigh set");
1143
1144        std::thread::sleep(Duration::from_millis(300));
1145
1146        let manager = CgroupManager::new().expect("create CgroupManager");
1147        let journal_dir = tempfile::tempdir().unwrap();
1148        let journal =
1149            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1150        let systemd = SystemdUser::connect();
1151        let effector = Effector::new(&manager, &journal, systemd.as_ref());
1152
1153        let original_high =
1154            cgfs::read_high(&cgroup).expect("systemd-run set an initial memory.high");
1155        assert_ne!(
1156            original_high, "max",
1157            "test needs a concrete prior MemoryHigh to distinguish from systemd's clear-to-max"
1158        );
1159        let property_before = unit_memory_high(&unit);
1160
1161        let res = Resolution {
1162            cgroup: cgroup.clone(),
1163            unit: Some(unit.clone()),
1164            verdict: Verdict::CapOnly,
1165            coverage: Coverage::Full,
1166            mechanism: Mechanism::Unit,
1167        };
1168
1169        effector
1170            .apply(&Action::Cap {
1171                res: res.clone(),
1172                name: "sleep".into(),
1173            })
1174            .expect("cap");
1175        assert_ne!(
1176            cgfs::read_high(&cgroup),
1177            Some(original_high.clone()),
1178            "cap should have changed memory.high"
1179        );
1180
1181        effector.apply(&Action::LiftCap { res }).expect("lift cap");
1182        assert_eq!(
1183            unit_memory_high(&unit),
1184            property_before,
1185            "cap/lift must not change the unit's MemoryHigh property"
1186        );
1187        assert_eq!(
1188            cgfs::read_high(&cgroup),
1189            Some(original_high),
1190            "lift must restore the raw pre-cap memory.high"
1191        );
1192        assert!(
1193            journal.entries().is_empty(),
1194            "journal entry removed after lift"
1195        );
1196
1197        let _ = child.kill();
1198        let _ = child.wait();
1199        let _ = Command::new("systemctl")
1200            .args(["--user", "stop", &unit])
1201            .status();
1202    }
1203
1204    /// Regression test for Promoted Minor B: a leaked `[Cap, Freeze]` chain
1205    /// (Cap journaled first, then the same cgroup frozen later without the
1206    /// Cap entry ever being cleared) must still restore the Cap's
1207    /// `prev_high` on thaw. Before the fix, liveness was judged against
1208    /// `entries.last()` — the Freeze entry, which has no `memory.high` of
1209    /// its own — so `restore_step` reported "nothing to restore" and
1210    /// `memory.high` stayed pinned at the guard's Cap value forever once the
1211    /// whole chain was cleared. Requires cgroup v2 delegation, so it's
1212    /// `#[ignore]`d.
1213    #[test]
1214    #[ignore = "requires cgroup v2 delegation; run manually"]
1215    fn chain_restores_cap_value_when_newest_entry_is_freeze() {
1216        use common::Limit;
1217        use std::process::Command;
1218
1219        let manager = CgroupManager::new().expect("create CgroupManager");
1220        let journal_dir = tempfile::tempdir().unwrap();
1221        let journal =
1222            Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1223        let effector = Effector::new(&manager, &journal, None);
1224
1225        let abs_path = manager
1226            .prepare_cgroup("test-chain-restore", &Limit::default())
1227            .expect("create test cgroup")
1228            .path;
1229        let cgroup = format!(
1230            "/{}",
1231            abs_path
1232                .strip_prefix("/sys/fs/cgroup")
1233                .expect("cgroup under /sys/fs/cgroup")
1234                .display()
1235        );
1236
1237        let mut child = Command::new("sleep")
1238            .arg("30")
1239            .spawn()
1240            .expect("spawn sleep");
1241        let pid = child.id();
1242        manager
1243            .add_to_cgroup(&abs_path, pid)
1244            .expect("add sleep to test cgroup");
1245
1246        let original_high = cgfs::read_high(&cgroup).expect("read initial memory.high");
1247        let res = test_resolution(cgroup.clone());
1248
1249        // Cap first (oldest entry)...
1250        effector
1251            .apply(&Action::Cap {
1252                res: res.clone(),
1253                name: "sleep".into(),
1254            })
1255            .expect("cap");
1256        assert_ne!(
1257            cgfs::read_high(&cgroup),
1258            Some(original_high.clone()),
1259            "cap should have changed memory.high"
1260        );
1261
1262        // ...then freeze the same cgroup without ever clearing the Cap
1263        // entry — this is the leaked-chain scenario: two coexisting entries,
1264        // Freeze newest.
1265        effector
1266            .apply(&Action::Freeze {
1267                res: res.clone(),
1268                name: "sleep".into(),
1269            })
1270            .expect("freeze");
1271
1272        let entries = journal.entries();
1273        assert_eq!(entries.len(), 2, "both Cap and Freeze entries coexist");
1274        assert_eq!(entries[0].action, JournalAction::Cap, "Cap is oldest");
1275        assert_eq!(entries[1].action, JournalAction::Freeze, "Freeze is newest");
1276
1277        // Thaw (the action that would naturally follow a Freeze) must still
1278        // restore the Cap's prev_high, not skip restoration just because the
1279        // newest entry is a Freeze with no memory.high of its own.
1280        effector.apply(&Action::Thaw { res }).expect("thaw");
1281        assert_eq!(
1282            cgfs::read_high(&cgroup),
1283            Some(original_high),
1284            "thaw must restore the chain's Cap value even though Freeze is the newest entry"
1285        );
1286        assert!(
1287            journal.entries().is_empty(),
1288            "both chain entries removed after thaw"
1289        );
1290
1291        let _ = child.kill();
1292        let _ = child.wait();
1293        let _ = manager.cleanup_cgroup("test-chain-restore");
1294    }
1295}