rlmctl_core/guard/effector.rs
1//! Executes [`Action`]s against real cgroups, acting in place on the cgroup a
2//! process already lives in (a systemd unit's scope/service, or an existing
3//! rlm rule cgroup) rather than moving it into an ephemeral `guard-<pid>`
4//! cgroup. Every action is best-effort and logged; a failure must never
5//! panic or otherwise crash the daemon loop. `apply` may return `Err` so the
6//! caller can log it. Desktop notifications are not sent from here: the
7//! daemon loop hands each applied action to `guard::notify`.
8//!
9//! # Write-ahead journal
10//! Freeze/Cap always `journal.append` (which fsyncs) *before* touching the
11//! cgroup, so a crash between the two still leaves a durable record that
12//! startup recovery (`sweep_leftovers`) can replay. `Journal` is internally
13//! mutex-serialized (see `journal.rs`), and the daemon calls into this
14//! `Effector` from a single thread (the tick loop in `rlm-guard`'s `main`),
15//! so `Effector` just holds a `&Journal` and relies on the daemon's
16//! single-threaded call discipline plus the journal's own internal locking
17//! for safety; it adds no locking of its own.
18//!
19//! # Mechanism: systemd unit vs. raw cgroupfs
20//! When a target resolved to a systemd unit (`Mechanism::Unit`), we prefer
21//! the D-Bus call (`FreezeUnit`/`ThawUnit`) with a hard
22//! 2s timeout (`systemd::SystemdUser`'s own enforced deadline); on `Err` (bus
23//! unavailable, call failed, or timed out) we fall back to the raw cgroupfs
24//! primitives in `cgfs`. Raw-mechanism targets (rlm's own rule cgroups) skip
25//! the D-Bus attempt entirely. Thawing is mechanism-independent: we always
26//! perform the raw `cgroup.freeze` write first, unconditionally, and only
27//! then attempt `ThawUnit` best-effort so systemd's view matches. This
28//! guarantees a cgroup is never left frozen because of a stale unit or a
29//! slow bus, and tolerates a missing cgroup (the raw write simply errors and
30//! we move on). On shutdown and startup replay every raw thaw and restore
31//! finishes before the first D-Bus call. Caps are the exception: they never
32//! go through systemd (see below).
33//!
34//! # Restoring `memory.high`
35//! The kernel truncates `memory.high` writes to page multiples, so the byte
36//! count we cap to is page-aligned *before* we journal/write it
37//! ([`page_align_down`]); as a second line of defense, [`Effector::cap`]
38//! reads `memory.high` back after writing and self-corrects the journal if
39//! reality still differs.
40//! Caps and restores are raw `memory.high` writes only, never systemd's
41//! `SetUnitProperties(MemoryHigh)`: a runtime property leaves a `/run`
42//! drop-in that outlives the cap and masks the unit's own configured
43//! `MemoryHigh` until reboot, and restoring through systemd wrote `infinity`
44//! over it. If systemd later re-applies the unit's value, that only lifts our
45//! cap early, and the journal's `our_high` check then skips the restore (the
46//! safe direction).
47//!
48//! If more than one journal entry ever coexists for the same
49//! cgroup (a leak from an incomplete prior removal), every restore path
50//! treats them as one chain: liveness is judged against the chain's `Cap`
51//! entry specifically (only a `Cap` has a `memory.high` to restore; a
52//! `Freeze` entry does not, so judging against `entries.last()` when it
53//! happens to be a newer `Freeze` would wrongly look like "nothing to
54//! restore" and strand `memory.high` at the guard's value forever), and the
55//! value restored is always the *oldest* `Cap` entry's `prev_high`, the
56//! true pre-intervention value, not an intermediate entry's `prev_high`
57//! (which is just our own previous `our_high`). See [`restore_target`].
58//! Liveness, however, is judged against the *newest* `Cap` entry, not the
59//! oldest: entries are strictly appended, so a later Cap's write always
60//! supersedes an earlier one's on disk, and `should_restore`'s string
61//! comparison must match what's actually there. See [`restore_decision`].
62
63use super::cgfs;
64use super::journal::{should_restore, Journal, JournalAction, JournalEntry};
65use super::resolve::{Mechanism, Resolution};
66use super::sampler::parse_meminfo;
67use super::systemd::SystemdUser;
68use super::types::Action;
69use crate::CgroupManager;
70use common::Result;
71use std::time::Duration;
72
73/// Floor for any soft cap. A cap below this is effectively a freeze for a
74/// desktop app, so small cgroups are never squeezed further than this.
75pub const MIN_CAP_BYTES: u64 = 256 * 1024 * 1024;
76
77/// What a successful [`Effector::apply`] did, beyond "it worked".
78#[derive(Debug, Clone, Copy, PartialEq, Eq)]
79pub struct Applied {
80 /// The `memory.high` a `Cap` wrote, in bytes. `None` for other actions.
81 pub cap_bytes: Option<u64>,
82}
83
84/// Hard deadline for every D-Bus call on the freeze-guard's storm path. On
85/// `Err` (including a timeout) callers fall back to raw cgroupfs writes.
86const DBUS_TIMEOUT: Duration = Duration::from_secs(2);
87
88/// Applies guard [`Action`]s in place on the resolved target cgroup, via the
89/// systemd-unit D-Bus path (with raw-cgroup fallback) and a write-ahead
90/// [`Journal`] for crash-safe restore.
91pub struct Effector<'a> {
92 /// Kept only for the legacy `guard-<pid>` sweep during the upgrade path
93 /// (see [`Effector::sweep_leftovers`]); acting in place no longer uses it.
94 manager: &'a CgroupManager,
95 journal: &'a Journal,
96 systemd: Option<&'a SystemdUser>,
97}
98
99impl<'a> Effector<'a> {
100 pub fn new(
101 manager: &'a CgroupManager,
102 journal: &'a Journal,
103 systemd: Option<&'a SystemdUser>,
104 ) -> Self {
105 Self {
106 manager,
107 journal,
108 systemd,
109 }
110 }
111
112 /// Apply a single action. Best-effort: returns `Err` only so the caller can
113 /// log it. On success, [`Applied::cap_bytes`] holds the `memory.high` a
114 /// `Cap` wrote.
115 pub fn apply(&self, action: &Action) -> Result<Applied> {
116 let done = Applied { cap_bytes: None };
117 match action {
118 Action::Freeze { res, name } => self.freeze(res, name).map(|()| done),
119 Action::Thaw { res } => self.thaw(res).map(|()| done),
120 Action::Cap { res, name } => self.cap(res, name).map(|bytes| Applied {
121 cap_bytes: Some(bytes),
122 }),
123 Action::LiftCap { res } => self.lift_cap(res).map(|()| done),
124 }
125 }
126
127 fn freeze(&self, res: &Resolution, name: &str) -> Result<()> {
128 // The inode is the guard that lets a later restore tell "this is
129 // still the same cgroup" from "this cgroup was torn down and
130 // recreated" (`should_restore`). `unwrap_or(0)` used to substitute a
131 // poison sentinel here: 0 is never a real inode, so the guard could
132 // never match again and the entry became permanently unrestorable,
133 // yet it was still journaled-and-acted-on, then later cleared and
134 // logged as if it were an intentional skip (Promoted Minor A). Fail
135 // closed instead: refuse to freeze at all rather than act with a
136 // record we can never safely restore from.
137 let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
138 tracing::warn!(
139 cgroup = %res.cgroup, name,
140 "cannot read cgroup inode; refusing to freeze (would be unrestorable)"
141 );
142 return Err(common::Error::Cgroup(format!(
143 "cannot read inode for {}; refusing to freeze",
144 res.cgroup
145 )));
146 };
147 let entry = JournalEntry {
148 cgroup: res.cgroup.clone(),
149 inode,
150 unit: res.unit.clone(),
151 action: JournalAction::Freeze,
152 prev_high: None,
153 our_high: None,
154 };
155 // Write-ahead: the entry must be durable (journal.append fsyncs)
156 // before we ever touch the cgroup, so a crash between the two still
157 // leaves a record startup recovery can act on.
158 self.journal.append(&entry)?;
159
160 tracing::info!(cgroup = %res.cgroup, name, "freezing cgroup");
161 if res.mechanism == Mechanism::Unit {
162 if let (Some(unit), Some(systemd)) = (&res.unit, self.systemd) {
163 match systemd.freeze_unit(unit, DBUS_TIMEOUT) {
164 Ok(()) => return Ok(()),
165 Err(e) => tracing::warn!(
166 cgroup = %res.cgroup, unit, error = %e,
167 "FreezeUnit failed; falling back to raw cgroup.freeze"
168 ),
169 }
170 }
171 }
172 cgfs::write_freeze(&res.cgroup, true)
173 }
174
175 fn thaw(&self, res: &Resolution) -> Result<()> {
176 tracing::info!(cgroup = %res.cgroup, "thawing cgroup");
177 // Gather any coexisting entries for this cgroup first: a `Thaw`
178 // action reverses a Frozen intervention, but if a Cap entry ever
179 // leaked alongside it (Important #3, Task 6 review) we must still
180 // restore its memory.high before wiping the journal record below,
181 // not just drop it silently.
182 let entries = self.entries_for(&res.cgroup);
183 let result = self.thaw_raw(&res.cgroup, res.unit.as_deref());
184 if let Err(e) = &result {
185 tracing::debug!(cgroup = %res.cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
186 }
187 self.restore_high_if_any(&res.cgroup, &entries);
188 self.journal.remove(&res.cgroup)?;
189 result
190 }
191
192 /// Returns the `memory.high` value written, in bytes.
193 fn cap(&self, res: &Resolution, name: &str) -> Result<u64> {
194 // See `freeze`'s matching comment (Promoted Minor A): a `0` inode
195 // sentinel here would make this Cap permanently unrestorable while
196 // looking like a real guard, so fail closed instead of
197 // journal-and-act with an unrestorable record.
198 let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
199 tracing::warn!(
200 cgroup = %res.cgroup, name,
201 "cannot read cgroup inode; refusing to cap (would be unrestorable)"
202 );
203 return Err(common::Error::Cgroup(format!(
204 "cannot read inode for {}; refusing to cap",
205 res.cgroup
206 )));
207 };
208 let prev_high = cgfs::read_high(&res.cgroup);
209 // Unknown swap total counts as 0: assuming anon is pinned only makes
210 // the cap gentler.
211 let swap_total_kb = std::fs::read_to_string("/proc/meminfo")
212 .ok()
213 .and_then(|m| parse_meminfo(&m))
214 .map_or(0, |m| m.swap_total_kb);
215 let anon_ok = cgfs::anon_reclaimable(&res.cgroup, swap_total_kb);
216 let Some(target) = cap_target(
217 cgfs::current_bytes(&res.cgroup),
218 cgfs::file_bytes(&res.cgroup),
219 anon_ok,
220 ) else {
221 tracing::warn!(
222 cgroup = %res.cgroup, name,
223 "cannot read memory.current; refusing to cap"
224 );
225 return Err(common::Error::Cgroup(
226 "cannot read memory.current; refusing to cap".into(),
227 ));
228 };
229 // The kernel truncates `memory.high` writes to page multiples, so we
230 // must journal/write the value it will actually store, not the raw
231 // target, or `should_restore`'s string-equality check can never
232 // pass again and the cap becomes permanent.
233 let our_bytes = page_align_down(target, page_size());
234 // Plain decimal bytes, no separators/whitespace: this must be
235 // exactly what a later `cgfs::read_high` (which only trims
236 // whitespace off the raw file contents) returns. See
237 // `our_high_string_is_plain_decimal_no_separators` below, and
238 // `reconcile_our_high` for the belt-and-braces check.
239 if !cap_tightens(prev_high.as_deref(), our_bytes) {
240 tracing::info!(
241 cgroup = %res.cgroup, name, prev_high = ?prev_high, our_bytes,
242 "existing memory.high already at or below the cap; not capping"
243 );
244 return Err(common::Error::Cgroup(
245 "existing memory.high already at or below the cap; refusing to cap".into(),
246 ));
247 }
248 let our_high = our_bytes.to_string();
249
250 let entry = JournalEntry {
251 cgroup: res.cgroup.clone(),
252 inode,
253 unit: res.unit.clone(),
254 action: JournalAction::Cap,
255 prev_high,
256 our_high: Some(our_high.clone()),
257 };
258 self.journal.append(&entry)?;
259
260 tracing::info!(
261 cgroup = %res.cgroup, name, our_high = %our_high, anon_reclaimable = anon_ok,
262 "soft-capping cgroup"
263 );
264 cgfs::write_high(&res.cgroup, &our_high)?;
265 self.reconcile_our_high(&entry);
266 Ok(our_bytes)
267 }
268
269 /// After a successful `Cap` write, read `memory.high` back; if what's
270 /// actually on disk differs from what we journaled (page truncation we
271 /// didn't fully pre-empt), correct the journal to match reality. Otherwise
272 /// `should_restore`'s string-equality check can never pass again and
273 /// the cap becomes permanent (Task 6 review, Critical #1). Rebuilds
274 /// only this cgroup's entries, preserving any others that might coexist
275 /// (Important #3), with `written`'s `our_high` corrected to the
276 /// read-back value. Uses `Journal::replace` (a single atomic rewrite),
277 /// not a separate `remove` then `append`: the latter has a window where
278 /// the cgroup has no journal record at all, so a crash/SIGKILL/append
279 /// failure right there would leave a permanent, unrecoverable cap,
280 /// exactly the invariant the journal exists to prevent (Task 6 review,
281 /// fix round 2).
282 fn reconcile_our_high(&self, written: &JournalEntry) {
283 let cgroup: &str = &written.cgroup;
284 let Some(actual) = cgfs::read_high(cgroup) else {
285 return;
286 };
287 if written.our_high.as_deref() == Some(actual.as_str()) {
288 return;
289 }
290 tracing::warn!(
291 cgroup = %cgroup, journaled = ?written.our_high, actual = %actual,
292 "memory.high on disk differs from what we journaled; correcting journal entry"
293 );
294 let mut entries = self.entries_for(cgroup);
295 let Some(pos) = entries.iter().rposition(|e| e == written) else {
296 // Already removed/replaced by something else (e.g. a concurrent
297 // Thaw/LiftCap); nothing left to correct.
298 return;
299 };
300 entries[pos].our_high = Some(actual);
301 if let Err(e) = self.journal.replace(cgroup, &entries) {
302 tracing::warn!(cgroup = %cgroup, error = %e, "failed to correct journal entry (atomic replace)");
303 }
304 }
305
306 fn lift_cap(&self, res: &Resolution) -> Result<()> {
307 tracing::info!(cgroup = %res.cgroup, "lifting cap");
308 let entries = self.entries_for(&res.cgroup);
309 // Mechanism-independent thaw always runs, regardless of whether any
310 // journal record exists: a dead-cgroup prune (carry-forward
311 // finding, Task 5 review) must never leave a target frozen just
312 // because we lost the journal entry.
313 let _ = self.thaw_raw(&res.cgroup, res.unit.as_deref());
314 self.restore_high_if_any(&res.cgroup, &entries);
315 self.journal.remove(&res.cgroup)
316 }
317
318 /// All journal entries for one cgroup, in the order `Journal::entries()`
319 /// returns them, oldest-first, since entries are strictly appended.
320 fn entries_for(&self, cgroup: &str) -> Vec<JournalEntry> {
321 self.journal
322 .entries()
323 .into_iter()
324 .filter(|e| e.cgroup == cgroup)
325 .collect()
326 }
327
328 /// Startup recovery: legacy `guard-<pid>` sweep (kept for one release as
329 /// an upgrade path from pre-act-in-place rlm), then replay every live
330 /// journal entry.
331 pub fn sweep_leftovers(&self) -> Result<()> {
332 if let Err(e) = self.manager.sweep_guard_leftovers() {
333 tracing::warn!(error = %e, "legacy guard-<pid> sweep failed (non-fatal)");
334 }
335 self.replay_and_clear()
336 }
337
338 /// Graceful shutdown: undo every live journal entry, then clear the
339 /// journal. Same replay as `sweep_leftovers`, minus the legacy sweep.
340 pub fn undo_all(&self) -> Result<()> {
341 self.replay_and_clear()
342 }
343
344 fn replay_and_clear(&self) -> Result<()> {
345 // Group by cgroup first (entries for the same cgroup aren't
346 // necessarily contiguous in the file) so each cgroup's chain is
347 // replayed as one unit (see `restore_high_if_any`) rather than
348 // entry-by-entry, which would mis-restore whenever more than one
349 // entry coexists for a cgroup (Important #3, Task 6 review).
350 let mut by_cgroup: std::collections::HashMap<String, Vec<JournalEntry>> =
351 std::collections::HashMap::new();
352 for e in self.journal.entries() {
353 by_cgroup.entry(e.cgroup.clone()).or_default().push(e);
354 }
355 // Pass 1: every raw thaw and memory.high restore, then clear the
356 // journal. No D-Bus call happens before this is done, so a slow
357 // session bus cannot push the undo past systemd's stop timeout.
358 for (cgroup, entries) in &by_cgroup {
359 if let Err(e) = cgfs::write_freeze(cgroup, false) {
360 tracing::debug!(cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
361 }
362 self.restore_high_if_any(cgroup, entries);
363 }
364 let cleared = self.journal.clear();
365 // Pass 2, best effort: tell systemd so its view of the units
366 // matches the kernel's. The processes already run again.
367 if let Some(systemd) = self.systemd {
368 for (cgroup, entries) in &by_cgroup {
369 if let Some(unit) = entries.last().and_then(|e| e.unit.as_deref()) {
370 if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
371 tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
372 }
373 }
374 }
375 }
376 cleared
377 }
378
379 /// Restore `memory.high` for one cgroup's journal `entries` (oldest-first),
380 /// if the chain is still live. Called *after* the caller has already
381 /// performed the mechanism-independent thaw (see module docs: no path
382 /// here ever means "leave frozen"). All the decision logic is delegated
383 /// to the pure [`restore_decision`]: liveness is judged against the
384 /// *newest* `Cap` entry (whose write is what's actually on disk right
385 /// now), while the value restored is [`restore_target`]'s *oldest*-entry
386 /// `prev_high` (the true pre-intervention value). Judging liveness
387 /// against the oldest Cap instead (NEW-1 regression) leaves a stacked
388 /// `[Cap, Cap]` chain's on-disk value permanently un-restorable, since
389 /// the oldest entry's `our_high` never matches what a later Cap actually
390 /// wrote. Judging liveness against `entries.last()` has the same failure
391 /// for a `[Cap, Freeze]` chain (Promoted Minor B): the newest entry is
392 /// the `Freeze`, which has no `memory.high` of its own. The restore is a
393 /// raw `memory.high` write only; systemd's `MemoryHigh` property is never
394 /// touched (see module docs).
395 fn restore_high_if_any(&self, cgroup: &str, entries: &[JournalEntry]) {
396 let inode = cgfs::dir_inode(cgroup);
397 let high = cgfs::read_high(cgroup);
398 match restore_decision(entries, inode, high.as_deref()) {
399 RestoreStep::ThawAndRestoreHigh { to } => {
400 if let Err(err) = cgfs::write_high(cgroup, &to) {
401 tracing::warn!(cgroup, error = %err, "failed to restore memory.high");
402 }
403 }
404 // No Cap entry anywhere in the chain (a Freeze-only chain,
405 // possibly with leaked duplicates): there is no memory.high to
406 // restore. The unconditional thaw already ran in the caller, so
407 // there's nothing left to do here.
408 RestoreStep::ThawOnly => {}
409 RestoreStep::SkipRemove => {
410 tracing::warn!(
411 cgroup,
412 "not restoring memory.high (cgroup recreated or value changed since our write)"
413 );
414 }
415 }
416 }
417
418 /// Mechanism-independent thaw: an *unconditional* raw `cgroup.freeze`
419 /// write first, so a cgroup is never left frozen and a slow bus cannot
420 /// delay it, then a best-effort `ThawUnit` (if we have a unit and a bus)
421 /// so systemd's view of the unit matches. Tolerates a missing cgroup:
422 /// the raw write then simply returns `Err`, which every caller here
423 /// treats as non-fatal.
424 fn thaw_raw(&self, cgroup: &str, unit: Option<&str>) -> Result<()> {
425 let result = cgfs::write_freeze(cgroup, false);
426 if let (Some(unit), Some(systemd)) = (unit, self.systemd) {
427 if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
428 tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
429 }
430 }
431 result
432 }
433}
434
435/// What to do with one journal entry's `memory.high` at restore time. Never
436/// speaks to freeze/thaw; that is unconditional and already handled by the
437/// caller (`Effector::thaw_raw`) before `restore_step` is even consulted, so
438/// no variant here can ever mean "leave it frozen".
439#[derive(Debug, PartialEq, Eq)]
440pub enum RestoreStep {
441 /// A `Freeze` entry whose guard still holds: nothing to restore, thaw
442 /// (already done by the caller) was the whole job.
443 ThawOnly,
444 /// A `Cap` entry whose guard still holds: restore `memory.high` to `to`
445 /// (the entry's `prev_high`, "max" if it was never recorded).
446 ThawAndRestoreHigh { to: String },
447 /// The guard no longer holds (cgroup recreated, or someone else changed
448 /// `memory.high` since our write): don't touch `memory.high`, just let
449 /// the caller remove the journal entry.
450 SkipRemove,
451}
452
453/// Pure: what to do for one journal entry at restore time, delegating the
454/// safety check to [`should_restore`].
455pub fn restore_step(e: &JournalEntry, inode: Option<u64>, high: Option<&str>) -> RestoreStep {
456 if !should_restore(e, inode, high) {
457 return RestoreStep::SkipRemove;
458 }
459 match e.action {
460 JournalAction::Freeze => RestoreStep::ThawOnly,
461 JournalAction::Cap => RestoreStep::ThawAndRestoreHigh {
462 to: e.prev_high.clone().unwrap_or_else(|| "max".into()),
463 },
464 }
465}
466
467/// Size a soft cap that slows an app down without stalling it.
468///
469/// memory.high applies to everything charged to the cgroup, page cache
470/// included, so the cap is sized from memory.current: it never asks the
471/// kernel to reclaim more than 10% of it. When anon memory cannot go to swap,
472/// only file pages can be reclaimed, so the cap never asks for more than 80%
473/// of them. Returns `None` when memory.current is unreadable: no cap is safer
474/// than a guessed one.
475pub fn cap_target(current: Option<u64>, file: Option<u64>, anon_reclaimable: bool) -> Option<u64> {
476 let current = current?;
477 let mut target = current / 10 * 9;
478 if !anon_reclaimable {
479 let file_floor = current.saturating_sub(file.unwrap_or(0).saturating_mul(8) / 10);
480 target = target.max(file_floor);
481 }
482 Some(target.max(MIN_CAP_BYTES))
483}
484
485/// Pure: whether writing `target` would tighten the existing `memory.high`.
486/// A numeric `prev_high` at or below `target` means the cap would loosen (or
487/// not change) the limit already in place, so the caller must not write it.
488/// `"max"`, or anything else that is not a byte count, is no limit at all.
489pub fn cap_tightens(prev_high: Option<&str>, target: u64) -> bool {
490 match prev_high.and_then(|s| s.trim().parse::<u64>().ok()) {
491 Some(prev) => prev > target,
492 None => true,
493 }
494}
495
496/// Pure: the full restore decision for one cgroup's journal `entries`
497/// (oldest-first), given the cgroup's current inode and on-disk
498/// `memory.high` (both already read by the caller, no IO here). Answers two
499/// orthogonal questions with two different entries on purpose:
500///
501/// - **Liveness** (is the guard we wrote still intact?) is judged against
502/// the *newest* `Cap` entry in the chain. Entries are strictly appended,
503/// so a later Cap's write always supersedes an earlier one's on disk, so
504/// `should_restore`'s string-equality check must compare against the
505/// entry whose `our_high` is what's actually there right now.
506/// - **Value** (what do we restore to?) is [`restore_target`]'s *oldest*
507/// `Cap` entry's `prev_high`, the true pre-intervention value, not an
508/// intermediate entry's `prev_high` (which is just our own previous
509/// `our_high` from an earlier cap in the same chain).
510///
511/// Judging both against the oldest Cap (NEW-1 regression) makes a stacked
512/// `[Cap, Cap]` chain's liveness check compare a stale `our_high` against
513/// the newer write on disk, so it never matches and the chain is treated as
514/// dead, stranding `memory.high` at the guard's value forever. Judging
515/// liveness against `entries.last()` fails the same way for `[Cap, Freeze]`
516/// (Promoted Minor B): the newest entry is the `Freeze`, which has no
517/// `memory.high` of its own. Returns [`RestoreStep::ThawOnly`] when there's
518/// no `Cap` entry anywhere in the chain (a `Freeze`-only chain): nothing to
519/// restore beyond the caller's unconditional thaw.
520pub fn restore_decision(
521 entries: &[JournalEntry],
522 inode: Option<u64>,
523 high: Option<&str>,
524) -> RestoreStep {
525 let Some(newest_cap) = entries
526 .iter()
527 .rev()
528 .find(|e| e.action == JournalAction::Cap)
529 else {
530 return RestoreStep::ThawOnly;
531 };
532 match restore_step(newest_cap, inode, high) {
533 RestoreStep::ThawAndRestoreHigh { .. } => match restore_target(entries) {
534 Some(to) => RestoreStep::ThawAndRestoreHigh { to },
535 // Unreachable: `newest_cap` being a Cap guarantees `restore_target`
536 // (which only needs *any* Cap entry) also finds one.
537 None => RestoreStep::ThawOnly,
538 },
539 other => other,
540 }
541}
542
543/// Pure: given all journal entries for one cgroup, oldest-first, select the
544/// `memory.high` value to restore, if the chain contains a `Cap` entry. The
545/// OLDEST `Cap` entry's `prev_high` is the true pre-intervention value; a
546/// later entry's `prev_high` is just our own previous `our_high` from an
547/// earlier cap in the same chain, not the original (Task 6 review,
548/// Important #3). Returns `None` if there's no `Cap` entry (e.g. a
549/// `Freeze`-only chain).
550pub fn restore_target(entries: &[JournalEntry]) -> Option<String> {
551 entries
552 .iter()
553 .find(|e| e.action == JournalAction::Cap)
554 .map(|e| e.prev_high.clone().unwrap_or_else(|| "max".into()))
555}
556
557/// Runtime page size in bytes, via `sysconf(_SC_PAGESIZE)`. Falls back to
558/// 4096 (by far the most common value) only if the syscall ever returns
559/// something nonsensical, a defensive fallback, not the source of truth,
560/// since the whole point is to match whatever the kernel actually enforces.
561fn page_size() -> u64 {
562 // SAFETY: sysconf(_SC_PAGESIZE) reads a static system parameter; no
563 // pointers involved, no side effects.
564 let p = unsafe { libc::sysconf(libc::_SC_PAGESIZE) };
565 if p > 0 {
566 p as u64
567 } else {
568 4096
569 }
570}
571
572/// Pure: round `bytes` down to a multiple of `page` (a no-op if `page` is 0).
573/// The kernel truncates `memory.high` writes to page multiples (Task 6
574/// review, Critical #1), so we must journal/write the value it will
575/// actually store, not the pre-truncation target; otherwise
576/// `should_restore`'s string-equality check can never pass again and a cap
577/// becomes permanent.
578fn page_align_down(bytes: u64, page: u64) -> u64 {
579 bytes.checked_div(page).map_or(bytes, |q| q * page)
580}
581
582#[cfg(test)]
583mod tests {
584 use super::super::resolve::{Coverage, Verdict};
585 use super::*;
586
587 const MIB: u64 = 1024 * 1024;
588 const GIB: u64 = 1024 * MIB;
589
590 #[test]
591 fn cap_never_demands_more_than_ten_percent_of_current() {
592 // Incident: Chrome scope with 1.5 GiB charged, mostly page cache, ~150 MiB anon.
593 let current = GIB * 3 / 2;
594 let cap = cap_target(Some(current), Some(GIB * 135 / 100), true).unwrap();
595 assert!(
596 cap >= current / 10 * 9,
597 "cap {cap} is below 90% of {current}"
598 );
599 }
600
601 #[test]
602 fn swapless_cap_only_asks_for_reclaimable_file_pages() {
603 assert_eq!(
604 cap_target(Some(GIB), Some(0), false),
605 Some(GIB),
606 "nothing reclaimable: demand nothing"
607 );
608 assert_eq!(
609 cap_target(Some(GIB), Some(512 * MIB), false),
610 Some(GIB / 10 * 9)
611 );
612 assert_eq!(
613 cap_target(Some(GIB), None, false),
614 Some(GIB),
615 "unknown file size: demand nothing"
616 );
617 }
618
619 #[test]
620 fn cap_has_a_256_mib_floor() {
621 assert_eq!(
622 cap_target(Some(100 * MIB), Some(0), true),
623 Some(MIN_CAP_BYTES)
624 );
625 }
626
627 #[test]
628 fn cap_never_loosens_an_existing_memory_high() {
629 let floor = cap_target(Some(100 * MIB), Some(0), true).unwrap();
630 assert_eq!(floor, MIN_CAP_BYTES);
631 let prev = (200 * MIB).to_string();
632 assert!(
633 !cap_tightens(Some(&prev), floor),
634 "200 MiB unit limit must not be raised to the 256 MiB floor"
635 );
636 assert!(
637 !cap_tightens(Some(&floor.to_string()), floor),
638 "equal: no-op"
639 );
640 assert!(cap_tightens(Some("max"), floor));
641 assert!(cap_tightens(None, floor));
642 let prev = (2 * GIB).to_string();
643 assert!(cap_tightens(Some(&prev), GIB));
644 }
645
646 #[test]
647 fn unreadable_current_refuses_to_cap() {
648 assert_eq!(cap_target(None, Some(1), true), None);
649 }
650
651 /// Carry-forward (Task 3 review): `our_high` must be a plain decimal
652 /// string with no grouping/whitespace, since `should_restore` compares
653 /// it by string equality against `cgfs::read_high` (which only trims).
654 #[test]
655 fn our_high_string_is_plain_decimal_no_separators() {
656 let bytes = cap_target(Some(12_345_678_900), None, true).unwrap();
657 let s = bytes.to_string();
658 assert!(
659 s.chars().all(|c| c.is_ascii_digit()),
660 "our_high must be plain digits, got {s:?}"
661 );
662 assert_eq!(s, format!("{bytes}"), "no formatting beyond plain decimal");
663 }
664
665 /// Task 6 review, Critical #1: the kernel truncates `memory.high`
666 /// writes to page multiples (the reviewer's own observed example:
667 /// writing 900_000_000 with a 4096-byte page reads back 899_997_696).
668 #[test]
669 fn page_align_down_rounds_to_page_multiple() {
670 assert_eq!(page_align_down(900_000_000, 4096), 899_997_696);
671 assert_eq!(
672 page_align_down(4096, 4096),
673 4096,
674 "already-aligned is a no-op"
675 );
676 assert_eq!(page_align_down(100, 0), 100, "page=0 guard is a no-op");
677 }
678
679 /// Task 6 review, Important #3: when duplicate/stacked `Cap` entries end
680 /// up coexisting for one cgroup (a leak from an incomplete prior
681 /// removal), the restore target must be the OLDEST entry's `prev_high`
682 /// (the true pre-intervention value). The newer entry's `prev_high` is
683 /// deliberately a different-looking value here to prove we don't
684 /// chain-follow it.
685 #[test]
686 fn restore_target_uses_oldest_caps_prev_high() {
687 let oldest = entry_cap("/x", 42, "max", "A");
688 let newest = entry_cap("/x", 42, "A-prime", "B");
689 assert_eq!(
690 restore_target(&[oldest, newest]),
691 Some("max".into()),
692 "must use the oldest entry's prev_high, not the newest's"
693 );
694 }
695
696 #[test]
697 fn restore_target_none_for_freeze_only_chain() {
698 assert_eq!(restore_target(&[entry_freeze("/x", 42)]), None);
699 }
700
701 /// Regression test for NEW-1: `restore_decision`'s liveness check must be
702 /// judged against the NEWEST `Cap` entry (whose `our_high` is what's
703 /// actually on disk), while the value restored stays the OLDEST `Cap`
704 /// entry's `prev_high`. A stacked `[Cap(prev="max", our="A"),
705 /// Cap(prev="A", our="B")]` chain with disk `memory.high == "B"` must
706 /// restore to `"max"`, not `SkipRemove`, which the old
707 /// `entries.iter().find()` (oldest-Cap-for-everything) produced: it
708 /// compared the oldest entry's `our_high` ("A") against disk ("B"),
709 /// never matched, and permanently stranded `memory.high`. Also covers
710 /// `[Cap, Freeze]` (Promoted Minor B's shape) and a bare `[Cap]` chain in
711 /// the same test so a future re-swap of either selection can't slip by.
712 #[test]
713 fn restore_decision_liveness_uses_newest_cap_value_uses_oldest() {
714 // [Cap, Cap]: disk holds the NEWEST Cap's our_high ("B"). Liveness
715 // must be checked against "B", not the oldest entry's "A".
716 let oldest = entry_cap("/x", 42, "max", "A");
717 let newest = entry_cap("/x", 42, "A", "B");
718 assert_eq!(
719 restore_decision(&[oldest.clone(), newest.clone()], Some(42), Some("B")),
720 RestoreStep::ThawAndRestoreHigh { to: "max".into() },
721 "must judge liveness against the newest Cap's our_high (matches disk \"B\"), \
722 but restore the oldest Cap's prev_high (\"max\")"
723 );
724 // Sanity: the stale intermediate value "A" is no longer live on disk,
725 // so checking against it (the old bug) would report SkipRemove.
726 assert_eq!(
727 restore_step(&oldest, Some(42), Some("B")),
728 RestoreStep::SkipRemove,
729 "confirms the bug this guards against: judging liveness against the oldest \
730 entry's our_high against the newer on-disk value mismatches"
731 );
732
733 // [Cap, Freeze]: newest entry has no memory.high of its own; must
734 // still fall through to the Cap for both liveness and value.
735 let cap = entry_cap("/x", 42, "max", "1000");
736 let frz = entry_freeze("/x", 42);
737 assert_eq!(
738 restore_decision(&[cap, frz], Some(42), Some("1000")),
739 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
740 );
741
742 // [Cap] alone: baseline single-entry behavior is unchanged.
743 let cap_only = entry_cap("/x", 42, "max", "1000");
744 assert_eq!(
745 restore_decision(&[cap_only], Some(42), Some("1000")),
746 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
747 );
748
749 // Freeze-only chain: nothing to restore.
750 assert_eq!(
751 restore_decision(&[entry_freeze("/x", 42)], Some(42), None),
752 RestoreStep::ThawOnly
753 );
754 }
755
756 fn entry_cap(cg: &str, inode: u64, prev: &str, our: &str) -> JournalEntry {
757 JournalEntry {
758 cgroup: cg.into(),
759 inode,
760 unit: None,
761 action: JournalAction::Cap,
762 prev_high: Some(prev.into()),
763 our_high: Some(our.into()),
764 }
765 }
766
767 fn entry_freeze(cg: &str, inode: u64) -> JournalEntry {
768 JournalEntry {
769 cgroup: cg.into(),
770 inode,
771 unit: None,
772 action: JournalAction::Freeze,
773 prev_high: None,
774 our_high: None,
775 }
776 }
777
778 #[test]
779 fn restore_step_matrix() {
780 let cap = entry_cap("/x", 42, "max", "1000");
781 assert_eq!(
782 restore_step(&cap, Some(42), Some("1000")),
783 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
784 );
785 assert_eq!(
786 restore_step(&cap, Some(43), Some("1000")),
787 RestoreStep::SkipRemove
788 );
789 assert_eq!(
790 restore_step(&cap, Some(42), Some("777")),
791 RestoreStep::SkipRemove
792 );
793 let frz = entry_freeze("/x", 42);
794 assert_eq!(
795 restore_step(&frz, Some(42), None),
796 RestoreStep::ThawOnly,
797 "alive freeze entry: nothing to restore beyond the unconditional thaw"
798 );
799 assert_eq!(
800 restore_step(&frz, None, None),
801 RestoreStep::SkipRemove,
802 "dead cgroup: restore_step only decides memory.high, never freeze; the \
803 unconditional thaw in Effector::thaw_raw already ran before this is consulted, \
804 so a still-frozen dead cgroup is never left behind (carry-forward: Task 5 review)"
805 );
806 }
807
808 /// Poll `cgfs::read_frozen` until it matches `want` or `timeout` elapses.
809 /// Writing `cgroup.freeze` only *requests* a state change; `cgroup.events`'s
810 /// `frozen` field (what `read_frozen` reads) only flips once the kernel has
811 /// actually quiesced every task in the cgroup, which can lag the write by a
812 /// few milliseconds under load, so a bare immediate read is flaky.
813 fn wait_for_frozen(cgroup: &str, want: bool, timeout: Duration) -> Option<bool> {
814 let deadline = std::time::Instant::now() + timeout;
815 loop {
816 let got = cgfs::read_frozen(cgroup);
817 if got == Some(want) || std::time::Instant::now() >= deadline {
818 return got;
819 }
820 std::thread::sleep(Duration::from_millis(20));
821 }
822 }
823
824 fn test_resolution(cgroup: String) -> Resolution {
825 Resolution {
826 cgroup,
827 unit: None,
828 verdict: Verdict::Freeze,
829 coverage: Coverage::Full,
830 mechanism: Mechanism::Raw,
831 }
832 }
833
834 /// Integration smoke test: freeze a real `sleep` via its (raw, rlm-created)
835 /// cgroup through the journal-backed Effector, confirm it's paused via
836 /// `cgroup.freeze` and journaled, then thaw and confirm the journal entry
837 /// is gone. Only works under cgroup v2 delegation, so it's `#[ignore]`d.
838 #[test]
839 #[ignore = "requires cgroup v2 delegation; run manually"]
840 fn freeze_thaw_real_process_raw_cgroup() {
841 use common::Limit;
842 use std::process::Command;
843
844 let manager = CgroupManager::new().expect("create CgroupManager");
845 let journal_dir = tempfile::tempdir().unwrap();
846 let journal =
847 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
848 let effector = Effector::new(&manager, &journal, None);
849
850 let abs_path = manager
851 .prepare_cgroup("test-freeze-thaw", &Limit::default())
852 .expect("create test cgroup")
853 .path;
854 let cgroup = format!(
855 "/{}",
856 abs_path
857 .strip_prefix("/sys/fs/cgroup")
858 .expect("cgroup under /sys/fs/cgroup")
859 .display()
860 );
861
862 let mut child = Command::new("sleep")
863 .arg("30")
864 .spawn()
865 .expect("spawn sleep");
866 let pid = child.id();
867 manager
868 .add_to_cgroup(&abs_path, pid)
869 .expect("add sleep to test cgroup");
870
871 let res = test_resolution(cgroup.clone());
872
873 effector
874 .apply(&Action::Freeze {
875 res: res.clone(),
876 name: "sleep".into(),
877 })
878 .expect("freeze");
879
880 assert_eq!(
881 wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
882 Some(true),
883 "cgroup should be frozen"
884 );
885 assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
886
887 effector
888 .apply(&Action::Thaw { res: res.clone() })
889 .expect("thaw");
890
891 assert_eq!(
892 wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
893 Some(false),
894 "cgroup should be thawed"
895 );
896 assert!(
897 journal.entries().is_empty(),
898 "journal entry should be removed after thaw"
899 );
900
901 let _ = child.kill();
902 let _ = child.wait();
903 let _ = manager.cleanup_cgroup("test-freeze-thaw");
904 }
905
906 /// Act-in-place integration test: freeze/thaw a real transient systemd
907 /// `--user --scope` unit through the Unit mechanism (D-Bus, with raw
908 /// fallback), confirming the journal-backed round trip end to end.
909 /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
910 #[test]
911 #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
912 fn freeze_thaw_real_transient_scope() {
913 use std::process::Command;
914
915 let uid_out = Command::new("id").arg("-u").output().expect("id -u");
916 let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
917 .trim()
918 .parse()
919 .expect("parse uid");
920
921 let unit_base = format!("rlm-e2e-{}", std::process::id());
922 let unit = format!("{unit_base}.scope");
923 let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
924
925 let mut child = Command::new("systemd-run")
926 .args([
927 "--user",
928 "--scope",
929 "--slice=app.slice",
930 &format!("--unit={unit_base}"),
931 "--",
932 "sleep",
933 "30",
934 ])
935 .spawn()
936 .expect("spawn systemd-run --scope");
937
938 // Give systemd a moment to register the transient scope's cgroup.
939 std::thread::sleep(Duration::from_millis(300));
940
941 let manager = CgroupManager::new().expect("create CgroupManager");
942 let journal_dir = tempfile::tempdir().unwrap();
943 let journal =
944 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
945 let systemd = SystemdUser::connect();
946 let effector = Effector::new(&manager, &journal, systemd.as_ref());
947
948 let res = Resolution {
949 cgroup: cgroup.clone(),
950 unit: Some(unit),
951 verdict: Verdict::Freeze,
952 coverage: Coverage::Full,
953 mechanism: Mechanism::Unit,
954 };
955
956 effector
957 .apply(&Action::Freeze {
958 res: res.clone(),
959 name: "sleep".into(),
960 })
961 .expect("freeze");
962 assert_eq!(
963 wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
964 Some(true),
965 "scope cgroup should be frozen"
966 );
967 assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
968
969 effector.apply(&Action::Thaw { res }).expect("thaw");
970 assert_eq!(
971 wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
972 Some(false),
973 "scope cgroup should be thawed"
974 );
975 assert!(
976 journal.entries().is_empty(),
977 "journal entry should be removed after thaw"
978 );
979
980 let _ = child.kill();
981 let _ = child.wait();
982 let _ = Command::new("systemctl")
983 .args(["--user", "stop", &format!("{unit_base}.scope")])
984 .status();
985 }
986
987 /// Regression test for Task 6 review Critical #1: cap a real cgroup and
988 /// assert `cgfs::read_high` matches the journaled `our_high` exactly,
989 /// i.e. the value we wrote never got silently page-truncated out from
990 /// under the journal. The target process holds enough random anon
991 /// memory (well above the 256 MiB `MIN_CAP_BYTES` floor, a round power
992 /// of two that would pass even without the fix and prove nothing) that
993 /// the sized cap is very unlikely to already sit on a page boundary. Requires cgroup v2
994 /// delegation, so it's `#[ignore]`d.
995 #[test]
996 #[ignore = "requires cgroup v2 delegation; run manually"]
997 fn cap_page_aligns_and_journal_matches_on_disk_value() {
998 use common::Limit;
999 use std::process::Command;
1000
1001 let manager = CgroupManager::new().expect("create CgroupManager");
1002 let journal_dir = tempfile::tempdir().unwrap();
1003 let journal =
1004 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1005 let effector = Effector::new(&manager, &journal, None);
1006
1007 let abs_path = manager
1008 .prepare_cgroup("test-cap-align", &Limit::default())
1009 .expect("create test cgroup")
1010 .path;
1011 let cgroup = format!(
1012 "/{}",
1013 abs_path
1014 .strip_prefix("/sys/fs/cgroup")
1015 .expect("cgroup under /sys/fs/cgroup")
1016 .display()
1017 );
1018
1019 // Hold ~400MB of anon memory (above the 256 MiB floor) whose sized
1020 // cap is very unlikely to land on a page boundary, then sleep.
1021 let mut child = Command::new("bash")
1022 .arg("-c")
1023 .arg("a=$(head -c 300000000 /dev/urandom | base64 -w0); sleep 30")
1024 .spawn()
1025 .expect("spawn memory-holding process");
1026 let pid = child.id();
1027 manager
1028 .add_to_cgroup(&abs_path, pid)
1029 .expect("add process to test cgroup");
1030 // Wait until the shell has actually built up the allocation, or the
1031 // cap would sit on the 256 MiB floor and prove nothing.
1032 let want = 300 * 1024 * 1024;
1033 let deadline = std::time::Instant::now() + Duration::from_secs(10);
1034 loop {
1035 let cur = cgfs::current_bytes(&cgroup).unwrap_or(0);
1036 if cur >= want {
1037 break;
1038 }
1039 if std::time::Instant::now() >= deadline {
1040 let _ = child.kill();
1041 let _ = child.wait();
1042 let _ = manager.cleanup_cgroup("test-cap-align");
1043 panic!("memory.current reached only {cur} bytes, need {want}, after 10s");
1044 }
1045 std::thread::sleep(Duration::from_millis(100));
1046 }
1047
1048 let res = test_resolution(cgroup.clone());
1049 effector
1050 .apply(&Action::Cap {
1051 res: res.clone(),
1052 name: "mem-hog".into(),
1053 })
1054 .expect("cap");
1055
1056 let entries = journal.entries();
1057 assert_eq!(entries.len(), 1, "cap should be journaled");
1058 let journaled = entries[0].our_high.clone().expect("our_high recorded");
1059
1060 let on_disk = cgfs::read_high(&cgroup).expect("read memory.high");
1061 assert_eq!(
1062 on_disk, journaled,
1063 "on-disk memory.high must match the journaled our_high exactly"
1064 );
1065
1066 let bytes: u64 = journaled.parse().expect("our_high is a plain decimal");
1067 assert_eq!(bytes % page_size(), 0, "written value must be page-aligned");
1068
1069 effector.apply(&Action::LiftCap { res }).expect("lift cap");
1070 assert!(
1071 journal.entries().is_empty(),
1072 "journal entry removed after lift"
1073 );
1074
1075 let _ = child.kill();
1076 let _ = child.wait();
1077 let _ = manager.cleanup_cgroup("test-cap-align");
1078 }
1079
1080 /// Cap and lift go through raw `memory.high` writes only, so the unit's
1081 /// own `MemoryHigh` property (as systemd reports it) must be the same
1082 /// before the cap and after the lift, and the raw `memory.high` must be
1083 /// back at its pre-cap value. A runtime `SetUnitProperties` would leave
1084 /// a `/run` drop-in behind that changes the reported property.
1085 /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
1086 #[test]
1087 #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
1088 fn lift_cap_leaves_unit_memory_high_property_untouched() {
1089 use std::process::Command;
1090
1091 let unit_memory_high = |unit: &str| -> String {
1092 let out = Command::new("systemctl")
1093 .args(["--user", "show", "-p", "MemoryHigh", "--value", unit])
1094 .output()
1095 .expect("systemctl --user show");
1096 String::from_utf8_lossy(&out.stdout).trim().to_string()
1097 };
1098
1099 let uid_out = Command::new("id").arg("-u").output().expect("id -u");
1100 let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
1101 .trim()
1102 .parse()
1103 .expect("parse uid");
1104
1105 let unit_base = format!("rlm-e2e-cap-{}", std::process::id());
1106 let unit = format!("{unit_base}.scope");
1107 let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
1108
1109 let mut child = Command::new("systemd-run")
1110 .args([
1111 "--user",
1112 "--scope",
1113 "--slice=app.slice",
1114 &format!("--unit={unit_base}"),
1115 // A genuine prior MemoryHigh, distinct from both "max" and
1116 // whatever Cap computes, so the restored value is
1117 // unambiguous.
1118 "--property=MemoryHigh=1500M",
1119 "--",
1120 "sleep",
1121 "30",
1122 ])
1123 .spawn()
1124 .expect("spawn systemd-run --scope with MemoryHigh set");
1125
1126 std::thread::sleep(Duration::from_millis(300));
1127
1128 let manager = CgroupManager::new().expect("create CgroupManager");
1129 let journal_dir = tempfile::tempdir().unwrap();
1130 let journal =
1131 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1132 let systemd = SystemdUser::connect();
1133 let effector = Effector::new(&manager, &journal, systemd.as_ref());
1134
1135 let original_high =
1136 cgfs::read_high(&cgroup).expect("systemd-run set an initial memory.high");
1137 assert_ne!(
1138 original_high, "max",
1139 "test needs a concrete prior MemoryHigh to distinguish from systemd's clear-to-max"
1140 );
1141 let property_before = unit_memory_high(&unit);
1142
1143 let res = Resolution {
1144 cgroup: cgroup.clone(),
1145 unit: Some(unit.clone()),
1146 verdict: Verdict::CapOnly,
1147 coverage: Coverage::Full,
1148 mechanism: Mechanism::Unit,
1149 };
1150
1151 effector
1152 .apply(&Action::Cap {
1153 res: res.clone(),
1154 name: "sleep".into(),
1155 })
1156 .expect("cap");
1157 assert_ne!(
1158 cgfs::read_high(&cgroup),
1159 Some(original_high.clone()),
1160 "cap should have changed memory.high"
1161 );
1162
1163 effector.apply(&Action::LiftCap { res }).expect("lift cap");
1164 assert_eq!(
1165 unit_memory_high(&unit),
1166 property_before,
1167 "cap/lift must not change the unit's MemoryHigh property"
1168 );
1169 assert_eq!(
1170 cgfs::read_high(&cgroup),
1171 Some(original_high),
1172 "lift must restore the raw pre-cap memory.high"
1173 );
1174 assert!(
1175 journal.entries().is_empty(),
1176 "journal entry removed after lift"
1177 );
1178
1179 let _ = child.kill();
1180 let _ = child.wait();
1181 let _ = Command::new("systemctl")
1182 .args(["--user", "stop", &unit])
1183 .status();
1184 }
1185
1186 /// Regression test for Promoted Minor B: a leaked `[Cap, Freeze]` chain
1187 /// (Cap journaled first, then the same cgroup frozen later without the
1188 /// Cap entry ever being cleared) must still restore the Cap's
1189 /// `prev_high` on thaw. Before the fix, liveness was judged against
1190 /// `entries.last()` (the Freeze entry, which has no `memory.high` of
1191 /// its own), so `restore_step` reported "nothing to restore" and
1192 /// `memory.high` stayed pinned at the guard's Cap value forever once the
1193 /// whole chain was cleared. Requires cgroup v2 delegation, so it's
1194 /// `#[ignore]`d.
1195 #[test]
1196 #[ignore = "requires cgroup v2 delegation; run manually"]
1197 fn chain_restores_cap_value_when_newest_entry_is_freeze() {
1198 use common::Limit;
1199 use std::process::Command;
1200
1201 let manager = CgroupManager::new().expect("create CgroupManager");
1202 let journal_dir = tempfile::tempdir().unwrap();
1203 let journal =
1204 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1205 let effector = Effector::new(&manager, &journal, None);
1206
1207 let abs_path = manager
1208 .prepare_cgroup("test-chain-restore", &Limit::default())
1209 .expect("create test cgroup")
1210 .path;
1211 let cgroup = format!(
1212 "/{}",
1213 abs_path
1214 .strip_prefix("/sys/fs/cgroup")
1215 .expect("cgroup under /sys/fs/cgroup")
1216 .display()
1217 );
1218
1219 let mut child = Command::new("sleep")
1220 .arg("30")
1221 .spawn()
1222 .expect("spawn sleep");
1223 let pid = child.id();
1224 manager
1225 .add_to_cgroup(&abs_path, pid)
1226 .expect("add sleep to test cgroup");
1227
1228 let original_high = cgfs::read_high(&cgroup).expect("read initial memory.high");
1229 let res = test_resolution(cgroup.clone());
1230
1231 // Cap first (oldest entry)...
1232 effector
1233 .apply(&Action::Cap {
1234 res: res.clone(),
1235 name: "sleep".into(),
1236 })
1237 .expect("cap");
1238 assert_ne!(
1239 cgfs::read_high(&cgroup),
1240 Some(original_high.clone()),
1241 "cap should have changed memory.high"
1242 );
1243
1244 // ...then freeze the same cgroup without ever clearing the Cap
1245 // entry. This is the leaked-chain scenario: two coexisting entries,
1246 // Freeze newest.
1247 effector
1248 .apply(&Action::Freeze {
1249 res: res.clone(),
1250 name: "sleep".into(),
1251 })
1252 .expect("freeze");
1253
1254 let entries = journal.entries();
1255 assert_eq!(entries.len(), 2, "both Cap and Freeze entries coexist");
1256 assert_eq!(entries[0].action, JournalAction::Cap, "Cap is oldest");
1257 assert_eq!(entries[1].action, JournalAction::Freeze, "Freeze is newest");
1258
1259 // Thaw (the action that would naturally follow a Freeze) must still
1260 // restore the Cap's prev_high, not skip restoration just because the
1261 // newest entry is a Freeze with no memory.high of its own.
1262 effector.apply(&Action::Thaw { res }).expect("thaw");
1263 assert_eq!(
1264 cgfs::read_high(&cgroup),
1265 Some(original_high),
1266 "thaw must restore the chain's Cap value even though Freeze is the newest entry"
1267 );
1268 assert!(
1269 journal.entries().is_empty(),
1270 "both chain entries removed after thaw"
1271 );
1272
1273 let _ = child.kill();
1274 let _ = child.wait();
1275 let _ = manager.cleanup_cgroup("test-chain-restore");
1276 }
1277}