rlmctl_core/guard/effector.rs
1//! Executes [`Action`]s against real cgroups, acting in place on the cgroup a
2//! process already lives in (a systemd unit's scope/service, or an existing
3//! rlm rule cgroup) rather than moving it into an ephemeral `guard-<pid>`
4//! cgroup. Every action is best-effort and logged; a failure must never
5//! panic or otherwise crash the daemon loop. `apply` may return `Err` so the
6//! caller can log it, but a missing `notify-send` (or any other notification
7//! failure) is never treated as an error.
8//!
9//! # Write-ahead journal
10//! Freeze/Cap always `journal.append` (which fsyncs) *before* touching the
11//! cgroup, so a crash between the two still leaves a durable record that
12//! startup recovery (`sweep_leftovers`) can replay. `Journal` is internally
13//! mutex-serialized (see `journal.rs`), and the daemon calls into this
14//! `Effector` from a single thread (the tick loop in `rlm-guard`'s `main`),
15//! so `Effector` just holds a `&Journal` and relies on the daemon's
16//! single-threaded call discipline plus the journal's own internal locking
17//! for safety — it adds no locking of its own.
18//!
19//! # Mechanism: systemd unit vs. raw cgroupfs
20//! When a target resolved to a systemd unit (`Mechanism::Unit`), we prefer
21//! the D-Bus call (`FreezeUnit`/`ThawUnit`) with a hard
22//! 2s timeout (`systemd::SystemdUser`'s own enforced deadline); on `Err` (bus
23//! unavailable, call failed, or timed out) we fall back to the raw cgroupfs
24//! primitives in `cgfs`. Raw-mechanism targets (rlm's own rule cgroups) skip
25//! the D-Bus attempt entirely. Thawing is mechanism-independent: we always
26//! perform the raw `cgroup.freeze` write first, unconditionally, and only
27//! then attempt `ThawUnit` best-effort so systemd's view matches. This
28//! guarantees a cgroup is never left frozen because of a stale unit or a
29//! slow bus, and tolerates a missing cgroup (the raw write simply errors and
30//! we move on). On shutdown and startup replay every raw thaw and restore
31//! finishes before the first D-Bus call. Caps are the exception: they never
32//! go through systemd (see below).
33//!
34//! # Restoring `memory.high`
35//! The kernel truncates `memory.high` writes to page multiples, so the byte
36//! count we cap to is page-aligned *before* we journal/write it
37//! ([`page_align_down`]); as a second line of defense, [`Effector::cap`]
38//! reads `memory.high` back after writing and self-corrects the journal if
39//! reality still differs.
40//! Caps and restores are raw `memory.high` writes only, never systemd's
41//! `SetUnitProperties(MemoryHigh)`: a runtime property leaves a `/run`
42//! drop-in that outlives the cap and masks the unit's own configured
43//! `MemoryHigh` until reboot, and restoring through systemd wrote `infinity`
44//! over it. If systemd later re-applies the unit's value, that only lifts our
45//! cap early, and the journal's `our_high` check then skips the restore (the
46//! safe direction).
47//!
48//! If more than one journal entry ever coexists for the same
49//! cgroup (a leak from an incomplete prior removal), every restore path
50//! treats them as one chain: liveness is judged against the chain's `Cap`
51//! entry specifically (only a `Cap` has a `memory.high` to restore — a
52//! `Freeze` entry does not, so judging against `entries.last()` when it
53//! happens to be a newer `Freeze` would wrongly look like "nothing to
54//! restore" and strand `memory.high` at the guard's value forever), and the
55//! value restored is always the *oldest* `Cap` entry's `prev_high` — the
56//! true pre-intervention value, not an intermediate entry's `prev_high`
57//! (which is just our own previous `our_high`). See [`restore_target`].
58//! Liveness, however, is judged against the *newest* `Cap` entry, not the
59//! oldest: entries are strictly appended, so a later Cap's write always
60//! supersedes an earlier one's on disk, and `should_restore`'s string
61//! comparison must match what's actually there. See [`restore_decision`].
62
63use super::cgfs;
64use super::journal::{should_restore, Journal, JournalAction, JournalEntry};
65use super::resolve::{Mechanism, Resolution};
66use super::sampler::parse_meminfo;
67use super::systemd::SystemdUser;
68use super::types::Action;
69use crate::CgroupManager;
70use common::Result;
71use std::process::Command;
72use std::time::Duration;
73
74/// Floor for any soft cap. A cap below this is effectively a freeze for a
75/// desktop app, so small cgroups are never squeezed further than this.
76pub const MIN_CAP_BYTES: u64 = 256 * 1024 * 1024;
77
78/// Hard deadline for every D-Bus call on the freeze-guard's storm path. On
79/// `Err` (including a timeout) callers fall back to raw cgroupfs writes.
80const DBUS_TIMEOUT: Duration = Duration::from_secs(2);
81
82/// Applies guard [`Action`]s in place on the resolved target cgroup, via the
83/// systemd-unit D-Bus path (with raw-cgroup fallback) and a write-ahead
84/// [`Journal`] for crash-safe restore.
85pub struct Effector<'a> {
86 /// Kept only for the legacy `guard-<pid>` sweep during the upgrade path
87 /// (see [`Effector::sweep_leftovers`]); acting in place no longer uses it.
88 manager: &'a CgroupManager,
89 journal: &'a Journal,
90 systemd: Option<&'a SystemdUser>,
91}
92
93impl<'a> Effector<'a> {
94 pub fn new(
95 manager: &'a CgroupManager,
96 journal: &'a Journal,
97 systemd: Option<&'a SystemdUser>,
98 ) -> Self {
99 Self {
100 manager,
101 journal,
102 systemd,
103 }
104 }
105
106 /// Apply a single action. Best-effort: returns `Err` only so the caller can
107 /// log it (a [`Action::Notify`] always returns `Ok`).
108 pub fn apply(&self, action: &Action) -> Result<()> {
109 match action {
110 Action::Freeze { res, name } => self.freeze(res, name),
111 Action::Thaw { res } => self.thaw(res),
112 Action::Cap { res, name } => self.cap(res, name),
113 Action::LiftCap { res } => self.lift_cap(res),
114 Action::Notify { message } => {
115 notify(message);
116 // Notification is always best-effort and never fails the caller.
117 Ok(())
118 }
119 }
120 }
121
122 fn freeze(&self, res: &Resolution, name: &str) -> Result<()> {
123 // The inode is the guard that lets a later restore tell "this is
124 // still the same cgroup" from "this cgroup was torn down and
125 // recreated" (`should_restore`). `unwrap_or(0)` used to substitute a
126 // poison sentinel here: 0 is never a real inode, so the guard could
127 // never match again and the entry became permanently unrestorable —
128 // yet it was still journaled-and-acted-on, then later cleared and
129 // logged as if it were an intentional skip (Promoted Minor A). Fail
130 // closed instead: refuse to freeze at all rather than act with a
131 // record we can never safely restore from.
132 let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
133 tracing::warn!(
134 cgroup = %res.cgroup, name,
135 "cannot read cgroup inode; refusing to freeze (would be unrestorable)"
136 );
137 return Err(common::Error::Cgroup(format!(
138 "cannot read inode for {}; refusing to freeze",
139 res.cgroup
140 )));
141 };
142 let entry = JournalEntry {
143 cgroup: res.cgroup.clone(),
144 inode,
145 unit: res.unit.clone(),
146 action: JournalAction::Freeze,
147 prev_high: None,
148 our_high: None,
149 };
150 // Write-ahead: the entry must be durable (journal.append fsyncs)
151 // before we ever touch the cgroup, so a crash between the two still
152 // leaves a record startup recovery can act on.
153 self.journal.append(&entry)?;
154
155 tracing::info!(cgroup = %res.cgroup, name, "freezing cgroup");
156 if res.mechanism == Mechanism::Unit {
157 if let (Some(unit), Some(systemd)) = (&res.unit, self.systemd) {
158 match systemd.freeze_unit(unit, DBUS_TIMEOUT) {
159 Ok(()) => return Ok(()),
160 Err(e) => tracing::warn!(
161 cgroup = %res.cgroup, unit, error = %e,
162 "FreezeUnit failed; falling back to raw cgroup.freeze"
163 ),
164 }
165 }
166 }
167 cgfs::write_freeze(&res.cgroup, true)
168 }
169
170 fn thaw(&self, res: &Resolution) -> Result<()> {
171 tracing::info!(cgroup = %res.cgroup, "thawing cgroup");
172 // Gather any coexisting entries for this cgroup first: a `Thaw`
173 // action reverses a Frozen intervention, but if a Cap entry ever
174 // leaked alongside it (Important #3, Task 6 review) we must still
175 // restore its memory.high before wiping the journal record below,
176 // not just drop it silently.
177 let entries = self.entries_for(&res.cgroup);
178 let result = self.thaw_raw(&res.cgroup, res.unit.as_deref());
179 if let Err(e) = &result {
180 tracing::debug!(cgroup = %res.cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
181 }
182 self.restore_high_if_any(&res.cgroup, &entries);
183 self.journal.remove(&res.cgroup)?;
184 result
185 }
186
187 fn cap(&self, res: &Resolution, name: &str) -> Result<()> {
188 // See `freeze`'s matching comment (Promoted Minor A): a `0` inode
189 // sentinel here would make this Cap permanently unrestorable while
190 // looking like a real guard, so fail closed instead of
191 // journal-and-act with an unrestorable record.
192 let Some(inode) = cgfs::dir_inode(&res.cgroup) else {
193 tracing::warn!(
194 cgroup = %res.cgroup, name,
195 "cannot read cgroup inode; refusing to cap (would be unrestorable)"
196 );
197 return Err(common::Error::Cgroup(format!(
198 "cannot read inode for {}; refusing to cap",
199 res.cgroup
200 )));
201 };
202 let prev_high = cgfs::read_high(&res.cgroup);
203 // Unknown swap total counts as 0: assuming anon is pinned only makes
204 // the cap gentler.
205 let swap_total_kb = std::fs::read_to_string("/proc/meminfo")
206 .ok()
207 .and_then(|m| parse_meminfo(&m))
208 .map_or(0, |m| m.swap_total_kb);
209 let anon_ok = cgfs::anon_reclaimable(&res.cgroup, swap_total_kb);
210 let Some(target) = cap_target(
211 cgfs::current_bytes(&res.cgroup),
212 cgfs::file_bytes(&res.cgroup),
213 anon_ok,
214 ) else {
215 tracing::warn!(
216 cgroup = %res.cgroup, name,
217 "cannot read memory.current; refusing to cap"
218 );
219 return Err(common::Error::Cgroup(
220 "cannot read memory.current; refusing to cap".into(),
221 ));
222 };
223 // The kernel truncates `memory.high` writes to page multiples, so we
224 // must journal/write the value it will actually store, not the raw
225 // target, or `should_restore`'s string-equality check can never
226 // pass again and the cap becomes permanent.
227 let our_bytes = page_align_down(target, page_size());
228 // Plain decimal bytes, no separators/whitespace: this must be
229 // exactly what a later `cgfs::read_high` (which only trims
230 // whitespace off the raw file contents) returns. See
231 // `our_high_string_is_plain_decimal_no_separators` below, and
232 // `reconcile_our_high` for the belt-and-braces check.
233 if !cap_tightens(prev_high.as_deref(), our_bytes) {
234 tracing::info!(
235 cgroup = %res.cgroup, name, prev_high = ?prev_high, our_bytes,
236 "existing memory.high already at or below the cap; not capping"
237 );
238 return Err(common::Error::Cgroup(
239 "existing memory.high already at or below the cap; refusing to cap".into(),
240 ));
241 }
242 let our_high = our_bytes.to_string();
243
244 let entry = JournalEntry {
245 cgroup: res.cgroup.clone(),
246 inode,
247 unit: res.unit.clone(),
248 action: JournalAction::Cap,
249 prev_high,
250 our_high: Some(our_high.clone()),
251 };
252 self.journal.append(&entry)?;
253
254 tracing::info!(
255 cgroup = %res.cgroup, name, our_high = %our_high, anon_reclaimable = anon_ok,
256 "soft-capping cgroup"
257 );
258 let result = cgfs::write_high(&res.cgroup, &our_high);
259
260 if result.is_ok() {
261 self.reconcile_our_high(&entry);
262 }
263 result
264 }
265
266 /// After a successful `Cap` write, read `memory.high` back; if what's
267 /// actually on disk differs from what we journaled (page truncation we
268 /// didn't fully pre-empt), correct the journal to match reality. Otherwise
269 /// `should_restore`'s string-equality check can never pass again and
270 /// the cap becomes permanent (Task 6 review, Critical #1). Rebuilds
271 /// only this cgroup's entries, preserving any others that might coexist
272 /// (Important #3), with `written`'s `our_high` corrected to the
273 /// read-back value. Uses `Journal::replace` (a single atomic rewrite),
274 /// not a separate `remove` then `append`: the latter has a window where
275 /// the cgroup has no journal record at all, so a crash/SIGKILL/append
276 /// failure right there would leave a permanent, unrecoverable cap —
277 /// exactly the invariant the journal exists to prevent (Task 6 review,
278 /// fix round 2).
279 fn reconcile_our_high(&self, written: &JournalEntry) {
280 let cgroup: &str = &written.cgroup;
281 let Some(actual) = cgfs::read_high(cgroup) else {
282 return;
283 };
284 if written.our_high.as_deref() == Some(actual.as_str()) {
285 return;
286 }
287 tracing::warn!(
288 cgroup = %cgroup, journaled = ?written.our_high, actual = %actual,
289 "memory.high on disk differs from what we journaled; correcting journal entry"
290 );
291 let mut entries = self.entries_for(cgroup);
292 let Some(pos) = entries.iter().rposition(|e| e == written) else {
293 // Already removed/replaced by something else (e.g. a concurrent
294 // Thaw/LiftCap) — nothing left to correct.
295 return;
296 };
297 entries[pos].our_high = Some(actual);
298 if let Err(e) = self.journal.replace(cgroup, &entries) {
299 tracing::warn!(cgroup = %cgroup, error = %e, "failed to correct journal entry (atomic replace)");
300 }
301 }
302
303 fn lift_cap(&self, res: &Resolution) -> Result<()> {
304 tracing::info!(cgroup = %res.cgroup, "lifting cap");
305 let entries = self.entries_for(&res.cgroup);
306 // Mechanism-independent thaw always runs, regardless of whether any
307 // journal record exists — a dead-cgroup prune (carry-forward
308 // finding, Task 5 review) must never leave a target frozen just
309 // because we lost the journal entry.
310 let _ = self.thaw_raw(&res.cgroup, res.unit.as_deref());
311 self.restore_high_if_any(&res.cgroup, &entries);
312 self.journal.remove(&res.cgroup)
313 }
314
315 /// All journal entries for one cgroup, in the order `Journal::entries()`
316 /// returns them — oldest-first, since entries are strictly appended.
317 fn entries_for(&self, cgroup: &str) -> Vec<JournalEntry> {
318 self.journal
319 .entries()
320 .into_iter()
321 .filter(|e| e.cgroup == cgroup)
322 .collect()
323 }
324
325 /// Startup recovery: legacy `guard-<pid>` sweep (kept for one release as
326 /// an upgrade path from pre-act-in-place rlm), then replay every live
327 /// journal entry.
328 pub fn sweep_leftovers(&self) -> Result<()> {
329 if let Err(e) = self.manager.sweep_guard_leftovers() {
330 tracing::warn!(error = %e, "legacy guard-<pid> sweep failed (non-fatal)");
331 }
332 self.replay_and_clear()
333 }
334
335 /// Graceful shutdown: undo every live journal entry, then clear the
336 /// journal. Same replay as `sweep_leftovers`, minus the legacy sweep.
337 pub fn undo_all(&self) -> Result<()> {
338 self.replay_and_clear()
339 }
340
341 fn replay_and_clear(&self) -> Result<()> {
342 // Group by cgroup first (entries for the same cgroup aren't
343 // necessarily contiguous in the file) so each cgroup's chain is
344 // replayed as one unit — see `restore_high_if_any` — rather than
345 // entry-by-entry, which would mis-restore whenever more than one
346 // entry coexists for a cgroup (Important #3, Task 6 review).
347 let mut by_cgroup: std::collections::HashMap<String, Vec<JournalEntry>> =
348 std::collections::HashMap::new();
349 for e in self.journal.entries() {
350 by_cgroup.entry(e.cgroup.clone()).or_default().push(e);
351 }
352 // Pass 1: every raw thaw and memory.high restore, then clear the
353 // journal. No D-Bus call happens before this is done, so a slow
354 // session bus cannot push the undo past systemd's stop timeout.
355 for (cgroup, entries) in &by_cgroup {
356 if let Err(e) = cgfs::write_freeze(cgroup, false) {
357 tracing::debug!(cgroup, error = %e, "raw thaw failed (cgroup may already be gone)");
358 }
359 self.restore_high_if_any(cgroup, entries);
360 }
361 let cleared = self.journal.clear();
362 // Pass 2, best effort: tell systemd so its view of the units
363 // matches the kernel's. The processes already run again.
364 if let Some(systemd) = self.systemd {
365 for (cgroup, entries) in &by_cgroup {
366 if let Some(unit) = entries.last().and_then(|e| e.unit.as_deref()) {
367 if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
368 tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
369 }
370 }
371 }
372 }
373 cleared
374 }
375
376 /// Restore `memory.high` for one cgroup's journal `entries` (oldest-first),
377 /// if the chain is still live — called *after* the caller has already
378 /// performed the mechanism-independent thaw (see module docs: no path
379 /// here ever means "leave frozen"). All the decision logic is delegated
380 /// to the pure [`restore_decision`]: liveness is judged against the
381 /// *newest* `Cap` entry (whose write is what's actually on disk right
382 /// now), while the value restored is [`restore_target`]'s *oldest*-entry
383 /// `prev_high` (the true pre-intervention value). Judging liveness
384 /// against the oldest Cap instead (NEW-1 regression) leaves a stacked
385 /// `[Cap, Cap]` chain's on-disk value permanently un-restorable, since
386 /// the oldest entry's `our_high` never matches what a later Cap actually
387 /// wrote. Judging liveness against `entries.last()` has the same failure
388 /// for a `[Cap, Freeze]` chain (Promoted Minor B): the newest entry is
389 /// the `Freeze`, which has no `memory.high` of its own. The restore is a
390 /// raw `memory.high` write only; systemd's `MemoryHigh` property is never
391 /// touched (see module docs).
392 fn restore_high_if_any(&self, cgroup: &str, entries: &[JournalEntry]) {
393 let inode = cgfs::dir_inode(cgroup);
394 let high = cgfs::read_high(cgroup);
395 match restore_decision(entries, inode, high.as_deref()) {
396 RestoreStep::ThawAndRestoreHigh { to } => {
397 if let Err(err) = cgfs::write_high(cgroup, &to) {
398 tracing::warn!(cgroup, error = %err, "failed to restore memory.high");
399 }
400 }
401 // No Cap entry anywhere in the chain (a Freeze-only chain,
402 // possibly with leaked duplicates): there is no memory.high to
403 // restore. The unconditional thaw already ran in the caller, so
404 // there's nothing left to do here.
405 RestoreStep::ThawOnly => {}
406 RestoreStep::SkipRemove => {
407 tracing::warn!(
408 cgroup,
409 "not restoring memory.high (cgroup recreated or value changed since our write)"
410 );
411 }
412 }
413 }
414
415 /// Mechanism-independent thaw: an *unconditional* raw `cgroup.freeze`
416 /// write first, so a cgroup is never left frozen and a slow bus cannot
417 /// delay it, then a best-effort `ThawUnit` (if we have a unit and a bus)
418 /// so systemd's view of the unit matches. Tolerates a missing cgroup:
419 /// the raw write then simply returns `Err`, which every caller here
420 /// treats as non-fatal.
421 fn thaw_raw(&self, cgroup: &str, unit: Option<&str>) -> Result<()> {
422 let result = cgfs::write_freeze(cgroup, false);
423 if let (Some(unit), Some(systemd)) = (unit, self.systemd) {
424 if let Err(e) = systemd.thaw_unit(unit, DBUS_TIMEOUT) {
425 tracing::debug!(cgroup, unit, error = %e, "ThawUnit failed after raw thaw");
426 }
427 }
428 result
429 }
430}
431
432/// What to do with one journal entry's `memory.high` at restore time. Never
433/// speaks to freeze/thaw — that is unconditional and already handled by the
434/// caller (`Effector::thaw_raw`) before `restore_step` is even consulted, so
435/// no variant here can ever mean "leave it frozen".
436#[derive(Debug, PartialEq, Eq)]
437pub enum RestoreStep {
438 /// A `Freeze` entry whose guard still holds: nothing to restore, thaw
439 /// (already done by the caller) was the whole job.
440 ThawOnly,
441 /// A `Cap` entry whose guard still holds: restore `memory.high` to `to`
442 /// (the entry's `prev_high`, "max" if it was never recorded).
443 ThawAndRestoreHigh { to: String },
444 /// The guard no longer holds (cgroup recreated, or someone else changed
445 /// `memory.high` since our write): don't touch `memory.high`, just let
446 /// the caller remove the journal entry.
447 SkipRemove,
448}
449
450/// Pure: what to do for one journal entry at restore time, delegating the
451/// safety check to [`should_restore`].
452pub fn restore_step(e: &JournalEntry, inode: Option<u64>, high: Option<&str>) -> RestoreStep {
453 if !should_restore(e, inode, high) {
454 return RestoreStep::SkipRemove;
455 }
456 match e.action {
457 JournalAction::Freeze => RestoreStep::ThawOnly,
458 JournalAction::Cap => RestoreStep::ThawAndRestoreHigh {
459 to: e.prev_high.clone().unwrap_or_else(|| "max".into()),
460 },
461 }
462}
463
464/// Size a soft cap that slows an app down without stalling it.
465///
466/// memory.high applies to everything charged to the cgroup, page cache
467/// included, so the cap is sized from memory.current: it never asks the
468/// kernel to reclaim more than 10% of it. When anon memory cannot go to swap,
469/// only file pages can be reclaimed, so the cap never asks for more than 80%
470/// of them. Returns `None` when memory.current is unreadable: no cap is safer
471/// than a guessed one.
472pub fn cap_target(current: Option<u64>, file: Option<u64>, anon_reclaimable: bool) -> Option<u64> {
473 let current = current?;
474 let mut target = current / 10 * 9;
475 if !anon_reclaimable {
476 let file_floor = current.saturating_sub(file.unwrap_or(0).saturating_mul(8) / 10);
477 target = target.max(file_floor);
478 }
479 Some(target.max(MIN_CAP_BYTES))
480}
481
482/// Pure: whether writing `target` would tighten the existing `memory.high`.
483/// A numeric `prev_high` at or below `target` means the cap would loosen (or
484/// not change) the limit already in place, so the caller must not write it.
485/// `"max"`, or anything else that is not a byte count, is no limit at all.
486pub fn cap_tightens(prev_high: Option<&str>, target: u64) -> bool {
487 match prev_high.and_then(|s| s.trim().parse::<u64>().ok()) {
488 Some(prev) => prev > target,
489 None => true,
490 }
491}
492
493/// Pure: the full restore decision for one cgroup's journal `entries`
494/// (oldest-first), given the cgroup's current inode and on-disk
495/// `memory.high` (both already read by the caller — no IO here). Answers two
496/// orthogonal questions with two different entries on purpose:
497///
498/// - **Liveness** (is the guard we wrote still intact?) is judged against
499/// the *newest* `Cap` entry in the chain. Entries are strictly appended,
500/// so a later Cap's write always supersedes an earlier one's on disk —
501/// `should_restore`'s string-equality check must compare against the
502/// entry whose `our_high` is what's actually there right now.
503/// - **Value** (what do we restore to?) is [`restore_target`]'s *oldest*
504/// `Cap` entry's `prev_high` — the true pre-intervention value, not an
505/// intermediate entry's `prev_high` (which is just our own previous
506/// `our_high` from an earlier cap in the same chain).
507///
508/// Judging both against the oldest Cap (NEW-1 regression) makes a stacked
509/// `[Cap, Cap]` chain's liveness check compare a stale `our_high` against
510/// the newer write on disk, so it never matches and the chain is treated as
511/// dead — stranding `memory.high` at the guard's value forever. Judging
512/// liveness against `entries.last()` fails the same way for `[Cap, Freeze]`
513/// (Promoted Minor B): the newest entry is the `Freeze`, which has no
514/// `memory.high` of its own. Returns [`RestoreStep::ThawOnly`] when there's
515/// no `Cap` entry anywhere in the chain (a `Freeze`-only chain) — nothing to
516/// restore beyond the caller's unconditional thaw.
517pub fn restore_decision(
518 entries: &[JournalEntry],
519 inode: Option<u64>,
520 high: Option<&str>,
521) -> RestoreStep {
522 let Some(newest_cap) = entries
523 .iter()
524 .rev()
525 .find(|e| e.action == JournalAction::Cap)
526 else {
527 return RestoreStep::ThawOnly;
528 };
529 match restore_step(newest_cap, inode, high) {
530 RestoreStep::ThawAndRestoreHigh { .. } => match restore_target(entries) {
531 Some(to) => RestoreStep::ThawAndRestoreHigh { to },
532 // Unreachable: `newest_cap` being a Cap guarantees `restore_target`
533 // (which only needs *any* Cap entry) also finds one.
534 None => RestoreStep::ThawOnly,
535 },
536 other => other,
537 }
538}
539
540/// Pure: given all journal entries for one cgroup, oldest-first, select the
541/// `memory.high` value to restore, if the chain contains a `Cap` entry. The
542/// OLDEST `Cap` entry's `prev_high` is the true pre-intervention value — a
543/// later entry's `prev_high` is just our own previous `our_high` from an
544/// earlier cap in the same chain, not the original (Task 6 review,
545/// Important #3). Returns `None` if there's no `Cap` entry (e.g. a
546/// `Freeze`-only chain).
547pub fn restore_target(entries: &[JournalEntry]) -> Option<String> {
548 entries
549 .iter()
550 .find(|e| e.action == JournalAction::Cap)
551 .map(|e| e.prev_high.clone().unwrap_or_else(|| "max".into()))
552}
553
554/// Runtime page size in bytes, via `sysconf(_SC_PAGESIZE)`. Falls back to
555/// 4096 (by far the most common value) only if the syscall ever returns
556/// something nonsensical — a defensive fallback, not the source of truth,
557/// since the whole point is to match whatever the kernel actually enforces.
558fn page_size() -> u64 {
559 // SAFETY: sysconf(_SC_PAGESIZE) reads a static system parameter; no
560 // pointers involved, no side effects.
561 let p = unsafe { libc::sysconf(libc::_SC_PAGESIZE) };
562 if p > 0 {
563 p as u64
564 } else {
565 4096
566 }
567}
568
569/// Pure: round `bytes` down to a multiple of `page` (a no-op if `page` is 0).
570/// The kernel truncates `memory.high` writes to page multiples (Task 6
571/// review, Critical #1), so we must journal/write the value it will
572/// actually store, not the pre-truncation target — otherwise
573/// `should_restore`'s string-equality check can never pass again and a cap
574/// becomes permanent.
575fn page_align_down(bytes: u64, page: u64) -> u64 {
576 bytes.checked_div(page).map_or(bytes, |q| q * page)
577}
578
579/// Best-effort desktop notification via `notify-send`. Silently does nothing if
580/// the binary is missing or the spawn fails — notifications must never break the
581/// guard.
582fn notify(message: &str) {
583 match Command::new("notify-send")
584 .arg("rlm-guard")
585 .arg(message)
586 .spawn()
587 {
588 Ok(mut child) => {
589 // Reap asynchronously so we don't block; ignore any wait error.
590 std::thread::spawn(move || {
591 let _ = child.wait();
592 });
593 }
594 Err(e) => {
595 tracing::debug!(error = %e, "notify-send unavailable; skipping notification");
596 }
597 }
598}
599
600#[cfg(test)]
601mod tests {
602 use super::super::resolve::{Coverage, Verdict};
603 use super::*;
604
605 const MIB: u64 = 1024 * 1024;
606 const GIB: u64 = 1024 * MIB;
607
608 #[test]
609 fn cap_never_demands_more_than_ten_percent_of_current() {
610 // Incident: Chrome scope with 1.5 GiB charged, mostly page cache, ~150 MiB anon.
611 let current = GIB * 3 / 2;
612 let cap = cap_target(Some(current), Some(GIB * 135 / 100), true).unwrap();
613 assert!(
614 cap >= current / 10 * 9,
615 "cap {cap} is below 90% of {current}"
616 );
617 }
618
619 #[test]
620 fn swapless_cap_only_asks_for_reclaimable_file_pages() {
621 assert_eq!(
622 cap_target(Some(GIB), Some(0), false),
623 Some(GIB),
624 "nothing reclaimable: demand nothing"
625 );
626 assert_eq!(
627 cap_target(Some(GIB), Some(512 * MIB), false),
628 Some(GIB / 10 * 9)
629 );
630 assert_eq!(
631 cap_target(Some(GIB), None, false),
632 Some(GIB),
633 "unknown file size: demand nothing"
634 );
635 }
636
637 #[test]
638 fn cap_has_a_256_mib_floor() {
639 assert_eq!(
640 cap_target(Some(100 * MIB), Some(0), true),
641 Some(MIN_CAP_BYTES)
642 );
643 }
644
645 #[test]
646 fn cap_never_loosens_an_existing_memory_high() {
647 let floor = cap_target(Some(100 * MIB), Some(0), true).unwrap();
648 assert_eq!(floor, MIN_CAP_BYTES);
649 let prev = (200 * MIB).to_string();
650 assert!(
651 !cap_tightens(Some(&prev), floor),
652 "200 MiB unit limit must not be raised to the 256 MiB floor"
653 );
654 assert!(
655 !cap_tightens(Some(&floor.to_string()), floor),
656 "equal: no-op"
657 );
658 assert!(cap_tightens(Some("max"), floor));
659 assert!(cap_tightens(None, floor));
660 let prev = (2 * GIB).to_string();
661 assert!(cap_tightens(Some(&prev), GIB));
662 }
663
664 #[test]
665 fn unreadable_current_refuses_to_cap() {
666 assert_eq!(cap_target(None, Some(1), true), None);
667 }
668
669 /// Carry-forward (Task 3 review): `our_high` must be a plain decimal
670 /// string with no grouping/whitespace, since `should_restore` compares
671 /// it by string equality against `cgfs::read_high` (which only trims).
672 #[test]
673 fn our_high_string_is_plain_decimal_no_separators() {
674 let bytes = cap_target(Some(12_345_678_900), None, true).unwrap();
675 let s = bytes.to_string();
676 assert!(
677 s.chars().all(|c| c.is_ascii_digit()),
678 "our_high must be plain digits, got {s:?}"
679 );
680 assert_eq!(s, format!("{bytes}"), "no formatting beyond plain decimal");
681 }
682
683 /// Task 6 review, Critical #1: the kernel truncates `memory.high`
684 /// writes to page multiples (the reviewer's own observed example:
685 /// writing 900_000_000 with a 4096-byte page reads back 899_997_696).
686 #[test]
687 fn page_align_down_rounds_to_page_multiple() {
688 assert_eq!(page_align_down(900_000_000, 4096), 899_997_696);
689 assert_eq!(
690 page_align_down(4096, 4096),
691 4096,
692 "already-aligned is a no-op"
693 );
694 assert_eq!(page_align_down(100, 0), 100, "page=0 guard is a no-op");
695 }
696
697 /// Task 6 review, Important #3: when duplicate/stacked `Cap` entries end
698 /// up coexisting for one cgroup (a leak from an incomplete prior
699 /// removal), the restore target must be the OLDEST entry's `prev_high`
700 /// — the true pre-intervention value. The newer entry's `prev_high` is
701 /// deliberately a different-looking value here to prove we don't
702 /// chain-follow it.
703 #[test]
704 fn restore_target_uses_oldest_caps_prev_high() {
705 let oldest = entry_cap("/x", 42, "max", "A");
706 let newest = entry_cap("/x", 42, "A-prime", "B");
707 assert_eq!(
708 restore_target(&[oldest, newest]),
709 Some("max".into()),
710 "must use the oldest entry's prev_high, not the newest's"
711 );
712 }
713
714 #[test]
715 fn restore_target_none_for_freeze_only_chain() {
716 assert_eq!(restore_target(&[entry_freeze("/x", 42)]), None);
717 }
718
719 /// Regression test for NEW-1: `restore_decision`'s liveness check must be
720 /// judged against the NEWEST `Cap` entry (whose `our_high` is what's
721 /// actually on disk), while the value restored stays the OLDEST `Cap`
722 /// entry's `prev_high`. A stacked `[Cap(prev="max", our="A"),
723 /// Cap(prev="A", our="B")]` chain with disk `memory.high == "B"` must
724 /// restore to `"max"` — not `SkipRemove`, which the old
725 /// `entries.iter().find()` (oldest-Cap-for-everything) produced: it
726 /// compared the oldest entry's `our_high` ("A") against disk ("B"),
727 /// never matched, and permanently stranded `memory.high`. Also covers
728 /// `[Cap, Freeze]` (Promoted Minor B's shape) and a bare `[Cap]` chain in
729 /// the same test so a future re-swap of either selection can't slip by.
730 #[test]
731 fn restore_decision_liveness_uses_newest_cap_value_uses_oldest() {
732 // [Cap, Cap]: disk holds the NEWEST Cap's our_high ("B"). Liveness
733 // must be checked against "B", not the oldest entry's "A".
734 let oldest = entry_cap("/x", 42, "max", "A");
735 let newest = entry_cap("/x", 42, "A", "B");
736 assert_eq!(
737 restore_decision(&[oldest.clone(), newest.clone()], Some(42), Some("B")),
738 RestoreStep::ThawAndRestoreHigh { to: "max".into() },
739 "must judge liveness against the newest Cap's our_high (matches disk \"B\"), \
740 but restore the oldest Cap's prev_high (\"max\")"
741 );
742 // Sanity: the stale intermediate value "A" is no longer live on disk,
743 // so checking against it (the old bug) would report SkipRemove.
744 assert_eq!(
745 restore_step(&oldest, Some(42), Some("B")),
746 RestoreStep::SkipRemove,
747 "confirms the bug this guards against: judging liveness against the oldest \
748 entry's our_high against the newer on-disk value mismatches"
749 );
750
751 // [Cap, Freeze]: newest entry has no memory.high of its own; must
752 // still fall through to the Cap for both liveness and value.
753 let cap = entry_cap("/x", 42, "max", "1000");
754 let frz = entry_freeze("/x", 42);
755 assert_eq!(
756 restore_decision(&[cap, frz], Some(42), Some("1000")),
757 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
758 );
759
760 // [Cap] alone: baseline single-entry behavior is unchanged.
761 let cap_only = entry_cap("/x", 42, "max", "1000");
762 assert_eq!(
763 restore_decision(&[cap_only], Some(42), Some("1000")),
764 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
765 );
766
767 // Freeze-only chain: nothing to restore.
768 assert_eq!(
769 restore_decision(&[entry_freeze("/x", 42)], Some(42), None),
770 RestoreStep::ThawOnly
771 );
772 }
773
774 fn entry_cap(cg: &str, inode: u64, prev: &str, our: &str) -> JournalEntry {
775 JournalEntry {
776 cgroup: cg.into(),
777 inode,
778 unit: None,
779 action: JournalAction::Cap,
780 prev_high: Some(prev.into()),
781 our_high: Some(our.into()),
782 }
783 }
784
785 fn entry_freeze(cg: &str, inode: u64) -> JournalEntry {
786 JournalEntry {
787 cgroup: cg.into(),
788 inode,
789 unit: None,
790 action: JournalAction::Freeze,
791 prev_high: None,
792 our_high: None,
793 }
794 }
795
796 #[test]
797 fn restore_step_matrix() {
798 let cap = entry_cap("/x", 42, "max", "1000");
799 assert_eq!(
800 restore_step(&cap, Some(42), Some("1000")),
801 RestoreStep::ThawAndRestoreHigh { to: "max".into() }
802 );
803 assert_eq!(
804 restore_step(&cap, Some(43), Some("1000")),
805 RestoreStep::SkipRemove
806 );
807 assert_eq!(
808 restore_step(&cap, Some(42), Some("777")),
809 RestoreStep::SkipRemove
810 );
811 let frz = entry_freeze("/x", 42);
812 assert_eq!(
813 restore_step(&frz, Some(42), None),
814 RestoreStep::ThawOnly,
815 "alive freeze entry: nothing to restore beyond the unconditional thaw"
816 );
817 assert_eq!(
818 restore_step(&frz, None, None),
819 RestoreStep::SkipRemove,
820 "dead cgroup: restore_step only decides memory.high, never freeze — the \
821 unconditional thaw in Effector::thaw_raw already ran before this is consulted, \
822 so a still-frozen dead cgroup is never left behind (carry-forward: Task 5 review)"
823 );
824 }
825
826 /// Poll `cgfs::read_frozen` until it matches `want` or `timeout` elapses.
827 /// Writing `cgroup.freeze` only *requests* a state change; `cgroup.events`'s
828 /// `frozen` field (what `read_frozen` reads) only flips once the kernel has
829 /// actually quiesced every task in the cgroup, which can lag the write by a
830 /// few milliseconds under load — so a bare immediate read is flaky.
831 fn wait_for_frozen(cgroup: &str, want: bool, timeout: Duration) -> Option<bool> {
832 let deadline = std::time::Instant::now() + timeout;
833 loop {
834 let got = cgfs::read_frozen(cgroup);
835 if got == Some(want) || std::time::Instant::now() >= deadline {
836 return got;
837 }
838 std::thread::sleep(Duration::from_millis(20));
839 }
840 }
841
842 fn test_resolution(cgroup: String) -> Resolution {
843 Resolution {
844 cgroup,
845 unit: None,
846 verdict: Verdict::Freeze,
847 coverage: Coverage::Full,
848 mechanism: Mechanism::Raw,
849 }
850 }
851
852 /// Integration smoke test: freeze a real `sleep` via its (raw, rlm-created)
853 /// cgroup through the journal-backed Effector, confirm it's paused via
854 /// `cgroup.freeze` and journaled, then thaw and confirm the journal entry
855 /// is gone. Only works under cgroup v2 delegation, so it's `#[ignore]`d.
856 #[test]
857 #[ignore = "requires cgroup v2 delegation; run manually"]
858 fn freeze_thaw_real_process_raw_cgroup() {
859 use common::Limit;
860 use std::process::Command;
861
862 let manager = CgroupManager::new().expect("create CgroupManager");
863 let journal_dir = tempfile::tempdir().unwrap();
864 let journal =
865 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
866 let effector = Effector::new(&manager, &journal, None);
867
868 let abs_path = manager
869 .prepare_cgroup("test-freeze-thaw", &Limit::default())
870 .expect("create test cgroup")
871 .path;
872 let cgroup = format!(
873 "/{}",
874 abs_path
875 .strip_prefix("/sys/fs/cgroup")
876 .expect("cgroup under /sys/fs/cgroup")
877 .display()
878 );
879
880 let mut child = Command::new("sleep")
881 .arg("30")
882 .spawn()
883 .expect("spawn sleep");
884 let pid = child.id();
885 manager
886 .add_to_cgroup(&abs_path, pid)
887 .expect("add sleep to test cgroup");
888
889 let res = test_resolution(cgroup.clone());
890
891 effector
892 .apply(&Action::Freeze {
893 res: res.clone(),
894 name: "sleep".into(),
895 })
896 .expect("freeze");
897
898 assert_eq!(
899 wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
900 Some(true),
901 "cgroup should be frozen"
902 );
903 assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
904
905 effector
906 .apply(&Action::Thaw { res: res.clone() })
907 .expect("thaw");
908
909 assert_eq!(
910 wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
911 Some(false),
912 "cgroup should be thawed"
913 );
914 assert!(
915 journal.entries().is_empty(),
916 "journal entry should be removed after thaw"
917 );
918
919 let _ = child.kill();
920 let _ = child.wait();
921 let _ = manager.cleanup_cgroup("test-freeze-thaw");
922 }
923
924 /// Act-in-place integration test: freeze/thaw a real transient systemd
925 /// `--user --scope` unit through the Unit mechanism (D-Bus, with raw
926 /// fallback), confirming the journal-backed round trip end to end.
927 /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
928 #[test]
929 #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
930 fn freeze_thaw_real_transient_scope() {
931 use std::process::Command;
932
933 let uid_out = Command::new("id").arg("-u").output().expect("id -u");
934 let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
935 .trim()
936 .parse()
937 .expect("parse uid");
938
939 let unit_base = format!("rlm-e2e-{}", std::process::id());
940 let unit = format!("{unit_base}.scope");
941 let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
942
943 let mut child = Command::new("systemd-run")
944 .args([
945 "--user",
946 "--scope",
947 "--slice=app.slice",
948 &format!("--unit={unit_base}"),
949 "--",
950 "sleep",
951 "30",
952 ])
953 .spawn()
954 .expect("spawn systemd-run --scope");
955
956 // Give systemd a moment to register the transient scope's cgroup.
957 std::thread::sleep(Duration::from_millis(300));
958
959 let manager = CgroupManager::new().expect("create CgroupManager");
960 let journal_dir = tempfile::tempdir().unwrap();
961 let journal =
962 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
963 let systemd = SystemdUser::connect();
964 let effector = Effector::new(&manager, &journal, systemd.as_ref());
965
966 let res = Resolution {
967 cgroup: cgroup.clone(),
968 unit: Some(unit),
969 verdict: Verdict::Freeze,
970 coverage: Coverage::Full,
971 mechanism: Mechanism::Unit,
972 };
973
974 effector
975 .apply(&Action::Freeze {
976 res: res.clone(),
977 name: "sleep".into(),
978 })
979 .expect("freeze");
980 assert_eq!(
981 wait_for_frozen(&cgroup, true, Duration::from_secs(2)),
982 Some(true),
983 "scope cgroup should be frozen"
984 );
985 assert_eq!(journal.entries().len(), 1, "freeze should be journaled");
986
987 effector.apply(&Action::Thaw { res }).expect("thaw");
988 assert_eq!(
989 wait_for_frozen(&cgroup, false, Duration::from_secs(2)),
990 Some(false),
991 "scope cgroup should be thawed"
992 );
993 assert!(
994 journal.entries().is_empty(),
995 "journal entry should be removed after thaw"
996 );
997
998 let _ = child.kill();
999 let _ = child.wait();
1000 let _ = Command::new("systemctl")
1001 .args(["--user", "stop", &format!("{unit_base}.scope")])
1002 .status();
1003 }
1004
1005 /// Regression test for Task 6 review Critical #1: cap a real cgroup and
1006 /// assert `cgfs::read_high` matches the journaled `our_high` exactly —
1007 /// i.e. the value we wrote never got silently page-truncated out from
1008 /// under the journal. The target process holds enough random anon
1009 /// memory (well above the 256 MiB `MIN_CAP_BYTES` floor, a round power
1010 /// of two that would pass even without the fix and prove nothing) that
1011 /// the sized cap is very unlikely to already sit on a page boundary. Requires cgroup v2
1012 /// delegation, so it's `#[ignore]`d.
1013 #[test]
1014 #[ignore = "requires cgroup v2 delegation; run manually"]
1015 fn cap_page_aligns_and_journal_matches_on_disk_value() {
1016 use common::Limit;
1017 use std::process::Command;
1018
1019 let manager = CgroupManager::new().expect("create CgroupManager");
1020 let journal_dir = tempfile::tempdir().unwrap();
1021 let journal =
1022 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1023 let effector = Effector::new(&manager, &journal, None);
1024
1025 let abs_path = manager
1026 .prepare_cgroup("test-cap-align", &Limit::default())
1027 .expect("create test cgroup")
1028 .path;
1029 let cgroup = format!(
1030 "/{}",
1031 abs_path
1032 .strip_prefix("/sys/fs/cgroup")
1033 .expect("cgroup under /sys/fs/cgroup")
1034 .display()
1035 );
1036
1037 // Hold ~400MB of anon memory (above the 256 MiB floor) whose sized
1038 // cap is very unlikely to land on a page boundary, then sleep.
1039 let mut child = Command::new("bash")
1040 .arg("-c")
1041 .arg("a=$(head -c 300000000 /dev/urandom | base64 -w0); sleep 30")
1042 .spawn()
1043 .expect("spawn memory-holding process");
1044 let pid = child.id();
1045 manager
1046 .add_to_cgroup(&abs_path, pid)
1047 .expect("add process to test cgroup");
1048 // Wait until the shell has actually built up the allocation, or the
1049 // cap would sit on the 256 MiB floor and prove nothing.
1050 let want = 300 * 1024 * 1024;
1051 let deadline = std::time::Instant::now() + Duration::from_secs(10);
1052 loop {
1053 let cur = cgfs::current_bytes(&cgroup).unwrap_or(0);
1054 if cur >= want {
1055 break;
1056 }
1057 if std::time::Instant::now() >= deadline {
1058 let _ = child.kill();
1059 let _ = child.wait();
1060 let _ = manager.cleanup_cgroup("test-cap-align");
1061 panic!("memory.current reached only {cur} bytes, need {want}, after 10s");
1062 }
1063 std::thread::sleep(Duration::from_millis(100));
1064 }
1065
1066 let res = test_resolution(cgroup.clone());
1067 effector
1068 .apply(&Action::Cap {
1069 res: res.clone(),
1070 name: "mem-hog".into(),
1071 })
1072 .expect("cap");
1073
1074 let entries = journal.entries();
1075 assert_eq!(entries.len(), 1, "cap should be journaled");
1076 let journaled = entries[0].our_high.clone().expect("our_high recorded");
1077
1078 let on_disk = cgfs::read_high(&cgroup).expect("read memory.high");
1079 assert_eq!(
1080 on_disk, journaled,
1081 "on-disk memory.high must match the journaled our_high exactly"
1082 );
1083
1084 let bytes: u64 = journaled.parse().expect("our_high is a plain decimal");
1085 assert_eq!(bytes % page_size(), 0, "written value must be page-aligned");
1086
1087 effector.apply(&Action::LiftCap { res }).expect("lift cap");
1088 assert!(
1089 journal.entries().is_empty(),
1090 "journal entry removed after lift"
1091 );
1092
1093 let _ = child.kill();
1094 let _ = child.wait();
1095 let _ = manager.cleanup_cgroup("test-cap-align");
1096 }
1097
1098 /// Cap and lift go through raw `memory.high` writes only, so the unit's
1099 /// own `MemoryHigh` property (as systemd reports it) must be the same
1100 /// before the cap and after the lift, and the raw `memory.high` must be
1101 /// back at its pre-cap value. A runtime `SetUnitProperties` would leave
1102 /// a `/run` drop-in behind that changes the reported property.
1103 /// Requires a session bus and cgroup v2 delegation, so it's `#[ignore]`d.
1104 #[test]
1105 #[ignore = "requires a session bus and cgroup v2 delegation; run manually"]
1106 fn lift_cap_leaves_unit_memory_high_property_untouched() {
1107 use std::process::Command;
1108
1109 let unit_memory_high = |unit: &str| -> String {
1110 let out = Command::new("systemctl")
1111 .args(["--user", "show", "-p", "MemoryHigh", "--value", unit])
1112 .output()
1113 .expect("systemctl --user show");
1114 String::from_utf8_lossy(&out.stdout).trim().to_string()
1115 };
1116
1117 let uid_out = Command::new("id").arg("-u").output().expect("id -u");
1118 let uid: u32 = String::from_utf8_lossy(&uid_out.stdout)
1119 .trim()
1120 .parse()
1121 .expect("parse uid");
1122
1123 let unit_base = format!("rlm-e2e-cap-{}", std::process::id());
1124 let unit = format!("{unit_base}.scope");
1125 let cgroup = format!("/user.slice/user-{uid}.slice/user@{uid}.service/app.slice/{unit}");
1126
1127 let mut child = Command::new("systemd-run")
1128 .args([
1129 "--user",
1130 "--scope",
1131 "--slice=app.slice",
1132 &format!("--unit={unit_base}"),
1133 // A genuine prior MemoryHigh, distinct from both "max" and
1134 // whatever Cap computes, so the restored value is
1135 // unambiguous.
1136 "--property=MemoryHigh=1500M",
1137 "--",
1138 "sleep",
1139 "30",
1140 ])
1141 .spawn()
1142 .expect("spawn systemd-run --scope with MemoryHigh set");
1143
1144 std::thread::sleep(Duration::from_millis(300));
1145
1146 let manager = CgroupManager::new().expect("create CgroupManager");
1147 let journal_dir = tempfile::tempdir().unwrap();
1148 let journal =
1149 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1150 let systemd = SystemdUser::connect();
1151 let effector = Effector::new(&manager, &journal, systemd.as_ref());
1152
1153 let original_high =
1154 cgfs::read_high(&cgroup).expect("systemd-run set an initial memory.high");
1155 assert_ne!(
1156 original_high, "max",
1157 "test needs a concrete prior MemoryHigh to distinguish from systemd's clear-to-max"
1158 );
1159 let property_before = unit_memory_high(&unit);
1160
1161 let res = Resolution {
1162 cgroup: cgroup.clone(),
1163 unit: Some(unit.clone()),
1164 verdict: Verdict::CapOnly,
1165 coverage: Coverage::Full,
1166 mechanism: Mechanism::Unit,
1167 };
1168
1169 effector
1170 .apply(&Action::Cap {
1171 res: res.clone(),
1172 name: "sleep".into(),
1173 })
1174 .expect("cap");
1175 assert_ne!(
1176 cgfs::read_high(&cgroup),
1177 Some(original_high.clone()),
1178 "cap should have changed memory.high"
1179 );
1180
1181 effector.apply(&Action::LiftCap { res }).expect("lift cap");
1182 assert_eq!(
1183 unit_memory_high(&unit),
1184 property_before,
1185 "cap/lift must not change the unit's MemoryHigh property"
1186 );
1187 assert_eq!(
1188 cgfs::read_high(&cgroup),
1189 Some(original_high),
1190 "lift must restore the raw pre-cap memory.high"
1191 );
1192 assert!(
1193 journal.entries().is_empty(),
1194 "journal entry removed after lift"
1195 );
1196
1197 let _ = child.kill();
1198 let _ = child.wait();
1199 let _ = Command::new("systemctl")
1200 .args(["--user", "stop", &unit])
1201 .status();
1202 }
1203
1204 /// Regression test for Promoted Minor B: a leaked `[Cap, Freeze]` chain
1205 /// (Cap journaled first, then the same cgroup frozen later without the
1206 /// Cap entry ever being cleared) must still restore the Cap's
1207 /// `prev_high` on thaw. Before the fix, liveness was judged against
1208 /// `entries.last()` — the Freeze entry, which has no `memory.high` of
1209 /// its own — so `restore_step` reported "nothing to restore" and
1210 /// `memory.high` stayed pinned at the guard's Cap value forever once the
1211 /// whole chain was cleared. Requires cgroup v2 delegation, so it's
1212 /// `#[ignore]`d.
1213 #[test]
1214 #[ignore = "requires cgroup v2 delegation; run manually"]
1215 fn chain_restores_cap_value_when_newest_entry_is_freeze() {
1216 use common::Limit;
1217 use std::process::Command;
1218
1219 let manager = CgroupManager::new().expect("create CgroupManager");
1220 let journal_dir = tempfile::tempdir().unwrap();
1221 let journal =
1222 Journal::open(journal_dir.path().join("j.jsonl"), "test-boot".into()).unwrap();
1223 let effector = Effector::new(&manager, &journal, None);
1224
1225 let abs_path = manager
1226 .prepare_cgroup("test-chain-restore", &Limit::default())
1227 .expect("create test cgroup")
1228 .path;
1229 let cgroup = format!(
1230 "/{}",
1231 abs_path
1232 .strip_prefix("/sys/fs/cgroup")
1233 .expect("cgroup under /sys/fs/cgroup")
1234 .display()
1235 );
1236
1237 let mut child = Command::new("sleep")
1238 .arg("30")
1239 .spawn()
1240 .expect("spawn sleep");
1241 let pid = child.id();
1242 manager
1243 .add_to_cgroup(&abs_path, pid)
1244 .expect("add sleep to test cgroup");
1245
1246 let original_high = cgfs::read_high(&cgroup).expect("read initial memory.high");
1247 let res = test_resolution(cgroup.clone());
1248
1249 // Cap first (oldest entry)...
1250 effector
1251 .apply(&Action::Cap {
1252 res: res.clone(),
1253 name: "sleep".into(),
1254 })
1255 .expect("cap");
1256 assert_ne!(
1257 cgfs::read_high(&cgroup),
1258 Some(original_high.clone()),
1259 "cap should have changed memory.high"
1260 );
1261
1262 // ...then freeze the same cgroup without ever clearing the Cap
1263 // entry — this is the leaked-chain scenario: two coexisting entries,
1264 // Freeze newest.
1265 effector
1266 .apply(&Action::Freeze {
1267 res: res.clone(),
1268 name: "sleep".into(),
1269 })
1270 .expect("freeze");
1271
1272 let entries = journal.entries();
1273 assert_eq!(entries.len(), 2, "both Cap and Freeze entries coexist");
1274 assert_eq!(entries[0].action, JournalAction::Cap, "Cap is oldest");
1275 assert_eq!(entries[1].action, JournalAction::Freeze, "Freeze is newest");
1276
1277 // Thaw (the action that would naturally follow a Freeze) must still
1278 // restore the Cap's prev_high, not skip restoration just because the
1279 // newest entry is a Freeze with no memory.high of its own.
1280 effector.apply(&Action::Thaw { res }).expect("thaw");
1281 assert_eq!(
1282 cgfs::read_high(&cgroup),
1283 Some(original_high),
1284 "thaw must restore the chain's Cap value even though Freeze is the newest entry"
1285 );
1286 assert!(
1287 journal.entries().is_empty(),
1288 "both chain entries removed after thaw"
1289 );
1290
1291 let _ = child.kill();
1292 let _ = child.wait();
1293 let _ = manager.cleanup_cgroup("test-chain-restore");
1294 }
1295}