Skip to main content

magi/
clean.rs

1//! The disk janitor: finished runs get their worktrees folded, worktrees whose
2//! run record is already gone get reclaimed too, and the shared build cache is
3//! pruned to its cap.
4//!
5//! A run's state being written by an older schema is not the same thing as it
6//! being unreadable, and this module used to conflate the two: [`fold_due`]
7//! treated any `run.json` its version check rejected exactly like one that
8//! failed to parse at all, so a single schema bump silently stopped every
9//! automatic fold in the fleet the moment it shipped, and did so with no
10//! counter and no log line to say so. A record magi genuinely cannot parse —
11//! missing fields, broken JSON, a schema newer than this build has ever heard
12//! of — is still left alone here, still counted in
13//! [`Housekeeping::unreadable`], and still only ever removed by an explicit
14//! operator action (`magi fold`, or the equivalent phone route). One written
15//! by a schema this build merely disagrees with the *meaning* of is not that:
16//! as long as it still parses, folding proceeds regardless of the number in
17//! its `schema` field.
18//!
19//! Everything policy-shaped — which statuses are foldable, how long a finished
20//! run is left alone, whether the cache is over its limit — is a pure function
21//! injected with numbers, so nothing here has to ask the operating system to
22//! be testable. The only I/O is the removal itself.
23
24use std::collections::BTreeSet;
25use std::path::{Path, PathBuf};
26
27use anyhow::{Context as _, Result, bail};
28use jiff::{SignedDuration, Timestamp};
29use serde::Deserialize;
30
31use crate::ask::Questions;
32use crate::config::Disk;
33use crate::run::{RunState, RunStatus, SCHEMA, short_of};
34
35use crate::disk::{Prune, dir_size};
36
37/// What one janitor pass did, for the caller's log line.
38#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
39pub struct Housekeeping {
40    /// Runs folded (worktrees dropped).
41    pub folded: usize,
42    /// Runs [`fold_due`] left alone because their `run.json` genuinely could
43    /// not be read - missing, broken JSON, a schema this build has never
44    /// heard of - as opposed to one merely written by a different schema
45    /// number, which is folded like any other (see the module docs). This was
46    /// defined but never incremented for a long stretch of this module's
47    /// history, which is exactly how 90 of 93 runs sat unfolded on one
48    /// operator's machine with nothing anywhere saying why: every one of them
49    /// was misclassified as unreadable by a schema check that has since been
50    /// narrowed to only the runs that actually are.
51    pub unreadable: usize,
52    /// Worktrees under the worktree bay reclaimed because no run record in
53    /// `runs/` claims them anymore (see [`fold_orphaned_worktrees`]).
54    pub orphaned_worktrees: usize,
55    /// Files dropped from the shared cache.
56    pub cache_files: usize,
57    /// Bytes freed from the shared cache.
58    pub cache_freed: u64,
59    /// Open questions abandoned because the run that asked them has already
60    /// settled where nothing is coming back to read an answer.
61    pub questions_abandoned: usize,
62}
63
64/// Run the janitor: fold due runs, reclaim orphaned worktrees, prune stale
65/// worktree registrations, then prune the cache if it is over its cap.
66///
67/// Every part is best-effort; a jammed cache lock or a run whose worktree
68/// another borrower holds must not stop the rest. Errors are reported through
69/// `tracing::warn` - this is housekeeping, and the daemon keeps serving
70/// either way.
71pub async fn housekeep(
72    cfg: &crate::config::Config,
73    home: &Path,
74    worktrees_root: &Path,
75    repo: &Path,
76    now: Timestamp,
77) -> Housekeeping {
78    let mut out = Housekeeping::default();
79    if cfg.disk.auto_fold {
80        let runs = home.join("runs");
81        match fold_due(&runs, home, worktrees_root, &cfg.disk, now).await {
82            Ok((folded, unreadable)) => {
83                out.folded = folded;
84                out.unreadable = unreadable;
85            }
86            Err(e) => {
87                tracing::warn!("housekeep: fold due runs: {e:#}");
88                // Stable wording: the reason varies per pass, and a changed
89                // message would relight the bell on every retry.
90                crate::notices::raise_in(
91                    home,
92                    crate::notices::Notice::warn(
93                        "housekeep:fold",
94                        "Automatic cleanup of finished runs failed; disk usage may keep growing.",
95                    ),
96                );
97            }
98        }
99        out.orphaned_worktrees =
100            fold_orphaned_worktrees(&runs, worktrees_root, home, cfg.disk.fold_grace_secs, now)
101                .await;
102        // Best-effort in the same sense as everything else here: a repository
103        // this janitor pass has nothing to do with (or none at all, in a unit
104        // test) must not turn a `warn` into a reason to skip the rest.
105        if let Err(e) = crate::git::worktree_prune(repo).await {
106            tracing::warn!("housekeep: prune worktree registrations: {e:#}");
107        }
108    }
109    match prune_cache_if_over_limit(cfg, home) {
110        Ok(Some(pruned)) => {
111            out.cache_files = pruned.files;
112            out.cache_freed = pruned.freed;
113        }
114        Ok(None) => {}
115        Err(e) => tracing::warn!("housekeep: prune cache: {e:#}"),
116    }
117    // Unconditional, unlike the two passes above: this is not a disk policy
118    // with a cap or an opt-out, it is closing a gap `graph::Runner` itself
119    // cannot - a run that reached `Merged`/`Ready`/`Failed` before this
120    // cleanup existed, or whose process died between saving that status and
121    // abandoning the question it leaves behind (see `Runner::settle_questions`).
122    // Left alone, that question sits `open` forever: the owner's badge,
123    // banner and title all keep counting a decision nobody is left to read.
124    out.questions_abandoned =
125        abandon_settled_questions(&Questions::at(home.join("questions")), &home.join("runs"));
126    out
127}
128
129/// Abandon every open question whose run has already settled into a status
130/// nothing comes back from, worded with what the run became - the same
131/// cleanup `graph::Runner::settle_questions` runs the moment `status` lands
132/// there, for questions that missed it.
133///
134/// Scans questions rather than runs: the open list is normally short, and a
135/// run that never asked anything costs nothing here. A run this cannot read,
136/// deleted or written by a schema this build does not speak, is left alone
137/// the same as everywhere else in this module; the question stays open
138/// rather than guessed at.
139pub fn abandon_settled_questions(store: &Questions, runs: &Path) -> usize {
140    let waiting_on: BTreeSet<String> = store
141        .list()
142        .into_iter()
143        .filter(|q| q.status.open())
144        .map(|q| q.run)
145        .collect();
146    let mut abandoned = 0;
147    for run in waiting_on {
148        let Ok(meta) = read_meta(runs, &run) else {
149            continue;
150        };
151        match store.settle_run(&run, meta.status) {
152            Ok(n) => abandoned += n,
153            Err(e) => tracing::warn!("housekeep: abandon questions for {run}: {e:#}"),
154        }
155    }
156    abandoned
157}
158
159/// Fold every run that is finished, older than the grace period, and not being
160/// worked on; return `(folded, unreadable)`.
161///
162/// A run whose `run.json` genuinely cannot be parsed — missing fields, broken
163/// JSON, a schema newer than this build has ever heard of — is left exactly
164/// as it is. Automatic housekeeping cannot tell a mid-write file from one that
165/// will never parse again, and `<home>/runs/<id>/` is the evidence `magi
166/// stats` and the deck read; when unsure whether it is safe to touch, the
167/// janitor keeps rather than deletes (see the module docs). Discarding a
168/// record this unreadable is an explicit operator action (`magi fold`, or the
169/// equivalent phone route), never something that happens unattended. Every
170/// such skip is counted in the returned `unreadable` and logged through
171/// `tracing::warn` with the parse failure that caused it - silence here is
172/// exactly the failure mode that let 90 of 93 runs sit unfolded with nothing
173/// to show for it.
174///
175/// A run merely written by a *different* schema number is not unreadable: as
176/// long as `run.json` still parses, it folds like any other terminal run (see
177/// the module docs for why the two are different questions).
178///
179/// Runnable statuses and runs newer than the grace period are also left
180/// alone; folding them would throw away work that is still the answer to
181/// somebody's question. `Merged` runs forget their winner's worktree (the
182/// merge already landed it); `Ready` and `Failed` runs keep it.
183pub async fn fold_due(
184    runs: &Path,
185    home: &Path,
186    _worktrees_root: &Path,
187    disk: &Disk,
188    now: Timestamp,
189) -> Result<(usize, usize)> {
190    let mut folded = 0usize;
191    let mut unreadable = 0usize;
192    let mut ids: Vec<String> = std::fs::read_dir(runs)
193        .into_iter()
194        .flatten()
195        .flatten()
196        .filter(|e| e.path().join("run.json").is_file())
197        .map(|e| e.file_name().to_string_lossy().into_owned())
198        .collect();
199    ids.sort_unstable();
200    for id in ids {
201        if crate::daemon::is_working_on(home, &id, now) {
202            continue;
203        }
204        let meta = match read_meta(runs, &id) {
205            Ok(meta) => meta,
206            Err(e) => {
207                unreadable += 1;
208                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
209                continue;
210            }
211        };
212        if meta.status.resumable() || !due(now, meta.updated_at, disk.fold_grace_secs) {
213            continue;
214        }
215        // `read_meta` already proved the file parses; `read_state` asks for
216        // the rest of the fields `graph::fold_run` needs (worktree paths,
217        // candidates, tally). A schema mismatch alone does not fail this -
218        // see the module docs - so reaching `Err` here means the JSON itself
219        // is broken in a way `read_meta` did not exercise, which is rare but
220        // not impossible (a body truncated between the two fields it reads
221        // and the rest). That must not cost every other run its turn through
222        // this loop, so it is a skip, not a `?`.
223        let mut state = match read_state(runs, &id) {
224            Ok(state) => state,
225            Err(e) => {
226                unreadable += 1;
227                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
228                continue;
229            }
230        };
231        if state.schema != SCHEMA {
232            tracing::info!(
233                "housekeep: run {id} was written by schema {} (this build speaks {SCHEMA}); \
234                 folding it anyway",
235                state.schema
236            );
237        }
238        let drop_winner = state.status == RunStatus::Merged;
239        // One run's fold must not cost every later run its turn. A worktree
240        // another borrower holds, a branch git refuses to delete, a repository
241        // that has since moved: each is a reason this run cannot be folded
242        // now, and none is a reason to stop the pass. Left unfolded, it is
243        // simply due again next time; a `?` here stopped automatic folding
244        // permanently at the first such run (finding R3-1-1 of run 51a3).
245        match crate::graph::fold_run(&mut state, drop_winner, home).await {
246            Ok(_) => folded += 1,
247            Err(e) => tracing::warn!("housekeep: fold {id}: {e:#}"),
248        }
249    }
250    Ok((folded, unreadable))
251}
252
253/// Is `updated` old enough, measured against `now`, that the run may fold?
254///
255/// Pure; the janitor compares against wallclock, tests inject both sides. The
256/// comparison is strict, so a run exactly at the edge of its grace period is
257/// left alone one more pass — the same convention as [`crate::disk::over_limit`].
258pub fn due(now: Timestamp, updated: Timestamp, grace_secs: u64) -> bool {
259    now.duration_since(updated) > SignedDuration::new(grace_secs as i64, 0)
260}
261
262/// The two fields the janitor decides on, read with a serde that tolerates
263/// everything else about the run being unreadable.
264#[derive(Deserialize)]
265struct Meta {
266    status: RunStatus,
267    updated_at: Timestamp,
268}
269
270/// Read `status` and `updated_at` straight off the state file, asking for
271/// nothing else. `Err` when the file is missing, not parseable, or a status in
272/// a version this build does not speak - all of which mean "unreadable".
273fn read_meta(runs: &Path, id: &str) -> Result<Meta> {
274    let path = runs.join(id).join("run.json");
275    let body =
276        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
277    let meta: Meta =
278        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
279    Ok(meta)
280}
281
282/// Read a whole run state from a runs directory, for folding only.
283///
284/// Deliberately more permissive than [`RunState::load`], which this does not
285/// call: `load` backs `--resume` and every hand-driven command, where a
286/// schema this build disagrees with the *meaning* of must refuse outright
287/// rather than resume a review round or a tally against stale semantics
288/// (`RunState::SCHEMA`'s own docs list what has changed meaning at each
289/// bump). Folding recomputes nothing - it only reads worktree paths, branch
290/// names and a tally winner off the struct to remove them - so an old
291/// schema's values are exactly as good here as a current one's; every schema
292/// bump so far has only ever added a field or a variant, never repurposed an
293/// existing one, and serde already fills an added field's default when an
294/// older record has nothing to say about it. What this cannot tolerate, and
295/// what still surfaces as an `Err`, is `run.json` failing to parse at all.
296fn read_state(runs: &Path, id: &str) -> Result<RunState> {
297    let path = runs.join(id).join("run.json");
298    let body =
299        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
300    let state: RunState =
301        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
302    Ok(state)
303}
304
305/// Remove a run that cannot be read: its state directory under `runs` and its
306/// worktree directory under `worktrees_root`.
307///
308/// The state file is the only record of a run's repository and branches, so a
309/// run this unreadable is discarded at the filesystem level - there is no
310/// candidate list to fold first. The worktrees live under
311/// [`crate::run::default_worktree_root`] unless the run's config relocated
312/// them, which an unreadable run cannot tell us; the default location is
313/// removed, and anything the run placed elsewhere is a leftover for whoever
314/// knows where it went.
315///
316/// Deleting a worktree directory by hand leaves its registration in git, and a
317/// registered path cannot be re-`worktree add`-ed until it is pruned - so every
318/// worktree is unregistered from its repository first, best-effort, via the
319/// `gitdir:` link git keeps inside the directory.
320pub async fn fold_unreadable(runs: &Path, worktrees_root: &Path, id: &str) -> Result<Vec<String>> {
321    let resolved = resolve_id_path(runs, id)?;
322    let mut removed = Vec::new();
323    let run_dir = runs.join(&resolved);
324    if run_dir.exists() {
325        std::fs::remove_dir_all(&run_dir)
326            .with_context(|| format!("remove {}", run_dir.display()))?;
327        removed.push(format!("runs/{resolved}"));
328    }
329    let wt = worktrees_root.join(short_of(&resolved));
330    if wt.exists() {
331        crate::git::remove_worktree_from_linked(&wt).await;
332        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
333            crate::git::remove_worktree_from_linked(&e.path()).await;
334        }
335        std::fs::remove_dir_all(&wt).with_context(|| format!("remove {}", wt.display()))?;
336        removed.push(wt.to_string_lossy().into_owned());
337    }
338    Ok(removed)
339}
340
341/// Reclaim worktrees under `worktrees_root` that no run record in `runs`
342/// claims anymore, and return how many were removed.
343///
344/// [`fold_due`] only ever sees a worktree by walking `runs/` first, so a
345/// worktree whose run record is already gone — `magi run rm`, or a record
346/// deleted before its worktree — never enters that loop at all: nothing there
347/// is looking for it. This walks the worktree bay directly instead, and
348/// removes any `<short>` directory that no run id maps to.
349///
350/// Two things must never happen, and this checks both before ever touching a
351/// directory:
352///
353/// - **A worktree bay is never the only kind of thing under `worktrees_root`,
354///   and this must not assume it is.** A hand-placed scratch directory, or
355///   anything else an operator or another tool left in the same bay, has the
356///   same "no run claims it" shape as a genuine orphan but is not one -
357///   [`looks_like_a_worktree_bay`] is the same tag shape [`crate::run::is_run_id`]
358///   already requires of a real run's short id, and anything else is left
359///   alone regardless of what else is true about it.
360/// - **A worktree that was only just created might not have a `run.json` yet
361///   for a reason that has nothing to do with being orphaned.** `Runner::start`
362///   and `Runner::review` both create the worktree before the first
363///   `RunState::save` lands, and that gap - several `git` subprocesses wide -
364///   is invisible to [`crate::daemon::is_working_on_short`] whenever the run
365///   is not being driven through this daemon's own `poll` loop at all (a
366///   `magi review` invocation, for one). A directory whose own modification
367///   time is within `grace_secs` of `now` is left alone on that basis alone,
368///   the same margin [`fold_due`] gives a run before treating it as truly
369///   finished - long enough that no realistic gap between a `worktree add`
370///   and its `run.json` could ever be mistaken for one.
371///
372/// The one failure this must never cause is deleting the worktree of a run
373/// that is genuinely in flight. [`crate::daemon::is_working_on_short`] is the
374/// same liveness check [`fold_due`] trusts everywhere else in this module,
375/// checked by short id because there is no full id to compare here; when it
376/// cannot tell, this leaves the directory alone. Best-effort like the rest of
377/// housekeeping: one directory git or the filesystem refuses to give up is a
378/// `tracing::warn`, not a reason to abandon the rest of the pass.
379pub async fn fold_orphaned_worktrees(
380    runs: &Path,
381    worktrees_root: &Path,
382    home: &Path,
383    grace_secs: u64,
384    now: Timestamp,
385) -> usize {
386    let known: std::collections::HashSet<String> = std::fs::read_dir(runs)
387        .into_iter()
388        .flatten()
389        .flatten()
390        .map(|e| e.file_name().to_string_lossy().into_owned())
391        .filter(|name| crate::run::is_run_id(name))
392        .map(|id| short_of(&id).to_owned())
393        .collect();
394
395    let mut folded = 0usize;
396    for entry in std::fs::read_dir(worktrees_root)
397        .into_iter()
398        .flatten()
399        .flatten()
400    {
401        if !entry.path().is_dir() {
402            continue;
403        }
404        let short = entry.file_name().to_string_lossy().into_owned();
405        if !looks_like_a_worktree_bay(&short) {
406            continue;
407        }
408        if known.contains(&short) || crate::daemon::is_working_on_short(home, &short, now) {
409            continue;
410        }
411        let wt = entry.path();
412        if !stale_enough(&wt, grace_secs, now) {
413            continue;
414        }
415        crate::git::remove_worktree_from_linked(&wt).await;
416        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
417            crate::git::remove_worktree_from_linked(&e.path()).await;
418        }
419        match std::fs::remove_dir_all(&wt) {
420            Ok(()) => folded += 1,
421            Err(e) => tracing::warn!(
422                "housekeep: remove orphaned worktree {}: {e:#}",
423                wt.display()
424            ),
425        }
426    }
427    folded
428}
429
430/// Does `name` have the shape a run's own worktree bay is named with: the
431/// same 4-character alphanumeric tag [`crate::run::is_run_id`] requires of a
432/// full id's trailing block (see [`short_of`])?
433///
434/// Anything else under `worktrees_root` is not a bay this function reclaims
435/// at all, claimed or not - answering "does a run claim this?" about a
436/// directory that was never a run's worktree in the first place is exactly
437/// the wrong question to ask before deleting it.
438fn looks_like_a_worktree_bay(name: &str) -> bool {
439    name.len() == 4 && name.bytes().all(|b| b.is_ascii_alphanumeric())
440}
441
442/// Is `dir`'s own modification time old enough, against `grace_secs` and
443/// `now`, that its emptiness of a run record can be trusted rather than
444/// caught mid-creation?
445///
446/// A directory this pass cannot stat at all - a race with its own removal, a
447/// permission error - is treated as not yet stale: unreadable metadata is not
448/// evidence of anything, and the janitor already keeps rather than deletes
449/// whenever it cannot tell (see the module docs).
450///
451/// `grace_secs` is floored at [`MIN_ORPHAN_AGE_SECS`] regardless of what the
452/// caller passes: `0` is a documented, legitimate value for
453/// [`crate::config::Disk::fold_grace_secs`] (`due`'s own "always due" case),
454/// because that grace answers a policy question the operator owns - how long
455/// a *known, finished* run's worktree lingers before cleanup. Whether an
456/// orphan worktree is actually a race with `Runner::review`'s `git worktree
457/// add` landing before its `run.json` is not a policy question, and must not
458/// collapse to zero just because the operator turned the other grace off -
459/// that would defeat the very check meant to catch it.
460fn stale_enough(dir: &Path, grace_secs: u64, now: Timestamp) -> bool {
461    let Ok(modified) = std::fs::metadata(dir).and_then(|m| m.modified()) else {
462        return false;
463    };
464    let Ok(ts) = Timestamp::try_from(modified) else {
465        return false;
466    };
467    due(now, ts, grace_secs.max(MIN_ORPHAN_AGE_SECS))
468}
469
470/// The floor under [`stale_enough`]'s grace, independent of
471/// [`crate::config::Disk::fold_grace_secs`].
472///
473/// Ample next to the race it guards: the gap between `Runner::start` or
474/// `Runner::review` creating a worktree and the first `RunState::save`
475/// landing is a handful of `git` subprocess calls, not minutes - but the
476/// janitor cannot tell "still mid-setup" from "orphaned" by any other signal
477/// for a run that never registers with `daemon::Status` at all (a `magi
478/// review` invocation, for one), so this is generous on purpose rather than
479/// tuned to the observed case.
480const MIN_ORPHAN_AGE_SECS: u64 = 5 * 60;
481
482/// Resolve an id or prefix against an explicit runs directory, exactly the way
483/// [`crate::run::resolve_id`] does against the global home.
484fn resolve_id_path(runs: &Path, prefix: &str) -> Result<String> {
485    // Keyed on the directory, not on a readable state file: the record this
486    // route exists to remove may be a lone `run.json.tmp` from a save that
487    // ran out of disk, and that is precisely the one a human needs a way to
488    // clear (see `crate::run::list_ids`).
489    if runs.join(prefix).is_dir() && crate::run::is_run_id(prefix) {
490        return Ok(prefix.to_owned());
491    }
492    let mut hits: Vec<String> = Vec::new();
493    for e in std::fs::read_dir(runs).into_iter().flatten().flatten() {
494        if !e.path().is_dir() {
495            continue;
496        }
497        let id = e.file_name().to_string_lossy().into_owned();
498        if crate::run::is_run_id(&id) && (id.starts_with(prefix) || id.ends_with(prefix)) {
499            hits.push(id);
500        }
501    }
502    match hits.len() {
503        1 => Ok(hits.into_iter().next().expect("exactly one hit")),
504        0 => bail!("no run matches `{prefix}`"),
505        _ => bail!(
506            "`{prefix}` matches {} runs: {}",
507            hits.len(),
508            hits.join(", ")
509        ),
510    }
511}
512
513/// `magi fold`'s recovery path for a run whose worktrees are already gone —
514/// so [`crate::graph::fold_run`] removed nothing — but whose `run.json` still
515/// lists active seats nobody is left to answer for: no live daemon claims the
516/// run, and every one of those seats has overrun its own timeout (see
517/// [`RunState::active_all_overrun`]). Clearing them and failing the run is
518/// what lets it be deleted afterward — [`RunState::ensure_can_delete`] only
519/// ever checks whether a live daemon is working on the run and whether its
520/// candidates are folded, not `status`, but a run stuck `implementing`
521/// forever with an empty worktree still reads as unresolved everywhere else
522/// (`magi show`, the deck, the phone) until this runs.
523///
524/// Returns `false` without changing anything when a live daemon still claims
525/// the run, or when some active seat has not actually overrun its budget yet
526/// — a run that is merely between waves must never be guessed at.
527pub fn clear_abandoned_active(state: &mut RunState, home: &Path, now: Timestamp) -> Result<bool> {
528    if crate::daemon::is_working_on(home, &state.id, now) || !state.active_all_overrun(now) {
529        return Ok(false);
530    }
531    state.abandon("fold");
532    state.save_under(home)?;
533    // The seat that asked is gone for good now - the same door
534    // `graph::Runner::settle_questions` closes the moment `status` lands
535    // somewhere non-resumable, see that method's own doc. Without this, an
536    // open question the abandoned seat left behind would keep badging the
537    // operator until the next daemon startup's `abandon_settled_questions`
538    // pass happened to notice it, or forever if nothing is running `magi
539    // serve` at all.
540    if let Err(e) = Questions::at(home.join("questions")).settle_run(&state.id, state.status) {
541        tracing::warn!("abandon questions for {}: {e:#}", state.id);
542    }
543    Ok(true)
544}
545
546/// Delete files from the shared build cache until it fits its cap — but only
547/// while nobody live is registered as using it. [`crate::cache::maintenance_prune`]
548/// takes out the same lease a build would, so a prune can never race a
549/// compile in flight (this run's own, another run's, or a human's `magi
550/// review`) into deleting a file that build still needs. `Ok(None)` when the
551/// cache is in use right now; the next pass catches it once the borrower
552/// releases it, the same way a cap of `0` or a missing `CARGO_TARGET_DIR`
553/// already meant "nothing to do this time" here.
554pub fn prune_cache(home: &Path, cache: &Path, limit_bytes: u64) -> Result<Option<Prune>> {
555    crate::cache::maintenance_prune(home, cache, limit_bytes)
556}
557
558/// [`prune_cache`], but resolving the operator's opt-out and missing
559/// `CARGO_TARGET_DIR` first — the same two checks [`housekeep`]'s idle pass
560/// makes before ever measuring the cache, factored out so
561/// [`crate::daemon`]'s between-runs check (see the module's own doc for why
562/// congestion can make "idle" arrive too rarely to matter) makes them
563/// identically rather than growing its own copy that could drift. `Ok(None)`
564/// covers a cap of `0` (see the module docs on `cache_limit_bytes`), a config
565/// that renders no `CARGO_TARGET_DIR` to aggregate at all, and a cache
566/// currently in use (see [`prune_cache`]).
567pub fn prune_cache_if_over_limit(
568    cfg: &crate::config::Config,
569    home: &Path,
570) -> Result<Option<Prune>> {
571    if cfg.disk.cache_limit_bytes == 0 {
572        return Ok(None);
573    }
574    let Some(cache) = cfg.cache_dir() else {
575        return Ok(None);
576    };
577    prune_cache(home, &cache, cfg.disk.cache_limit_bytes)
578}
579
580/// The cache's path, size and cap, for `magi cache show` and the health view.
581/// `None` when the config declares no `CARGO_TARGET_DIR` to aggregate.
582///
583/// A cap of `0` means the operator opted out of pruning; the size is then
584/// reported but never acted on.
585pub fn cache_report(cfg: &crate::config::Config) -> Option<(PathBuf, u64, u64)> {
586    let cache = cfg.cache_dir()?;
587    Some((
588        cache.clone(),
589        cache_size(&cache),
590        cfg.disk.cache_limit_bytes,
591    ))
592}
593
594/// Size in bytes of the shared build cache.
595pub fn cache_size(cache: &Path) -> u64 {
596    dir_size(cache)
597}
598
599#[cfg(test)]
600mod tests {
601    use super::*;
602    use crate::config::Disk;
603    use std::fs;
604
605    fn ts(s: &str) -> Timestamp {
606        s.parse().expect("rfc3339")
607    }
608
609    fn block_on<F: std::future::Future>(f: F) -> F::Output {
610        tokio::runtime::Runtime::new().expect("runtime").block_on(f)
611    }
612
613    #[test]
614    fn a_run_is_due_after_its_grace_and_not_before() {
615        let now = ts("2026-09-05T00:00:00Z");
616        let grace = 600;
617        let old = now - SignedDuration::new(601, 0);
618        let fresh = now - SignedDuration::new(599, 0);
619        assert!(due(now, old, grace));
620        assert!(!due(now, fresh, grace));
621        // Exactly at the edge: not yet due.
622        let edge = now - SignedDuration::new(600, 0);
623        assert!(!due(now, edge, grace));
624        // A zero grace folds everything, ever.
625        assert!(due(now, old, 0));
626    }
627
628    #[test]
629    fn the_meta_reader_is_tolerant_of_everything_except_the_deciders() {
630        let dir = tempfile::tempdir().unwrap();
631        let runs = dir.path().join("runs");
632        let id = "20260905-000000-abcd";
633        std::fs::create_dir_all(runs.join(id)).unwrap();
634        std::fs::write(
635            runs.join(id).join("run.json"),
636            r#"{"schema": 99, "id": "20260905-000000-abcd", "updated_at": "2026-09-05T00:00:00Z", "status": "ready", "junk_from_another_build": [1, 2, 3]}"#,
637        )
638        .unwrap();
639        let meta = read_meta(&runs, id).expect("readable");
640        assert_eq!(meta.status, RunStatus::Ready);
641        assert_eq!(meta.updated_at, ts("2026-09-05T00:00:00Z"));
642        assert!(read_meta(&runs, "nope").is_err(), "missing file unreadable");
643        std::fs::write(runs.join(id).join("run.json"), "not json at all").unwrap();
644        assert!(read_meta(&runs, id).is_err(), "garbage unreadable");
645    }
646
647    #[test]
648    fn fold_unreadable_releases_run_dir_and_worktrees() {
649        let dir = tempfile::tempdir().unwrap();
650        let runs = dir.path().join("runs");
651        let wt = dir.path().join("wt");
652        let id = "20260905-000000-abcd";
653        std::fs::create_dir_all(runs.join(id)).unwrap();
654        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
655        std::fs::create_dir_all(wt.join("abcd")).unwrap();
656        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
657
658        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold");
659        assert_eq!(removed.len(), 2);
660        assert!(!runs.join(id).exists(), "run dir gone");
661        assert!(!wt.join("abcd").exists(), "worktrees gone");
662
663        // A prefix resolves like `run::resolve_id` does.
664        std::fs::create_dir_all(runs.join(id)).unwrap();
665        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
666        std::fs::create_dir_all(wt.join("abcd")).unwrap();
667        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
668        let removed = block_on(fold_unreadable(&runs, &wt, "20260905")).expect("by prefix");
669        assert_eq!(removed.len(), 2);
670        // Once gone, `id` cannot be resolved at all - same as `run::resolve_id`
671        // on an id nothing on disk matches - so a repeat pass errors rather
672        // than silently reporting nothing removed.
673        assert!(
674            block_on(fold_unreadable(&runs, &wt, id)).is_err(),
675            "a run already gone cannot be resolved again"
676        );
677    }
678
679    #[test]
680    fn prune_cache_sheds_the_oldest_generation_until_it_fits() {
681        let home = tempfile::tempdir().unwrap();
682        let dir = tempfile::tempdir().unwrap();
683        // Same size, different age: only the age decides, and the newest
684        // generation - the one the next build reuses - is what survives.
685        fs::write(dir.path().join("old"), b"xx").unwrap();
686        fs::write(dir.path().join("new"), b"yy").unwrap();
687        touch(&dir.path().join("old"), 1_000_000);
688        touch(&dir.path().join("new"), 2_000_000);
689
690        let out = prune_cache(home.path(), dir.path(), 2)
691            .expect("prune")
692            .expect("the cache is free");
693        assert_eq!(out.files, 1, "one deletion is enough to reach the cap");
694        assert_eq!(out.remaining, 2);
695        assert!(!dir.path().join("old").exists(), "the older file went");
696        assert!(dir.path().join("new").exists(), "the newer one stayed");
697
698        // A whole generation shares one timestamp tick, so the tie has to be
699        // decided too: largest first, which reaches the cap in the fewest
700        // deletions. Left to `read_dir` and an unstable sort this deleted
701        // both files on Linux and one on Windows.
702        let tied = tempfile::tempdir().unwrap();
703        fs::write(tied.path().join("big"), b"xxxx").unwrap();
704        fs::write(tied.path().join("small"), b"yy").unwrap();
705        touch(&tied.path().join("big"), 1_000_000);
706        touch(&tied.path().join("small"), 1_000_000);
707        let out = prune_cache(home.path(), tied.path(), 2)
708            .expect("prune")
709            .expect("the cache is free");
710        assert_eq!(out.files, 1, "the big one alone gets under the cap");
711        assert_eq!(out.remaining, 2);
712        assert!(tied.path().join("small").exists());
713    }
714
715    #[test]
716    fn prune_cache_if_over_limit_resolves_the_opt_outs_before_ever_measuring() {
717        let home = tempfile::tempdir().unwrap();
718        let dir = tempfile::tempdir().unwrap();
719        fs::write(dir.path().join("big"), vec![0u8; 10]).unwrap();
720
721        let mut cfg = crate::config::Config::default();
722        cfg.verify.gate = vec![format!(
723            "CARGO_TARGET_DIR={} cargo make check",
724            dir.path().display()
725        )];
726
727        // A cap of `0` is the operator's opt-out: never measured, never
728        // pruned, regardless of what is actually on disk.
729        cfg.disk.cache_limit_bytes = 0;
730        assert_eq!(
731            prune_cache_if_over_limit(&cfg, home.path()).unwrap(),
732            None,
733            "a zero cap must not even look at the directory"
734        );
735        assert!(dir.path().join("big").exists());
736
737        // No `CARGO_TARGET_DIR` in either verify command: nothing to
738        // aggregate, so there is nothing to prune either.
739        let mut no_cache = crate::config::Config::default();
740        no_cache.disk.cache_limit_bytes = 1;
741        assert_eq!(
742            prune_cache_if_over_limit(&no_cache, home.path()).unwrap(),
743            None
744        );
745
746        // Over the cap and configured: pruned exactly like `prune_cache`
747        // itself would.
748        cfg.disk.cache_limit_bytes = 1;
749        let pruned = prune_cache_if_over_limit(&cfg, home.path())
750            .unwrap()
751            .expect("a real cache dir over its cap prunes");
752        assert_eq!(pruned.files, 1);
753        assert!(!dir.path().join("big").exists());
754    }
755
756    /// Pin a file's mtime, so a test asserts the policy and not the runner's
757    /// timestamp granularity.
758    fn touch(path: &Path, secs: u64) {
759        let f = fs::File::options().write(true).open(path).unwrap();
760        f.set_times(fs::FileTimes::new().set_modified(
761            std::time::SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(secs),
762        ))
763        .unwrap();
764    }
765
766    /// The disk-full casualty: a run whose first save left `run.json.tmp` and
767    /// nothing else. It has to be clearable, or the record is permanent.
768    #[test]
769    fn fold_unreadable_clears_a_run_whose_state_never_landed() {
770        let dir = tempfile::tempdir().unwrap();
771        let runs = dir.path().join("runs");
772        let wt = dir.path().join("wt");
773        let id = "20260904-014540-88c0";
774        std::fs::create_dir_all(runs.join(id)).unwrap();
775        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
776
777        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold by id");
778        assert_eq!(removed, vec![format!("runs/{id}")]);
779        assert!(!runs.join(id).exists(), "record gone");
780
781        // And by prefix, the way the deck and the phone address a run.
782        std::fs::create_dir_all(runs.join(id)).unwrap();
783        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
784        assert!(
785            block_on(fold_unreadable(&runs, &wt, "88c0")).is_ok(),
786            "by prefix"
787        );
788
789        // A directory under `runs` that is not a run is never a fold target.
790        std::fs::create_dir_all(runs.join("scratch")).unwrap();
791        assert!(
792            block_on(fold_unreadable(&runs, &wt, "scratch")).is_err(),
793            "a stray directory is not a run"
794        );
795    }
796
797    #[test]
798    fn fold_due_folds_terminal_runs_of_any_schema_but_leaves_genuinely_unreadable_ones() {
799        let dir = tempfile::tempdir().unwrap();
800        let runs = dir.path().join("runs");
801        let wt = dir.path().join("wt");
802        let home = dir.path().to_path_buf();
803        let disk = Disk::default();
804        let now = ts("2026-09-05T00:00:00Z");
805
806        // 1. Runnable (judging): never folded, however old.
807        let judging = "20260801-000000-0001";
808        write_meta(&runs, judging, "judging", "2026-08-01T00:00:00Z");
809
810        // 2. Finished but fresh: grace not elapsed. Within the default 6h
811        //    grace of `now`, so `fold_due` must stop at the freshness check
812        //    and never even reach `read_state` - `write_meta`'s minimal JSON
813        //    would fail that full parse anyway, and this case exists to
814        //    prove freshness is why the run survives, not an accident of the
815        //    fixture being unparseable as a whole `RunState`.
816        let ready_fresh = "20260904-220000-0002";
817        write_meta(&runs, ready_fresh, "ready", "2026-09-04T22:00:00Z");
818
819        // 3. Genuinely unreadable: broken JSON, not merely an unfamiliar
820        //    schema number. Left alone and counted - this is the one case
821        //    automatic housekeeping must never touch (see `fold_due`'s docs);
822        //    discarding it is an explicit operator action, not something a
823        //    background pass does.
824        let garbage = "20260901-000000-0004";
825        std::fs::create_dir_all(runs.join(garbage)).unwrap();
826        std::fs::write(runs.join(garbage).join("run.json"), "not json").unwrap();
827        std::fs::create_dir_all(wt.join("0004")).unwrap();
828
829        // 4. Finished, well past grace, current schema: the ordinary case
830        //    `fold_due` has always acted on.
831        let due_ready = due_run(&runs, &wt, "20260801-000000-ffff", SCHEMA);
832
833        // 5. Finished, well past grace, but written by a schema number this
834        //    build no longer matches - the defect this task exists to fix.
835        //    It still parses cleanly, so only the version number differs, and
836        //    that alone must not block folding.
837        let due_old_schema = due_run(&runs, &wt, "20260801-000000-eeee", SCHEMA - 1);
838
839        let (folded, unreadable) =
840            block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
841        assert_eq!(
842            folded, 2,
843            "both due, parseable runs fold regardless of their schema number"
844        );
845        assert_eq!(
846            unreadable, 1,
847            "only the run with broken JSON counts as unreadable"
848        );
849        assert!(runs.join(judging).exists(), "runnable never folded");
850        assert!(runs.join(ready_fresh).exists(), "fresh never folded");
851        assert!(runs.join(garbage).exists(), "unreadable record kept");
852        assert!(wt.join("0004").exists(), "unreadable worktree kept");
853        assert!(
854            runs.join(&due_ready).exists(),
855            "folding drops worktrees, not the record"
856        );
857        assert!(
858            runs.join(&due_old_schema).exists(),
859            "an old-schema record survives its fold exactly like a current one"
860        );
861        // `graph::fold_run` saves the state it just folded back to disk. That
862        // write must land under this test's own `runs` - the argument it
863        // passed to `fold_due`, not the process-global `run::home` - or a
864        // fold that landed somewhere else entirely would still be counted
865        // above as one of the two `folded` runs.
866        for id in [&due_ready, &due_old_schema] {
867            let saved = read_meta(&runs, id).expect("folded run still parses");
868            assert_ne!(
869                saved.updated_at,
870                ts("2026-08-01T00:00:00Z"),
871                "fold_run must have saved the updated state back through the \
872                 `runs` directory this test passed to fold_due"
873            );
874        }
875    }
876
877    /// `[disk] auto_fold = false` must leave the janitor's fold-and-reclaim
878    /// passes completely inert - a due run's worktree and record both
879    /// survive exactly as if `housekeep` had never run at all. Cache pruning
880    /// is a separate opt-out (`cache_limit_bytes`) and stays disabled here
881    /// too, so this test is only ever about `auto_fold`.
882    #[tokio::test]
883    async fn housekeep_leaves_everything_alone_when_auto_fold_is_disabled() {
884        let dir = tempfile::tempdir().unwrap();
885        let runs = dir.path().join("runs");
886        let wt = dir.path().join("wt");
887        let home = dir.path().to_path_buf();
888        crate::run::set_home(dir.path().to_path_buf());
889
890        let due_id = due_run(&runs, &wt, "20260801-000000-abcd", SCHEMA);
891        std::fs::create_dir_all(wt.join("orphan").join("cand-A")).unwrap();
892
893        let mut cfg = crate::config::Config::default();
894        cfg.disk.auto_fold = false;
895        cfg.disk.cache_limit_bytes = 0;
896
897        let out = housekeep(&cfg, &home, &wt, &dir.path().join("repo"), Timestamp::now()).await;
898
899        assert_eq!(out.folded, 0);
900        assert_eq!(out.unreadable, 0);
901        assert_eq!(out.orphaned_worktrees, 0);
902        assert!(
903            runs.join(&due_id).exists(),
904            "a due run's record survives untouched"
905        );
906        assert!(
907            wt.join("orphan").exists(),
908            "an orphaned worktree survives untouched: the reclaim pass never ran"
909        );
910    }
911
912    #[test]
913    fn fold_orphaned_worktrees_removes_only_worktrees_no_run_claims_and_none_in_flight() {
914        let dir = tempfile::tempdir().unwrap();
915        let runs = dir.path().join("runs");
916        let wt = dir.path().join("wt");
917        let home = dir.path().to_path_buf();
918
919        // A run record exists for this one: its worktree is claimed, not
920        // orphaned, however old the record.
921        write_meta(
922            &runs,
923            "20260801-000000-aaaa",
924            "ready",
925            "2026-08-01T00:00:00Z",
926        );
927        std::fs::create_dir_all(wt.join("aaaa").join("cand-A")).unwrap();
928
929        // No run record at all, and nobody is working on it: this is the
930        // leftover `fold_due` can never see, because it only ever walks
931        // `runs/`.
932        std::fs::create_dir_all(wt.join("bbbb").join("cand-A")).unwrap();
933
934        // No run record either, but a live daemon status names a run with
935        // this short id - the save-timing gap between the daemon claiming a
936        // task and `RunState::new` writing its first `run.json`. Must survive
937        // untouched.
938        std::fs::create_dir_all(wt.join("cccc")).unwrap();
939
940        // Not shaped like a run's short id at all - a scratch directory an
941        // operator or another tool left in the same bay - so it is never a
942        // reclaim target regardless of what runs claim it or not.
943        std::fs::create_dir_all(wt.join("scratch")).unwrap();
944
945        // `now` pushed comfortably past `MIN_ORPHAN_AGE_SECS`, so a zero
946        // grace - the same "always due" escape hatch `due` itself documents
947        // - still reclaims once a worktree is genuinely old, without faking
948        // an mtime: real directory creation just above is already in the
949        // past relative to this `now`, by design rather than by timing.
950        let now = Timestamp::now() + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
951        let mut status = crate::daemon::Status::new();
952        status.current = vec![crate::daemon::Current {
953            task: "20260905-000000-t111".to_owned(),
954            run: "20260905-000000-cccc".to_owned(),
955        }];
956        status.updated_at = now;
957        crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
958
959        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
960        assert_eq!(
961            folded, 1,
962            "only the truly orphaned, idle, bay-shaped worktree is removed"
963        );
964        assert!(wt.join("aaaa").exists(), "claimed by a run record");
965        assert!(!wt.join("bbbb").exists(), "orphaned and idle: reclaimed");
966        assert!(wt.join("cccc").exists(), "a run in flight is never touched");
967        assert!(
968            wt.join("scratch").exists(),
969            "not shaped like a worktree bay, so never a reclaim target"
970        );
971    }
972
973    /// The gap this closes: `Runner::review` (`magi review`) creates the
974    /// worktree with `git worktree add` before `RunState::save` ever writes a
975    /// `run.json`, and that path never runs through the daemon's own `poll`
976    /// loop at all, so `daemon::Status` never names it either. Without a
977    /// grace window, a janitor pass landing in that gap would read the
978    /// worktree as an orphan nothing is waiting on and delete a review still
979    /// being set up.
980    #[test]
981    fn fold_orphaned_worktrees_leaves_a_freshly_created_bay_alone() {
982        let dir = tempfile::tempdir().unwrap();
983        let runs = dir.path().join("runs");
984        let wt = dir.path().join("wt");
985        let home = dir.path().to_path_buf();
986
987        std::fs::create_dir_all(wt.join("dddd").join("under-review")).unwrap();
988
989        let now = Timestamp::now();
990        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 6 * 60 * 60, now));
991        assert_eq!(
992            folded, 0,
993            "too fresh to tell apart from a run still being set up"
994        );
995        assert!(wt.join("dddd").exists());
996    }
997
998    /// A fresh open question on `run`, stored and handed back for assertions.
999    fn open_question(store: &Questions, run: &str) -> crate::ask::Question {
1000        let mut q = crate::ask::Question::new(
1001            run.to_owned(),
1002            "implement".to_owned(),
1003            "impl-A".to_owned(),
1004            "Which storage backend should the cache use?".to_owned(),
1005            String::new(),
1006            vec!["SQLite".to_owned(), "Redis".to_owned()],
1007        );
1008        store.put(&mut q).unwrap();
1009        q
1010    }
1011
1012    /// The exact ghost the phone showed: a run that already finished, with a
1013    /// question its dead seat asked still sitting `open` because it reached
1014    /// that status before `graph::Runner::settle_questions` existed (or
1015    /// missed it in the crash window `daemon::reclaim_orphaned_running`
1016    /// covers). This sweep is the second door to the same fact.
1017    #[test]
1018    fn a_finished_runs_open_question_is_swept_up() {
1019        let dir = tempfile::tempdir().unwrap();
1020        let runs = dir.path().join("runs");
1021        let store = Questions::at(dir.path().join("questions"));
1022
1023        let failed = "20260908-205802-c9eb";
1024        write_meta(&runs, failed, "failed", "2026-09-08T20:58:02Z");
1025        let failed_q = open_question(&store, failed);
1026
1027        let merged = "20260908-205501-ca67";
1028        write_meta(&runs, merged, "merged", "2026-09-08T20:55:01Z");
1029        let merged_q = open_question(&store, merged);
1030
1031        let n = abandon_settled_questions(&store, &runs);
1032        assert_eq!(n, 2, "both dead runs' questions are swept in one pass");
1033
1034        for (id, run) in [(&failed_q.id, failed), (&merged_q.id, merged)] {
1035            let back = store.get(id).unwrap();
1036            assert!(!back.status.open(), "{run} is done; nobody reads an answer");
1037            assert!(back.detail.contains(run), "{}", back.detail);
1038        }
1039    }
1040
1041    #[test]
1042    fn a_still_alive_runs_open_question_survives_the_sweep() {
1043        let dir = tempfile::tempdir().unwrap();
1044        let runs = dir.path().join("runs");
1045        let store = Questions::at(dir.path().join("questions"));
1046
1047        // `Blocked` and `Stalled` are `RunStatus::resumable`: the run can
1048        // still be picked back up, so its question may yet get a real
1049        // answer. A run still mid-competition is even more obviously alive.
1050        for (id, status) in [
1051            ("20260908-000000-b10c", "blocked"),
1052            ("20260908-000000-5ta1", "stalled"),
1053            ("20260908-000000-jud6", "judging"),
1054        ] {
1055            write_meta(&runs, id, status, "2026-09-08T00:00:00Z");
1056            let q = open_question(&store, id);
1057
1058            let n = abandon_settled_questions(&store, &runs);
1059            assert_eq!(n, 0, "{status} run is not done; nothing to sweep");
1060            assert!(
1061                store.get(&q.id).unwrap().status.open(),
1062                "{status} run's question must still be waiting"
1063            );
1064        }
1065    }
1066
1067    #[test]
1068    fn the_sweep_leaves_an_answered_question_and_an_unreadable_run_alone() {
1069        let dir = tempfile::tempdir().unwrap();
1070        let runs = dir.path().join("runs");
1071        let store = Questions::at(dir.path().join("questions"));
1072
1073        // Already decided: a sweep must never revisit it, whatever the run
1074        // that asked went on to become.
1075        let done = "20260908-000000-answ";
1076        write_meta(&runs, done, "failed", "2026-09-08T00:00:00Z");
1077        let mut answered = open_question(&store, done);
1078        answered
1079            .answer(crate::ask::Answer::Choice("SQLite".to_owned()))
1080            .unwrap();
1081        store.put(&mut answered).unwrap();
1082
1083        // No `run.json` at all for this one - deleted, or never landed.
1084        let gone = "20260908-000000-gone";
1085        let orphan = open_question(&store, gone);
1086
1087        assert_eq!(abandon_settled_questions(&store, &runs), 0);
1088        assert_eq!(
1089            store.get(&answered.id).unwrap().status,
1090            crate::ask::QuestionStatus::Answered,
1091            "a real answer is never overwritten by a sweep"
1092        );
1093        assert!(
1094            store.get(&orphan.id).unwrap().status.open(),
1095            "a run this sweep cannot read is left exactly as it was, not guessed at"
1096        );
1097    }
1098
1099    /// A grace of `0` is a legitimate, documented value for the operator's
1100    /// own `Disk::fold_grace_secs` - `due`'s "always due" case - but the
1101    /// freshness check this guards is not that policy, and must not collapse
1102    /// to it: a `0` handed straight through would reclaim a worktree the
1103    /// instant it exists, exactly the race `fold_orphaned_worktrees_leaves_a_
1104    /// freshly_created_bay_alone` exists to rule out, just with the operator
1105    /// having turned the other grace off instead of leaving it at its
1106    /// default.
1107    #[test]
1108    fn fold_orphaned_worktrees_floors_a_zero_grace_at_the_race_safe_minimum() {
1109        let dir = tempfile::tempdir().unwrap();
1110        let runs = dir.path().join("runs");
1111        let wt = dir.path().join("wt");
1112        let home = dir.path().to_path_buf();
1113
1114        std::fs::create_dir_all(wt.join("eeee").join("under-review")).unwrap();
1115
1116        // Too fresh, even with the grace argument at zero.
1117        let now = Timestamp::now();
1118        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1119        assert_eq!(
1120            folded, 0,
1121            "a zero grace must not defeat the race-safety floor"
1122        );
1123        assert!(wt.join("eeee").exists());
1124
1125        // Once genuinely past the floor, a zero grace reclaims it - the
1126        // floor is a minimum, not a replacement policy that never fires.
1127        let later = now + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1128        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, later));
1129        assert_eq!(folded, 1, "old enough now, regardless of the zero grace");
1130        assert!(!wt.join("eeee").exists());
1131    }
1132
1133    #[test]
1134    fn clear_abandoned_active_only_acts_once_dead_and_overrun() {
1135        let dir = tempfile::tempdir().unwrap();
1136        // Harmless if another test in this binary already pinned the global
1137        // home first (see `run::set_home`'s own doc): this test only checks
1138        // the in-memory mutation `clear_abandoned_active` makes, never a
1139        // write that landed under this exact directory.
1140        crate::run::set_home(dir.path().to_path_buf());
1141        let home = dir.path().to_path_buf();
1142        let now = ts("2026-09-14T12:00:00Z");
1143        let overrun_seat = || crate::run::ActiveSeat {
1144            node: "implement".to_owned(),
1145            started_at: now - SignedDuration::new(21_000, 0),
1146            timeout_secs: 3_600,
1147            attempt: 0,
1148            task: None,
1149            command: None,
1150            index: None,
1151            total: None,
1152        };
1153
1154        let mut state = RunState::new(
1155            PathBuf::from("/repo"),
1156            "main".to_owned(),
1157            "abc1234".to_owned(),
1158            "fixture".to_owned(),
1159            crate::config::Config::default(),
1160        );
1161        state.status = RunStatus::Implementing;
1162        state.active.insert("impl-A".to_owned(), overrun_seat());
1163
1164        // A seat still within its own budget: not provably dead yet, so this
1165        // must change nothing.
1166        let mut fresh = state.clone();
1167        fresh.active.insert(
1168            "impl-B".to_owned(),
1169            crate::run::ActiveSeat {
1170                node: "implement".to_owned(),
1171                started_at: now,
1172                timeout_secs: 3_600,
1173                attempt: 0,
1174                task: None,
1175                command: None,
1176                index: None,
1177                total: None,
1178            },
1179        );
1180        assert!(!clear_abandoned_active(&mut fresh, &home, now).unwrap());
1181        assert!(!fresh.active.is_empty());
1182        assert_eq!(fresh.status, RunStatus::Implementing);
1183
1184        let store = Questions::at(home.join("questions"));
1185        let q = open_question(&store, &state.id);
1186
1187        assert!(clear_abandoned_active(&mut state, &home, now).unwrap());
1188        assert!(state.active.is_empty());
1189        assert_eq!(state.status, RunStatus::Failed);
1190        assert!(
1191            !store.get(&q.id).unwrap().status.open(),
1192            "the abandoned seat's own open question must not keep badging the \
1193             operator until some later daemon startup notices it"
1194        );
1195    }
1196
1197    /// Write a whole `run.json` that magi can read, over the given state.
1198    fn write_meta(runs: &Path, id: &str, status: &str, updated_at: &str) {
1199        let day = &updated_at[..10];
1200        std::fs::create_dir_all(runs.join(id)).unwrap();
1201        let body = format!(
1202            r#"{{"schema": {SCHEMA}, "id": "{id}", "repo": "/nonexistent/repo", "base_branch": "main", "base_commit": "0000000000000000000000000000000000000000", "instruction": "", "created_at": "{day}T00:00:00Z", "updated_at": "{updated_at}", "status": "{status}", "seed": 1}}"#
1203        );
1204        std::fs::write(runs.join(id).join("run.json"), body).unwrap();
1205    }
1206
1207    /// Write a fully-formed, `Ready`, well-past-grace `run.json` tagged with
1208    /// an arbitrary schema number - so a test can write one this build's own
1209    /// `RunState::new` could never produce on its own. Returns the id.
1210    ///
1211    /// `wt` becomes this run's `graph.worktree_root`: left at the config
1212    /// default, `RunState::worktree_root` falls through to
1213    /// `run::default_worktree_root` - the operator's real `~/wt/magi` - and
1214    /// `graph::fold_run`'s second sweep would then `read_dir` and remove
1215    /// worktrees there instead of anything this test owns.
1216    fn due_run(runs: &Path, wt: &Path, id: &str, schema: u32) -> String {
1217        let mut config = crate::config::Config::default();
1218        config.graph.worktree_root = Some(wt.to_path_buf());
1219        let mut state = RunState::new(
1220            PathBuf::from("/nonexistent/repo"),
1221            "main".to_owned(),
1222            "0000000000000000000000000000000000000000".to_owned(),
1223            String::new(),
1224            config,
1225        );
1226        state.id = id.to_owned();
1227        state.status = RunStatus::Ready;
1228        state.updated_at = ts("2026-08-01T00:00:00Z");
1229        let mut value = serde_json::to_value(&state).unwrap();
1230        value["schema"] = serde_json::json!(schema);
1231        std::fs::create_dir_all(runs.join(id)).unwrap();
1232        std::fs::write(
1233            runs.join(id).join("run.json"),
1234            serde_json::to_string_pretty(&value).unwrap(),
1235        )
1236        .unwrap();
1237        id.to_owned()
1238    }
1239}