Skip to main content

magi/
clean.rs

1//! The disk janitor: finished runs get their worktrees folded, worktrees whose
2//! run record is already gone get reclaimed too, and the shared build cache is
3//! pruned to its cap.
4//!
5//! A run's state being written by an older schema is not the same thing as it
6//! being unreadable, and this module used to conflate the two: [`fold_due`]
7//! treated any `run.json` its version check rejected exactly like one that
8//! failed to parse at all, so a single schema bump silently stopped every
9//! automatic fold in the fleet the moment it shipped, and did so with no
10//! counter and no log line to say so. A record magi genuinely cannot parse —
11//! missing fields, broken JSON, a schema newer than this build has ever heard
12//! of — is still left alone here, still counted in
13//! [`Housekeeping::unreadable`], and still only ever removed by an explicit
14//! operator action (`magi fold`, or the equivalent phone route). One written
15//! by a schema this build merely disagrees with the *meaning* of is not that:
16//! as long as it still parses, folding proceeds regardless of the number in
17//! its `schema` field.
18//!
19//! Everything policy-shaped — which statuses are foldable, how long a finished
20//! run is left alone, whether the cache is over its limit — is a pure function
21//! injected with numbers, so nothing here has to ask the operating system to
22//! be testable. The only I/O is the removal itself.
23
24use std::collections::BTreeSet;
25use std::path::{Path, PathBuf};
26
27use anyhow::{Context as _, Result, bail};
28use jiff::{SignedDuration, Timestamp};
29use serde::Deserialize;
30
31use crate::ask::Questions;
32use crate::config::Disk;
33use crate::run::{RunState, RunStatus, SCHEMA, short_of};
34
35use crate::disk::{Prune, dir_size};
36
37/// What one janitor pass did, for the caller's log line.
38#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
39pub struct Housekeeping {
40    /// Runs folded (worktrees dropped).
41    pub folded: usize,
42    /// Runs [`fold_due`] left alone because their `run.json` genuinely could
43    /// not be read - missing, broken JSON, a schema this build has never
44    /// heard of - as opposed to one merely written by a different schema
45    /// number, which is folded like any other (see the module docs). This was
46    /// defined but never incremented for a long stretch of this module's
47    /// history, which is exactly how 90 of 93 runs sat unfolded on one
48    /// operator's machine with nothing anywhere saying why: every one of them
49    /// was misclassified as unreadable by a schema check that has since been
50    /// narrowed to only the runs that actually are.
51    pub unreadable: usize,
52    /// Worktrees under the worktree bay reclaimed because no run record in
53    /// `runs/` claims them anymore (see [`fold_orphaned_worktrees`]).
54    pub orphaned_worktrees: usize,
55    /// Files dropped from the shared cache.
56    pub cache_files: usize,
57    /// Bytes freed from the shared cache.
58    pub cache_freed: u64,
59    /// Open questions abandoned because the run that asked them has already
60    /// settled where nothing is coming back to read an answer.
61    pub questions_abandoned: usize,
62    /// Runs found blocked on a merge magi never recorded — a pull request the
63    /// operator merged by hand while `land::land` never got as far as
64    /// opening one itself — and corrected automatically. See
65    /// [`reconcile_external_merges`].
66    pub external_merges_recorded: usize,
67}
68
69/// Run the janitor: fold due runs, reclaim orphaned worktrees, prune stale
70/// worktree registrations, then prune the cache if it is over its cap.
71///
72/// Every part is best-effort; a jammed cache lock or a run whose worktree
73/// another borrower holds must not stop the rest. Errors are reported through
74/// `tracing::warn` - this is housekeeping, and the daemon keeps serving
75/// either way.
76pub async fn housekeep(
77    cfg: &crate::config::Config,
78    home: &Path,
79    worktrees_root: &Path,
80    repo: &Path,
81    now: Timestamp,
82) -> Housekeeping {
83    let mut out = Housekeeping::default();
84    if cfg.disk.auto_fold {
85        let runs = home.join("runs");
86        match fold_due(&runs, home, worktrees_root, &cfg.disk, now).await {
87            Ok((folded, unreadable)) => {
88                out.folded = folded;
89                out.unreadable = unreadable;
90            }
91            Err(e) => {
92                tracing::warn!("housekeep: fold due runs: {e:#}");
93                // Stable wording: the reason varies per pass, and a changed
94                // message would relight the bell on every retry.
95                crate::notices::raise_in(
96                    home,
97                    crate::notices::Notice::warn(
98                        "housekeep:fold",
99                        "Automatic cleanup of finished runs failed; disk usage may keep growing.",
100                    ),
101                );
102            }
103        }
104        out.orphaned_worktrees =
105            fold_orphaned_worktrees(&runs, worktrees_root, home, cfg.disk.fold_grace_secs, now)
106                .await;
107        // Best-effort in the same sense as everything else here: a repository
108        // this janitor pass has nothing to do with (or none at all, in a unit
109        // test) must not turn a `warn` into a reason to skip the rest.
110        if let Err(e) = crate::git::worktree_prune(repo).await {
111            tracing::warn!("housekeep: prune worktree registrations: {e:#}");
112        }
113        out.external_merges_recorded = reconcile_external_merges(&runs, home, &cfg.disk, now).await;
114    }
115    match prune_cache_if_over_limit(cfg, home) {
116        Ok(Some(pruned)) => {
117            out.cache_files = pruned.files;
118            out.cache_freed = pruned.freed;
119        }
120        Ok(None) => {}
121        Err(e) => tracing::warn!("housekeep: prune cache: {e:#}"),
122    }
123    // Unconditional, unlike the two passes above: this is not a disk policy
124    // with a cap or an opt-out, it is closing a gap `graph::Runner` itself
125    // cannot - a run that reached `Merged`/`Ready`/`Failed` before this
126    // cleanup existed, or whose process died between saving that status and
127    // abandoning the question it leaves behind (see `Runner::settle_questions`).
128    // Left alone, that question sits `open` forever: the owner's badge,
129    // banner and title all keep counting a decision nobody is left to read.
130    out.questions_abandoned =
131        abandon_settled_questions(&Questions::at(home.join("questions")), &home.join("runs"));
132    out
133}
134
135/// Abandon every open question whose run has already settled into a status
136/// nothing comes back from, worded with what the run became - the same
137/// cleanup `graph::Runner::settle_questions` runs the moment `status` lands
138/// there, for questions that missed it.
139///
140/// Scans questions rather than runs: the open list is normally short, and a
141/// run that never asked anything costs nothing here. A run this cannot read,
142/// deleted or written by a schema this build does not speak, is left alone
143/// the same as everywhere else in this module; the question stays open
144/// rather than guessed at.
145pub fn abandon_settled_questions(store: &Questions, runs: &Path) -> usize {
146    let waiting_on: BTreeSet<String> = store
147        .list()
148        .into_iter()
149        .filter(|q| q.status.open())
150        .map(|q| q.run)
151        .collect();
152    let mut abandoned = 0;
153    for run in waiting_on {
154        let Ok(meta) = read_meta(runs, &run) else {
155            continue;
156        };
157        match store.settle_run(&run, meta.status) {
158            Ok(n) => abandoned += n,
159            Err(e) => tracing::warn!("housekeep: abandon questions for {run}: {e:#}"),
160        }
161    }
162    abandoned
163}
164
165/// Fold every run that is finished, older than the grace period, and not being
166/// worked on; return `(folded, unreadable)`.
167///
168/// A run whose `run.json` genuinely cannot be parsed — missing fields, broken
169/// JSON, a schema newer than this build has ever heard of — is left exactly
170/// as it is. Automatic housekeeping cannot tell a mid-write file from one that
171/// will never parse again, and `<home>/runs/<id>/` is the evidence `magi
172/// stats` and the deck read; when unsure whether it is safe to touch, the
173/// janitor keeps rather than deletes (see the module docs). Discarding a
174/// record this unreadable is an explicit operator action (`magi fold`, or the
175/// equivalent phone route), never something that happens unattended. Every
176/// such skip is counted in the returned `unreadable` and logged through
177/// `tracing::warn` with the parse failure that caused it - silence here is
178/// exactly the failure mode that let 90 of 93 runs sit unfolded with nothing
179/// to show for it.
180///
181/// A run merely written by a *different* schema number is not unreadable: as
182/// long as `run.json` still parses, it folds like any other terminal run (see
183/// the module docs for why the two are different questions).
184///
185/// Runnable statuses and runs newer than the grace period are also left
186/// alone; folding them would throw away work that is still the answer to
187/// somebody's question. `Merged` runs forget their winner's worktree (the
188/// merge already landed it); `Ready` and `Failed` runs keep it.
189pub async fn fold_due(
190    runs: &Path,
191    home: &Path,
192    _worktrees_root: &Path,
193    disk: &Disk,
194    now: Timestamp,
195) -> Result<(usize, usize)> {
196    let mut folded = 0usize;
197    let mut unreadable = 0usize;
198    let mut ids: Vec<String> = std::fs::read_dir(runs)
199        .into_iter()
200        .flatten()
201        .flatten()
202        .filter(|e| e.path().join("run.json").is_file())
203        .map(|e| e.file_name().to_string_lossy().into_owned())
204        .collect();
205    ids.sort_unstable();
206    for id in ids {
207        if crate::daemon::is_working_on(home, &id, now) {
208            continue;
209        }
210        let meta = match read_meta(runs, &id) {
211            Ok(meta) => meta,
212            Err(e) => {
213                unreadable += 1;
214                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
215                continue;
216            }
217        };
218        if meta.status.resumable() || !due(now, meta.updated_at, disk.fold_grace_secs) {
219            continue;
220        }
221        // `read_meta` already proved the file parses; `read_state` asks for
222        // the rest of the fields `graph::fold_run` needs (worktree paths,
223        // candidates, tally). A schema mismatch alone does not fail this -
224        // see the module docs - so reaching `Err` here means the JSON itself
225        // is broken in a way `read_meta` did not exercise, which is rare but
226        // not impossible (a body truncated between the two fields it reads
227        // and the rest). That must not cost every other run its turn through
228        // this loop, so it is a skip, not a `?`.
229        let mut state = match read_state(runs, &id) {
230            Ok(state) => state,
231            Err(e) => {
232                unreadable += 1;
233                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
234                continue;
235            }
236        };
237        if state.schema != SCHEMA {
238            tracing::info!(
239                "housekeep: run {id} was written by schema {} (this build speaks {SCHEMA}); \
240                 folding it anyway",
241                state.schema
242            );
243        }
244        // `Merged`'s winner branch is safe to drop because it already landed
245        // in the real target branch; `Superseded`'s winner branch is safe to
246        // drop for a different reason - a later attempt at the same task is
247        // what actually landed (or didn't), so *this* run's own winner is
248        // never going to be merged by anyone. Every other terminal status
249        // keeps the winner: `Ready`/`Blocked`/`Stalled`/`Failed`/`VerifiedNoop`
250        // may still have a human's decision pending on that exact branch.
251        // `AlreadyInBase` is the same: the change is on the base under other
252        // commit ids, so the branch holds nothing anyone still needs.
253        let drop_winner = matches!(
254            state.status,
255            RunStatus::Merged | RunStatus::Superseded | RunStatus::AlreadyInBase
256        );
257        // One run's fold must not cost every later run its turn. A worktree
258        // another borrower holds, a branch git refuses to delete, a repository
259        // that has since moved: each is a reason this run cannot be folded
260        // now, and none is a reason to stop the pass. Left unfolded, it is
261        // simply due again next time; a `?` here stopped automatic folding
262        // permanently at the first such run (finding R3-1-1 of run 51a3).
263        match crate::graph::fold_run(&mut state, drop_winner, home).await {
264            Ok(_) => folded += 1,
265            Err(e) => tracing::warn!("housekeep: fold {id}: {e:#}"),
266        }
267    }
268    Ok((folded, unreadable))
269}
270
271/// Is `updated` old enough, measured against `now`, that the run may fold?
272///
273/// Pure; the janitor compares against wallclock, tests inject both sides. The
274/// comparison is strict, so a run exactly at the edge of its grace period is
275/// left alone one more pass — the same convention as [`crate::disk::over_limit`].
276pub fn due(now: Timestamp, updated: Timestamp, grace_secs: u64) -> bool {
277    now.duration_since(updated) > SignedDuration::new(grace_secs as i64, 0)
278}
279
280/// The two fields the janitor decides on, read with a serde that tolerates
281/// everything else about the run being unreadable.
282#[derive(Deserialize)]
283struct Meta {
284    status: RunStatus,
285    updated_at: Timestamp,
286}
287
288/// Read `status` and `updated_at` straight off the state file, asking for
289/// nothing else. `Err` when the file is missing, not parseable, or a status in
290/// a version this build does not speak - all of which mean "unreadable".
291fn read_meta(runs: &Path, id: &str) -> Result<Meta> {
292    let path = runs.join(id).join("run.json");
293    let body =
294        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
295    let meta: Meta =
296        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
297    Ok(meta)
298}
299
300/// Runs checked against GitHub in one janitor pass, at most.
301///
302/// A fleet with many stuck runs must not turn every idle tick into a burst of
303/// `gh pr list` calls; a run left unchecked this pass is checked again next
304/// pass, same as an unfolded due run is.
305const MAX_EXTERNAL_MERGE_CHECKS_PER_PASS: usize = 5;
306
307/// The fields [`reconcile_external_merges`] filters on before paying for a
308/// full [`RunState`] parse or a `gh` round trip.
309#[derive(Deserialize)]
310struct ExternalMergeMeta {
311    status: RunStatus,
312    updated_at: Timestamp,
313    #[serde(default)]
314    merge: Option<crate::run::MergeOutcome>,
315}
316
317fn read_external_merge_meta(runs: &Path, id: &str) -> Result<ExternalMergeMeta> {
318    let path = runs.join(id).join("run.json");
319    let body =
320        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
321    let meta: ExternalMergeMeta =
322        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
323    Ok(meta)
324}
325
326/// Is a run worth asking GitHub about?
327///
328/// Pure, so the filter itself is assertable without a fixture that writes
329/// `run.json` files or spawns `gh`. Narrower than "blocked": `merge_recorded`
330/// excludes a run `land::land` already finished landing (successfully or
331/// not) - that path already carries its own record of what happened - and
332/// `due` gives an operator who is about to fix a run by hand a grace window
333/// before the janitor starts asking the same question about it.
334pub fn eligible_for_external_merge_check(
335    status: RunStatus,
336    merge_recorded: bool,
337    updated_at: Timestamp,
338    now: Timestamp,
339    grace_secs: u64,
340) -> bool {
341    status == RunStatus::Blocked && !merge_recorded && due(now, updated_at, grace_secs)
342}
343
344/// Rotate `ids` so a fixed per-pass cap does not always land on the same
345/// prefix of the list, starting at `start` (already reduced modulo nothing -
346/// callers pass whatever a persisted cursor last recorded, and this takes it
347/// modulo `ids.len()` itself).
348///
349/// Pure: the offset is the caller's problem, not wall-clock time. An earlier
350/// version derived `start` from `now.as_second() % ids.len()`, which looked
351/// independent of any state to persist but is not actually independent of
352/// how often the janitor runs - a poll interval and a pass's own duration
353/// that alias onto the same handful of `now.as_second()` values (say, a
354/// ten-second poll and a pass that finishes in under a second) revisit only
355/// those same few residues forever and never advance into the rest of the
356/// list at all. [`reconcile_external_merges`] instead persists a cursor
357/// across passes ([`read_external_merge_cursor`] /
358/// [`write_external_merge_cursor`]) and advances it by exactly how many ids
359/// this pass actually looked at, which is the only quantity that is
360/// guaranteed to move forward pass over pass regardless of timing.
361fn rotate_from(ids: &[String], start: usize) -> Vec<String> {
362    if ids.is_empty() {
363        return Vec::new();
364    }
365    let start = start % ids.len();
366    ids[start..]
367        .iter()
368        .chain(ids[..start].iter())
369        .cloned()
370        .collect()
371}
372
373/// Where [`reconcile_external_merges`] remembers how far it got, so a fixed
374/// per-pass cap advances through a fleet's eligible runs pass over pass
375/// instead of camping on whichever ones happen to sort first.
376const EXTERNAL_MERGE_CURSOR_FILE: &str = "external-merge-cursor";
377
378/// `0` for anything this cannot read as a plain number - a first run, a
379/// corrupt or missing file, a build that has never written one - which is
380/// exactly as good a starting point as any: the cursor's whole job is to
381/// keep moving, not to encode any particular position as meaningful.
382fn read_external_merge_cursor(home: &Path) -> usize {
383    std::fs::read_to_string(home.join(EXTERNAL_MERGE_CURSOR_FILE))
384        .ok()
385        .and_then(|s| s.trim().parse().ok())
386        .unwrap_or(0)
387}
388
389/// Best-effort, like every other write in this module's sweep: a failure to
390/// persist the cursor costs the fleet one pass's worth of forward progress
391/// on this sweep, never the run whose housekeeping it is racing to keep up
392/// with.
393fn write_external_merge_cursor(home: &Path, cursor: usize) {
394    if let Err(e) = std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), cursor.to_string()) {
395        tracing::warn!("housekeep: persist the external-merge sweep's cursor: {e:#}");
396    }
397}
398
399/// Close the gap `magi fold --merged` exists for, without waiting for an
400/// operator to notice and go find the pull request URL: a run that stopped
401/// `Blocked` with no `merge` recorded may have been merged anyway, by hand,
402/// on a pull request `land::land` never opened or never got to observe as
403/// merged. `land::find_external_merge` asks GitHub about the run's own
404/// winning branch, and only a pull request it can uniquely tie back to this
405/// run is ever acted on - see that function's own doc for what "uniquely"
406/// excludes.
407///
408/// A run this finds and fixes goes through
409/// [`crate::land::correct_confirmed_external_merge`] - the same correction
410/// `magi fold --merged` performs, minus the same-repo guard that command
411/// needs and this loop doesn't (see that function's own doc: the URL here
412/// was never operator-supplied, it came from asking `state.repo`'s own
413/// remote) - then through [`crate::graph::fold_run`] so it stops holding
414/// worktrees the moment it stops needing them. A run this cannot decide
415/// about - `gh` unreachable, more than one candidate pull
416/// request, nothing found at all - is left exactly as it is; only an error
417/// asking GitHub raises a notice, since "nothing found" is the ordinary,
418/// expected shape of a run that really is just blocked.
419async fn reconcile_external_merges(runs: &Path, home: &Path, disk: &Disk, now: Timestamp) -> usize {
420    let mut ids: Vec<String> = std::fs::read_dir(runs)
421        .into_iter()
422        .flatten()
423        .flatten()
424        .filter(|e| e.path().join("run.json").is_file())
425        .map(|e| e.file_name().to_string_lossy().into_owned())
426        .collect();
427    ids.sort_unstable();
428
429    // Every eligible id first, cheaply (a `Meta` read, no `gh`), then rotated
430    // before the cap is applied. Capping the sorted order directly would
431    // always land on the same lexicographic prefix - a fleet's oldest run ids
432    // - so a handful of long-blocked runs that never turn out to be merged
433    // would starve every id after them of a `gh` check, forever.
434    let mut eligible: Vec<String> = Vec::new();
435    for id in ids {
436        if crate::daemon::is_working_on(home, &id, now) {
437            continue;
438        }
439        let meta = match read_external_merge_meta(runs, &id) {
440            Ok(meta) => meta,
441            // Already counted as unreadable by `fold_due`'s own pass; not
442            // worth a second warning for the same file.
443            Err(_) => continue,
444        };
445        if eligible_for_external_merge_check(
446            meta.status,
447            meta.merge.is_some(),
448            meta.updated_at,
449            now,
450            disk.fold_grace_secs,
451        ) {
452            eligible.push(id);
453        }
454    }
455
456    if eligible.is_empty() {
457        return 0;
458    }
459    let cursor = read_external_merge_cursor(home);
460    let window: Vec<String> = rotate_from(&eligible, cursor)
461        .into_iter()
462        .take(MAX_EXTERNAL_MERGE_CHECKS_PER_PASS)
463        .collect();
464    // Advances by exactly how many ids this pass looked at, regardless of
465    // what came of looking - a run that turned out not to be merged after
466    // all must not be revisited before every other eligible run has had its
467    // own turn.
468    write_external_merge_cursor(home, (cursor + window.len()) % eligible.len());
469
470    let mut reconciled = 0usize;
471    for id in window {
472        let mut state = match read_state(runs, &id) {
473            Ok(state) => state,
474            Err(_) => continue,
475        };
476        match crate::land::find_external_merge(&state).await {
477            Ok(Some(found)) => {
478                match crate::land::correct_confirmed_external_merge(&mut state, &found.url).await {
479                    Ok(_) => {
480                        if let Err(e) = crate::graph::fold_run(&mut state, true, home).await {
481                            tracing::warn!(
482                                "housekeep: fold {id} after recording its external merge: {e:#}"
483                            );
484                        }
485                        reconciled += 1;
486                    }
487                    Err(e) => tracing::warn!(
488                        "housekeep: record external merge {} for {id}: {e:#}",
489                        found.url
490                    ),
491                }
492            }
493            Ok(None) => {}
494            Err(e) => {
495                tracing::warn!("housekeep: check external merge for {id}: {e:#}");
496                crate::notices::raise_in(
497                    home,
498                    crate::notices::Notice::warn(
499                        &format!("merged-unrecorded:{id}"),
500                        format!(
501                            "Run {id} is blocked with no recorded merge, and checking GitHub \
502                             for a merge failed; check by hand."
503                        ),
504                    ),
505                );
506            }
507        }
508    }
509    reconciled
510}
511
512/// Read a whole run state from a runs directory, for folding only.
513///
514/// Deliberately more permissive than [`RunState::load`], which this does not
515/// call: `load` backs `--resume` and every hand-driven command, where a
516/// schema this build disagrees with the *meaning* of must refuse outright
517/// rather than resume a review round or a tally against stale semantics
518/// (`RunState::SCHEMA`'s own docs list what has changed meaning at each
519/// bump). Folding recomputes nothing - it only reads worktree paths, branch
520/// names and a tally winner off the struct to remove them - so an old
521/// schema's values are exactly as good here as a current one's; every schema
522/// bump so far has only ever added a field or a variant, never repurposed an
523/// existing one, and serde already fills an added field's default when an
524/// older record has nothing to say about it. What this cannot tolerate, and
525/// what still surfaces as an `Err`, is `run.json` failing to parse at all.
526fn read_state(runs: &Path, id: &str) -> Result<RunState> {
527    let path = runs.join(id).join("run.json");
528    let body =
529        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
530    let state: RunState =
531        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
532    Ok(state)
533}
534
535/// Remove a run that cannot be read: its state directory under `runs` and its
536/// worktree directory under `worktrees_root`.
537///
538/// The state file is the only record of a run's repository and branches, so a
539/// run this unreadable is discarded at the filesystem level - there is no
540/// candidate list to fold first. The worktrees live under
541/// [`crate::run::default_worktree_root`] unless the run's config relocated
542/// them, which an unreadable run cannot tell us; the default location is
543/// removed, and anything the run placed elsewhere is a leftover for whoever
544/// knows where it went.
545///
546/// Deleting a worktree directory by hand leaves its registration in git, and a
547/// registered path cannot be re-`worktree add`-ed until it is pruned - so every
548/// worktree is unregistered from its repository first, best-effort, via the
549/// `gitdir:` link git keeps inside the directory.
550pub async fn fold_unreadable(runs: &Path, worktrees_root: &Path, id: &str) -> Result<Vec<String>> {
551    let resolved = resolve_id_path(runs, id)?;
552    let mut removed = Vec::new();
553    let run_dir = runs.join(&resolved);
554    if run_dir.exists() {
555        std::fs::remove_dir_all(&run_dir)
556            .with_context(|| format!("remove {}", run_dir.display()))?;
557        removed.push(format!("runs/{resolved}"));
558    }
559    let wt = worktrees_root.join(short_of(&resolved));
560    if wt.exists() {
561        crate::git::remove_worktree_from_linked(&wt).await;
562        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
563            crate::git::remove_worktree_from_linked(&e.path()).await;
564        }
565        std::fs::remove_dir_all(&wt).with_context(|| format!("remove {}", wt.display()))?;
566        removed.push(wt.to_string_lossy().into_owned());
567    }
568    Ok(removed)
569}
570
571/// Reclaim worktrees under `worktrees_root` that no run record in `runs`
572/// claims anymore, and return how many were removed.
573///
574/// [`fold_due`] only ever sees a worktree by walking `runs/` first, so a
575/// worktree whose run record is already gone — `magi run rm`, or a record
576/// deleted before its worktree — never enters that loop at all: nothing there
577/// is looking for it. This walks the worktree bay directly instead, and
578/// removes any `<short>` directory that no run id maps to.
579///
580/// Two things must never happen, and this checks both before ever touching a
581/// directory:
582///
583/// - **A worktree bay is never the only kind of thing under `worktrees_root`,
584///   and this must not assume it is.** A hand-placed scratch directory, or
585///   anything else an operator or another tool left in the same bay, has the
586///   same "no run claims it" shape as a genuine orphan but is not one -
587///   [`looks_like_a_worktree_bay`] is the same tag shape [`crate::run::is_run_id`]
588///   already requires of a real run's short id, and anything else is left
589///   alone regardless of what else is true about it.
590/// - **A worktree that was only just created might not have a `run.json` yet
591///   for a reason that has nothing to do with being orphaned.** `Runner::start`
592///   and `Runner::review` both create the worktree before the first
593///   `RunState::save` lands, and that gap - several `git` subprocesses wide -
594///   is invisible to [`crate::daemon::is_working_on_short`] whenever the run
595///   is not being driven through this daemon's own `poll` loop at all (a
596///   `magi review` invocation, for one). A directory whose own modification
597///   time is within `grace_secs` of `now` is left alone on that basis alone,
598///   the same margin [`fold_due`] gives a run before treating it as truly
599///   finished - long enough that no realistic gap between a `worktree add`
600///   and its `run.json` could ever be mistaken for one.
601///
602/// The one failure this must never cause is deleting the worktree of a run
603/// that is genuinely in flight. [`crate::daemon::is_working_on_short`] is the
604/// same liveness check [`fold_due`] trusts everywhere else in this module,
605/// checked by short id because there is no full id to compare here; when it
606/// cannot tell, this leaves the directory alone. Best-effort like the rest of
607/// housekeeping: one directory git or the filesystem refuses to give up is a
608/// `tracing::warn`, not a reason to abandon the rest of the pass.
609pub async fn fold_orphaned_worktrees(
610    runs: &Path,
611    worktrees_root: &Path,
612    home: &Path,
613    grace_secs: u64,
614    now: Timestamp,
615) -> usize {
616    let known: std::collections::HashSet<String> = std::fs::read_dir(runs)
617        .into_iter()
618        .flatten()
619        .flatten()
620        .map(|e| e.file_name().to_string_lossy().into_owned())
621        .filter(|name| crate::run::is_run_id(name))
622        .map(|id| short_of(&id).to_owned())
623        .collect();
624
625    let mut folded = 0usize;
626    for entry in std::fs::read_dir(worktrees_root)
627        .into_iter()
628        .flatten()
629        .flatten()
630    {
631        if !entry.path().is_dir() {
632            continue;
633        }
634        let short = entry.file_name().to_string_lossy().into_owned();
635        if !looks_like_a_worktree_bay(&short) {
636            continue;
637        }
638        if known.contains(&short) || crate::daemon::is_working_on_short(home, &short, now) {
639            continue;
640        }
641        let wt = entry.path();
642        if !stale_enough(&wt, grace_secs, now) {
643            continue;
644        }
645        crate::git::remove_worktree_from_linked(&wt).await;
646        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
647            crate::git::remove_worktree_from_linked(&e.path()).await;
648        }
649        match std::fs::remove_dir_all(&wt) {
650            Ok(()) => folded += 1,
651            Err(e) => tracing::warn!(
652                "housekeep: remove orphaned worktree {}: {e:#}",
653                wt.display()
654            ),
655        }
656    }
657    folded
658}
659
660/// Does `name` have the shape a run's own worktree bay is named with: the
661/// same 4-character alphanumeric tag [`crate::run::is_run_id`] requires of a
662/// full id's trailing block (see [`short_of`])?
663///
664/// Anything else under `worktrees_root` is not a bay this function reclaims
665/// at all, claimed or not - answering "does a run claim this?" about a
666/// directory that was never a run's worktree in the first place is exactly
667/// the wrong question to ask before deleting it.
668fn looks_like_a_worktree_bay(name: &str) -> bool {
669    name.len() == 4 && name.bytes().all(|b| b.is_ascii_alphanumeric())
670}
671
672/// Is `dir`'s own modification time old enough, against `grace_secs` and
673/// `now`, that its emptiness of a run record can be trusted rather than
674/// caught mid-creation?
675///
676/// A directory this pass cannot stat at all - a race with its own removal, a
677/// permission error - is treated as not yet stale: unreadable metadata is not
678/// evidence of anything, and the janitor already keeps rather than deletes
679/// whenever it cannot tell (see the module docs).
680///
681/// `grace_secs` is floored at [`MIN_ORPHAN_AGE_SECS`] regardless of what the
682/// caller passes: `0` is a documented, legitimate value for
683/// [`crate::config::Disk::fold_grace_secs`] (`due`'s own "always due" case),
684/// because that grace answers a policy question the operator owns - how long
685/// a *known, finished* run's worktree lingers before cleanup. Whether an
686/// orphan worktree is actually a race with `Runner::review`'s `git worktree
687/// add` landing before its `run.json` is not a policy question, and must not
688/// collapse to zero just because the operator turned the other grace off -
689/// that would defeat the very check meant to catch it.
690fn stale_enough(dir: &Path, grace_secs: u64, now: Timestamp) -> bool {
691    let Ok(modified) = std::fs::metadata(dir).and_then(|m| m.modified()) else {
692        return false;
693    };
694    let Ok(ts) = Timestamp::try_from(modified) else {
695        return false;
696    };
697    due(now, ts, grace_secs.max(MIN_ORPHAN_AGE_SECS))
698}
699
700/// The floor under [`stale_enough`]'s grace, independent of
701/// [`crate::config::Disk::fold_grace_secs`].
702///
703/// Ample next to the race it guards: the gap between `Runner::start` or
704/// `Runner::review` creating a worktree and the first `RunState::save`
705/// landing is a handful of `git` subprocess calls, not minutes - but the
706/// janitor cannot tell "still mid-setup" from "orphaned" by any other signal
707/// for a run that never registers with `daemon::Status` at all (a `magi
708/// review` invocation, for one), so this is generous on purpose rather than
709/// tuned to the observed case.
710const MIN_ORPHAN_AGE_SECS: u64 = 5 * 60;
711
712/// Resolve an id or prefix against an explicit runs directory, exactly the way
713/// [`crate::run::resolve_id`] does against the global home.
714fn resolve_id_path(runs: &Path, prefix: &str) -> Result<String> {
715    // Keyed on the directory, not on a readable state file: the record this
716    // route exists to remove may be a lone `run.json.tmp` from a save that
717    // ran out of disk, and that is precisely the one a human needs a way to
718    // clear (see `crate::run::list_ids`).
719    if runs.join(prefix).is_dir() && crate::run::is_run_id(prefix) {
720        return Ok(prefix.to_owned());
721    }
722    let mut hits: Vec<String> = Vec::new();
723    for e in std::fs::read_dir(runs).into_iter().flatten().flatten() {
724        if !e.path().is_dir() {
725            continue;
726        }
727        let id = e.file_name().to_string_lossy().into_owned();
728        if crate::run::is_run_id(&id) && (id.starts_with(prefix) || id.ends_with(prefix)) {
729            hits.push(id);
730        }
731    }
732    match hits.len() {
733        1 => Ok(hits.into_iter().next().expect("exactly one hit")),
734        0 => bail!("no run matches `{prefix}`"),
735        _ => bail!(
736            "`{prefix}` matches {} runs: {}",
737            hits.len(),
738            hits.join(", ")
739        ),
740    }
741}
742
743/// `magi fold`'s recovery path for a run whose worktrees are already gone —
744/// so [`crate::graph::fold_run`] removed nothing — but whose `run.json` still
745/// lists active seats nobody is left to answer for: no live daemon claims the
746/// run, and every one of those seats has overrun its own timeout (see
747/// [`RunState::active_all_overrun`]). Clearing them and failing the run is
748/// what lets it be deleted afterward — [`RunState::ensure_can_delete`] only
749/// ever checks whether a live daemon is working on the run and whether its
750/// candidates are folded, not `status`, but a run stuck `implementing`
751/// forever with an empty worktree still reads as unresolved everywhere else
752/// (`magi show`, the deck, the phone) until this runs.
753///
754/// Returns `false` without changing anything when a live daemon still claims
755/// the run, or when some active seat has not actually overrun its budget yet
756/// — a run that is merely between waves must never be guessed at.
757pub fn clear_abandoned_active(state: &mut RunState, home: &Path, now: Timestamp) -> Result<bool> {
758    if crate::daemon::is_working_on(home, &state.id, now) || !state.active_all_overrun(now) {
759        return Ok(false);
760    }
761    state.abandon("fold");
762    state.save_under(home)?;
763    // The seat that asked is gone for good now - the same door
764    // `graph::Runner::settle_questions` closes the moment `status` lands
765    // somewhere non-resumable, see that method's own doc. Without this, an
766    // open question the abandoned seat left behind would keep badging the
767    // operator until the next daemon startup's `abandon_settled_questions`
768    // pass happened to notice it, or forever if nothing is running `magi
769    // serve` at all.
770    if let Err(e) = Questions::at(home.join("questions")).settle_run(&state.id, state.status) {
771        tracing::warn!("abandon questions for {}: {e:#}", state.id);
772    }
773    Ok(true)
774}
775
776/// Delete files from the shared build cache until it fits its cap — but only
777/// while nobody live is registered as using it. [`crate::cache::maintenance_prune`]
778/// takes out the same lease a build would, so a prune can never race a
779/// compile in flight (this run's own, another run's, or a human's `magi
780/// review`) into deleting a file that build still needs. `Ok(None)` when the
781/// cache is in use right now; the next pass catches it once the borrower
782/// releases it, the same way a cap of `0` or a missing `CARGO_TARGET_DIR`
783/// already meant "nothing to do this time" here.
784pub fn prune_cache(home: &Path, cache: &Path, limit_bytes: u64) -> Result<Option<Prune>> {
785    crate::cache::maintenance_prune(home, cache, limit_bytes)
786}
787
788/// [`prune_cache`], but resolving the operator's opt-out and missing
789/// `CARGO_TARGET_DIR` first — the same two checks [`housekeep`]'s idle pass
790/// makes before ever measuring the cache, factored out so
791/// [`crate::daemon`]'s between-runs check (see the module's own doc for why
792/// congestion can make "idle" arrive too rarely to matter) makes them
793/// identically rather than growing its own copy that could drift. `Ok(None)`
794/// covers a cap of `0` (see the module docs on `cache_limit_bytes`), a config
795/// that renders no `CARGO_TARGET_DIR` to aggregate at all, and a cache
796/// currently in use (see [`prune_cache`]).
797pub fn prune_cache_if_over_limit(
798    cfg: &crate::config::Config,
799    home: &Path,
800) -> Result<Option<Prune>> {
801    if cfg.disk.cache_limit_bytes == 0 {
802        return Ok(None);
803    }
804    let Some(cache) = cfg.cache_dir() else {
805        return Ok(None);
806    };
807    prune_cache(home, &cache, cfg.disk.cache_limit_bytes)
808}
809
810/// The cache's path, size and cap, for `magi cache show` and the health view.
811/// `None` when the config declares no `CARGO_TARGET_DIR` to aggregate.
812///
813/// A cap of `0` means the operator opted out of pruning; the size is then
814/// reported but never acted on.
815pub fn cache_report(cfg: &crate::config::Config) -> Option<(PathBuf, u64, u64)> {
816    let cache = cfg.cache_dir()?;
817    Some((
818        cache.clone(),
819        cache_size(&cache),
820        cfg.disk.cache_limit_bytes,
821    ))
822}
823
824/// Size in bytes of the shared build cache.
825pub fn cache_size(cache: &Path) -> u64 {
826    dir_size(cache)
827}
828
829#[cfg(test)]
830mod tests {
831    use super::*;
832    use crate::config::Disk;
833    use std::fs;
834
835    fn ts(s: &str) -> Timestamp {
836        s.parse().expect("rfc3339")
837    }
838
839    fn block_on<F: std::future::Future>(f: F) -> F::Output {
840        tokio::runtime::Runtime::new().expect("runtime").block_on(f)
841    }
842
843    #[test]
844    fn a_run_is_due_after_its_grace_and_not_before() {
845        let now = ts("2026-09-05T00:00:00Z");
846        let grace = 600;
847        let old = now - SignedDuration::new(601, 0);
848        let fresh = now - SignedDuration::new(599, 0);
849        assert!(due(now, old, grace));
850        assert!(!due(now, fresh, grace));
851        // Exactly at the edge: not yet due.
852        let edge = now - SignedDuration::new(600, 0);
853        assert!(!due(now, edge, grace));
854        // A zero grace folds everything, ever.
855        assert!(due(now, old, 0));
856    }
857
858    #[test]
859    fn rotate_from_moves_the_starting_point_as_the_cursor_advances() {
860        let ids: Vec<String> = ["a", "b", "c", "d", "e"]
861            .iter()
862            .map(|s| s.to_string())
863            .collect();
864
865        // A cursor of 0 starts at the front, same as an unrotated list.
866        assert_eq!(rotate_from(&ids, 0), ids);
867
868        // A cursor of 2 starts two ids further along, with the skipped
869        // front wrapping to the tail rather than being dropped.
870        let at_2 = rotate_from(&ids, 2);
871        assert_eq!(at_2, vec!["c", "d", "e", "a", "b"]);
872        for id in &ids {
873            assert!(at_2.contains(id));
874        }
875
876        // A cursor past the list's own length wraps rather than panicking -
877        // the caller never has to reduce it modulo anything itself.
878        assert_eq!(rotate_from(&ids, 7), rotate_from(&ids, 2));
879
880        // An empty list has no start point to compute and must not panic.
881        assert_eq!(rotate_from(&[], 0), Vec::<String>::new());
882    }
883
884    /// A fleet with more eligible runs than one pass can check must not park
885    /// the same handful at the front of the window forever: advancing the
886    /// cursor by exactly how many ids one pass actually looked at - the
887    /// arithmetic [`reconcile_external_merges`] itself does - walks every id
888    /// to the front within one full rotation, with no reliance on how often
889    /// or how regularly passes happen to run.
890    #[test]
891    fn advancing_the_cursor_by_the_window_size_gives_every_id_a_turn() {
892        let ids: Vec<String> = (0..12).map(|n| format!("run-{n}")).collect();
893        let cap = 5usize;
894        let mut cursor = 0usize;
895        let mut ever_seen: std::collections::HashSet<String> = std::collections::HashSet::new();
896        for _ in 0..ids.len() {
897            let window: Vec<String> = rotate_from(&ids, cursor).into_iter().take(cap).collect();
898            ever_seen.extend(window.iter().cloned());
899            cursor = (cursor + window.len()) % ids.len();
900        }
901        assert_eq!(
902            ever_seen.len(),
903            ids.len(),
904            "every id must be checked at least once across a full rotation, whatever \
905             the cadence between passes"
906        );
907    }
908
909    #[test]
910    fn the_external_merge_cursor_round_trips_through_a_files_absence() {
911        let dir = tempfile::tempdir().unwrap();
912        let home = dir.path();
913
914        // Nothing written yet: starts at 0, not an error.
915        assert_eq!(read_external_merge_cursor(home), 0);
916
917        write_external_merge_cursor(home, 7);
918        assert_eq!(read_external_merge_cursor(home), 7);
919
920        // Anything unreadable as a plain number falls back to 0 rather than
921        // wedging the sweep on a corrupt file.
922        std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), "not a number").unwrap();
923        assert_eq!(read_external_merge_cursor(home), 0);
924    }
925
926    #[test]
927    fn a_run_qualifies_for_an_external_merge_check_only_when_blocked_unmerged_and_due() {
928        let now = ts("2026-09-05T00:00:00Z");
929        let grace = 600;
930        let old = now - SignedDuration::new(601, 0);
931        let fresh = now - SignedDuration::new(599, 0);
932
933        assert!(
934            eligible_for_external_merge_check(RunStatus::Blocked, false, old, now, grace),
935            "blocked, unmerged, and past its grace period is exactly the run this exists for"
936        );
937        assert!(
938            !eligible_for_external_merge_check(RunStatus::Blocked, false, fresh, now, grace),
939            "an operator mid-fix deserves the same grace window `fold_due` gives before \
940             the janitor starts asking GitHub about it"
941        );
942        assert!(
943            !eligible_for_external_merge_check(RunStatus::Blocked, true, old, now, grace),
944            "a run `land::land` already recorded a merge outcome for has its own answer \
945             already; this check is only for a run with nothing recorded at all"
946        );
947        assert!(
948            !eligible_for_external_merge_check(RunStatus::Stalled, false, old, now, grace),
949            "stalled is not blocked - it means the tally never reached quorum, which a \
950             pull request cannot fix"
951        );
952        assert!(
953            !eligible_for_external_merge_check(RunStatus::Ready, false, old, now, grace),
954            "ready has nothing to correct - it was never landed by design"
955        );
956        assert!(
957            !eligible_for_external_merge_check(RunStatus::Superseded, false, old, now, grace),
958            "a superseded run's task already has its answer from a later attempt; \
959             there is nothing left for GitHub to confirm here"
960        );
961    }
962
963    /// A `Blocked` run with a decided winner, so `land::find_external_merge`
964    /// has a branch to ask `gh` about instead of returning early for lack of
965    /// one - which is what makes an unreachable `/nonexistent/repo` fail
966    /// loudly (an `Err`, raising a notice) rather than silently (an early
967    /// `Ok(None)`, raising nothing) when this run's sweep is exercised.
968    fn write_blocked_run(runs: &Path, id: &str, updated_at: Timestamp) {
969        use crate::run::{Candidate, Tally};
970        use std::collections::BTreeMap;
971
972        let mut state = RunState::new(
973            PathBuf::from("/nonexistent/repo"),
974            "main".to_owned(),
975            "0000000000000000000000000000000000000000".to_owned(),
976            String::new(),
977            crate::config::Config::default(),
978        );
979        state.id = id.to_owned();
980        state.status = RunStatus::Blocked;
981        state.updated_at = updated_at;
982        state.candidates.push(Candidate {
983            index: 0,
984            label: 'A',
985            agent: "agent".to_owned(),
986            branch: format!("magi/{id}/A"),
987            worktree: PathBuf::from("/nonexistent/repo"),
988            summary: String::new(),
989            stat: String::new(),
990            files: 0,
991            commits: 0,
992            empty: false,
993            failed: None,
994            verified_noop: None,
995            duration_ms: 0,
996            folded: false,
997        });
998        state.tally = Some(Tally {
999            first_choice: BTreeMap::from([('A', 1)]),
1000            borda: BTreeMap::new(),
1001            winner: 'A',
1002            rankings: 1,
1003            unanimous_initial: true,
1004            deliberated: false,
1005            changed_votes: 0,
1006            unanimous_final: true,
1007            tie_break: None,
1008            judges: 1,
1009            present: 1,
1010            quorum: 1,
1011            met_quorum: true,
1012            uncontested: None,
1013        });
1014        std::fs::create_dir_all(runs.join(id)).unwrap();
1015        std::fs::write(
1016            runs.join(id).join("run.json"),
1017            serde_json::to_string_pretty(&state).unwrap(),
1018        )
1019        .unwrap();
1020    }
1021
1022    /// No run here can be confirmed merged - one has no repository `gh` can
1023    /// even ask about, one has not sat blocked long enough, and one is
1024    /// claimed by a live daemon - so the sweep must leave every one of them
1025    /// exactly as it found them and never panic on the failures in between.
1026    #[tokio::test]
1027    async fn reconcile_external_merges_leaves_ineligible_and_unconfirmable_runs_alone() {
1028        let dir = tempfile::tempdir().unwrap();
1029        let runs = dir.path().join("runs");
1030        let home = dir.path().to_path_buf();
1031        std::fs::create_dir_all(&runs).unwrap();
1032
1033        let now = ts("2026-09-05T00:00:00Z");
1034        let disk = Disk {
1035            fold_grace_secs: 600,
1036            ..Disk::default()
1037        };
1038
1039        let due_id = "20260905-000000-blkd";
1040        write_blocked_run(&runs, due_id, ts("2026-08-01T00:00:00Z"));
1041
1042        let fresh_id = "20260905-000000-fres";
1043        write_blocked_run(&runs, fresh_id, now);
1044
1045        let live_id = "20260905-000000-live";
1046        write_blocked_run(&runs, live_id, ts("2026-08-01T00:00:00Z"));
1047        let mut status = crate::daemon::Status::new();
1048        status.current = vec![crate::daemon::Current {
1049            task: "20260905-000000-task".to_owned(),
1050            run: live_id.to_owned(),
1051        }];
1052        status.updated_at = now;
1053        crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1054
1055        let reconciled = reconcile_external_merges(&runs, &home, &disk, now).await;
1056        assert_eq!(
1057            reconciled, 0,
1058            "an unreachable repository can never be confirmed merged"
1059        );
1060        for id in [due_id, fresh_id, live_id] {
1061            assert_eq!(
1062                read_meta(&runs, id).unwrap().status,
1063                RunStatus::Blocked,
1064                "{id} must be left exactly as it was found"
1065            );
1066        }
1067    }
1068
1069    /// The bug this exists to pin: a first version rotated on
1070    /// `now.as_second() % len`, which is fully determined by wall-clock time
1071    /// and nothing else. Two passes seconds apart - as they would be when a
1072    /// short poll interval and a fast pass alias onto the same handful of
1073    /// `now.as_second()` residues - landed on the *same* starting point and
1074    /// never covered a fleet bigger than the per-pass cap, however many
1075    /// times housekeeping ran. Persisting a cursor makes forward progress a
1076    /// function of how many ids got looked at, not of when the clock reads
1077    /// this pass happened to run.
1078    #[tokio::test]
1079    async fn reconcile_external_merges_covers_a_larger_fleet_across_repeated_passes_at_one_instant()
1080    {
1081        let dir = tempfile::tempdir().unwrap();
1082        let runs = dir.path().join("runs");
1083        let home = dir.path().to_path_buf();
1084        std::fs::create_dir_all(&runs).unwrap();
1085
1086        let now = ts("2026-09-05T00:00:00Z");
1087        let disk = Disk {
1088            fold_grace_secs: 600,
1089            ..Disk::default()
1090        };
1091        let old = ts("2026-08-01T00:00:00Z");
1092
1093        let ids: Vec<String> = (0..8).map(|n| format!("20260905-000000-r{n:03}")).collect();
1094        for id in &ids {
1095            write_blocked_run(&runs, id, old);
1096        }
1097
1098        let notices = crate::notices::Notices::at(home.join("notifications"));
1099
1100        // Two passes at the exact same `now`, exactly what a fast pass on a
1101        // short poll interval looks like. Each of the 8 unreachable-repo
1102        // runs raises its own notice the first time it is looked at, so the
1103        // count of distinct notices is a direct readout of how many distinct
1104        // ids have been checked so far.
1105        reconcile_external_merges(&runs, &home, &disk, now).await;
1106        let after_first = notices.list().len();
1107        assert_eq!(
1108            after_first, MAX_EXTERNAL_MERGE_CHECKS_PER_PASS,
1109            "the first pass checks exactly one cap's worth"
1110        );
1111
1112        reconcile_external_merges(&runs, &home, &disk, now).await;
1113        let after_second = notices.list().len();
1114        assert_eq!(
1115            after_second,
1116            ids.len(),
1117            "a second pass at the same instant must still reach every id the \
1118             first pass had no room for, not repeat the same cap's worth"
1119        );
1120    }
1121
1122    #[test]
1123    fn the_meta_reader_is_tolerant_of_everything_except_the_deciders() {
1124        let dir = tempfile::tempdir().unwrap();
1125        let runs = dir.path().join("runs");
1126        let id = "20260905-000000-abcd";
1127        std::fs::create_dir_all(runs.join(id)).unwrap();
1128        std::fs::write(
1129            runs.join(id).join("run.json"),
1130            r#"{"schema": 99, "id": "20260905-000000-abcd", "updated_at": "2026-09-05T00:00:00Z", "status": "ready", "junk_from_another_build": [1, 2, 3]}"#,
1131        )
1132        .unwrap();
1133        let meta = read_meta(&runs, id).expect("readable");
1134        assert_eq!(meta.status, RunStatus::Ready);
1135        assert_eq!(meta.updated_at, ts("2026-09-05T00:00:00Z"));
1136        assert!(read_meta(&runs, "nope").is_err(), "missing file unreadable");
1137        std::fs::write(runs.join(id).join("run.json"), "not json at all").unwrap();
1138        assert!(read_meta(&runs, id).is_err(), "garbage unreadable");
1139    }
1140
1141    #[test]
1142    fn fold_unreadable_releases_run_dir_and_worktrees() {
1143        let dir = tempfile::tempdir().unwrap();
1144        let runs = dir.path().join("runs");
1145        let wt = dir.path().join("wt");
1146        let id = "20260905-000000-abcd";
1147        std::fs::create_dir_all(runs.join(id)).unwrap();
1148        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1149        std::fs::create_dir_all(wt.join("abcd")).unwrap();
1150        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1151
1152        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold");
1153        assert_eq!(removed.len(), 2);
1154        assert!(!runs.join(id).exists(), "run dir gone");
1155        assert!(!wt.join("abcd").exists(), "worktrees gone");
1156
1157        // A prefix resolves like `run::resolve_id` does.
1158        std::fs::create_dir_all(runs.join(id)).unwrap();
1159        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1160        std::fs::create_dir_all(wt.join("abcd")).unwrap();
1161        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1162        let removed = block_on(fold_unreadable(&runs, &wt, "20260905")).expect("by prefix");
1163        assert_eq!(removed.len(), 2);
1164        // Once gone, `id` cannot be resolved at all - same as `run::resolve_id`
1165        // on an id nothing on disk matches - so a repeat pass errors rather
1166        // than silently reporting nothing removed.
1167        assert!(
1168            block_on(fold_unreadable(&runs, &wt, id)).is_err(),
1169            "a run already gone cannot be resolved again"
1170        );
1171    }
1172
1173    #[test]
1174    fn prune_cache_sheds_the_oldest_generation_until_it_fits() {
1175        let home = tempfile::tempdir().unwrap();
1176        let dir = tempfile::tempdir().unwrap();
1177        // Same size, different age: only the age decides, and the newest
1178        // generation - the one the next build reuses - is what survives.
1179        fs::write(dir.path().join("old"), b"xx").unwrap();
1180        fs::write(dir.path().join("new"), b"yy").unwrap();
1181        touch(&dir.path().join("old"), 1_000_000);
1182        touch(&dir.path().join("new"), 2_000_000);
1183
1184        let out = prune_cache(home.path(), dir.path(), 2)
1185            .expect("prune")
1186            .expect("the cache is free");
1187        assert_eq!(out.files, 1, "one deletion is enough to reach the cap");
1188        assert_eq!(out.remaining, 2);
1189        assert!(!dir.path().join("old").exists(), "the older file went");
1190        assert!(dir.path().join("new").exists(), "the newer one stayed");
1191
1192        // A whole generation shares one timestamp tick, so the tie has to be
1193        // decided too: largest first, which reaches the cap in the fewest
1194        // deletions. Left to `read_dir` and an unstable sort this deleted
1195        // both files on Linux and one on Windows.
1196        let tied = tempfile::tempdir().unwrap();
1197        fs::write(tied.path().join("big"), b"xxxx").unwrap();
1198        fs::write(tied.path().join("small"), b"yy").unwrap();
1199        touch(&tied.path().join("big"), 1_000_000);
1200        touch(&tied.path().join("small"), 1_000_000);
1201        let out = prune_cache(home.path(), tied.path(), 2)
1202            .expect("prune")
1203            .expect("the cache is free");
1204        assert_eq!(out.files, 1, "the big one alone gets under the cap");
1205        assert_eq!(out.remaining, 2);
1206        assert!(tied.path().join("small").exists());
1207    }
1208
1209    #[test]
1210    fn prune_cache_if_over_limit_resolves_the_opt_outs_before_ever_measuring() {
1211        let home = tempfile::tempdir().unwrap();
1212        let dir = tempfile::tempdir().unwrap();
1213        fs::write(dir.path().join("big"), vec![0u8; 10]).unwrap();
1214
1215        let mut cfg = crate::config::Config::default();
1216        cfg.verify.gate = vec![format!(
1217            "CARGO_TARGET_DIR={} cargo make check",
1218            dir.path().display()
1219        )];
1220
1221        // A cap of `0` is the operator's opt-out: never measured, never
1222        // pruned, regardless of what is actually on disk.
1223        cfg.disk.cache_limit_bytes = 0;
1224        assert_eq!(
1225            prune_cache_if_over_limit(&cfg, home.path()).unwrap(),
1226            None,
1227            "a zero cap must not even look at the directory"
1228        );
1229        assert!(dir.path().join("big").exists());
1230
1231        // No `CARGO_TARGET_DIR` in either verify command: nothing to
1232        // aggregate, so there is nothing to prune either.
1233        let mut no_cache = crate::config::Config::default();
1234        no_cache.disk.cache_limit_bytes = 1;
1235        assert_eq!(
1236            prune_cache_if_over_limit(&no_cache, home.path()).unwrap(),
1237            None
1238        );
1239
1240        // Over the cap and configured: pruned exactly like `prune_cache`
1241        // itself would.
1242        cfg.disk.cache_limit_bytes = 1;
1243        let pruned = prune_cache_if_over_limit(&cfg, home.path())
1244            .unwrap()
1245            .expect("a real cache dir over its cap prunes");
1246        assert_eq!(pruned.files, 1);
1247        assert!(!dir.path().join("big").exists());
1248    }
1249
1250    /// Pin a file's mtime, so a test asserts the policy and not the runner's
1251    /// timestamp granularity.
1252    fn touch(path: &Path, secs: u64) {
1253        let f = fs::File::options().write(true).open(path).unwrap();
1254        f.set_times(fs::FileTimes::new().set_modified(
1255            std::time::SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(secs),
1256        ))
1257        .unwrap();
1258    }
1259
1260    /// The disk-full casualty: a run whose first save left `run.json.tmp` and
1261    /// nothing else. It has to be clearable, or the record is permanent.
1262    #[test]
1263    fn fold_unreadable_clears_a_run_whose_state_never_landed() {
1264        let dir = tempfile::tempdir().unwrap();
1265        let runs = dir.path().join("runs");
1266        let wt = dir.path().join("wt");
1267        let id = "20260904-014540-88c0";
1268        std::fs::create_dir_all(runs.join(id)).unwrap();
1269        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1270
1271        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold by id");
1272        assert_eq!(removed, vec![format!("runs/{id}")]);
1273        assert!(!runs.join(id).exists(), "record gone");
1274
1275        // And by prefix, the way the deck and the phone address a run.
1276        std::fs::create_dir_all(runs.join(id)).unwrap();
1277        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1278        assert!(
1279            block_on(fold_unreadable(&runs, &wt, "88c0")).is_ok(),
1280            "by prefix"
1281        );
1282
1283        // A directory under `runs` that is not a run is never a fold target.
1284        std::fs::create_dir_all(runs.join("scratch")).unwrap();
1285        assert!(
1286            block_on(fold_unreadable(&runs, &wt, "scratch")).is_err(),
1287            "a stray directory is not a run"
1288        );
1289    }
1290
1291    #[test]
1292    fn fold_due_folds_terminal_runs_of_any_schema_but_leaves_genuinely_unreadable_ones() {
1293        let dir = tempfile::tempdir().unwrap();
1294        let runs = dir.path().join("runs");
1295        let wt = dir.path().join("wt");
1296        let home = dir.path().to_path_buf();
1297        let disk = Disk::default();
1298        let now = ts("2026-09-05T00:00:00Z");
1299
1300        // 1. Runnable (judging): never folded, however old.
1301        let judging = "20260801-000000-0001";
1302        write_meta(&runs, judging, "judging", "2026-08-01T00:00:00Z");
1303
1304        // 2. Finished but fresh: grace not elapsed. Within the default 6h
1305        //    grace of `now`, so `fold_due` must stop at the freshness check
1306        //    and never even reach `read_state` - `write_meta`'s minimal JSON
1307        //    would fail that full parse anyway, and this case exists to
1308        //    prove freshness is why the run survives, not an accident of the
1309        //    fixture being unparseable as a whole `RunState`.
1310        let ready_fresh = "20260904-220000-0002";
1311        write_meta(&runs, ready_fresh, "ready", "2026-09-04T22:00:00Z");
1312
1313        // 3. Genuinely unreadable: broken JSON, not merely an unfamiliar
1314        //    schema number. Left alone and counted - this is the one case
1315        //    automatic housekeeping must never touch (see `fold_due`'s docs);
1316        //    discarding it is an explicit operator action, not something a
1317        //    background pass does.
1318        let garbage = "20260901-000000-0004";
1319        std::fs::create_dir_all(runs.join(garbage)).unwrap();
1320        std::fs::write(runs.join(garbage).join("run.json"), "not json").unwrap();
1321        std::fs::create_dir_all(wt.join("0004")).unwrap();
1322
1323        // 4. Finished, well past grace, current schema: the ordinary case
1324        //    `fold_due` has always acted on.
1325        let due_ready = due_run(&runs, &wt, "20260801-000000-ffff", SCHEMA);
1326
1327        // 5. Finished, well past grace, but written by a schema number this
1328        //    build no longer matches - the defect this task exists to fix.
1329        //    It still parses cleanly, so only the version number differs, and
1330        //    that alone must not block folding.
1331        let due_old_schema = due_run(&runs, &wt, "20260801-000000-eeee", SCHEMA - 1);
1332
1333        let (folded, unreadable) =
1334            block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
1335        assert_eq!(
1336            folded, 2,
1337            "both due, parseable runs fold regardless of their schema number"
1338        );
1339        assert_eq!(
1340            unreadable, 1,
1341            "only the run with broken JSON counts as unreadable"
1342        );
1343        assert!(runs.join(judging).exists(), "runnable never folded");
1344        assert!(runs.join(ready_fresh).exists(), "fresh never folded");
1345        assert!(runs.join(garbage).exists(), "unreadable record kept");
1346        assert!(wt.join("0004").exists(), "unreadable worktree kept");
1347        assert!(
1348            runs.join(&due_ready).exists(),
1349            "folding drops worktrees, not the record"
1350        );
1351        assert!(
1352            runs.join(&due_old_schema).exists(),
1353            "an old-schema record survives its fold exactly like a current one"
1354        );
1355        // `graph::fold_run` saves the state it just folded back to disk. That
1356        // write must land under this test's own `runs` - the argument it
1357        // passed to `fold_due`, not the process-global `run::home` - or a
1358        // fold that landed somewhere else entirely would still be counted
1359        // above as one of the two `folded` runs.
1360        for id in [&due_ready, &due_old_schema] {
1361            let saved = read_meta(&runs, id).expect("folded run still parses");
1362            assert_ne!(
1363                saved.updated_at,
1364                ts("2026-08-01T00:00:00Z"),
1365                "fold_run must have saved the updated state back through the \
1366                 `runs` directory this test passed to fold_due"
1367            );
1368        }
1369    }
1370
1371    /// `[disk] auto_fold = false` must leave the janitor's fold-and-reclaim
1372    /// passes completely inert - a due run's worktree and record both
1373    /// survive exactly as if `housekeep` had never run at all. Cache pruning
1374    /// is a separate opt-out (`cache_limit_bytes`) and stays disabled here
1375    /// too, so this test is only ever about `auto_fold`.
1376    #[tokio::test]
1377    async fn housekeep_leaves_everything_alone_when_auto_fold_is_disabled() {
1378        let dir = tempfile::tempdir().unwrap();
1379        let runs = dir.path().join("runs");
1380        let wt = dir.path().join("wt");
1381        let home = dir.path().to_path_buf();
1382        crate::run::set_home(dir.path().to_path_buf());
1383
1384        let due_id = due_run(&runs, &wt, "20260801-000000-abcd", SCHEMA);
1385        std::fs::create_dir_all(wt.join("orphan").join("cand-A")).unwrap();
1386
1387        let mut cfg = crate::config::Config::default();
1388        cfg.disk.auto_fold = false;
1389        cfg.disk.cache_limit_bytes = 0;
1390
1391        let out = housekeep(&cfg, &home, &wt, &dir.path().join("repo"), Timestamp::now()).await;
1392
1393        assert_eq!(out.folded, 0);
1394        assert_eq!(out.unreadable, 0);
1395        assert_eq!(out.orphaned_worktrees, 0);
1396        assert!(
1397            runs.join(&due_id).exists(),
1398            "a due run's record survives untouched"
1399        );
1400        assert!(
1401            wt.join("orphan").exists(),
1402            "an orphaned worktree survives untouched: the reclaim pass never ran"
1403        );
1404    }
1405
1406    #[test]
1407    fn fold_orphaned_worktrees_removes_only_worktrees_no_run_claims_and_none_in_flight() {
1408        let dir = tempfile::tempdir().unwrap();
1409        let runs = dir.path().join("runs");
1410        let wt = dir.path().join("wt");
1411        let home = dir.path().to_path_buf();
1412
1413        // A run record exists for this one: its worktree is claimed, not
1414        // orphaned, however old the record.
1415        write_meta(
1416            &runs,
1417            "20260801-000000-aaaa",
1418            "ready",
1419            "2026-08-01T00:00:00Z",
1420        );
1421        std::fs::create_dir_all(wt.join("aaaa").join("cand-A")).unwrap();
1422
1423        // No run record at all, and nobody is working on it: this is the
1424        // leftover `fold_due` can never see, because it only ever walks
1425        // `runs/`.
1426        std::fs::create_dir_all(wt.join("bbbb").join("cand-A")).unwrap();
1427
1428        // No run record either, but a live daemon status names a run with
1429        // this short id - the save-timing gap between the daemon claiming a
1430        // task and `RunState::new` writing its first `run.json`. Must survive
1431        // untouched.
1432        std::fs::create_dir_all(wt.join("cccc")).unwrap();
1433
1434        // Not shaped like a run's short id at all - a scratch directory an
1435        // operator or another tool left in the same bay - so it is never a
1436        // reclaim target regardless of what runs claim it or not.
1437        std::fs::create_dir_all(wt.join("scratch")).unwrap();
1438
1439        // `now` pushed comfortably past `MIN_ORPHAN_AGE_SECS`, so a zero
1440        // grace - the same "always due" escape hatch `due` itself documents
1441        // - still reclaims once a worktree is genuinely old, without faking
1442        // an mtime: real directory creation just above is already in the
1443        // past relative to this `now`, by design rather than by timing.
1444        let now = Timestamp::now() + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1445        let mut status = crate::daemon::Status::new();
1446        status.current = vec![crate::daemon::Current {
1447            task: "20260905-000000-t111".to_owned(),
1448            run: "20260905-000000-cccc".to_owned(),
1449        }];
1450        status.updated_at = now;
1451        crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1452
1453        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1454        assert_eq!(
1455            folded, 1,
1456            "only the truly orphaned, idle, bay-shaped worktree is removed"
1457        );
1458        assert!(wt.join("aaaa").exists(), "claimed by a run record");
1459        assert!(!wt.join("bbbb").exists(), "orphaned and idle: reclaimed");
1460        assert!(wt.join("cccc").exists(), "a run in flight is never touched");
1461        assert!(
1462            wt.join("scratch").exists(),
1463            "not shaped like a worktree bay, so never a reclaim target"
1464        );
1465    }
1466
1467    /// The gap this closes: `Runner::review` (`magi review`) creates the
1468    /// worktree with `git worktree add` before `RunState::save` ever writes a
1469    /// `run.json`, and that path never runs through the daemon's own `poll`
1470    /// loop at all, so `daemon::Status` never names it either. Without a
1471    /// grace window, a janitor pass landing in that gap would read the
1472    /// worktree as an orphan nothing is waiting on and delete a review still
1473    /// being set up.
1474    #[test]
1475    fn fold_orphaned_worktrees_leaves_a_freshly_created_bay_alone() {
1476        let dir = tempfile::tempdir().unwrap();
1477        let runs = dir.path().join("runs");
1478        let wt = dir.path().join("wt");
1479        let home = dir.path().to_path_buf();
1480
1481        std::fs::create_dir_all(wt.join("dddd").join("under-review")).unwrap();
1482
1483        let now = Timestamp::now();
1484        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 6 * 60 * 60, now));
1485        assert_eq!(
1486            folded, 0,
1487            "too fresh to tell apart from a run still being set up"
1488        );
1489        assert!(wt.join("dddd").exists());
1490    }
1491
1492    /// A fresh open question on `run`, stored and handed back for assertions.
1493    fn open_question(store: &Questions, run: &str) -> crate::ask::Question {
1494        let mut q = crate::ask::Question::new(
1495            run.to_owned(),
1496            "implement".to_owned(),
1497            "impl-A".to_owned(),
1498            "Which storage backend should the cache use?".to_owned(),
1499            String::new(),
1500            vec!["SQLite".to_owned(), "Redis".to_owned()],
1501        );
1502        store.put(&mut q).unwrap();
1503        q
1504    }
1505
1506    /// The exact ghost the phone showed: a run that already finished, with a
1507    /// question its dead seat asked still sitting `open` because it reached
1508    /// that status before `graph::Runner::settle_questions` existed (or
1509    /// missed it in the crash window `daemon::reclaim_orphaned_running`
1510    /// covers). This sweep is the second door to the same fact.
1511    #[test]
1512    fn a_finished_runs_open_question_is_swept_up() {
1513        let dir = tempfile::tempdir().unwrap();
1514        let runs = dir.path().join("runs");
1515        let store = Questions::at(dir.path().join("questions"));
1516
1517        let failed = "20260908-205802-c9eb";
1518        write_meta(&runs, failed, "failed", "2026-09-08T20:58:02Z");
1519        let failed_q = open_question(&store, failed);
1520
1521        let merged = "20260908-205501-ca67";
1522        write_meta(&runs, merged, "merged", "2026-09-08T20:55:01Z");
1523        let merged_q = open_question(&store, merged);
1524
1525        let n = abandon_settled_questions(&store, &runs);
1526        assert_eq!(n, 2, "both dead runs' questions are swept in one pass");
1527
1528        for (id, run) in [(&failed_q.id, failed), (&merged_q.id, merged)] {
1529            let back = store.get(id).unwrap();
1530            assert!(!back.status.open(), "{run} is done; nobody reads an answer");
1531            assert!(back.detail.contains(run), "{}", back.detail);
1532        }
1533    }
1534
1535    #[test]
1536    fn a_still_alive_runs_open_question_survives_the_sweep() {
1537        let dir = tempfile::tempdir().unwrap();
1538        let runs = dir.path().join("runs");
1539        let store = Questions::at(dir.path().join("questions"));
1540
1541        // `Blocked` and `Stalled` are `RunStatus::resumable`: the run can
1542        // still be picked back up, so its question may yet get a real
1543        // answer. A run still mid-competition is even more obviously alive.
1544        for (id, status) in [
1545            ("20260908-000000-b10c", "blocked"),
1546            ("20260908-000000-5ta1", "stalled"),
1547            ("20260908-000000-jud6", "judging"),
1548        ] {
1549            write_meta(&runs, id, status, "2026-09-08T00:00:00Z");
1550            let q = open_question(&store, id);
1551
1552            let n = abandon_settled_questions(&store, &runs);
1553            assert_eq!(n, 0, "{status} run is not done; nothing to sweep");
1554            assert!(
1555                store.get(&q.id).unwrap().status.open(),
1556                "{status} run's question must still be waiting"
1557            );
1558        }
1559    }
1560
1561    #[test]
1562    fn the_sweep_leaves_an_answered_question_and_an_unreadable_run_alone() {
1563        let dir = tempfile::tempdir().unwrap();
1564        let runs = dir.path().join("runs");
1565        let store = Questions::at(dir.path().join("questions"));
1566
1567        // Already decided: a sweep must never revisit it, whatever the run
1568        // that asked went on to become.
1569        let done = "20260908-000000-answ";
1570        write_meta(&runs, done, "failed", "2026-09-08T00:00:00Z");
1571        let mut answered = open_question(&store, done);
1572        answered
1573            .answer(crate::ask::Answer::Choice("SQLite".to_owned()))
1574            .unwrap();
1575        store.put(&mut answered).unwrap();
1576
1577        // No `run.json` at all for this one - deleted, or never landed.
1578        let gone = "20260908-000000-gone";
1579        let orphan = open_question(&store, gone);
1580
1581        assert_eq!(abandon_settled_questions(&store, &runs), 0);
1582        assert_eq!(
1583            store.get(&answered.id).unwrap().status,
1584            crate::ask::QuestionStatus::Answered,
1585            "a real answer is never overwritten by a sweep"
1586        );
1587        assert!(
1588            store.get(&orphan.id).unwrap().status.open(),
1589            "a run this sweep cannot read is left exactly as it was, not guessed at"
1590        );
1591    }
1592
1593    /// A grace of `0` is a legitimate, documented value for the operator's
1594    /// own `Disk::fold_grace_secs` - `due`'s "always due" case - but the
1595    /// freshness check this guards is not that policy, and must not collapse
1596    /// to it: a `0` handed straight through would reclaim a worktree the
1597    /// instant it exists, exactly the race `fold_orphaned_worktrees_leaves_a_
1598    /// freshly_created_bay_alone` exists to rule out, just with the operator
1599    /// having turned the other grace off instead of leaving it at its
1600    /// default.
1601    #[test]
1602    fn fold_orphaned_worktrees_floors_a_zero_grace_at_the_race_safe_minimum() {
1603        let dir = tempfile::tempdir().unwrap();
1604        let runs = dir.path().join("runs");
1605        let wt = dir.path().join("wt");
1606        let home = dir.path().to_path_buf();
1607
1608        std::fs::create_dir_all(wt.join("eeee").join("under-review")).unwrap();
1609
1610        // Too fresh, even with the grace argument at zero.
1611        let now = Timestamp::now();
1612        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1613        assert_eq!(
1614            folded, 0,
1615            "a zero grace must not defeat the race-safety floor"
1616        );
1617        assert!(wt.join("eeee").exists());
1618
1619        // Once genuinely past the floor, a zero grace reclaims it - the
1620        // floor is a minimum, not a replacement policy that never fires.
1621        let later = now + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1622        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, later));
1623        assert_eq!(folded, 1, "old enough now, regardless of the zero grace");
1624        assert!(!wt.join("eeee").exists());
1625    }
1626
1627    #[test]
1628    fn clear_abandoned_active_only_acts_once_dead_and_overrun() {
1629        let dir = tempfile::tempdir().unwrap();
1630        // Harmless if another test in this binary already pinned the global
1631        // home first (see `run::set_home`'s own doc): this test only checks
1632        // the in-memory mutation `clear_abandoned_active` makes, never a
1633        // write that landed under this exact directory.
1634        crate::run::set_home(dir.path().to_path_buf());
1635        let home = dir.path().to_path_buf();
1636        let now = ts("2026-09-14T12:00:00Z");
1637        let overrun_seat = || crate::run::ActiveSeat {
1638            node: "implement".to_owned(),
1639            started_at: now - SignedDuration::new(21_000, 0),
1640            timeout_secs: 3_600,
1641            attempt: 0,
1642            task: None,
1643            command: None,
1644            index: None,
1645            total: None,
1646        };
1647
1648        let mut state = RunState::new(
1649            PathBuf::from("/repo"),
1650            "main".to_owned(),
1651            "abc1234".to_owned(),
1652            "fixture".to_owned(),
1653            crate::config::Config::default(),
1654        );
1655        state.status = RunStatus::Implementing;
1656        state.active.insert("impl-A".to_owned(), overrun_seat());
1657
1658        // A seat still within its own budget: not provably dead yet, so this
1659        // must change nothing.
1660        let mut fresh = state.clone();
1661        fresh.active.insert(
1662            "impl-B".to_owned(),
1663            crate::run::ActiveSeat {
1664                node: "implement".to_owned(),
1665                started_at: now,
1666                timeout_secs: 3_600,
1667                attempt: 0,
1668                task: None,
1669                command: None,
1670                index: None,
1671                total: None,
1672            },
1673        );
1674        assert!(!clear_abandoned_active(&mut fresh, &home, now).unwrap());
1675        assert!(!fresh.active.is_empty());
1676        assert_eq!(fresh.status, RunStatus::Implementing);
1677
1678        let store = Questions::at(home.join("questions"));
1679        let q = open_question(&store, &state.id);
1680
1681        assert!(clear_abandoned_active(&mut state, &home, now).unwrap());
1682        assert!(state.active.is_empty());
1683        assert_eq!(state.status, RunStatus::Failed);
1684        assert!(
1685            !store.get(&q.id).unwrap().status.open(),
1686            "the abandoned seat's own open question must not keep badging the \
1687             operator until some later daemon startup notices it"
1688        );
1689    }
1690
1691    /// Write a whole `run.json` that magi can read, over the given state.
1692    fn write_meta(runs: &Path, id: &str, status: &str, updated_at: &str) {
1693        let day = &updated_at[..10];
1694        std::fs::create_dir_all(runs.join(id)).unwrap();
1695        let body = format!(
1696            r#"{{"schema": {SCHEMA}, "id": "{id}", "repo": "/nonexistent/repo", "base_branch": "main", "base_commit": "0000000000000000000000000000000000000000", "instruction": "", "created_at": "{day}T00:00:00Z", "updated_at": "{updated_at}", "status": "{status}", "seed": 1}}"#
1697        );
1698        std::fs::write(runs.join(id).join("run.json"), body).unwrap();
1699    }
1700
1701    /// Write a fully-formed, `Ready`, well-past-grace `run.json` tagged with
1702    /// an arbitrary schema number - so a test can write one this build's own
1703    /// `RunState::new` could never produce on its own. Returns the id.
1704    ///
1705    /// `wt` becomes this run's `graph.worktree_root`: left at the config
1706    /// default, `RunState::worktree_root` falls through to
1707    /// `run::default_worktree_root` - the operator's real `~/wt/magi` - and
1708    /// `graph::fold_run`'s second sweep would then `read_dir` and remove
1709    /// worktrees there instead of anything this test owns.
1710    fn due_run(runs: &Path, wt: &Path, id: &str, schema: u32) -> String {
1711        due_run_with_status(runs, wt, id, schema, RunStatus::Ready)
1712    }
1713
1714    /// [`due_run`], with the terminal status a caller wants instead of the
1715    /// `Ready` every other caller here happens to want.
1716    fn due_run_with_status(
1717        runs: &Path,
1718        wt: &Path,
1719        id: &str,
1720        schema: u32,
1721        status: RunStatus,
1722    ) -> String {
1723        let mut config = crate::config::Config::default();
1724        config.graph.worktree_root = Some(wt.to_path_buf());
1725        let mut state = RunState::new(
1726            PathBuf::from("/nonexistent/repo"),
1727            "main".to_owned(),
1728            "0000000000000000000000000000000000000000".to_owned(),
1729            String::new(),
1730            config,
1731        );
1732        state.id = id.to_owned();
1733        state.status = status;
1734        state.updated_at = ts("2026-08-01T00:00:00Z");
1735        let mut value = serde_json::to_value(&state).unwrap();
1736        value["schema"] = serde_json::json!(schema);
1737        std::fs::create_dir_all(runs.join(id)).unwrap();
1738        std::fs::write(
1739            runs.join(id).join("run.json"),
1740            serde_json::to_string_pretty(&value).unwrap(),
1741        )
1742        .unwrap();
1743        id.to_owned()
1744    }
1745
1746    #[test]
1747    fn fold_due_folds_a_superseded_run_same_as_any_other_terminal_one() {
1748        let dir = tempfile::tempdir().unwrap();
1749        let runs = dir.path().join("runs");
1750        let wt = dir.path().join("wt");
1751        let home = dir.path().to_path_buf();
1752        let disk = Disk::default();
1753        let now = ts("2026-09-05T00:00:00Z");
1754
1755        let superseded = due_run_with_status(
1756            &runs,
1757            &wt,
1758            "20260801-000000-cccc",
1759            SCHEMA,
1760            RunStatus::Superseded,
1761        );
1762
1763        let (folded, unreadable) =
1764            block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
1765        assert_eq!(
1766            folded, 1,
1767            "a superseded run has nothing left for a human to check, so it folds \
1768             exactly like a merged one"
1769        );
1770        assert_eq!(unreadable, 0);
1771        assert!(
1772            read_meta(&runs, &superseded).is_ok(),
1773            "folding drops the worktree, not the record"
1774        );
1775    }
1776
1777    /// A throwaway repo with one commit on `main`, for a test that needs a
1778    /// real winner worktree `fold_run` can actually remove.
1779    fn init_repo(dir: &Path) {
1780        use crate::proc::Quiet as _;
1781        let run = |args: &[&str]| {
1782            let out = std::process::Command::new("git")
1783                .args(args)
1784                .current_dir(dir)
1785                .quiet()
1786                .output()
1787                .expect("spawn git");
1788            assert!(
1789                out.status.success(),
1790                "git {args:?} failed: {}",
1791                String::from_utf8_lossy(&out.stderr)
1792            );
1793        };
1794        run(&["init", "-b", "main"]);
1795        run(&["config", "user.name", "magi test"]);
1796        run(&["config", "user.email", "magi@example.com"]);
1797        std::fs::write(dir.join("README.md"), "# fixture\n").unwrap();
1798        run(&["add", "-A"]);
1799        run(&["commit", "-m", "init"]);
1800    }
1801
1802    /// A due, terminal run with a decided winner that still has a real
1803    /// worktree and branch on disk - so a test can tell whether folding it
1804    /// actually removed the winner or only left it registered as folded.
1805    async fn due_run_with_winner_worktree(
1806        runs: &Path,
1807        wt: &Path,
1808        repo: &Path,
1809        id: &str,
1810        status: RunStatus,
1811    ) -> (String, PathBuf) {
1812        use crate::run::{Candidate, Tally};
1813        use std::collections::BTreeMap;
1814
1815        let mut config = crate::config::Config::default();
1816        config.graph.worktree_root = Some(wt.to_path_buf());
1817        let mut state = RunState::new(
1818            repo.to_path_buf(),
1819            "main".to_owned(),
1820            "0000000000000000000000000000000000000000".to_owned(),
1821            String::new(),
1822            config,
1823        );
1824        state.id = id.to_owned();
1825        state.status = status;
1826        state.updated_at = ts("2026-08-01T00:00:00Z");
1827
1828        let winner_wt = state.worktree_root().join("cand-A");
1829        crate::git::worktree_add_branch(repo, &winner_wt, &format!("magi/{id}/A"), "main")
1830            .await
1831            .expect("winner worktree");
1832
1833        state.candidates.push(Candidate {
1834            index: 0,
1835            label: 'A',
1836            agent: "agent".to_owned(),
1837            branch: format!("magi/{id}/A"),
1838            worktree: winner_wt.clone(),
1839            summary: String::new(),
1840            stat: String::new(),
1841            files: 0,
1842            commits: 0,
1843            empty: false,
1844            failed: None,
1845            verified_noop: None,
1846            duration_ms: 0,
1847            folded: false,
1848        });
1849        state.tally = Some(Tally {
1850            first_choice: BTreeMap::from([('A', 1)]),
1851            borda: BTreeMap::new(),
1852            winner: 'A',
1853            rankings: 1,
1854            unanimous_initial: true,
1855            deliberated: false,
1856            changed_votes: 0,
1857            unanimous_final: true,
1858            tie_break: None,
1859            judges: 1,
1860            present: 1,
1861            quorum: 1,
1862            met_quorum: true,
1863            uncontested: None,
1864        });
1865        std::fs::create_dir_all(runs.join(id)).unwrap();
1866        std::fs::write(
1867            runs.join(id).join("run.json"),
1868            serde_json::to_string_pretty(&state).unwrap(),
1869        )
1870        .unwrap();
1871        (id.to_owned(), winner_wt)
1872    }
1873
1874    #[tokio::test]
1875    async fn fold_due_drops_a_superseded_runs_own_winner_worktree_but_keeps_a_readys() {
1876        let dir = tempfile::tempdir().unwrap();
1877        let runs = dir.path().join("runs");
1878        let wt = dir.path().join("wt");
1879        let repo = dir.path().join("repo");
1880        let home = dir.path().to_path_buf();
1881        let disk = Disk::default();
1882        let now = ts("2026-09-05T00:00:00Z");
1883        std::fs::create_dir_all(&repo).unwrap();
1884        init_repo(&repo);
1885
1886        let (superseded_id, superseded_wt) = due_run_with_winner_worktree(
1887            &runs,
1888            &wt,
1889            &repo,
1890            "20260801-000000-supw",
1891            RunStatus::Superseded,
1892        )
1893        .await;
1894        // `Ready` is auto-folded exactly like `Superseded` (both terminal and
1895        // non-resumable), but never landed anywhere - its own winner is
1896        // deliberately kept, which is what makes it the right contrast here.
1897        // `Blocked`/`Stalled` are `resumable()` and so never even reach
1898        // `fold_run` in the first place; they would not exercise the
1899        // `drop_winner` choice this test is about.
1900        let (ready_id, ready_wt) = due_run_with_winner_worktree(
1901            &runs,
1902            &wt,
1903            &repo,
1904            "20260801-000000-rdyw",
1905            RunStatus::Ready,
1906        )
1907        .await;
1908
1909        let (folded, unreadable) = fold_due(&runs, &home, &wt, &disk, now)
1910            .await
1911            .expect("fold_due");
1912        assert_eq!(folded, 2);
1913        assert_eq!(unreadable, 0);
1914
1915        assert!(
1916            !superseded_wt.exists(),
1917            "a superseded run's own winner never lands anywhere else, so its worktree \
1918             must be dropped exactly like a merged run's"
1919        );
1920        assert!(
1921            ready_wt.exists(),
1922            "a still-ready run's winner may yet be merged by hand - folding must not \
1923             touch it"
1924        );
1925
1926        // Both records survive the fold; only the worktrees differ.
1927        assert!(read_meta(&runs, &superseded_id).is_ok());
1928        assert!(read_meta(&runs, &ready_id).is_ok());
1929    }
1930}