Skip to main content

magi/
clean.rs

1//! The disk janitor: finished runs get their worktrees folded, worktrees whose
2//! run record is already gone get reclaimed too, and the shared build cache is
3//! pruned to its cap.
4//!
5//! A run's state being written by an older schema is not the same thing as it
6//! being unreadable, and this module used to conflate the two: [`fold_due`]
7//! treated any `run.json` its version check rejected exactly like one that
8//! failed to parse at all, so a single schema bump silently stopped every
9//! automatic fold in the fleet the moment it shipped, and did so with no
10//! counter and no log line to say so. A record magi genuinely cannot parse —
11//! missing fields, broken JSON, a schema newer than this build has ever heard
12//! of — is still left alone here, still counted in
13//! [`Housekeeping::unreadable`], and still only ever removed by an explicit
14//! operator action (`magi fold`, or the equivalent phone route). One written
15//! by a schema this build merely disagrees with the *meaning* of is not that:
16//! as long as it still parses, folding proceeds regardless of the number in
17//! its `schema` field.
18//!
19//! Everything policy-shaped — which statuses are foldable, how long a finished
20//! run is left alone, whether the cache is over its limit — is a pure function
21//! injected with numbers, so nothing here has to ask the operating system to
22//! be testable. The only I/O is the removal itself.
23
24use std::collections::BTreeSet;
25use std::path::{Path, PathBuf};
26
27use anyhow::{Context as _, Result, bail};
28use jiff::{SignedDuration, Timestamp};
29use serde::Deserialize;
30
31use crate::ask::Questions;
32use crate::config::Disk;
33use crate::run::{RunState, RunStatus, SCHEMA, short_of};
34
35use crate::disk::{Prune, dir_size};
36
37/// What one janitor pass did, for the caller's log line.
38#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
39pub struct Housekeeping {
40    /// Runs folded (worktrees dropped).
41    pub folded: usize,
42    /// Runs [`fold_due`] left alone because their `run.json` genuinely could
43    /// not be read - missing, broken JSON, a schema this build has never
44    /// heard of - as opposed to one merely written by a different schema
45    /// number, which is folded like any other (see the module docs). This was
46    /// defined but never incremented for a long stretch of this module's
47    /// history, which is exactly how 90 of 93 runs sat unfolded on one
48    /// operator's machine with nothing anywhere saying why: every one of them
49    /// was misclassified as unreadable by a schema check that has since been
50    /// narrowed to only the runs that actually are.
51    pub unreadable: usize,
52    /// Worktrees under the worktree bay reclaimed because no run record in
53    /// `runs/` claims them anymore (see [`fold_orphaned_worktrees`]).
54    pub orphaned_worktrees: usize,
55    /// Files dropped from the shared cache.
56    pub cache_files: usize,
57    /// Bytes freed from the shared cache.
58    pub cache_freed: u64,
59    /// Open questions abandoned because the run that asked them has already
60    /// settled where nothing is coming back to read an answer.
61    pub questions_abandoned: usize,
62    /// Runs found blocked on a merge magi never recorded — a pull request the
63    /// operator merged by hand while `land::land` never got as far as
64    /// opening one itself — and corrected automatically. See
65    /// [`reconcile_external_merges`].
66    pub external_merges_recorded: usize,
67    /// Run records that still called their pull request open after it was
68    /// merged or closed, and were rewritten (see
69    /// [`crate::land::repair_stale_pr_states`]).
70    pub stale_pr_states_repaired: usize,
71}
72
73/// Run the janitor: fold due runs, reclaim orphaned worktrees, prune stale
74/// worktree registrations, then prune the cache if it is over its cap.
75///
76/// Every part is best-effort; a jammed cache lock or a run whose worktree
77/// another borrower holds must not stop the rest. Errors are reported through
78/// `tracing::warn` - this is housekeeping, and the daemon keeps serving
79/// either way.
80pub async fn housekeep(
81    cfg: &crate::config::Config,
82    home: &Path,
83    worktrees_root: &Path,
84    repo: &Path,
85    now: Timestamp,
86) -> Housekeeping {
87    let mut out = Housekeeping::default();
88    if cfg.disk.auto_fold {
89        let runs = home.join("runs");
90        match fold_due(&runs, home, worktrees_root, &cfg.disk, now).await {
91            Ok((folded, unreadable)) => {
92                out.folded = folded;
93                out.unreadable = unreadable;
94            }
95            Err(e) => {
96                tracing::warn!("housekeep: fold due runs: {e:#}");
97                // Stable wording: the reason varies per pass, and a changed
98                // message would relight the bell on every retry.
99                crate::notices::raise_in(
100                    home,
101                    crate::notices::Notice::warn(
102                        "housekeep:fold",
103                        "Automatic cleanup of finished runs failed; disk usage may keep growing.",
104                    ),
105                );
106            }
107        }
108        out.orphaned_worktrees =
109            fold_orphaned_worktrees(&runs, worktrees_root, home, cfg.disk.fold_grace_secs, now)
110                .await;
111        // Best-effort in the same sense as everything else here: a repository
112        // this janitor pass has nothing to do with (or none at all, in a unit
113        // test) must not turn a `warn` into a reason to skip the rest.
114        if let Err(e) = crate::git::worktree_prune(repo).await {
115            tracing::warn!("housekeep: prune worktree registrations: {e:#}");
116        }
117        out.external_merges_recorded = reconcile_external_merges(&runs, home, &cfg.disk, now).await;
118        // Bounded: a handful of forge lookups a pass keeps this cheap, and the
119        // pass is idempotent, so a backlog drains over several passes.
120        out.stale_pr_states_repaired = crate::land::repair_stale_pr_states(home, 5).await.0;
121    }
122    match prune_cache_if_over_limit(cfg, home) {
123        Ok(Some(pruned)) => {
124            out.cache_files = pruned.files;
125            out.cache_freed = pruned.freed;
126        }
127        Ok(None) => {}
128        Err(e) => tracing::warn!("housekeep: prune cache: {e:#}"),
129    }
130    // Unconditional, unlike the two passes above: this is not a disk policy
131    // with a cap or an opt-out, it is closing a gap `graph::Runner` itself
132    // cannot - a run that reached `Merged`/`Ready`/`Failed` before this
133    // cleanup existed, or whose process died between saving that status and
134    // abandoning the question it leaves behind (see `Runner::settle_questions`).
135    // Left alone, that question sits `open` forever: the owner's badge,
136    // banner and title all keep counting a decision nobody is left to read.
137    out.questions_abandoned =
138        abandon_settled_questions(&Questions::at(home.join("questions")), &home.join("runs"));
139    out
140}
141
142/// Abandon every open question whose run has already settled into a status
143/// nothing comes back from, worded with what the run became - the same
144/// cleanup `graph::Runner::settle_questions` runs the moment `status` lands
145/// there, for questions that missed it.
146///
147/// Scans questions rather than runs: the open list is normally short, and a
148/// run that never asked anything costs nothing here. A run this cannot read,
149/// deleted or written by a schema this build does not speak, is left alone
150/// the same as everywhere else in this module; the question stays open
151/// rather than guessed at.
152pub fn abandon_settled_questions(store: &Questions, runs: &Path) -> usize {
153    let waiting_on: BTreeSet<String> = store
154        .list()
155        .into_iter()
156        .filter(|q| q.status.open())
157        .map(|q| q.run)
158        .collect();
159    let mut abandoned = 0;
160    for run in waiting_on {
161        let Ok(meta) = read_meta(runs, &run) else {
162            continue;
163        };
164        match store.settle_run(&run, meta.status) {
165            Ok(n) => abandoned += n,
166            Err(e) => tracing::warn!("housekeep: abandon questions for {run}: {e:#}"),
167        }
168    }
169    abandoned
170}
171
172/// Fold every run that is finished, older than the grace period, and not being
173/// worked on; return `(folded, unreadable)`.
174///
175/// A run whose `run.json` genuinely cannot be parsed — missing fields, broken
176/// JSON, a schema newer than this build has ever heard of — is left exactly
177/// as it is. Automatic housekeeping cannot tell a mid-write file from one that
178/// will never parse again, and `<home>/runs/<id>/` is the evidence `magi
179/// stats` and the deck read; when unsure whether it is safe to touch, the
180/// janitor keeps rather than deletes (see the module docs). Discarding a
181/// record this unreadable is an explicit operator action (`magi fold`, or the
182/// equivalent phone route), never something that happens unattended. Every
183/// such skip is counted in the returned `unreadable` and logged through
184/// `tracing::warn` with the parse failure that caused it - silence here is
185/// exactly the failure mode that let 90 of 93 runs sit unfolded with nothing
186/// to show for it.
187///
188/// A run merely written by a *different* schema number is not unreadable: as
189/// long as `run.json` still parses, it folds like any other terminal run (see
190/// the module docs for why the two are different questions).
191///
192/// Runnable statuses and runs newer than the grace period are also left
193/// alone; folding them would throw away work that is still the answer to
194/// somebody's question. `Merged` runs forget their winner's worktree (the
195/// merge already landed it); `Ready` and `Failed` runs keep it.
196pub async fn fold_due(
197    runs: &Path,
198    home: &Path,
199    _worktrees_root: &Path,
200    disk: &Disk,
201    now: Timestamp,
202) -> Result<(usize, usize)> {
203    let mut folded = 0usize;
204    let mut unreadable = 0usize;
205    let mut ids: Vec<String> = std::fs::read_dir(runs)
206        .into_iter()
207        .flatten()
208        .flatten()
209        .filter(|e| e.path().join("run.json").is_file())
210        .map(|e| e.file_name().to_string_lossy().into_owned())
211        .collect();
212    ids.sort_unstable();
213    for id in ids {
214        if crate::daemon::is_working_on(home, &id, now) {
215            continue;
216        }
217        let meta = match read_meta(runs, &id) {
218            Ok(meta) => meta,
219            Err(e) => {
220                unreadable += 1;
221                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
222                continue;
223            }
224        };
225        if meta.status.resumable() || !due(now, meta.updated_at, disk.fold_grace_secs) {
226            continue;
227        }
228        // `read_meta` already proved the file parses; `read_state` asks for
229        // the rest of the fields `graph::fold_run` needs (worktree paths,
230        // candidates, tally). A schema mismatch alone does not fail this -
231        // see the module docs - so reaching `Err` here means the JSON itself
232        // is broken in a way `read_meta` did not exercise, which is rare but
233        // not impossible (a body truncated between the two fields it reads
234        // and the rest). That must not cost every other run its turn through
235        // this loop, so it is a skip, not a `?`.
236        let mut state = match read_state(runs, &id) {
237            Ok(state) => state,
238            Err(e) => {
239                unreadable += 1;
240                tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
241                continue;
242            }
243        };
244        if state.schema != SCHEMA {
245            tracing::info!(
246                "housekeep: run {id} was written by schema {} (this build speaks {SCHEMA}); \
247                 folding it anyway",
248                state.schema
249            );
250        }
251        // `Merged`'s winner branch is safe to drop because it already landed
252        // in the real target branch; `Superseded`'s winner branch is safe to
253        // drop for a different reason - a later attempt at the same task is
254        // what actually landed (or didn't), so *this* run's own winner is
255        // never going to be merged by anyone. Every other terminal status
256        // keeps the winner: `Ready`/`Blocked`/`Stalled`/`Failed`/`VerifiedNoop`
257        // may still have a human's decision pending on that exact branch.
258        // `AlreadyInBase` is the same: the change is on the base under other
259        // commit ids, so the branch holds nothing anyone still needs.
260        let drop_winner = matches!(
261            state.status,
262            RunStatus::Merged | RunStatus::Superseded | RunStatus::AlreadyInBase
263        );
264        // One run's fold must not cost every later run its turn. A worktree
265        // another borrower holds, a branch git refuses to delete, a repository
266        // that has since moved: each is a reason this run cannot be folded
267        // now, and none is a reason to stop the pass. Left unfolded, it is
268        // simply due again next time; a `?` here stopped automatic folding
269        // permanently at the first such run (finding R3-1-1 of run 51a3).
270        match crate::graph::fold_run(&mut state, drop_winner, home).await {
271            Ok(_) => folded += 1,
272            Err(e) => tracing::warn!("housekeep: fold {id}: {e:#}"),
273        }
274    }
275    Ok((folded, unreadable))
276}
277
278/// What one [`fetch_origins`] pass did.
279#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
280pub struct FetchReport {
281    /// Repos whose fetch succeeded.
282    pub fetched: usize,
283    /// Repos with no `origin` remote, skipped.
284    pub no_origin: usize,
285    /// Repos whose fetch failed or timed out.
286    pub failed: usize,
287}
288
289/// Run `git fetch origin` in every checkout under `roots` (the set `magi
290/// repos` lists), so `origin/main` stays current.
291///
292/// Fetch only: never a checkout, merge or fast-forward, so no HEAD, branch or
293/// working tree moves. Best-effort like [`fold_due`]: a repo with no `origin`
294/// is skipped silently, a failure or timeout is a warning, and the pass never
295/// errors. `stop` is checked between repos.
296pub async fn fetch_origins(
297    roots: &[PathBuf],
298    per_repo: std::time::Duration,
299    stop: impl Fn() -> bool,
300) -> FetchReport {
301    let mut report = FetchReport::default();
302    for repo in crate::repos::scan(roots) {
303        if stop() {
304            break;
305        }
306        match crate::git::git_raw(&repo.path, &["remote", "get-url", "origin"]).await {
307            Ok(out) if out.ok() => {}
308            _ => {
309                report.no_origin += 1;
310                continue;
311            }
312        }
313        match crate::git::fetch_origin(&repo.path, per_repo).await {
314            Ok(()) => report.fetched += 1,
315            Err(e) => {
316                tracing::warn!("fetch origin in {}: {e:#}", repo.name);
317                report.failed += 1;
318            }
319        }
320    }
321    report
322}
323
324/// Is `updated` old enough, measured against `now`, that the run may fold?
325///
326/// Pure; the janitor compares against wallclock, tests inject both sides. The
327/// comparison is strict, so a run exactly at the edge of its grace period is
328/// left alone one more pass — the same convention as [`crate::disk::over_limit`].
329pub fn due(now: Timestamp, updated: Timestamp, grace_secs: u64) -> bool {
330    now.duration_since(updated) > SignedDuration::new(grace_secs as i64, 0)
331}
332
333/// The two fields the janitor decides on, read with a serde that tolerates
334/// everything else about the run being unreadable.
335#[derive(Deserialize)]
336struct Meta {
337    status: RunStatus,
338    updated_at: Timestamp,
339}
340
341/// Read `status` and `updated_at` straight off the state file, asking for
342/// nothing else. `Err` when the file is missing, not parseable, or a status in
343/// a version this build does not speak - all of which mean "unreadable".
344fn read_meta(runs: &Path, id: &str) -> Result<Meta> {
345    let path = runs.join(id).join("run.json");
346    let body =
347        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
348    let meta: Meta =
349        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
350    Ok(meta)
351}
352
353/// Runs checked against GitHub in one janitor pass, at most.
354///
355/// A fleet with many stuck runs must not turn every idle tick into a burst of
356/// `gh pr list` calls; a run left unchecked this pass is checked again next
357/// pass, same as an unfolded due run is.
358const MAX_EXTERNAL_MERGE_CHECKS_PER_PASS: usize = 5;
359
360/// The fields [`reconcile_external_merges`] filters on before paying for a
361/// full [`RunState`] parse or a `gh` round trip.
362#[derive(Deserialize)]
363struct ExternalMergeMeta {
364    status: RunStatus,
365    updated_at: Timestamp,
366    #[serde(default)]
367    merge: Option<crate::run::MergeOutcome>,
368}
369
370fn read_external_merge_meta(runs: &Path, id: &str) -> Result<ExternalMergeMeta> {
371    let path = runs.join(id).join("run.json");
372    let body =
373        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
374    let meta: ExternalMergeMeta =
375        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
376    Ok(meta)
377}
378
379/// Is a run worth asking GitHub about?
380///
381/// Pure, so the filter itself is assertable without a fixture that writes
382/// `run.json` files or spawns `gh`. Narrower than "blocked": `merge_recorded`
383/// excludes a run `land::land` already finished landing (successfully or
384/// not) - that path already carries its own record of what happened - and
385/// `due` gives an operator who is about to fix a run by hand a grace window
386/// before the janitor starts asking the same question about it.
387pub fn eligible_for_external_merge_check(
388    status: RunStatus,
389    merge_recorded: bool,
390    updated_at: Timestamp,
391    now: Timestamp,
392    grace_secs: u64,
393) -> bool {
394    status == RunStatus::Blocked && !merge_recorded && due(now, updated_at, grace_secs)
395}
396
397/// Rotate `ids` so a fixed per-pass cap does not always land on the same
398/// prefix of the list, starting at `start` (already reduced modulo nothing -
399/// callers pass whatever a persisted cursor last recorded, and this takes it
400/// modulo `ids.len()` itself).
401///
402/// Pure: the offset is the caller's problem, not wall-clock time. An earlier
403/// version derived `start` from `now.as_second() % ids.len()`, which looked
404/// independent of any state to persist but is not actually independent of
405/// how often the janitor runs - a poll interval and a pass's own duration
406/// that alias onto the same handful of `now.as_second()` values (say, a
407/// ten-second poll and a pass that finishes in under a second) revisit only
408/// those same few residues forever and never advance into the rest of the
409/// list at all. [`reconcile_external_merges`] instead persists a cursor
410/// across passes ([`read_external_merge_cursor`] /
411/// [`write_external_merge_cursor`]) and advances it by exactly how many ids
412/// this pass actually looked at, which is the only quantity that is
413/// guaranteed to move forward pass over pass regardless of timing.
414fn rotate_from(ids: &[String], start: usize) -> Vec<String> {
415    if ids.is_empty() {
416        return Vec::new();
417    }
418    let start = start % ids.len();
419    ids[start..]
420        .iter()
421        .chain(ids[..start].iter())
422        .cloned()
423        .collect()
424}
425
426/// Where [`reconcile_external_merges`] remembers how far it got, so a fixed
427/// per-pass cap advances through a fleet's eligible runs pass over pass
428/// instead of camping on whichever ones happen to sort first.
429const EXTERNAL_MERGE_CURSOR_FILE: &str = "external-merge-cursor";
430
431/// `0` for anything this cannot read as a plain number - a first run, a
432/// corrupt or missing file, a build that has never written one - which is
433/// exactly as good a starting point as any: the cursor's whole job is to
434/// keep moving, not to encode any particular position as meaningful.
435fn read_external_merge_cursor(home: &Path) -> usize {
436    std::fs::read_to_string(home.join(EXTERNAL_MERGE_CURSOR_FILE))
437        .ok()
438        .and_then(|s| s.trim().parse().ok())
439        .unwrap_or(0)
440}
441
442/// Best-effort, like every other write in this module's sweep: a failure to
443/// persist the cursor costs the fleet one pass's worth of forward progress
444/// on this sweep, never the run whose housekeeping it is racing to keep up
445/// with.
446fn write_external_merge_cursor(home: &Path, cursor: usize) {
447    if let Err(e) = std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), cursor.to_string()) {
448        tracing::warn!("housekeep: persist the external-merge sweep's cursor: {e:#}");
449    }
450}
451
452/// Close the gap `magi fold --merged` exists for, without waiting for an
453/// operator to notice and go find the pull request URL: a run that stopped
454/// `Blocked` with no `merge` recorded may have been merged anyway, by hand,
455/// on a pull request `land::land` never opened or never got to observe as
456/// merged. `land::find_external_merge` asks GitHub about the run's own
457/// winning branch, and only a pull request it can uniquely tie back to this
458/// run is ever acted on - see that function's own doc for what "uniquely"
459/// excludes.
460///
461/// A run this finds and fixes goes through
462/// [`crate::land::correct_confirmed_external_merge`] - the same correction
463/// `magi fold --merged` performs, minus the same-repo guard that command
464/// needs and this loop doesn't (see that function's own doc: the URL here
465/// was never operator-supplied, it came from asking `state.repo`'s own
466/// remote) - then through [`crate::graph::fold_run`] so it stops holding
467/// worktrees the moment it stops needing them. A run this cannot decide
468/// about - `gh` unreachable, more than one candidate pull
469/// request, nothing found at all - is left exactly as it is; only an error
470/// asking GitHub raises a notice, since "nothing found" is the ordinary,
471/// expected shape of a run that really is just blocked.
472async fn reconcile_external_merges(runs: &Path, home: &Path, disk: &Disk, now: Timestamp) -> usize {
473    let mut ids: Vec<String> = std::fs::read_dir(runs)
474        .into_iter()
475        .flatten()
476        .flatten()
477        .filter(|e| e.path().join("run.json").is_file())
478        .map(|e| e.file_name().to_string_lossy().into_owned())
479        .collect();
480    ids.sort_unstable();
481
482    // Every eligible id first, cheaply (a `Meta` read, no `gh`), then rotated
483    // before the cap is applied. Capping the sorted order directly would
484    // always land on the same lexicographic prefix - a fleet's oldest run ids
485    // - so a handful of long-blocked runs that never turn out to be merged
486    // would starve every id after them of a `gh` check, forever.
487    let mut eligible: Vec<String> = Vec::new();
488    for id in ids {
489        if crate::daemon::is_working_on(home, &id, now) {
490            continue;
491        }
492        let meta = match read_external_merge_meta(runs, &id) {
493            Ok(meta) => meta,
494            // Already counted as unreadable by `fold_due`'s own pass; not
495            // worth a second warning for the same file.
496            Err(_) => continue,
497        };
498        if eligible_for_external_merge_check(
499            meta.status,
500            meta.merge.is_some(),
501            meta.updated_at,
502            now,
503            disk.fold_grace_secs,
504        ) {
505            eligible.push(id);
506        }
507    }
508
509    if eligible.is_empty() {
510        return 0;
511    }
512    let cursor = read_external_merge_cursor(home);
513    let window: Vec<String> = rotate_from(&eligible, cursor)
514        .into_iter()
515        .take(MAX_EXTERNAL_MERGE_CHECKS_PER_PASS)
516        .collect();
517    // Advances by exactly how many ids this pass looked at, regardless of
518    // what came of looking - a run that turned out not to be merged after
519    // all must not be revisited before every other eligible run has had its
520    // own turn.
521    write_external_merge_cursor(home, (cursor + window.len()) % eligible.len());
522
523    let mut reconciled = 0usize;
524    for id in window {
525        let mut state = match read_state(runs, &id) {
526            Ok(state) => state,
527            Err(_) => continue,
528        };
529        match crate::land::find_external_merge(&state).await {
530            Ok(Some(found)) => {
531                match crate::land::correct_confirmed_external_merge(&mut state, &found.url).await {
532                    Ok(_) => {
533                        if let Err(e) = crate::graph::fold_run(&mut state, true, home).await {
534                            tracing::warn!(
535                                "housekeep: fold {id} after recording its external merge: {e:#}"
536                            );
537                        }
538                        reconciled += 1;
539                    }
540                    Err(e) => tracing::warn!(
541                        "housekeep: record external merge {} for {id}: {e:#}",
542                        found.url
543                    ),
544                }
545            }
546            Ok(None) => {}
547            Err(e) => {
548                tracing::warn!("housekeep: check external merge for {id}: {e:#}");
549                crate::notices::raise_in(
550                    home,
551                    crate::notices::Notice::warn(
552                        &format!("merged-unrecorded:{id}"),
553                        format!(
554                            "Run {id} is blocked with no recorded merge, and checking GitHub \
555                             for a merge failed; check by hand."
556                        ),
557                    )
558                    .link(crate::notices::Link::Run { id: id.clone() }),
559                );
560            }
561        }
562    }
563    reconciled
564}
565
566/// Read a whole run state from a runs directory, for folding only.
567///
568/// Deliberately more permissive than [`RunState::load`], which this does not
569/// call: `load` backs `--resume` and every hand-driven command, where a
570/// schema this build disagrees with the *meaning* of must refuse outright
571/// rather than resume a review round or a tally against stale semantics
572/// (`RunState::SCHEMA`'s own docs list what has changed meaning at each
573/// bump). Folding recomputes nothing - it only reads worktree paths, branch
574/// names and a tally winner off the struct to remove them - so an old
575/// schema's values are exactly as good here as a current one's; every schema
576/// bump so far has only ever added a field or a variant, never repurposed an
577/// existing one, and serde already fills an added field's default when an
578/// older record has nothing to say about it. What this cannot tolerate, and
579/// what still surfaces as an `Err`, is `run.json` failing to parse at all.
580fn read_state(runs: &Path, id: &str) -> Result<RunState> {
581    let path = runs.join(id).join("run.json");
582    let body =
583        std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
584    let state: RunState =
585        serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
586    Ok(state)
587}
588
589/// Remove a run that cannot be read: its state directory under `runs` and its
590/// worktree directory under `worktrees_root`.
591///
592/// The state file is the only record of a run's repository and branches, so a
593/// run this unreadable is discarded at the filesystem level - there is no
594/// candidate list to fold first. The worktrees live under
595/// [`crate::run::default_worktree_root`] unless the run's config relocated
596/// them, which an unreadable run cannot tell us; the default location is
597/// removed, and anything the run placed elsewhere is a leftover for whoever
598/// knows where it went.
599///
600/// Deleting a worktree directory by hand leaves its registration in git, and a
601/// registered path cannot be re-`worktree add`-ed until it is pruned - so every
602/// worktree is unregistered from its repository first, best-effort, via the
603/// `gitdir:` link git keeps inside the directory.
604pub async fn fold_unreadable(runs: &Path, worktrees_root: &Path, id: &str) -> Result<Vec<String>> {
605    let resolved = resolve_id_path(runs, id)?;
606    let mut removed = Vec::new();
607    let run_dir = runs.join(&resolved);
608    if run_dir.exists() {
609        std::fs::remove_dir_all(&run_dir)
610            .with_context(|| format!("remove {}", run_dir.display()))?;
611        removed.push(format!("runs/{resolved}"));
612    }
613    let wt = worktrees_root.join(short_of(&resolved));
614    if wt.exists() {
615        crate::git::remove_worktree_from_linked(&wt).await;
616        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
617            crate::git::remove_worktree_from_linked(&e.path()).await;
618        }
619        std::fs::remove_dir_all(&wt).with_context(|| format!("remove {}", wt.display()))?;
620        removed.push(wt.to_string_lossy().into_owned());
621    }
622    Ok(removed)
623}
624
625/// Reclaim worktrees under `worktrees_root` that no run record in `runs`
626/// claims anymore, and return how many were removed.
627///
628/// [`fold_due`] only ever sees a worktree by walking `runs/` first, so a
629/// worktree whose run record is already gone — `magi run rm`, or a record
630/// deleted before its worktree — never enters that loop at all: nothing there
631/// is looking for it. This walks the worktree bay directly instead, and
632/// removes any `<short>` directory that no run id maps to.
633///
634/// Two things must never happen, and this checks both before ever touching a
635/// directory:
636///
637/// - **A worktree bay is never the only kind of thing under `worktrees_root`,
638///   and this must not assume it is.** A hand-placed scratch directory, or
639///   anything else an operator or another tool left in the same bay, has the
640///   same "no run claims it" shape as a genuine orphan but is not one -
641///   [`looks_like_a_worktree_bay`] is the same tag shape [`crate::run::is_run_id`]
642///   already requires of a real run's short id, and anything else is left
643///   alone regardless of what else is true about it.
644/// - **A worktree that was only just created might not have a `run.json` yet
645///   for a reason that has nothing to do with being orphaned.** `Runner::start`
646///   and `Runner::review` both create the worktree before the first
647///   `RunState::save` lands, and that gap - several `git` subprocesses wide -
648///   is invisible to [`crate::daemon::is_working_on_short`] whenever the run
649///   is not being driven through this daemon's own `poll` loop at all (a
650///   `magi review` invocation, for one). A directory whose own modification
651///   time is within `grace_secs` of `now` is left alone on that basis alone,
652///   the same margin [`fold_due`] gives a run before treating it as truly
653///   finished - long enough that no realistic gap between a `worktree add`
654///   and its `run.json` could ever be mistaken for one.
655///
656/// The one failure this must never cause is deleting the worktree of a run
657/// that is genuinely in flight. [`crate::daemon::is_working_on_short`] is the
658/// same liveness check [`fold_due`] trusts everywhere else in this module,
659/// checked by short id because there is no full id to compare here; when it
660/// cannot tell, this leaves the directory alone. Best-effort like the rest of
661/// housekeeping: one directory git or the filesystem refuses to give up is a
662/// `tracing::warn`, not a reason to abandon the rest of the pass.
663pub async fn fold_orphaned_worktrees(
664    runs: &Path,
665    worktrees_root: &Path,
666    home: &Path,
667    grace_secs: u64,
668    now: Timestamp,
669) -> usize {
670    let known: std::collections::HashSet<String> = std::fs::read_dir(runs)
671        .into_iter()
672        .flatten()
673        .flatten()
674        .map(|e| e.file_name().to_string_lossy().into_owned())
675        .filter(|name| crate::run::is_run_id(name))
676        .map(|id| short_of(&id).to_owned())
677        .collect();
678
679    let mut folded = 0usize;
680    for entry in std::fs::read_dir(worktrees_root)
681        .into_iter()
682        .flatten()
683        .flatten()
684    {
685        if !entry.path().is_dir() {
686            continue;
687        }
688        let short = entry.file_name().to_string_lossy().into_owned();
689        if !looks_like_a_worktree_bay(&short) {
690            continue;
691        }
692        if known.contains(&short) || crate::daemon::is_working_on_short(home, &short, now) {
693            continue;
694        }
695        let wt = entry.path();
696        if !stale_enough(&wt, grace_secs, now) {
697            continue;
698        }
699        crate::git::remove_worktree_from_linked(&wt).await;
700        for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
701            crate::git::remove_worktree_from_linked(&e.path()).await;
702        }
703        match std::fs::remove_dir_all(&wt) {
704            Ok(()) => folded += 1,
705            Err(e) => tracing::warn!(
706                "housekeep: remove orphaned worktree {}: {e:#}",
707                wt.display()
708            ),
709        }
710    }
711    folded
712}
713
714/// Does `name` have the shape a run's own worktree bay is named with: the
715/// same 4-character alphanumeric tag [`crate::run::is_run_id`] requires of a
716/// full id's trailing block (see [`short_of`])?
717///
718/// Anything else under `worktrees_root` is not a bay this function reclaims
719/// at all, claimed or not - answering "does a run claim this?" about a
720/// directory that was never a run's worktree in the first place is exactly
721/// the wrong question to ask before deleting it.
722fn looks_like_a_worktree_bay(name: &str) -> bool {
723    name.len() == 4 && name.bytes().all(|b| b.is_ascii_alphanumeric())
724}
725
726/// Is `dir`'s own modification time old enough, against `grace_secs` and
727/// `now`, that its emptiness of a run record can be trusted rather than
728/// caught mid-creation?
729///
730/// A directory this pass cannot stat at all - a race with its own removal, a
731/// permission error - is treated as not yet stale: unreadable metadata is not
732/// evidence of anything, and the janitor already keeps rather than deletes
733/// whenever it cannot tell (see the module docs).
734///
735/// `grace_secs` is floored at [`MIN_ORPHAN_AGE_SECS`] regardless of what the
736/// caller passes: `0` is a documented, legitimate value for
737/// [`crate::config::Disk::fold_grace_secs`] (`due`'s own "always due" case),
738/// because that grace answers a policy question the operator owns - how long
739/// a *known, finished* run's worktree lingers before cleanup. Whether an
740/// orphan worktree is actually a race with `Runner::review`'s `git worktree
741/// add` landing before its `run.json` is not a policy question, and must not
742/// collapse to zero just because the operator turned the other grace off -
743/// that would defeat the very check meant to catch it.
744fn stale_enough(dir: &Path, grace_secs: u64, now: Timestamp) -> bool {
745    let Ok(modified) = std::fs::metadata(dir).and_then(|m| m.modified()) else {
746        return false;
747    };
748    let Ok(ts) = Timestamp::try_from(modified) else {
749        return false;
750    };
751    due(now, ts, grace_secs.max(MIN_ORPHAN_AGE_SECS))
752}
753
754/// The floor under [`stale_enough`]'s grace, independent of
755/// [`crate::config::Disk::fold_grace_secs`].
756///
757/// Ample next to the race it guards: the gap between `Runner::start` or
758/// `Runner::review` creating a worktree and the first `RunState::save`
759/// landing is a handful of `git` subprocess calls, not minutes - but the
760/// janitor cannot tell "still mid-setup" from "orphaned" by any other signal
761/// for a run that never registers with `daemon::Status` at all (a `magi
762/// review` invocation, for one), so this is generous on purpose rather than
763/// tuned to the observed case.
764const MIN_ORPHAN_AGE_SECS: u64 = 5 * 60;
765
766/// Resolve an id or prefix against an explicit runs directory, exactly the way
767/// [`crate::run::resolve_id`] does against the global home.
768fn resolve_id_path(runs: &Path, prefix: &str) -> Result<String> {
769    // Keyed on the directory, not on a readable state file: the record this
770    // route exists to remove may be a lone `run.json.tmp` from a save that
771    // ran out of disk, and that is precisely the one a human needs a way to
772    // clear (see `crate::run::list_ids`).
773    if runs.join(prefix).is_dir() && crate::run::is_run_id(prefix) {
774        return Ok(prefix.to_owned());
775    }
776    let mut hits: Vec<String> = Vec::new();
777    for e in std::fs::read_dir(runs).into_iter().flatten().flatten() {
778        if !e.path().is_dir() {
779            continue;
780        }
781        let id = e.file_name().to_string_lossy().into_owned();
782        if crate::run::is_run_id(&id) && (id.starts_with(prefix) || id.ends_with(prefix)) {
783            hits.push(id);
784        }
785    }
786    match hits.len() {
787        1 => Ok(hits.into_iter().next().expect("exactly one hit")),
788        0 => bail!("no run matches `{prefix}`"),
789        _ => bail!(
790            "`{prefix}` matches {} runs: {}",
791            hits.len(),
792            hits.join(", ")
793        ),
794    }
795}
796
797/// `magi fold`'s recovery path for a run whose worktrees are already gone —
798/// so [`crate::graph::fold_run`] removed nothing — but whose `run.json` still
799/// lists active seats nobody is left to answer for: no live daemon claims the
800/// run, and every one of those seats has overrun its own timeout (see
801/// [`RunState::active_all_overrun`]). Clearing them and failing the run is
802/// what lets it be deleted afterward — [`RunState::ensure_can_delete`] only
803/// ever checks whether a live daemon is working on the run and whether its
804/// candidates are folded, not `status`, but a run stuck `implementing`
805/// forever with an empty worktree still reads as unresolved everywhere else
806/// (`magi show`, the deck, the phone) until this runs.
807///
808/// Returns `false` without changing anything when a live daemon still claims
809/// the run, or when some active seat has not actually overrun its budget yet
810/// — a run that is merely between waves must never be guessed at.
811pub fn clear_abandoned_active(state: &mut RunState, home: &Path, now: Timestamp) -> Result<bool> {
812    if crate::daemon::is_working_on(home, &state.id, now) || !state.active_all_overrun(now) {
813        return Ok(false);
814    }
815    state.abandon("fold");
816    state.save_under(home)?;
817    // The seat that asked is gone for good now - the same door
818    // `graph::Runner::settle_questions` closes the moment `status` lands
819    // somewhere non-resumable, see that method's own doc. Without this, an
820    // open question the abandoned seat left behind would keep badging the
821    // operator until the next daemon startup's `abandon_settled_questions`
822    // pass happened to notice it, or forever if nothing is running `magi
823    // serve` at all.
824    if let Err(e) = Questions::at(home.join("questions")).settle_run(&state.id, state.status) {
825        tracing::warn!("abandon questions for {}: {e:#}", state.id);
826    }
827    Ok(true)
828}
829
830/// Delete files from the shared build cache until it fits its cap — but only
831/// while nobody live is registered as using it. [`crate::cache::maintenance_prune`]
832/// takes out the same lease a build would, so a prune can never race a
833/// compile in flight (this run's own, another run's, or a human's `magi
834/// review`) into deleting a file that build still needs. `Ok(None)` when the
835/// cache is in use right now; the next pass catches it once the borrower
836/// releases it, the same way a cap of `0` or a missing `CARGO_TARGET_DIR`
837/// already meant "nothing to do this time" here.
838pub fn prune_cache(home: &Path, cache: &Path, limit_bytes: u64) -> Result<Option<Prune>> {
839    crate::cache::maintenance_prune(home, cache, limit_bytes)
840}
841
842/// [`prune_cache`], but resolving the operator's opt-out and missing
843/// `CARGO_TARGET_DIR` first — the same two checks [`housekeep`]'s idle pass
844/// makes before ever measuring the cache, factored out so
845/// [`crate::daemon`]'s between-runs check (see the module's own doc for why
846/// congestion can make "idle" arrive too rarely to matter) makes them
847/// identically rather than growing its own copy that could drift. `Ok(None)`
848/// covers a cap of `0` (see the module docs on `cache_limit_bytes`), a config
849/// that renders no `CARGO_TARGET_DIR` to aggregate at all, and a cache
850/// currently in use (see [`prune_cache`]).
851pub fn prune_cache_if_over_limit(
852    cfg: &crate::config::Config,
853    home: &Path,
854) -> Result<Option<Prune>> {
855    if cfg.disk.cache_limit_bytes == 0 {
856        return Ok(None);
857    }
858    let Some(cache) = cfg.cache_dir() else {
859        return Ok(None);
860    };
861    prune_cache(home, &cache, cfg.disk.cache_limit_bytes)
862}
863
864/// The cache's path, size and cap, for `magi cache show` and the health view.
865/// `None` when the config declares no `CARGO_TARGET_DIR` to aggregate.
866///
867/// A cap of `0` means the operator opted out of pruning; the size is then
868/// reported but never acted on.
869pub fn cache_report(cfg: &crate::config::Config) -> Option<(PathBuf, u64, u64)> {
870    let cache = cfg.cache_dir()?;
871    Some((
872        cache.clone(),
873        cache_size(&cache),
874        cfg.disk.cache_limit_bytes,
875    ))
876}
877
878/// Size in bytes of the shared build cache.
879pub fn cache_size(cache: &Path) -> u64 {
880    dir_size(cache)
881}
882
883#[cfg(test)]
884mod tests {
885
886    async fn sh(cwd: &Path, args: &[&str]) {
887        let out = crate::git::git_raw(cwd, args).await.expect("spawn git");
888        assert!(out.ok(), "git {args:?}: {}", out.stderr);
889    }
890
891    async fn commit(cwd: &Path, file: &str, body: &str) {
892        std::fs::write(cwd.join(file), body).unwrap();
893        sh(cwd, &["add", file]).await;
894        sh(
895            cwd,
896            &[
897                "-c",
898                "user.name=t",
899                "-c",
900                "user.email=t@example.com",
901                "commit",
902                "-q",
903                "-m",
904                body,
905            ],
906        )
907        .await;
908    }
909
910    const LONG: std::time::Duration = std::time::Duration::from_secs(30);
911
912    #[tokio::test]
913    async fn fetch_advances_origin_main_and_leaves_the_checkout_alone() {
914        let tmp = tempfile::tempdir().unwrap();
915        let t = tmp.path();
916        let bare = t.join("remote.git");
917        sh(
918            t,
919            &["init", "-q", "--bare", "-b", "main", bare.to_str().unwrap()],
920        )
921        .await;
922        let dir = t.join("root/h/o/r");
923        std::fs::create_dir_all(dir.parent().unwrap()).unwrap();
924        sh(
925            t,
926            &["clone", "-q", bare.to_str().unwrap(), dir.to_str().unwrap()],
927        )
928        .await;
929        sh(&dir, &["checkout", "-q", "-b", "main"]).await;
930        commit(&dir, "a.txt", "one").await;
931        sh(&dir, &["push", "-q", "origin", "main"]).await;
932        sh(&dir, &["checkout", "-q", "--detach"]).await;
933        std::fs::write(dir.join("a.txt"), "dirty").unwrap();
934
935        let other = t.join("other");
936        sh(
937            t,
938            &[
939                "clone",
940                "-q",
941                bare.to_str().unwrap(),
942                other.to_str().unwrap(),
943            ],
944        )
945        .await;
946        commit(&other, "b.txt", "two").await;
947        sh(&other, &["push", "-q", "origin", "HEAD:main"]).await;
948
949        let rev = |d: PathBuf, r: &'static str| async move {
950            crate::git::git(&d, &["rev-parse", r]).await.unwrap()
951        };
952        let head = rev(dir.clone(), "HEAD").await;
953        let before = rev(dir.clone(), "origin/main").await;
954        let status = crate::git::git(&dir, &["status", "--porcelain"])
955            .await
956            .unwrap();
957
958        let r = fetch_origins(&[t.join("root")], LONG, || false).await;
959        assert_eq!(
960            r,
961            FetchReport {
962                fetched: 1,
963                no_origin: 0,
964                failed: 0
965            }
966        );
967
968        assert_ne!(rev(dir.clone(), "origin/main").await, before);
969        assert_eq!(rev(dir.clone(), "HEAD").await, head);
970        assert_eq!(
971            crate::git::git(&dir, &["status", "--porcelain"])
972                .await
973                .unwrap(),
974            status
975        );
976        assert_eq!(std::fs::read_to_string(dir.join("a.txt")).unwrap(), "dirty");
977    }
978
979    #[tokio::test]
980    async fn fetch_never_writes_a_local_branch_whatever_the_configured_refspec() {
981        let tmp = tempfile::tempdir().unwrap();
982        let t = tmp.path();
983        let bare = t.join("remote.git");
984        sh(
985            t,
986            &["init", "-q", "--bare", "-b", "main", bare.to_str().unwrap()],
987        )
988        .await;
989        let dir = t.join("root/h/o/r");
990        std::fs::create_dir_all(dir.parent().unwrap()).unwrap();
991        sh(
992            t,
993            &["clone", "-q", bare.to_str().unwrap(), dir.to_str().unwrap()],
994        )
995        .await;
996        sh(&dir, &["checkout", "-q", "-b", "main"]).await;
997        commit(&dir, "a.txt", "one").await;
998        sh(&dir, &["push", "-q", "origin", "main"]).await;
999        sh(&dir, &["checkout", "-q", "--detach"]).await;
1000        // A hostile mapping onto the local branch, and no origin/* at all.
1001        sh(
1002            &dir,
1003            &[
1004                "config",
1005                "remote.origin.fetch",
1006                "+refs/heads/main:refs/heads/main",
1007            ],
1008        )
1009        .await;
1010        let _ = crate::git::git_raw(&dir, &["update-ref", "-d", "refs/remotes/origin/main"]).await;
1011
1012        let other = t.join("other");
1013        sh(
1014            t,
1015            &[
1016                "clone",
1017                "-q",
1018                bare.to_str().unwrap(),
1019                other.to_str().unwrap(),
1020            ],
1021        )
1022        .await;
1023        commit(&other, "b.txt", "two").await;
1024        sh(&other, &["push", "-q", "origin", "HEAD:main"]).await;
1025        let remote_tip = crate::git::git(&other, &["rev-parse", "HEAD"])
1026            .await
1027            .unwrap();
1028        let local_main = crate::git::git(&dir, &["rev-parse", "refs/heads/main"])
1029            .await
1030            .unwrap();
1031
1032        let r = fetch_origins(&[t.join("root")], LONG, || false).await;
1033        assert_eq!(r.fetched, 1);
1034        assert_eq!(
1035            crate::git::git(&dir, &["rev-parse", "refs/remotes/origin/main"])
1036                .await
1037                .unwrap(),
1038            remote_tip
1039        );
1040        assert_eq!(
1041            crate::git::git(&dir, &["rev-parse", "refs/heads/main"])
1042                .await
1043                .unwrap(),
1044            local_main
1045        );
1046    }
1047
1048    #[tokio::test]
1049    async fn fetch_skips_no_origin_and_survives_an_unreachable_origin() {
1050        let tmp = tempfile::tempdir().unwrap();
1051        let t = tmp.path();
1052        let root = t.join("root");
1053        let bare = t.join("remote.git");
1054        sh(t, &["init", "-q", "--bare", bare.to_str().unwrap()]).await;
1055
1056        let none = root.join("h/o/none");
1057        let dead = root.join("h/o/dead");
1058        let good = root.join("h/o/good");
1059        for d in [&none, &dead, &good] {
1060            std::fs::create_dir_all(d).unwrap();
1061            sh(d, &["init", "-q"]).await;
1062        }
1063        sh(
1064            &dead,
1065            &[
1066                "remote",
1067                "add",
1068                "origin",
1069                t.join("missing").to_str().unwrap(),
1070            ],
1071        )
1072        .await;
1073        sh(&good, &["remote", "add", "origin", bare.to_str().unwrap()]).await;
1074
1075        let r = fetch_origins(&[root], LONG, || false).await;
1076        assert_eq!(
1077            r,
1078            FetchReport {
1079                fetched: 1,
1080                no_origin: 1,
1081                failed: 1
1082            }
1083        );
1084    }
1085    use super::*;
1086    use crate::config::Disk;
1087    use std::fs;
1088
1089    fn ts(s: &str) -> Timestamp {
1090        s.parse().expect("rfc3339")
1091    }
1092
1093    fn block_on<F: std::future::Future>(f: F) -> F::Output {
1094        tokio::runtime::Runtime::new().expect("runtime").block_on(f)
1095    }
1096
1097    #[test]
1098    fn a_run_is_due_after_its_grace_and_not_before() {
1099        let now = ts("2026-09-05T00:00:00Z");
1100        let grace = 600;
1101        let old = now - SignedDuration::new(601, 0);
1102        let fresh = now - SignedDuration::new(599, 0);
1103        assert!(due(now, old, grace));
1104        assert!(!due(now, fresh, grace));
1105        // Exactly at the edge: not yet due.
1106        let edge = now - SignedDuration::new(600, 0);
1107        assert!(!due(now, edge, grace));
1108        // A zero grace folds everything, ever.
1109        assert!(due(now, old, 0));
1110    }
1111
1112    #[test]
1113    fn rotate_from_moves_the_starting_point_as_the_cursor_advances() {
1114        let ids: Vec<String> = ["a", "b", "c", "d", "e"]
1115            .iter()
1116            .map(|s| s.to_string())
1117            .collect();
1118
1119        // A cursor of 0 starts at the front, same as an unrotated list.
1120        assert_eq!(rotate_from(&ids, 0), ids);
1121
1122        // A cursor of 2 starts two ids further along, with the skipped
1123        // front wrapping to the tail rather than being dropped.
1124        let at_2 = rotate_from(&ids, 2);
1125        assert_eq!(at_2, vec!["c", "d", "e", "a", "b"]);
1126        for id in &ids {
1127            assert!(at_2.contains(id));
1128        }
1129
1130        // A cursor past the list's own length wraps rather than panicking -
1131        // the caller never has to reduce it modulo anything itself.
1132        assert_eq!(rotate_from(&ids, 7), rotate_from(&ids, 2));
1133
1134        // An empty list has no start point to compute and must not panic.
1135        assert_eq!(rotate_from(&[], 0), Vec::<String>::new());
1136    }
1137
1138    /// A fleet with more eligible runs than one pass can check must not park
1139    /// the same handful at the front of the window forever: advancing the
1140    /// cursor by exactly how many ids one pass actually looked at - the
1141    /// arithmetic [`reconcile_external_merges`] itself does - walks every id
1142    /// to the front within one full rotation, with no reliance on how often
1143    /// or how regularly passes happen to run.
1144    #[test]
1145    fn advancing_the_cursor_by_the_window_size_gives_every_id_a_turn() {
1146        let ids: Vec<String> = (0..12).map(|n| format!("run-{n}")).collect();
1147        let cap = 5usize;
1148        let mut cursor = 0usize;
1149        let mut ever_seen: std::collections::HashSet<String> = std::collections::HashSet::new();
1150        for _ in 0..ids.len() {
1151            let window: Vec<String> = rotate_from(&ids, cursor).into_iter().take(cap).collect();
1152            ever_seen.extend(window.iter().cloned());
1153            cursor = (cursor + window.len()) % ids.len();
1154        }
1155        assert_eq!(
1156            ever_seen.len(),
1157            ids.len(),
1158            "every id must be checked at least once across a full rotation, whatever \
1159             the cadence between passes"
1160        );
1161    }
1162
1163    #[test]
1164    fn the_external_merge_cursor_round_trips_through_a_files_absence() {
1165        let dir = tempfile::tempdir().unwrap();
1166        let home = dir.path();
1167
1168        // Nothing written yet: starts at 0, not an error.
1169        assert_eq!(read_external_merge_cursor(home), 0);
1170
1171        write_external_merge_cursor(home, 7);
1172        assert_eq!(read_external_merge_cursor(home), 7);
1173
1174        // Anything unreadable as a plain number falls back to 0 rather than
1175        // wedging the sweep on a corrupt file.
1176        std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), "not a number").unwrap();
1177        assert_eq!(read_external_merge_cursor(home), 0);
1178    }
1179
1180    #[test]
1181    fn a_run_qualifies_for_an_external_merge_check_only_when_blocked_unmerged_and_due() {
1182        let now = ts("2026-09-05T00:00:00Z");
1183        let grace = 600;
1184        let old = now - SignedDuration::new(601, 0);
1185        let fresh = now - SignedDuration::new(599, 0);
1186
1187        assert!(
1188            eligible_for_external_merge_check(RunStatus::Blocked, false, old, now, grace),
1189            "blocked, unmerged, and past its grace period is exactly the run this exists for"
1190        );
1191        assert!(
1192            !eligible_for_external_merge_check(RunStatus::Blocked, false, fresh, now, grace),
1193            "an operator mid-fix deserves the same grace window `fold_due` gives before \
1194             the janitor starts asking GitHub about it"
1195        );
1196        assert!(
1197            !eligible_for_external_merge_check(RunStatus::Blocked, true, old, now, grace),
1198            "a run `land::land` already recorded a merge outcome for has its own answer \
1199             already; this check is only for a run with nothing recorded at all"
1200        );
1201        assert!(
1202            !eligible_for_external_merge_check(RunStatus::Stalled, false, old, now, grace),
1203            "stalled is not blocked - it means the tally never reached quorum, which a \
1204             pull request cannot fix"
1205        );
1206        assert!(
1207            !eligible_for_external_merge_check(RunStatus::Ready, false, old, now, grace),
1208            "ready has nothing to correct - it was never landed by design"
1209        );
1210        assert!(
1211            !eligible_for_external_merge_check(RunStatus::Superseded, false, old, now, grace),
1212            "a superseded run's task already has its answer from a later attempt; \
1213             there is nothing left for GitHub to confirm here"
1214        );
1215    }
1216
1217    /// A `Blocked` run with a decided winner, so `land::find_external_merge`
1218    /// has a branch to ask `gh` about instead of returning early for lack of
1219    /// one - which is what makes an unreachable `/nonexistent/repo` fail
1220    /// loudly (an `Err`, raising a notice) rather than silently (an early
1221    /// `Ok(None)`, raising nothing) when this run's sweep is exercised.
1222    fn write_blocked_run(runs: &Path, id: &str, updated_at: Timestamp) {
1223        use crate::run::{Candidate, Tally};
1224        use std::collections::BTreeMap;
1225
1226        let mut state = RunState::new(
1227            PathBuf::from("/nonexistent/repo"),
1228            "main".to_owned(),
1229            "0000000000000000000000000000000000000000".to_owned(),
1230            String::new(),
1231            crate::config::Config::default(),
1232        );
1233        state.id = id.to_owned();
1234        state.status = RunStatus::Blocked;
1235        state.updated_at = updated_at;
1236        state.candidates.push(Candidate {
1237            index: 0,
1238            label: 'A',
1239            agent: "agent".to_owned(),
1240            branch: format!("magi/{id}/A"),
1241            worktree: PathBuf::from("/nonexistent/repo"),
1242            summary: String::new(),
1243            stat: String::new(),
1244            files: 0,
1245            commits: 0,
1246            empty: false,
1247            failed: None,
1248            verified_noop: None,
1249            duration_ms: 0,
1250            folded: false,
1251        });
1252        state.tally = Some(Tally {
1253            first_choice: BTreeMap::from([('A', 1)]),
1254            borda: BTreeMap::new(),
1255            winner: 'A',
1256            rankings: 1,
1257            unanimous_initial: true,
1258            deliberated: false,
1259            changed_votes: 0,
1260            unanimous_final: true,
1261            tie_break: None,
1262            judges: 1,
1263            present: 1,
1264            quorum: 1,
1265            met_quorum: true,
1266            uncontested: None,
1267        });
1268        std::fs::create_dir_all(runs.join(id)).unwrap();
1269        std::fs::write(
1270            runs.join(id).join("run.json"),
1271            serde_json::to_string_pretty(&state).unwrap(),
1272        )
1273        .unwrap();
1274    }
1275
1276    /// No run here can be confirmed merged - one has no repository `gh` can
1277    /// even ask about, one has not sat blocked long enough, and one is
1278    /// claimed by a live daemon - so the sweep must leave every one of them
1279    /// exactly as it found them and never panic on the failures in between.
1280    #[tokio::test]
1281    async fn reconcile_external_merges_leaves_ineligible_and_unconfirmable_runs_alone() {
1282        let dir = tempfile::tempdir().unwrap();
1283        let runs = dir.path().join("runs");
1284        let home = dir.path().to_path_buf();
1285        std::fs::create_dir_all(&runs).unwrap();
1286
1287        let now = ts("2026-09-05T00:00:00Z");
1288        let disk = Disk {
1289            fold_grace_secs: 600,
1290            ..Disk::default()
1291        };
1292
1293        let due_id = "20260905-000000-blkd";
1294        write_blocked_run(&runs, due_id, ts("2026-08-01T00:00:00Z"));
1295
1296        let fresh_id = "20260905-000000-fres";
1297        write_blocked_run(&runs, fresh_id, now);
1298
1299        let live_id = "20260905-000000-live";
1300        write_blocked_run(&runs, live_id, ts("2026-08-01T00:00:00Z"));
1301        let mut status = crate::daemon::Status::new();
1302        status.current = vec![crate::daemon::Current {
1303            task: "20260905-000000-task".to_owned(),
1304            run: live_id.to_owned(),
1305        }];
1306        status.updated_at = now;
1307        crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1308
1309        let reconciled = reconcile_external_merges(&runs, &home, &disk, now).await;
1310        assert_eq!(
1311            reconciled, 0,
1312            "an unreachable repository can never be confirmed merged"
1313        );
1314        for id in [due_id, fresh_id, live_id] {
1315            assert_eq!(
1316                read_meta(&runs, id).unwrap().status,
1317                RunStatus::Blocked,
1318                "{id} must be left exactly as it was found"
1319            );
1320        }
1321        let notices = crate::notices::Notices::at(home.join("notifications")).list();
1322        let due_notice = notices
1323            .iter()
1324            .find(|n| n.key == format!("merged-unrecorded:{due_id}"))
1325            .expect("notice for due_id must be raised");
1326        assert_eq!(
1327            due_notice.link,
1328            Some(crate::notices::Link::Run {
1329                id: due_id.to_owned(),
1330            })
1331        );
1332    }
1333
1334    /// The bug this exists to pin: a first version rotated on
1335    /// `now.as_second() % len`, which is fully determined by wall-clock time
1336    /// and nothing else. Two passes seconds apart - as they would be when a
1337    /// short poll interval and a fast pass alias onto the same handful of
1338    /// `now.as_second()` residues - landed on the *same* starting point and
1339    /// never covered a fleet bigger than the per-pass cap, however many
1340    /// times housekeeping ran. Persisting a cursor makes forward progress a
1341    /// function of how many ids got looked at, not of when the clock reads
1342    /// this pass happened to run.
1343    #[tokio::test]
1344    async fn reconcile_external_merges_covers_a_larger_fleet_across_repeated_passes_at_one_instant()
1345    {
1346        let dir = tempfile::tempdir().unwrap();
1347        let runs = dir.path().join("runs");
1348        let home = dir.path().to_path_buf();
1349        std::fs::create_dir_all(&runs).unwrap();
1350
1351        let now = ts("2026-09-05T00:00:00Z");
1352        let disk = Disk {
1353            fold_grace_secs: 600,
1354            ..Disk::default()
1355        };
1356        let old = ts("2026-08-01T00:00:00Z");
1357
1358        let ids: Vec<String> = (0..8).map(|n| format!("20260905-000000-r{n:03}")).collect();
1359        for id in &ids {
1360            write_blocked_run(&runs, id, old);
1361        }
1362
1363        let notices = crate::notices::Notices::at(home.join("notifications"));
1364
1365        // Two passes at the exact same `now`, exactly what a fast pass on a
1366        // short poll interval looks like. Each of the 8 unreachable-repo
1367        // runs raises its own notice the first time it is looked at, so the
1368        // count of distinct notices is a direct readout of how many distinct
1369        // ids have been checked so far.
1370        reconcile_external_merges(&runs, &home, &disk, now).await;
1371        let after_first = notices.list().len();
1372        assert_eq!(
1373            after_first, MAX_EXTERNAL_MERGE_CHECKS_PER_PASS,
1374            "the first pass checks exactly one cap's worth"
1375        );
1376
1377        reconcile_external_merges(&runs, &home, &disk, now).await;
1378        let after_second = notices.list().len();
1379        assert_eq!(
1380            after_second,
1381            ids.len(),
1382            "a second pass at the same instant must still reach every id the \
1383             first pass had no room for, not repeat the same cap's worth"
1384        );
1385    }
1386
1387    #[test]
1388    fn the_meta_reader_is_tolerant_of_everything_except_the_deciders() {
1389        let dir = tempfile::tempdir().unwrap();
1390        let runs = dir.path().join("runs");
1391        let id = "20260905-000000-abcd";
1392        std::fs::create_dir_all(runs.join(id)).unwrap();
1393        std::fs::write(
1394            runs.join(id).join("run.json"),
1395            r#"{"schema": 99, "id": "20260905-000000-abcd", "updated_at": "2026-09-05T00:00:00Z", "status": "ready", "junk_from_another_build": [1, 2, 3]}"#,
1396        )
1397        .unwrap();
1398        let meta = read_meta(&runs, id).expect("readable");
1399        assert_eq!(meta.status, RunStatus::Ready);
1400        assert_eq!(meta.updated_at, ts("2026-09-05T00:00:00Z"));
1401        assert!(read_meta(&runs, "nope").is_err(), "missing file unreadable");
1402        std::fs::write(runs.join(id).join("run.json"), "not json at all").unwrap();
1403        assert!(read_meta(&runs, id).is_err(), "garbage unreadable");
1404    }
1405
1406    #[test]
1407    fn fold_unreadable_releases_run_dir_and_worktrees() {
1408        let dir = tempfile::tempdir().unwrap();
1409        let runs = dir.path().join("runs");
1410        let wt = dir.path().join("wt");
1411        let id = "20260905-000000-abcd";
1412        std::fs::create_dir_all(runs.join(id)).unwrap();
1413        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1414        std::fs::create_dir_all(wt.join("abcd")).unwrap();
1415        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1416
1417        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold");
1418        assert_eq!(removed.len(), 2);
1419        assert!(!runs.join(id).exists(), "run dir gone");
1420        assert!(!wt.join("abcd").exists(), "worktrees gone");
1421
1422        // A prefix resolves like `run::resolve_id` does.
1423        std::fs::create_dir_all(runs.join(id)).unwrap();
1424        std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1425        std::fs::create_dir_all(wt.join("abcd")).unwrap();
1426        std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1427        let removed = block_on(fold_unreadable(&runs, &wt, "20260905")).expect("by prefix");
1428        assert_eq!(removed.len(), 2);
1429        // Once gone, `id` cannot be resolved at all - same as `run::resolve_id`
1430        // on an id nothing on disk matches - so a repeat pass errors rather
1431        // than silently reporting nothing removed.
1432        assert!(
1433            block_on(fold_unreadable(&runs, &wt, id)).is_err(),
1434            "a run already gone cannot be resolved again"
1435        );
1436    }
1437
1438    #[test]
1439    fn prune_cache_sheds_the_oldest_generation_until_it_fits() {
1440        let home = tempfile::tempdir().unwrap();
1441        let dir = tempfile::tempdir().unwrap();
1442        // Same size, different age: only the age decides, and the newest
1443        // generation - the one the next build reuses - is what survives.
1444        fs::write(dir.path().join("old"), b"xx").unwrap();
1445        fs::write(dir.path().join("new"), b"yy").unwrap();
1446        touch(&dir.path().join("old"), 1_000_000);
1447        touch(&dir.path().join("new"), 2_000_000);
1448
1449        let out = prune_cache(home.path(), dir.path(), 2)
1450            .expect("prune")
1451            .expect("the cache is free");
1452        assert_eq!(out.files, 1, "one deletion is enough to reach the cap");
1453        assert_eq!(out.remaining, 2);
1454        assert!(!dir.path().join("old").exists(), "the older file went");
1455        assert!(dir.path().join("new").exists(), "the newer one stayed");
1456
1457        // A whole generation shares one timestamp tick, so the tie has to be
1458        // decided too: largest first, which reaches the cap in the fewest
1459        // deletions. Left to `read_dir` and an unstable sort this deleted
1460        // both files on Linux and one on Windows.
1461        let tied = tempfile::tempdir().unwrap();
1462        fs::write(tied.path().join("big"), b"xxxx").unwrap();
1463        fs::write(tied.path().join("small"), b"yy").unwrap();
1464        touch(&tied.path().join("big"), 1_000_000);
1465        touch(&tied.path().join("small"), 1_000_000);
1466        let out = prune_cache(home.path(), tied.path(), 2)
1467            .expect("prune")
1468            .expect("the cache is free");
1469        assert_eq!(out.files, 1, "the big one alone gets under the cap");
1470        assert_eq!(out.remaining, 2);
1471        assert!(tied.path().join("small").exists());
1472    }
1473
1474    #[test]
1475    fn prune_cache_if_over_limit_resolves_the_opt_outs_before_ever_measuring() {
1476        let home = tempfile::tempdir().unwrap();
1477        let dir = tempfile::tempdir().unwrap();
1478        fs::write(dir.path().join("big"), vec![0u8; 10]).unwrap();
1479
1480        let mut cfg = crate::config::Config::default();
1481        cfg.verify.gate = vec![format!(
1482            "CARGO_TARGET_DIR={} cargo make check",
1483            dir.path().display()
1484        )];
1485
1486        // A cap of `0` is the operator's opt-out: never measured, never
1487        // pruned, regardless of what is actually on disk.
1488        cfg.disk.cache_limit_bytes = 0;
1489        assert_eq!(
1490            prune_cache_if_over_limit(&cfg, home.path()).unwrap(),
1491            None,
1492            "a zero cap must not even look at the directory"
1493        );
1494        assert!(dir.path().join("big").exists());
1495
1496        // No `CARGO_TARGET_DIR` in either verify command: nothing to
1497        // aggregate, so there is nothing to prune either.
1498        let mut no_cache = crate::config::Config::default();
1499        no_cache.disk.cache_limit_bytes = 1;
1500        assert_eq!(
1501            prune_cache_if_over_limit(&no_cache, home.path()).unwrap(),
1502            None
1503        );
1504
1505        // Over the cap and configured: pruned exactly like `prune_cache`
1506        // itself would.
1507        cfg.disk.cache_limit_bytes = 1;
1508        let pruned = prune_cache_if_over_limit(&cfg, home.path())
1509            .unwrap()
1510            .expect("a real cache dir over its cap prunes");
1511        assert_eq!(pruned.files, 1);
1512        assert!(!dir.path().join("big").exists());
1513    }
1514
1515    /// Pin a file's mtime, so a test asserts the policy and not the runner's
1516    /// timestamp granularity.
1517    fn touch(path: &Path, secs: u64) {
1518        let f = fs::File::options().write(true).open(path).unwrap();
1519        f.set_times(fs::FileTimes::new().set_modified(
1520            std::time::SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(secs),
1521        ))
1522        .unwrap();
1523    }
1524
1525    /// The disk-full casualty: a run whose first save left `run.json.tmp` and
1526    /// nothing else. It has to be clearable, or the record is permanent.
1527    #[test]
1528    fn fold_unreadable_clears_a_run_whose_state_never_landed() {
1529        let dir = tempfile::tempdir().unwrap();
1530        let runs = dir.path().join("runs");
1531        let wt = dir.path().join("wt");
1532        let id = "20260904-014540-88c0";
1533        std::fs::create_dir_all(runs.join(id)).unwrap();
1534        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1535
1536        let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold by id");
1537        assert_eq!(removed, vec![format!("runs/{id}")]);
1538        assert!(!runs.join(id).exists(), "record gone");
1539
1540        // And by prefix, the way the deck and the phone address a run.
1541        std::fs::create_dir_all(runs.join(id)).unwrap();
1542        std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1543        assert!(
1544            block_on(fold_unreadable(&runs, &wt, "88c0")).is_ok(),
1545            "by prefix"
1546        );
1547
1548        // A directory under `runs` that is not a run is never a fold target.
1549        std::fs::create_dir_all(runs.join("scratch")).unwrap();
1550        assert!(
1551            block_on(fold_unreadable(&runs, &wt, "scratch")).is_err(),
1552            "a stray directory is not a run"
1553        );
1554    }
1555
1556    #[test]
1557    fn fold_due_folds_terminal_runs_of_any_schema_but_leaves_genuinely_unreadable_ones() {
1558        let dir = tempfile::tempdir().unwrap();
1559        let runs = dir.path().join("runs");
1560        let wt = dir.path().join("wt");
1561        let home = dir.path().to_path_buf();
1562        let disk = Disk::default();
1563        let now = ts("2026-09-05T00:00:00Z");
1564
1565        // 1. Runnable (judging): never folded, however old.
1566        let judging = "20260801-000000-0001";
1567        write_meta(&runs, judging, "judging", "2026-08-01T00:00:00Z");
1568
1569        // 2. Finished but fresh: grace not elapsed. Within the default 6h
1570        //    grace of `now`, so `fold_due` must stop at the freshness check
1571        //    and never even reach `read_state` - `write_meta`'s minimal JSON
1572        //    would fail that full parse anyway, and this case exists to
1573        //    prove freshness is why the run survives, not an accident of the
1574        //    fixture being unparseable as a whole `RunState`.
1575        let ready_fresh = "20260904-220000-0002";
1576        write_meta(&runs, ready_fresh, "ready", "2026-09-04T22:00:00Z");
1577
1578        // 3. Genuinely unreadable: broken JSON, not merely an unfamiliar
1579        //    schema number. Left alone and counted - this is the one case
1580        //    automatic housekeeping must never touch (see `fold_due`'s docs);
1581        //    discarding it is an explicit operator action, not something a
1582        //    background pass does.
1583        let garbage = "20260901-000000-0004";
1584        std::fs::create_dir_all(runs.join(garbage)).unwrap();
1585        std::fs::write(runs.join(garbage).join("run.json"), "not json").unwrap();
1586        std::fs::create_dir_all(wt.join("0004")).unwrap();
1587
1588        // 4. Finished, well past grace, current schema: the ordinary case
1589        //    `fold_due` has always acted on.
1590        let due_ready = due_run(&runs, &wt, "20260801-000000-ffff", SCHEMA);
1591
1592        // 5. Finished, well past grace, but written by a schema number this
1593        //    build no longer matches - the defect this task exists to fix.
1594        //    It still parses cleanly, so only the version number differs, and
1595        //    that alone must not block folding.
1596        let due_old_schema = due_run(&runs, &wt, "20260801-000000-eeee", SCHEMA - 1);
1597
1598        let (folded, unreadable) =
1599            block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
1600        assert_eq!(
1601            folded, 2,
1602            "both due, parseable runs fold regardless of their schema number"
1603        );
1604        assert_eq!(
1605            unreadable, 1,
1606            "only the run with broken JSON counts as unreadable"
1607        );
1608        assert!(runs.join(judging).exists(), "runnable never folded");
1609        assert!(runs.join(ready_fresh).exists(), "fresh never folded");
1610        assert!(runs.join(garbage).exists(), "unreadable record kept");
1611        assert!(wt.join("0004").exists(), "unreadable worktree kept");
1612        assert!(
1613            runs.join(&due_ready).exists(),
1614            "folding drops worktrees, not the record"
1615        );
1616        assert!(
1617            runs.join(&due_old_schema).exists(),
1618            "an old-schema record survives its fold exactly like a current one"
1619        );
1620        // `graph::fold_run` saves the state it just folded back to disk. That
1621        // write must land under this test's own `runs` - the argument it
1622        // passed to `fold_due`, not the process-global `run::home` - or a
1623        // fold that landed somewhere else entirely would still be counted
1624        // above as one of the two `folded` runs.
1625        for id in [&due_ready, &due_old_schema] {
1626            let saved = read_meta(&runs, id).expect("folded run still parses");
1627            assert_ne!(
1628                saved.updated_at,
1629                ts("2026-08-01T00:00:00Z"),
1630                "fold_run must have saved the updated state back through the \
1631                 `runs` directory this test passed to fold_due"
1632            );
1633        }
1634    }
1635
1636    /// `[disk] auto_fold = false` must leave the janitor's fold-and-reclaim
1637    /// passes completely inert - a due run's worktree and record both
1638    /// survive exactly as if `housekeep` had never run at all. Cache pruning
1639    /// is a separate opt-out (`cache_limit_bytes`) and stays disabled here
1640    /// too, so this test is only ever about `auto_fold`.
1641    #[tokio::test]
1642    async fn housekeep_leaves_everything_alone_when_auto_fold_is_disabled() {
1643        let dir = tempfile::tempdir().unwrap();
1644        let runs = dir.path().join("runs");
1645        let wt = dir.path().join("wt");
1646        let home = dir.path().to_path_buf();
1647        crate::run::set_home(dir.path().to_path_buf());
1648
1649        let due_id = due_run(&runs, &wt, "20260801-000000-abcd", SCHEMA);
1650        std::fs::create_dir_all(wt.join("orphan").join("cand-A")).unwrap();
1651
1652        let mut cfg = crate::config::Config::default();
1653        cfg.disk.auto_fold = false;
1654        cfg.disk.cache_limit_bytes = 0;
1655
1656        let out = housekeep(&cfg, &home, &wt, &dir.path().join("repo"), Timestamp::now()).await;
1657
1658        assert_eq!(out.folded, 0);
1659        assert_eq!(out.unreadable, 0);
1660        assert_eq!(out.orphaned_worktrees, 0);
1661        assert!(
1662            runs.join(&due_id).exists(),
1663            "a due run's record survives untouched"
1664        );
1665        assert!(
1666            wt.join("orphan").exists(),
1667            "an orphaned worktree survives untouched: the reclaim pass never ran"
1668        );
1669    }
1670
1671    #[test]
1672    fn fold_orphaned_worktrees_removes_only_worktrees_no_run_claims_and_none_in_flight() {
1673        let dir = tempfile::tempdir().unwrap();
1674        let runs = dir.path().join("runs");
1675        let wt = dir.path().join("wt");
1676        let home = dir.path().to_path_buf();
1677
1678        // A run record exists for this one: its worktree is claimed, not
1679        // orphaned, however old the record.
1680        write_meta(
1681            &runs,
1682            "20260801-000000-aaaa",
1683            "ready",
1684            "2026-08-01T00:00:00Z",
1685        );
1686        std::fs::create_dir_all(wt.join("aaaa").join("cand-A")).unwrap();
1687
1688        // No run record at all, and nobody is working on it: this is the
1689        // leftover `fold_due` can never see, because it only ever walks
1690        // `runs/`.
1691        std::fs::create_dir_all(wt.join("bbbb").join("cand-A")).unwrap();
1692
1693        // No run record either, but a live daemon status names a run with
1694        // this short id - the save-timing gap between the daemon claiming a
1695        // task and `RunState::new` writing its first `run.json`. Must survive
1696        // untouched.
1697        std::fs::create_dir_all(wt.join("cccc")).unwrap();
1698
1699        // Not shaped like a run's short id at all - a scratch directory an
1700        // operator or another tool left in the same bay - so it is never a
1701        // reclaim target regardless of what runs claim it or not.
1702        std::fs::create_dir_all(wt.join("scratch")).unwrap();
1703
1704        // `now` pushed comfortably past `MIN_ORPHAN_AGE_SECS`, so a zero
1705        // grace - the same "always due" escape hatch `due` itself documents
1706        // - still reclaims once a worktree is genuinely old, without faking
1707        // an mtime: real directory creation just above is already in the
1708        // past relative to this `now`, by design rather than by timing.
1709        let now = Timestamp::now() + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1710        let mut status = crate::daemon::Status::new();
1711        status.current = vec![crate::daemon::Current {
1712            task: "20260905-000000-t111".to_owned(),
1713            run: "20260905-000000-cccc".to_owned(),
1714        }];
1715        status.updated_at = now;
1716        crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1717
1718        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1719        assert_eq!(
1720            folded, 1,
1721            "only the truly orphaned, idle, bay-shaped worktree is removed"
1722        );
1723        assert!(wt.join("aaaa").exists(), "claimed by a run record");
1724        assert!(!wt.join("bbbb").exists(), "orphaned and idle: reclaimed");
1725        assert!(wt.join("cccc").exists(), "a run in flight is never touched");
1726        assert!(
1727            wt.join("scratch").exists(),
1728            "not shaped like a worktree bay, so never a reclaim target"
1729        );
1730    }
1731
1732    /// The gap this closes: `Runner::review` (`magi review`) creates the
1733    /// worktree with `git worktree add` before `RunState::save` ever writes a
1734    /// `run.json`, and that path never runs through the daemon's own `poll`
1735    /// loop at all, so `daemon::Status` never names it either. Without a
1736    /// grace window, a janitor pass landing in that gap would read the
1737    /// worktree as an orphan nothing is waiting on and delete a review still
1738    /// being set up.
1739    #[test]
1740    fn fold_orphaned_worktrees_leaves_a_freshly_created_bay_alone() {
1741        let dir = tempfile::tempdir().unwrap();
1742        let runs = dir.path().join("runs");
1743        let wt = dir.path().join("wt");
1744        let home = dir.path().to_path_buf();
1745
1746        std::fs::create_dir_all(wt.join("dddd").join("under-review")).unwrap();
1747
1748        let now = Timestamp::now();
1749        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 6 * 60 * 60, now));
1750        assert_eq!(
1751            folded, 0,
1752            "too fresh to tell apart from a run still being set up"
1753        );
1754        assert!(wt.join("dddd").exists());
1755    }
1756
1757    /// A fresh open question on `run`, stored and handed back for assertions.
1758    fn open_question(store: &Questions, run: &str) -> crate::ask::Question {
1759        let mut q = crate::ask::Question::new(
1760            run.to_owned(),
1761            "implement".to_owned(),
1762            "impl-A".to_owned(),
1763            "Which storage backend should the cache use?".to_owned(),
1764            String::new(),
1765            vec!["SQLite".to_owned(), "Redis".to_owned()],
1766        );
1767        store.put(&mut q).unwrap();
1768        q
1769    }
1770
1771    /// The exact ghost the phone showed: a run that already finished, with a
1772    /// question its dead seat asked still sitting `open` because it reached
1773    /// that status before `graph::Runner::settle_questions` existed (or
1774    /// missed it in the crash window `daemon::reclaim_orphaned_running`
1775    /// covers). This sweep is the second door to the same fact.
1776    #[test]
1777    fn a_finished_runs_open_question_is_swept_up() {
1778        let dir = tempfile::tempdir().unwrap();
1779        let runs = dir.path().join("runs");
1780        let store = Questions::at(dir.path().join("questions"));
1781
1782        let failed = "20260908-205802-c9eb";
1783        write_meta(&runs, failed, "failed", "2026-09-08T20:58:02Z");
1784        let failed_q = open_question(&store, failed);
1785
1786        let merged = "20260908-205501-ca67";
1787        write_meta(&runs, merged, "merged", "2026-09-08T20:55:01Z");
1788        let merged_q = open_question(&store, merged);
1789
1790        let n = abandon_settled_questions(&store, &runs);
1791        assert_eq!(n, 2, "both dead runs' questions are swept in one pass");
1792
1793        for (id, run) in [(&failed_q.id, failed), (&merged_q.id, merged)] {
1794            let back = store.get(id).unwrap();
1795            assert!(!back.status.open(), "{run} is done; nobody reads an answer");
1796            assert!(back.detail.contains(run), "{}", back.detail);
1797        }
1798    }
1799
1800    #[test]
1801    fn a_still_alive_runs_open_question_survives_the_sweep() {
1802        let dir = tempfile::tempdir().unwrap();
1803        let runs = dir.path().join("runs");
1804        let store = Questions::at(dir.path().join("questions"));
1805
1806        // `Blocked` and `Stalled` are `RunStatus::resumable`: the run can
1807        // still be picked back up, so its question may yet get a real
1808        // answer. A run still mid-competition is even more obviously alive.
1809        for (id, status) in [
1810            ("20260908-000000-b10c", "blocked"),
1811            ("20260908-000000-5ta1", "stalled"),
1812            ("20260908-000000-jud6", "judging"),
1813        ] {
1814            write_meta(&runs, id, status, "2026-09-08T00:00:00Z");
1815            let q = open_question(&store, id);
1816
1817            let n = abandon_settled_questions(&store, &runs);
1818            assert_eq!(n, 0, "{status} run is not done; nothing to sweep");
1819            assert!(
1820                store.get(&q.id).unwrap().status.open(),
1821                "{status} run's question must still be waiting"
1822            );
1823        }
1824    }
1825
1826    #[test]
1827    fn the_sweep_leaves_an_answered_question_and_an_unreadable_run_alone() {
1828        let dir = tempfile::tempdir().unwrap();
1829        let runs = dir.path().join("runs");
1830        let store = Questions::at(dir.path().join("questions"));
1831
1832        // Already decided: a sweep must never revisit it, whatever the run
1833        // that asked went on to become.
1834        let done = "20260908-000000-answ";
1835        write_meta(&runs, done, "failed", "2026-09-08T00:00:00Z");
1836        let mut answered = open_question(&store, done);
1837        answered
1838            .answer(crate::ask::Answer::Choice("SQLite".to_owned()))
1839            .unwrap();
1840        store.put(&mut answered).unwrap();
1841
1842        // No `run.json` at all for this one - deleted, or never landed.
1843        let gone = "20260908-000000-gone";
1844        let orphan = open_question(&store, gone);
1845
1846        assert_eq!(abandon_settled_questions(&store, &runs), 0);
1847        assert_eq!(
1848            store.get(&answered.id).unwrap().status,
1849            crate::ask::QuestionStatus::Answered,
1850            "a real answer is never overwritten by a sweep"
1851        );
1852        assert!(
1853            store.get(&orphan.id).unwrap().status.open(),
1854            "a run this sweep cannot read is left exactly as it was, not guessed at"
1855        );
1856    }
1857
1858    /// A grace of `0` is a legitimate, documented value for the operator's
1859    /// own `Disk::fold_grace_secs` - `due`'s "always due" case - but the
1860    /// freshness check this guards is not that policy, and must not collapse
1861    /// to it: a `0` handed straight through would reclaim a worktree the
1862    /// instant it exists, exactly the race `fold_orphaned_worktrees_leaves_a_
1863    /// freshly_created_bay_alone` exists to rule out, just with the operator
1864    /// having turned the other grace off instead of leaving it at its
1865    /// default.
1866    #[test]
1867    fn fold_orphaned_worktrees_floors_a_zero_grace_at_the_race_safe_minimum() {
1868        let dir = tempfile::tempdir().unwrap();
1869        let runs = dir.path().join("runs");
1870        let wt = dir.path().join("wt");
1871        let home = dir.path().to_path_buf();
1872
1873        std::fs::create_dir_all(wt.join("eeee").join("under-review")).unwrap();
1874
1875        // Too fresh, even with the grace argument at zero.
1876        let now = Timestamp::now();
1877        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1878        assert_eq!(
1879            folded, 0,
1880            "a zero grace must not defeat the race-safety floor"
1881        );
1882        assert!(wt.join("eeee").exists());
1883
1884        // Once genuinely past the floor, a zero grace reclaims it - the
1885        // floor is a minimum, not a replacement policy that never fires.
1886        let later = now + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1887        let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, later));
1888        assert_eq!(folded, 1, "old enough now, regardless of the zero grace");
1889        assert!(!wt.join("eeee").exists());
1890    }
1891
1892    #[test]
1893    fn clear_abandoned_active_only_acts_once_dead_and_overrun() {
1894        let dir = tempfile::tempdir().unwrap();
1895        // Harmless if another test in this binary already pinned the global
1896        // home first (see `run::set_home`'s own doc): this test only checks
1897        // the in-memory mutation `clear_abandoned_active` makes, never a
1898        // write that landed under this exact directory.
1899        crate::run::set_home(dir.path().to_path_buf());
1900        let home = dir.path().to_path_buf();
1901        let now = ts("2026-09-14T12:00:00Z");
1902        let overrun_seat = || crate::run::ActiveSeat {
1903            node: "implement".to_owned(),
1904            started_at: now - SignedDuration::new(21_000, 0),
1905            timeout_secs: 3_600,
1906            attempt: 0,
1907            task: None,
1908            command: None,
1909            index: None,
1910            total: None,
1911        };
1912
1913        let mut state = RunState::new(
1914            PathBuf::from("/repo"),
1915            "main".to_owned(),
1916            "abc1234".to_owned(),
1917            "fixture".to_owned(),
1918            crate::config::Config::default(),
1919        );
1920        state.status = RunStatus::Implementing;
1921        state.active.insert("impl-A".to_owned(), overrun_seat());
1922
1923        // A seat still within its own budget: not provably dead yet, so this
1924        // must change nothing.
1925        let mut fresh = state.clone();
1926        fresh.active.insert(
1927            "impl-B".to_owned(),
1928            crate::run::ActiveSeat {
1929                node: "implement".to_owned(),
1930                started_at: now,
1931                timeout_secs: 3_600,
1932                attempt: 0,
1933                task: None,
1934                command: None,
1935                index: None,
1936                total: None,
1937            },
1938        );
1939        assert!(!clear_abandoned_active(&mut fresh, &home, now).unwrap());
1940        assert!(!fresh.active.is_empty());
1941        assert_eq!(fresh.status, RunStatus::Implementing);
1942
1943        let store = Questions::at(home.join("questions"));
1944        let q = open_question(&store, &state.id);
1945
1946        assert!(clear_abandoned_active(&mut state, &home, now).unwrap());
1947        assert!(state.active.is_empty());
1948        assert_eq!(state.status, RunStatus::Failed);
1949        assert!(
1950            !store.get(&q.id).unwrap().status.open(),
1951            "the abandoned seat's own open question must not keep badging the \
1952             operator until some later daemon startup notices it"
1953        );
1954    }
1955
1956    /// Write a whole `run.json` that magi can read, over the given state.
1957    fn write_meta(runs: &Path, id: &str, status: &str, updated_at: &str) {
1958        let day = &updated_at[..10];
1959        std::fs::create_dir_all(runs.join(id)).unwrap();
1960        let body = format!(
1961            r#"{{"schema": {SCHEMA}, "id": "{id}", "repo": "/nonexistent/repo", "base_branch": "main", "base_commit": "0000000000000000000000000000000000000000", "instruction": "", "created_at": "{day}T00:00:00Z", "updated_at": "{updated_at}", "status": "{status}", "seed": 1}}"#
1962        );
1963        std::fs::write(runs.join(id).join("run.json"), body).unwrap();
1964    }
1965
1966    /// Write a fully-formed, `Ready`, well-past-grace `run.json` tagged with
1967    /// an arbitrary schema number - so a test can write one this build's own
1968    /// `RunState::new` could never produce on its own. Returns the id.
1969    ///
1970    /// `wt` becomes this run's `graph.worktree_root`: left at the config
1971    /// default, `RunState::worktree_root` falls through to
1972    /// `run::default_worktree_root` - the operator's real `~/wt/magi` - and
1973    /// `graph::fold_run`'s second sweep would then `read_dir` and remove
1974    /// worktrees there instead of anything this test owns.
1975    fn due_run(runs: &Path, wt: &Path, id: &str, schema: u32) -> String {
1976        due_run_with_status(runs, wt, id, schema, RunStatus::Ready)
1977    }
1978
1979    /// [`due_run`], with the terminal status a caller wants instead of the
1980    /// `Ready` every other caller here happens to want.
1981    fn due_run_with_status(
1982        runs: &Path,
1983        wt: &Path,
1984        id: &str,
1985        schema: u32,
1986        status: RunStatus,
1987    ) -> String {
1988        let mut config = crate::config::Config::default();
1989        config.graph.worktree_root = Some(wt.to_path_buf());
1990        let mut state = RunState::new(
1991            PathBuf::from("/nonexistent/repo"),
1992            "main".to_owned(),
1993            "0000000000000000000000000000000000000000".to_owned(),
1994            String::new(),
1995            config,
1996        );
1997        state.id = id.to_owned();
1998        state.status = status;
1999        state.updated_at = ts("2026-08-01T00:00:00Z");
2000        let mut value = serde_json::to_value(&state).unwrap();
2001        value["schema"] = serde_json::json!(schema);
2002        std::fs::create_dir_all(runs.join(id)).unwrap();
2003        std::fs::write(
2004            runs.join(id).join("run.json"),
2005            serde_json::to_string_pretty(&value).unwrap(),
2006        )
2007        .unwrap();
2008        id.to_owned()
2009    }
2010
2011    #[test]
2012    fn fold_due_folds_a_superseded_run_same_as_any_other_terminal_one() {
2013        let dir = tempfile::tempdir().unwrap();
2014        let runs = dir.path().join("runs");
2015        let wt = dir.path().join("wt");
2016        let home = dir.path().to_path_buf();
2017        let disk = Disk::default();
2018        let now = ts("2026-09-05T00:00:00Z");
2019
2020        let superseded = due_run_with_status(
2021            &runs,
2022            &wt,
2023            "20260801-000000-cccc",
2024            SCHEMA,
2025            RunStatus::Superseded,
2026        );
2027
2028        let (folded, unreadable) =
2029            block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
2030        assert_eq!(
2031            folded, 1,
2032            "a superseded run has nothing left for a human to check, so it folds \
2033             exactly like a merged one"
2034        );
2035        assert_eq!(unreadable, 0);
2036        assert!(
2037            read_meta(&runs, &superseded).is_ok(),
2038            "folding drops the worktree, not the record"
2039        );
2040    }
2041
2042    /// A throwaway repo with one commit on `main`, for a test that needs a
2043    /// real winner worktree `fold_run` can actually remove.
2044    fn init_repo(dir: &Path) {
2045        use crate::proc::Quiet as _;
2046        let run = |args: &[&str]| {
2047            let out = std::process::Command::new("git")
2048                .args(args)
2049                .current_dir(dir)
2050                .quiet()
2051                .output()
2052                .expect("spawn git");
2053            assert!(
2054                out.status.success(),
2055                "git {args:?} failed: {}",
2056                String::from_utf8_lossy(&out.stderr)
2057            );
2058        };
2059        run(&["init", "-b", "main"]);
2060        run(&["config", "user.name", "magi test"]);
2061        run(&["config", "user.email", "magi@example.com"]);
2062        std::fs::write(dir.join("README.md"), "# fixture\n").unwrap();
2063        run(&["add", "-A"]);
2064        run(&["commit", "-m", "init"]);
2065    }
2066
2067    /// A due, terminal run with a decided winner that still has a real
2068    /// worktree and branch on disk - so a test can tell whether folding it
2069    /// actually removed the winner or only left it registered as folded.
2070    async fn due_run_with_winner_worktree(
2071        runs: &Path,
2072        wt: &Path,
2073        repo: &Path,
2074        id: &str,
2075        status: RunStatus,
2076    ) -> (String, PathBuf) {
2077        use crate::run::{Candidate, Tally};
2078        use std::collections::BTreeMap;
2079
2080        let mut config = crate::config::Config::default();
2081        config.graph.worktree_root = Some(wt.to_path_buf());
2082        let mut state = RunState::new(
2083            repo.to_path_buf(),
2084            "main".to_owned(),
2085            "0000000000000000000000000000000000000000".to_owned(),
2086            String::new(),
2087            config,
2088        );
2089        state.id = id.to_owned();
2090        state.status = status;
2091        state.updated_at = ts("2026-08-01T00:00:00Z");
2092
2093        let winner_wt = state.worktree_root().join("cand-A");
2094        crate::git::worktree_add_branch(repo, &winner_wt, &format!("magi/{id}/A"), "main")
2095            .await
2096            .expect("winner worktree");
2097
2098        state.candidates.push(Candidate {
2099            index: 0,
2100            label: 'A',
2101            agent: "agent".to_owned(),
2102            branch: format!("magi/{id}/A"),
2103            worktree: winner_wt.clone(),
2104            summary: String::new(),
2105            stat: String::new(),
2106            files: 0,
2107            commits: 0,
2108            empty: false,
2109            failed: None,
2110            verified_noop: None,
2111            duration_ms: 0,
2112            folded: false,
2113        });
2114        state.tally = Some(Tally {
2115            first_choice: BTreeMap::from([('A', 1)]),
2116            borda: BTreeMap::new(),
2117            winner: 'A',
2118            rankings: 1,
2119            unanimous_initial: true,
2120            deliberated: false,
2121            changed_votes: 0,
2122            unanimous_final: true,
2123            tie_break: None,
2124            judges: 1,
2125            present: 1,
2126            quorum: 1,
2127            met_quorum: true,
2128            uncontested: None,
2129        });
2130        std::fs::create_dir_all(runs.join(id)).unwrap();
2131        std::fs::write(
2132            runs.join(id).join("run.json"),
2133            serde_json::to_string_pretty(&state).unwrap(),
2134        )
2135        .unwrap();
2136        (id.to_owned(), winner_wt)
2137    }
2138
2139    #[tokio::test]
2140    async fn fold_due_drops_a_superseded_runs_own_winner_worktree_but_keeps_a_readys() {
2141        let dir = tempfile::tempdir().unwrap();
2142        let runs = dir.path().join("runs");
2143        let wt = dir.path().join("wt");
2144        let repo = dir.path().join("repo");
2145        let home = dir.path().to_path_buf();
2146        let disk = Disk::default();
2147        let now = ts("2026-09-05T00:00:00Z");
2148        std::fs::create_dir_all(&repo).unwrap();
2149        init_repo(&repo);
2150
2151        let (superseded_id, superseded_wt) = due_run_with_winner_worktree(
2152            &runs,
2153            &wt,
2154            &repo,
2155            "20260801-000000-supw",
2156            RunStatus::Superseded,
2157        )
2158        .await;
2159        // `Ready` is auto-folded exactly like `Superseded` (both terminal and
2160        // non-resumable), but never landed anywhere - its own winner is
2161        // deliberately kept, which is what makes it the right contrast here.
2162        // `Blocked`/`Stalled` are `resumable()` and so never even reach
2163        // `fold_run` in the first place; they would not exercise the
2164        // `drop_winner` choice this test is about.
2165        let (ready_id, ready_wt) = due_run_with_winner_worktree(
2166            &runs,
2167            &wt,
2168            &repo,
2169            "20260801-000000-rdyw",
2170            RunStatus::Ready,
2171        )
2172        .await;
2173
2174        let (folded, unreadable) = fold_due(&runs, &home, &wt, &disk, now)
2175            .await
2176            .expect("fold_due");
2177        assert_eq!(folded, 2);
2178        assert_eq!(unreadable, 0);
2179
2180        assert!(
2181            !superseded_wt.exists(),
2182            "a superseded run's own winner never lands anywhere else, so its worktree \
2183             must be dropped exactly like a merged run's"
2184        );
2185        assert!(
2186            ready_wt.exists(),
2187            "a still-ready run's winner may yet be merged by hand - folding must not \
2188             touch it"
2189        );
2190
2191        // Both records survive the fold; only the worktrees differ.
2192        assert!(read_meta(&runs, &superseded_id).is_ok());
2193        assert!(read_meta(&runs, &ready_id).is_ok());
2194    }
2195}