magi/clean.rs
1//! The disk janitor: finished runs get their worktrees folded, worktrees whose
2//! run record is already gone get reclaimed too, and the shared build cache is
3//! pruned to its cap.
4//!
5//! A run's state being written by an older schema is not the same thing as it
6//! being unreadable, and this module used to conflate the two: [`fold_due`]
7//! treated any `run.json` its version check rejected exactly like one that
8//! failed to parse at all, so a single schema bump silently stopped every
9//! automatic fold in the fleet the moment it shipped, and did so with no
10//! counter and no log line to say so. A record magi genuinely cannot parse —
11//! missing fields, broken JSON, a schema newer than this build has ever heard
12//! of — is still left alone here, still counted in
13//! [`Housekeeping::unreadable`], and still only ever removed by an explicit
14//! operator action (`magi fold`, or the equivalent phone route). One written
15//! by a schema this build merely disagrees with the *meaning* of is not that:
16//! as long as it still parses, folding proceeds regardless of the number in
17//! its `schema` field.
18//!
19//! Everything policy-shaped — which statuses are foldable, how long a finished
20//! run is left alone, whether the cache is over its limit — is a pure function
21//! injected with numbers, so nothing here has to ask the operating system to
22//! be testable. The only I/O is the removal itself.
23
24use std::collections::BTreeSet;
25use std::path::{Path, PathBuf};
26
27use anyhow::{Context as _, Result, bail};
28use jiff::{SignedDuration, Timestamp};
29use serde::Deserialize;
30
31use crate::ask::Questions;
32use crate::config::Disk;
33use crate::run::{RunState, RunStatus, SCHEMA, short_of};
34
35use crate::disk::{Prune, dir_size};
36
37/// What one janitor pass did, for the caller's log line.
38#[derive(Debug, Clone, Copy, Default, PartialEq, Eq)]
39pub struct Housekeeping {
40 /// Runs folded (worktrees dropped).
41 pub folded: usize,
42 /// Runs [`fold_due`] left alone because their `run.json` genuinely could
43 /// not be read - missing, broken JSON, a schema this build has never
44 /// heard of - as opposed to one merely written by a different schema
45 /// number, which is folded like any other (see the module docs). This was
46 /// defined but never incremented for a long stretch of this module's
47 /// history, which is exactly how 90 of 93 runs sat unfolded on one
48 /// operator's machine with nothing anywhere saying why: every one of them
49 /// was misclassified as unreadable by a schema check that has since been
50 /// narrowed to only the runs that actually are.
51 pub unreadable: usize,
52 /// Worktrees under the worktree bay reclaimed because no run record in
53 /// `runs/` claims them anymore (see [`fold_orphaned_worktrees`]).
54 pub orphaned_worktrees: usize,
55 /// Files dropped from the shared cache.
56 pub cache_files: usize,
57 /// Bytes freed from the shared cache.
58 pub cache_freed: u64,
59 /// Open questions abandoned because the run that asked them has already
60 /// settled where nothing is coming back to read an answer.
61 pub questions_abandoned: usize,
62 /// Runs found blocked on a merge magi never recorded — a pull request the
63 /// operator merged by hand while `land::land` never got as far as
64 /// opening one itself — and corrected automatically. See
65 /// [`reconcile_external_merges`].
66 pub external_merges_recorded: usize,
67 /// Run records that still called their pull request open after it was
68 /// merged or closed, and were rewritten (see
69 /// [`crate::land::repair_stale_pr_states`]).
70 pub stale_pr_states_repaired: usize,
71}
72
73/// Run the janitor: fold due runs, reclaim orphaned worktrees, prune stale
74/// worktree registrations, then prune the cache if it is over its cap.
75///
76/// Every part is best-effort; a jammed cache lock or a run whose worktree
77/// another borrower holds must not stop the rest. Errors are reported through
78/// `tracing::warn` - this is housekeeping, and the daemon keeps serving
79/// either way.
80pub async fn housekeep(
81 cfg: &crate::config::Config,
82 home: &Path,
83 worktrees_root: &Path,
84 repo: &Path,
85 now: Timestamp,
86) -> Housekeeping {
87 let mut out = Housekeeping::default();
88 if cfg.disk.auto_fold {
89 let runs = home.join("runs");
90 match fold_due(&runs, home, worktrees_root, &cfg.disk, now).await {
91 Ok((folded, unreadable)) => {
92 out.folded = folded;
93 out.unreadable = unreadable;
94 }
95 Err(e) => {
96 tracing::warn!("housekeep: fold due runs: {e:#}");
97 // Stable wording: the reason varies per pass, and a changed
98 // message would relight the bell on every retry.
99 crate::notices::raise_in(
100 home,
101 crate::notices::Notice::warn(
102 "housekeep:fold",
103 "Automatic cleanup of finished runs failed; disk usage may keep growing.",
104 ),
105 );
106 }
107 }
108 out.orphaned_worktrees =
109 fold_orphaned_worktrees(&runs, worktrees_root, home, cfg.disk.fold_grace_secs, now)
110 .await;
111 // Best-effort in the same sense as everything else here: a repository
112 // this janitor pass has nothing to do with (or none at all, in a unit
113 // test) must not turn a `warn` into a reason to skip the rest.
114 if let Err(e) = crate::git::worktree_prune(repo).await {
115 tracing::warn!("housekeep: prune worktree registrations: {e:#}");
116 }
117 out.external_merges_recorded = reconcile_external_merges(&runs, home, &cfg.disk, now).await;
118 // Bounded: a handful of forge lookups a pass keeps this cheap, and the
119 // pass is idempotent, so a backlog drains over several passes.
120 out.stale_pr_states_repaired = crate::land::repair_stale_pr_states(home, 5).await.0;
121 }
122 match prune_cache_if_over_limit(cfg, home) {
123 Ok(Some(pruned)) => {
124 out.cache_files = pruned.files;
125 out.cache_freed = pruned.freed;
126 }
127 Ok(None) => {}
128 Err(e) => tracing::warn!("housekeep: prune cache: {e:#}"),
129 }
130 // Unconditional, unlike the two passes above: this is not a disk policy
131 // with a cap or an opt-out, it is closing a gap `graph::Runner` itself
132 // cannot - a run that reached `Merged`/`Ready`/`Failed` before this
133 // cleanup existed, or whose process died between saving that status and
134 // abandoning the question it leaves behind (see `Runner::settle_questions`).
135 // Left alone, that question sits `open` forever: the owner's badge,
136 // banner and title all keep counting a decision nobody is left to read.
137 out.questions_abandoned =
138 abandon_settled_questions(&Questions::at(home.join("questions")), &home.join("runs"));
139 out
140}
141
142/// Abandon every open question whose run has already settled into a status
143/// nothing comes back from, worded with what the run became - the same
144/// cleanup `graph::Runner::settle_questions` runs the moment `status` lands
145/// there, for questions that missed it.
146///
147/// Scans questions rather than runs: the open list is normally short, and a
148/// run that never asked anything costs nothing here. A run this cannot read,
149/// deleted or written by a schema this build does not speak, is left alone
150/// the same as everywhere else in this module; the question stays open
151/// rather than guessed at.
152pub fn abandon_settled_questions(store: &Questions, runs: &Path) -> usize {
153 let waiting_on: BTreeSet<String> = store
154 .list()
155 .into_iter()
156 .filter(|q| q.status.open())
157 .map(|q| q.run)
158 .collect();
159 let mut abandoned = 0;
160 for run in waiting_on {
161 let Ok(meta) = read_meta(runs, &run) else {
162 continue;
163 };
164 match store.settle_run(&run, meta.status) {
165 Ok(n) => abandoned += n,
166 Err(e) => tracing::warn!("housekeep: abandon questions for {run}: {e:#}"),
167 }
168 }
169 abandoned
170}
171
172/// Fold every run that is finished, older than the grace period, and not being
173/// worked on; return `(folded, unreadable)`.
174///
175/// A run whose `run.json` genuinely cannot be parsed — missing fields, broken
176/// JSON, a schema newer than this build has ever heard of — is left exactly
177/// as it is. Automatic housekeeping cannot tell a mid-write file from one that
178/// will never parse again, and `<home>/runs/<id>/` is the evidence `magi
179/// stats` and the deck read; when unsure whether it is safe to touch, the
180/// janitor keeps rather than deletes (see the module docs). Discarding a
181/// record this unreadable is an explicit operator action (`magi fold`, or the
182/// equivalent phone route), never something that happens unattended. Every
183/// such skip is counted in the returned `unreadable` and logged through
184/// `tracing::warn` with the parse failure that caused it - silence here is
185/// exactly the failure mode that let 90 of 93 runs sit unfolded with nothing
186/// to show for it.
187///
188/// A run merely written by a *different* schema number is not unreadable: as
189/// long as `run.json` still parses, it folds like any other terminal run (see
190/// the module docs for why the two are different questions).
191///
192/// Runnable statuses and runs newer than the grace period are also left
193/// alone; folding them would throw away work that is still the answer to
194/// somebody's question. `Merged` runs forget their winner's worktree (the
195/// merge already landed it); `Ready` and `Failed` runs keep it.
196pub async fn fold_due(
197 runs: &Path,
198 home: &Path,
199 _worktrees_root: &Path,
200 disk: &Disk,
201 now: Timestamp,
202) -> Result<(usize, usize)> {
203 let mut folded = 0usize;
204 let mut unreadable = 0usize;
205 let mut ids: Vec<String> = std::fs::read_dir(runs)
206 .into_iter()
207 .flatten()
208 .flatten()
209 .filter(|e| e.path().join("run.json").is_file())
210 .map(|e| e.file_name().to_string_lossy().into_owned())
211 .collect();
212 ids.sort_unstable();
213 for id in ids {
214 if crate::daemon::is_working_on(home, &id, now) {
215 continue;
216 }
217 let meta = match read_meta(runs, &id) {
218 Ok(meta) => meta,
219 Err(e) => {
220 unreadable += 1;
221 tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
222 continue;
223 }
224 };
225 if meta.status.resumable() || !due(now, meta.updated_at, disk.fold_grace_secs) {
226 continue;
227 }
228 // `read_meta` already proved the file parses; `read_state` asks for
229 // the rest of the fields `graph::fold_run` needs (worktree paths,
230 // candidates, tally). A schema mismatch alone does not fail this -
231 // see the module docs - so reaching `Err` here means the JSON itself
232 // is broken in a way `read_meta` did not exercise, which is rare but
233 // not impossible (a body truncated between the two fields it reads
234 // and the rest). That must not cost every other run its turn through
235 // this loop, so it is a skip, not a `?`.
236 let mut state = match read_state(runs, &id) {
237 Ok(state) => state,
238 Err(e) => {
239 unreadable += 1;
240 tracing::warn!("housekeep: run {id} unreadable, left alone: {e:#}");
241 continue;
242 }
243 };
244 if state.schema != SCHEMA {
245 tracing::info!(
246 "housekeep: run {id} was written by schema {} (this build speaks {SCHEMA}); \
247 folding it anyway",
248 state.schema
249 );
250 }
251 // `Merged`'s winner branch is safe to drop because it already landed
252 // in the real target branch; `Superseded`'s winner branch is safe to
253 // drop for a different reason - a later attempt at the same task is
254 // what actually landed (or didn't), so *this* run's own winner is
255 // never going to be merged by anyone. Every other terminal status
256 // keeps the winner: `Ready`/`Blocked`/`Stalled`/`Failed`/`VerifiedNoop`
257 // may still have a human's decision pending on that exact branch.
258 // `AlreadyInBase` is the same: the change is on the base under other
259 // commit ids, so the branch holds nothing anyone still needs.
260 let drop_winner = matches!(
261 state.status,
262 RunStatus::Merged | RunStatus::Superseded | RunStatus::AlreadyInBase
263 );
264 // One run's fold must not cost every later run its turn. A worktree
265 // another borrower holds, a branch git refuses to delete, a repository
266 // that has since moved: each is a reason this run cannot be folded
267 // now, and none is a reason to stop the pass. Left unfolded, it is
268 // simply due again next time; a `?` here stopped automatic folding
269 // permanently at the first such run (finding R3-1-1 of run 51a3).
270 match crate::graph::fold_run(&mut state, drop_winner, home).await {
271 Ok(_) => folded += 1,
272 Err(e) => tracing::warn!("housekeep: fold {id}: {e:#}"),
273 }
274 }
275 Ok((folded, unreadable))
276}
277
278/// Is `updated` old enough, measured against `now`, that the run may fold?
279///
280/// Pure; the janitor compares against wallclock, tests inject both sides. The
281/// comparison is strict, so a run exactly at the edge of its grace period is
282/// left alone one more pass — the same convention as [`crate::disk::over_limit`].
283pub fn due(now: Timestamp, updated: Timestamp, grace_secs: u64) -> bool {
284 now.duration_since(updated) > SignedDuration::new(grace_secs as i64, 0)
285}
286
287/// The two fields the janitor decides on, read with a serde that tolerates
288/// everything else about the run being unreadable.
289#[derive(Deserialize)]
290struct Meta {
291 status: RunStatus,
292 updated_at: Timestamp,
293}
294
295/// Read `status` and `updated_at` straight off the state file, asking for
296/// nothing else. `Err` when the file is missing, not parseable, or a status in
297/// a version this build does not speak - all of which mean "unreadable".
298fn read_meta(runs: &Path, id: &str) -> Result<Meta> {
299 let path = runs.join(id).join("run.json");
300 let body =
301 std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
302 let meta: Meta =
303 serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
304 Ok(meta)
305}
306
307/// Runs checked against GitHub in one janitor pass, at most.
308///
309/// A fleet with many stuck runs must not turn every idle tick into a burst of
310/// `gh pr list` calls; a run left unchecked this pass is checked again next
311/// pass, same as an unfolded due run is.
312const MAX_EXTERNAL_MERGE_CHECKS_PER_PASS: usize = 5;
313
314/// The fields [`reconcile_external_merges`] filters on before paying for a
315/// full [`RunState`] parse or a `gh` round trip.
316#[derive(Deserialize)]
317struct ExternalMergeMeta {
318 status: RunStatus,
319 updated_at: Timestamp,
320 #[serde(default)]
321 merge: Option<crate::run::MergeOutcome>,
322}
323
324fn read_external_merge_meta(runs: &Path, id: &str) -> Result<ExternalMergeMeta> {
325 let path = runs.join(id).join("run.json");
326 let body =
327 std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
328 let meta: ExternalMergeMeta =
329 serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
330 Ok(meta)
331}
332
333/// Is a run worth asking GitHub about?
334///
335/// Pure, so the filter itself is assertable without a fixture that writes
336/// `run.json` files or spawns `gh`. Narrower than "blocked": `merge_recorded`
337/// excludes a run `land::land` already finished landing (successfully or
338/// not) - that path already carries its own record of what happened - and
339/// `due` gives an operator who is about to fix a run by hand a grace window
340/// before the janitor starts asking the same question about it.
341pub fn eligible_for_external_merge_check(
342 status: RunStatus,
343 merge_recorded: bool,
344 updated_at: Timestamp,
345 now: Timestamp,
346 grace_secs: u64,
347) -> bool {
348 status == RunStatus::Blocked && !merge_recorded && due(now, updated_at, grace_secs)
349}
350
351/// Rotate `ids` so a fixed per-pass cap does not always land on the same
352/// prefix of the list, starting at `start` (already reduced modulo nothing -
353/// callers pass whatever a persisted cursor last recorded, and this takes it
354/// modulo `ids.len()` itself).
355///
356/// Pure: the offset is the caller's problem, not wall-clock time. An earlier
357/// version derived `start` from `now.as_second() % ids.len()`, which looked
358/// independent of any state to persist but is not actually independent of
359/// how often the janitor runs - a poll interval and a pass's own duration
360/// that alias onto the same handful of `now.as_second()` values (say, a
361/// ten-second poll and a pass that finishes in under a second) revisit only
362/// those same few residues forever and never advance into the rest of the
363/// list at all. [`reconcile_external_merges`] instead persists a cursor
364/// across passes ([`read_external_merge_cursor`] /
365/// [`write_external_merge_cursor`]) and advances it by exactly how many ids
366/// this pass actually looked at, which is the only quantity that is
367/// guaranteed to move forward pass over pass regardless of timing.
368fn rotate_from(ids: &[String], start: usize) -> Vec<String> {
369 if ids.is_empty() {
370 return Vec::new();
371 }
372 let start = start % ids.len();
373 ids[start..]
374 .iter()
375 .chain(ids[..start].iter())
376 .cloned()
377 .collect()
378}
379
380/// Where [`reconcile_external_merges`] remembers how far it got, so a fixed
381/// per-pass cap advances through a fleet's eligible runs pass over pass
382/// instead of camping on whichever ones happen to sort first.
383const EXTERNAL_MERGE_CURSOR_FILE: &str = "external-merge-cursor";
384
385/// `0` for anything this cannot read as a plain number - a first run, a
386/// corrupt or missing file, a build that has never written one - which is
387/// exactly as good a starting point as any: the cursor's whole job is to
388/// keep moving, not to encode any particular position as meaningful.
389fn read_external_merge_cursor(home: &Path) -> usize {
390 std::fs::read_to_string(home.join(EXTERNAL_MERGE_CURSOR_FILE))
391 .ok()
392 .and_then(|s| s.trim().parse().ok())
393 .unwrap_or(0)
394}
395
396/// Best-effort, like every other write in this module's sweep: a failure to
397/// persist the cursor costs the fleet one pass's worth of forward progress
398/// on this sweep, never the run whose housekeeping it is racing to keep up
399/// with.
400fn write_external_merge_cursor(home: &Path, cursor: usize) {
401 if let Err(e) = std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), cursor.to_string()) {
402 tracing::warn!("housekeep: persist the external-merge sweep's cursor: {e:#}");
403 }
404}
405
406/// Close the gap `magi fold --merged` exists for, without waiting for an
407/// operator to notice and go find the pull request URL: a run that stopped
408/// `Blocked` with no `merge` recorded may have been merged anyway, by hand,
409/// on a pull request `land::land` never opened or never got to observe as
410/// merged. `land::find_external_merge` asks GitHub about the run's own
411/// winning branch, and only a pull request it can uniquely tie back to this
412/// run is ever acted on - see that function's own doc for what "uniquely"
413/// excludes.
414///
415/// A run this finds and fixes goes through
416/// [`crate::land::correct_confirmed_external_merge`] - the same correction
417/// `magi fold --merged` performs, minus the same-repo guard that command
418/// needs and this loop doesn't (see that function's own doc: the URL here
419/// was never operator-supplied, it came from asking `state.repo`'s own
420/// remote) - then through [`crate::graph::fold_run`] so it stops holding
421/// worktrees the moment it stops needing them. A run this cannot decide
422/// about - `gh` unreachable, more than one candidate pull
423/// request, nothing found at all - is left exactly as it is; only an error
424/// asking GitHub raises a notice, since "nothing found" is the ordinary,
425/// expected shape of a run that really is just blocked.
426async fn reconcile_external_merges(runs: &Path, home: &Path, disk: &Disk, now: Timestamp) -> usize {
427 let mut ids: Vec<String> = std::fs::read_dir(runs)
428 .into_iter()
429 .flatten()
430 .flatten()
431 .filter(|e| e.path().join("run.json").is_file())
432 .map(|e| e.file_name().to_string_lossy().into_owned())
433 .collect();
434 ids.sort_unstable();
435
436 // Every eligible id first, cheaply (a `Meta` read, no `gh`), then rotated
437 // before the cap is applied. Capping the sorted order directly would
438 // always land on the same lexicographic prefix - a fleet's oldest run ids
439 // - so a handful of long-blocked runs that never turn out to be merged
440 // would starve every id after them of a `gh` check, forever.
441 let mut eligible: Vec<String> = Vec::new();
442 for id in ids {
443 if crate::daemon::is_working_on(home, &id, now) {
444 continue;
445 }
446 let meta = match read_external_merge_meta(runs, &id) {
447 Ok(meta) => meta,
448 // Already counted as unreadable by `fold_due`'s own pass; not
449 // worth a second warning for the same file.
450 Err(_) => continue,
451 };
452 if eligible_for_external_merge_check(
453 meta.status,
454 meta.merge.is_some(),
455 meta.updated_at,
456 now,
457 disk.fold_grace_secs,
458 ) {
459 eligible.push(id);
460 }
461 }
462
463 if eligible.is_empty() {
464 return 0;
465 }
466 let cursor = read_external_merge_cursor(home);
467 let window: Vec<String> = rotate_from(&eligible, cursor)
468 .into_iter()
469 .take(MAX_EXTERNAL_MERGE_CHECKS_PER_PASS)
470 .collect();
471 // Advances by exactly how many ids this pass looked at, regardless of
472 // what came of looking - a run that turned out not to be merged after
473 // all must not be revisited before every other eligible run has had its
474 // own turn.
475 write_external_merge_cursor(home, (cursor + window.len()) % eligible.len());
476
477 let mut reconciled = 0usize;
478 for id in window {
479 let mut state = match read_state(runs, &id) {
480 Ok(state) => state,
481 Err(_) => continue,
482 };
483 match crate::land::find_external_merge(&state).await {
484 Ok(Some(found)) => {
485 match crate::land::correct_confirmed_external_merge(&mut state, &found.url).await {
486 Ok(_) => {
487 if let Err(e) = crate::graph::fold_run(&mut state, true, home).await {
488 tracing::warn!(
489 "housekeep: fold {id} after recording its external merge: {e:#}"
490 );
491 }
492 reconciled += 1;
493 }
494 Err(e) => tracing::warn!(
495 "housekeep: record external merge {} for {id}: {e:#}",
496 found.url
497 ),
498 }
499 }
500 Ok(None) => {}
501 Err(e) => {
502 tracing::warn!("housekeep: check external merge for {id}: {e:#}");
503 crate::notices::raise_in(
504 home,
505 crate::notices::Notice::warn(
506 &format!("merged-unrecorded:{id}"),
507 format!(
508 "Run {id} is blocked with no recorded merge, and checking GitHub \
509 for a merge failed; check by hand."
510 ),
511 ),
512 );
513 }
514 }
515 }
516 reconciled
517}
518
519/// Read a whole run state from a runs directory, for folding only.
520///
521/// Deliberately more permissive than [`RunState::load`], which this does not
522/// call: `load` backs `--resume` and every hand-driven command, where a
523/// schema this build disagrees with the *meaning* of must refuse outright
524/// rather than resume a review round or a tally against stale semantics
525/// (`RunState::SCHEMA`'s own docs list what has changed meaning at each
526/// bump). Folding recomputes nothing - it only reads worktree paths, branch
527/// names and a tally winner off the struct to remove them - so an old
528/// schema's values are exactly as good here as a current one's; every schema
529/// bump so far has only ever added a field or a variant, never repurposed an
530/// existing one, and serde already fills an added field's default when an
531/// older record has nothing to say about it. What this cannot tolerate, and
532/// what still surfaces as an `Err`, is `run.json` failing to parse at all.
533fn read_state(runs: &Path, id: &str) -> Result<RunState> {
534 let path = runs.join(id).join("run.json");
535 let body =
536 std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
537 let state: RunState =
538 serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
539 Ok(state)
540}
541
542/// Remove a run that cannot be read: its state directory under `runs` and its
543/// worktree directory under `worktrees_root`.
544///
545/// The state file is the only record of a run's repository and branches, so a
546/// run this unreadable is discarded at the filesystem level - there is no
547/// candidate list to fold first. The worktrees live under
548/// [`crate::run::default_worktree_root`] unless the run's config relocated
549/// them, which an unreadable run cannot tell us; the default location is
550/// removed, and anything the run placed elsewhere is a leftover for whoever
551/// knows where it went.
552///
553/// Deleting a worktree directory by hand leaves its registration in git, and a
554/// registered path cannot be re-`worktree add`-ed until it is pruned - so every
555/// worktree is unregistered from its repository first, best-effort, via the
556/// `gitdir:` link git keeps inside the directory.
557pub async fn fold_unreadable(runs: &Path, worktrees_root: &Path, id: &str) -> Result<Vec<String>> {
558 let resolved = resolve_id_path(runs, id)?;
559 let mut removed = Vec::new();
560 let run_dir = runs.join(&resolved);
561 if run_dir.exists() {
562 std::fs::remove_dir_all(&run_dir)
563 .with_context(|| format!("remove {}", run_dir.display()))?;
564 removed.push(format!("runs/{resolved}"));
565 }
566 let wt = worktrees_root.join(short_of(&resolved));
567 if wt.exists() {
568 crate::git::remove_worktree_from_linked(&wt).await;
569 for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
570 crate::git::remove_worktree_from_linked(&e.path()).await;
571 }
572 std::fs::remove_dir_all(&wt).with_context(|| format!("remove {}", wt.display()))?;
573 removed.push(wt.to_string_lossy().into_owned());
574 }
575 Ok(removed)
576}
577
578/// Reclaim worktrees under `worktrees_root` that no run record in `runs`
579/// claims anymore, and return how many were removed.
580///
581/// [`fold_due`] only ever sees a worktree by walking `runs/` first, so a
582/// worktree whose run record is already gone — `magi run rm`, or a record
583/// deleted before its worktree — never enters that loop at all: nothing there
584/// is looking for it. This walks the worktree bay directly instead, and
585/// removes any `<short>` directory that no run id maps to.
586///
587/// Two things must never happen, and this checks both before ever touching a
588/// directory:
589///
590/// - **A worktree bay is never the only kind of thing under `worktrees_root`,
591/// and this must not assume it is.** A hand-placed scratch directory, or
592/// anything else an operator or another tool left in the same bay, has the
593/// same "no run claims it" shape as a genuine orphan but is not one -
594/// [`looks_like_a_worktree_bay`] is the same tag shape [`crate::run::is_run_id`]
595/// already requires of a real run's short id, and anything else is left
596/// alone regardless of what else is true about it.
597/// - **A worktree that was only just created might not have a `run.json` yet
598/// for a reason that has nothing to do with being orphaned.** `Runner::start`
599/// and `Runner::review` both create the worktree before the first
600/// `RunState::save` lands, and that gap - several `git` subprocesses wide -
601/// is invisible to [`crate::daemon::is_working_on_short`] whenever the run
602/// is not being driven through this daemon's own `poll` loop at all (a
603/// `magi review` invocation, for one). A directory whose own modification
604/// time is within `grace_secs` of `now` is left alone on that basis alone,
605/// the same margin [`fold_due`] gives a run before treating it as truly
606/// finished - long enough that no realistic gap between a `worktree add`
607/// and its `run.json` could ever be mistaken for one.
608///
609/// The one failure this must never cause is deleting the worktree of a run
610/// that is genuinely in flight. [`crate::daemon::is_working_on_short`] is the
611/// same liveness check [`fold_due`] trusts everywhere else in this module,
612/// checked by short id because there is no full id to compare here; when it
613/// cannot tell, this leaves the directory alone. Best-effort like the rest of
614/// housekeeping: one directory git or the filesystem refuses to give up is a
615/// `tracing::warn`, not a reason to abandon the rest of the pass.
616pub async fn fold_orphaned_worktrees(
617 runs: &Path,
618 worktrees_root: &Path,
619 home: &Path,
620 grace_secs: u64,
621 now: Timestamp,
622) -> usize {
623 let known: std::collections::HashSet<String> = std::fs::read_dir(runs)
624 .into_iter()
625 .flatten()
626 .flatten()
627 .map(|e| e.file_name().to_string_lossy().into_owned())
628 .filter(|name| crate::run::is_run_id(name))
629 .map(|id| short_of(&id).to_owned())
630 .collect();
631
632 let mut folded = 0usize;
633 for entry in std::fs::read_dir(worktrees_root)
634 .into_iter()
635 .flatten()
636 .flatten()
637 {
638 if !entry.path().is_dir() {
639 continue;
640 }
641 let short = entry.file_name().to_string_lossy().into_owned();
642 if !looks_like_a_worktree_bay(&short) {
643 continue;
644 }
645 if known.contains(&short) || crate::daemon::is_working_on_short(home, &short, now) {
646 continue;
647 }
648 let wt = entry.path();
649 if !stale_enough(&wt, grace_secs, now) {
650 continue;
651 }
652 crate::git::remove_worktree_from_linked(&wt).await;
653 for e in std::fs::read_dir(&wt).into_iter().flatten().flatten() {
654 crate::git::remove_worktree_from_linked(&e.path()).await;
655 }
656 match std::fs::remove_dir_all(&wt) {
657 Ok(()) => folded += 1,
658 Err(e) => tracing::warn!(
659 "housekeep: remove orphaned worktree {}: {e:#}",
660 wt.display()
661 ),
662 }
663 }
664 folded
665}
666
667/// Does `name` have the shape a run's own worktree bay is named with: the
668/// same 4-character alphanumeric tag [`crate::run::is_run_id`] requires of a
669/// full id's trailing block (see [`short_of`])?
670///
671/// Anything else under `worktrees_root` is not a bay this function reclaims
672/// at all, claimed or not - answering "does a run claim this?" about a
673/// directory that was never a run's worktree in the first place is exactly
674/// the wrong question to ask before deleting it.
675fn looks_like_a_worktree_bay(name: &str) -> bool {
676 name.len() == 4 && name.bytes().all(|b| b.is_ascii_alphanumeric())
677}
678
679/// Is `dir`'s own modification time old enough, against `grace_secs` and
680/// `now`, that its emptiness of a run record can be trusted rather than
681/// caught mid-creation?
682///
683/// A directory this pass cannot stat at all - a race with its own removal, a
684/// permission error - is treated as not yet stale: unreadable metadata is not
685/// evidence of anything, and the janitor already keeps rather than deletes
686/// whenever it cannot tell (see the module docs).
687///
688/// `grace_secs` is floored at [`MIN_ORPHAN_AGE_SECS`] regardless of what the
689/// caller passes: `0` is a documented, legitimate value for
690/// [`crate::config::Disk::fold_grace_secs`] (`due`'s own "always due" case),
691/// because that grace answers a policy question the operator owns - how long
692/// a *known, finished* run's worktree lingers before cleanup. Whether an
693/// orphan worktree is actually a race with `Runner::review`'s `git worktree
694/// add` landing before its `run.json` is not a policy question, and must not
695/// collapse to zero just because the operator turned the other grace off -
696/// that would defeat the very check meant to catch it.
697fn stale_enough(dir: &Path, grace_secs: u64, now: Timestamp) -> bool {
698 let Ok(modified) = std::fs::metadata(dir).and_then(|m| m.modified()) else {
699 return false;
700 };
701 let Ok(ts) = Timestamp::try_from(modified) else {
702 return false;
703 };
704 due(now, ts, grace_secs.max(MIN_ORPHAN_AGE_SECS))
705}
706
707/// The floor under [`stale_enough`]'s grace, independent of
708/// [`crate::config::Disk::fold_grace_secs`].
709///
710/// Ample next to the race it guards: the gap between `Runner::start` or
711/// `Runner::review` creating a worktree and the first `RunState::save`
712/// landing is a handful of `git` subprocess calls, not minutes - but the
713/// janitor cannot tell "still mid-setup" from "orphaned" by any other signal
714/// for a run that never registers with `daemon::Status` at all (a `magi
715/// review` invocation, for one), so this is generous on purpose rather than
716/// tuned to the observed case.
717const MIN_ORPHAN_AGE_SECS: u64 = 5 * 60;
718
719/// Resolve an id or prefix against an explicit runs directory, exactly the way
720/// [`crate::run::resolve_id`] does against the global home.
721fn resolve_id_path(runs: &Path, prefix: &str) -> Result<String> {
722 // Keyed on the directory, not on a readable state file: the record this
723 // route exists to remove may be a lone `run.json.tmp` from a save that
724 // ran out of disk, and that is precisely the one a human needs a way to
725 // clear (see `crate::run::list_ids`).
726 if runs.join(prefix).is_dir() && crate::run::is_run_id(prefix) {
727 return Ok(prefix.to_owned());
728 }
729 let mut hits: Vec<String> = Vec::new();
730 for e in std::fs::read_dir(runs).into_iter().flatten().flatten() {
731 if !e.path().is_dir() {
732 continue;
733 }
734 let id = e.file_name().to_string_lossy().into_owned();
735 if crate::run::is_run_id(&id) && (id.starts_with(prefix) || id.ends_with(prefix)) {
736 hits.push(id);
737 }
738 }
739 match hits.len() {
740 1 => Ok(hits.into_iter().next().expect("exactly one hit")),
741 0 => bail!("no run matches `{prefix}`"),
742 _ => bail!(
743 "`{prefix}` matches {} runs: {}",
744 hits.len(),
745 hits.join(", ")
746 ),
747 }
748}
749
750/// `magi fold`'s recovery path for a run whose worktrees are already gone —
751/// so [`crate::graph::fold_run`] removed nothing — but whose `run.json` still
752/// lists active seats nobody is left to answer for: no live daemon claims the
753/// run, and every one of those seats has overrun its own timeout (see
754/// [`RunState::active_all_overrun`]). Clearing them and failing the run is
755/// what lets it be deleted afterward — [`RunState::ensure_can_delete`] only
756/// ever checks whether a live daemon is working on the run and whether its
757/// candidates are folded, not `status`, but a run stuck `implementing`
758/// forever with an empty worktree still reads as unresolved everywhere else
759/// (`magi show`, the deck, the phone) until this runs.
760///
761/// Returns `false` without changing anything when a live daemon still claims
762/// the run, or when some active seat has not actually overrun its budget yet
763/// — a run that is merely between waves must never be guessed at.
764pub fn clear_abandoned_active(state: &mut RunState, home: &Path, now: Timestamp) -> Result<bool> {
765 if crate::daemon::is_working_on(home, &state.id, now) || !state.active_all_overrun(now) {
766 return Ok(false);
767 }
768 state.abandon("fold");
769 state.save_under(home)?;
770 // The seat that asked is gone for good now - the same door
771 // `graph::Runner::settle_questions` closes the moment `status` lands
772 // somewhere non-resumable, see that method's own doc. Without this, an
773 // open question the abandoned seat left behind would keep badging the
774 // operator until the next daemon startup's `abandon_settled_questions`
775 // pass happened to notice it, or forever if nothing is running `magi
776 // serve` at all.
777 if let Err(e) = Questions::at(home.join("questions")).settle_run(&state.id, state.status) {
778 tracing::warn!("abandon questions for {}: {e:#}", state.id);
779 }
780 Ok(true)
781}
782
783/// Delete files from the shared build cache until it fits its cap — but only
784/// while nobody live is registered as using it. [`crate::cache::maintenance_prune`]
785/// takes out the same lease a build would, so a prune can never race a
786/// compile in flight (this run's own, another run's, or a human's `magi
787/// review`) into deleting a file that build still needs. `Ok(None)` when the
788/// cache is in use right now; the next pass catches it once the borrower
789/// releases it, the same way a cap of `0` or a missing `CARGO_TARGET_DIR`
790/// already meant "nothing to do this time" here.
791pub fn prune_cache(home: &Path, cache: &Path, limit_bytes: u64) -> Result<Option<Prune>> {
792 crate::cache::maintenance_prune(home, cache, limit_bytes)
793}
794
795/// [`prune_cache`], but resolving the operator's opt-out and missing
796/// `CARGO_TARGET_DIR` first — the same two checks [`housekeep`]'s idle pass
797/// makes before ever measuring the cache, factored out so
798/// [`crate::daemon`]'s between-runs check (see the module's own doc for why
799/// congestion can make "idle" arrive too rarely to matter) makes them
800/// identically rather than growing its own copy that could drift. `Ok(None)`
801/// covers a cap of `0` (see the module docs on `cache_limit_bytes`), a config
802/// that renders no `CARGO_TARGET_DIR` to aggregate at all, and a cache
803/// currently in use (see [`prune_cache`]).
804pub fn prune_cache_if_over_limit(
805 cfg: &crate::config::Config,
806 home: &Path,
807) -> Result<Option<Prune>> {
808 if cfg.disk.cache_limit_bytes == 0 {
809 return Ok(None);
810 }
811 let Some(cache) = cfg.cache_dir() else {
812 return Ok(None);
813 };
814 prune_cache(home, &cache, cfg.disk.cache_limit_bytes)
815}
816
817/// The cache's path, size and cap, for `magi cache show` and the health view.
818/// `None` when the config declares no `CARGO_TARGET_DIR` to aggregate.
819///
820/// A cap of `0` means the operator opted out of pruning; the size is then
821/// reported but never acted on.
822pub fn cache_report(cfg: &crate::config::Config) -> Option<(PathBuf, u64, u64)> {
823 let cache = cfg.cache_dir()?;
824 Some((
825 cache.clone(),
826 cache_size(&cache),
827 cfg.disk.cache_limit_bytes,
828 ))
829}
830
831/// Size in bytes of the shared build cache.
832pub fn cache_size(cache: &Path) -> u64 {
833 dir_size(cache)
834}
835
836#[cfg(test)]
837mod tests {
838 use super::*;
839 use crate::config::Disk;
840 use std::fs;
841
842 fn ts(s: &str) -> Timestamp {
843 s.parse().expect("rfc3339")
844 }
845
846 fn block_on<F: std::future::Future>(f: F) -> F::Output {
847 tokio::runtime::Runtime::new().expect("runtime").block_on(f)
848 }
849
850 #[test]
851 fn a_run_is_due_after_its_grace_and_not_before() {
852 let now = ts("2026-09-05T00:00:00Z");
853 let grace = 600;
854 let old = now - SignedDuration::new(601, 0);
855 let fresh = now - SignedDuration::new(599, 0);
856 assert!(due(now, old, grace));
857 assert!(!due(now, fresh, grace));
858 // Exactly at the edge: not yet due.
859 let edge = now - SignedDuration::new(600, 0);
860 assert!(!due(now, edge, grace));
861 // A zero grace folds everything, ever.
862 assert!(due(now, old, 0));
863 }
864
865 #[test]
866 fn rotate_from_moves_the_starting_point_as_the_cursor_advances() {
867 let ids: Vec<String> = ["a", "b", "c", "d", "e"]
868 .iter()
869 .map(|s| s.to_string())
870 .collect();
871
872 // A cursor of 0 starts at the front, same as an unrotated list.
873 assert_eq!(rotate_from(&ids, 0), ids);
874
875 // A cursor of 2 starts two ids further along, with the skipped
876 // front wrapping to the tail rather than being dropped.
877 let at_2 = rotate_from(&ids, 2);
878 assert_eq!(at_2, vec!["c", "d", "e", "a", "b"]);
879 for id in &ids {
880 assert!(at_2.contains(id));
881 }
882
883 // A cursor past the list's own length wraps rather than panicking -
884 // the caller never has to reduce it modulo anything itself.
885 assert_eq!(rotate_from(&ids, 7), rotate_from(&ids, 2));
886
887 // An empty list has no start point to compute and must not panic.
888 assert_eq!(rotate_from(&[], 0), Vec::<String>::new());
889 }
890
891 /// A fleet with more eligible runs than one pass can check must not park
892 /// the same handful at the front of the window forever: advancing the
893 /// cursor by exactly how many ids one pass actually looked at - the
894 /// arithmetic [`reconcile_external_merges`] itself does - walks every id
895 /// to the front within one full rotation, with no reliance on how often
896 /// or how regularly passes happen to run.
897 #[test]
898 fn advancing_the_cursor_by_the_window_size_gives_every_id_a_turn() {
899 let ids: Vec<String> = (0..12).map(|n| format!("run-{n}")).collect();
900 let cap = 5usize;
901 let mut cursor = 0usize;
902 let mut ever_seen: std::collections::HashSet<String> = std::collections::HashSet::new();
903 for _ in 0..ids.len() {
904 let window: Vec<String> = rotate_from(&ids, cursor).into_iter().take(cap).collect();
905 ever_seen.extend(window.iter().cloned());
906 cursor = (cursor + window.len()) % ids.len();
907 }
908 assert_eq!(
909 ever_seen.len(),
910 ids.len(),
911 "every id must be checked at least once across a full rotation, whatever \
912 the cadence between passes"
913 );
914 }
915
916 #[test]
917 fn the_external_merge_cursor_round_trips_through_a_files_absence() {
918 let dir = tempfile::tempdir().unwrap();
919 let home = dir.path();
920
921 // Nothing written yet: starts at 0, not an error.
922 assert_eq!(read_external_merge_cursor(home), 0);
923
924 write_external_merge_cursor(home, 7);
925 assert_eq!(read_external_merge_cursor(home), 7);
926
927 // Anything unreadable as a plain number falls back to 0 rather than
928 // wedging the sweep on a corrupt file.
929 std::fs::write(home.join(EXTERNAL_MERGE_CURSOR_FILE), "not a number").unwrap();
930 assert_eq!(read_external_merge_cursor(home), 0);
931 }
932
933 #[test]
934 fn a_run_qualifies_for_an_external_merge_check_only_when_blocked_unmerged_and_due() {
935 let now = ts("2026-09-05T00:00:00Z");
936 let grace = 600;
937 let old = now - SignedDuration::new(601, 0);
938 let fresh = now - SignedDuration::new(599, 0);
939
940 assert!(
941 eligible_for_external_merge_check(RunStatus::Blocked, false, old, now, grace),
942 "blocked, unmerged, and past its grace period is exactly the run this exists for"
943 );
944 assert!(
945 !eligible_for_external_merge_check(RunStatus::Blocked, false, fresh, now, grace),
946 "an operator mid-fix deserves the same grace window `fold_due` gives before \
947 the janitor starts asking GitHub about it"
948 );
949 assert!(
950 !eligible_for_external_merge_check(RunStatus::Blocked, true, old, now, grace),
951 "a run `land::land` already recorded a merge outcome for has its own answer \
952 already; this check is only for a run with nothing recorded at all"
953 );
954 assert!(
955 !eligible_for_external_merge_check(RunStatus::Stalled, false, old, now, grace),
956 "stalled is not blocked - it means the tally never reached quorum, which a \
957 pull request cannot fix"
958 );
959 assert!(
960 !eligible_for_external_merge_check(RunStatus::Ready, false, old, now, grace),
961 "ready has nothing to correct - it was never landed by design"
962 );
963 assert!(
964 !eligible_for_external_merge_check(RunStatus::Superseded, false, old, now, grace),
965 "a superseded run's task already has its answer from a later attempt; \
966 there is nothing left for GitHub to confirm here"
967 );
968 }
969
970 /// A `Blocked` run with a decided winner, so `land::find_external_merge`
971 /// has a branch to ask `gh` about instead of returning early for lack of
972 /// one - which is what makes an unreachable `/nonexistent/repo` fail
973 /// loudly (an `Err`, raising a notice) rather than silently (an early
974 /// `Ok(None)`, raising nothing) when this run's sweep is exercised.
975 fn write_blocked_run(runs: &Path, id: &str, updated_at: Timestamp) {
976 use crate::run::{Candidate, Tally};
977 use std::collections::BTreeMap;
978
979 let mut state = RunState::new(
980 PathBuf::from("/nonexistent/repo"),
981 "main".to_owned(),
982 "0000000000000000000000000000000000000000".to_owned(),
983 String::new(),
984 crate::config::Config::default(),
985 );
986 state.id = id.to_owned();
987 state.status = RunStatus::Blocked;
988 state.updated_at = updated_at;
989 state.candidates.push(Candidate {
990 index: 0,
991 label: 'A',
992 agent: "agent".to_owned(),
993 branch: format!("magi/{id}/A"),
994 worktree: PathBuf::from("/nonexistent/repo"),
995 summary: String::new(),
996 stat: String::new(),
997 files: 0,
998 commits: 0,
999 empty: false,
1000 failed: None,
1001 verified_noop: None,
1002 duration_ms: 0,
1003 folded: false,
1004 });
1005 state.tally = Some(Tally {
1006 first_choice: BTreeMap::from([('A', 1)]),
1007 borda: BTreeMap::new(),
1008 winner: 'A',
1009 rankings: 1,
1010 unanimous_initial: true,
1011 deliberated: false,
1012 changed_votes: 0,
1013 unanimous_final: true,
1014 tie_break: None,
1015 judges: 1,
1016 present: 1,
1017 quorum: 1,
1018 met_quorum: true,
1019 uncontested: None,
1020 });
1021 std::fs::create_dir_all(runs.join(id)).unwrap();
1022 std::fs::write(
1023 runs.join(id).join("run.json"),
1024 serde_json::to_string_pretty(&state).unwrap(),
1025 )
1026 .unwrap();
1027 }
1028
1029 /// No run here can be confirmed merged - one has no repository `gh` can
1030 /// even ask about, one has not sat blocked long enough, and one is
1031 /// claimed by a live daemon - so the sweep must leave every one of them
1032 /// exactly as it found them and never panic on the failures in between.
1033 #[tokio::test]
1034 async fn reconcile_external_merges_leaves_ineligible_and_unconfirmable_runs_alone() {
1035 let dir = tempfile::tempdir().unwrap();
1036 let runs = dir.path().join("runs");
1037 let home = dir.path().to_path_buf();
1038 std::fs::create_dir_all(&runs).unwrap();
1039
1040 let now = ts("2026-09-05T00:00:00Z");
1041 let disk = Disk {
1042 fold_grace_secs: 600,
1043 ..Disk::default()
1044 };
1045
1046 let due_id = "20260905-000000-blkd";
1047 write_blocked_run(&runs, due_id, ts("2026-08-01T00:00:00Z"));
1048
1049 let fresh_id = "20260905-000000-fres";
1050 write_blocked_run(&runs, fresh_id, now);
1051
1052 let live_id = "20260905-000000-live";
1053 write_blocked_run(&runs, live_id, ts("2026-08-01T00:00:00Z"));
1054 let mut status = crate::daemon::Status::new();
1055 status.current = vec![crate::daemon::Current {
1056 task: "20260905-000000-task".to_owned(),
1057 run: live_id.to_owned(),
1058 }];
1059 status.updated_at = now;
1060 crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1061
1062 let reconciled = reconcile_external_merges(&runs, &home, &disk, now).await;
1063 assert_eq!(
1064 reconciled, 0,
1065 "an unreachable repository can never be confirmed merged"
1066 );
1067 for id in [due_id, fresh_id, live_id] {
1068 assert_eq!(
1069 read_meta(&runs, id).unwrap().status,
1070 RunStatus::Blocked,
1071 "{id} must be left exactly as it was found"
1072 );
1073 }
1074 }
1075
1076 /// The bug this exists to pin: a first version rotated on
1077 /// `now.as_second() % len`, which is fully determined by wall-clock time
1078 /// and nothing else. Two passes seconds apart - as they would be when a
1079 /// short poll interval and a fast pass alias onto the same handful of
1080 /// `now.as_second()` residues - landed on the *same* starting point and
1081 /// never covered a fleet bigger than the per-pass cap, however many
1082 /// times housekeeping ran. Persisting a cursor makes forward progress a
1083 /// function of how many ids got looked at, not of when the clock reads
1084 /// this pass happened to run.
1085 #[tokio::test]
1086 async fn reconcile_external_merges_covers_a_larger_fleet_across_repeated_passes_at_one_instant()
1087 {
1088 let dir = tempfile::tempdir().unwrap();
1089 let runs = dir.path().join("runs");
1090 let home = dir.path().to_path_buf();
1091 std::fs::create_dir_all(&runs).unwrap();
1092
1093 let now = ts("2026-09-05T00:00:00Z");
1094 let disk = Disk {
1095 fold_grace_secs: 600,
1096 ..Disk::default()
1097 };
1098 let old = ts("2026-08-01T00:00:00Z");
1099
1100 let ids: Vec<String> = (0..8).map(|n| format!("20260905-000000-r{n:03}")).collect();
1101 for id in &ids {
1102 write_blocked_run(&runs, id, old);
1103 }
1104
1105 let notices = crate::notices::Notices::at(home.join("notifications"));
1106
1107 // Two passes at the exact same `now`, exactly what a fast pass on a
1108 // short poll interval looks like. Each of the 8 unreachable-repo
1109 // runs raises its own notice the first time it is looked at, so the
1110 // count of distinct notices is a direct readout of how many distinct
1111 // ids have been checked so far.
1112 reconcile_external_merges(&runs, &home, &disk, now).await;
1113 let after_first = notices.list().len();
1114 assert_eq!(
1115 after_first, MAX_EXTERNAL_MERGE_CHECKS_PER_PASS,
1116 "the first pass checks exactly one cap's worth"
1117 );
1118
1119 reconcile_external_merges(&runs, &home, &disk, now).await;
1120 let after_second = notices.list().len();
1121 assert_eq!(
1122 after_second,
1123 ids.len(),
1124 "a second pass at the same instant must still reach every id the \
1125 first pass had no room for, not repeat the same cap's worth"
1126 );
1127 }
1128
1129 #[test]
1130 fn the_meta_reader_is_tolerant_of_everything_except_the_deciders() {
1131 let dir = tempfile::tempdir().unwrap();
1132 let runs = dir.path().join("runs");
1133 let id = "20260905-000000-abcd";
1134 std::fs::create_dir_all(runs.join(id)).unwrap();
1135 std::fs::write(
1136 runs.join(id).join("run.json"),
1137 r#"{"schema": 99, "id": "20260905-000000-abcd", "updated_at": "2026-09-05T00:00:00Z", "status": "ready", "junk_from_another_build": [1, 2, 3]}"#,
1138 )
1139 .unwrap();
1140 let meta = read_meta(&runs, id).expect("readable");
1141 assert_eq!(meta.status, RunStatus::Ready);
1142 assert_eq!(meta.updated_at, ts("2026-09-05T00:00:00Z"));
1143 assert!(read_meta(&runs, "nope").is_err(), "missing file unreadable");
1144 std::fs::write(runs.join(id).join("run.json"), "not json at all").unwrap();
1145 assert!(read_meta(&runs, id).is_err(), "garbage unreadable");
1146 }
1147
1148 #[test]
1149 fn fold_unreadable_releases_run_dir_and_worktrees() {
1150 let dir = tempfile::tempdir().unwrap();
1151 let runs = dir.path().join("runs");
1152 let wt = dir.path().join("wt");
1153 let id = "20260905-000000-abcd";
1154 std::fs::create_dir_all(runs.join(id)).unwrap();
1155 std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1156 std::fs::create_dir_all(wt.join("abcd")).unwrap();
1157 std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1158
1159 let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold");
1160 assert_eq!(removed.len(), 2);
1161 assert!(!runs.join(id).exists(), "run dir gone");
1162 assert!(!wt.join("abcd").exists(), "worktrees gone");
1163
1164 // A prefix resolves like `run::resolve_id` does.
1165 std::fs::create_dir_all(runs.join(id)).unwrap();
1166 std::fs::write(runs.join(id).join("run.json"), "garbage").unwrap();
1167 std::fs::create_dir_all(wt.join("abcd")).unwrap();
1168 std::fs::write(wt.join("abcd").join("leftover"), b"x").unwrap();
1169 let removed = block_on(fold_unreadable(&runs, &wt, "20260905")).expect("by prefix");
1170 assert_eq!(removed.len(), 2);
1171 // Once gone, `id` cannot be resolved at all - same as `run::resolve_id`
1172 // on an id nothing on disk matches - so a repeat pass errors rather
1173 // than silently reporting nothing removed.
1174 assert!(
1175 block_on(fold_unreadable(&runs, &wt, id)).is_err(),
1176 "a run already gone cannot be resolved again"
1177 );
1178 }
1179
1180 #[test]
1181 fn prune_cache_sheds_the_oldest_generation_until_it_fits() {
1182 let home = tempfile::tempdir().unwrap();
1183 let dir = tempfile::tempdir().unwrap();
1184 // Same size, different age: only the age decides, and the newest
1185 // generation - the one the next build reuses - is what survives.
1186 fs::write(dir.path().join("old"), b"xx").unwrap();
1187 fs::write(dir.path().join("new"), b"yy").unwrap();
1188 touch(&dir.path().join("old"), 1_000_000);
1189 touch(&dir.path().join("new"), 2_000_000);
1190
1191 let out = prune_cache(home.path(), dir.path(), 2)
1192 .expect("prune")
1193 .expect("the cache is free");
1194 assert_eq!(out.files, 1, "one deletion is enough to reach the cap");
1195 assert_eq!(out.remaining, 2);
1196 assert!(!dir.path().join("old").exists(), "the older file went");
1197 assert!(dir.path().join("new").exists(), "the newer one stayed");
1198
1199 // A whole generation shares one timestamp tick, so the tie has to be
1200 // decided too: largest first, which reaches the cap in the fewest
1201 // deletions. Left to `read_dir` and an unstable sort this deleted
1202 // both files on Linux and one on Windows.
1203 let tied = tempfile::tempdir().unwrap();
1204 fs::write(tied.path().join("big"), b"xxxx").unwrap();
1205 fs::write(tied.path().join("small"), b"yy").unwrap();
1206 touch(&tied.path().join("big"), 1_000_000);
1207 touch(&tied.path().join("small"), 1_000_000);
1208 let out = prune_cache(home.path(), tied.path(), 2)
1209 .expect("prune")
1210 .expect("the cache is free");
1211 assert_eq!(out.files, 1, "the big one alone gets under the cap");
1212 assert_eq!(out.remaining, 2);
1213 assert!(tied.path().join("small").exists());
1214 }
1215
1216 #[test]
1217 fn prune_cache_if_over_limit_resolves_the_opt_outs_before_ever_measuring() {
1218 let home = tempfile::tempdir().unwrap();
1219 let dir = tempfile::tempdir().unwrap();
1220 fs::write(dir.path().join("big"), vec![0u8; 10]).unwrap();
1221
1222 let mut cfg = crate::config::Config::default();
1223 cfg.verify.gate = vec![format!(
1224 "CARGO_TARGET_DIR={} cargo make check",
1225 dir.path().display()
1226 )];
1227
1228 // A cap of `0` is the operator's opt-out: never measured, never
1229 // pruned, regardless of what is actually on disk.
1230 cfg.disk.cache_limit_bytes = 0;
1231 assert_eq!(
1232 prune_cache_if_over_limit(&cfg, home.path()).unwrap(),
1233 None,
1234 "a zero cap must not even look at the directory"
1235 );
1236 assert!(dir.path().join("big").exists());
1237
1238 // No `CARGO_TARGET_DIR` in either verify command: nothing to
1239 // aggregate, so there is nothing to prune either.
1240 let mut no_cache = crate::config::Config::default();
1241 no_cache.disk.cache_limit_bytes = 1;
1242 assert_eq!(
1243 prune_cache_if_over_limit(&no_cache, home.path()).unwrap(),
1244 None
1245 );
1246
1247 // Over the cap and configured: pruned exactly like `prune_cache`
1248 // itself would.
1249 cfg.disk.cache_limit_bytes = 1;
1250 let pruned = prune_cache_if_over_limit(&cfg, home.path())
1251 .unwrap()
1252 .expect("a real cache dir over its cap prunes");
1253 assert_eq!(pruned.files, 1);
1254 assert!(!dir.path().join("big").exists());
1255 }
1256
1257 /// Pin a file's mtime, so a test asserts the policy and not the runner's
1258 /// timestamp granularity.
1259 fn touch(path: &Path, secs: u64) {
1260 let f = fs::File::options().write(true).open(path).unwrap();
1261 f.set_times(fs::FileTimes::new().set_modified(
1262 std::time::SystemTime::UNIX_EPOCH + std::time::Duration::from_secs(secs),
1263 ))
1264 .unwrap();
1265 }
1266
1267 /// The disk-full casualty: a run whose first save left `run.json.tmp` and
1268 /// nothing else. It has to be clearable, or the record is permanent.
1269 #[test]
1270 fn fold_unreadable_clears_a_run_whose_state_never_landed() {
1271 let dir = tempfile::tempdir().unwrap();
1272 let runs = dir.path().join("runs");
1273 let wt = dir.path().join("wt");
1274 let id = "20260904-014540-88c0";
1275 std::fs::create_dir_all(runs.join(id)).unwrap();
1276 std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1277
1278 let removed = block_on(fold_unreadable(&runs, &wt, id)).expect("fold by id");
1279 assert_eq!(removed, vec![format!("runs/{id}")]);
1280 assert!(!runs.join(id).exists(), "record gone");
1281
1282 // And by prefix, the way the deck and the phone address a run.
1283 std::fs::create_dir_all(runs.join(id)).unwrap();
1284 std::fs::write(runs.join(id).join("run.json.tmp"), b"").unwrap();
1285 assert!(
1286 block_on(fold_unreadable(&runs, &wt, "88c0")).is_ok(),
1287 "by prefix"
1288 );
1289
1290 // A directory under `runs` that is not a run is never a fold target.
1291 std::fs::create_dir_all(runs.join("scratch")).unwrap();
1292 assert!(
1293 block_on(fold_unreadable(&runs, &wt, "scratch")).is_err(),
1294 "a stray directory is not a run"
1295 );
1296 }
1297
1298 #[test]
1299 fn fold_due_folds_terminal_runs_of_any_schema_but_leaves_genuinely_unreadable_ones() {
1300 let dir = tempfile::tempdir().unwrap();
1301 let runs = dir.path().join("runs");
1302 let wt = dir.path().join("wt");
1303 let home = dir.path().to_path_buf();
1304 let disk = Disk::default();
1305 let now = ts("2026-09-05T00:00:00Z");
1306
1307 // 1. Runnable (judging): never folded, however old.
1308 let judging = "20260801-000000-0001";
1309 write_meta(&runs, judging, "judging", "2026-08-01T00:00:00Z");
1310
1311 // 2. Finished but fresh: grace not elapsed. Within the default 6h
1312 // grace of `now`, so `fold_due` must stop at the freshness check
1313 // and never even reach `read_state` - `write_meta`'s minimal JSON
1314 // would fail that full parse anyway, and this case exists to
1315 // prove freshness is why the run survives, not an accident of the
1316 // fixture being unparseable as a whole `RunState`.
1317 let ready_fresh = "20260904-220000-0002";
1318 write_meta(&runs, ready_fresh, "ready", "2026-09-04T22:00:00Z");
1319
1320 // 3. Genuinely unreadable: broken JSON, not merely an unfamiliar
1321 // schema number. Left alone and counted - this is the one case
1322 // automatic housekeeping must never touch (see `fold_due`'s docs);
1323 // discarding it is an explicit operator action, not something a
1324 // background pass does.
1325 let garbage = "20260901-000000-0004";
1326 std::fs::create_dir_all(runs.join(garbage)).unwrap();
1327 std::fs::write(runs.join(garbage).join("run.json"), "not json").unwrap();
1328 std::fs::create_dir_all(wt.join("0004")).unwrap();
1329
1330 // 4. Finished, well past grace, current schema: the ordinary case
1331 // `fold_due` has always acted on.
1332 let due_ready = due_run(&runs, &wt, "20260801-000000-ffff", SCHEMA);
1333
1334 // 5. Finished, well past grace, but written by a schema number this
1335 // build no longer matches - the defect this task exists to fix.
1336 // It still parses cleanly, so only the version number differs, and
1337 // that alone must not block folding.
1338 let due_old_schema = due_run(&runs, &wt, "20260801-000000-eeee", SCHEMA - 1);
1339
1340 let (folded, unreadable) =
1341 block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
1342 assert_eq!(
1343 folded, 2,
1344 "both due, parseable runs fold regardless of their schema number"
1345 );
1346 assert_eq!(
1347 unreadable, 1,
1348 "only the run with broken JSON counts as unreadable"
1349 );
1350 assert!(runs.join(judging).exists(), "runnable never folded");
1351 assert!(runs.join(ready_fresh).exists(), "fresh never folded");
1352 assert!(runs.join(garbage).exists(), "unreadable record kept");
1353 assert!(wt.join("0004").exists(), "unreadable worktree kept");
1354 assert!(
1355 runs.join(&due_ready).exists(),
1356 "folding drops worktrees, not the record"
1357 );
1358 assert!(
1359 runs.join(&due_old_schema).exists(),
1360 "an old-schema record survives its fold exactly like a current one"
1361 );
1362 // `graph::fold_run` saves the state it just folded back to disk. That
1363 // write must land under this test's own `runs` - the argument it
1364 // passed to `fold_due`, not the process-global `run::home` - or a
1365 // fold that landed somewhere else entirely would still be counted
1366 // above as one of the two `folded` runs.
1367 for id in [&due_ready, &due_old_schema] {
1368 let saved = read_meta(&runs, id).expect("folded run still parses");
1369 assert_ne!(
1370 saved.updated_at,
1371 ts("2026-08-01T00:00:00Z"),
1372 "fold_run must have saved the updated state back through the \
1373 `runs` directory this test passed to fold_due"
1374 );
1375 }
1376 }
1377
1378 /// `[disk] auto_fold = false` must leave the janitor's fold-and-reclaim
1379 /// passes completely inert - a due run's worktree and record both
1380 /// survive exactly as if `housekeep` had never run at all. Cache pruning
1381 /// is a separate opt-out (`cache_limit_bytes`) and stays disabled here
1382 /// too, so this test is only ever about `auto_fold`.
1383 #[tokio::test]
1384 async fn housekeep_leaves_everything_alone_when_auto_fold_is_disabled() {
1385 let dir = tempfile::tempdir().unwrap();
1386 let runs = dir.path().join("runs");
1387 let wt = dir.path().join("wt");
1388 let home = dir.path().to_path_buf();
1389 crate::run::set_home(dir.path().to_path_buf());
1390
1391 let due_id = due_run(&runs, &wt, "20260801-000000-abcd", SCHEMA);
1392 std::fs::create_dir_all(wt.join("orphan").join("cand-A")).unwrap();
1393
1394 let mut cfg = crate::config::Config::default();
1395 cfg.disk.auto_fold = false;
1396 cfg.disk.cache_limit_bytes = 0;
1397
1398 let out = housekeep(&cfg, &home, &wt, &dir.path().join("repo"), Timestamp::now()).await;
1399
1400 assert_eq!(out.folded, 0);
1401 assert_eq!(out.unreadable, 0);
1402 assert_eq!(out.orphaned_worktrees, 0);
1403 assert!(
1404 runs.join(&due_id).exists(),
1405 "a due run's record survives untouched"
1406 );
1407 assert!(
1408 wt.join("orphan").exists(),
1409 "an orphaned worktree survives untouched: the reclaim pass never ran"
1410 );
1411 }
1412
1413 #[test]
1414 fn fold_orphaned_worktrees_removes_only_worktrees_no_run_claims_and_none_in_flight() {
1415 let dir = tempfile::tempdir().unwrap();
1416 let runs = dir.path().join("runs");
1417 let wt = dir.path().join("wt");
1418 let home = dir.path().to_path_buf();
1419
1420 // A run record exists for this one: its worktree is claimed, not
1421 // orphaned, however old the record.
1422 write_meta(
1423 &runs,
1424 "20260801-000000-aaaa",
1425 "ready",
1426 "2026-08-01T00:00:00Z",
1427 );
1428 std::fs::create_dir_all(wt.join("aaaa").join("cand-A")).unwrap();
1429
1430 // No run record at all, and nobody is working on it: this is the
1431 // leftover `fold_due` can never see, because it only ever walks
1432 // `runs/`.
1433 std::fs::create_dir_all(wt.join("bbbb").join("cand-A")).unwrap();
1434
1435 // No run record either, but a live daemon status names a run with
1436 // this short id - the save-timing gap between the daemon claiming a
1437 // task and `RunState::new` writing its first `run.json`. Must survive
1438 // untouched.
1439 std::fs::create_dir_all(wt.join("cccc")).unwrap();
1440
1441 // Not shaped like a run's short id at all - a scratch directory an
1442 // operator or another tool left in the same bay - so it is never a
1443 // reclaim target regardless of what runs claim it or not.
1444 std::fs::create_dir_all(wt.join("scratch")).unwrap();
1445
1446 // `now` pushed comfortably past `MIN_ORPHAN_AGE_SECS`, so a zero
1447 // grace - the same "always due" escape hatch `due` itself documents
1448 // - still reclaims once a worktree is genuinely old, without faking
1449 // an mtime: real directory creation just above is already in the
1450 // past relative to this `now`, by design rather than by timing.
1451 let now = Timestamp::now() + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1452 let mut status = crate::daemon::Status::new();
1453 status.current = vec![crate::daemon::Current {
1454 task: "20260905-000000-t111".to_owned(),
1455 run: "20260905-000000-cccc".to_owned(),
1456 }];
1457 status.updated_at = now;
1458 crate::daemon::write_status_to(&home.join("daemon.json"), &status).unwrap();
1459
1460 let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1461 assert_eq!(
1462 folded, 1,
1463 "only the truly orphaned, idle, bay-shaped worktree is removed"
1464 );
1465 assert!(wt.join("aaaa").exists(), "claimed by a run record");
1466 assert!(!wt.join("bbbb").exists(), "orphaned and idle: reclaimed");
1467 assert!(wt.join("cccc").exists(), "a run in flight is never touched");
1468 assert!(
1469 wt.join("scratch").exists(),
1470 "not shaped like a worktree bay, so never a reclaim target"
1471 );
1472 }
1473
1474 /// The gap this closes: `Runner::review` (`magi review`) creates the
1475 /// worktree with `git worktree add` before `RunState::save` ever writes a
1476 /// `run.json`, and that path never runs through the daemon's own `poll`
1477 /// loop at all, so `daemon::Status` never names it either. Without a
1478 /// grace window, a janitor pass landing in that gap would read the
1479 /// worktree as an orphan nothing is waiting on and delete a review still
1480 /// being set up.
1481 #[test]
1482 fn fold_orphaned_worktrees_leaves_a_freshly_created_bay_alone() {
1483 let dir = tempfile::tempdir().unwrap();
1484 let runs = dir.path().join("runs");
1485 let wt = dir.path().join("wt");
1486 let home = dir.path().to_path_buf();
1487
1488 std::fs::create_dir_all(wt.join("dddd").join("under-review")).unwrap();
1489
1490 let now = Timestamp::now();
1491 let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 6 * 60 * 60, now));
1492 assert_eq!(
1493 folded, 0,
1494 "too fresh to tell apart from a run still being set up"
1495 );
1496 assert!(wt.join("dddd").exists());
1497 }
1498
1499 /// A fresh open question on `run`, stored and handed back for assertions.
1500 fn open_question(store: &Questions, run: &str) -> crate::ask::Question {
1501 let mut q = crate::ask::Question::new(
1502 run.to_owned(),
1503 "implement".to_owned(),
1504 "impl-A".to_owned(),
1505 "Which storage backend should the cache use?".to_owned(),
1506 String::new(),
1507 vec!["SQLite".to_owned(), "Redis".to_owned()],
1508 );
1509 store.put(&mut q).unwrap();
1510 q
1511 }
1512
1513 /// The exact ghost the phone showed: a run that already finished, with a
1514 /// question its dead seat asked still sitting `open` because it reached
1515 /// that status before `graph::Runner::settle_questions` existed (or
1516 /// missed it in the crash window `daemon::reclaim_orphaned_running`
1517 /// covers). This sweep is the second door to the same fact.
1518 #[test]
1519 fn a_finished_runs_open_question_is_swept_up() {
1520 let dir = tempfile::tempdir().unwrap();
1521 let runs = dir.path().join("runs");
1522 let store = Questions::at(dir.path().join("questions"));
1523
1524 let failed = "20260908-205802-c9eb";
1525 write_meta(&runs, failed, "failed", "2026-09-08T20:58:02Z");
1526 let failed_q = open_question(&store, failed);
1527
1528 let merged = "20260908-205501-ca67";
1529 write_meta(&runs, merged, "merged", "2026-09-08T20:55:01Z");
1530 let merged_q = open_question(&store, merged);
1531
1532 let n = abandon_settled_questions(&store, &runs);
1533 assert_eq!(n, 2, "both dead runs' questions are swept in one pass");
1534
1535 for (id, run) in [(&failed_q.id, failed), (&merged_q.id, merged)] {
1536 let back = store.get(id).unwrap();
1537 assert!(!back.status.open(), "{run} is done; nobody reads an answer");
1538 assert!(back.detail.contains(run), "{}", back.detail);
1539 }
1540 }
1541
1542 #[test]
1543 fn a_still_alive_runs_open_question_survives_the_sweep() {
1544 let dir = tempfile::tempdir().unwrap();
1545 let runs = dir.path().join("runs");
1546 let store = Questions::at(dir.path().join("questions"));
1547
1548 // `Blocked` and `Stalled` are `RunStatus::resumable`: the run can
1549 // still be picked back up, so its question may yet get a real
1550 // answer. A run still mid-competition is even more obviously alive.
1551 for (id, status) in [
1552 ("20260908-000000-b10c", "blocked"),
1553 ("20260908-000000-5ta1", "stalled"),
1554 ("20260908-000000-jud6", "judging"),
1555 ] {
1556 write_meta(&runs, id, status, "2026-09-08T00:00:00Z");
1557 let q = open_question(&store, id);
1558
1559 let n = abandon_settled_questions(&store, &runs);
1560 assert_eq!(n, 0, "{status} run is not done; nothing to sweep");
1561 assert!(
1562 store.get(&q.id).unwrap().status.open(),
1563 "{status} run's question must still be waiting"
1564 );
1565 }
1566 }
1567
1568 #[test]
1569 fn the_sweep_leaves_an_answered_question_and_an_unreadable_run_alone() {
1570 let dir = tempfile::tempdir().unwrap();
1571 let runs = dir.path().join("runs");
1572 let store = Questions::at(dir.path().join("questions"));
1573
1574 // Already decided: a sweep must never revisit it, whatever the run
1575 // that asked went on to become.
1576 let done = "20260908-000000-answ";
1577 write_meta(&runs, done, "failed", "2026-09-08T00:00:00Z");
1578 let mut answered = open_question(&store, done);
1579 answered
1580 .answer(crate::ask::Answer::Choice("SQLite".to_owned()))
1581 .unwrap();
1582 store.put(&mut answered).unwrap();
1583
1584 // No `run.json` at all for this one - deleted, or never landed.
1585 let gone = "20260908-000000-gone";
1586 let orphan = open_question(&store, gone);
1587
1588 assert_eq!(abandon_settled_questions(&store, &runs), 0);
1589 assert_eq!(
1590 store.get(&answered.id).unwrap().status,
1591 crate::ask::QuestionStatus::Answered,
1592 "a real answer is never overwritten by a sweep"
1593 );
1594 assert!(
1595 store.get(&orphan.id).unwrap().status.open(),
1596 "a run this sweep cannot read is left exactly as it was, not guessed at"
1597 );
1598 }
1599
1600 /// A grace of `0` is a legitimate, documented value for the operator's
1601 /// own `Disk::fold_grace_secs` - `due`'s "always due" case - but the
1602 /// freshness check this guards is not that policy, and must not collapse
1603 /// to it: a `0` handed straight through would reclaim a worktree the
1604 /// instant it exists, exactly the race `fold_orphaned_worktrees_leaves_a_
1605 /// freshly_created_bay_alone` exists to rule out, just with the operator
1606 /// having turned the other grace off instead of leaving it at its
1607 /// default.
1608 #[test]
1609 fn fold_orphaned_worktrees_floors_a_zero_grace_at_the_race_safe_minimum() {
1610 let dir = tempfile::tempdir().unwrap();
1611 let runs = dir.path().join("runs");
1612 let wt = dir.path().join("wt");
1613 let home = dir.path().to_path_buf();
1614
1615 std::fs::create_dir_all(wt.join("eeee").join("under-review")).unwrap();
1616
1617 // Too fresh, even with the grace argument at zero.
1618 let now = Timestamp::now();
1619 let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, now));
1620 assert_eq!(
1621 folded, 0,
1622 "a zero grace must not defeat the race-safety floor"
1623 );
1624 assert!(wt.join("eeee").exists());
1625
1626 // Once genuinely past the floor, a zero grace reclaims it - the
1627 // floor is a minimum, not a replacement policy that never fires.
1628 let later = now + SignedDuration::new((MIN_ORPHAN_AGE_SECS + 1) as i64, 0);
1629 let folded = block_on(fold_orphaned_worktrees(&runs, &wt, &home, 0, later));
1630 assert_eq!(folded, 1, "old enough now, regardless of the zero grace");
1631 assert!(!wt.join("eeee").exists());
1632 }
1633
1634 #[test]
1635 fn clear_abandoned_active_only_acts_once_dead_and_overrun() {
1636 let dir = tempfile::tempdir().unwrap();
1637 // Harmless if another test in this binary already pinned the global
1638 // home first (see `run::set_home`'s own doc): this test only checks
1639 // the in-memory mutation `clear_abandoned_active` makes, never a
1640 // write that landed under this exact directory.
1641 crate::run::set_home(dir.path().to_path_buf());
1642 let home = dir.path().to_path_buf();
1643 let now = ts("2026-09-14T12:00:00Z");
1644 let overrun_seat = || crate::run::ActiveSeat {
1645 node: "implement".to_owned(),
1646 started_at: now - SignedDuration::new(21_000, 0),
1647 timeout_secs: 3_600,
1648 attempt: 0,
1649 task: None,
1650 command: None,
1651 index: None,
1652 total: None,
1653 };
1654
1655 let mut state = RunState::new(
1656 PathBuf::from("/repo"),
1657 "main".to_owned(),
1658 "abc1234".to_owned(),
1659 "fixture".to_owned(),
1660 crate::config::Config::default(),
1661 );
1662 state.status = RunStatus::Implementing;
1663 state.active.insert("impl-A".to_owned(), overrun_seat());
1664
1665 // A seat still within its own budget: not provably dead yet, so this
1666 // must change nothing.
1667 let mut fresh = state.clone();
1668 fresh.active.insert(
1669 "impl-B".to_owned(),
1670 crate::run::ActiveSeat {
1671 node: "implement".to_owned(),
1672 started_at: now,
1673 timeout_secs: 3_600,
1674 attempt: 0,
1675 task: None,
1676 command: None,
1677 index: None,
1678 total: None,
1679 },
1680 );
1681 assert!(!clear_abandoned_active(&mut fresh, &home, now).unwrap());
1682 assert!(!fresh.active.is_empty());
1683 assert_eq!(fresh.status, RunStatus::Implementing);
1684
1685 let store = Questions::at(home.join("questions"));
1686 let q = open_question(&store, &state.id);
1687
1688 assert!(clear_abandoned_active(&mut state, &home, now).unwrap());
1689 assert!(state.active.is_empty());
1690 assert_eq!(state.status, RunStatus::Failed);
1691 assert!(
1692 !store.get(&q.id).unwrap().status.open(),
1693 "the abandoned seat's own open question must not keep badging the \
1694 operator until some later daemon startup notices it"
1695 );
1696 }
1697
1698 /// Write a whole `run.json` that magi can read, over the given state.
1699 fn write_meta(runs: &Path, id: &str, status: &str, updated_at: &str) {
1700 let day = &updated_at[..10];
1701 std::fs::create_dir_all(runs.join(id)).unwrap();
1702 let body = format!(
1703 r#"{{"schema": {SCHEMA}, "id": "{id}", "repo": "/nonexistent/repo", "base_branch": "main", "base_commit": "0000000000000000000000000000000000000000", "instruction": "", "created_at": "{day}T00:00:00Z", "updated_at": "{updated_at}", "status": "{status}", "seed": 1}}"#
1704 );
1705 std::fs::write(runs.join(id).join("run.json"), body).unwrap();
1706 }
1707
1708 /// Write a fully-formed, `Ready`, well-past-grace `run.json` tagged with
1709 /// an arbitrary schema number - so a test can write one this build's own
1710 /// `RunState::new` could never produce on its own. Returns the id.
1711 ///
1712 /// `wt` becomes this run's `graph.worktree_root`: left at the config
1713 /// default, `RunState::worktree_root` falls through to
1714 /// `run::default_worktree_root` - the operator's real `~/wt/magi` - and
1715 /// `graph::fold_run`'s second sweep would then `read_dir` and remove
1716 /// worktrees there instead of anything this test owns.
1717 fn due_run(runs: &Path, wt: &Path, id: &str, schema: u32) -> String {
1718 due_run_with_status(runs, wt, id, schema, RunStatus::Ready)
1719 }
1720
1721 /// [`due_run`], with the terminal status a caller wants instead of the
1722 /// `Ready` every other caller here happens to want.
1723 fn due_run_with_status(
1724 runs: &Path,
1725 wt: &Path,
1726 id: &str,
1727 schema: u32,
1728 status: RunStatus,
1729 ) -> String {
1730 let mut config = crate::config::Config::default();
1731 config.graph.worktree_root = Some(wt.to_path_buf());
1732 let mut state = RunState::new(
1733 PathBuf::from("/nonexistent/repo"),
1734 "main".to_owned(),
1735 "0000000000000000000000000000000000000000".to_owned(),
1736 String::new(),
1737 config,
1738 );
1739 state.id = id.to_owned();
1740 state.status = status;
1741 state.updated_at = ts("2026-08-01T00:00:00Z");
1742 let mut value = serde_json::to_value(&state).unwrap();
1743 value["schema"] = serde_json::json!(schema);
1744 std::fs::create_dir_all(runs.join(id)).unwrap();
1745 std::fs::write(
1746 runs.join(id).join("run.json"),
1747 serde_json::to_string_pretty(&value).unwrap(),
1748 )
1749 .unwrap();
1750 id.to_owned()
1751 }
1752
1753 #[test]
1754 fn fold_due_folds_a_superseded_run_same_as_any_other_terminal_one() {
1755 let dir = tempfile::tempdir().unwrap();
1756 let runs = dir.path().join("runs");
1757 let wt = dir.path().join("wt");
1758 let home = dir.path().to_path_buf();
1759 let disk = Disk::default();
1760 let now = ts("2026-09-05T00:00:00Z");
1761
1762 let superseded = due_run_with_status(
1763 &runs,
1764 &wt,
1765 "20260801-000000-cccc",
1766 SCHEMA,
1767 RunStatus::Superseded,
1768 );
1769
1770 let (folded, unreadable) =
1771 block_on(fold_due(&runs, &home, &wt, &disk, now)).expect("fold_due");
1772 assert_eq!(
1773 folded, 1,
1774 "a superseded run has nothing left for a human to check, so it folds \
1775 exactly like a merged one"
1776 );
1777 assert_eq!(unreadable, 0);
1778 assert!(
1779 read_meta(&runs, &superseded).is_ok(),
1780 "folding drops the worktree, not the record"
1781 );
1782 }
1783
1784 /// A throwaway repo with one commit on `main`, for a test that needs a
1785 /// real winner worktree `fold_run` can actually remove.
1786 fn init_repo(dir: &Path) {
1787 use crate::proc::Quiet as _;
1788 let run = |args: &[&str]| {
1789 let out = std::process::Command::new("git")
1790 .args(args)
1791 .current_dir(dir)
1792 .quiet()
1793 .output()
1794 .expect("spawn git");
1795 assert!(
1796 out.status.success(),
1797 "git {args:?} failed: {}",
1798 String::from_utf8_lossy(&out.stderr)
1799 );
1800 };
1801 run(&["init", "-b", "main"]);
1802 run(&["config", "user.name", "magi test"]);
1803 run(&["config", "user.email", "magi@example.com"]);
1804 std::fs::write(dir.join("README.md"), "# fixture\n").unwrap();
1805 run(&["add", "-A"]);
1806 run(&["commit", "-m", "init"]);
1807 }
1808
1809 /// A due, terminal run with a decided winner that still has a real
1810 /// worktree and branch on disk - so a test can tell whether folding it
1811 /// actually removed the winner or only left it registered as folded.
1812 async fn due_run_with_winner_worktree(
1813 runs: &Path,
1814 wt: &Path,
1815 repo: &Path,
1816 id: &str,
1817 status: RunStatus,
1818 ) -> (String, PathBuf) {
1819 use crate::run::{Candidate, Tally};
1820 use std::collections::BTreeMap;
1821
1822 let mut config = crate::config::Config::default();
1823 config.graph.worktree_root = Some(wt.to_path_buf());
1824 let mut state = RunState::new(
1825 repo.to_path_buf(),
1826 "main".to_owned(),
1827 "0000000000000000000000000000000000000000".to_owned(),
1828 String::new(),
1829 config,
1830 );
1831 state.id = id.to_owned();
1832 state.status = status;
1833 state.updated_at = ts("2026-08-01T00:00:00Z");
1834
1835 let winner_wt = state.worktree_root().join("cand-A");
1836 crate::git::worktree_add_branch(repo, &winner_wt, &format!("magi/{id}/A"), "main")
1837 .await
1838 .expect("winner worktree");
1839
1840 state.candidates.push(Candidate {
1841 index: 0,
1842 label: 'A',
1843 agent: "agent".to_owned(),
1844 branch: format!("magi/{id}/A"),
1845 worktree: winner_wt.clone(),
1846 summary: String::new(),
1847 stat: String::new(),
1848 files: 0,
1849 commits: 0,
1850 empty: false,
1851 failed: None,
1852 verified_noop: None,
1853 duration_ms: 0,
1854 folded: false,
1855 });
1856 state.tally = Some(Tally {
1857 first_choice: BTreeMap::from([('A', 1)]),
1858 borda: BTreeMap::new(),
1859 winner: 'A',
1860 rankings: 1,
1861 unanimous_initial: true,
1862 deliberated: false,
1863 changed_votes: 0,
1864 unanimous_final: true,
1865 tie_break: None,
1866 judges: 1,
1867 present: 1,
1868 quorum: 1,
1869 met_quorum: true,
1870 uncontested: None,
1871 });
1872 std::fs::create_dir_all(runs.join(id)).unwrap();
1873 std::fs::write(
1874 runs.join(id).join("run.json"),
1875 serde_json::to_string_pretty(&state).unwrap(),
1876 )
1877 .unwrap();
1878 (id.to_owned(), winner_wt)
1879 }
1880
1881 #[tokio::test]
1882 async fn fold_due_drops_a_superseded_runs_own_winner_worktree_but_keeps_a_readys() {
1883 let dir = tempfile::tempdir().unwrap();
1884 let runs = dir.path().join("runs");
1885 let wt = dir.path().join("wt");
1886 let repo = dir.path().join("repo");
1887 let home = dir.path().to_path_buf();
1888 let disk = Disk::default();
1889 let now = ts("2026-09-05T00:00:00Z");
1890 std::fs::create_dir_all(&repo).unwrap();
1891 init_repo(&repo);
1892
1893 let (superseded_id, superseded_wt) = due_run_with_winner_worktree(
1894 &runs,
1895 &wt,
1896 &repo,
1897 "20260801-000000-supw",
1898 RunStatus::Superseded,
1899 )
1900 .await;
1901 // `Ready` is auto-folded exactly like `Superseded` (both terminal and
1902 // non-resumable), but never landed anywhere - its own winner is
1903 // deliberately kept, which is what makes it the right contrast here.
1904 // `Blocked`/`Stalled` are `resumable()` and so never even reach
1905 // `fold_run` in the first place; they would not exercise the
1906 // `drop_winner` choice this test is about.
1907 let (ready_id, ready_wt) = due_run_with_winner_worktree(
1908 &runs,
1909 &wt,
1910 &repo,
1911 "20260801-000000-rdyw",
1912 RunStatus::Ready,
1913 )
1914 .await;
1915
1916 let (folded, unreadable) = fold_due(&runs, &home, &wt, &disk, now)
1917 .await
1918 .expect("fold_due");
1919 assert_eq!(folded, 2);
1920 assert_eq!(unreadable, 0);
1921
1922 assert!(
1923 !superseded_wt.exists(),
1924 "a superseded run's own winner never lands anywhere else, so its worktree \
1925 must be dropped exactly like a merged run's"
1926 );
1927 assert!(
1928 ready_wt.exists(),
1929 "a still-ready run's winner may yet be merged by hand - folding must not \
1930 touch it"
1931 );
1932
1933 // Both records survive the fold; only the worktrees differ.
1934 assert!(read_meta(&runs, &superseded_id).is_ok());
1935 assert!(read_meta(&runs, &ready_id).is_ok());
1936 }
1937}