magi/run.rs
1//! Run state: what happened, where it is stored, and how a run is resumed.
2//!
3//! Every node writes its result into [`RunState`] and the whole struct is
4//! flushed to `run.json` before the next node starts. That is what makes a run
5//! resumable: a competition can take an hour, and dying in review round four
6//! should not throw away three implementations, nine judge reads and a
7//! deliberation.
8//!
9//! Patches and raw agent transcripts are *not* in `run.json` — they live beside
10//! it under `artifacts/`, so the state file stays small enough to read by hand.
11use std::collections::BTreeMap;
12use std::path::{Path, PathBuf};
13
14use anyhow::{Context as _, Result, bail};
15use jiff::{Timestamp, Zoned};
16use serde::{Deserialize, Serialize};
17
18use crate::agent::SeatState;
19use crate::blind::Leak;
20use crate::config::{Config, MergeMode};
21use crate::verdict::{Finding, Rejection, ReviewVote, Severity};
22
23/// On-disk format version. Bumped when a field changes meaning, so a resumed
24/// run never half-reads a state file written by a different magi.
25///
26/// 2: added `RunStatus::Stalled`, `RunState::quota` (rate-limit losses), and
27/// the quorum fields on `Tally`. `RunState::load` already fails loudly and
28/// clearly on a schema mismatch; an old `run.json` from schema 1 now says so
29/// instead of silently half-reading.
30///
31/// 3: added `RunState::judge_skipped`. A solo candidate makes `judge` write
32/// only an event, leaving `judgements` empty forever — indistinguishable from
33/// "not yet judged" on every later reentry, which is what let `judge` re-run
34/// on a finished run and clobber its status back to `Judging`. The flag is
35/// the missing record of the fact that judging was skipped on purpose.
36///
37/// Also 3: a single-viable-candidate tally records `Tally::judges` as `0` and
38/// fills `Tally::uncontested`, instead of leaving the full roster size sitting
39/// next to a panel that never sat. A schema-2 record keeps reading as "0 of 3
40/// judges present" forever, because a tally is computed once and never
41/// recomputed on resume; the bump keeps that stale reading from being mixed
42/// with the new meaning.
43///
44/// 4: added `ReviewRound::progressed`. `graph::STAGNANT_LIMIT` counts
45/// consecutive rounds with `progressed == false` to decide whether the
46/// review loop should give up early, and a schema-3 record's default
47/// `false` would misreport a round that, at the time, actually committed a
48/// real diff — the field simply did not exist yet to say so. Without the
49/// bump, resuming an old multi-round review could spuriously trip the
50/// stagnation check on rounds that were never stagnant.
51///
52/// 5: added `RunStatus::Landing`. A run inside [`crate::land`]'s post-merge
53/// loop used to carry whatever status `merge` set before calling it forward
54/// unchanged - `Merged`, even while still watching CI or waiting on the
55/// owner's approval - which is also the one status [`RunStatus::resumable`]
56/// treats as finished. A daemon that gave this run's slot back to poll
57/// something else while an approval was outstanding, or one that simply
58/// crashed mid-land, had no way to tell "still landing" from "actually
59/// merged" and would either restart the whole competition or leave the run
60/// stuck reading as done. A schema-4 record has no notion of `Landing` at
61/// all, so this is a meaning a resumed old run cannot be guessed into rather
62/// than a value it can default to - hence the bump, not a `#[serde(default)]`.
63///
64/// 6: a deferred e2e is represented by an empty outcome list plus
65/// `ReviewRound::e2e_deferred`. Schema 5 treated that same empty list as an
66/// unconfigured, successful check, so schema-5 records are migrated with the
67/// old (not-deferred) meaning while older binaries reject schema-6 records.
68///
69/// 7: added `RunState::gate_ran`. An empty `RunState::gate` used to carry two
70/// meanings at once — "never attempted, or the last attempt was
71/// resource-blocked" (`graph::Runner::gate`'s retry case) and "attempted,
72/// zero commands configured, vacuously passed" (a repo with no
73/// `verify.gate`) — and nothing told them apart. `graph::Runner::merge`
74/// therefore read the second case as the first and refused forever: a
75/// review-only run with no gate commands configured reached `Gating` and
76/// then could never leave it. A schema-6 record's non-empty `gate` is
77/// migrated to `gate_ran = true` (a recorded attempt, real or historical,
78/// should not be spent again); an empty one migrates to `gate_ran = false`
79/// and is simply re-attempted by the next `gate()` call, which self-heals
80/// instantly for the zero-commands case.
81///
82/// 8: `ReviewRound::verified_head` used to be `None` for the overwhelming
83/// majority of rounds — every ordinary round that ran e2e against its own
84/// `head` in the main review loop never set it at all, leaving only the
85/// rare catch-up-on-a-different-commit case populated. A reader (a review
86/// prompt, `magi show`, the web UI) had no field to ask "which commit did
87/// this round's `e2e` actually check" and fell back to assuming it was
88/// always `head`, which is also what let a stale round's red output get
89/// quoted to a later round's reviewers as if it were about their patch, not
90/// an earlier one (see `ReviewRound::verification_summary`, which now exists
91/// so nowhere else has to guess). `verified_head` is now set whenever `e2e`
92/// held a real attempt (`E2eStatus::Passed`/`Failed`), always naming the
93/// commit actually checked instead of only the divergent case, and
94/// `ReviewRound::verified_at` is new alongside it. A schema-7 round's own
95/// unconditional main-loop check was always against `head` whether or not
96/// this field said so, so a `None` with a non-empty `e2e` migrates to
97/// `Some(head)` — a reconstruction of a fact that was always true, not a
98/// guess. `verified_at` has no historical value to reconstruct and stays
99/// `None`, which reads through `verification_summary` as "checked at:
100/// unknown" — an honest gap, not a fabricated time.
101/// 9: added [`RunState::operator_fixes`] — one record per `magi fix`
102/// invocation, routing specific, already-recorded findings to a fixer as a
103/// targeted, out-of-band fix outside the normal round sequence. Kept in a
104/// channel of its own rather than folded into [`ReviewRound`], because a
105/// reviewer's own severity and vote (copied verbatim onto
106/// [`OperatorFixFinding`]) must never be rewritten to look like the operator
107/// manufactured a blocking verdict — see `graph::Runner::fix_selected`. A
108/// schema-8 record has no operator-fix history at all, and
109/// `#[serde(default)]` reads an empty list as exactly that: "none happened",
110/// not an unknown gap. Nothing about an existing field's meaning changes.
111///
112/// 10: added `RunStatus::VerifiedNoop` and `Candidate::verified_noop`. Before
113/// this, an implementer that correctly concluded (with evidence) that a
114/// task's request was already satisfied elsewhere had no way to say so: the
115/// run ended the same way as one where every candidate simply failed to
116/// write anything — `after_implement` bailing with "no candidate produced a
117/// change; nothing to judge" and the run settling as a plain `Failed`. That
118/// conflated two very different facts (investigation run 391f's audit is
119/// what surfaced it: two attempts that had, correctly, found their fix
120/// already on `main`). A schema-9 record has no notion of either the new
121/// status or field, so a `VerifiedNoop` value is a meaning that cannot be
122/// reconstructed from an old record — hence the bump, not a
123/// `#[serde(default)]` for the status. `Candidate::verified_noop` alone
124/// *does* default-read as `None` on an old record, which is the honest
125/// reading: a run written before this schema never made the claim.
126///
127/// The report task 391f itself was raised from also named `6c5e`, `8df3` and
128/// `e9ce` as three more tasks whose implement wave ended the same
129/// diff-zero way, and the investigation traced all three — they do not
130/// share one cause.
131///
132/// `6c5e` and `8df3` are the same already-landed pattern as `391f`, not a
133/// coincidence: all three were re-queued together by a same-day audit of
134/// `done`-but-unlanded magi tasks (queue talk `20260912-115153-7216`,
135/// 2026-09-12 02:51–04:24), which found 17 magi tasks marked `done` with no
136/// merge to show for it and re-queued 16 of them, `6c5e` (a fix for the
137/// owner's `magi ask --thread` back-and-forth) and `8df3` (release
138/// automation) included. A second, same-day audit (talk
139/// `20260912-222053-07fe`, 13:20–13:36) then found 12 of those re-queued
140/// tasks — `391f`, `6c5e` and `8df3` among them — already merged by another
141/// route, and the owner had them deleted (`magi task rm`); `391f` alone
142/// survived because a daemon still held its run at the moment of deletion,
143/// which is the only reason any record of this group still exists to audit.
144/// Quoted directly from that second audit's own turn (talk `07fe`, so this
145/// reads without needing access to that talk store), naming both by id:
146///
147/// > 12件がマージ済み(対応不要)、3件が未実装(妥当)、2件が部分実装(要確認)でした。
148/// > **マージ済み → hold/rmを推奨:** 6c5e, 1ddc, fcf5, e25b, cea2, 391f, 3202,
149/// > b0a1, 5365, af85, 9f26, 8df3
150///
151/// — followed by the owner answering "削除!" and the agent confirming "11件
152/// 削除完了。391f はいま実行中のdaemonが掴んでいて削除できませんでした."
153/// `git log` independently confirms both fixes: the ask-back feature `6c5e`
154/// wanted landed as `f0df474` ("let the owner ask back on a question...",
155/// #93) on 2026-09-06, and the release-bump automation `8df3` wanted landed
156/// as `61005dd`/`bedd925` (open a release-bump PR on merge) on 2026-09-07
157/// and `116fcdc` (proportional version bump, #108) on 2026-09-08 — all
158/// before the 09-12 requeue. No run record survives the deletion for either
159/// task, so this schema's evidence is the audit transcript plus the
160/// independently re-checked `git log`, not a `run.json`.
161///
162/// `e9ce` is not that pattern at all, and is the reason the adoption guard
163/// below is all-or-nothing rather than "any candidate said so": its task
164/// asked an implementer to merge the real repository's `main` and cut a
165/// GitHub release — a destructive, out-of-worktree operation `AGENTS.md`
166/// names explicitly as not something to hand to an unattended candidate.
167/// Both of its runs (`20260912-053352-49ad`, `20260912-062629-bab1`)
168/// correctly refused, filed `magi ask` (questions `6196`, `6c9a`), and ended
169/// with an empty diff only because no answer arrived before the implement
170/// node's timeout — `49ad` looped `magi ask --wait` in the foreground for
171/// roughly 50 minutes as instructed before the timeout cut it off; `bab1`
172/// ended its turn moments after filing its question without ever actually
173/// blocking on the wait, a separate protocol slip this schema change does
174/// not attempt to fix. `49ad`'s own `candidates[0].summary` (quoted here
175/// because both records predate schema 10 and, separately, predate a
176/// still-unrelated struct change that already makes today's `magi show`
177/// refuse to parse either of them — `unknown field 'planner'` — so this is
178/// read straight from `run.json` on disk, not through that command):
179///
180/// > タスクの内容(READY 状態の run を実リポジトリの main に `merge --no-ff`
181/// > する、GitHub Release を作る)を精査した結果、これは全てこのワーカーの
182/// > worktree の外にある実リポジトリと GitHub 上の共有状態に対する不可逆な
183/// > 操作であり […] 私自身の運用ルール「Work only inside this worktree.
184/// > Nothing outside it is yours.」と正面から矛盾すると判断しました。
185///
186/// Neither candidate's reply carries the
187/// `NO CHANGE NEEDED` marker below, so both runs correctly stay `Failed`
188/// under this schema, not `VerifiedNoop`: a run blocked on an unanswered
189/// authorization question is not a verified no-op, and reading the two
190/// alike is exactly the misclassification the guard's per-candidate and
191/// whole-run conditions exist to refuse.
192///
193/// `RunStatus::Superseded` (added without a bump, still schema 10) is
194/// additive in the ordinary sense — an old build reading a run written under
195/// a newer one only had `RunStatus` variants to worry about before this, and
196/// a record already on disk never contained the new variant to begin with —
197/// but it is not additive in the sense every earlier schema bump on this
198/// constant was: `daemon::supersede_prior_runs` now rewrites a `Blocked` or
199/// `Stalled` run's `status` field *after* it was first written, once a later
200/// attempt at the same task lands. Any tooling outside magi that reads
201/// `run.json` and assumes a terminal `status` is permanent once set — a
202/// monitoring script polling for `blocked`, say — needs to know that
203/// `superseded` is where some of those records now go instead of staying put
204/// forever; see that function's own doc for exactly when.
205///
206/// Schema 11 adds [`RunState::released_to`] / [`RunState::released_branches`]:
207/// a run whose worktree a later attempt at the same task took over
208/// (`crate::handover`) can no longer be resumed from where it was. The fields
209/// are additive, but an older build would happily resume such a run into a
210/// worktree that no longer exists, which is the one thing the bump is for.
211///
212/// Schema 12 adds [`RunState::driver_exited`]: the driver records that it
213/// stopped walking the graph. Additive, but an older build would keep reading
214/// a long-lived daemon's pid as a live driver of a run that ended long ago.
215pub const SCHEMA: u32 = 12;
216
217/// Where a run got to.
218#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
219#[serde(rename_all = "snake_case")]
220pub enum RunStatus {
221 /// Worktrees being prepared.
222 Prep,
223 /// Candidates being implemented.
224 Implementing,
225 /// Judges ranking blind.
226 Judging,
227 /// Judges deliberating after a split.
228 Deliberating,
229 /// Final votes being collected privately.
230 Voting,
231 /// Winner in the review + verification loop.
232 Reviewing,
233 /// Gate commands running. Transient: `graph::Runner::gate` and
234 /// `graph::Runner::merge` always move a run on from here, whether or not
235 /// any gate commands are configured — see [`RunState::gate_status`] and
236 /// `SCHEMA`'s doc for schema 7, which fixed a repo with an empty
237 /// `verify.gate` stranding a review-only run in `Gating` forever.
238 Gating,
239 /// Inside [`crate::land`]'s post-merge loop: watching CI, running a fix
240 /// round, rebasing onto a moved base, or waiting on the owner's merge
241 /// approval. A run parked here while an approval is outstanding has
242 /// handed its daemon slot back — see [`crate::daemon`] — and resumes
243 /// through exactly this status, not a fresh competition.
244 Landing,
245 /// Winner merged.
246 Merged,
247 /// Winner passed the gate; merge was not requested.
248 Ready,
249 /// The judgement did not gather enough judges (e.g. rate limiting took out
250 /// seats), so the verdict is not trustworthy. The run stopped and kept its
251 /// work so it can be resumed or folded — it must never be confused with a
252 /// healthy `Ready`.
253 Stalled,
254 /// Review rounds exhausted with findings still open, or the gate failed.
255 Blocked,
256 /// The graph could not complete.
257 Failed,
258 /// Every candidate wrote nothing, and every one of them said why in a way
259 /// that survived [`crate::graph::Runner`]'s adoption guard: a clean CLI
260 /// exit, an actually-empty tree, no command left with an unconfirmed
261 /// exit status, and non-empty evidence. Distinct from `Failed` on
262 /// purpose — see `SCHEMA`'s doc for schema 10 — because the two read
263 /// identically to an operator glancing at a card ("nothing happened")
264 /// while meaning opposite things: one is an agent that could not do the
265 /// work, the other is an agent that checked and the work was already
266 /// done. Settles the task through [`crate::queue::Task::handed_off`], not
267 /// [`crate::queue::Task::fail`]: a human still has to look — the claim is
268 /// unverified by magi itself — and `Held` (not `Failed`-and-requeued)
269 /// means nothing retries the task unattended on the same unconfirmed
270 /// claim while that look is pending.
271 VerifiedNoop,
272 /// A later attempt at the same task already finished the job — see
273 /// `crate::queue::Task::superseded_attempts` — so this run's own
274 /// `Blocked`/`Stalled` no longer means anyone has to look at it. Written
275 /// only over one of those two statuses, only once the task itself is
276 /// `crate::queue::TaskStatus::Done`, and only onto an attempt that comes
277 /// before the one that succeeded. Unlike them, not [`Self::resumable`]:
278 /// there is nothing left to resume towards, the task already has its
279 /// answer, and a fold is free to clean this run's worktree away without
280 /// waiting on a human to confirm that first.
281 Superseded,
282}
283
284impl RunStatus {
285 /// Is this a terminal state?
286 pub fn done(self) -> bool {
287 matches!(
288 self,
289 Self::Merged
290 | Self::Ready
291 | Self::Stalled
292 | Self::Blocked
293 | Self::Failed
294 | Self::VerifiedNoop
295 | Self::Superseded
296 )
297 }
298
299 /// The name this status is written and shown under, matching the
300 /// `snake_case` serde spelling so a log line, an error message and the
301 /// JSON a phone reads all say the same word.
302 pub fn as_str(self) -> &'static str {
303 match self {
304 Self::Prep => "prep",
305 Self::Implementing => "implementing",
306 Self::Judging => "judging",
307 Self::Deliberating => "deliberating",
308 Self::Voting => "voting",
309 Self::Reviewing => "reviewing",
310 Self::Gating => "gating",
311 Self::Landing => "landing",
312 Self::Merged => "merged",
313 Self::Ready => "ready",
314 Self::Stalled => "stalled",
315 Self::Blocked => "blocked",
316 Self::Failed => "failed",
317 Self::VerifiedNoop => "verified_noop",
318 Self::Superseded => "superseded",
319 }
320 }
321
322 /// Label for a human-facing listing or report — the same word as
323 /// [`Self::as_str`] except where the machine spelling would read harsher
324 /// than the state actually is. `VerifiedNoop` is the one case: its own
325 /// `as_str` exists for logs, JSON and event messages, none of which
326 /// should quietly grow a second vocabulary, but a bare "verified_noop" in
327 /// a report reads like an error code, not the qualified, evidence-backed
328 /// claim it actually is.
329 pub fn display_label(self) -> &'static str {
330 match self {
331 Self::VerifiedNoop => "agent-verified no-op",
332 other => other.as_str(),
333 }
334 }
335
336 /// Can this run be carried on from where it stopped?
337 ///
338 /// Everything except a finished run and a failed one. `execute` skips
339 /// nodes already recorded, so re-entering is cheap wherever the run
340 /// stopped, and the alternative is always a fresh competition against
341 /// work that already exists.
342 ///
343 /// - `Stalled` re-asks only the seats whose absence collapsed the panel,
344 /// keeping the candidates that were already paid for.
345 /// - `Blocked` re-enters the review loop against a branch that is built.
346 /// - **A non-terminal status** means the run was interrupted: a parked
347 /// run waiting for its upgrade, or one whose daemon was killed. This
348 /// used to be excluded, which left run 4043 stuck at `reviewing` with
349 /// the deck telling the operator it could not be resumed - the one
350 /// state where resuming is the only sensible answer.
351 ///
352 /// `Failed` does not qualify: the graph could not complete and there is
353 /// no established point to continue from. Nor does a finished run, whose
354 /// answer is a new competition. Nor does `VerifiedNoop`: every candidate
355 /// already agreed nothing belongs in this worktree, and resuming would
356 /// only re-ask the same question — the answer is for a human to check
357 /// the evidence, not for the graph to run again. Nor does `Superseded`:
358 /// a later attempt at the same task already finished it, so there is
359 /// nothing left this run's own answer could still contribute.
360 ///
361 /// Whether anything is *already* driving the run is a separate question,
362 /// answered by `daemon::is_working_on` at the callers that need it.
363 pub fn resumable(self) -> bool {
364 !matches!(
365 self,
366 Self::Merged | Self::Ready | Self::Failed | Self::VerifiedNoop | Self::Superseded
367 )
368 }
369}
370
371/// One candidate implementation.
372#[derive(Debug, Clone, Serialize, Deserialize)]
373pub struct Candidate {
374 /// Position in the implementer list.
375 pub index: usize,
376 /// Blind label as presented to judges.
377 pub label: char,
378 /// Which agent wrote it. Recorded for the stats tables, never shown to a
379 /// judge.
380 pub agent: String,
381 /// Branch, named after the label so judges can inspect it without learning
382 /// the author.
383 pub branch: String,
384 /// Worktree path.
385 pub worktree: PathBuf,
386 /// Sanitized author summary.
387 #[serde(default)]
388 pub summary: String,
389 /// `git diff --stat`.
390 #[serde(default)]
391 pub stat: String,
392 /// Files touched.
393 #[serde(default)]
394 pub files: usize,
395 /// Commits ahead of base.
396 #[serde(default)]
397 pub commits: usize,
398 /// True when the agent produced no change at all.
399 #[serde(default)]
400 pub empty: bool,
401 /// Why this candidate is not in the running.
402 #[serde(default)]
403 pub failed: Option<String>,
404 /// The evidence this candidate gave for writing no change on purpose —
405 /// the `NO CHANGE NEEDED:` marker `prompt::implement`'s reply format
406 /// documents, verbatim. `Some` only when [`crate::graph`]'s adoption
407 /// guard accepted the claim: the CLI exited cleanly, the tree really is
408 /// empty, no command in the reply was left with an unconfirmed exit
409 /// status, and the evidence itself is non-empty. A candidate that wrote
410 /// nothing and said nothing about why — the ordinary empty loss — always
411 /// reads `None` here, same as one written before schema 10 ever existed.
412 #[serde(default)]
413 pub verified_noop: Option<String>,
414 /// Wall-clock time for the implementation.
415 #[serde(default)]
416 pub duration_ms: u64,
417 /// Whether the worktree has been folded away.
418 #[serde(default)]
419 pub folded: bool,
420}
421
422impl Candidate {
423 /// Can this candidate be judged?
424 pub fn viable(&self) -> bool {
425 self.failed.is_none() && !self.empty
426 }
427}
428
429/// One judge's independent ranking.
430#[derive(Debug, Clone, Serialize, Deserialize)]
431pub struct Judgement {
432 /// Judge seat number, 1-based.
433 pub judge: usize,
434 /// Seat key.
435 pub seat: String,
436 /// Agent occupying the seat.
437 pub agent: String,
438 /// Best-first labels.
439 #[serde(default)]
440 pub ranking: Vec<char>,
441 /// Per-label justification.
442 #[serde(default)]
443 pub reasons: BTreeMap<String, String>,
444 /// Self-reported confidence.
445 #[serde(default)]
446 pub confidence: Option<u8>,
447 /// Order the candidates were presented in, as candidate indices.
448 #[serde(default)]
449 pub order: Vec<usize>,
450 /// Why this judge has no ranking.
451 #[serde(default)]
452 pub failed: Option<String>,
453 /// Wall-clock time.
454 #[serde(default)]
455 pub duration_ms: u64,
456}
457
458/// One judge's turn in a deliberation round.
459#[derive(Debug, Clone, Serialize, Deserialize)]
460pub struct DeliberationTurn {
461 /// Judge seat number, 1-based.
462 pub judge: usize,
463 /// Agent occupying the seat.
464 pub agent: String,
465 /// The argument, as written.
466 pub body: String,
467 /// Where the judge stood at the end of the turn.
468 #[serde(default)]
469 pub tentative: Option<char>,
470}
471
472/// A deliberation round.
473#[derive(Debug, Clone, Serialize, Deserialize)]
474pub struct DeliberationRound {
475 /// 1-based round number.
476 pub round: usize,
477 /// Turns, in the order they were taken.
478 pub turns: Vec<DeliberationTurn>,
479}
480
481/// A final vote, collected privately.
482#[derive(Debug, Clone, Serialize, Deserialize)]
483pub struct VoteRecord {
484 /// Judge seat number, 1-based.
485 pub judge: usize,
486 /// Agent occupying the seat.
487 pub agent: String,
488 /// The vote.
489 #[serde(default)]
490 pub vote: Option<char>,
491 /// Why.
492 #[serde(default)]
493 pub reason: String,
494 /// Did this judge move from its initial first choice?
495 #[serde(default)]
496 pub changed: bool,
497}
498
499/// A seat that was taken out by a CLI rate limit / quota, recorded so a run
500/// whose panel collapsed does not masquerade as a healthy one.
501#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
502pub struct QuotaLoss {
503 /// Seat key, e.g. `judge-1` or `review-2`.
504 pub seat: String,
505 /// Node that was running, e.g. `judge`, `vote`, `review`.
506 pub node: String,
507 /// When the CLI reported the limit.
508 pub at: Timestamp,
509 /// Reset hint if the CLI printed one, free text.
510 #[serde(default)]
511 pub reset: Option<String>,
512}
513
514/// A newly created lockfile of a package manager the directory does not use,
515/// which a rescue commit left untracked instead of committing. Recorded so the
516/// omission is visible: the file may be one the task really wanted.
517#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
518pub struct Withheld {
519 /// Repo-relative path.
520 pub path: String,
521 /// The manager the file belongs to.
522 pub manager: String,
523 /// The tracked file (or missing manifest) that made it foreign.
524 pub kept_by: String,
525 /// Node whose rescue commit withheld it.
526 pub node: String,
527 /// When.
528 pub at: Timestamp,
529}
530
531/// The mechanical count.
532#[derive(Debug, Clone, Serialize, Deserialize)]
533pub struct Tally {
534 /// First-choice votes per label.
535 pub first_choice: BTreeMap<char, usize>,
536 /// Borda points from the initial rankings, used only to break a tie.
537 pub borda: BTreeMap<char, usize>,
538 /// The winning label.
539 pub winner: char,
540 /// How many judges produced a usable ranking. A panel of one is not a
541 /// consensus and must not be reported as a split.
542 #[serde(default)]
543 pub rankings: usize,
544 /// Did every judge's *initial* first choice agree?
545 pub unanimous_initial: bool,
546 /// Was deliberation run?
547 pub deliberated: bool,
548 /// Judges who moved between their initial ranking and their final vote.
549 pub changed_votes: usize,
550 /// Did the final votes agree?
551 pub unanimous_final: bool,
552 /// How the tie was broken, when it had to be.
553 #[serde(default)]
554 pub tie_break: Option<String>,
555 /// Configured judge count — the size of the full panel. `0` when no
556 /// panel was asked (see `uncontested`), not the roster size a panel that
557 /// never sat would have had.
558 #[serde(default)]
559 pub judges: usize,
560 /// Judges who actually contributed to the decision (not taken out by a
561 /// rate limit and producing a usable rank or vote).
562 #[serde(default)]
563 pub present: usize,
564 /// How many judges are required for a trustworthy verdict. Chosen as a
565 /// strict majority (`judges / 2 + 1`): a verdict backed by a minority must
566 /// never be presented as a healthy one, while a bare majority is still
567 /// real signal. A one-candidate run needs no quorum.
568 #[serde(default)]
569 pub quorum: usize,
570 /// `present >= quorum`, or no quorum was required.
571 #[serde(default)]
572 pub met_quorum: bool,
573 /// Why no panel was asked, when none was: a single viable candidate, or
574 /// a review-only run that never competed. `None` when judges actually
575 /// ranked and voted — including when too few of them survived to reach
576 /// quorum, which is a collapse and must keep reading as one.
577 #[serde(default)]
578 pub uncontested: Option<String>,
579}
580
581/// One reviewer's report in a round.
582#[derive(Debug, Clone, Serialize, Deserialize)]
583pub struct ReviewRecord {
584 /// Reviewer seat number, 1-based.
585 pub reviewer: usize,
586 /// Agent occupying the seat.
587 pub agent: String,
588 /// Reviewer prose.
589 #[serde(default)]
590 pub summary: String,
591 /// Findings, with magi-assigned ids.
592 #[serde(default)]
593 pub findings: Vec<Finding>,
594 /// This seat's initial vote. `None` on a record predating votes, exactly
595 /// like a round that genuinely had none cast — never a stand-in for a
596 /// vote that was lost.
597 #[serde(default)]
598 pub vote: Option<ReviewVote>,
599 /// Why this reviewer produced nothing.
600 #[serde(default)]
601 pub failed: Option<String>,
602 /// Wall-clock time.
603 #[serde(default)]
604 pub duration_ms: u64,
605 /// How many times this seat was asked before it settled — 0 for a first
606 /// answer, N after N nudges (`ask_json_wave`'s retry loop). Read this
607 /// together with [`Self::failed`], never `failed` alone: `failed: Some(_)`
608 /// with `attempts == 0` is a seat that never answered at all, while
609 /// `failed: None` with `attempts > 0` is one that only came back after a
610 /// nudge — recovered, not silent — and the two must not look the same in
611 /// history. A record written before this field existed defaults to `0`,
612 /// which under-reports a pre-existing retry rather than inventing one;
613 /// see `ask_json_wave`'s own doc for where this is filled in.
614 #[serde(default)]
615 pub attempts: usize,
616}
617
618/// One seat's revote during a round's reconsideration (see
619/// [`ReviewRound::reconsideration`]).
620#[derive(Debug, Clone, Serialize, Deserialize)]
621pub struct ReviewRevoteRecord {
622 /// Reviewer seat number, 1-based.
623 pub reviewer: usize,
624 /// Agent occupying the seat.
625 pub agent: String,
626 /// The revote. `None` when the seat did not answer.
627 #[serde(default)]
628 pub vote: Option<ReviewVote>,
629 /// Why.
630 #[serde(default)]
631 pub reason: String,
632 /// Why this seat produced no revote.
633 #[serde(default)]
634 pub failed: Option<String>,
635}
636
637/// One fix round spent on a failing `verify.gate`: which fixer, what it did to
638/// the tree, and the gate output it was shown.
639#[derive(Debug, Clone, Serialize, Deserialize)]
640pub struct GateFixRecord {
641 /// Agent that applied the fix.
642 pub agent: String,
643 /// The gate output the fixer was handed, tail-truncated.
644 pub failed: Vec<CommandOutcome>,
645 /// What the fixer said it changed.
646 #[serde(default)]
647 pub notes: String,
648 /// Did the fix produce a commit?
649 #[serde(default)]
650 pub committed: bool,
651 /// Why the fix step produced nothing.
652 #[serde(default)]
653 pub error: Option<String>,
654}
655
656/// One fixer round spent on a rebase that stopped on a conflict: which fixer,
657/// what it was shown and whether git says the rebase then finished.
658///
659/// The number of these is the budget spent (`graph.review_rounds` is the cap),
660/// written to disk *before* the fixer runs so a park or crash cannot hand the
661/// round back.
662#[derive(Debug, Clone, Serialize, Deserialize)]
663pub struct RebaseFixRecord {
664 /// Agent asked to resolve the conflict.
665 pub agent: String,
666 /// Paths git reported unmerged when the round started.
667 pub paths: Vec<String>,
668 /// The branch tip the rebase started from, so a resumed run can tell a
669 /// worktree still holding those files from one with edits of its own.
670 #[serde(default)]
671 pub from: Option<String>,
672 /// Did git report the rebase finished after the round?
673 #[serde(default)]
674 pub finished: bool,
675 /// Why the round produced nothing (agent failure, quota).
676 #[serde(default)]
677 pub error: Option<String>,
678}
679
680/// The fixer's response to a round.
681#[derive(Debug, Clone, Serialize, Deserialize)]
682pub struct FixRecord {
683 /// Agent that applied the fixes.
684 pub agent: String,
685 /// Finding ids acted on.
686 #[serde(default)]
687 pub addressed: Vec<String>,
688 /// Findings declined, with reasons.
689 #[serde(default)]
690 pub rejected: Vec<Rejection>,
691 /// What changed.
692 #[serde(default)]
693 pub notes: String,
694 /// Did the fix produce a commit?
695 #[serde(default)]
696 pub committed: bool,
697 /// Why the fix step produced nothing.
698 #[serde(default)]
699 pub failed: Option<String>,
700 /// Wall-clock time.
701 #[serde(default)]
702 pub duration_ms: u64,
703 /// How the fixer's own seat was made to answer when its CLI turn ended
704 /// cleanly but without an addressed/rejected report — see
705 /// [`graph::Runner::continue_fix_report`]. `None` for a record written
706 /// before this existed, which must read as "unknown", not as
707 /// [`ContinuationOutcome::NotNeeded`]: an old run really may have hit
708 /// this exact gap and simply had no mechanism to say so.
709 #[serde(default)]
710 pub continuation: Option<ContinuationRecord>,
711}
712
713/// How a node recovered — or failed to recover — a structured report after
714/// the CLI's own turn ended cleanly (a usable, non-empty, exit-0 reply)
715/// without it. A clean CLI turn is not the same fact as the node's own work
716/// being done — see the `fix` node's `continue_fix_report`, which is what
717/// produces this.
718#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
719pub enum ContinuationOutcome {
720 /// The first reply already carried the report; nothing was resumed.
721 NotNeeded,
722 /// A follow-up call in the same session recovered the report.
723 Resumed,
724 /// The continuation budget was spent without ever recovering it.
725 Exhausted,
726 /// A continuation attempt hit the CLI's rate limit; not retried further
727 /// — a quota fails the same way again immediately.
728 QuotaLost,
729 /// No session was left to resume into, so nothing was attempted.
730 NoSession,
731}
732
733/// Cost and outcome of one node's attempt to recover a missing report by
734/// resuming its own seat.
735#[derive(Debug, Clone, Copy, Serialize, Deserialize)]
736pub struct ContinuationRecord {
737 /// Follow-up calls made to the same seat. `0` when the outcome is
738 /// [`ContinuationOutcome::NotNeeded`] or [`ContinuationOutcome::NoSession`].
739 pub attempts: usize,
740 /// Wall-clock time spent on those follow-up calls, summed — not counting
741 /// the original call whose reply this is recovering from.
742 pub cumulative_wait_ms: u64,
743 /// What ended the loop.
744 pub outcome: ContinuationOutcome,
745}
746
747impl ContinuationRecord {
748 /// The report was already there on the first try.
749 pub fn not_needed() -> Self {
750 Self {
751 attempts: 0,
752 cumulative_wait_ms: 0,
753 outcome: ContinuationOutcome::NotNeeded,
754 }
755 }
756}
757
758/// What happened to one operator-selected finding after the fixer ran, as
759/// part of an [`OperatorFixRequest`].
760#[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)]
761#[serde(rename_all = "snake_case")]
762pub enum OperatorFixOutcome {
763 /// The request has not run yet, or never got far enough to report.
764 #[default]
765 Pending,
766 /// The fixer's adoption report named this finding as addressed.
767 Addressed,
768 /// The fixer's adoption report declined it, with an argument.
769 Rejected {
770 /// The fixer's own reason.
771 why: String,
772 },
773 /// The fixer never delivered a usable adoption report at all — a
774 /// dropped stream, a quota hit, or a continuation budget spent without
775 /// recovering one (see `graph::Runner::continue_fix_report`). Distinct
776 /// from `Rejected`, which needs an argument this never produced, and
777 /// never written back as "addressed" or silently left `Pending` — a
778 /// gap in the report is its own outcome, not evidence either way about
779 /// the finding.
780 Unreported,
781}
782
783/// One finding an operator selected for [`OperatorFixRequest`], with the
784/// provenance a reviewer originally gave it, copied here verbatim.
785///
786/// Severity and vote are snapshots, never recomputed and never treated as
787/// blocking just because an operator picked the finding — only
788/// [`Severity::blocks`] on the original [`ReviewRecord`] decides that. This
789/// type exists so an operator's selection is an auditable *addition* to the
790/// record, not a rewrite of what a reviewer actually said.
791#[derive(Debug, Clone, Serialize, Deserialize)]
792pub struct OperatorFixFinding {
793 /// Finding id, e.g. `R2-1-3`.
794 pub id: String,
795 /// Severity as the reviewer recorded it.
796 pub severity: Severity,
797 /// The reviewer seat's overall vote for the round this finding came
798 /// from, if one was cast.
799 #[serde(default)]
800 pub reviewer_vote: Option<ReviewVote>,
801 /// Review round the finding was raised in.
802 pub round: usize,
803 /// That round's own head — the commit the finding was actually raised
804 /// against, used for the freshness check against the branch's current
805 /// head at request time.
806 pub round_head: String,
807 /// Reviewer seat number, 1-based.
808 pub reviewer: usize,
809 /// Agent occupying that seat.
810 pub agent: String,
811 /// File the finding concerns.
812 #[serde(default)]
813 pub file: Option<String>,
814 /// Line the finding concerns.
815 #[serde(default)]
816 pub line: Option<u32>,
817 /// One-line summary.
818 pub title: String,
819 /// The argument.
820 #[serde(default)]
821 pub detail: String,
822 /// What happened to this finding after the fixer ran.
823 #[serde(default)]
824 pub outcome: OperatorFixOutcome,
825}
826
827/// One `magi fix` invocation: the operator's own record of which
828/// already-recorded findings they routed to a fixer, why, and what came
829/// back. See [`SCHEMA`]'s doc for schema 9 on why this is a channel of its
830/// own rather than a field on [`ReviewRound`].
831#[derive(Debug, Clone, Serialize, Deserialize)]
832pub struct OperatorFixRequest {
833 /// When `magi fix` was invoked.
834 pub requested_at: Timestamp,
835 /// The operator's own reasoning. Required and never empty at the CLI —
836 /// the audit trail this feature exists for.
837 pub reason: String,
838 /// The findings selected, each with its own provenance and outcome.
839 pub findings: Vec<OperatorFixFinding>,
840 /// The branch's head at the moment this request started executing.
841 pub head_at_request: String,
842 /// Did the operator pass `--allow-stale`?
843 pub allow_stale: bool,
844 /// Did any selected finding's own `round_head` differ from
845 /// `head_at_request`? Kept distinct from `allow_stale` — flipping that
846 /// flag does not retroactively make a request that was actually fresh
847 /// read as stale, or the reverse.
848 pub stale: bool,
849 /// The fixer's own attempt, once dispatched.
850 #[serde(default)]
851 pub fix: Option<FixRecord>,
852 /// Head after the fixer's commit, when it produced one.
853 #[serde(default)]
854 pub result_head: Option<String>,
855 /// The review-only run opened to re-verify the change, when one was
856 /// actually committed. `None` when nothing changed, so there was
857 /// nothing new to re-review — never left implicit as "not gotten to
858 /// yet".
859 #[serde(default)]
860 pub follow_up_review_run: Option<String>,
861}
862
863/// A command a seat's own CLI reported running, kept for `magi show` and for
864/// telling "this seat's turn ended" apart from "the process it started is
865/// done" — see `agent::CommandEvidence`, which is the only source this is
866/// ever built from. Never something magi polled or supervised; a command the
867/// CLI never reported finishing (or a CLI this crate has no adapter for at
868/// all) simply has no entry here, which must read as "unknown", not as
869/// "nothing ran".
870#[derive(Debug, Clone, Serialize, Deserialize)]
871pub struct JobRecord {
872 /// Graph node the seat belongs to, e.g. `"implement"`, `"fix"`.
873 pub node: String,
874 /// Review round this job belongs to, for a `"review"`/`"fix"` node —
875 /// `None` for every other node, where rounds do not apply, and for every
876 /// record written before this was tracked. Lets a reader ask "what did
877 /// this seat itself actually run this round", distinct from and never
878 /// substituted for magi's own recorded `ReviewRound::e2e` — an absent
879 /// entry here means unobserved, not that nothing ran (see this type's
880 /// own doc).
881 #[serde(default)]
882 pub round: Option<usize>,
883 /// Seat key, e.g. `"impl-A"`.
884 pub seat: String,
885 /// The CLI's own id for this command.
886 pub id: String,
887 /// The command itself, as the CLI reported it.
888 pub description: String,
889 /// When this evidence was captured — the moment this seat's reply
890 /// carrying it was read, not the command's own start time, which no
891 /// adapter here currently has. A lower bound on staleness only.
892 pub checked_at: Timestamp,
893 /// What the CLI reported for it.
894 pub status: JobStatus,
895 /// Exit code the CLI reported.
896 pub exit_code: Option<i32>,
897 /// Tail of the command's own output, when reported.
898 #[serde(default)]
899 pub result_summary: String,
900 /// Which CLI/event stream this came from, e.g. `"codex"`.
901 pub source: String,
902}
903
904/// What a [`JobRecord`]'s own CLI reported for it. There is no `Running`
905/// variant: nothing here is ever polled live, so "still running" and
906/// "finished but never reported" are the same absence of evidence, not a
907/// state this type can name — see [`JobRecord`]'s own doc.
908#[derive(Debug, Clone, Copy, Serialize, Deserialize, PartialEq, Eq)]
909pub enum JobStatus {
910 /// The command's own reported exit code was `0`.
911 Completed,
912 /// The command's own reported exit code was non-zero.
913 Failed,
914 /// The CLI reported this command but not a readable exit code.
915 Unknown,
916}
917
918/// Outcome of one shell command.
919#[derive(Debug, Clone, Serialize, Deserialize)]
920pub struct CommandOutcome {
921 /// The command, as configured.
922 pub command: String,
923 /// Exit code, `None` on timeout or signal.
924 pub code: Option<i32>,
925 /// Tail of the combined output, for the report and the fix prompt.
926 #[serde(default)]
927 pub output_tail: String,
928 /// Wall-clock time.
929 #[serde(default)]
930 pub duration_ms: u64,
931 /// Set only by magi itself, never inferred from `output_tail`: the
932 /// configured command was never actually run because a resource it
933 /// needs — right now, only the shared build cache's lease or the
934 /// freshness check that must precede using it — was not available
935 /// within budget. Distinct from an ordinary failure or timeout (both of
936 /// which *did* run something and are evidence about the patch); this is
937 /// evidence about the machine, and must never be read as a verdict on
938 /// the tree it named. `#[serde(default)]` so every record written
939 /// before this field existed keeps reading as `false` — exactly what it
940 /// was.
941 #[serde(default)]
942 pub resource_blocked: bool,
943}
944
945/// Substrings that mark a Cargo/rustc/link failure: the toolchain could not
946/// produce a binary to run at all, as opposed to producing one that ran and
947/// failed. A Windows link race against a shared `CARGO_TARGET_DIR` (see
948/// AGENTS.md, "Running magi on magi") looks exactly like a red command
949/// otherwise, and a run has concluded `Blocked` on nothing but that race.
950const BUILD_FAILURE_MARKERS: &[&str] = &[
951 "error: could not compile",
952 "error: linking with",
953 "LINK : fatal error",
954 "fatal error LNK",
955];
956
957impl CommandOutcome {
958 /// Did it pass?
959 pub fn ok(&self) -> bool {
960 self.code == Some(0)
961 }
962
963 /// Did this command fail because the code could not be built or linked,
964 /// rather than because it ran and produced a wrong result? A failure here
965 /// is not a verdict on the patch under review.
966 pub fn build_failed(&self) -> bool {
967 !self.ok()
968 && BUILD_FAILURE_MARKERS
969 .iter()
970 .any(|m| self.output_tail.contains(m))
971 }
972}
973
974/// One review + verify + fix round.
975#[derive(Debug, Clone, Serialize, Deserialize)]
976pub struct ReviewRound {
977 /// 1-based round number.
978 pub round: usize,
979 /// Commit the round reviewed.
980 pub head: String,
981 /// The commit `e2e` was actually attempted against. Set whenever an
982 /// attempt was dispatched (`e2e_status()` reads `Passed`, `Failed`, or
983 /// `ResourceBlocked`), naming that commit even when it equals `head` —
984 /// never left implicit, because an implicit "must have been `head`" is
985 /// exactly what let a later round quote an earlier round's result
986 /// without saying which commit it came from. A resource-blocked attempt
987 /// still targeted a specific commit even though no command finished, and
988 /// leaving that unrecorded is exactly what made a *fresh* blocked
989 /// attempt read the same as an untracked one from before schema 8.
990 /// `None` only when nothing was attempted at all (`NotConfigured`,
991 /// `Deferred`). See `SCHEMA`'s doc for schema 8 for why this broadened
992 /// from only the catch-up-on-a-different-commit case.
993 #[serde(default)]
994 pub verified_head: Option<String>,
995 /// When the attempt behind `verified_head` actually ran. `None` on
996 /// every record written before schema 8, and on a round where nothing
997 /// ran —
998 /// both read as "unknown", not as "now" or "never asked".
999 #[serde(default)]
1000 pub verified_at: Option<Timestamp>,
1001 /// Reviewer reports.
1002 pub reviews: Vec<ReviewRecord>,
1003 /// E2E command outcomes for this round.
1004 #[serde(default)]
1005 pub e2e: Vec<CommandOutcome>,
1006 /// True when the first verify attempt this round could not build or
1007 /// link, and `e2e` above holds a second attempt run before concluding.
1008 /// A run must never be decided on a red it could not tell from an
1009 /// unrelated build race.
1010 #[serde(default)]
1011 pub verify_retried: bool,
1012 /// True when `e2e` was intentionally left empty this round: the round
1013 /// already had blocking findings and another round was available, so
1014 /// `graph::Runner::review_loop` sent the fixer straight at them instead
1015 /// of spending a full verify run on a head it already knew would need
1016 /// another fix. Distinct from an `e2e` that is simply empty because
1017 /// `verify.e2e` has no commands configured — `e2e.is_empty()` alone
1018 /// cannot tell those apart, and conflating them is exactly how a
1019 /// deferred check would get painted green. A record written before this
1020 /// field existed defaults to `false`, which is the truth for it: every
1021 /// round used to run e2e unconditionally.
1022 #[serde(default)]
1023 pub e2e_deferred: bool,
1024 /// Why `e2e` was deferred, set only when [`Self::e2e_deferred`] is true.
1025 /// Carried to the fixer's prompt and shown in the report so "deferred"
1026 /// never reads as silence.
1027 #[serde(default)]
1028 pub e2e_defer_reason: Option<String>,
1029 /// Fixer response, absent when the round was already clean.
1030 #[serde(default)]
1031 pub fix: Option<FixRecord>,
1032 /// Findings that hold the merge.
1033 #[serde(default)]
1034 pub blocking: usize,
1035 /// Reviewer seats that answered (did not time out, crash, or return
1036 /// something unparsable).
1037 #[serde(default)]
1038 pub answered: usize,
1039 /// Reviewer seats the round expected an answer from — normally
1040 /// `graph.reviewers`, but recorded per round so a config change between
1041 /// runs never has to be inferred from history.
1042 #[serde(default)]
1043 pub expected: usize,
1044 /// Round ended with no blocking findings and green verification, judged
1045 /// against the seats that answered. See [`Self::incomplete`] for whether
1046 /// that verdict is missing input.
1047 #[serde(default)]
1048 pub clean: bool,
1049 /// Did the tree actually move against `base` this round, comparing the
1050 /// diff after the fix to the diff the reviewers saw at the start of the
1051 /// round?
1052 ///
1053 /// Never derived from the fixer's own `addressed`/`rejected` count: that
1054 /// self-report has been caught lying twice on this workload (runs `b455`
1055 /// and `6218`, both of which committed a real, substantial diff while
1056 /// reporting `0 addressed`). `git` does not lie about whether the tree
1057 /// changed, so this is what `graph::Runner::review_loop` counts rounds of
1058 /// no progress against. Absent on a round with no fix attempt (already
1059 /// clean, or the round the budget ran out on), where it defaults to
1060 /// `false` and is not consulted.
1061 #[serde(default)]
1062 pub progressed: bool,
1063 /// Did the seats' initial votes ([`ReviewRecord::vote`]) disagree?
1064 #[serde(default)]
1065 pub vote_split: bool,
1066 /// One round of revoting, run only when `vote_split`: each seat that cast
1067 /// an initial vote reads every seat's findings and votes, then revotes.
1068 /// Empty when the initial votes already agreed, the same as a solo
1069 /// candidate leaving `deliberation` empty.
1070 #[serde(default)]
1071 pub reconsideration: Vec<ReviewRevoteRecord>,
1072 /// The round's verdict: the most cautious vote among the seats that
1073 /// answered, using each seat's revote where reconsideration ran and its
1074 /// initial vote otherwise. `None` when no seat produced a usable vote —
1075 /// including every record written before votes existed, which is the
1076 /// truth for those rounds, not a gap in this one.
1077 #[serde(default)]
1078 pub verdict: Option<ReviewVote>,
1079}
1080
1081impl ReviewRound {
1082 /// Did at least one reviewer seat fail to answer this round?
1083 pub fn incomplete(&self) -> bool {
1084 self.answered < self.expected
1085 }
1086
1087 /// The honest state of this round's e2e leg.
1088 ///
1089 /// Never derive this from `e2e.is_empty()` alone anywhere else in the
1090 /// codebase — `NotConfigured` and `Deferred` both leave it empty, and
1091 /// only this method (backed by [`Self::e2e_deferred`]) tells them apart.
1092 /// A resource-blocked attempt is checked first and ahead of both: `e2e`
1093 /// is non-empty for it too, but `CommandOutcome::resource_blocked` says
1094 /// no command actually ran, and reading that as `Failed` is exactly how
1095 /// shared build-cache contention gets misreported as a verdict on the
1096 /// patch (see `CommandOutcome::resource_blocked`'s own doc).
1097 pub fn e2e_status(&self) -> E2eStatus {
1098 if self.e2e.iter().any(|o| o.resource_blocked) {
1099 E2eStatus::ResourceBlocked
1100 } else if !self.e2e.is_empty() {
1101 if self.e2e.iter().all(CommandOutcome::ok) {
1102 E2eStatus::Passed
1103 } else {
1104 E2eStatus::Failed
1105 }
1106 } else if self.e2e_deferred {
1107 E2eStatus::Deferred
1108 } else {
1109 E2eStatus::NotConfigured
1110 }
1111 }
1112
1113 /// Facts about this round's verification leg, judged against
1114 /// `current_head` — the commit whoever is asking is actually looking at
1115 /// right now. `None` when there is nothing worth surfacing: no
1116 /// `verify.e2e` configured, or the round's own check came back green (a
1117 /// passing result needs no skepticism attached to it, and an unread
1118 /// `None` is exactly what keeps a quiet round quiet instead of padding
1119 /// every prompt with "everything was fine").
1120 ///
1121 /// This is the single place that turns `e2e`/`e2e_deferred`/
1122 /// `verified_head`/`verified_at` into text. Every prompt and report that
1123 /// shows a round's verification result must build its wording from this,
1124 /// not re-derive its own summary at the call site — a hand-rolled
1125 /// version at one more place is exactly how "an old red read as today's
1126 /// answer" comes back through a different door (see the incident this
1127 /// type exists to prevent, recorded alongside `SCHEMA`'s doc for schema
1128 /// 8).
1129 pub fn verification_summary(&self, current_head: &str) -> Option<VerificationSummary> {
1130 let status = self.e2e_status();
1131 if matches!(status, E2eStatus::NotConfigured | E2eStatus::Passed) {
1132 return None;
1133 }
1134 let commit = match &self.verified_head {
1135 Some(h) if h == current_head => {
1136 format!("commit {} (this is the head being looked at now)", short(h))
1137 }
1138 Some(h) => format!("commit {} (an earlier head, since superseded)", short(h)),
1139 None => "commit unknown (no command finished checking one)".to_owned(),
1140 };
1141 let checked_at = match self.verified_at {
1142 Some(t) => format!("checked at {t}"),
1143 None => "checked at: unknown (recorded before this was tracked)".to_owned(),
1144 };
1145 let result = match status {
1146 E2eStatus::NotConfigured | E2eStatus::Passed => unreachable!("checked above"),
1147 E2eStatus::Failed => "result: FAILED".to_owned(),
1148 E2eStatus::Deferred => format!(
1149 "result: not run this round yet — deferred to the fixer{}. Not passed, not \
1150 failed.",
1151 self.e2e_defer_reason
1152 .as_deref()
1153 .map(|why| format!(" ({why})"))
1154 .unwrap_or_default()
1155 ),
1156 E2eStatus::ResourceBlocked => "result: could not run — the shared build cache was \
1157 not available. This is evidence about the machine, \
1158 not about the patch."
1159 .to_owned(),
1160 };
1161 let label = format!("round {}, {commit}, {checked_at}\n{result}", self.round);
1162 // `Failed` names the command that actually ran and failed;
1163 // `ResourceBlocked` names the operation magi was waiting on (or the
1164 // freshness check it could not confirm) — `command` still says what
1165 // was attempted even though nothing finished, and leaving it out is
1166 // exactly how a reviewer or fixer lost the one thing this leg *can*
1167 // still tell them: what was being checked, not whether it passed.
1168 let tail = matches!(status, E2eStatus::Failed | E2eStatus::ResourceBlocked).then(|| {
1169 self.e2e
1170 .iter()
1171 .filter(|o| !o.ok())
1172 .map(|o| format!("$ {}\n{}\n", o.command, o.output_tail))
1173 .collect::<String>()
1174 });
1175 Some(VerificationSummary { label, tail })
1176 }
1177}
1178
1179/// The honest state of a round's e2e leg. See [`ReviewRound::e2e_status`].
1180#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1181pub enum E2eStatus {
1182 /// `verify.e2e` has no commands configured.
1183 NotConfigured,
1184 /// Skipped this round on purpose: blocking findings already required a
1185 /// fix, so the round went straight to the fixer instead of spending a
1186 /// full verify run on a head it already knew would need another pass.
1187 Deferred,
1188 /// Ran, and every command exited 0.
1189 Passed,
1190 /// Ran, and at least one command did not exit 0.
1191 Failed,
1192 /// Magi could not even get a command to run — the shared build cache's
1193 /// lease or freshness check was not available within budget. Evidence
1194 /// about the machine, never a verdict on the tree it named; must not be
1195 /// shown or counted the same as [`Self::Failed`].
1196 ResourceBlocked,
1197}
1198
1199/// [`ReviewRound::verification_summary`]'s output: the facts, pre-worded, for
1200/// a prompt or report to place under its own heading. Kept as two pieces
1201/// rather than one pre-joined string so a caller that wants to insert its own
1202/// note between the label and the raw command tail (see `prompt::review`) can
1203/// do so without re-parsing text back apart.
1204#[derive(Debug, Clone)]
1205pub struct VerificationSummary {
1206 /// Round, commit, freshness and result — always present.
1207 pub label: String,
1208 /// Raw `$ command` / output tail, present for `result: FAILED` and for
1209 /// a resource-blocked attempt (naming the operation magi was waiting on,
1210 /// even though nothing finished) — absent for every other result, which
1211 /// has nothing to add past the label.
1212 pub tail: Option<String>,
1213}
1214
1215/// The honest state of a run's final gate. See [`RunState::gate_status`].
1216#[derive(Debug, Clone, Copy, PartialEq, Eq)]
1217pub enum GateStatus {
1218 /// Never attempted, or the last attempt was resource-blocked (the shared
1219 /// build cache could not be acquired or confirmed fresh in time) and
1220 /// needs a retry.
1221 NotRun,
1222 /// Ran with zero commands configured (`verify.gate` is empty) and
1223 /// therefore vacuously passed — there was nothing to check.
1224 PassedWithNoCommands,
1225 /// Ran one or more commands, and every one of them exited 0.
1226 Passed,
1227 /// Ran one or more commands, and at least one did not exit 0.
1228 Failed,
1229}
1230
1231impl GateStatus {
1232 /// May a run in this state proceed to merge?
1233 pub fn ok(self) -> bool {
1234 matches!(self, Self::PassedWithNoCommands | Self::Passed)
1235 }
1236}
1237
1238/// What happened to the winning branch.
1239#[derive(Debug, Clone, Serialize, Deserialize)]
1240pub struct MergeOutcome {
1241 /// Requested mode.
1242 pub mode: MergeMode,
1243 /// Did it land?
1244 pub ok: bool,
1245 /// Command output, or the command the operator should run.
1246 #[serde(default)]
1247 pub detail: String,
1248 /// The winner had no commits ahead of the base, so no pull request was
1249 /// attempted. Distinct from a `gh` failure: nothing was wrong with the
1250 /// tooling, there was simply nothing to land.
1251 #[serde(default)]
1252 pub empty: bool,
1253}
1254
1255/// A seat currently mid-answer: a prompt was sent and no reply has landed yet.
1256///
1257/// This is not the whole story of "is it alive" — a daemon killed mid-wave
1258/// leaves its last wave's entries here forever, since nothing ran to clear
1259/// them. A reader must cross-check a live daemon's heartbeat
1260/// (`daemon::is_working_on`) before trusting one of these as "still running"
1261/// rather than "abandoned". [`RunState::clear_active`] is what keeps that
1262/// leftover from surviving into the next attempt at this run: `execute` calls
1263/// it before doing anything else, so a resumed run never carries a stale
1264/// entry into its own report before the next wave repopulates it.
1265///
1266/// Deliberately carries no agent id: an implementer's agent is no secret, but
1267/// a judge or reviewer seat is blind (`SeatState::key` is keyed by seat, never
1268/// agent, for exactly this reason), and this struct has no way to tell which
1269/// kind of seat it describes. The seat key alone — already in the map this
1270/// lives under — is what every caller needs to say which seat is running.
1271///
1272/// The same map also carries entries for shell-command work that runs
1273/// outside any seat — `verify.e2e`, `verify.gate` — keyed by the task's own
1274/// name (`"e2e"`, `"gate"`) rather than a seat key. [`Self::task`] is `Some`
1275/// only for those; it is how a reader tells the two kinds of entry apart
1276/// without a second map, a second route, or a second SSE reason to poll for
1277/// — see [`RunState::seats_active`] / [`RunState::tasks_active`] for the
1278/// accessors that split them back apart. A task entry is exactly as blind as
1279/// a seat entry: no agent runs it, so there is nothing to leak, and
1280/// [`Self::command`] carries only the shell command being run, never
1281/// anything about who is running it.
1282#[derive(Debug, Clone, Serialize, Deserialize)]
1283pub struct ActiveSeat {
1284 /// Node the seat is answering for, e.g. `implement`, `judge`, `review`.
1285 /// For a task entry, the node the command list runs under (`verify`,
1286 /// `gate`).
1287 pub node: String,
1288 /// When this attempt — or, for a task entry, this one command — was
1289 /// started. A task entry's timer resets at every command boundary,
1290 /// because `verify.e2e` / `verify.gate` apply their timeout per command,
1291 /// not once across the whole list — see [`RunState::task_command`].
1292 pub started_at: Timestamp,
1293 /// The wall-clock budget for this attempt (a seat) or this one command
1294 /// (a task entry).
1295 pub timeout_secs: u64,
1296 /// 0 for the first ask, N for the Nth nudge or resume. For a task entry,
1297 /// 0 for the first pass over the command list, N for the Nth retry (see
1298 /// `run_e2e_with_retry`'s build/link retry).
1299 #[serde(default)]
1300 pub attempt: usize,
1301 /// `None` for a seat; `Some("e2e")` / `Some("gate")` for a running
1302 /// command-list task. This is the type tag that lets both kinds of entry
1303 /// share one map without a task ever being mistaken for a (blind) seat —
1304 /// see this struct's own doc. Omitted from JSON when absent (the common,
1305 /// seat case), rather than written out as a literal `null` on every one
1306 /// of a run's seat entries.
1307 #[serde(default, skip_serializing_if = "Option::is_none")]
1308 pub task: Option<String>,
1309 /// The command currently running, task entries only. Never set on a
1310 /// seat entry — a seat has no command, only a prompt, and a prompt is
1311 /// not safe to show mid-run (see this struct's blindness note).
1312 #[serde(default, skip_serializing_if = "Option::is_none")]
1313 pub command: Option<String>,
1314 /// 1-based position of [`Self::command`] within the task's command list.
1315 #[serde(default, skip_serializing_if = "Option::is_none")]
1316 pub index: Option<usize>,
1317 /// Number of commands in the task's list.
1318 #[serde(default, skip_serializing_if = "Option::is_none")]
1319 pub total: Option<usize>,
1320}
1321
1322impl ActiveSeat {
1323 /// Seconds since this attempt was sent.
1324 #[must_use]
1325 pub fn elapsed_secs(&self, now: Timestamp) -> i64 {
1326 (now.as_second() - self.started_at.as_second()).max(0)
1327 }
1328
1329 /// Seconds left before this attempt's own timeout fires, floored at zero
1330 /// rather than going negative once the CLI has overrun its budget.
1331 #[must_use]
1332 pub fn remaining_secs(&self, now: Timestamp) -> i64 {
1333 (self.timeout_secs as i64 - self.elapsed_secs(now)).max(0)
1334 }
1335}
1336
1337/// Whether a process is provably still driving a run, provably not, or
1338/// neither — see [`RunState::liveness`]. Serialized as a lowercase string
1339/// (`"live"` / `"dead"` / `"unknown"`) rather than a bool: a bool has no room
1340/// for "could not tell", and folding that case into either `true` or `false`
1341/// is exactly the wrong call for a display an operator uses to decide
1342/// whether to wait or to act — see the schema-10 field doc on
1343/// [`RunState::driver_pid`] for the report it used to produce instead.
1344#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
1345#[serde(rename_all = "lowercase")]
1346pub enum Liveness {
1347 /// Proven: a daemon's heartbeat claims the run, or `driver_pid` answers
1348 /// alive under the same identity (`driver_started_at`) this run recorded
1349 /// for it.
1350 Live,
1351 /// Proven: no daemon claim, and either `driver_pid` answers dead outright
1352 /// or it answers alive under a *different* identity than recorded — a
1353 /// pid the OS has since handed to an unrelated process is exactly as
1354 /// good as proof the original driver is gone (see
1355 /// [`RunState::driver_started_at`]'s own doc).
1356 Dead,
1357 /// Neither proven — no daemon claim, and either no `driver_pid` to ask,
1358 /// the platform could not answer for it, or a live pid with nothing (or
1359 /// nothing queryable) to corroborate its identity against. Never treated
1360 /// as `Dead`: see [`RunState::liveness_with`].
1361 Unknown,
1362}
1363
1364/// A timestamped note about a node.
1365#[derive(Debug, Clone, Serialize, Deserialize)]
1366pub struct Event {
1367 /// When.
1368 pub at: Timestamp,
1369 /// Node name.
1370 pub node: String,
1371 /// What happened.
1372 pub message: String,
1373}
1374
1375/// How far the winner's tree trailed the landing base, last time it was
1376/// checked, and what came of trying to close that gap.
1377///
1378/// Set by `graph::Runner::sync_to_base`, which runs before the review loop and
1379/// again before the gate: verifying against a tree that does not yet contain
1380/// the base's tip answers "green on the commit this run branched from", not
1381/// "green on what is about to land", and a merge on that answer can revert
1382/// whatever landed elsewhere while the run was thinking.
1383#[derive(Debug, Clone, Serialize, Deserialize)]
1384pub struct BaseSync {
1385 /// `<remote>/<base>` tip the tree was last checked against.
1386 pub tip: String,
1387 /// Commits `tip` was ahead of the tree at that check, before any rebase
1388 /// this round tried to close the gap. Zero means the tree already
1389 /// contained `tip`.
1390 pub behind: usize,
1391 /// Rebase attempts spent so far this run, bounded by
1392 /// `graph::BASE_SYNC_ROUNDS`.
1393 pub attempts: usize,
1394 /// What git said, if the most recent rebase attempt conflicted or could
1395 /// not be pushed. `Some` here is what makes a `Blocked` run read as
1396 /// "stopped on the base, not on review or the gate" - the rebase is not
1397 /// retried again while this is set; a person has to look.
1398 #[serde(default)]
1399 pub conflict: Option<String>,
1400}
1401
1402/// What the land loop saw last time it looked at the pull request.
1403///
1404/// Strings for `state` and `checks` on purpose: they are `gh`'s vocabulary, and
1405/// pinning them into an enum here would mean a new GitHub check conclusion
1406/// turns a readable status into a deserialisation error on a run someone is
1407/// trying to look at.
1408#[derive(Debug, Clone, Serialize, Deserialize)]
1409pub struct PrRecord {
1410 /// Pull request url.
1411 pub url: String,
1412 /// Pull request number.
1413 pub number: u64,
1414 /// `open`, `merged` or `closed`.
1415 pub state: String,
1416 /// `pending`, `green`, `red` or `unknown`.
1417 pub checks: String,
1418 /// Land round, 1-based, or 0 before the first fix.
1419 pub round: usize,
1420 /// Land round budget.
1421 pub rounds: usize,
1422 /// Checks that were red when the pull request merged anyway (the forge
1423 /// said they do not block it). Empty for a green merge, and for any run
1424 /// recorded before this field existed - so empty is not proof nothing was
1425 /// red.
1426 #[serde(default, skip_serializing_if = "Vec::is_empty")]
1427 pub red_at_merge: Vec<String>,
1428}
1429
1430/// What the post-merge release-bump step did, and whether it needs a human.
1431///
1432/// Recorded next to the `bump` events rather than instead of them: the events
1433/// are the narrative, this is what `magi show`, the status word and the
1434/// notification read without parsing prose. `status` stays `Merged` - the
1435/// merge did happen - so [`RunState::needs_attention`] is the one place that
1436/// says a merged run is not fully done.
1437#[derive(Debug, Clone, Default, PartialEq, Eq, Serialize, Deserialize)]
1438pub struct ReleaseBump {
1439 /// The release pull request, when one was opened.
1440 #[serde(default)]
1441 pub pr_url: Option<String>,
1442 /// The version it releases.
1443 #[serde(default)]
1444 pub version: Option<String>,
1445 /// Did enabling automerge on `pr_url` succeed?
1446 #[serde(default)]
1447 pub automerge_enabled: bool,
1448 /// GitHub refused automerge because CI had already finished (the pull
1449 /// request was in clean status), so magi merged it directly. Kept apart
1450 /// from `automerge_enabled`, which stays the plain fact it says it is.
1451 #[serde(default)]
1452 pub merged_directly: bool,
1453 /// What went wrong, verbatim from the tool that said it.
1454 #[serde(default)]
1455 pub problem: Option<String>,
1456 /// What the operator has to do about it. `Some` is what makes a merged run
1457 /// read as needing attention.
1458 #[serde(default)]
1459 pub action_required: Option<String>,
1460}
1461
1462/// The whole run.
1463#[derive(Debug, Clone, Serialize, Deserialize)]
1464pub struct RunState {
1465 /// On-disk format version.
1466 pub schema: u32,
1467 /// Run id, e.g. `20260830-153012-a1b2`.
1468 pub id: String,
1469 /// Repository the run operates on.
1470 pub repo: PathBuf,
1471 /// Branch the run started from.
1472 pub base_branch: String,
1473 /// Commit the run started from.
1474 pub base_commit: String,
1475 /// The task, verbatim.
1476 pub instruction: String,
1477 /// When the run was created.
1478 pub created_at: Timestamp,
1479 /// Last state flush.
1480 pub updated_at: Timestamp,
1481 /// Current status.
1482 pub status: RunStatus,
1483 /// Seed for labels and session ids.
1484 pub seed: u64,
1485 /// Config snapshot, so a resumed run behaves like the original.
1486 pub config: Config,
1487 /// Did this run take a reference on `extensions.worktreeConfig` being on
1488 /// (see [`crate::git::acquire_worktree_config`])? If so, cleanup releases
1489 /// it - which only actually turns the setting back off once every other
1490 /// run sharing this repository has released its own reference too.
1491 #[serde(default)]
1492 pub enabled_worktree_config: bool,
1493 /// Candidates.
1494 #[serde(default)]
1495 pub candidates: Vec<Candidate>,
1496 /// Initial blind rankings.
1497 #[serde(default)]
1498 pub judgements: Vec<Judgement>,
1499 /// `judge` decided a solo candidate needs no panel and only logged it.
1500 ///
1501 /// `judgements` stays empty in that case — nothing to distinguish from
1502 /// "not yet judged" — so this is the record that makes the skip
1503 /// idempotent: without it, every reentry re-ran `judge`, re-logged the
1504 /// same event, and rewrote `status` to `Judging` over whatever a later
1505 /// node had already concluded.
1506 #[serde(default)]
1507 pub judge_skipped: bool,
1508 /// Deliberation, if it happened.
1509 #[serde(default)]
1510 pub deliberation: Vec<DeliberationRound>,
1511 /// Private final votes.
1512 #[serde(default)]
1513 pub votes: Vec<VoteRecord>,
1514 /// The count.
1515 #[serde(default)]
1516 pub tally: Option<Tally>,
1517 /// Review rounds.
1518 #[serde(default)]
1519 pub reviews: Vec<ReviewRound>,
1520 /// Final gate.
1521 ///
1522 /// Never derive whether the gate has run from `gate.is_empty()` alone —
1523 /// use [`Self::gate_status`] instead. An empty list is ambiguous on its
1524 /// own: it is what an unattempted gate looks like, what a
1525 /// resource-blocked attempt leaves behind (see `graph::Runner::gate`'s
1526 /// own doc), and also what a repo with no `verify.gate` commands
1527 /// configured produces once it *has* run. [`Self::gate_ran`] is what
1528 /// tells the third case apart from the first two.
1529 #[serde(default)]
1530 pub gate: Vec<CommandOutcome>,
1531 /// Did `gate()` actually record an attempt — zero commands configured
1532 /// and vacuously passed, or one or more commands that ran to
1533 /// completion — as opposed to never having run, or having last hit a
1534 /// resource-blocked retry?
1535 ///
1536 /// `gate.is_empty()` cannot tell those apart by itself: a repo with no
1537 /// `verify.gate` commands leaves `gate` empty exactly like an
1538 /// unattempted or resource-blocked one does, and reading that empty list
1539 /// as "not yet run" is what stranded a review-only run in
1540 /// `RunStatus::Gating` forever on such a repo — see `SCHEMA`'s doc for
1541 /// schema 7. A record written before this field existed defaults to
1542 /// `false` and is migrated in [`migrate_schema`].
1543 #[serde(default)]
1544 pub gate_ran: bool,
1545 /// Fix rounds a failing gate has already spent, oldest first. Persisted
1546 /// right after each fixer call so a resumed run spends only what is left
1547 /// of `graph.gate_fix_rounds` instead of starting the budget over.
1548 #[serde(default)]
1549 pub gate_fixes: Vec<GateFixRecord>,
1550 /// Fixer rounds spent on a rebase conflict (base sync and land share one
1551 /// budget, `graph.review_rounds`), oldest first. Persisted right before
1552 /// each fixer call, so a resumed run spends only what is left.
1553 #[serde(default)]
1554 pub rebase_fixes: Vec<RebaseFixRecord>,
1555 /// Outcomes of the `verify.pre_gate` commands from the latest time they
1556 /// ran on the winner. Informational only: a failure here never blocks the
1557 /// run, the gate stays the single arbiter. Overwritten on each pass.
1558 #[serde(default)]
1559 pub pre_gate: Vec<CommandOutcome>,
1560 /// The latest commit a `pre_gate` pass made, if any pass left changes.
1561 #[serde(default)]
1562 pub pre_gate_commit: Option<String>,
1563 /// Merge outcome.
1564 #[serde(default)]
1565 pub merge: Option<MergeOutcome>,
1566 /// Vendor tokens seen in judged material.
1567 #[serde(default)]
1568 pub leaks: Vec<Leak>,
1569 /// Seats lost to a CLI rate limit / quota, in the order they hit.
1570 #[serde(default)]
1571 pub quota: Vec<QuotaLoss>,
1572 /// Stray foreign lockfiles a rescue commit left out, one entry per path.
1573 #[serde(default)]
1574 pub withheld: Vec<Withheld>,
1575 /// Parked at a node boundary, waiting to be resumed.
1576 ///
1577 /// A run that is neither finished nor being worked on is otherwise
1578 /// indistinguishable from one whose daemon was killed, and the two want
1579 /// opposite things from an operator: the first is expected to be resumed,
1580 /// the second is a leftover. Cleared by the resume that carries it on.
1581 #[serde(default)]
1582 pub parked: bool,
1583 /// The run that took this run's worktree over, when one did — see
1584 /// [`crate::handover`]. `Some` means the worktree is gone on purpose and
1585 /// the run can no longer be resumed from it ([`Self::released`]).
1586 #[serde(default)]
1587 pub released_to: Option<String>,
1588 /// Branches that outlived this run's released worktree and now belong to
1589 /// the run in [`Self::released_to`] (and to its pull request). A later
1590 /// fold must leave them alone.
1591 #[serde(default)]
1592 pub released_branches: Vec<String>,
1593 /// Existing branches and commits the task text names, and what the
1594 /// repository says about each (`crate::refs`). Unmerged ones seed the
1595 /// candidates; `base_commit` stays the comparison point.
1596 #[serde(default)]
1597 pub seeds: Vec<crate::refs::Seed>,
1598 /// Per-seat conversation state.
1599 #[serde(default)]
1600 pub seats: BTreeMap<String, SeatState>,
1601 /// Seats currently mid-answer, keyed by seat.
1602 ///
1603 /// An entry exists from the moment a prompt is sent until a reply (of any
1604 /// kind — success, failure, quota, drop) comes back, so its keys are
1605 /// exactly "who hasn't answered yet" for whichever node populated it. See
1606 /// [`ActiveSeat`] for why a reader still has to check a live daemon
1607 /// before trusting one of these as "running" rather than "abandoned".
1608 #[serde(default)]
1609 pub active: BTreeMap<String, ActiveSeat>,
1610 /// Process id of whichever `execute()` call last drove this run —
1611 /// written at the very top of that method, the same place
1612 /// [`Self::clear_active`] runs, so a fresh reentry always overwrites the
1613 /// pid a previous, possibly-dead process left behind.
1614 ///
1615 /// A daemon-claimed run already has a stronger signal
1616 /// (`daemon::is_working_on`), but a `magi run` / `magi review` typed
1617 /// straight into a terminal claims nothing there — before this field
1618 /// existed, [`report::active_seats`] had no way to tell that run apart
1619 /// from one a killed process abandoned, and printed the same "no live
1620 /// daemon claims this run" warning over a run that was, in fact, still
1621 /// answering. See [`Liveness`] for how this and the daemon claim combine.
1622 #[serde(default)]
1623 pub driver_pid: Option<u32>,
1624 /// The OS-reported moment [`Self::driver_pid`] started, recorded in the
1625 /// same breath as the pid itself — an opaque marker
1626 /// (`crate::proc::process_started_at`), compared only for equality.
1627 ///
1628 /// A pid alone never proves a live process is *this run's* driver: pids
1629 /// get reused, sometimes within minutes on a busy machine, and a killed
1630 /// manual `magi run` whose pid a later, wholly unrelated process happens
1631 /// to receive would otherwise read back as `Liveness::Live` from that
1632 /// coincidence alone. [`Self::liveness`] re-queries the current holder
1633 /// of `driver_pid` and requires this marker to still match before
1634 /// trusting a live answer — a mismatch means a different process now
1635 /// answers to that number, and no marker to compare (an old run, or a
1636 /// platform this build could not ask at record time) means neither
1637 /// extreme can be proven.
1638 #[serde(default)]
1639 pub driver_started_at: Option<String>,
1640 /// The process named by [`Self::driver_pid`] has stopped driving this run:
1641 /// its graph walk returned, whatever the outcome. Cleared at the start of
1642 /// every walk, set as the last write of one.
1643 ///
1644 /// A long-lived `magi serve` records its own pid and keeps it after the
1645 /// run ends, so that pid answers alive with a matching identity for as
1646 /// long as the daemon lives, and every blocked, failed or parked run it
1647 /// ever drove would read as being worked on right now. A crash or kill
1648 /// never sets this, and needs nothing to: the pid is dead then.
1649 #[serde(default)]
1650 pub driver_exited: bool,
1651 /// Last observation of the winner's pull request, when a land loop ran.
1652 ///
1653 /// Persisted rather than derived from the event log because the phone asks
1654 /// two questions about a run that has opened a PR - how are its checks and
1655 /// which round is it on - and parsing prose out of events to answer them
1656 /// would break the first time an event message was reworded.
1657 #[serde(default)]
1658 pub pr: Option<PrRecord>,
1659 /// The post-merge release-bump step's outcome, when it got as far as
1660 /// having one to report. See [`ReleaseBump`].
1661 #[serde(default)]
1662 pub release_bump: Option<ReleaseBump>,
1663 /// The last look at how far the winner's tree trailed the landing base,
1664 /// and the rebase(s) tried to close that gap. `None` until the tree has a
1665 /// winner to check.
1666 #[serde(default)]
1667 pub base_sync: Option<BaseSync>,
1668 /// The design-deliberation stage's output, when `[graph] advise` ran it:
1669 /// one record per advisor seat, plus the synthesis blended into the
1670 /// implementer's prompt. `None` when the stage is off, has not run yet,
1671 /// or could not even resolve its seats - see
1672 /// [`crate::graph::Runner::advise`].
1673 #[serde(default)]
1674 pub advice: Option<crate::advise::Advice>,
1675 /// Whether the design-deliberation stage has already been attempted this
1676 /// run, whatever it produced. The idempotency marker `Runner::advise`
1677 /// checks on reentry, the same role [`Self::judge_skipped`] plays for
1678 /// `judge` - without it a resumed run whose stage failed (a misconfigured
1679 /// `[roles] advisors`, every seat quota'd) would re-run it, and re-spend
1680 /// the agent calls, on every single reentry before `implement`.
1681 #[serde(default)]
1682 pub advise_attempted: bool,
1683 /// Node log.
1684 #[serde(default)]
1685 pub events: Vec<Event>,
1686 /// Commands seats' own CLIs reported running, across every node — see
1687 /// [`JobRecord`]. Populated in [`crate::graph::wave`] as each seat
1688 /// answers, so a resumed run keeps what earlier waves already collected
1689 /// rather than losing it to a reentry. Empty on a record written before
1690 /// this existed, or wherever no adapter reads structured job events for
1691 /// the backend a seat used — both read as "no evidence", not "nothing
1692 /// ran".
1693 #[serde(default)]
1694 pub jobs: Vec<JobRecord>,
1695 /// Operator-triggered targeted fixes — see [`OperatorFixRequest`] and
1696 /// `SCHEMA`'s doc for schema 9. Empty on every record written before
1697 /// this existed, which reads correctly as "no operator fix ever
1698 /// requested".
1699 #[serde(default)]
1700 pub operator_fixes: Vec<OperatorFixRequest>,
1701 /// Absolute paths of the files the task was filed with (see
1702 /// `crate::queue::Task::attachments`), listed in the implementers' prompt
1703 /// and made readable to their seats. Refreshed from the task on every
1704 /// start and resume, so a file attached while the task was held reaches a
1705 /// resumed run. Additive and `#[serde(default)]`, so `SCHEMA` stays put.
1706 #[serde(default)]
1707 pub attachments: Vec<PathBuf>,
1708}
1709
1710impl RunState {
1711 /// A fresh run.
1712 pub fn new(
1713 repo: PathBuf,
1714 base_branch: String,
1715 base_commit: String,
1716 instruction: String,
1717 config: Config,
1718 ) -> Self {
1719 let now = Timestamp::now();
1720 let seed = config.blind.seed.unwrap_or_else(crate::rng::entropy);
1721 Self {
1722 schema: SCHEMA,
1723 id: new_id(),
1724 repo,
1725 base_branch,
1726 base_commit,
1727 instruction,
1728 created_at: now,
1729 updated_at: now,
1730 status: RunStatus::Prep,
1731 seed,
1732 config,
1733 enabled_worktree_config: false,
1734 candidates: Vec::new(),
1735 judgements: Vec::new(),
1736 judge_skipped: false,
1737 deliberation: Vec::new(),
1738 votes: Vec::new(),
1739 tally: None,
1740 reviews: Vec::new(),
1741 gate: Vec::new(),
1742 gate_ran: false,
1743 gate_fixes: Vec::new(),
1744 rebase_fixes: Vec::new(),
1745 pre_gate: Vec::new(),
1746 pre_gate_commit: None,
1747 merge: None,
1748 leaks: Vec::new(),
1749 quota: Vec::new(),
1750 withheld: Vec::new(),
1751 parked: false,
1752 released_to: None,
1753 released_branches: Vec::new(),
1754 seeds: Vec::new(),
1755 seats: BTreeMap::new(),
1756 active: BTreeMap::new(),
1757 driver_pid: None,
1758 driver_started_at: None,
1759 driver_exited: false,
1760 pr: None,
1761 release_bump: None,
1762 base_sync: None,
1763 advice: None,
1764 advise_attempted: false,
1765 attachments: Vec::new(),
1766 events: Vec::new(),
1767 jobs: Vec::new(),
1768 operator_fixes: Vec::new(),
1769 }
1770 }
1771
1772 /// Directory holding this run's state and artifacts.
1773 pub fn dir(&self) -> PathBuf {
1774 run_dir(&self.id)
1775 }
1776
1777 /// Short form used in branch names and reports.
1778 pub fn short(&self) -> &str {
1779 short_of(&self.id)
1780 }
1781
1782 /// Branch name for a label.
1783 pub fn branch_for(&self, label: char) -> String {
1784 format!("magi/{}/{}", self.short(), label)
1785 }
1786
1787 /// Root of this run's worktrees.
1788 pub fn worktree_root(&self) -> PathBuf {
1789 self.config
1790 .graph
1791 .worktree_root
1792 .clone()
1793 .unwrap_or_else(default_worktree_root)
1794 .join(self.short())
1795 }
1796
1797 /// Record lockfiles a rescue commit withheld; a path already recorded by an
1798 /// earlier round is not repeated.
1799 pub fn note_withheld(&mut self, node: &str, strays: &[crate::git::Stray]) {
1800 for s in strays {
1801 if self.withheld.iter().any(|w| w.path == s.path) {
1802 continue;
1803 }
1804 self.event(
1805 node,
1806 format!(
1807 "withheld stray lockfile {} ({}; the directory uses {})",
1808 s.path, s.manager, s.kept_by
1809 ),
1810 );
1811 self.withheld.push(Withheld {
1812 path: s.path.clone(),
1813 manager: s.manager.clone(),
1814 kept_by: s.kept_by.clone(),
1815 node: node.to_owned(),
1816 at: Timestamp::now(),
1817 });
1818 }
1819 }
1820
1821 /// Note something in the run log and on the tracing stream.
1822 pub fn event(&mut self, node: &str, message: impl Into<String>) {
1823 let message = message.into();
1824 tracing::info!(node, "{message}");
1825 self.events.push(Event {
1826 at: Timestamp::now(),
1827 node: node.to_owned(),
1828 message,
1829 });
1830 }
1831
1832 /// The honest state of the final gate.
1833 ///
1834 /// Never derive this from `gate.is_empty()` alone anywhere else in the
1835 /// codebase — `NotRun` and `PassedWithNoCommands` both leave `gate`
1836 /// empty, and only this method (backed by [`Self::gate_ran`]) tells them
1837 /// apart. See `SCHEMA`'s doc for schema 7 for what conflating them used
1838 /// to do.
1839 pub fn gate_status(&self) -> GateStatus {
1840 if !self.gate_ran {
1841 GateStatus::NotRun
1842 } else if self.gate.is_empty() {
1843 GateStatus::PassedWithNoCommands
1844 } else if self.gate.iter().all(CommandOutcome::ok) {
1845 GateStatus::Passed
1846 } else {
1847 GateStatus::Failed
1848 }
1849 }
1850
1851 /// Record that `seat` was just sent a prompt for `node`, with the given
1852 /// wall-clock budget. `attempt` is 0 for the first ask and N for the Nth
1853 /// nudge or resume, purely for display — it does not change how the seat
1854 /// is treated.
1855 pub fn seat_started(
1856 &mut self,
1857 node: &str,
1858 seat: &str,
1859 timeout: std::time::Duration,
1860 attempt: usize,
1861 ) {
1862 self.active.insert(
1863 seat.to_owned(),
1864 ActiveSeat {
1865 node: node.to_owned(),
1866 started_at: Timestamp::now(),
1867 timeout_secs: timeout.as_secs(),
1868 attempt,
1869 task: None,
1870 command: None,
1871 index: None,
1872 total: None,
1873 },
1874 );
1875 }
1876
1877 /// Record that `seat` has answered, whatever the answer was.
1878 pub fn seat_finished(&mut self, seat: &str) {
1879 self.active.remove(seat);
1880 }
1881
1882 /// Record that `task` (`"e2e"` or `"gate"` — a shell-command list run
1883 /// outside any seat) has just started `command`, the `index`-th of
1884 /// `total`. Called at every command boundary, not once for the whole
1885 /// list: `verify.e2e` / `verify.gate` apply `timeout` per command, so
1886 /// this is the only way a reader can tell "how long is left" for
1887 /// whichever command is actually running right now, rather than a stale
1888 /// budget left over from the first one.
1889 #[allow(clippy::too_many_arguments)]
1890 pub fn task_command(
1891 &mut self,
1892 task: &str,
1893 node: &str,
1894 attempt: usize,
1895 command: &str,
1896 index: usize,
1897 total: usize,
1898 timeout: std::time::Duration,
1899 ) {
1900 self.active.insert(
1901 task.to_owned(),
1902 ActiveSeat {
1903 node: node.to_owned(),
1904 started_at: Timestamp::now(),
1905 timeout_secs: timeout.as_secs(),
1906 attempt,
1907 task: Some(task.to_owned()),
1908 command: Some(command.to_owned()),
1909 index: Some(index),
1910 total: Some(total),
1911 },
1912 );
1913 }
1914
1915 /// Record that `task` has finished its whole command list for this
1916 /// attempt.
1917 pub fn task_finished(&mut self, task: &str) {
1918 self.active.remove(task);
1919 }
1920
1921 /// The seats — never task entries — currently mid-answer. What
1922 /// `report::active_seats` and the phone's "who has not answered yet"
1923 /// note need: a seat's identifier is safe to show ([`ActiveSeat`]'s doc),
1924 /// so nothing here filters anything out beyond the type tag itself.
1925 pub fn seats_active(&self) -> impl Iterator<Item = (&String, &ActiveSeat)> {
1926 self.active.iter().filter(|(_, a)| a.task.is_none())
1927 }
1928
1929 /// The command-list tasks — never seat entries — currently running.
1930 /// Counterpart to [`Self::seats_active`]; see [`ActiveSeat::task`] for
1931 /// the tag both read.
1932 pub fn tasks_active(&self) -> impl Iterator<Item = (&String, &ActiveSeat)> {
1933 self.active.iter().filter(|(_, a)| a.task.is_some())
1934 }
1935
1936 /// Drop every seat this state still lists as answering, reporting whether
1937 /// anything was dropped.
1938 ///
1939 /// Called first thing in `execute`, on every entry — fresh, resumed, or
1940 /// recovering a stall — because an entry here only means something while
1941 /// the process that wrote it is still asking that seat something. A
1942 /// process killed mid-wave leaves its last batch of seats here with
1943 /// nobody left to clear them, and the next process to touch this run must
1944 /// not let that leftover read as "still going" before it has asked
1945 /// anyone anything.
1946 pub fn clear_active(&mut self) -> bool {
1947 if self.active.is_empty() {
1948 return false;
1949 }
1950 self.active.clear();
1951 true
1952 }
1953
1954 /// Does every seat this run still lists as [`Self::active`] sit past its
1955 /// own [`ActiveSeat::timeout_secs`]? `false` when nothing is active at
1956 /// all — an empty map is not evidence of anything overrunning.
1957 ///
1958 /// This alone is not proof the run is dead: a seat's own attempt can
1959 /// legitimately run a little past its budget while the process driving it
1960 /// is still tearing the attempt down. Every caller pairs this with its own
1961 /// `!live` reading (`daemon::is_working_on`) before treating the run as
1962 /// abandoned — this module cannot check that itself without depending on
1963 /// `crate::daemon`, and callers already have to ask that question anyway.
1964 #[must_use]
1965 pub fn active_all_overrun(&self, now: Timestamp) -> bool {
1966 !self.active.is_empty()
1967 && self
1968 .active
1969 .values()
1970 .all(|a| a.elapsed_secs(now) > a.timeout_secs as i64)
1971 }
1972
1973 /// Whether a process is actually still driving this run, given whether a
1974 /// daemon's heartbeat claims it and process-liveness/identity queries for
1975 /// [`Self::driver_pid`].
1976 ///
1977 /// A daemon claim wins outright when present — it is the stronger,
1978 /// independently-heartbeating signal. Absent that (every manual `magi
1979 /// run` / `magi review`, and every daemon-driven run whose daemon has
1980 /// since exited cleanly), `driver_pid` is asked directly. A live answer
1981 /// alone is not enough to trust, though: pids get reused, so `identity`
1982 /// re-queries whoever currently holds that pid and the result must still
1983 /// match [`Self::driver_started_at`] — the marker recorded at the same
1984 /// moment `driver_pid` was — before this reads `Live`. A mismatch means
1985 /// a *different* process now answers to that number, which is exactly as
1986 /// good as proof the original driver is gone, so that reads `Dead`; no
1987 /// marker to compare against (an old run, or a platform this build could
1988 /// not ask at record time) or a `None` from either query, and this
1989 /// cannot tell either way, so it reads [`Liveness::Unknown`] — never
1990 /// guessed as [`Liveness::Dead`] out of mere silence. A display that
1991 /// guessed "dead" out of missing information would be exactly the
1992 /// mtime-and-task-manager guessing this type exists to replace.
1993 ///
1994 /// Kept generic over `query` and `identity` so a test can inject answers
1995 /// without spawning a real process query — production code goes through
1996 /// [`Self::liveness`], which supplies [`crate::proc::pid_status`] and
1997 /// [`crate::proc::process_started_at`].
1998 #[must_use]
1999 pub fn liveness_with<F, G>(&self, daemon_claims: bool, query: F, identity: G) -> Liveness
2000 where
2001 F: FnOnce(u32) -> Option<bool>,
2002 G: FnOnce(u32) -> Option<String>,
2003 {
2004 if daemon_claims {
2005 return Liveness::Live;
2006 }
2007 // The driver said itself that it stopped; its pid may well be alive
2008 // (a daemon outlives the runs it drives), so it is not asked.
2009 if self.driver_exited {
2010 return Liveness::Dead;
2011 }
2012 let Some(pid) = self.driver_pid else {
2013 return Liveness::Unknown;
2014 };
2015 match query(pid) {
2016 Some(false) => Liveness::Dead,
2017 None => Liveness::Unknown,
2018 Some(true) => match (&self.driver_started_at, identity(pid)) {
2019 (Some(recorded), Some(current)) if *recorded == current => Liveness::Live,
2020 (Some(_), Some(_)) => Liveness::Dead,
2021 _ => Liveness::Unknown,
2022 },
2023 }
2024 }
2025
2026 /// [`Self::liveness_with`], backed by the real process-liveness and
2027 /// identity queries.
2028 #[must_use]
2029 pub fn liveness(&self, daemon_claims: bool) -> Liveness {
2030 self.liveness_with(
2031 daemon_claims,
2032 crate::proc::pid_status,
2033 crate::proc::process_started_at,
2034 )
2035 }
2036
2037 /// Clear every seat this run still lists as active and fail it, unless it
2038 /// had already reached a terminal status some other way.
2039 ///
2040 /// Callers must already have proven this run is dead — [`Self::active_all_overrun`]
2041 /// plus their own `!live` reading — before calling this; it does not
2042 /// check either itself. Unlike [`Self::clear_active`] (dropping a resumed
2043 /// run's own stale wave before repopulating it, called unconditionally at
2044 /// the top of every `execute()`), this is a verdict: a run left this way
2045 /// has nothing left to repopulate the wave, ever, and must stop reading as
2046 /// `implementing` (or whichever node) forever.
2047 pub fn abandon(&mut self, by: &str) {
2048 let seats: Vec<String> = self.active.keys().cloned().collect();
2049 self.clear_active();
2050 if !self.status.done() {
2051 self.status = RunStatus::Failed;
2052 }
2053 self.event(
2054 by,
2055 format!(
2056 "abandoned: seat(s) {} left behind by a killed process, past their own \
2057 timeout with no live daemon claiming this run",
2058 seats.join(", ")
2059 ),
2060 );
2061 }
2062
2063 /// Flush to `run.json`, atomically, under the process-global [`home`].
2064 pub fn save(&mut self) -> Result<()> {
2065 let home = home();
2066 self.save_under(&home)
2067 }
2068
2069 /// [`Self::save`], rooted at an explicit `home` instead of the
2070 /// process-global one.
2071 ///
2072 /// For a caller that was already handed its own `home` explicitly — a
2073 /// housekeeping pass, mainly, for the same reason `Queue::at` and the
2074 /// daemon status path are parameters rather than resolved here (see
2075 /// `daemon::drive`'s own doc) — falling through to the global would write
2076 /// back through whichever directory some *other* process or test pinned
2077 /// into that `OnceLock` first, not the one this call was actually handed.
2078 pub fn save_under(&mut self, home: &Path) -> Result<()> {
2079 self.updated_at = Timestamp::now();
2080 let dir = home.join("runs").join(&self.id);
2081 std::fs::create_dir_all(&dir).with_context(|| format!("create {}", dir.display()))?;
2082 let body = serde_json::to_string_pretty(self).context("serialize run state")?;
2083 let tmp = dir.join("run.json.tmp");
2084 std::fs::write(&tmp, &body).with_context(|| format!("write {}", tmp.display()))?;
2085 std::fs::rename(&tmp, dir.join("run.json")).with_context(|| "replace run.json")?;
2086 Ok(())
2087 }
2088
2089 /// Load a run by id or unambiguous id prefix.
2090 pub fn load(id: &str) -> Result<Self> {
2091 let resolved = resolve_id(id)?;
2092 let path = run_dir(&resolved).join("run.json");
2093 let body =
2094 std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
2095 let state: Self =
2096 serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
2097 migrate_schema(state)
2098 }
2099
2100 /// [`Self::load`], rooted at an explicit `home` instead of the
2101 /// process-global one, and for a full id rather than a prefix — a caller
2102 /// with its own `home` already has the exact id (from a `RunState` it
2103 /// already read, or from `Task::runs`) and has no `runs_root` to search
2104 /// for a prefix against in the first place. Same reasoning as
2105 /// [`Self::save_under`]: a caller holding its own `home` explicitly must
2106 /// not read back through whichever directory some *other* process or
2107 /// test pinned into the global [`home`] `OnceLock` first.
2108 pub fn load_under(id: &str, home: &Path) -> Result<Self> {
2109 let path = home.join("runs").join(id).join("run.json");
2110 let body =
2111 std::fs::read_to_string(&path).with_context(|| format!("read {}", path.display()))?;
2112 let state: Self =
2113 serde_json::from_str(&body).with_context(|| format!("parse {}", path.display()))?;
2114 migrate_schema(state)
2115 }
2116}
2117
2118fn migrate_schema(mut state: RunState) -> Result<RunState> {
2119 // Schema 5 predates deferred e2e. Its empty e2e lists therefore mean
2120 // "not configured", never "deferred"; serde's field defaults retain
2121 // exactly that representation while this migration permits resumes.
2122 if state.schema == 5 {
2123 state.schema = 6;
2124 }
2125 // Schema 6 predates `gate_ran` and could not tell "never attempted or
2126 // resource-blocked" apart from "ran with zero commands configured" — see
2127 // `SCHEMA`'s doc for schema 7. A non-empty `gate` is a real recorded
2128 // attempt either way, so it is trusted as `gate_ran = true` rather than
2129 // spent again; an empty one is simply handed back to the next `gate()`
2130 // call, which re-attempts it and, for the zero-commands case, resolves
2131 // instantly.
2132 if state.schema == 6 {
2133 state.gate_ran = !state.gate.is_empty();
2134 state.schema = 7;
2135 }
2136 // Schema 7's main review loop always checked `e2e` against the round's
2137 // own `head`, it just never wrote that fact into `verified_head` unless
2138 // a catch-up run had checked a *different* commit — see `SCHEMA`'s doc
2139 // for schema 8. Reconstructing `Some(head)` for a round whose `e2e` held
2140 // a real attempt restores a fact that was always true; `verified_at` has
2141 // no historical value to recover and stays `None`.
2142 if state.schema == 7 {
2143 for round in &mut state.reviews {
2144 if round.verified_head.is_none()
2145 && matches!(round.e2e_status(), E2eStatus::Passed | E2eStatus::Failed)
2146 {
2147 round.verified_head = Some(round.head.clone());
2148 }
2149 }
2150 state.schema = 8;
2151 }
2152 // Schema 8 predates `operator_fixes`. There is nothing to reconstruct —
2153 // an old run simply never had one requested — so `#[serde(default)]`
2154 // already left it as the correct empty `Vec`; this only advances the
2155 // version number.
2156 if state.schema == 8 {
2157 state.schema = 9;
2158 }
2159 // Schema 9 predates `RunStatus::VerifiedNoop` and
2160 // `Candidate::verified_noop`. Nothing to reconstruct: an old record never
2161 // made the claim, `#[serde(default)]` already reads `verified_noop` as
2162 // `None` on every candidate, and a `VerifiedNoop` status cannot appear in
2163 // a schema-9 record at all — see `SCHEMA`'s doc for schema 10. This only
2164 // advances the version number.
2165 if state.schema == 9 {
2166 state.schema = 10;
2167 }
2168 // Schema 10 predates `released_to`: no old run ever had its worktree
2169 // handed over, and `#[serde(default)]` reads exactly that. Only the
2170 // version number advances.
2171 if state.schema == 10 {
2172 state.schema = 11;
2173 }
2174 // Schema 11 predates `driver_exited`. A run that had already ended
2175 // Blocked / Failed / Stalled, or was parked in Landing (which hands its
2176 // daemon slot back), stopped walking the graph, but its recorded
2177 // pid may belong to a daemon still alive doing other work, which would pin
2178 // it as running for good; those are marked exited. Any other status keeps
2179 // `false`, and its pid is asked as before. This is an inference from
2180 // status, sound because a schema-11 record was written by a build that is
2181 // gone once this one is installed (and a daemon claim, which wins
2182 // regardless, still protects a run a live daemon is driving).
2183 if state.schema == 11 {
2184 if matches!(
2185 state.status,
2186 RunStatus::Blocked | RunStatus::Failed | RunStatus::Stalled | RunStatus::Landing
2187 ) {
2188 state.driver_exited = true;
2189 }
2190 state.schema = SCHEMA;
2191 }
2192 if state.schema != SCHEMA {
2193 bail!(
2194 "run {} was written by a different magi (schema {}, this build \
2195 speaks {SCHEMA})",
2196 state.id,
2197 state.schema
2198 );
2199 }
2200 Ok(state)
2201}
2202
2203impl RunState {
2204 /// Whether a later attempt took this run's worktree over, so a resume has
2205 /// nothing left to continue from.
2206 pub fn released(&self) -> bool {
2207 self.released_to.is_some()
2208 }
2209
2210 /// The winning candidate, once the tally has run.
2211 pub fn winner(&self) -> Option<&Candidate> {
2212 let label = self.tally.as_ref()?.winner;
2213 self.candidates.iter().find(|c| c.label == label)
2214 }
2215
2216 /// Candidates eligible for judging.
2217 pub fn viable(&self) -> Vec<&Candidate> {
2218 self.candidates.iter().filter(|c| c.viable()).collect()
2219 }
2220
2221 /// Did every candidate write nothing, and every one of them back it with
2222 /// evidence [`crate::graph`]'s adoption guard accepted?
2223 ///
2224 /// All-or-nothing on purpose: one candidate declaring `NO CHANGE NEEDED`
2225 /// while another simply failed to produce anything is not agreement, it
2226 /// is one candidate's unverified claim next to an ordinary loss, and the
2227 /// run must still read as the `Failed` it is. Only ever meaningful when
2228 /// [`Self::viable`] is already empty — a run with any real patch to judge
2229 /// never reaches the caller that asks this.
2230 pub fn all_candidates_verified_noop(&self) -> bool {
2231 !self.candidates.is_empty()
2232 && self
2233 .candidates
2234 .iter()
2235 .all(|c| c.empty && c.verified_noop.is_some())
2236 }
2237
2238 /// Findings still open when the review loop stopped trying: the last
2239 /// round's, exactly when that round was not clean. Empty on a run that
2240 /// never reviewed, or whose last round was clean.
2241 ///
2242 /// This is the last round's findings regardless of what the fixer claims
2243 /// to have addressed in that same round: a round that stopped the loop
2244 /// (round budget spent, or no tree progress for
2245 /// [`crate::graph::STAGNANT_LIMIT`] rounds) never had a *following* round
2246 /// to confirm the fix actually landed, and the self-reported adoption
2247 /// count is not trusted for that judgement either — see
2248 /// [`ReviewRound::progressed`].
2249 pub fn open_findings(&self) -> Vec<&Finding> {
2250 match self.reviews.last() {
2251 Some(r) if !r.clean => r
2252 .reviews
2253 .iter()
2254 .flat_map(|rec| rec.findings.iter())
2255 .collect(),
2256 _ => Vec::new(),
2257 }
2258 }
2259
2260 /// Every finding raised in the most recent review round, regardless of
2261 /// that round's own severity mix — unlike [`Self::open_findings`], not
2262 /// filtered to a round that was not clean. This is the pool `magi fix`
2263 /// reports as available to pick from: a round can conclude clean (no
2264 /// finding blocked merge) while still carrying minor findings nobody
2265 /// has acted on.
2266 pub fn last_round_findings(&self) -> Vec<&Finding> {
2267 self.reviews
2268 .last()
2269 .into_iter()
2270 .flat_map(|r| r.reviews.iter())
2271 .flat_map(|rec| rec.findings.iter())
2272 .collect()
2273 }
2274
2275 /// Look up a finding by id anywhere in this run's review history,
2276 /// together with the round and reviewer record that raised it — the
2277 /// provenance `magi fix` snapshots onto [`OperatorFixFinding`].
2278 pub fn finding(&self, id: &str) -> Option<(&ReviewRound, &ReviewRecord, &Finding)> {
2279 self.reviews.iter().find_map(|round| {
2280 round.reviews.iter().find_map(|rec| {
2281 rec.findings
2282 .iter()
2283 .find(|f| f.id == id)
2284 .map(|f| (round, rec, f))
2285 })
2286 })
2287 }
2288
2289 /// Did this run reach a mergeable status (`Ready` or `Merged`) with
2290 /// review findings still open?
2291 ///
2292 /// That combination is the point of the review hand-off: the review
2293 /// round budget (or an unproductive round, see [`ReviewRound::progressed`])
2294 /// was spent while gate and e2e stayed green, so the run was handed off
2295 /// rather than blocked — but the findings did not disappear, and whoever
2296 /// reads the result should be told they are still there.
2297 pub fn handed_off_with_open_findings(&self) -> bool {
2298 matches!(self.status, RunStatus::Ready | RunStatus::Merged)
2299 && self.reviews.last().is_some_and(|r| !r.clean)
2300 }
2301
2302 /// Reached `Ready` because `[merge] mode = "none"` left it there by
2303 /// design, never to be picked up by the PR-polling merge watcher — as
2304 /// opposed to a `Ready` that is still a plausible landing candidate (a
2305 /// PR closed without merging, or a re-entry onto an already-concluded
2306 /// node). Both leave `status` at `Ready`; only this one leaves the
2307 /// winning branch permanently unwatched, which is what a caller needs to
2308 /// know before labelling the run in a listing.
2309 pub fn unmerged_by_design(&self) -> bool {
2310 self.status == RunStatus::Ready
2311 && self
2312 .merge
2313 .as_ref()
2314 .is_some_and(|m| m.mode == MergeMode::None)
2315 }
2316
2317 /// Did a step after the merge leave something only a human can finish?
2318 ///
2319 /// Derived, not a status: the merge happened, so `status` stays `Merged`
2320 /// and everything that branches on it keeps working. Display code asks
2321 /// this so a merged run with an unmerged release PR does not read as
2322 /// plain green.
2323 pub fn needs_attention(&self) -> bool {
2324 self.status == RunStatus::Merged
2325 && self
2326 .release_bump
2327 .as_ref()
2328 .is_some_and(|b| b.action_required.is_some())
2329 }
2330
2331 /// Local-time creation stamp for reports.
2332 pub fn created_local(&self) -> String {
2333 self.created_at
2334 .to_zoned(jiff::tz::TimeZone::system())
2335 .strftime("%Y-%m-%d %H:%M:%S")
2336 .to_string()
2337 }
2338
2339 /// Assert that this run is safe to delete.
2340 ///
2341 /// Refuses a run a live daemon is working on, and refuses any run whose
2342 /// candidate worktrees and branches have not been folded away with `magi
2343 /// fold`. The fold requirement is the real protection: it is what makes
2344 /// "delete" mean "remove a record" rather than "throw away a worktree
2345 /// somebody may still be editing".
2346 ///
2347 /// `in_flight` has to come from the caller, because a run's own status
2348 /// cannot answer the question. A daemon killed mid-run leaves its status at
2349 /// `implementing` forever, and a guard that trusted that would make every
2350 /// interrupted run permanently undeletable - the operator's only recourse
2351 /// being to edit `run.json` by hand, which is exactly the sort of thing
2352 /// this command exists to avoid. The queue already treats an orphaned
2353 /// `.lock` from a `SIGKILL`ed daemon the same way; this is that rule for
2354 /// runs.
2355 pub fn ensure_can_delete(&self, in_flight: bool) -> Result<()> {
2356 if in_flight {
2357 bail!(
2358 "run {} is being worked on by a live daemon right now",
2359 self.short()
2360 );
2361 }
2362 if self.candidates.iter().any(|c| !c.folded) {
2363 bail!(
2364 "run {} has unfolded candidates; fold first with `magi fold`",
2365 self.short()
2366 );
2367 }
2368 Ok(())
2369 }
2370}
2371
2372/// The short form of a commit, for a label a human or an LLM reads.
2373fn short(commit: &str) -> String {
2374 commit.chars().take(7).collect()
2375}
2376
2377/// The short form of a run id: the trailing block after the last `-`.
2378///
2379/// A free function as well as [`RunState::short`], because callers that have
2380/// only an id - an error message, a daemon status, a route handler - were
2381/// otherwise reimplementing the split, and two spellings of "short id" is one
2382/// rename away from branch names that no longer match their run.
2383pub fn short_of(id: &str) -> &str {
2384 id.split('-').next_back().unwrap_or(id)
2385}
2386
2387/// Where magi keeps its runs.
2388///
2389/// `MAGI_HOME` overrides the default, and [`set_home`] overrides both — which
2390/// is what lets the integration tests drive a whole graph without writing into
2391/// the operator's real history.
2392///
2393/// In a unit test build (`cfg(test)`), falling through to the real
2394/// `<data_local>/magi` is not a fallback worth having: it is exactly how
2395/// three broken fixture runs ended up in the operator's actual history and
2396/// were counted as `unreadable` by the deck. A test that reaches this point
2397/// forgot to call [`set_home`] (or set `MAGI_HOME`) - that is a bug in the
2398/// test, not a case to serve, so it panics instead of writing anywhere.
2399pub fn home() -> PathBuf {
2400 resolve_home(HOME.get().cloned(), std::env::var_os("MAGI_HOME"))
2401}
2402
2403/// The decision `home` makes, taking its two overrides as plain values
2404/// instead of reading the `OnceLock` and the environment itself.
2405///
2406/// Pulled out so the `cfg(test)` panic is asserted directly against a
2407/// `None, None` input, rather than racing every other unit test in the
2408/// binary for who touches the process-global `HOME` first.
2409fn resolve_home(pinned: Option<PathBuf>, magi_home_env: Option<std::ffi::OsString>) -> PathBuf {
2410 if let Some(dir) = pinned {
2411 return dir;
2412 }
2413 if let Some(dir) = magi_home_env {
2414 return PathBuf::from(dir);
2415 }
2416 #[cfg(test)]
2417 {
2418 panic!(
2419 "run::home() was reached in a test without run::set_home() or \
2420 MAGI_HOME; this would write into the operator's real \
2421 <data_local>/magi. Call `run::set_home(temp_dir)` before any \
2422 code path that touches a RunState."
2423 );
2424 }
2425 #[cfg(not(test))]
2426 {
2427 dirs::data_local_dir()
2428 .unwrap_or_else(|| PathBuf::from("."))
2429 .join("magi")
2430 }
2431}
2432
2433/// [`home`] without the test panic: `None` in a test that never pinned a
2434/// home, so best-effort writers (see [`crate::notices::raise`]) skip the write
2435/// instead of touching the operator's real state or aborting the test.
2436pub fn try_home() -> Option<PathBuf> {
2437 if cfg!(test) {
2438 HOME.get()
2439 .cloned()
2440 .or_else(|| std::env::var_os("MAGI_HOME").map(PathBuf::from))
2441 } else {
2442 Some(home())
2443 }
2444}
2445
2446/// Pin the run home for this process. The first call wins.
2447pub fn set_home(dir: PathBuf) {
2448 let _ = HOME.set(dir);
2449}
2450
2451static HOME: std::sync::OnceLock<PathBuf> = std::sync::OnceLock::new();
2452
2453/// `<home>/runs`.
2454pub fn runs_root() -> PathBuf {
2455 home().join("runs")
2456}
2457
2458/// The worktree root a run uses when the config sets none: `~/wt/magi`.
2459///
2460/// One definition of the default, so the folder the janitor folds and the
2461/// folder the health view sizes cannot drift apart: a run with no configured
2462/// [`crate::config::Graph::worktree_root`] lays its worktrees exactly here.
2463pub fn default_worktree_root() -> PathBuf {
2464 dirs::home_dir()
2465 .unwrap_or_else(|| PathBuf::from("."))
2466 .join("wt")
2467 .join("magi")
2468}
2469
2470/// Directory for one run id.
2471pub fn run_dir(id: &str) -> PathBuf {
2472 runs_root().join(id)
2473}
2474
2475/// Every run id on disk, newest first.
2476///
2477/// A directory is a run because of its **name**, not because it holds a
2478/// readable `run.json`. A run whose very first save lost the machine's last
2479/// free bytes leaves `<id>/run.json.tmp` and nothing else, and filtering on
2480/// `run.json` made that run invisible everywhere: not in `magi list`, not in
2481/// `runs_unreadable`, not on the phone, so nothing could report it and no
2482/// route could clear it. `88c0` sat like that for two days. Unreadable is
2483/// counted, never hidden - the readers already say why each one cannot be
2484/// read, and `fold_unreadable` is how a record like this leaves.
2485pub fn list_ids() -> Vec<String> {
2486 let mut ids: Vec<String> = std::fs::read_dir(runs_root())
2487 .into_iter()
2488 .flatten()
2489 .flatten()
2490 .filter(|e| e.path().is_dir())
2491 .map(|e| e.file_name().to_string_lossy().into_owned())
2492 .filter(|name| is_run_id(name))
2493 .collect();
2494 // Ids start with a sortable timestamp.
2495 ids.sort_unstable_by(|a, b| b.cmp(a));
2496 ids
2497}
2498
2499/// Does `name` have the shape [`new_id`] mints: `YYYYMMDD-HHMMSS-xxxx`?
2500///
2501/// The test for "this directory is a run", so a stray folder under
2502/// `<home>/runs` is not reported as a broken run.
2503///
2504/// The tag is checked for length and for being alphanumeric, not for being
2505/// hex: real ids are hex, but fixtures across this crate name runs
2506/// `...-dead` / `...-gone` / `...-once`, and a predicate that disowned those
2507/// would be asserting the fixtures' spelling rather than the shape.
2508pub fn is_run_id(name: &str) -> bool {
2509 let mut parts = name.split('-');
2510 let (Some(day), Some(time), Some(tag), None) =
2511 (parts.next(), parts.next(), parts.next(), parts.next())
2512 else {
2513 return false;
2514 };
2515 day.len() == 8
2516 && day.bytes().all(|b| b.is_ascii_digit())
2517 && time.len() == 6
2518 && time.bytes().all(|b| b.is_ascii_digit())
2519 && tag.len() == 4
2520 && tag.bytes().all(|b| b.is_ascii_alphanumeric())
2521}
2522
2523/// Expand an id prefix to exactly one run id.
2524pub fn resolve_id(prefix: &str) -> Result<String> {
2525 // A whole id names its directory, readable state or not: the run whose
2526 // `run.json` never landed still has to be reachable by `magi show` and
2527 // by the fold route, which is the only way its record ever leaves.
2528 if is_run_id(prefix) && run_dir(prefix).is_dir() {
2529 return Ok(prefix.to_owned());
2530 }
2531 let hits: Vec<String> = list_ids()
2532 .into_iter()
2533 .filter(|id| id.starts_with(prefix) || id.ends_with(prefix))
2534 .collect();
2535 match hits.len() {
2536 1 => Ok(hits.into_iter().next().expect("exactly one hit")),
2537 0 => bail!("no run matches `{prefix}`"),
2538 _ => bail!(
2539 "`{prefix}` matches {} runs: {}",
2540 hits.len(),
2541 hits.join(", ")
2542 ),
2543 }
2544}
2545
2546/// The most recent run, if any.
2547pub fn latest_id() -> Option<String> {
2548 list_ids().into_iter().next()
2549}
2550
2551/// `YYYYMMDD-HHMMSS-xxxx`, sortable and short enough for a branch name.
2552///
2553/// The four hex digits are fresh entropy, **not** `blind.seed`. They were the
2554/// seed, and a pinned seed then made the whole id a function of the second it
2555/// started in: two runs a second apart were distinguishable, two in the same
2556/// second were not. Everything keyed on the id collided with them - the run
2557/// directory, `artifacts/`, and the candidate worktrees under
2558/// `wt/magi/<short>/`.
2559///
2560/// `tests/common` pins the seed on purpose, so its integration tests all share
2561/// one suffix. On Windows the suite is slow enough that the seconds differ and
2562/// nothing showed; on Linux `graph_dropped_stream`'s three tests run inside
2563/// 16s, so two of them shared a run directory and the second read an artifact
2564/// the first had written (`impl-B-resume.out`) - a failure that looked like the
2565/// resume logic misbehaving and was really two runs in one directory.
2566///
2567/// A seed exists to make the *blind* decisions reproducible: label assignment
2568/// and per-judge presentation order. It was never meant to name the run, and
2569/// `RunState::seed` still carries it for what it is for.
2570fn new_id() -> String {
2571 let stamp = Zoned::now().strftime("%Y%m%d-%H%M%S");
2572 let entropy = crate::rng::entropy();
2573 format!("{stamp}-{:04x}", (entropy ^ (entropy >> 32)) & 0xffff)
2574}
2575
2576/// Keep the last `max` bytes of `text`, on a line boundary.
2577pub fn tail(text: &str, max: usize) -> String {
2578 if text.len() <= max {
2579 return text.to_owned();
2580 }
2581 let mut cut = text.len() - max;
2582 while cut < text.len() && !text.is_char_boundary(cut) {
2583 cut += 1;
2584 }
2585 let slice = &text[cut..];
2586 let start = slice.find('\n').map_or(0, |i| i + 1);
2587 format!(
2588 "[... {} earlier bytes omitted ...]\n{}",
2589 cut,
2590 &slice[start..]
2591 )
2592}
2593
2594/// Path of a run artifact.
2595pub fn artifact_path(run: &RunState, name: &str) -> PathBuf {
2596 run.dir().join("artifacts").join(name)
2597}
2598
2599/// Write an artifact, creating the directory if needed.
2600pub fn write_artifact(run: &RunState, name: &str, body: &str) -> Result<PathBuf> {
2601 let path = artifact_path(run, name);
2602 if let Some(parent) = path.parent() {
2603 std::fs::create_dir_all(parent).with_context(|| format!("create {}", parent.display()))?;
2604 }
2605 std::fs::write(&path, body).with_context(|| format!("write {}", path.display()))?;
2606 Ok(path)
2607}
2608
2609/// Read an artifact back, e.g. a stored patch on resume.
2610pub fn read_artifact(run: &RunState, name: &str) -> Option<String> {
2611 std::fs::read_to_string(artifact_path(run, name)).ok()
2612}
2613
2614#[cfg(test)]
2615mod tests {
2616 use super::*;
2617
2618 fn state() -> RunState {
2619 RunState::new(
2620 PathBuf::from("/repo"),
2621 "main".to_owned(),
2622 "abc1234def".to_owned(),
2623 "add retries".to_owned(),
2624 Config::default(),
2625 )
2626 }
2627
2628 #[test]
2629 fn superseded_is_terminal_but_not_resumable() {
2630 assert!(RunStatus::Superseded.done());
2631 assert!(!RunStatus::Superseded.resumable());
2632 assert_eq!(RunStatus::Superseded.as_str(), "superseded");
2633 }
2634
2635 #[test]
2636 fn load_under_reads_back_exactly_what_save_under_wrote_at_an_explicit_home() {
2637 // Both rooted at an explicit `home` rather than the process-global
2638 // one - a caller with its own `home` (a test fixture, a housekeeping
2639 // pass) must round-trip through exactly that directory, never
2640 // through whichever home some other test in the same binary pinned
2641 // into the global `OnceLock` first.
2642 let dir = tempfile::tempdir().unwrap();
2643 let mut original = state();
2644 original.id = "20260101-000000-load".to_owned();
2645 original.status = RunStatus::Blocked;
2646 original.save_under(dir.path()).unwrap();
2647
2648 let reloaded = RunState::load_under(&original.id, dir.path()).unwrap();
2649 assert_eq!(reloaded.id, original.id);
2650 assert_eq!(reloaded.status, RunStatus::Blocked);
2651
2652 assert!(
2653 RunState::load_under("20260101-000000-none", dir.path()).is_err(),
2654 "an id with nothing saved under this home must not silently read something else"
2655 );
2656 }
2657
2658 #[test]
2659 fn resolve_home_prefers_the_pin_then_the_env_var() {
2660 let pinned = PathBuf::from("/pinned");
2661 assert_eq!(
2662 resolve_home(Some(pinned.clone()), Some("/env".into())),
2663 pinned,
2664 "a pin wins even over MAGI_HOME"
2665 );
2666 assert_eq!(
2667 resolve_home(None, Some("/env".into())),
2668 PathBuf::from("/env")
2669 );
2670 }
2671
2672 #[test]
2673 #[should_panic(expected = "run::set_home()")]
2674 fn resolve_home_refuses_to_fall_back_to_the_operators_real_home() {
2675 // Neither override present is exactly the state a test reaches by
2676 // forgetting `set_home`/`MAGI_HOME` - the accident that put three
2677 // broken fixture runs into the operator's real history. Asserted
2678 // against the pure decision directly, not `home()` itself, because
2679 // `HOME` is a process-wide `OnceLock` another test may have already
2680 // set - this must not depend on test execution order.
2681 resolve_home(None, None);
2682 }
2683
2684 #[test]
2685 fn a_run_is_named_by_shape_so_a_state_less_directory_is_still_a_run() {
2686 // The shape `new_id` mints. A directory answering to it is a run even
2687 // with no readable `run.json`: that is how a save that ran out of
2688 // disk stays visible instead of vanishing from every listing.
2689 assert!(is_run_id(&new_id()));
2690 assert!(is_run_id("20260904-014540-88c0"));
2691 // Not runs: a stray folder, a truncated id, a non-hex tag, and an id
2692 // with an extra segment (a worktree label, say).
2693 assert!(!is_run_id("scratch"));
2694 assert!(!is_run_id("20260904-014540"));
2695 assert!(!is_run_id("20260904-014540-88c0f"));
2696 assert!(!is_run_id("2026090x-014540-88c0"));
2697 assert!(!is_run_id("20260904-014540-88c0-A"));
2698 }
2699
2700 #[test]
2701 fn ids_are_sortable_and_short_suffixed() {
2702 let s = state();
2703 let parts: Vec<&str> = s.id.split('-').collect();
2704 assert_eq!(parts.len(), 3);
2705 assert_eq!(parts[0].len(), 8);
2706 assert_eq!(parts[1].len(), 6);
2707 assert_eq!(parts[2].len(), 4);
2708 assert_eq!(s.short(), parts[2]);
2709 }
2710
2711 #[test]
2712 fn branch_names_carry_the_label_not_the_author() {
2713 let s = state();
2714 let b = s.branch_for('B');
2715 assert_eq!(b, format!("magi/{}/B", s.short()));
2716 assert!(!b.contains("claude"));
2717 }
2718
2719 /// A pinned seed reproduces the blind decisions. It must **not** reproduce
2720 /// the run's identity.
2721 ///
2722 /// `assert_eq!(a.short(), b.short())` used to stand where the last
2723 /// assertion is now, and it was pinning the defect: with the id's suffix
2724 /// derived from the seed, two runs started in the same second were the
2725 /// same run as far as the filesystem was concerned - one directory, one
2726 /// `artifacts/`, one set of candidate worktrees. `tests/common` pins a
2727 /// seed for every integration test, so on Linux, where the suite is fast,
2728 /// two tests in `graph_dropped_stream` shared a directory and one read the
2729 /// other's artifact.
2730 #[test]
2731 fn a_pinned_seed_is_reproducible_but_never_the_run_id() {
2732 let mut cfg = Config::default();
2733 cfg.blind.seed = Some(1234);
2734 let a = RunState::new(
2735 PathBuf::from("/r"),
2736 "main".to_owned(),
2737 "c".to_owned(),
2738 "t".to_owned(),
2739 cfg.clone(),
2740 );
2741 let b = RunState::new(
2742 PathBuf::from("/r"),
2743 "main".to_owned(),
2744 "c".to_owned(),
2745 "t".to_owned(),
2746 cfg,
2747 );
2748 // What the seed is for: the same shuffles, run after run.
2749 assert_eq!(a.seed, 1234);
2750 assert_eq!(a.seed, b.seed);
2751 // What it is not for. Two runs are two runs, in the same second or
2752 // not, and everything keyed on the id depends on that.
2753 assert_ne!(
2754 a.id, b.id,
2755 "two runs sharing an id share a directory, artifacts and worktrees"
2756 );
2757 }
2758
2759 #[test]
2760 fn status_terminality() {
2761 assert!(RunStatus::Merged.done());
2762 assert!(RunStatus::Blocked.done());
2763 assert!(!RunStatus::Reviewing.done());
2764 }
2765
2766 fn overrun_seat(now: Timestamp, elapsed_secs: i64, timeout_secs: u64) -> ActiveSeat {
2767 ActiveSeat {
2768 node: "implement".to_owned(),
2769 started_at: now - jiff::SignedDuration::new(elapsed_secs, 0),
2770 timeout_secs,
2771 attempt: 0,
2772 task: None,
2773 command: None,
2774 index: None,
2775 total: None,
2776 }
2777 }
2778
2779 #[test]
2780 fn active_all_overrun_requires_every_seat_past_its_own_timeout() {
2781 let mut s = state();
2782 let now = Timestamp::now();
2783 assert!(
2784 !s.active_all_overrun(now),
2785 "nothing active is not evidence of anything"
2786 );
2787
2788 s.active
2789 .insert("impl-A".to_owned(), overrun_seat(now, 21_000, 3_600));
2790 assert!(
2791 s.active_all_overrun(now),
2792 "21000s elapsed against a 3600s budget"
2793 );
2794
2795 // A seat still well within its own budget means the run is not
2796 // provably dead, however far its sibling has overrun.
2797 s.active
2798 .insert("impl-B".to_owned(), overrun_seat(now, 0, 3_600));
2799 assert!(!s.active_all_overrun(now));
2800 }
2801
2802 /// A schema-11 run that ended Blocked cannot be pinned by a daemon pid
2803 /// that is still alive; one still mid-walk keeps asking the pid.
2804 #[test]
2805 fn migrating_schema_11_marks_only_ended_runs_as_exited() {
2806 for (status, exited) in [
2807 (RunStatus::Blocked, true),
2808 (RunStatus::Failed, true),
2809 (RunStatus::Stalled, true),
2810 (RunStatus::Landing, true),
2811 (RunStatus::Reviewing, false),
2812 ] {
2813 let mut s = state();
2814 s.schema = 11;
2815 s.status = status;
2816 let m = migrate_schema(s).unwrap();
2817 assert_eq!(m.schema, SCHEMA);
2818 assert_eq!(m.driver_exited, exited, "{status:?}");
2819 }
2820 }
2821
2822 /// A driver that recorded its own exit reads dead without its pid being
2823 /// asked: a daemon's pid outlives the runs it drove.
2824 #[test]
2825 fn an_exited_driver_is_dead_even_when_its_pid_answers_alive() {
2826 let mut s = state();
2827 s.driver_pid = Some(4242);
2828 s.driver_started_at = Some("2026-09-22T10:00:00Z".to_owned());
2829 s.driver_exited = true;
2830 let never = |_| -> Option<bool> { panic!("the pid must not be asked") };
2831 assert_eq!(
2832 s.liveness_with(false, never, |_| panic!("nor its identity")),
2833 Liveness::Dead
2834 );
2835 assert_eq!(s.liveness_with(true, never, |_| None), Liveness::Live);
2836 }
2837
2838 /// A daemon claim wins outright, whatever `driver_pid` or either query
2839 /// says — the stronger, independently-heartbeating signal. Neither query
2840 /// closure is even called: a daemon claim short-circuits before either
2841 /// one, which panicking closures here prove.
2842 #[test]
2843 fn liveness_reads_live_from_a_daemon_claim_alone() {
2844 let mut s = state();
2845 s.driver_pid = None;
2846 assert_eq!(
2847 s.liveness_with(
2848 true,
2849 |_| panic!("a daemon claim needs no pid query"),
2850 |_| panic!("a daemon claim needs no identity query")
2851 ),
2852 Liveness::Live,
2853 "a daemon claim needs no pid to back it up"
2854 );
2855 }
2856
2857 /// The gap `driver_pid` closes: no daemon claim (every manual `magi run`
2858 /// / `magi review`), but the recorded pid answers alive *and* the
2859 /// process currently holding it still carries the same start-time
2860 /// marker this run recorded — proof it is genuinely the same process,
2861 /// not merely the same number.
2862 #[test]
2863 fn liveness_reads_live_from_a_confirmed_pid_with_a_matching_identity() {
2864 let mut s = state();
2865 s.driver_pid = Some(4242);
2866 s.driver_started_at = Some("2026-09-22T10:00:00Z".to_owned());
2867 assert_eq!(
2868 s.liveness_with(
2869 false,
2870 |pid| {
2871 assert_eq!(pid, 4242);
2872 Some(true)
2873 },
2874 |pid| {
2875 assert_eq!(pid, 4242);
2876 Some("2026-09-22T10:00:00Z".to_owned())
2877 }
2878 ),
2879 Liveness::Live
2880 );
2881 }
2882
2883 /// No daemon claim and the recorded pid confirmed gone by the OS itself
2884 /// — dead outright, and the identity query is never even reached (a
2885 /// panicking closure proves it), since there is nothing left to
2886 /// corroborate.
2887 #[test]
2888 fn liveness_reads_dead_from_a_confirmed_dead_pid() {
2889 let mut s = state();
2890 s.driver_pid = Some(4242);
2891 assert_eq!(
2892 s.liveness_with(
2893 false,
2894 |_| Some(false),
2895 |_| panic!("a confirmed-dead pid needs no identity query")
2896 ),
2897 Liveness::Dead
2898 );
2899 }
2900
2901 /// The gap this task's review round exists to close: a killed manual
2902 /// run's pid gets handed to a wholly unrelated later process. `pid_status`
2903 /// alone would read that as `Live` — the reused pid really is alive —
2904 /// but the process now holding it started at a different moment than the
2905 /// one this run recorded, so this must read `Dead`, not `Live`: a
2906 /// mismatch is exactly as good as proof the original driver is gone.
2907 #[test]
2908 fn liveness_reads_dead_when_a_live_pid_no_longer_matches_the_recorded_start_time() {
2909 let mut s = state();
2910 s.driver_pid = Some(4242);
2911 s.driver_started_at = Some("2026-09-22T10:00:00Z".to_owned());
2912 assert_eq!(
2913 s.liveness_with(
2914 false,
2915 |_| Some(true),
2916 |_| Some("2026-09-22T11:30:00Z".to_owned())
2917 ),
2918 Liveness::Dead,
2919 "the pid is alive, but under a different process than the one this run recorded"
2920 );
2921 }
2922
2923 /// Missing information never collapses to `Dead`: an old run with no
2924 /// `driver_pid` at all, a `driver_pid` this build could not query, a live
2925 /// pid with no recorded start time to compare (an even older run, before
2926 /// that field existed), and a live pid whose current identity this build
2927 /// could not re-query, all read as `Unknown` — never a guess in either
2928 /// direction.
2929 #[test]
2930 fn liveness_never_guesses_out_of_missing_information() {
2931 let mut s = state();
2932 s.driver_pid = None;
2933 assert_eq!(
2934 s.liveness_with(
2935 false,
2936 |_| panic!("no pid to query"),
2937 |_| panic!("no pid to query")
2938 ),
2939 Liveness::Unknown,
2940 "no driver_pid recorded at all — an old run predating this field"
2941 );
2942
2943 s.driver_pid = Some(4242);
2944 assert_eq!(
2945 s.liveness_with(false, |_| None, |_| panic!("inconclusive already")),
2946 Liveness::Unknown,
2947 "a pid to ask, but the platform could not answer for it"
2948 );
2949
2950 s.driver_started_at = None;
2951 assert_eq!(
2952 s.liveness_with(false, |_| Some(true), |_| Some("anything".to_owned())),
2953 Liveness::Unknown,
2954 "a live pid, but no recorded marker to corroborate it against — an old run \
2955 predating `driver_started_at`"
2956 );
2957
2958 s.driver_started_at = Some("2026-09-22T10:00:00Z".to_owned());
2959 assert_eq!(
2960 s.liveness_with(false, |_| Some(true), |_| None),
2961 Liveness::Unknown,
2962 "a live pid and a recorded marker, but the identity re-query itself failed"
2963 );
2964 }
2965
2966 #[test]
2967 fn abandon_clears_active_and_fails_a_non_terminal_run() {
2968 let mut s = state();
2969 s.status = RunStatus::Implementing;
2970 let now = Timestamp::now();
2971 s.active
2972 .insert("impl-A".to_owned(), overrun_seat(now, 21_000, 3_600));
2973
2974 s.abandon("daemon");
2975
2976 assert!(s.active.is_empty());
2977 assert_eq!(s.status, RunStatus::Failed);
2978 assert!(
2979 s.events
2980 .last()
2981 .expect("an event was logged")
2982 .message
2983 .contains("impl-A"),
2984 "the event names the abandoned seat"
2985 );
2986 }
2987
2988 #[test]
2989 fn abandon_never_overwrites_a_status_already_terminal() {
2990 let mut s = state();
2991 s.status = RunStatus::Ready;
2992 let now = Timestamp::now();
2993 s.active
2994 .insert("impl-A".to_owned(), overrun_seat(now, 21_000, 3_600));
2995
2996 s.abandon("daemon");
2997
2998 assert!(s.active.is_empty());
2999 assert_eq!(
3000 s.status,
3001 RunStatus::Ready,
3002 "a run already done must not be relabelled Failed"
3003 );
3004 }
3005
3006 #[test]
3007 fn candidate_viability_excludes_empty_and_failed() {
3008 let mut c = Candidate {
3009 index: 0,
3010 label: 'A',
3011 agent: "a".to_owned(),
3012 branch: "b".to_owned(),
3013 worktree: PathBuf::from("/w"),
3014 summary: String::new(),
3015 stat: String::new(),
3016 files: 1,
3017 commits: 1,
3018 empty: false,
3019 failed: None,
3020 verified_noop: None,
3021 duration_ms: 0,
3022 folded: false,
3023 };
3024 assert!(c.viable());
3025 c.empty = true;
3026 assert!(!c.viable());
3027 c.empty = false;
3028 c.failed = Some("timeout".to_owned());
3029 assert!(!c.viable());
3030 }
3031
3032 #[test]
3033 fn build_failure_is_distinguished_from_a_failing_test() {
3034 let link_race = CommandOutcome {
3035 command: "cargo test".to_owned(),
3036 code: Some(1),
3037 output_tail: "LINK : fatal error LNK1104: cannot open file \
3038 'graph_dirty_tree-71d4dc8e.exe'\n\
3039 error: could not compile `magi-cli` (test \"graph_dirty_tree\")"
3040 .to_owned(),
3041 duration_ms: 500,
3042 resource_blocked: false,
3043 };
3044 assert!(!link_race.ok());
3045 assert!(link_race.build_failed());
3046
3047 let failing_test = CommandOutcome {
3048 command: "cargo test".to_owned(),
3049 code: Some(101),
3050 output_tail: "thread 'it_works' panicked at 'assertion failed'".to_owned(),
3051 duration_ms: 500,
3052 resource_blocked: false,
3053 };
3054 assert!(!failing_test.ok());
3055 assert!(
3056 !failing_test.build_failed(),
3057 "a real test failure must not be classed as a build failure"
3058 );
3059
3060 let passing = CommandOutcome {
3061 command: "cargo test".to_owned(),
3062 code: Some(0),
3063 output_tail: String::new(),
3064 duration_ms: 500,
3065 resource_blocked: false,
3066 };
3067 assert!(passing.ok());
3068 assert!(!passing.build_failed());
3069 }
3070
3071 #[test]
3072 fn tail_keeps_the_end_on_a_line_boundary() {
3073 let text = (0..100).map(|i| format!("line {i}\n")).collect::<String>();
3074 let t = tail(&text, 40);
3075 assert!(t.starts_with("[..."));
3076 assert!(t.ends_with("line 99\n"));
3077 assert!(t.len() < 120);
3078 assert_eq!(tail("short", 40), "short");
3079 }
3080
3081 #[test]
3082 fn tail_survives_multibyte_cuts() {
3083 let text = "あ".repeat(50);
3084 let t = tail(&text, 10);
3085 assert!(t.contains("earlier bytes omitted"));
3086 assert!(t.ends_with('あ'));
3087 }
3088
3089 fn finding(id: &str, severity: crate::verdict::Severity) -> crate::verdict::Finding {
3090 crate::verdict::Finding {
3091 id: id.to_owned(),
3092 severity,
3093 file: None,
3094 line: None,
3095 title: "x".to_owned(),
3096 detail: String::new(),
3097 }
3098 }
3099
3100 fn round(clean: bool, findings: Vec<crate::verdict::Finding>) -> ReviewRound {
3101 ReviewRound {
3102 round: 1,
3103 head: "h".to_owned(),
3104 verified_head: None,
3105 verified_at: None,
3106 reviews: vec![ReviewRecord {
3107 attempts: 0,
3108 reviewer: 1,
3109 agent: "a".to_owned(),
3110 summary: String::new(),
3111 findings,
3112 vote: None,
3113 failed: None,
3114 duration_ms: 0,
3115 }],
3116 e2e: Vec::new(),
3117 verify_retried: false,
3118 e2e_deferred: false,
3119 e2e_defer_reason: None,
3120 fix: None,
3121 blocking: 0,
3122 answered: 1,
3123 expected: 1,
3124 clean,
3125 progressed: false,
3126 vote_split: false,
3127 reconsideration: Vec::new(),
3128 verdict: None,
3129 }
3130 }
3131
3132 #[test]
3133 fn e2e_status_tells_deferred_apart_from_not_configured() {
3134 let mut r = round(false, Vec::new());
3135 assert_eq!(r.e2e_status(), E2eStatus::NotConfigured);
3136
3137 r.e2e_deferred = true;
3138 assert_eq!(
3139 r.e2e_status(),
3140 E2eStatus::Deferred,
3141 "an empty e2e must not read as unconfigured once it was deferred on purpose"
3142 );
3143
3144 r.e2e = vec![CommandOutcome {
3145 command: "test".to_owned(),
3146 code: Some(0),
3147 output_tail: String::new(),
3148 duration_ms: 0,
3149 resource_blocked: false,
3150 }];
3151 assert_eq!(
3152 r.e2e_status(),
3153 E2eStatus::Passed,
3154 "a round with real outcomes is never read as deferred, even if the flag is still set"
3155 );
3156 }
3157
3158 #[test]
3159 fn e2e_status_reports_a_real_failure_as_failed_not_deferred() {
3160 let mut r = round(false, Vec::new());
3161 r.e2e = vec![CommandOutcome {
3162 command: "test".to_owned(),
3163 code: Some(1),
3164 output_tail: "boom".to_owned(),
3165 duration_ms: 0,
3166 resource_blocked: false,
3167 }];
3168 assert_eq!(r.e2e_status(), E2eStatus::Failed);
3169 }
3170
3171 #[test]
3172 fn e2e_status_never_reads_a_resource_block_as_a_failure() {
3173 // The exact shape of contention on the shared build cache: `e2e`
3174 // holds one outcome, and it is `resource_blocked`, never a command
3175 // that actually ran and produced a red exit code.
3176 let mut r = round(false, Vec::new());
3177 r.e2e = vec![CommandOutcome {
3178 command: "(waiting for the shared build cache)".to_owned(),
3179 code: None,
3180 output_tail: "contended".to_owned(),
3181 duration_ms: 0,
3182 resource_blocked: true,
3183 }];
3184 assert_eq!(
3185 r.e2e_status(),
3186 E2eStatus::ResourceBlocked,
3187 "magi's own inability to get a command to run must not read as a verdict on the \
3188 patch"
3189 );
3190 }
3191
3192 #[test]
3193 fn verification_summary_is_silent_when_there_is_nothing_worth_saying() {
3194 let mut r = round(true, Vec::new());
3195 assert!(
3196 r.verification_summary("h").is_none(),
3197 "no verify.e2e configured: nothing to surface"
3198 );
3199 r.e2e = vec![CommandOutcome {
3200 command: "test".to_owned(),
3201 code: Some(0),
3202 output_tail: String::new(),
3203 duration_ms: 0,
3204 resource_blocked: false,
3205 }];
3206 assert!(
3207 r.verification_summary("h").is_none(),
3208 "a green result needs no skepticism attached to it"
3209 );
3210 }
3211
3212 #[test]
3213 fn verification_summary_tells_the_current_head_apart_from_an_earlier_one() {
3214 let mut r = round(false, Vec::new());
3215 r.head = "h1".to_owned();
3216 r.e2e = vec![CommandOutcome {
3217 command: "test".to_owned(),
3218 code: Some(1),
3219 output_tail: "boom".to_owned(),
3220 duration_ms: 0,
3221 resource_blocked: false,
3222 }];
3223 r.verified_head = Some("h1".to_owned());
3224 r.verified_at = Some(Timestamp::now());
3225
3226 let fresh = r.verification_summary("h1").expect("a failure is surfaced");
3227 assert!(
3228 fresh.label.contains("this is the head being looked at now"),
3229 "{}",
3230 fresh.label
3231 );
3232 assert_eq!(fresh.tail.as_deref(), Some("$ test\nboom\n"));
3233
3234 let stale = r.verification_summary("h2").expect("still surfaced");
3235 assert!(
3236 stale.label.contains("an earlier head, since superseded"),
3237 "a result about a different commit than the one being looked at now must say so, \
3238 not read as current: {}",
3239 stale.label
3240 );
3241 }
3242
3243 #[test]
3244 fn verification_summary_marks_a_resource_block_and_a_deferral_distinctly_from_a_failure() {
3245 let mut r = round(false, Vec::new());
3246 r.e2e = vec![CommandOutcome {
3247 command: "(waiting for the shared build cache)".to_owned(),
3248 code: None,
3249 output_tail: "contended".to_owned(),
3250 duration_ms: 0,
3251 resource_blocked: true,
3252 }];
3253 let blocked = r
3254 .verification_summary("h")
3255 .expect("a resource block is still surfaced, never silent");
3256 assert!(blocked.label.contains("could not run"));
3257 // No command actually ran, but which operation was attempted is
3258 // still a fact worth showing — never silent past the label either.
3259 let tail = blocked
3260 .tail
3261 .expect("the attempted operation is still named");
3262 assert!(tail.contains("(waiting for the shared build cache)"));
3263 assert!(tail.contains("contended"));
3264
3265 let mut d = round(false, Vec::new());
3266 d.e2e_deferred = true;
3267 d.e2e_defer_reason = Some("2 blocking finding(s) already required a fix".to_owned());
3268 let deferred = d.verification_summary("h").expect("deferred is surfaced");
3269 assert!(deferred.label.contains("deferred to the fixer"));
3270 assert!(deferred.label.contains("2 blocking finding(s)"));
3271 assert!(deferred.tail.is_none());
3272 }
3273
3274 #[test]
3275 fn verification_summary_says_unknown_rather_than_guessing_a_time_or_a_commit() {
3276 let mut r = round(false, Vec::new());
3277 r.e2e = vec![CommandOutcome {
3278 command: "test".to_owned(),
3279 code: Some(1),
3280 output_tail: "boom".to_owned(),
3281 duration_ms: 0,
3282 resource_blocked: false,
3283 }];
3284 // verified_head/verified_at left at their default `None` — exactly
3285 // the shape a schema-7 round with no reconstructable timestamp has.
3286 let summary = r.verification_summary("h").expect("a failure is surfaced");
3287 assert!(summary.label.contains("commit unknown"));
3288 assert!(summary.label.contains("checked at: unknown"));
3289 }
3290
3291 #[test]
3292 fn gate_status_tells_not_run_apart_from_passed_with_no_commands() {
3293 let mut s = state();
3294 assert_eq!(s.gate_status(), GateStatus::NotRun);
3295
3296 s.gate_ran = true;
3297 assert_eq!(
3298 s.gate_status(),
3299 GateStatus::PassedWithNoCommands,
3300 "an empty gate must read as a real pass once gate_ran says it actually ran"
3301 );
3302
3303 s.gate = vec![CommandOutcome {
3304 command: "cargo make check".to_owned(),
3305 code: Some(0),
3306 output_tail: String::new(),
3307 duration_ms: 0,
3308 resource_blocked: false,
3309 }];
3310 assert_eq!(s.gate_status(), GateStatus::Passed);
3311
3312 s.gate[0].code = Some(1);
3313 assert_eq!(s.gate_status(), GateStatus::Failed);
3314
3315 s.gate_ran = false;
3316 assert_eq!(
3317 s.gate_status(),
3318 GateStatus::NotRun,
3319 "gate_ran false must win even over a non-empty gate left from a stale record"
3320 );
3321 }
3322
3323 #[test]
3324 fn open_findings_is_empty_when_the_last_round_was_clean() {
3325 let mut s = state();
3326 s.reviews = vec![round(
3327 true,
3328 vec![finding("R1-1-1", crate::verdict::Severity::Minor)],
3329 )];
3330 assert!(s.open_findings().is_empty());
3331 }
3332
3333 #[test]
3334 fn open_findings_reads_the_last_non_clean_round() {
3335 let mut s = state();
3336 s.reviews = vec![round(
3337 false,
3338 vec![finding("R1-1-1", crate::verdict::Severity::Major)],
3339 )];
3340 let open = s.open_findings();
3341 assert_eq!(open.len(), 1);
3342 assert_eq!(open[0].id, "R1-1-1");
3343 }
3344
3345 #[test]
3346 fn handed_off_with_open_findings_needs_a_mergeable_status_and_an_open_round() {
3347 let mut s = state();
3348 s.reviews = vec![round(
3349 false,
3350 vec![finding("R1-1-1", crate::verdict::Severity::Major)],
3351 )];
3352
3353 s.status = RunStatus::Blocked;
3354 assert!(
3355 !s.handed_off_with_open_findings(),
3356 "a blocked run is not a hand-off"
3357 );
3358
3359 s.status = RunStatus::Ready;
3360 assert!(s.handed_off_with_open_findings());
3361
3362 s.reviews = vec![round(true, Vec::new())];
3363 assert!(
3364 !s.handed_off_with_open_findings(),
3365 "a clean last round has nothing to hand off"
3366 );
3367 }
3368
3369 #[test]
3370 fn unmerged_by_design_is_only_ready_reached_via_merge_mode_none() {
3371 let mut s = state();
3372
3373 s.status = RunStatus::Ready;
3374 assert!(
3375 !s.unmerged_by_design(),
3376 "no merge outcome recorded at all must not be flagged"
3377 );
3378
3379 s.merge = Some(MergeOutcome {
3380 mode: MergeMode::None,
3381 ok: true,
3382 detail: "git merge --no-ff magi/x/A".to_owned(),
3383 empty: false,
3384 });
3385 assert!(
3386 s.unmerged_by_design(),
3387 "Ready reached through mode none is the case this exists to flag"
3388 );
3389
3390 // A PR closed without merging also leaves `status` at `Ready`, but
3391 // through `mode = "pr"` — a run that may still have been landable by
3392 // a person watching the PR, unlike the honest mode-none no-op.
3393 s.merge = Some(MergeOutcome {
3394 mode: MergeMode::Pr,
3395 ok: false,
3396 detail: "https://example.com/pr/1 was closed without merging".to_owned(),
3397 empty: false,
3398 });
3399 assert!(
3400 !s.unmerged_by_design(),
3401 "a closed pull request is a different Ready and must not be relabelled"
3402 );
3403
3404 // Same signal must not fire before the run actually got there.
3405 s.status = RunStatus::Gating;
3406 s.merge = Some(MergeOutcome {
3407 mode: MergeMode::None,
3408 ok: true,
3409 detail: "git merge --no-ff magi/x/A".to_owned(),
3410 empty: false,
3411 });
3412 assert!(
3413 !s.unmerged_by_design(),
3414 "status must actually be Ready, not merely have a stale mode-none merge record"
3415 );
3416 }
3417
3418 #[test]
3419 fn state_round_trips_through_json() {
3420 let s = state();
3421 let body = serde_json::to_string(&s).unwrap();
3422 let back: RunState = serde_json::from_str(&body).unwrap();
3423 assert_eq!(back.id, s.id);
3424 assert_eq!(back.instruction, "add retries");
3425 assert_eq!(back.status, RunStatus::Prep);
3426 }
3427
3428 #[test]
3429 fn a_round_recorded_before_e2e_deferral_existed_still_loads() {
3430 // Exactly the shape a pre-existing `run.json` has for a round: no
3431 // `e2e_deferred`, no `e2e_defer_reason`. Every round used to run e2e
3432 // unconditionally, so the honest reading of an old record's silence
3433 // on this is "it was not deferred" — `false`/`None`, not a load
3434 // failure and not a schema bump (see the `SCHEMA` doc comment: a
3435 // purely additive field whose absence has one unambiguous meaning
3436 // does not need one).
3437 let body = r#"{
3438 "round": 1,
3439 "head": "deadbeef",
3440 "reviews": [],
3441 "e2e": [],
3442 "verify_retried": false,
3443 "fix": null,
3444 "blocking": 0,
3445 "answered": 1,
3446 "expected": 1,
3447 "clean": true
3448 }"#;
3449 let r: ReviewRound = serde_json::from_str(body).expect("an old-shaped round must load");
3450 assert!(!r.e2e_deferred);
3451 assert!(r.e2e_defer_reason.is_none());
3452 assert_eq!(r.e2e_status(), E2eStatus::NotConfigured);
3453 }
3454
3455 #[test]
3456 fn schema_five_state_migrates_old_empty_e2e_and_legacy_verify_budget() {
3457 let mut value = serde_json::to_value(state()).expect("serialize state");
3458 let object = value.as_object_mut().expect("state object");
3459 object.insert("schema".to_owned(), serde_json::json!(5));
3460 let graph = object["config"]["graph"]
3461 .as_object_mut()
3462 .expect("graph object");
3463 graph.insert("timeout_review".to_owned(), serde_json::json!(3600));
3464 graph.remove("timeout_verify");
3465 let review = object["reviews"].as_array_mut().expect("reviews");
3466 review.push(serde_json::json!({
3467 "round": 1, "head": "old", "reviews": [], "e2e": [],
3468 "verify_retried": false, "blocking": 0, "answered": 1,
3469 "expected": 1, "clean": true
3470 }));
3471 let old: RunState = serde_json::from_value(value).expect("schema-5 shape parses");
3472 let migrated = migrate_schema(old).expect("schema 5 migrates");
3473 assert_eq!(migrated.schema, SCHEMA);
3474 assert_eq!(migrated.config.graph.verify_timeout(), 3600);
3475 assert_eq!(migrated.reviews[0].e2e_status(), E2eStatus::NotConfigured);
3476 }
3477
3478 #[test]
3479 fn schema_six_state_with_a_recorded_gate_migrates_to_gate_ran_true() {
3480 let mut value = serde_json::to_value(state()).expect("serialize state");
3481 let object = value.as_object_mut().expect("state object");
3482 object.insert("schema".to_owned(), serde_json::json!(6));
3483 object.insert(
3484 "gate".to_owned(),
3485 serde_json::json!([{
3486 "command": "cargo make check",
3487 "code": 0,
3488 "output_tail": "",
3489 "duration_ms": 0,
3490 "resource_blocked": false
3491 }]),
3492 );
3493 let old: RunState = serde_json::from_value(value).expect("schema-6 shape parses");
3494 let migrated = migrate_schema(old).expect("schema 6 migrates");
3495 assert_eq!(migrated.schema, SCHEMA);
3496 assert!(
3497 migrated.gate_ran,
3498 "a non-empty recorded gate is a real attempt, not an unrun one"
3499 );
3500 assert_eq!(migrated.gate_status(), GateStatus::Passed);
3501 }
3502
3503 #[test]
3504 fn schema_six_state_with_an_empty_gate_migrates_to_gate_ran_false_and_is_retried() {
3505 // The exact shape of the stuck `shoka` run this schema bump fixes:
3506 // `verify.gate` empty, `gate` empty, schema 6. It must come back as
3507 // "not yet run" so the next `gate()` call re-attempts it — and for a
3508 // repo with no gate commands configured, that resolves instantly to
3509 // `PassedWithNoCommands` instead of staying stuck forever.
3510 let mut value = serde_json::to_value(state()).expect("serialize state");
3511 let object = value.as_object_mut().expect("state object");
3512 object.insert("schema".to_owned(), serde_json::json!(6));
3513 object.insert("gate".to_owned(), serde_json::json!([]));
3514 let old: RunState = serde_json::from_value(value).expect("schema-6 shape parses");
3515 let migrated = migrate_schema(old).expect("schema 6 migrates");
3516 assert_eq!(migrated.schema, SCHEMA);
3517 assert!(
3518 !migrated.gate_ran,
3519 "an empty gate on schema 6 is ambiguous and must be treated as unrun"
3520 );
3521 assert_eq!(migrated.gate_status(), GateStatus::NotRun);
3522 }
3523
3524 #[test]
3525 fn schema_seven_state_reconstructs_verified_head_for_a_round_that_actually_ran_e2e() {
3526 // Schema 7's main review loop always checked `e2e` against the
3527 // round's own `head` — it just never wrote that into `verified_head`
3528 // unless a catch-up run had checked a *different* commit. Migrating
3529 // to schema 8 restores that always-true fact instead of leaving a
3530 // reader to assume it.
3531 let mut value = serde_json::to_value(state()).expect("serialize state");
3532 let object = value.as_object_mut().expect("state object");
3533 object.insert("schema".to_owned(), serde_json::json!(7));
3534 let reviews = object["reviews"].as_array_mut().expect("reviews");
3535 reviews.push(serde_json::json!({
3536 "round": 1, "head": "deadbeef", "reviews": [],
3537 "e2e": [{
3538 "command": "cargo test", "code": 0, "output_tail": "",
3539 "duration_ms": 0, "resource_blocked": false
3540 }],
3541 "verify_retried": false, "blocking": 0, "answered": 1,
3542 "expected": 1, "clean": true
3543 }));
3544 let old: RunState = serde_json::from_value(value).expect("schema-7 shape parses");
3545 let migrated = migrate_schema(old).expect("schema 7 migrates");
3546 assert_eq!(migrated.schema, SCHEMA);
3547 assert_eq!(
3548 migrated.reviews[0].verified_head.as_deref(),
3549 Some("deadbeef"),
3550 "a schema-7 round's main-loop e2e was always against its own head, even though the \
3551 field never said so"
3552 );
3553 assert!(
3554 migrated.reviews[0].verified_at.is_none(),
3555 "no historical timestamp exists to reconstruct; unknown stays unknown, not a \
3556 guessed 'now'"
3557 );
3558 }
3559
3560 #[test]
3561 fn schema_seven_state_leaves_a_deferred_round_with_no_verified_head() {
3562 let mut value = serde_json::to_value(state()).expect("serialize state");
3563 let object = value.as_object_mut().expect("state object");
3564 object.insert("schema".to_owned(), serde_json::json!(7));
3565 let reviews = object["reviews"].as_array_mut().expect("reviews");
3566 reviews.push(serde_json::json!({
3567 "round": 1, "head": "deadbeef", "reviews": [],
3568 "e2e": [], "e2e_deferred": true,
3569 "verify_retried": false, "blocking": 1, "answered": 1,
3570 "expected": 1, "clean": false
3571 }));
3572 let old: RunState = serde_json::from_value(value).expect("schema-7 shape parses");
3573 let migrated = migrate_schema(old).expect("schema 7 migrates");
3574 assert!(
3575 migrated.reviews[0].verified_head.is_none(),
3576 "a deferred round never ran e2e; there is nothing to reconstruct"
3577 );
3578 }
3579
3580 #[test]
3581 fn schema_six_serialization_is_rejected_by_a_schema_five_reader() {
3582 let body = serde_json::to_value(state()).expect("serialize state");
3583 assert_eq!(body["schema"], serde_json::json!(SCHEMA));
3584 assert_ne!(body["schema"], serde_json::json!(5));
3585 }
3586
3587 #[test]
3588 fn seat_started_and_finished_track_who_has_not_answered_yet() {
3589 let mut s = state();
3590 s.seat_started("judge", "judge-1", std::time::Duration::from_secs(60), 0);
3591 s.seat_started("judge", "judge-2", std::time::Duration::from_secs(60), 0);
3592 assert_eq!(s.active.len(), 2, "both seats are still out");
3593
3594 s.seat_finished("judge-1");
3595 assert_eq!(
3596 s.active.keys().collect::<Vec<_>>(),
3597 vec!["judge-2"],
3598 "only the seat that answered drops out; judge-2 is still waited on"
3599 );
3600 }
3601
3602 /// `seats_active` / `tasks_active` are the accessors report/web read
3603 /// instead of `active` directly, so neither ever counts the other kind of
3604 /// entry as a seat — a `verify.e2e` task must never inflate a quorum or
3605 /// seat count, and a seat must never show up in a task listing.
3606 #[test]
3607 fn seats_active_and_tasks_active_never_cross_over() {
3608 let mut s = state();
3609 s.seat_started("judge", "judge-1", std::time::Duration::from_secs(60), 0);
3610 s.task_command(
3611 "e2e",
3612 "verify",
3613 0,
3614 "cargo test",
3615 1,
3616 2,
3617 std::time::Duration::from_secs(600),
3618 );
3619
3620 assert_eq!(
3621 s.seats_active()
3622 .map(|(k, _)| k.as_str())
3623 .collect::<Vec<_>>(),
3624 vec!["judge-1"]
3625 );
3626 assert_eq!(
3627 s.tasks_active()
3628 .map(|(k, _)| k.as_str())
3629 .collect::<Vec<_>>(),
3630 vec!["e2e"]
3631 );
3632
3633 // A command boundary updates the same entry in place — still one
3634 // task, never a second one accumulating alongside it.
3635 s.task_command(
3636 "e2e",
3637 "verify",
3638 0,
3639 "cargo clippy",
3640 2,
3641 2,
3642 std::time::Duration::from_secs(600),
3643 );
3644 assert_eq!(s.tasks_active().count(), 1);
3645 assert_eq!(s.active["e2e"].command.as_deref(), Some("cargo clippy"));
3646
3647 s.task_finished("e2e");
3648 assert!(s.tasks_active().next().is_none());
3649 assert_eq!(
3650 s.seats_active()
3651 .map(|(k, _)| k.as_str())
3652 .collect::<Vec<_>>(),
3653 vec!["judge-1"],
3654 "clearing the task must not touch the seat entry"
3655 );
3656 }
3657
3658 #[test]
3659 fn a_retry_is_recorded_as_a_later_attempt_on_the_same_seat() {
3660 let mut s = state();
3661 s.seat_started("review", "review-2", std::time::Duration::from_secs(30), 0);
3662 s.seat_finished("review-2");
3663 // A nudge re-asks the same seat; attempt says this is not the first
3664 // time, which is the only trace a nudge otherwise leaves behind.
3665 s.seat_started("review", "review-2", std::time::Duration::from_secs(30), 1);
3666 assert_eq!(s.active["review-2"].attempt, 1);
3667 }
3668
3669 #[test]
3670 fn active_seat_reports_elapsed_and_remaining_time() {
3671 let now = Timestamp::now();
3672 let started = now - jiff::SignedDuration::from_secs(30);
3673 let seat = ActiveSeat {
3674 node: "judge".to_owned(),
3675 started_at: started,
3676 timeout_secs: 100,
3677 attempt: 0,
3678 task: None,
3679 command: None,
3680 index: None,
3681 total: None,
3682 };
3683 assert_eq!(seat.elapsed_secs(now), 30);
3684 assert_eq!(seat.remaining_secs(now), 70);
3685 }
3686
3687 #[test]
3688 fn remaining_time_never_goes_negative_past_the_timeout() {
3689 // `agy`'s own print-timeout occasionally overruns by a hair before the
3690 // kill lands; a naive subtraction would print a negative "time left".
3691 let now = Timestamp::now();
3692 let started = now - jiff::SignedDuration::from_secs(200);
3693 let seat = ActiveSeat {
3694 node: "implement".to_owned(),
3695 started_at: started,
3696 timeout_secs: 100,
3697 attempt: 1,
3698 task: None,
3699 command: None,
3700 index: None,
3701 total: None,
3702 };
3703 assert_eq!(seat.remaining_secs(now), 0);
3704 }
3705
3706 #[test]
3707 fn clear_active_drops_stale_seats_and_reports_whether_it_did() {
3708 let mut s = state();
3709 assert!(!s.clear_active(), "nothing to clear on a fresh run");
3710 s.seat_started(
3711 "implement",
3712 "impl-B",
3713 std::time::Duration::from_secs(3600),
3714 0,
3715 );
3716 assert!(s.clear_active(), "a leftover entry is reported as cleared");
3717 assert!(s.active.is_empty());
3718 }
3719
3720 #[test]
3721 fn active_seat_carries_nothing_that_could_be_read_as_output_bytes() {
3722 // `agy` prints exactly one JSON object, at the very end (see
3723 // `agent::dropped_stream`'s doc comment) — a seat can sit at zero
3724 // captured bytes for its whole timeout while working normally. So
3725 // `ActiveSeat` records only the wall-clock facts (when it started,
3726 // its budget, which attempt), never a byte count, which is what
3727 // keeps a reader from being able to build "0 bytes => dead" out of
3728 // it even by accident.
3729 let seat = ActiveSeat {
3730 node: "implement".to_owned(),
3731 started_at: Timestamp::now(),
3732 timeout_secs: 60,
3733 attempt: 0,
3734 task: None,
3735 command: None,
3736 index: None,
3737 total: None,
3738 };
3739 let value = serde_json::to_value(&seat).unwrap();
3740 let keys: std::collections::BTreeSet<String> =
3741 value.as_object().unwrap().keys().cloned().collect();
3742 assert_eq!(
3743 keys,
3744 std::collections::BTreeSet::from([
3745 "node".to_owned(),
3746 "started_at".to_owned(),
3747 "timeout_secs".to_owned(),
3748 "attempt".to_owned(),
3749 ]),
3750 "a byte count here would be a lever to declare a silent-but-healthy seat dead, and \
3751 the task-only fields must stay absent (not null) on an ordinary seat entry"
3752 );
3753 }
3754
3755 /// The same guarantee as
3756 /// [`active_seat_carries_nothing_that_could_be_read_as_output_bytes`],
3757 /// extended to a task entry: `verify.e2e` / `verify.gate` are exactly as
3758 /// silent as `agy` between commands, so a running command-list task must
3759 /// never carry anything a reader could mistake for output-byte evidence
3760 /// either.
3761 #[test]
3762 fn task_active_seat_carries_nothing_that_could_be_read_as_output_bytes() {
3763 let seat = ActiveSeat {
3764 node: "verify".to_owned(),
3765 started_at: Timestamp::now(),
3766 timeout_secs: 600,
3767 attempt: 0,
3768 task: Some("e2e".to_owned()),
3769 command: Some("cargo test".to_owned()),
3770 index: Some(1),
3771 total: Some(3),
3772 };
3773 let value = serde_json::to_value(&seat).unwrap();
3774 let keys: std::collections::BTreeSet<String> =
3775 value.as_object().unwrap().keys().cloned().collect();
3776 assert_eq!(
3777 keys,
3778 std::collections::BTreeSet::from([
3779 "node".to_owned(),
3780 "started_at".to_owned(),
3781 "timeout_secs".to_owned(),
3782 "attempt".to_owned(),
3783 "task".to_owned(),
3784 "command".to_owned(),
3785 "index".to_owned(),
3786 "total".to_owned(),
3787 ]),
3788 );
3789 }
3790
3791 #[test]
3792 fn an_old_run_json_without_active_seats_still_loads() {
3793 // Schema did not bump for this field: an already-written run.json
3794 // simply lacks the key, and `#[serde(default)]` must fill it in
3795 // rather than fail the whole read.
3796 let s = state();
3797 let mut value = serde_json::to_value(&s).unwrap();
3798 value.as_object_mut().unwrap().remove("active");
3799 let back: RunState = serde_json::from_value(value).unwrap();
3800 assert!(back.active.is_empty());
3801 assert_eq!(back.schema, SCHEMA);
3802 }
3803
3804 #[test]
3805 fn an_old_run_json_without_jobs_still_loads() {
3806 // No schema bump for this field either, for the same reason: an
3807 // empty `jobs` list on an old record means exactly what it always
3808 // meant for that record — no adapter existed yet to report one —
3809 // and `#[serde(default)]` fills it in rather than failing the read.
3810 let s = state();
3811 let mut value = serde_json::to_value(&s).unwrap();
3812 value.as_object_mut().unwrap().remove("jobs");
3813 let back: RunState = serde_json::from_value(value).unwrap();
3814 assert!(back.jobs.is_empty());
3815 assert_eq!(back.schema, SCHEMA);
3816 }
3817
3818 #[test]
3819 fn ensure_can_delete_guards_live_and_unfolded_runs() {
3820 let mut s = state();
3821 // 1. A daemon is working on it right now.
3822 s.status = RunStatus::Prep;
3823 let err = s.ensure_can_delete(true).unwrap_err().to_string();
3824 assert!(err.contains("live daemon"), "{err}");
3825
3826 // 2. The same unfinished run with no daemon behind it is a leftover
3827 // from a killed process, and deletable. Without this an interrupted
3828 // run could never be removed: its status stays `prep` forever.
3829 assert!(s.ensure_can_delete(false).is_ok());
3830
3831 // 3. Unfolded candidates are refused either way — that is the guard
3832 // that stops a delete from discarding a worktree.
3833 s.status = RunStatus::Merged;
3834 s.candidates.push(Candidate {
3835 index: 0,
3836 label: 'A',
3837 agent: "a".to_owned(),
3838 branch: "b".to_owned(),
3839 worktree: PathBuf::from("/w"),
3840 summary: String::new(),
3841 stat: String::new(),
3842 files: 1,
3843 commits: 1,
3844 empty: false,
3845 failed: None,
3846 verified_noop: None,
3847 duration_ms: 0,
3848 folded: false,
3849 });
3850 let err = s.ensure_can_delete(false).unwrap_err().to_string();
3851 assert!(
3852 err.contains("magi fold"),
3853 "error must suggest `magi fold`: {err}"
3854 );
3855
3856 // 4. Folded and nobody working on it.
3857 s.candidates[0].folded = true;
3858 assert!(s.ensure_can_delete(false).is_ok());
3859 }
3860}