Skip to main content

leviath_core/
run_meta.rs

1//! Plain, serializable run-state data types.
2//!
3//! These are pure data (`serde`-derived structs/enums plus trivial constructors)
4//! with no filesystem or async dependencies, so they can be named by both
5//! `leviath-cli` and the `leviath-runtime` engine. All on-disk IO for
6//! these types (reading/writing `meta.json`, run directories, snapshots, etc.)
7//! lives in `leviath_cli::runstate`.
8
9use serde::{Deserialize, Serialize};
10use std::collections::HashMap;
11
12mod stage_ledger;
13
14// Re-exported flat rather than left behind a path of their own: the stage ledger
15// moved out of this file because the file got long, and that is a fact about
16// where the source lives, not about what a caller should have to type.
17pub use stage_ledger::{
18    MAX_STAGE_VISITS, StageCall, StageRecord, StageRunStatus, StageVisitRecord,
19};
20
21/// Current status of a background run.
22#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
23#[serde(rename_all = "snake_case")]
24pub enum RunStatus {
25    /// Accepted and being set up; no inference has been issued yet.
26    Starting,
27    /// Working: inferring, calling tools, or moving between stages.
28    Running,
29    /// Blocked on a person. A human-in-the-loop tool or an interaction point is
30    /// waiting for an answer, and the run holds its concurrency slot until it
31    /// gets one.
32    WaitingInput,
33    /// Finished, with nothing further to accept.
34    Complete,
35    /// All required stages done; agent still accepts optional follow-up input.
36    /// Shown as "Complete" in the dashboard - no kill option, input still enabled.
37    CompleteInteractive,
38    /// Paused by the user; resumes on request and is restored paused after a
39    /// daemon restart.
40    Paused,
41    /// Stopped by a failure. `RunMeta::error` carries what went wrong.
42    Error,
43    /// Stopped from outside, by `lev kill` or a shutting-down daemon. Distinct
44    /// from [`Error`](Self::Error): nothing went wrong, someone decided.
45    Cancelled,
46}
47
48impl RunStatus {
49    /// The word this status goes on the wire as: `snake_case`, the same
50    /// spelling serde writes into `meta.json` and into every JSON body that
51    /// carries a whole run.
52    ///
53    /// Here rather than left to each caller because a status reaches a client
54    /// three ways - serialized inside a run, rendered into a `status` string by
55    /// a route that builds its own response shape, and forwarded off the
56    /// engine's event stream - and all three have to spell one state one way.
57    /// [`Display`](std::fmt::Display) is PascalCase and is
58    /// for a person reading a terminal; this is for a client matching on it.
59    pub fn wire(&self) -> &'static str {
60        match self {
61            RunStatus::Starting => "starting",
62            RunStatus::Running => "running",
63            RunStatus::WaitingInput => "waiting_input",
64            RunStatus::Complete => "complete",
65            RunStatus::CompleteInteractive => "complete_interactive",
66            RunStatus::Paused => "paused",
67            RunStatus::Error => "error",
68            RunStatus::Cancelled => "cancelled",
69        }
70    }
71}
72
73impl std::fmt::Display for RunStatus {
74    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
75        match self {
76            RunStatus::Starting => write!(f, "Starting"),
77            RunStatus::Running => write!(f, "Running"),
78            RunStatus::WaitingInput => write!(f, "WaitingInput"),
79            RunStatus::Complete => write!(f, "Complete"),
80            RunStatus::CompleteInteractive => write!(f, "CompleteInteractive"),
81            RunStatus::Paused => write!(f, "Paused"),
82            RunStatus::Error => write!(f, "Error"),
83            RunStatus::Cancelled => write!(f, "Cancelled"),
84        }
85    }
86}
87
88/// Why a run's status is [`RunStatus::WaitingInput`].
89///
90/// `WaitingInput` alone is several unrelated situations wearing one word, and
91/// they call for opposite responses: a fan-out parent whose workers are
92/// churning is healthy and needs nothing, while a run parked on a
93/// tool-approval prompt is stopped dead until a person answers it. With the
94/// two indistinguishable, an operator reading `waiting` across a factory
95/// concludes it has stalled and starts killing healthy runs, and every client
96/// that reads `meta.json` is left guessing the same way.
97///
98/// Derived on demand from markers the engine already sets, by
99/// [`wait_reason_from`]; nothing tracks it separately, so it cannot fall out of
100/// sync with the status it explains. It lives here rather than in the runtime
101/// because it is both reported live over the control socket and written to
102/// `meta.json`, and one vocabulary across those two is the whole point.
103///
104/// Deliberately not new [`RunStatus`] variants: the status is matched
105/// exhaustively across the codebase and serialized two ways on the wire, so
106/// splitting it would break every consumer to express something that is not a
107/// new state. The run really is waiting; this says on what.
108#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)]
109#[serde(rename_all = "snake_case", tag = "reason")]
110pub enum WaitReason {
111    /// Blocked on a tool-approval prompt. Needs a person (or `--yolo`).
112    ToolApproval,
113
114    /// Blocked on a question the agent itself asked (`ask_user_*`,
115    /// `present_for_review`). Needs a person.
116    UserPrompt,
117
118    /// Blocked on a taint-gate clearance prompt. Needs a person.
119    TaintGate,
120
121    /// Blocked on a blueprint stage-boundary checkpoint. Needs a person.
122    InteractionPoint,
123
124    /// Parked while fan-out workers run. Healthy; resolves on its own.
125    FanOutWorkers {
126        /// Workers still to finish, counting both running and not-yet-started.
127        outstanding: usize,
128    },
129
130    /// Parked while spawned sub-agents run (`requires_children`). Healthy;
131    /// resolves on its own.
132    Children {
133        /// Children that have not reached a terminal status.
134        outstanding: usize,
135    },
136
137    /// Parked because something on the machine has to change before this run
138    /// can go on: a provider it needs is not configured, a key was rejected,
139    /// an account is out of credits.
140    ///
141    /// These are all deterministic and all outside the run's control, so
142    /// ending the run would throw away everything it had done to punish a
143    /// person for a typo in `config.toml`. The run holds its place instead,
144    /// and `lev resume` picks it up once the machine is fixed.
145    ///
146    /// The distinction that matters is not "is there a fix" but "does the fix
147    /// let *this* run continue": a broken blueprint is equally deterministic
148    /// and equally fixable, and still cannot be resumed into, because the
149    /// blueprint was read at spawn.
150    NeedsSetup {
151        /// Which kind of problem, so a client can offer the right thing to do
152        /// rather than parse the sentence below.
153        blocker: SetupBlocker,
154        /// What to do about it, in a sentence, for whoever reads the run.
155        remedy: String,
156    },
157}
158
159/// What is stopping a [`WaitReason::NeedsSetup`] run, in a form a client can
160/// branch on.
161///
162/// One variant per remedy, not per error: these are the cases whose *fixes*
163/// differ. Topping up an account, adding a provider to `config.toml` and
164/// replacing a rejected key are three different screens, and a console that
165/// had only the sentence would be reduced to matching on its wording.
166#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
167#[serde(rename_all = "snake_case")]
168pub enum SetupBlocker {
169    /// The stage names a provider this install has not configured.
170    ProviderMissing,
171    /// The account behind the provider is out of credits.
172    CreditsExhausted,
173    /// The key was rejected.
174    AuthFailed,
175    /// The key is valid but not allowed to use the model.
176    Forbidden,
177    /// Every candidate is out of service, for reasons that do not agree or are
178    /// not known. The remedy names what was tried last.
179    ProvidersUnavailable,
180    /// The provider could not be reached at all: the name did not resolve,
181    /// the connection was refused, or the TLS handshake failed. The network
182    /// or the address is what to check.
183    ProviderUnreachable,
184    /// The provider was reached and did not answer in time. It is up, but
185    /// slow, or the request was large; a resume tries again.
186    ProviderTimedOut,
187    /// The provider was reached and failed: a server error, a reply that
188    /// stopped part-way, or one that could not be read. Nothing about the
189    /// setup is known to be wrong; a resume tries again once it recovers.
190    ProviderFailed,
191}
192
193impl std::fmt::Display for SetupBlocker {
194    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
195        match self {
196            Self::ProviderMissing => f.write_str("provider"),
197            Self::CreditsExhausted => f.write_str("credits"),
198            Self::AuthFailed => f.write_str("key"),
199            Self::Forbidden => f.write_str("access"),
200            Self::ProvidersUnavailable => f.write_str("providers"),
201            Self::ProviderUnreachable => f.write_str("unreachable"),
202            Self::ProviderTimedOut => f.write_str("timed out"),
203            Self::ProviderFailed => f.write_str("failed"),
204        }
205    }
206}
207
208/// What a parked run needs, gathered where the markers are visible.
209#[derive(Debug, Clone, PartialEq, Eq)]
210pub struct SetupNeeded {
211    /// Which kind of problem it is.
212    pub blocker: SetupBlocker,
213    /// What to do about it.
214    pub remedy: String,
215}
216
217impl WaitReason {
218    /// Whether clearing this needs a person. `false` means the run is parked on
219    /// other work and will move on by itself.
220    pub fn needs_a_person(&self) -> bool {
221        !matches!(self, Self::FanOutWorkers { .. } | Self::Children { .. })
222    }
223}
224
225/// A stopwatch that runs only while the thing it measures is actually working.
226///
227/// Wall-clock age and working time are different questions, and the difference
228/// is the whole point of this type: a run paused overnight is twelve hours old
229/// and spent eleven of them doing nothing. Age comes from `started_at`; this is
230/// what a reader means by "how long has this taken".
231///
232/// Kept as banked seconds plus the start of the span in progress, rather than a
233/// single total, so a reader that polls between writes still sees the number
234/// climb instead of stepping once per heartbeat.
235#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
236pub struct ActiveClock {
237    /// Seconds banked from spans that have already ended.
238    #[serde(default)]
239    pub banked_secs: u64,
240    /// When the span in progress began, or `None` while the clock is stopped.
241    #[serde(default)]
242    pub since: Option<i64>,
243}
244
245impl ActiveClock {
246    /// Start or stop the clock to match `running`, banking the span that ends.
247    ///
248    /// Idempotent: called every persist tick with the same answer, it does
249    /// nothing, so only genuine transitions move the accounting.
250    pub fn observe(&mut self, now: i64, running: bool) {
251        match (running, self.since) {
252            (true, None) => self.since = Some(now),
253            (false, Some(started)) => {
254                self.banked_secs += crate::duration::between(started, now);
255                self.since = None;
256            }
257            _ => {}
258        }
259    }
260
261    /// Close the span in progress at `as_of`, banking it.
262    ///
263    /// For a clock read back from disk: whatever the run was doing stopped when
264    /// the daemon holding it did, which is the last moment the record was
265    /// written - not now. Without this, a run reloaded after the daemon was down
266    /// for a day comes back claiming a day's work.
267    pub fn settle(&mut self, as_of: i64) {
268        self.observe(as_of, false);
269    }
270
271    /// Working seconds at `now`, counting the span in progress.
272    pub fn total_secs(&self, now: i64) -> u64 {
273        self.banked_secs + self.since.map_or(0, |s| crate::duration::between(s, now))
274    }
275}
276
277/// Whether a run in this state has its clock running.
278///
279/// It runs while the run is working, and while the run is parked on work of its
280/// own - fan-out workers, sub-agents - because that time is the run taking as
281/// long as it takes. It stops for everything that is not the run's doing:
282/// paused, blocked on a person, parked until the machine is fixed, finished.
283///
284/// A `waiting_input` run with nothing claiming the wait is treated as blocked on
285/// a person, which is the only way it gets there.
286pub fn clock_runs(status: &RunStatus, waiting_on: Option<&WaitReason>) -> bool {
287    match status {
288        RunStatus::Starting | RunStatus::Running => true,
289        RunStatus::WaitingInput => waiting_on.is_some_and(|r| !r.needs_a_person()),
290        RunStatus::Paused
291        | RunStatus::Complete
292        | RunStatus::CompleteInteractive
293        | RunStatus::Error
294        | RunStatus::Cancelled => false,
295    }
296}
297
298impl std::fmt::Display for WaitReason {
299    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
300        match self {
301            Self::ToolApproval => f.write_str("tool approval"),
302            Self::UserPrompt => f.write_str("user prompt"),
303            Self::TaintGate => f.write_str("taint gate"),
304            Self::InteractionPoint => f.write_str("checkpoint"),
305            Self::FanOutWorkers { outstanding } => write!(f, "workers({outstanding})"),
306            Self::Children { outstanding } => write!(f, "children({outstanding})"),
307            // The remedy is a sentence; this is a table cell. The blocker is
308            // the half that fits, and the half that says which screen to open.
309            // The three that describe the provider rather than something the
310            // install lacks say what happened to it.
311            Self::NeedsSetup {
312                blocker:
313                    blocker @ (SetupBlocker::ProviderUnreachable
314                    | SetupBlocker::ProviderTimedOut
315                    | SetupBlocker::ProviderFailed),
316                ..
317            } => write!(f, "provider {blocker}"),
318            Self::NeedsSetup { blocker, .. } => write!(f, "needs {blocker}"),
319        }
320    }
321}
322
323/// The parking markers an agent carries, gathered by whoever can see them.
324///
325/// The live listing reads these straight off the world; the persistence system
326/// reads them off its query. Both then hand them here, so the precedence below
327/// is written once instead of once per surface - two copies of it would
328/// disagree the first time either was edited.
329#[derive(Debug, Clone, Default, PartialEq)]
330pub struct WaitMarkers {
331    /// A taint-gate clearance prompt is outstanding.
332    pub gate_prompt: bool,
333    /// A blueprint stage-boundary checkpoint is holding.
334    pub interaction_point: bool,
335    /// Fan-out workers still to finish, when this run is a fan-out parent.
336    pub fan_out_outstanding: Option<usize>,
337    /// Sub-agents still running, when this run is held for its children.
338    pub children_outstanding: Option<usize>,
339    /// The kind of hub request holding this run, when one is.
340    pub interaction: Option<crate::interaction::InteractionKind>,
341    /// Whether a hub request is holding it at all. Separate from the kind
342    /// because the kind can be unknown while the block is real.
343    pub awaiting_interaction: bool,
344    /// The run is parked until the machine is fixed, and this is what it
345    /// needs.
346    pub needs_setup: Option<SetupNeeded>,
347}
348
349/// Why a parked run is parked, or `None` when it is not parked or nothing has
350/// claimed it.
351///
352/// Order matters, and it is the specific claim first. A taint-gate block and a
353/// stage checkpoint each open a hub request of their own, so both also look
354/// like a generic prompt; asking the specific markers first is what keeps them
355/// from all reporting as one.
356pub fn wait_reason_from(parked: bool, markers: &WaitMarkers) -> Option<WaitReason> {
357    if !parked {
358        return None;
359    }
360    // First, because it outranks everything: a run whose provider is missing
361    // is not going to be unblocked by answering a prompt.
362    if let Some(need) = &markers.needs_setup {
363        return Some(WaitReason::NeedsSetup {
364            blocker: need.blocker,
365            remedy: need.remedy.clone(),
366        });
367    }
368    if markers.gate_prompt {
369        return Some(WaitReason::TaintGate);
370    }
371    if markers.interaction_point {
372        return Some(WaitReason::InteractionPoint);
373    }
374    if let Some(outstanding) = markers.fan_out_outstanding {
375        return Some(WaitReason::FanOutWorkers { outstanding });
376    }
377    if let Some(outstanding) = markers.children_outstanding {
378        return Some(WaitReason::Children { outstanding });
379    }
380    if markers.awaiting_interaction {
381        return Some(match markers.interaction {
382            Some(crate::interaction::InteractionKind::ToolApproval) => WaitReason::ToolApproval,
383            _ => WaitReason::UserPrompt,
384        });
385    }
386    None
387}
388
389/// Metadata for a single background agent run.
390#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
391pub struct RunMeta {
392    /// Identifies the run everywhere, and names its directory under
393    /// `~/.leviath/runs/`. Assigned at spawn and never reused.
394    pub run_id: String,
395    /// The blueprint's `[agent] name`, not the file it was loaded from. Two runs
396    /// of the same agent from different paths share this.
397    pub agent_name: String,
398    /// Absolute path to the agent manifest directory
399    pub agent_path: String,
400    /// The task text the run was started with, verbatim.
401    pub task: String,
402    /// The `provider/model` actually resolved for the entry stage, or `None`
403    /// before resolution. Later stages may use a different one; this is not
404    /// rewritten to follow them.
405    pub model: Option<String>,
406    /// Always 0. There is no worker process per run: the daemon hosts every run
407    /// as an entity in one shared world, so no run has a pid of its own.
408    ///
409    /// Kept because it is written into every `meta.json` there has ever been,
410    /// and served from `GET /api/agents`. Do not key liveness on it. `pid == 0`
411    /// is true of a run that is working, a run that has finished, and a run
412    /// nothing is driving, so a sweeper that reverts on it reverts everything.
413    /// Ask the daemon (`lev ps`) whether it is still hosting the run, and read
414    /// `status` and `last_progress_at` off disk for what became of it.
415    #[serde(default)]
416    pub pid: u32,
417    /// Where the run stands. The durable counterpart to the ECS world's live
418    /// `AgentStatus`, and the one that survives a daemon restart.
419    pub status: RunStatus,
420    /// Name of the stage the run is in, matching a key under `[stages]`.
421    pub current_stage: String,
422    /// Zero-based position of `current_stage` in the blueprint's stage list.
423    /// Not a progress measure: stages can loop and revisit.
424    pub stage_index: usize,
425    /// How many stages the blueprint declares, so a reader can render
426    /// `stage_index` as "3 of 7" without loading the manifest.
427    pub num_stages: usize,
428    /// Inference turns taken in the current stage, reset on entering a new one.
429    /// Compared against the stage's `max_iterations`.
430    pub iteration: usize,
431    /// Cumulative input tokens billed across every inference this run has made,
432    /// including retries.
433    pub prompt_tokens: usize,
434    /// Cumulative output tokens billed across every inference this run has made.
435    pub completion_tokens: usize,
436    /// Cumulative tokens read from provider cache.
437    #[serde(default)]
438    pub cached_tokens: usize,
439    /// Cumulative tokens written to provider cache.
440    #[serde(default)]
441    pub cache_write_tokens: usize,
442    /// Total number of tool calls made across all iterations.
443    #[serde(default)]
444    pub tool_calls: usize,
445    /// What this run has spent, when every call could be priced.
446    ///
447    /// `None` means *unknown*, never free: some call was served by a model with
448    /// no reported cost and no known rates, so any total would be understating
449    /// by an unknown amount. See [`unpriced_calls`](Self::unpriced_calls) for
450    /// how many, and [`cost_is_exact`](Self::cost_is_exact) for whether the
451    /// figure is the provider's own or reconstructed from rate cards.
452    #[serde(default)]
453    pub cost_usd: Option<f64>,
454    /// Calls that could not be priced at all. Non-zero forces
455    /// [`cost_usd`](Self::cost_usd) to `None`.
456    #[serde(default)]
457    pub unpriced_calls: usize,
458    /// Whether every priced call carried the provider's own cost figure rather
459    /// than one computed from published rates. `false` means the total is this
460    /// process's best reconstruction of the invoice, not the invoice.
461    #[serde(default)]
462    pub cost_is_exact: bool,
463    /// The priced subtotal, kept even when `cost_usd` is `None` so a resumed run
464    /// does not restart its accounting from zero.
465    #[serde(default)]
466    pub cost_priced_usd: f64,
467    /// Absolute path to the working directory for tool execution
468    pub workdir: String,
469    /// Unix timestamp (seconds)
470    pub started_at: i64,
471    /// Unix timestamp (seconds)
472    pub updated_at: i64,
473    /// Unix seconds when this run last actually moved: a new iteration, a new
474    /// stage, or a change of status. `None` before the first snapshot lands, and
475    /// on runs written by a daemon older than this field.
476    ///
477    /// Distinct from `updated_at`, which also advances on the 30-second
478    /// persistence heartbeat and so stays fresh on a run that is wedged. A fresh
479    /// `updated_at` is evidence the daemon is alive, and no evidence at all about
480    /// the run. Anything that ages a run must read this instead. Note that a
481    /// daemon restart resets it: a reloaded run really is re-driven from its
482    /// saved context, so it really has just moved.
483    #[serde(default)]
484    pub last_progress_at: Option<i64>,
485    /// How long this run has actually been working, as against how long it has
486    /// existed. See [`ActiveClock`], and read it through
487    /// [`RunMeta::active_runtime_secs`] rather than directly.
488    ///
489    /// `None` means no clock was kept - a run written by a daemon older than
490    /// this field. Deliberately an `Option` rather than a zeroed clock: a run
491    /// that finished inside a second has a genuine total of zero, and the two
492    /// have different right answers.
493    #[serde(default)]
494    pub active: Option<ActiveClock>,
495    /// What went wrong, set alongside [`RunStatus::Error`]. `None` on every
496    /// other status.
497    pub error: Option<String>,
498    /// Short human-readable title generated from the task prompt (None until generated).
499    #[serde(default)]
500    pub title: Option<String>,
501    /// Why [`Self::title`] is still `None`, once title generation has given up
502    /// - the provider it could not reach, or what came back instead of a title.
503    ///
504    /// `None` is the ordinary state: titling has not finished yet, or was never
505    /// asked for. `Some` means it ran and could not produce a name, which is
506    /// otherwise indistinguishable from either.
507    #[serde(default)]
508    pub title_error: Option<String>,
509    /// Custom key-value pairs from the spawn request (API metadata).
510    #[serde(default)]
511    pub metadata: HashMap<String, String>,
512    /// Webhook URL to POST on agent completion/error.
513    #[serde(default)]
514    pub callback_url: Option<String>,
515    /// Optional shared secret used to HMAC-SHA256 sign the webhook body
516    /// (`X-Leviath-Signature` header) so the receiver can verify authenticity.
517    ///
518    /// Persisted, because the daemon must still be able to sign a webhook for a
519    /// run it reloaded after a restart. **Never serve it** - strip it with
520    /// [`RunMeta::redacted`] before any of this struct leaves the process.
521    #[serde(default)]
522    pub callback_secret: Option<String>,
523    /// Links sub-agent runs to their parent run.
524    #[serde(default)]
525    pub parent_run_id: Option<String>,
526    /// Run-ids of this agent's direct sub-agents (sub-agent-tool spawns and
527    /// fan-out workers). Persisted so the daemon can rebuild the exact
528    /// parent→children tree on restart rather than reload children as orphans.
529    #[serde(default)]
530    pub children: Vec<String>,
531    /// This agent's depth in the sub-agent tree (0 for a top-level run).
532    /// Persisted so a reloaded child enforces its remaining spawn-depth budget.
533    #[serde(default)]
534    pub depth: usize,
535    /// The sub-agent depth cap this agent imposes on its own children
536    /// (0 when it has none). Restores `SubAgentChildren::max_child_depth`.
537    #[serde(default)]
538    pub max_child_depth: usize,
539    /// Why this run may have produced nothing useful - see [`RunFlags`].
540    #[serde(default)]
541    pub flags: RunFlags,
542    /// Whether the run was launched unattended (`--yolo`), so a daemon restart
543    /// resumes it the way it was started.
544    ///
545    /// Persisted rather than dropped on reload. Forgetting a launch override
546    /// only ever prompts more, never less, which is why dropping it reads as
547    /// safe; what it actually does is convert an unattended run into one
548    /// parked on a prompt nobody is watching for, discarding consent the
549    /// operator gave at launch. Runs written before this field existed default
550    /// to attended, so nothing is escalated retroactively.
551    #[serde(default)]
552    pub yolo: bool,
553    /// The named yolo profile (`--yolo=<name>`) the run was launched under,
554    /// persisted with `yolo` for the same reason: a restart that dropped the
555    /// name would resume a carefully scoped run under bare `--yolo`, which is
556    /// the escalating direction. Absent for the bare flag and for runs written
557    /// before profiles existed.
558    #[serde(default, skip_serializing_if = "Option::is_none")]
559    pub yolo_profile: Option<String>,
560    /// How much of the blueprint's `[read_paths]` the config granted, as
561    /// resolved at spawn. `None` for a blueprint that declared none, and for
562    /// runs written before this field existed.
563    #[serde(default, skip_serializing_if = "Option::is_none")]
564    pub read_paths: Option<ReadPathGrantCounts>,
565    /// What the agent handed back, if it submitted anything: everything about
566    /// the answer except the bytes.
567    ///
568    /// This is the run's answer, as distinct from `error` (why it failed) and
569    /// from the stage logs (what it did along the way). The content itself is
570    /// in a sidecar file beside this one, because this file is parsed for every
571    /// run on every listing and must stay small no matter how long an answer is.
572    #[serde(default, skip_serializing_if = "Option::is_none")]
573    pub final_output: Option<crate::output::FinalOutputDescriptor>,
574
575    /// Why this run is parked, when it is. `None` on every other status, and
576    /// on a run written before this field existed. Same vocabulary the live
577    /// listing reports, so `lev ps` and a client reading this file describe a
578    /// run the same way.
579    ///
580    /// Additive on purpose: `default` means a `meta.json` from an older build
581    /// still loads, and `skip_serializing_if` means a run that is not parked
582    /// writes exactly the file it wrote before, so an older build reading a
583    /// newer run sees nothing new either.
584    #[serde(default, skip_serializing_if = "Option::is_none")]
585    pub waiting_on: Option<WaitReason>,
586    /// The output shape this run was launched asking for, when the caller
587    /// overrode the blueprint's.
588    ///
589    /// Persisted for the same reason `yolo` is: a daemon restart rebuilds the
590    /// run's spawn arguments from this file, and dropping the request would
591    /// silently revert the run to the blueprint's shape partway through. The
592    /// caller asked once and should not have to ask again.
593    #[serde(default, skip_serializing_if = "Option::is_none")]
594    pub output_request: Option<crate::output::OutputSpec>,
595    /// The `--model` the run was launched with, exactly as given
596    /// (`provider/model` or a bare model), when the caller gave one.
597    ///
598    /// Distinct from `model`, which is what the entry stage *resolved to* and
599    /// is recorded whether or not anything was overridden. A daemon restart
600    /// rebuilds the run's spawn arguments from this file, and must hand back
601    /// this field rather than `model`: handing back `model` pins every stage
602    /// of a run launched with no `--model` to its first stage's provider and
603    /// model, and loses its failover list. This field is what was actually
604    /// asked for, so a reload asks for the same thing - and for a run that
605    /// asked for nothing, resolves each stage afresh, as the launch did.
606    ///
607    /// Runs written before this field existed reload with no override. That
608    /// loses a `--model` given to such a run, which is the smaller harm: the
609    /// stage falls back to its blueprint's list rather than being pinned to a
610    /// pair the user may never have named.
611    #[serde(default, skip_serializing_if = "Option::is_none")]
612    pub model_override: Option<String>,
613}
614
615/// How many `[read_paths]` entries a run's blueprint declared, and how many of
616/// them the user's config actually granted.
617///
618/// Declaring is not granting: an ungranted entry is inert, and the reads it was
619/// meant to allow are refused. Recorded at spawn, because that is when the
620/// policy the run enforces is fixed - editing the config afterwards changes
621/// nothing for a run already in flight.
622#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
623pub struct ReadPathGrantCounts {
624    /// Entries the blueprint declares.
625    pub declared: usize,
626    /// Entries the config grants.
627    pub granted: usize,
628}
629
630/// Post-hoc diagnosis of a run's productivity, persisted in `meta.json` so a
631/// harness (or the dashboard) can tell an empty run from a successful one
632/// without inspecting the workspace or parsing logs.
633///
634/// The motivating failure: 13/300 SWE-bench runs completed their whole stage
635/// pipeline and produced no file changes at all. Nothing on disk said so, or
636/// said why.
637#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
638pub struct RunFlags {
639    /// Paths passed to file-modifying tools that succeeded, in first-touch
640    /// order. Capped at [`MAX_TRACKED_MODIFIED_FILES`]; `modified_file_count`
641    /// keeps the true total.
642    #[serde(default)]
643    pub modified_files: Vec<String>,
644    /// Total successful file-modifying tool calls across the run (uncapped).
645    #[serde(default)]
646    pub modified_file_count: usize,
647    /// The run reached a terminal status having modified nothing, and its
648    /// blueprint gave it a way to modify something. See [`Self::no_output_tools`].
649    #[serde(default)]
650    pub empty_output: bool,
651    /// No stage of the blueprint advertised a file-modifying tool, so this run
652    /// could never have produced the file changes `empty_output` looks for.
653    ///
654    /// Recorded because "modified no files" only diagnoses an agent that was
655    /// supposed to modify files. A router that spawns sub-agents, or an agent
656    /// whose answer is its text, would otherwise report itself empty on every
657    /// successful run. The framework has no basis to judge such a run, so it
658    /// says nothing rather than accusing.
659    ///
660    /// This mirrors the escape the runtime's `gate_blocks` already applies per
661    /// stage: a `require_modifications` gate on a stage that advertises no
662    /// modifying tool is skipped, because it could never pass.
663    ///
664    /// Phrased negatively so the `false` that [`Default`] and `serde(default)`
665    /// produce means "was capable" - the behavior every `meta.json` written
666    /// before this field had.
667    #[serde(default)]
668    pub no_output_tools: bool,
669    /// `web_search` calls this run made, across every stage.
670    ///
671    /// Zero for an agent that never had the tool, which is most of them - the
672    /// pair says nothing about a run that could not have searched, the same
673    /// escape [`Self::no_output_tools`] applies to file modifications.
674    #[serde(default)]
675    pub searches_run: usize,
676    /// Of those, how many came back with nothing usable: no results, or a
677    /// diagnostic saying the search could not run.
678    ///
679    /// A research run whose searches all came back empty still finishes
680    /// `complete` and still writes a confident, fully cited report, because a
681    /// model handed an empty result set fills the gap from its training data
682    /// and cites what it remembers. One did exactly that across 47 consecutive
683    /// failed searches - no engine was configured - and nothing on disk
684    /// recorded that the report rested on nothing.
685    ///
686    /// `searches_empty == searches_run` with `searches_run > 0` is that run.
687    #[serde(default)]
688    pub searches_empty: usize,
689    /// How many stages exhausted their `max_iterations`.
690    #[serde(default)]
691    pub max_iterations_hit: usize,
692    /// How many transitions proceeded past an unsatisfied gate because the
693    /// gate's re-run budget ran out.
694    #[serde(default)]
695    pub gates_forced: usize,
696    /// Regions declared `required` that a stage gave up on and that are **still
697    /// empty**, in the order they were abandoned.
698    ///
699    /// The mechanism re-runs the stage a bounded number of times and then
700    /// proceeds with a log line, which nothing downstream reads: a run whose
701    /// agent wrote its plan and a run where we asked twice and moved on both
702    /// finished `complete`, with the second silently missing the artifact every
703    /// later stage's prompt says to work from. Names rather than a count
704    /// because knowing *which* region was abandoned is what makes it
705    /// actionable, and a run cannot abandon many.
706    ///
707    /// A later stage can fill a region an earlier one gave up on, and when that
708    /// happens the name is dropped from here - the artifact exists, so a reader
709    /// told it is missing would be told something false. A `deep-researcher` run
710    /// abandoned `sources_index` in `gather`, `analyze` wrote it, and the run
711    /// finished with a fifty-citation bibliography while still reporting the
712    /// region as never written. The moment is kept in the log; this field
713    /// answers "what is actually missing", which is the question a consumer is
714    /// asking when it renders a warning.
715    #[serde(default)]
716    pub required_regions_abandoned: Vec<String>,
717    /// The working directory disappeared mid-run.
718    #[serde(default)]
719    pub workspace_lost: bool,
720    /// The run submitted a final output.
721    ///
722    /// Counts as having produced something, alongside file modifications.
723    /// Without this an agent whose whole deliverable is its answer - a
724    /// researcher, a reviewer, a router - reported itself empty on every
725    /// successful run, which is the same mistake [`Self::no_output_tools`] was
726    /// added to correct from the other direction.
727    #[serde(default)]
728    pub produced_output: bool,
729    /// How many stages transitioned without the final output they required,
730    /// because the re-run budget ran out.
731    ///
732    /// The counterpart to [`Self::gates_forced`]: the run finished, and this
733    /// says the answer it hands back may be missing.
734    #[serde(default)]
735    pub output_forced: usize,
736    /// How many fan-out splits were unusable and were degraded to an empty
737    /// fan-out because the blueprint declared no `error` and no `dead_end`
738    /// escape from the stage.
739    ///
740    /// Ending the run on a split that cannot be parsed throws away everything
741    /// the parent has already done, workers finished and later stages still
742    /// pending. The stage moves on instead, and this is what says the fan-out
743    /// it moved on from produced nothing. Non-zero means the merge stage
744    /// worked from less than it was meant to.
745    #[serde(default)]
746    pub splits_degraded: usize,
747
748    /// Rhai scripts this run needed that could not be used, by name.
749    ///
750    /// A script that will not compile, or that throws where the runtime has to
751    /// carry on regardless, is skipped rather than fatal - which is the right
752    /// call and used to be completely silent; this list is the trace. An
753    /// output validator that cannot run is the exception: by default the
754    /// submission is rejected and the script's own error goes back to the
755    /// model as retry feedback, while `on_validator_error = "accept"` records
756    /// the submission unchecked and leaves this list as the only trace of an
757    /// answer nothing checked. The script is named here in both modes.
758    ///
759    /// Named rather than counted, because the useful question is which one -
760    /// the answer tells you which file to open.
761    #[serde(default, skip_serializing_if = "Vec::is_empty")]
762    pub broken_scripts: Vec<String>,
763}
764
765/// How many distinct modified paths [`RunFlags`] records before it stops
766/// growing (the count keeps rising). Bounds `meta.json` for a long run.
767pub const MAX_TRACKED_MODIFIED_FILES: usize = 200;
768
769impl RunFlags {
770    /// Record a successful modifying tool call on `path`.
771    pub fn record_modification(&mut self, path: &str) {
772        self.modified_file_count += 1;
773        if self.modified_files.len() < MAX_TRACKED_MODIFIED_FILES
774            && !self.modified_files.iter().any(|p| p == path)
775        {
776            self.modified_files.push(path.to_string());
777        }
778    }
779
780    /// Note a path that changed on disk without a modifying tool naming it -
781    /// a file a `shell` command created or rewrote, found by scanning the
782    /// working directory. It joins the list (deduped, capped) but does not
783    /// touch `modified_file_count`, which counts modifying *tool calls*: a
784    /// shell call is not one, and a scan that ran twice must not double-count
785    /// the same file. Returns whether the path was newly added.
786    pub fn note_modified_path(&mut self, path: &str) -> bool {
787        if self.modified_files.len() >= MAX_TRACKED_MODIFIED_FILES
788            || self.modified_files.iter().any(|p| p == path)
789        {
790            return false;
791        }
792        self.modified_files.push(path.to_string());
793        true
794    }
795}
796
797impl RunMeta {
798    /// This run's metadata with the webhook signing secret removed, for anything
799    /// that leaves the process.
800    ///
801    /// `GET /api/agents`, `/api/agents/{id}` and `/api/agents/{id}/children`
802    /// all serialize `RunMeta` whole, so without this any holder of the API
803    /// token reads every run's `callback_secret` - the key that authenticates
804    /// Leviath's webhooks to their receivers. Mirrors the `RedactedConfig`
805    /// pattern the `/api/config` handler uses.
806    ///
807    /// Returns an owned copy rather than mutating in place so a caller cannot
808    /// accidentally redact the record the daemon still needs for signing.
809    #[must_use]
810    pub fn redacted(&self) -> Self {
811        Self {
812            callback_secret: None,
813            ..self.clone()
814        }
815    }
816
817    /// A newly accepted run: [`RunStatus::Starting`], both timestamps now, every
818    /// counter at zero and every optional field unset.
819    ///
820    /// Only the seven values a caller genuinely knows at spawn are parameters.
821    /// Everything else is filled in by the daemon as the run proceeds, so taking
822    /// them here would invite a caller to invent a stage or a token count.
823    pub fn new(
824        run_id: String,
825        agent_name: String,
826        agent_path: String,
827        task: String,
828        model: Option<String>,
829        workdir: String,
830        num_stages: usize,
831    ) -> Self {
832        let now = crate::duration::now_secs();
833        Self {
834            run_id,
835            agent_name,
836            agent_path,
837            task,
838            model,
839            pid: 0,
840            status: RunStatus::Starting,
841            current_stage: String::new(),
842            stage_index: 0,
843            num_stages,
844            iteration: 0,
845            prompt_tokens: 0,
846            completion_tokens: 0,
847            cached_tokens: 0,
848            cache_write_tokens: 0,
849            tool_calls: 0,
850            // A run that has made no calls has spent nothing, and that
851            // zero IS known - unlike a run whose calls could not be priced.
852            cost_usd: Some(0.0),
853            unpriced_calls: 0,
854            cost_is_exact: true,
855            cost_priced_usd: 0.0,
856            workdir,
857            started_at: now,
858            updated_at: now,
859            last_progress_at: None,
860            active: None,
861            error: None,
862            title: None,
863            title_error: None,
864            metadata: HashMap::new(),
865            callback_url: None,
866            callback_secret: None,
867            parent_run_id: None,
868            children: Vec::new(),
869            depth: 0,
870            max_child_depth: 0,
871            final_output: None,
872            waiting_on: None,
873            output_request: None,
874            model_override: None,
875            flags: RunFlags::default(),
876            yolo: false,
877            yolo_profile: None,
878            read_paths: None,
879        }
880    }
881
882    /// How long ago this run was launched, at `now`.
883    ///
884    /// One of the three spans a reader can ask a run about, and they answer
885    /// different questions - keep them apart:
886    ///
887    /// - **age** ([`Self::age_secs`]): how long since it was launched. Says
888    ///   nothing about whether it has done anything.
889    /// - **working** ([`Self::active_runtime_secs`]): how long it actually spent
890    ///   working. This is the one to call a run's duration.
891    /// - **last moved** (from [`Self::last_progress_at`]): how long since it made
892    ///   progress. A health signal, not a duration: it is how a wedged run is
893    ///   told from a slow one. No accessor, because the surfaces that show it
894    ///   read it off a live listing row rather than off a `RunMeta`.
895    ///
896    /// Every surface reads these rather than doing the arithmetic itself, so
897    /// `lev ps`, the dashboard and the HTTP API cannot disagree about what a run
898    /// has been doing.
899    pub fn age_secs(&self, now: i64) -> u64 {
900        crate::duration::between(self.started_at, now)
901    }
902
903    /// How long this run has actually been working, at `now`.
904    ///
905    /// This is the number to show as a run's duration. Wall-clock age answers a
906    /// different question, and answers it misleadingly: a run left paused, or
907    /// sitting on a question nobody has answered, kept climbing while nothing
908    /// was happening on its behalf. See the sibling spans on [`Self::age_secs`].
909    ///
910    /// A run written before the clock existed carries no spans at all, so it
911    /// falls back to the wall-clock span - a finished run that claims to have
912    /// taken no time is the worse answer of the two.
913    pub fn active_runtime_secs(&self, now: i64) -> u64 {
914        match self.active {
915            Some(clock) => clock.total_secs(now),
916            None => crate::duration::between(self.started_at, self.updated_at),
917        }
918    }
919
920    /// Stamp `updated_at` with the current time.
921    ///
922    /// Deliberately does **not** touch `last_progress_at`: the 30-second
923    /// persistence heartbeat calls this, and a run that is wedged must not look
924    /// like one that just moved. See [`RunMeta::last_progress_at`].
925    pub fn touch(&mut self) {
926        self.updated_at = crate::duration::now_secs();
927    }
928}
929
930/// One content entry within a region, captured at snapshot time.
931#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
932pub struct RegionEntrySnapshot {
933    /// The entry's parts, exactly as they sat in the live region. Reads as
934    /// text; a plain string in an older snapshot loads as one text part.
935    pub content: crate::region::EntryContent,
936    /// The entry's token cost as counted when it was added, carried through the
937    /// snapshot so a reload does not have to re-tokenize to rebuild budgets.
938    pub tokens: usize,
939    /// The entry's role/kind, so a snapshot round-trips faithfully when the
940    /// daemon reloads it on restart. Defaults to `Text` for older snapshots.
941    #[serde(default)]
942    pub kind: crate::region::EntryKind,
943    /// Free-form structured data an entry writer attached, passed through
944    /// untouched. Nothing in the engine interprets it.
945    #[serde(default, skip_serializing_if = "Option::is_none")]
946    pub metadata: Option<serde_json::Value>,
947    /// Key for HashMap region entries (file paths, section names, etc.)
948    #[serde(default, skip_serializing_if = "Option::is_none")]
949    pub key: Option<String>,
950    /// How sensitive this entry is.
951    ///
952    /// Persisted because taint was not, and a restore that dropped it silently
953    /// disarmed the gate: the reloaded run re-enabled taint tracking, found
954    /// every region back at `Public`, and let outbound tools through that had
955    /// been blocked a moment earlier. Any restart, crash-recovery, `resume`, or
956    /// page-in did it.
957    ///
958    /// Defaults to `Public` for snapshots written before this field existed -
959    /// the same value they were being restored with anyway, so nothing is worse
960    /// than it was, and new runs are correct from their first write.
961    #[serde(default)]
962    pub taint: crate::taint::TaintLevel,
963    /// The opaque provider token this turn has to be replayed with.
964    ///
965    /// Persisted for the same reason `taint` is: a restore that dropped it
966    /// would silently break reasoning continuity on a stateless backend, and
967    /// the run would look fine while paying to re-derive its chain of thought
968    /// every turn. Defaults to absent for snapshots written before the field,
969    /// which is what they were being restored with anyway.
970    #[serde(default, skip_serializing_if = "Option::is_none")]
971    pub reasoning: Option<String>,
972}
973
974/// Per-region token snapshot written by the background worker after each inference.
975#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
976pub struct RegionSnapshot {
977    /// The region's name, matching its key under `[context.regions]`.
978    pub name: String,
979    /// Stringified kind, spelled the way the blueprint spells it: `pinned`,
980    /// `temporary`, `clearable`, `sliding_window`, `compacting`,
981    /// `compact_history`, `hashmap`, `checklist`, `custom`.
982    ///
983    /// A snapshot written by an older build says `sliding` and `history` for
984    /// those two, and those files stay on disk, so a reader that renders this
985    /// accepts both spellings.
986    pub kind: String,
987    /// Tokens the region held when the snapshot was taken.
988    pub current_tokens: usize,
989    /// The region's ceiling at snapshot time, already resolved against the
990    /// model in front of it, so a percentage budget appears here as a number.
991    pub max_tokens: usize,
992    /// Actual content entries stored in this region (empty for zero-token regions).
993    #[serde(default, skip_serializing_if = "Vec::is_empty")]
994    pub entries: Vec<RegionEntrySnapshot>,
995    /// What the blueprint says this region is for, when it says.
996    ///
997    /// Carried on the snapshot so every reader of `context.json` can show it -
998    /// the dashboard, the history API, a console - rather than each having to
999    /// find and re-parse the manifest to explain a region it is already
1000    /// displaying.
1001    #[serde(default, skip_serializing_if = "Option::is_none")]
1002    pub description: Option<String>,
1003}
1004
1005/// Snapshot of the full context window, written to `context.json` alongside `meta.json`.
1006#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
1007pub struct ContextSnapshot {
1008    /// The stage the run was in when this was written.
1009    pub stage_name: String,
1010    /// Tokens held across every region, which is what the next request costs
1011    /// before the model's reply.
1012    pub total_tokens: usize,
1013    /// The whole window's budget, from the blueprint's `total_budget_tokens` or
1014    /// the model's own limit.
1015    pub max_tokens: usize,
1016    /// Every region, in layout order.
1017    pub regions: Vec<RegionSnapshot>,
1018}
1019
1020/// Current Unix time in seconds (saturating to 0 before the epoch).
1021#[cfg(test)]
1022mod tests {
1023    use super::*;
1024
1025    #[test]
1026    fn active_clock_banks_only_the_spans_it_was_running_for() {
1027        let mut clock = ActiveClock::default();
1028        clock.observe(100, true);
1029        assert_eq!(clock.since, Some(100));
1030        assert_eq!(clock.total_secs(140), 40, "the open span counts");
1031
1032        // Repeating the same answer is a no-op, so a per-tick call cannot
1033        // restart the span and lose the time already in it.
1034        clock.observe(120, true);
1035        assert_eq!(clock.since, Some(100));
1036
1037        clock.observe(140, false);
1038        assert_eq!(clock.banked_secs, 40);
1039        assert_eq!(clock.since, None);
1040        // Stopped: the reading no longer moves however long the caller waits.
1041        assert_eq!(clock.total_secs(9_999), 40);
1042
1043        clock.observe(1_000, true);
1044        assert_eq!(clock.total_secs(1_010), 50, "a resume adds to the bank");
1045    }
1046
1047    #[test]
1048    fn active_clock_ignores_a_clock_that_runs_backwards() {
1049        let mut clock = ActiveClock {
1050            banked_secs: 5,
1051            since: Some(500),
1052        };
1053        // An `as_of` before the span began (a corrected system clock) banks
1054        // nothing rather than wrapping the unsigned total.
1055        clock.settle(400);
1056        assert_eq!(clock.banked_secs, 5);
1057        assert_eq!(clock.since, None);
1058        assert_eq!(clock.total_secs(0), 5);
1059    }
1060
1061    #[test]
1062    fn settle_closes_an_open_span_at_the_moment_given() {
1063        let mut clock = ActiveClock {
1064            banked_secs: 10,
1065            since: Some(100),
1066        };
1067        // The daemon died at 130 and the run is reloaded much later: only the
1068        // 30 seconds it was actually up count.
1069        clock.settle(130);
1070        assert_eq!(clock.banked_secs, 40);
1071        assert_eq!(clock.total_secs(1_000_000), 40);
1072    }
1073
1074    #[test]
1075    fn clock_runs_while_working_or_held_for_the_runs_own_children() {
1076        // Working.
1077        assert!(clock_runs(&RunStatus::Starting, None));
1078        assert!(clock_runs(&RunStatus::Running, None));
1079        // Parked on work of its own: still the run taking as long as it takes.
1080        assert!(clock_runs(
1081            &RunStatus::WaitingInput,
1082            Some(&WaitReason::Children { outstanding: 2 })
1083        ));
1084        assert!(clock_runs(
1085            &RunStatus::WaitingInput,
1086            Some(&WaitReason::FanOutWorkers { outstanding: 5 })
1087        ));
1088        // Waiting on a person, in each of the ways it can happen.
1089        for reason in [
1090            WaitReason::ToolApproval,
1091            WaitReason::UserPrompt,
1092            WaitReason::TaintGate,
1093            WaitReason::InteractionPoint,
1094            WaitReason::NeedsSetup {
1095                blocker: SetupBlocker::CreditsExhausted,
1096                remedy: "top up".to_string(),
1097            },
1098        ] {
1099            assert!(
1100                !clock_runs(&RunStatus::WaitingInput, Some(&reason)),
1101                "{reason} should stop the clock"
1102            );
1103        }
1104        // An unclaimed wait is a wait: nothing is driving the run.
1105        assert!(!clock_runs(&RunStatus::WaitingInput, None));
1106        // Paused and every terminal state.
1107        for status in [
1108            RunStatus::Paused,
1109            RunStatus::Complete,
1110            RunStatus::CompleteInteractive,
1111            RunStatus::Error,
1112            RunStatus::Cancelled,
1113        ] {
1114            assert!(!clock_runs(&status, None), "{status} should stop the clock");
1115        }
1116    }
1117
1118    /// Age and working time are different questions, and a paused run is where
1119    /// they come apart.
1120    #[test]
1121    fn age_is_the_wall_clock_span_whatever_the_run_was_doing() {
1122        let mut meta = sample_meta();
1123        meta.started_at = 1_000;
1124        meta.active = Some(ActiveClock {
1125            banked_secs: 20,
1126            since: None,
1127        });
1128        assert_eq!(meta.age_secs(4_600), 3_600, "an hour old");
1129        assert_eq!(
1130            meta.active_runtime_secs(4_600),
1131            20,
1132            "twenty seconds of work"
1133        );
1134        // A clock corrected backwards reads as brand new, not as a huge age.
1135        assert_eq!(meta.age_secs(500), 0);
1136    }
1137
1138    #[test]
1139    fn active_runtime_secs_falls_back_to_the_wall_clock_span_when_unrecorded() {
1140        // A run written by a build that kept no clock: the wall-clock span is
1141        // a better answer than "this run took no time".
1142        let mut meta = sample_meta();
1143        meta.started_at = 1_000;
1144        meta.updated_at = 1_600;
1145        assert_eq!(meta.active_runtime_secs(9_000), 600);
1146
1147        // Once the clock exists it is the authority, and the wall-clock span
1148        // stops mattering.
1149        meta.active = Some(ActiveClock {
1150            banked_secs: 20,
1151            since: None,
1152        });
1153        assert_eq!(meta.active_runtime_secs(9_000), 20);
1154
1155        // And a run that genuinely took no time reports zero, rather than
1156        // falling back to a span that would invent one.
1157        meta.active = Some(ActiveClock::default());
1158        assert_eq!(meta.active_runtime_secs(9_000), 0);
1159    }
1160
1161    #[test]
1162    fn stage_active_runtime_secs_falls_back_the_same_way() {
1163        let mut rec = StageRecord::new("plan".to_string(), 0);
1164        // Never entered: no time, and no timestamp to guess from.
1165        assert_eq!(rec.active_runtime_secs(500), 0);
1166
1167        rec.started_at = Some(100);
1168        assert_eq!(rec.active_runtime_secs(160), 60, "still running: up to now");
1169        rec.ended_at = Some(130);
1170        assert_eq!(rec.active_runtime_secs(160), 30, "finished: up to the end");
1171
1172        rec.active = Some(ActiveClock {
1173            banked_secs: 7,
1174            since: None,
1175        });
1176        assert_eq!(rec.active_runtime_secs(160), 7);
1177    }
1178
1179    fn sample_meta() -> RunMeta {
1180        RunMeta::new(
1181            "run-1".to_string(),
1182            "agent".to_string(),
1183            "/agents/agent".to_string(),
1184            "do the thing".to_string(),
1185            Some("claude-sonnet-4-6".to_string()),
1186            "/work".to_string(),
1187            3,
1188        )
1189    }
1190
1191    /// `wire()` has to say exactly what serde says, because the two spellings
1192    /// reach the same client from different routes - one from a whole run
1193    /// serialized as JSON, the other from a route that builds its own `status`
1194    /// string. Derived from serde here rather than compared to a second hand
1195    /// written list, so a renamed variant fails this instead of shipping two
1196    /// words for one state.
1197    #[test]
1198    fn the_wire_word_is_the_word_serde_writes() {
1199        for status in [
1200            RunStatus::Starting,
1201            RunStatus::Running,
1202            RunStatus::WaitingInput,
1203            RunStatus::Complete,
1204            RunStatus::CompleteInteractive,
1205            RunStatus::Paused,
1206            RunStatus::Error,
1207            RunStatus::Cancelled,
1208        ] {
1209            let serde_word = serde_json::to_value(&status).unwrap();
1210            assert_eq!(serde_word, serde_json::json!(status.wire()));
1211        }
1212    }
1213
1214    /// The webhook signing secret must not survive into anything served over
1215    /// the API - an unredacted meta lets `GET /api/agents` hand it to any
1216    /// token holder.
1217    #[test]
1218    fn redacted_drops_the_callback_secret_and_keeps_everything_else() {
1219        let mut m = sample_meta();
1220        m.callback_secret = Some("shhh".to_string());
1221        m.callback_url = Some("https://example.com/hook".to_string());
1222
1223        let r = m.redacted();
1224        assert_eq!(r.callback_secret, None);
1225        // The URL is not a secret and stays: a caller needs to see where its own
1226        // webhook was pointed.
1227        assert_eq!(r.callback_url.as_deref(), Some("https://example.com/hook"));
1228        assert_eq!(r.run_id, m.run_id);
1229        assert_eq!(r.task, m.task);
1230
1231        // Serializing the redacted form must not mention it at all - a `None`
1232        // that still emitted `"callback_secret": null` would be fine, but an
1233        // assertion on the wire format is what a reviewer actually checks.
1234        let json = serde_json::to_string(&r).unwrap();
1235        assert!(!json.contains("shhh"), "{json}");
1236
1237        // ...and the original is untouched, because the daemon still needs it to
1238        // sign the webhook for a run it reloaded after a restart.
1239        assert_eq!(m.callback_secret.as_deref(), Some("shhh"));
1240    }
1241
1242    #[test]
1243    fn run_meta_new_sets_defaults() {
1244        let m = sample_meta();
1245        assert_eq!(m.run_id, "run-1");
1246        assert_eq!(m.agent_name, "agent");
1247        assert_eq!(m.agent_path, "/agents/agent");
1248        assert_eq!(m.task, "do the thing");
1249        assert_eq!(m.model.as_deref(), Some("claude-sonnet-4-6"));
1250        assert_eq!(m.workdir, "/work");
1251        assert_eq!(m.num_stages, 3);
1252        assert_eq!(m.pid, 0);
1253        assert_eq!(m.status, RunStatus::Starting);
1254        assert_eq!(m.stage_index, 0);
1255        assert_eq!(m.iteration, 0);
1256        assert_eq!(m.prompt_tokens, 0);
1257        assert_eq!(m.completion_tokens, 0);
1258        assert_eq!(m.cached_tokens, 0);
1259        assert_eq!(m.cache_write_tokens, 0);
1260        assert_eq!(m.tool_calls, 0);
1261        assert!(m.error.is_none());
1262        assert!(m.title.is_none());
1263        assert!(m.metadata.is_empty());
1264        assert!(m.callback_url.is_none());
1265        assert!(m.callback_secret.is_none());
1266        assert!(m.parent_run_id.is_none());
1267        assert!(m.children.is_empty());
1268        assert_eq!(m.depth, 0);
1269        assert_eq!(m.max_child_depth, 0);
1270        assert!(m.current_stage.is_empty());
1271        assert_eq!(m.started_at, m.updated_at);
1272    }
1273
1274    #[test]
1275    fn run_meta_touch_advances_updated_at() {
1276        let mut m = sample_meta();
1277        m.updated_at = 0;
1278        m.touch();
1279        assert!(m.updated_at > 0);
1280    }
1281
1282    /// A `meta.json` written before `waiting_on` existed still loads.
1283    ///
1284    /// This is the whole compatibility question for the field, and it is worth
1285    /// a test rather than a reading of the serde attributes: every run already
1286    /// on disk was written by a build that had never heard of it, and a
1287    /// deserialize that insisted on the key would make every one of them
1288    /// unreadable.
1289    #[test]
1290    fn a_run_written_before_waiting_on_existed_still_loads() {
1291        let mut original = sample_meta();
1292        original.status = RunStatus::WaitingInput;
1293        let mut value = serde_json::to_value(&original).unwrap();
1294        // Whatever the current build writes, an older file simply has no such
1295        // key. Removing it reproduces that exactly.
1296        value
1297            .as_object_mut()
1298            .expect("meta is an object")
1299            .remove("waiting_on");
1300        assert!(value.get("waiting_on").is_none(), "the old shape");
1301
1302        let back: RunMeta = serde_json::from_value(value).unwrap();
1303        assert_eq!(back.waiting_on, None);
1304        assert_eq!(back.status, RunStatus::WaitingInput);
1305        assert_eq!(back.run_id, original.run_id);
1306    }
1307
1308    /// A run that is not parked writes the file it always wrote, so an older
1309    /// build reading a newer run sees nothing it does not understand.
1310    #[test]
1311    fn a_run_that_is_not_parked_writes_no_waiting_on_key() {
1312        let mut m = sample_meta();
1313        m.status = RunStatus::Running;
1314        m.waiting_on = None;
1315        let json = serde_json::to_value(&m).unwrap();
1316        assert!(json.get("waiting_on").is_none(), "{json}");
1317
1318        m.waiting_on = Some(WaitReason::FanOutWorkers { outstanding: 3 });
1319        let json = serde_json::to_value(&m).unwrap();
1320        assert_eq!(
1321            json["waiting_on"],
1322            serde_json::json!({"reason": "fan_out_workers", "outstanding": 3})
1323        );
1324    }
1325
1326    /// Every variant is on the wire in snake_case, the way `RunStatus` is, and
1327    /// round-trips. The counted ones carry their number with them, which is
1328    /// what lets a client say "waiting on 3 of them" rather than "waiting".
1329    #[test]
1330    fn wait_reason_serializes_in_snake_case() {
1331        for (variant, wire) in [
1332            (WaitReason::ToolApproval, "tool_approval"),
1333            (WaitReason::UserPrompt, "user_prompt"),
1334            (WaitReason::TaintGate, "taint_gate"),
1335            (WaitReason::InteractionPoint, "interaction_point"),
1336            (
1337                WaitReason::FanOutWorkers { outstanding: 3 },
1338                "fan_out_workers",
1339            ),
1340            (WaitReason::Children { outstanding: 1 }, "children"),
1341        ] {
1342            let json = serde_json::to_value(&variant).unwrap();
1343            assert_eq!(json["reason"], serde_json::json!(wire));
1344            let back: WaitReason = serde_json::from_value(json).unwrap();
1345            assert_eq!(back, variant);
1346        }
1347    }
1348
1349    /// Each marker names its own reason, and the counted ones carry the count.
1350    ///
1351    /// The precedence is the specific claim first: a taint gate and a
1352    /// checkpoint each open a hub request of their own, so a generic-prompt
1353    /// answer would swallow both.
1354    #[test]
1355    fn each_marker_names_its_own_reason_specific_first() {
1356        let cases = [
1357            (
1358                WaitMarkers {
1359                    gate_prompt: true,
1360                    awaiting_interaction: true,
1361                    ..Default::default()
1362                },
1363                WaitReason::TaintGate,
1364            ),
1365            (
1366                WaitMarkers {
1367                    interaction_point: true,
1368                    awaiting_interaction: true,
1369                    ..Default::default()
1370                },
1371                WaitReason::InteractionPoint,
1372            ),
1373            (
1374                WaitMarkers {
1375                    fan_out_outstanding: Some(5),
1376                    ..Default::default()
1377                },
1378                WaitReason::FanOutWorkers { outstanding: 5 },
1379            ),
1380            (
1381                WaitMarkers {
1382                    children_outstanding: Some(4),
1383                    ..Default::default()
1384                },
1385                WaitReason::Children { outstanding: 4 },
1386            ),
1387        ];
1388        for (markers, expected) in cases {
1389            assert_eq!(
1390                wait_reason_from(true, &markers),
1391                Some(expected),
1392                "{markers:?}"
1393            );
1394        }
1395        // A parent holding both kinds of sub-work reports the more specific one.
1396        assert_eq!(
1397            wait_reason_from(
1398                true,
1399                &WaitMarkers {
1400                    fan_out_outstanding: Some(2),
1401                    children_outstanding: Some(9),
1402                    ..Default::default()
1403                }
1404            ),
1405            Some(WaitReason::FanOutWorkers { outstanding: 2 })
1406        );
1407    }
1408
1409    /// A run parked until the machine is fixed says so before anything else.
1410    ///
1411    /// It outranks every other marker on purpose: answering a prompt does not
1412    /// help a run whose provider is not configured, so sending someone to the
1413    /// prompt would be sending them to the wrong screen.
1414    #[test]
1415    fn needing_setup_outranks_every_other_reason() {
1416        let need = SetupNeeded {
1417            blocker: SetupBlocker::ProviderMissing,
1418            remedy: "add it to config.toml".to_string(),
1419        };
1420        let reason = wait_reason_from(
1421            true,
1422            &WaitMarkers {
1423                needs_setup: Some(need.clone()),
1424                // Everything else at once, so precedence is being tested
1425                // rather than the absence of competition.
1426                gate_prompt: true,
1427                interaction_point: true,
1428                fan_out_outstanding: Some(2),
1429                children_outstanding: Some(3),
1430                awaiting_interaction: true,
1431                interaction: Some(crate::interaction::InteractionKind::ToolApproval),
1432            },
1433        );
1434        assert_eq!(
1435            reason,
1436            Some(WaitReason::NeedsSetup {
1437                blocker: SetupBlocker::ProviderMissing,
1438                remedy: "add it to config.toml".to_string(),
1439            })
1440        );
1441        assert!(
1442            reason.unwrap().needs_a_person(),
1443            "nothing resolves this without somebody"
1444        );
1445    }
1446
1447    /// Each blocker is its own value on the wire, so a console can offer the
1448    /// right remedy instead of matching on the sentence.
1449    #[test]
1450    fn every_blocker_has_its_own_wire_name_and_label() {
1451        for (blocker, wire, label) in [
1452            (
1453                SetupBlocker::ProviderMissing,
1454                "provider_missing",
1455                "provider",
1456            ),
1457            (
1458                SetupBlocker::CreditsExhausted,
1459                "credits_exhausted",
1460                "credits",
1461            ),
1462            (SetupBlocker::AuthFailed, "auth_failed", "key"),
1463            (SetupBlocker::Forbidden, "forbidden", "access"),
1464            (
1465                SetupBlocker::ProvidersUnavailable,
1466                "providers_unavailable",
1467                "providers",
1468            ),
1469            (
1470                SetupBlocker::ProviderUnreachable,
1471                "provider_unreachable",
1472                "unreachable",
1473            ),
1474            (
1475                SetupBlocker::ProviderTimedOut,
1476                "provider_timed_out",
1477                "timed out",
1478            ),
1479            (SetupBlocker::ProviderFailed, "provider_failed", "failed"),
1480        ] {
1481            assert_eq!(serde_json::to_value(blocker).unwrap(), wire);
1482            assert_eq!(blocker.to_string(), label);
1483            let back: SetupBlocker = serde_json::from_value(serde_json::json!(wire)).unwrap();
1484            assert_eq!(back, blocker);
1485            // The row renders the kind, not the sentence: a remedy is a
1486            // sentence and this is a table cell. A blocker that describes the
1487            // provider says what happened to it rather than what is needed.
1488            let lead = match blocker {
1489                SetupBlocker::ProviderUnreachable
1490                | SetupBlocker::ProviderTimedOut
1491                | SetupBlocker::ProviderFailed => "provider",
1492                _ => "needs",
1493            };
1494            assert_eq!(
1495                WaitReason::NeedsSetup {
1496                    blocker,
1497                    remedy: "a whole sentence that would not fit".to_string(),
1498                }
1499                .to_string(),
1500                format!("{lead} {label}")
1501            );
1502        }
1503    }
1504
1505    /// A generic hub block reports what kind of prompt it is, so "approve this
1506    /// tool call" and "answer this question" are not the same row.
1507    #[test]
1508    fn a_hub_block_reports_the_kind_of_prompt_holding_it() {
1509        let held = |kind| WaitMarkers {
1510            awaiting_interaction: true,
1511            interaction: kind,
1512            ..Default::default()
1513        };
1514        assert_eq!(
1515            wait_reason_from(
1516                true,
1517                &held(Some(crate::interaction::InteractionKind::ToolApproval))
1518            ),
1519            Some(WaitReason::ToolApproval)
1520        );
1521        // Anything else the agent asked for is a question for a person. The
1522        // kind can also be unknown while the block is real, which reads the
1523        // same way: somebody is being waited on.
1524        assert_eq!(
1525            wait_reason_from(
1526                true,
1527                &held(Some(crate::interaction::InteractionKind::FreeText))
1528            ),
1529            Some(WaitReason::UserPrompt)
1530        );
1531        assert_eq!(
1532            wait_reason_from(true, &held(None)),
1533            Some(WaitReason::UserPrompt)
1534        );
1535    }
1536
1537    /// Parked with nothing claiming it: the field is left off rather than
1538    /// filled with a guess, and a run that is not parked never has one.
1539    #[test]
1540    fn nothing_claiming_a_parked_run_reports_no_reason() {
1541        assert_eq!(wait_reason_from(true, &WaitMarkers::default()), None);
1542        assert_eq!(
1543            wait_reason_from(
1544                false,
1545                &WaitMarkers {
1546                    gate_prompt: true,
1547                    ..Default::default()
1548                }
1549            ),
1550            None,
1551            "a run that is not waiting is not waiting on anything"
1552        );
1553    }
1554
1555    /// The rendered form every text surface uses, counts included. Narrow
1556    /// enough for a table column, which is why it is not the variant name.
1557    #[test]
1558    fn every_reason_renders_for_a_narrow_column() {
1559        assert_eq!(WaitReason::ToolApproval.to_string(), "tool approval");
1560        assert_eq!(WaitReason::UserPrompt.to_string(), "user prompt");
1561        assert_eq!(WaitReason::TaintGate.to_string(), "taint gate");
1562        assert_eq!(WaitReason::InteractionPoint.to_string(), "checkpoint");
1563        assert_eq!(
1564            WaitReason::FanOutWorkers { outstanding: 3 }.to_string(),
1565            "workers(3)"
1566        );
1567        assert_eq!(
1568            WaitReason::Children { outstanding: 2 }.to_string(),
1569            "children(2)"
1570        );
1571    }
1572
1573    /// Only the two engine-side reasons resolve on their own; the rest are a
1574    /// person's to clear. This is the predicate a badge should be built on.
1575    #[test]
1576    fn only_the_engine_side_reasons_need_nobody() {
1577        assert!(WaitReason::ToolApproval.needs_a_person());
1578        assert!(WaitReason::UserPrompt.needs_a_person());
1579        assert!(WaitReason::TaintGate.needs_a_person());
1580        assert!(WaitReason::InteractionPoint.needs_a_person());
1581        assert!(!WaitReason::FanOutWorkers { outstanding: 2 }.needs_a_person());
1582        assert!(!WaitReason::Children { outstanding: 2 }.needs_a_person());
1583    }
1584
1585    #[test]
1586    fn run_meta_serde_roundtrip() {
1587        let mut m = sample_meta();
1588        m.status = RunStatus::Running;
1589        m.metadata.insert("k".to_string(), "v".to_string());
1590        m.title = Some("A title".to_string());
1591        m.callback_secret = Some("shh".to_string());
1592        m.parent_run_id = Some("parent-1".to_string());
1593        m.children = vec!["child-a".to_string(), "child-b".to_string()];
1594        m.depth = 2;
1595        m.max_child_depth = 5;
1596        let json = serde_json::to_string(&m).unwrap();
1597        let back: RunMeta = serde_json::from_str(&json).unwrap();
1598        assert_eq!(back.run_id, m.run_id);
1599        assert_eq!(back.status, RunStatus::Running);
1600        assert_eq!(back.metadata.get("k").map(String::as_str), Some("v"));
1601        assert_eq!(back.title.as_deref(), Some("A title"));
1602        assert_eq!(back.callback_secret.as_deref(), Some("shh"));
1603        assert_eq!(back.parent_run_id.as_deref(), Some("parent-1"));
1604        assert_eq!(
1605            back.children,
1606            vec!["child-a".to_string(), "child-b".to_string()]
1607        );
1608        assert_eq!(back.depth, 2);
1609        assert_eq!(back.max_child_depth, 5);
1610    }
1611
1612    #[test]
1613    fn run_status_display_all_variants() {
1614        assert_eq!(RunStatus::Starting.to_string(), "Starting");
1615        assert_eq!(RunStatus::Running.to_string(), "Running");
1616        assert_eq!(RunStatus::WaitingInput.to_string(), "WaitingInput");
1617        assert_eq!(RunStatus::Complete.to_string(), "Complete");
1618        assert_eq!(
1619            RunStatus::CompleteInteractive.to_string(),
1620            "CompleteInteractive"
1621        );
1622        assert_eq!(RunStatus::Paused.to_string(), "Paused");
1623        assert_eq!(RunStatus::Error.to_string(), "Error");
1624        assert_eq!(RunStatus::Cancelled.to_string(), "Cancelled");
1625    }
1626
1627    #[test]
1628    fn run_status_serde_snake_case_roundtrip() {
1629        for s in [
1630            RunStatus::Starting,
1631            RunStatus::Running,
1632            RunStatus::WaitingInput,
1633            RunStatus::Complete,
1634            RunStatus::CompleteInteractive,
1635            RunStatus::Paused,
1636            RunStatus::Error,
1637            RunStatus::Cancelled,
1638        ] {
1639            let json = serde_json::to_string(&s).unwrap();
1640            let back: RunStatus = serde_json::from_str(&json).unwrap();
1641            assert_eq!(back, s);
1642        }
1643        assert_eq!(
1644            serde_json::to_string(&RunStatus::WaitingInput).unwrap(),
1645            "\"waiting_input\""
1646        );
1647        assert_eq!(
1648            serde_json::to_string(&RunStatus::Paused).unwrap(),
1649            "\"paused\""
1650        );
1651    }
1652
1653    #[test]
1654    fn context_snapshot_serde_roundtrip() {
1655        let snap = ContextSnapshot {
1656            stage_name: "plan".to_string(),
1657            total_tokens: 42,
1658            max_tokens: 100,
1659            regions: vec![RegionSnapshot {
1660                name: "history".to_string(),
1661                kind: "sliding".to_string(),
1662                current_tokens: 10,
1663                max_tokens: 50,
1664                entries: vec![RegionEntrySnapshot {
1665                    content: "hi".into(),
1666                    tokens: 1,
1667                    kind: crate::region::EntryKind::UserMessage,
1668                    metadata: Some(serde_json::json!({"a": 1})),
1669                    key: Some("k".to_string()),
1670                    taint: Default::default(),
1671                    reasoning: None,
1672                }],
1673                description: None,
1674            }],
1675        };
1676        let json = serde_json::to_string(&snap).unwrap();
1677        let back: ContextSnapshot = serde_json::from_str(&json).unwrap();
1678        assert_eq!(back.stage_name, "plan");
1679        assert_eq!(back.regions.len(), 1);
1680        assert_eq!(back.regions[0].entries.len(), 1);
1681        assert_eq!(back.regions[0].entries[0].content, "hi");
1682        assert_eq!(back.regions[0].entries[0].key.as_deref(), Some("k"));
1683    }
1684
1685    #[test]
1686    fn region_snapshot_skips_empty_entries_in_json() {
1687        let snap = RegionSnapshot {
1688            name: "r".to_string(),
1689            kind: "pinned".to_string(),
1690            current_tokens: 0,
1691            max_tokens: 0,
1692            entries: vec![],
1693            description: None,
1694        };
1695        let json = serde_json::to_string(&snap).unwrap();
1696        assert!(!json.contains("entries"));
1697    }
1698
1699    #[test]
1700    fn stage_run_status_display_all_variants() {
1701        assert_eq!(StageRunStatus::Pending.to_string(), "Pending");
1702        assert_eq!(StageRunStatus::Skipped.to_string(), "Skipped");
1703        assert_eq!(StageRunStatus::Active.to_string(), "Active");
1704        assert_eq!(StageRunStatus::WaitingInput.to_string(), "WaitingInput");
1705        assert_eq!(StageRunStatus::Complete.to_string(), "Complete");
1706        assert_eq!(StageRunStatus::Error.to_string(), "Error");
1707    }
1708
1709    #[test]
1710    fn run_flags_record_modification_dedups_paths_and_caps_the_list() {
1711        let mut flags = RunFlags::default();
1712        flags.record_modification("src/a.rs");
1713        flags.record_modification("src/a.rs");
1714        flags.record_modification("src/b.rs");
1715        assert_eq!(flags.modified_file_count, 3);
1716        assert_eq!(flags.modified_files, vec!["src/a.rs", "src/b.rs"]);
1717
1718        // Past the cap the count keeps rising but the list stops growing, so a
1719        // long run can't bloat meta.json.
1720        for i in 0..MAX_TRACKED_MODIFIED_FILES {
1721            flags.record_modification(&format!("f{i}.rs"));
1722        }
1723        assert_eq!(flags.modified_files.len(), MAX_TRACKED_MODIFIED_FILES);
1724        assert_eq!(flags.modified_file_count, 3 + MAX_TRACKED_MODIFIED_FILES);
1725    }
1726
1727    #[test]
1728    fn note_modified_path_joins_the_list_without_touching_the_call_count() {
1729        let mut flags = RunFlags::default();
1730        // A modifying tool call: counts and lists.
1731        flags.record_modification("plot_chart.py");
1732        // A shell creation, noted from a workdir scan: lists, but is not a
1733        // modifying tool call, so the count stays 1 - and a second scan that
1734        // finds it again neither re-adds nor re-counts.
1735        assert!(flags.note_modified_path("chart.png"));
1736        assert!(!flags.note_modified_path("chart.png"));
1737        assert_eq!(flags.modified_file_count, 1);
1738        assert_eq!(flags.modified_files, vec!["plot_chart.py", "chart.png"]);
1739
1740        // Past the cap it stops adding and says so.
1741        let mut full = RunFlags::default();
1742        for i in 0..MAX_TRACKED_MODIFIED_FILES {
1743            assert!(full.note_modified_path(&format!("f{i}.png")));
1744        }
1745        assert!(!full.note_modified_path("one-too-many.png"));
1746        assert_eq!(full.modified_files.len(), MAX_TRACKED_MODIFIED_FILES);
1747        assert_eq!(full.modified_file_count, 0);
1748    }
1749
1750    #[test]
1751    fn run_meta_flags_default_for_older_files() {
1752        // A meta.json written before `flags` existed has no such key at all.
1753        let mut meta = RunMeta::new(
1754            "r".to_string(),
1755            "a".to_string(),
1756            "/p".to_string(),
1757            "t".to_string(),
1758            None,
1759            "/w".to_string(),
1760            1,
1761        );
1762        meta.flags.empty_output = true;
1763        // Drop the key structurally rather than by string surgery: a literal
1764        // spelling of the serialized flags silently stops matching the moment a
1765        // field is added, and the test then passes for the wrong reason.
1766        let mut json = serde_json::to_value(&meta).unwrap();
1767        json.as_object_mut().unwrap().remove("flags").unwrap();
1768        assert!(!json.to_string().contains("flags"));
1769        let back: RunMeta = serde_json::from_value(json).unwrap();
1770        assert_eq!(back.flags, RunFlags::default());
1771    }
1772}