leviath_core/run_meta.rs
1//! Plain, serializable run-state data types.
2//!
3//! These are pure data (`serde`-derived structs/enums plus trivial constructors)
4//! with no filesystem or async dependencies, so they can be named by both
5//! `leviath-cli` and the `leviath-runtime` engine. All on-disk IO for
6//! these types (reading/writing `meta.json`, run directories, snapshots, etc.)
7//! lives in `leviath_cli::runstate`.
8
9use serde::{Deserialize, Serialize};
10use std::collections::HashMap;
11
12mod stage_ledger;
13
14// Re-exported flat rather than left behind a path of their own: the stage ledger
15// moved out of this file because the file got long, and that is a fact about
16// where the source lives, not about what a caller should have to type.
17pub use stage_ledger::{
18 MAX_STAGE_VISITS, StageCall, StageRecord, StageRunStatus, StageVisitRecord,
19};
20
21/// Current status of a background run.
22#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
23#[serde(rename_all = "snake_case")]
24pub enum RunStatus {
25 /// Accepted and being set up; no inference has been issued yet.
26 Starting,
27 /// Working: inferring, calling tools, or moving between stages.
28 Running,
29 /// Blocked on a person. A human-in-the-loop tool or an interaction point is
30 /// waiting for an answer, and the run holds its concurrency slot until it
31 /// gets one.
32 WaitingInput,
33 /// Finished, with nothing further to accept.
34 Complete,
35 /// All required stages done; agent still accepts optional follow-up input.
36 /// Shown as "Complete" in the dashboard - no kill option, input still enabled.
37 CompleteInteractive,
38 /// Paused by the user; resumes on request and is restored paused after a
39 /// daemon restart.
40 Paused,
41 /// Stopped by a failure. `RunMeta::error` carries what went wrong.
42 Error,
43 /// Stopped from outside, by `lev kill` or a shutting-down daemon. Distinct
44 /// from [`Error`](Self::Error): nothing went wrong, someone decided.
45 Cancelled,
46}
47
48impl RunStatus {
49 /// The word this status goes on the wire as: `snake_case`, the same
50 /// spelling serde writes into `meta.json` and into every JSON body that
51 /// carries a whole run.
52 ///
53 /// Here rather than left to each caller because a status reaches a client
54 /// three ways - serialized inside a run, rendered into a `status` string by
55 /// a route that builds its own response shape, and forwarded off the
56 /// engine's event stream - and all three have to spell one state one way.
57 /// [`Display`](std::fmt::Display) is PascalCase and is
58 /// for a person reading a terminal; this is for a client matching on it.
59 pub fn wire(&self) -> &'static str {
60 match self {
61 RunStatus::Starting => "starting",
62 RunStatus::Running => "running",
63 RunStatus::WaitingInput => "waiting_input",
64 RunStatus::Complete => "complete",
65 RunStatus::CompleteInteractive => "complete_interactive",
66 RunStatus::Paused => "paused",
67 RunStatus::Error => "error",
68 RunStatus::Cancelled => "cancelled",
69 }
70 }
71}
72
73impl std::fmt::Display for RunStatus {
74 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
75 match self {
76 RunStatus::Starting => write!(f, "Starting"),
77 RunStatus::Running => write!(f, "Running"),
78 RunStatus::WaitingInput => write!(f, "WaitingInput"),
79 RunStatus::Complete => write!(f, "Complete"),
80 RunStatus::CompleteInteractive => write!(f, "CompleteInteractive"),
81 RunStatus::Paused => write!(f, "Paused"),
82 RunStatus::Error => write!(f, "Error"),
83 RunStatus::Cancelled => write!(f, "Cancelled"),
84 }
85 }
86}
87
88/// Why a run's status is [`RunStatus::WaitingInput`].
89///
90/// `WaitingInput` alone is several unrelated situations wearing one word, and
91/// they call for opposite responses: a fan-out parent whose workers are
92/// churning is healthy and needs nothing, while a run parked on a
93/// tool-approval prompt is stopped dead until a person answers it. With the
94/// two indistinguishable, an operator reading `waiting` across a factory
95/// concludes it has stalled and starts killing healthy runs, and every client
96/// that reads `meta.json` is left guessing the same way.
97///
98/// Derived on demand from markers the engine already sets, by
99/// [`wait_reason_from`]; nothing tracks it separately, so it cannot fall out of
100/// sync with the status it explains. It lives here rather than in the runtime
101/// because it is both reported live over the control socket and written to
102/// `meta.json`, and one vocabulary across those two is the whole point.
103///
104/// Deliberately not new [`RunStatus`] variants: the status is matched
105/// exhaustively across the codebase and serialized two ways on the wire, so
106/// splitting it would break every consumer to express something that is not a
107/// new state. The run really is waiting; this says on what.
108#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)]
109#[serde(rename_all = "snake_case", tag = "reason")]
110pub enum WaitReason {
111 /// Blocked on a tool-approval prompt. Needs a person (or `--yolo`).
112 ToolApproval,
113
114 /// Blocked on a question the agent itself asked (`ask_user_*`,
115 /// `present_for_review`). Needs a person.
116 UserPrompt,
117
118 /// Blocked on a taint-gate clearance prompt. Needs a person.
119 TaintGate,
120
121 /// Blocked on a blueprint stage-boundary checkpoint. Needs a person.
122 InteractionPoint,
123
124 /// Parked while fan-out workers run. Healthy; resolves on its own.
125 FanOutWorkers {
126 /// Workers still to finish, counting both running and not-yet-started.
127 outstanding: usize,
128 },
129
130 /// Parked while spawned sub-agents run (`requires_children`). Healthy;
131 /// resolves on its own.
132 Children {
133 /// Children that have not reached a terminal status.
134 outstanding: usize,
135 },
136
137 /// Parked because something on the machine has to change before this run
138 /// can go on: a provider it needs is not configured, a key was rejected,
139 /// an account is out of credits.
140 ///
141 /// These are all deterministic and all outside the run's control, so
142 /// ending the run would throw away everything it had done to punish a
143 /// person for a typo in `config.toml`. The run holds its place instead,
144 /// and `lev resume` picks it up once the machine is fixed.
145 ///
146 /// The distinction that matters is not "is there a fix" but "does the fix
147 /// let *this* run continue": a broken blueprint is equally deterministic
148 /// and equally fixable, and still cannot be resumed into, because the
149 /// blueprint was read at spawn.
150 NeedsSetup {
151 /// Which kind of problem, so a client can offer the right thing to do
152 /// rather than parse the sentence below.
153 blocker: SetupBlocker,
154 /// What to do about it, in a sentence, for whoever reads the run.
155 remedy: String,
156 },
157}
158
159/// What is stopping a [`WaitReason::NeedsSetup`] run, in a form a client can
160/// branch on.
161///
162/// One variant per remedy, not per error: these are the cases whose *fixes*
163/// differ. Topping up an account, adding a provider to `config.toml` and
164/// replacing a rejected key are three different screens, and a console that
165/// had only the sentence would be reduced to matching on its wording.
166#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
167#[serde(rename_all = "snake_case")]
168pub enum SetupBlocker {
169 /// The stage names a provider this install has not configured.
170 ProviderMissing,
171 /// The account behind the provider is out of credits.
172 CreditsExhausted,
173 /// The key was rejected.
174 AuthFailed,
175 /// The key is valid but not allowed to use the model.
176 Forbidden,
177 /// Every candidate is out of service, for reasons that do not agree or are
178 /// not known. The remedy names what was tried last.
179 ProvidersUnavailable,
180 /// The provider could not be reached at all: the name did not resolve,
181 /// the connection was refused, or the TLS handshake failed. The network
182 /// or the address is what to check.
183 ProviderUnreachable,
184 /// The provider was reached and did not answer in time. It is up, but
185 /// slow, or the request was large; a resume tries again.
186 ProviderTimedOut,
187 /// The provider was reached and failed: a server error, a reply that
188 /// stopped part-way, or one that could not be read. Nothing about the
189 /// setup is known to be wrong; a resume tries again once it recovers.
190 ProviderFailed,
191}
192
193impl std::fmt::Display for SetupBlocker {
194 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
195 match self {
196 Self::ProviderMissing => f.write_str("provider"),
197 Self::CreditsExhausted => f.write_str("credits"),
198 Self::AuthFailed => f.write_str("key"),
199 Self::Forbidden => f.write_str("access"),
200 Self::ProvidersUnavailable => f.write_str("providers"),
201 Self::ProviderUnreachable => f.write_str("unreachable"),
202 Self::ProviderTimedOut => f.write_str("timed out"),
203 Self::ProviderFailed => f.write_str("failed"),
204 }
205 }
206}
207
208/// What a parked run needs, gathered where the markers are visible.
209#[derive(Debug, Clone, PartialEq, Eq)]
210pub struct SetupNeeded {
211 /// Which kind of problem it is.
212 pub blocker: SetupBlocker,
213 /// What to do about it.
214 pub remedy: String,
215}
216
217impl WaitReason {
218 /// Whether clearing this needs a person. `false` means the run is parked on
219 /// other work and will move on by itself.
220 pub fn needs_a_person(&self) -> bool {
221 !matches!(self, Self::FanOutWorkers { .. } | Self::Children { .. })
222 }
223}
224
225/// A stopwatch that runs only while the thing it measures is actually working.
226///
227/// Wall-clock age and working time are different questions, and the difference
228/// is the whole point of this type: a run paused overnight is twelve hours old
229/// and spent eleven of them doing nothing. Age comes from `started_at`; this is
230/// what a reader means by "how long has this taken".
231///
232/// Kept as banked seconds plus the start of the span in progress, rather than a
233/// single total, so a reader that polls between writes still sees the number
234/// climb instead of stepping once per heartbeat.
235#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, Serialize, Deserialize)]
236pub struct ActiveClock {
237 /// Seconds banked from spans that have already ended.
238 #[serde(default)]
239 pub banked_secs: u64,
240 /// When the span in progress began, or `None` while the clock is stopped.
241 #[serde(default)]
242 pub since: Option<i64>,
243}
244
245impl ActiveClock {
246 /// Start or stop the clock to match `running`, banking the span that ends.
247 ///
248 /// Idempotent: called every persist tick with the same answer, it does
249 /// nothing, so only genuine transitions move the accounting.
250 pub fn observe(&mut self, now: i64, running: bool) {
251 match (running, self.since) {
252 (true, None) => self.since = Some(now),
253 (false, Some(started)) => {
254 self.banked_secs += crate::duration::between(started, now);
255 self.since = None;
256 }
257 _ => {}
258 }
259 }
260
261 /// Close the span in progress at `as_of`, banking it.
262 ///
263 /// For a clock read back from disk: whatever the run was doing stopped when
264 /// the daemon holding it did, which is the last moment the record was
265 /// written - not now. Without this, a run reloaded after the daemon was down
266 /// for a day comes back claiming a day's work.
267 pub fn settle(&mut self, as_of: i64) {
268 self.observe(as_of, false);
269 }
270
271 /// Working seconds at `now`, counting the span in progress.
272 pub fn total_secs(&self, now: i64) -> u64 {
273 self.banked_secs + self.since.map_or(0, |s| crate::duration::between(s, now))
274 }
275}
276
277/// Whether a run in this state has its clock running.
278///
279/// It runs while the run is working, and while the run is parked on work of its
280/// own - fan-out workers, sub-agents - because that time is the run taking as
281/// long as it takes. It stops for everything that is not the run's doing:
282/// paused, blocked on a person, parked until the machine is fixed, finished.
283///
284/// A `waiting_input` run with nothing claiming the wait is treated as blocked on
285/// a person, which is the only way it gets there.
286pub fn clock_runs(status: &RunStatus, waiting_on: Option<&WaitReason>) -> bool {
287 match status {
288 RunStatus::Starting | RunStatus::Running => true,
289 RunStatus::WaitingInput => waiting_on.is_some_and(|r| !r.needs_a_person()),
290 RunStatus::Paused
291 | RunStatus::Complete
292 | RunStatus::CompleteInteractive
293 | RunStatus::Error
294 | RunStatus::Cancelled => false,
295 }
296}
297
298impl std::fmt::Display for WaitReason {
299 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
300 match self {
301 Self::ToolApproval => f.write_str("tool approval"),
302 Self::UserPrompt => f.write_str("user prompt"),
303 Self::TaintGate => f.write_str("taint gate"),
304 Self::InteractionPoint => f.write_str("checkpoint"),
305 Self::FanOutWorkers { outstanding } => write!(f, "workers({outstanding})"),
306 Self::Children { outstanding } => write!(f, "children({outstanding})"),
307 // The remedy is a sentence; this is a table cell. The blocker is
308 // the half that fits, and the half that says which screen to open.
309 // The three that describe the provider rather than something the
310 // install lacks say what happened to it.
311 Self::NeedsSetup {
312 blocker:
313 blocker @ (SetupBlocker::ProviderUnreachable
314 | SetupBlocker::ProviderTimedOut
315 | SetupBlocker::ProviderFailed),
316 ..
317 } => write!(f, "provider {blocker}"),
318 Self::NeedsSetup { blocker, .. } => write!(f, "needs {blocker}"),
319 }
320 }
321}
322
323/// The parking markers an agent carries, gathered by whoever can see them.
324///
325/// The live listing reads these straight off the world; the persistence system
326/// reads them off its query. Both then hand them here, so the precedence below
327/// is written once instead of once per surface - two copies of it would
328/// disagree the first time either was edited.
329#[derive(Debug, Clone, Default, PartialEq)]
330pub struct WaitMarkers {
331 /// A taint-gate clearance prompt is outstanding.
332 pub gate_prompt: bool,
333 /// A blueprint stage-boundary checkpoint is holding.
334 pub interaction_point: bool,
335 /// Fan-out workers still to finish, when this run is a fan-out parent.
336 pub fan_out_outstanding: Option<usize>,
337 /// Sub-agents still running, when this run is held for its children.
338 pub children_outstanding: Option<usize>,
339 /// The kind of hub request holding this run, when one is.
340 pub interaction: Option<crate::interaction::InteractionKind>,
341 /// Whether a hub request is holding it at all. Separate from the kind
342 /// because the kind can be unknown while the block is real.
343 pub awaiting_interaction: bool,
344 /// The run is parked until the machine is fixed, and this is what it
345 /// needs.
346 pub needs_setup: Option<SetupNeeded>,
347}
348
349/// Why a parked run is parked, or `None` when it is not parked or nothing has
350/// claimed it.
351///
352/// Order matters, and it is the specific claim first. A taint-gate block and a
353/// stage checkpoint each open a hub request of their own, so both also look
354/// like a generic prompt; asking the specific markers first is what keeps them
355/// from all reporting as one.
356pub fn wait_reason_from(parked: bool, markers: &WaitMarkers) -> Option<WaitReason> {
357 if !parked {
358 return None;
359 }
360 // First, because it outranks everything: a run whose provider is missing
361 // is not going to be unblocked by answering a prompt.
362 if let Some(need) = &markers.needs_setup {
363 return Some(WaitReason::NeedsSetup {
364 blocker: need.blocker,
365 remedy: need.remedy.clone(),
366 });
367 }
368 if markers.gate_prompt {
369 return Some(WaitReason::TaintGate);
370 }
371 if markers.interaction_point {
372 return Some(WaitReason::InteractionPoint);
373 }
374 if let Some(outstanding) = markers.fan_out_outstanding {
375 return Some(WaitReason::FanOutWorkers { outstanding });
376 }
377 if let Some(outstanding) = markers.children_outstanding {
378 return Some(WaitReason::Children { outstanding });
379 }
380 if markers.awaiting_interaction {
381 return Some(match markers.interaction {
382 Some(crate::interaction::InteractionKind::ToolApproval) => WaitReason::ToolApproval,
383 _ => WaitReason::UserPrompt,
384 });
385 }
386 None
387}
388
389/// Metadata for a single background agent run.
390#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
391pub struct RunMeta {
392 /// Identifies the run everywhere, and names its directory under
393 /// `~/.leviath/runs/`. Assigned at spawn and never reused.
394 pub run_id: String,
395 /// The blueprint's `[agent] name`, not the file it was loaded from. Two runs
396 /// of the same agent from different paths share this.
397 pub agent_name: String,
398 /// Absolute path to the agent manifest directory
399 pub agent_path: String,
400 /// The task text the run was started with, verbatim.
401 pub task: String,
402 /// The `provider/model` actually resolved for the entry stage, or `None`
403 /// before resolution. Later stages may use a different one; this is not
404 /// rewritten to follow them.
405 pub model: Option<String>,
406 /// Always 0. There is no worker process per run: the daemon hosts every run
407 /// as an entity in one shared world, so no run has a pid of its own.
408 ///
409 /// Kept because it is written into every `meta.json` there has ever been,
410 /// and served from `GET /api/agents`. Do not key liveness on it. `pid == 0`
411 /// is true of a run that is working, a run that has finished, and a run
412 /// nothing is driving, so a sweeper that reverts on it reverts everything.
413 /// Ask the daemon (`lev ps`) whether it is still hosting the run, and read
414 /// `status` and `last_progress_at` off disk for what became of it.
415 #[serde(default)]
416 pub pid: u32,
417 /// Where the run stands. The durable counterpart to the ECS world's live
418 /// `AgentStatus`, and the one that survives a daemon restart.
419 pub status: RunStatus,
420 /// Name of the stage the run is in, matching a key under `[stages]`.
421 pub current_stage: String,
422 /// Zero-based position of `current_stage` in the blueprint's stage list.
423 /// Not a progress measure: stages can loop and revisit.
424 pub stage_index: usize,
425 /// How many stages the blueprint declares, so a reader can render
426 /// `stage_index` as "3 of 7" without loading the manifest.
427 pub num_stages: usize,
428 /// Inference turns taken in the current stage, reset on entering a new one.
429 /// Compared against the stage's `max_iterations`.
430 pub iteration: usize,
431 /// Cumulative input tokens billed across every inference this run has made,
432 /// including retries.
433 pub prompt_tokens: usize,
434 /// Cumulative output tokens billed across every inference this run has made.
435 pub completion_tokens: usize,
436 /// Cumulative tokens read from provider cache.
437 #[serde(default)]
438 pub cached_tokens: usize,
439 /// Cumulative tokens written to provider cache.
440 #[serde(default)]
441 pub cache_write_tokens: usize,
442 /// Total number of tool calls made across all iterations.
443 #[serde(default)]
444 pub tool_calls: usize,
445 /// What this run has spent, when every call could be priced.
446 ///
447 /// `None` means *unknown*, never free: some call was served by a model with
448 /// no reported cost and no known rates, so any total would be understating
449 /// by an unknown amount. See [`unpriced_calls`](Self::unpriced_calls) for
450 /// how many, and [`cost_is_exact`](Self::cost_is_exact) for whether the
451 /// figure is the provider's own or reconstructed from rate cards.
452 #[serde(default)]
453 pub cost_usd: Option<f64>,
454 /// Calls that could not be priced at all. Non-zero forces
455 /// [`cost_usd`](Self::cost_usd) to `None`.
456 #[serde(default)]
457 pub unpriced_calls: usize,
458 /// Whether every priced call carried the provider's own cost figure rather
459 /// than one computed from published rates. `false` means the total is this
460 /// process's best reconstruction of the invoice, not the invoice.
461 #[serde(default)]
462 pub cost_is_exact: bool,
463 /// The priced subtotal, kept even when `cost_usd` is `None` so a resumed run
464 /// does not restart its accounting from zero.
465 #[serde(default)]
466 pub cost_priced_usd: f64,
467 /// Absolute path to the working directory for tool execution
468 pub workdir: String,
469 /// Unix timestamp (seconds)
470 pub started_at: i64,
471 /// Unix timestamp (seconds)
472 pub updated_at: i64,
473 /// Unix seconds when this run last actually moved: a new iteration, a new
474 /// stage, or a change of status. `None` before the first snapshot lands, and
475 /// on runs written by a daemon older than this field.
476 ///
477 /// Distinct from `updated_at`, which also advances on the 30-second
478 /// persistence heartbeat and so stays fresh on a run that is wedged. A fresh
479 /// `updated_at` is evidence the daemon is alive, and no evidence at all about
480 /// the run. Anything that ages a run must read this instead. Note that a
481 /// daemon restart resets it: a reloaded run really is re-driven from its
482 /// saved context, so it really has just moved.
483 #[serde(default)]
484 pub last_progress_at: Option<i64>,
485 /// How long this run has actually been working, as against how long it has
486 /// existed. See [`ActiveClock`], and read it through
487 /// [`RunMeta::active_runtime_secs`] rather than directly.
488 ///
489 /// `None` means no clock was kept - a run written by a daemon older than
490 /// this field. Deliberately an `Option` rather than a zeroed clock: a run
491 /// that finished inside a second has a genuine total of zero, and the two
492 /// have different right answers.
493 #[serde(default)]
494 pub active: Option<ActiveClock>,
495 /// What went wrong, set alongside [`RunStatus::Error`]. `None` on every
496 /// other status.
497 pub error: Option<String>,
498 /// Short human-readable title generated from the task prompt (None until generated).
499 #[serde(default)]
500 pub title: Option<String>,
501 /// Why [`Self::title`] is still `None`, once title generation has given up
502 /// - the provider it could not reach, or what came back instead of a title.
503 ///
504 /// `None` is the ordinary state: titling has not finished yet, or was never
505 /// asked for. `Some` means it ran and could not produce a name, which is
506 /// otherwise indistinguishable from either.
507 #[serde(default)]
508 pub title_error: Option<String>,
509 /// Custom key-value pairs from the spawn request (API metadata).
510 #[serde(default)]
511 pub metadata: HashMap<String, String>,
512 /// Webhook URL to POST on agent completion/error.
513 #[serde(default)]
514 pub callback_url: Option<String>,
515 /// Optional shared secret used to HMAC-SHA256 sign the webhook body
516 /// (`X-Leviath-Signature` header) so the receiver can verify authenticity.
517 ///
518 /// Persisted, because the daemon must still be able to sign a webhook for a
519 /// run it reloaded after a restart. **Never serve it** - strip it with
520 /// [`RunMeta::redacted`] before any of this struct leaves the process.
521 #[serde(default)]
522 pub callback_secret: Option<String>,
523 /// Links sub-agent runs to their parent run.
524 #[serde(default)]
525 pub parent_run_id: Option<String>,
526 /// Run-ids of this agent's direct sub-agents (sub-agent-tool spawns and
527 /// fan-out workers). Persisted so the daemon can rebuild the exact
528 /// parent→children tree on restart rather than reload children as orphans.
529 #[serde(default)]
530 pub children: Vec<String>,
531 /// This agent's depth in the sub-agent tree (0 for a top-level run).
532 /// Persisted so a reloaded child enforces its remaining spawn-depth budget.
533 #[serde(default)]
534 pub depth: usize,
535 /// The sub-agent depth cap this agent imposes on its own children
536 /// (0 when it has none). Restores `SubAgentChildren::max_child_depth`.
537 #[serde(default)]
538 pub max_child_depth: usize,
539 /// Why this run may have produced nothing useful - see [`RunFlags`].
540 #[serde(default)]
541 pub flags: RunFlags,
542 /// Whether the run was launched unattended (`--yolo`), so a daemon restart
543 /// resumes it the way it was started.
544 ///
545 /// Persisted rather than dropped on reload. Forgetting a launch override
546 /// only ever prompts more, never less, which is why dropping it reads as
547 /// safe; what it actually does is convert an unattended run into one
548 /// parked on a prompt nobody is watching for, discarding consent the
549 /// operator gave at launch. Runs written before this field existed default
550 /// to attended, so nothing is escalated retroactively.
551 #[serde(default)]
552 pub yolo: bool,
553 /// The named yolo profile (`--yolo=<name>`) the run was launched under,
554 /// persisted with `yolo` for the same reason: a restart that dropped the
555 /// name would resume a carefully scoped run under bare `--yolo`, which is
556 /// the escalating direction. Absent for the bare flag and for runs written
557 /// before profiles existed.
558 #[serde(default, skip_serializing_if = "Option::is_none")]
559 pub yolo_profile: Option<String>,
560 /// How much of the blueprint's `[read_paths]` the config granted, as
561 /// resolved at spawn. `None` for a blueprint that declared none, and for
562 /// runs written before this field existed.
563 #[serde(default, skip_serializing_if = "Option::is_none")]
564 pub read_paths: Option<ReadPathGrantCounts>,
565 /// What the agent handed back, if it submitted anything: everything about
566 /// the answer except the bytes.
567 ///
568 /// This is the run's answer, as distinct from `error` (why it failed) and
569 /// from the stage logs (what it did along the way). The content itself is
570 /// in a sidecar file beside this one, because this file is parsed for every
571 /// run on every listing and must stay small no matter how long an answer is.
572 #[serde(default, skip_serializing_if = "Option::is_none")]
573 pub final_output: Option<crate::output::FinalOutputDescriptor>,
574
575 /// Why this run is parked, when it is. `None` on every other status, and
576 /// on a run written before this field existed. Same vocabulary the live
577 /// listing reports, so `lev ps` and a client reading this file describe a
578 /// run the same way.
579 ///
580 /// Additive on purpose: `default` means a `meta.json` from an older build
581 /// still loads, and `skip_serializing_if` means a run that is not parked
582 /// writes exactly the file it wrote before, so an older build reading a
583 /// newer run sees nothing new either.
584 #[serde(default, skip_serializing_if = "Option::is_none")]
585 pub waiting_on: Option<WaitReason>,
586 /// The output shape this run was launched asking for, when the caller
587 /// overrode the blueprint's.
588 ///
589 /// Persisted for the same reason `yolo` is: a daemon restart rebuilds the
590 /// run's spawn arguments from this file, and dropping the request would
591 /// silently revert the run to the blueprint's shape partway through. The
592 /// caller asked once and should not have to ask again.
593 #[serde(default, skip_serializing_if = "Option::is_none")]
594 pub output_request: Option<crate::output::OutputSpec>,
595 /// The `--model` the run was launched with, exactly as given
596 /// (`provider/model` or a bare model), when the caller gave one.
597 ///
598 /// Distinct from `model`, which is what the entry stage *resolved to* and
599 /// is recorded whether or not anything was overridden. A daemon restart
600 /// rebuilds the run's spawn arguments from this file, and must hand back
601 /// this field rather than `model`: handing back `model` pins every stage
602 /// of a run launched with no `--model` to its first stage's provider and
603 /// model, and loses its failover list. This field is what was actually
604 /// asked for, so a reload asks for the same thing - and for a run that
605 /// asked for nothing, resolves each stage afresh, as the launch did.
606 ///
607 /// Runs written before this field existed reload with no override. That
608 /// loses a `--model` given to such a run, which is the smaller harm: the
609 /// stage falls back to its blueprint's list rather than being pinned to a
610 /// pair the user may never have named.
611 #[serde(default, skip_serializing_if = "Option::is_none")]
612 pub model_override: Option<String>,
613}
614
615/// How many `[read_paths]` entries a run's blueprint declared, and how many of
616/// them the user's config actually granted.
617///
618/// Declaring is not granting: an ungranted entry is inert, and the reads it was
619/// meant to allow are refused. Recorded at spawn, because that is when the
620/// policy the run enforces is fixed - editing the config afterwards changes
621/// nothing for a run already in flight.
622#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq, Eq)]
623pub struct ReadPathGrantCounts {
624 /// Entries the blueprint declares.
625 pub declared: usize,
626 /// Entries the config grants.
627 pub granted: usize,
628}
629
630/// Post-hoc diagnosis of a run's productivity, persisted in `meta.json` so a
631/// harness (or the dashboard) can tell an empty run from a successful one
632/// without inspecting the workspace or parsing logs.
633///
634/// The motivating failure: 13/300 SWE-bench runs completed their whole stage
635/// pipeline and produced no file changes at all. Nothing on disk said so, or
636/// said why.
637#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
638pub struct RunFlags {
639 /// Paths passed to file-modifying tools that succeeded, in first-touch
640 /// order. Capped at [`MAX_TRACKED_MODIFIED_FILES`]; `modified_file_count`
641 /// keeps the true total.
642 #[serde(default)]
643 pub modified_files: Vec<String>,
644 /// Total successful file-modifying tool calls across the run (uncapped).
645 #[serde(default)]
646 pub modified_file_count: usize,
647 /// The run reached a terminal status having modified nothing, and its
648 /// blueprint gave it a way to modify something. See [`Self::no_output_tools`].
649 #[serde(default)]
650 pub empty_output: bool,
651 /// No stage of the blueprint advertised a file-modifying tool, so this run
652 /// could never have produced the file changes `empty_output` looks for.
653 ///
654 /// Recorded because "modified no files" only diagnoses an agent that was
655 /// supposed to modify files. A router that spawns sub-agents, or an agent
656 /// whose answer is its text, would otherwise report itself empty on every
657 /// successful run. The framework has no basis to judge such a run, so it
658 /// says nothing rather than accusing.
659 ///
660 /// This mirrors the escape the runtime's `gate_blocks` already applies per
661 /// stage: a `require_modifications` gate on a stage that advertises no
662 /// modifying tool is skipped, because it could never pass.
663 ///
664 /// Phrased negatively so the `false` that [`Default`] and `serde(default)`
665 /// produce means "was capable" - the behavior every `meta.json` written
666 /// before this field had.
667 #[serde(default)]
668 pub no_output_tools: bool,
669 /// `web_search` calls this run made, across every stage.
670 ///
671 /// Zero for an agent that never had the tool, which is most of them - the
672 /// pair says nothing about a run that could not have searched, the same
673 /// escape [`Self::no_output_tools`] applies to file modifications.
674 #[serde(default)]
675 pub searches_run: usize,
676 /// Of those, how many came back with nothing usable: no results, or a
677 /// diagnostic saying the search could not run.
678 ///
679 /// A research run whose searches all came back empty still finishes
680 /// `complete` and still writes a confident, fully cited report, because a
681 /// model handed an empty result set fills the gap from its training data
682 /// and cites what it remembers. One did exactly that across 47 consecutive
683 /// failed searches - no engine was configured - and nothing on disk
684 /// recorded that the report rested on nothing.
685 ///
686 /// `searches_empty == searches_run` with `searches_run > 0` is that run.
687 #[serde(default)]
688 pub searches_empty: usize,
689 /// How many stages exhausted their `max_iterations`.
690 #[serde(default)]
691 pub max_iterations_hit: usize,
692 /// How many transitions proceeded past an unsatisfied gate because the
693 /// gate's re-run budget ran out.
694 #[serde(default)]
695 pub gates_forced: usize,
696 /// Regions declared `required` that a stage gave up on and that are **still
697 /// empty**, in the order they were abandoned.
698 ///
699 /// The mechanism re-runs the stage a bounded number of times and then
700 /// proceeds with a log line, which nothing downstream reads: a run whose
701 /// agent wrote its plan and a run where we asked twice and moved on both
702 /// finished `complete`, with the second silently missing the artifact every
703 /// later stage's prompt says to work from. Names rather than a count
704 /// because knowing *which* region was abandoned is what makes it
705 /// actionable, and a run cannot abandon many.
706 ///
707 /// A later stage can fill a region an earlier one gave up on, and when that
708 /// happens the name is dropped from here - the artifact exists, so a reader
709 /// told it is missing would be told something false. A `deep-researcher` run
710 /// abandoned `sources_index` in `gather`, `analyze` wrote it, and the run
711 /// finished with a fifty-citation bibliography while still reporting the
712 /// region as never written. The moment is kept in the log; this field
713 /// answers "what is actually missing", which is the question a consumer is
714 /// asking when it renders a warning.
715 #[serde(default)]
716 pub required_regions_abandoned: Vec<String>,
717 /// The working directory disappeared mid-run.
718 #[serde(default)]
719 pub workspace_lost: bool,
720 /// The run submitted a final output.
721 ///
722 /// Counts as having produced something, alongside file modifications.
723 /// Without this an agent whose whole deliverable is its answer - a
724 /// researcher, a reviewer, a router - reported itself empty on every
725 /// successful run, which is the same mistake [`Self::no_output_tools`] was
726 /// added to correct from the other direction.
727 #[serde(default)]
728 pub produced_output: bool,
729 /// How many stages transitioned without the final output they required,
730 /// because the re-run budget ran out.
731 ///
732 /// The counterpart to [`Self::gates_forced`]: the run finished, and this
733 /// says the answer it hands back may be missing.
734 #[serde(default)]
735 pub output_forced: usize,
736 /// How many fan-out splits were unusable and were degraded to an empty
737 /// fan-out because the blueprint declared no `error` and no `dead_end`
738 /// escape from the stage.
739 ///
740 /// Ending the run on a split that cannot be parsed throws away everything
741 /// the parent has already done, workers finished and later stages still
742 /// pending. The stage moves on instead, and this is what says the fan-out
743 /// it moved on from produced nothing. Non-zero means the merge stage
744 /// worked from less than it was meant to.
745 #[serde(default)]
746 pub splits_degraded: usize,
747
748 /// Rhai scripts this run needed that could not be used, by name.
749 ///
750 /// A script that will not compile, or that throws where the runtime has to
751 /// carry on regardless, is skipped rather than fatal - which is the right
752 /// call and used to be completely silent; this list is the trace. An
753 /// output validator that cannot run is the exception: by default the
754 /// submission is rejected and the script's own error goes back to the
755 /// model as retry feedback, while `on_validator_error = "accept"` records
756 /// the submission unchecked and leaves this list as the only trace of an
757 /// answer nothing checked. The script is named here in both modes.
758 ///
759 /// Named rather than counted, because the useful question is which one -
760 /// the answer tells you which file to open.
761 #[serde(default, skip_serializing_if = "Vec::is_empty")]
762 pub broken_scripts: Vec<String>,
763}
764
765/// How many distinct modified paths [`RunFlags`] records before it stops
766/// growing (the count keeps rising). Bounds `meta.json` for a long run.
767pub const MAX_TRACKED_MODIFIED_FILES: usize = 200;
768
769impl RunFlags {
770 /// Record a successful modifying tool call on `path`.
771 pub fn record_modification(&mut self, path: &str) {
772 self.modified_file_count += 1;
773 if self.modified_files.len() < MAX_TRACKED_MODIFIED_FILES
774 && !self.modified_files.iter().any(|p| p == path)
775 {
776 self.modified_files.push(path.to_string());
777 }
778 }
779
780 /// Note a path that changed on disk without a modifying tool naming it -
781 /// a file a `shell` command created or rewrote, found by scanning the
782 /// working directory. It joins the list (deduped, capped) but does not
783 /// touch `modified_file_count`, which counts modifying *tool calls*: a
784 /// shell call is not one, and a scan that ran twice must not double-count
785 /// the same file. Returns whether the path was newly added.
786 pub fn note_modified_path(&mut self, path: &str) -> bool {
787 if self.modified_files.len() >= MAX_TRACKED_MODIFIED_FILES
788 || self.modified_files.iter().any(|p| p == path)
789 {
790 return false;
791 }
792 self.modified_files.push(path.to_string());
793 true
794 }
795}
796
797impl RunMeta {
798 /// This run's metadata with the webhook signing secret removed, for anything
799 /// that leaves the process.
800 ///
801 /// `GET /api/agents`, `/api/agents/{id}` and `/api/agents/{id}/children`
802 /// all serialize `RunMeta` whole, so without this any holder of the API
803 /// token reads every run's `callback_secret` - the key that authenticates
804 /// Leviath's webhooks to their receivers. Mirrors the `RedactedConfig`
805 /// pattern the `/api/config` handler uses.
806 ///
807 /// Returns an owned copy rather than mutating in place so a caller cannot
808 /// accidentally redact the record the daemon still needs for signing.
809 #[must_use]
810 pub fn redacted(&self) -> Self {
811 Self {
812 callback_secret: None,
813 ..self.clone()
814 }
815 }
816
817 /// A newly accepted run: [`RunStatus::Starting`], both timestamps now, every
818 /// counter at zero and every optional field unset.
819 ///
820 /// Only the seven values a caller genuinely knows at spawn are parameters.
821 /// Everything else is filled in by the daemon as the run proceeds, so taking
822 /// them here would invite a caller to invent a stage or a token count.
823 pub fn new(
824 run_id: String,
825 agent_name: String,
826 agent_path: String,
827 task: String,
828 model: Option<String>,
829 workdir: String,
830 num_stages: usize,
831 ) -> Self {
832 let now = crate::duration::now_secs();
833 Self {
834 run_id,
835 agent_name,
836 agent_path,
837 task,
838 model,
839 pid: 0,
840 status: RunStatus::Starting,
841 current_stage: String::new(),
842 stage_index: 0,
843 num_stages,
844 iteration: 0,
845 prompt_tokens: 0,
846 completion_tokens: 0,
847 cached_tokens: 0,
848 cache_write_tokens: 0,
849 tool_calls: 0,
850 // A run that has made no calls has spent nothing, and that
851 // zero IS known - unlike a run whose calls could not be priced.
852 cost_usd: Some(0.0),
853 unpriced_calls: 0,
854 cost_is_exact: true,
855 cost_priced_usd: 0.0,
856 workdir,
857 started_at: now,
858 updated_at: now,
859 last_progress_at: None,
860 active: None,
861 error: None,
862 title: None,
863 title_error: None,
864 metadata: HashMap::new(),
865 callback_url: None,
866 callback_secret: None,
867 parent_run_id: None,
868 children: Vec::new(),
869 depth: 0,
870 max_child_depth: 0,
871 final_output: None,
872 waiting_on: None,
873 output_request: None,
874 model_override: None,
875 flags: RunFlags::default(),
876 yolo: false,
877 yolo_profile: None,
878 read_paths: None,
879 }
880 }
881
882 /// How long ago this run was launched, at `now`.
883 ///
884 /// One of the three spans a reader can ask a run about, and they answer
885 /// different questions - keep them apart:
886 ///
887 /// - **age** ([`Self::age_secs`]): how long since it was launched. Says
888 /// nothing about whether it has done anything.
889 /// - **working** ([`Self::active_runtime_secs`]): how long it actually spent
890 /// working. This is the one to call a run's duration.
891 /// - **last moved** (from [`Self::last_progress_at`]): how long since it made
892 /// progress. A health signal, not a duration: it is how a wedged run is
893 /// told from a slow one. No accessor, because the surfaces that show it
894 /// read it off a live listing row rather than off a `RunMeta`.
895 ///
896 /// Every surface reads these rather than doing the arithmetic itself, so
897 /// `lev ps`, the dashboard and the HTTP API cannot disagree about what a run
898 /// has been doing.
899 pub fn age_secs(&self, now: i64) -> u64 {
900 crate::duration::between(self.started_at, now)
901 }
902
903 /// How long this run has actually been working, at `now`.
904 ///
905 /// This is the number to show as a run's duration. Wall-clock age answers a
906 /// different question, and answers it misleadingly: a run left paused, or
907 /// sitting on a question nobody has answered, kept climbing while nothing
908 /// was happening on its behalf. See the sibling spans on [`Self::age_secs`].
909 ///
910 /// A run written before the clock existed carries no spans at all, so it
911 /// falls back to the wall-clock span - a finished run that claims to have
912 /// taken no time is the worse answer of the two.
913 pub fn active_runtime_secs(&self, now: i64) -> u64 {
914 match self.active {
915 Some(clock) => clock.total_secs(now),
916 None => crate::duration::between(self.started_at, self.updated_at),
917 }
918 }
919
920 /// Stamp `updated_at` with the current time.
921 ///
922 /// Deliberately does **not** touch `last_progress_at`: the 30-second
923 /// persistence heartbeat calls this, and a run that is wedged must not look
924 /// like one that just moved. See [`RunMeta::last_progress_at`].
925 pub fn touch(&mut self) {
926 self.updated_at = crate::duration::now_secs();
927 }
928}
929
930/// One content entry within a region, captured at snapshot time.
931#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
932pub struct RegionEntrySnapshot {
933 /// The entry's parts, exactly as they sat in the live region. Reads as
934 /// text; a plain string in an older snapshot loads as one text part.
935 pub content: crate::region::EntryContent,
936 /// The entry's token cost as counted when it was added, carried through the
937 /// snapshot so a reload does not have to re-tokenize to rebuild budgets.
938 pub tokens: usize,
939 /// The entry's role/kind, so a snapshot round-trips faithfully when the
940 /// daemon reloads it on restart. Defaults to `Text` for older snapshots.
941 #[serde(default)]
942 pub kind: crate::region::EntryKind,
943 /// Free-form structured data an entry writer attached, passed through
944 /// untouched. Nothing in the engine interprets it.
945 #[serde(default, skip_serializing_if = "Option::is_none")]
946 pub metadata: Option<serde_json::Value>,
947 /// Key for HashMap region entries (file paths, section names, etc.)
948 #[serde(default, skip_serializing_if = "Option::is_none")]
949 pub key: Option<String>,
950 /// How sensitive this entry is.
951 ///
952 /// Persisted because taint was not, and a restore that dropped it silently
953 /// disarmed the gate: the reloaded run re-enabled taint tracking, found
954 /// every region back at `Public`, and let outbound tools through that had
955 /// been blocked a moment earlier. Any restart, crash-recovery, `resume`, or
956 /// page-in did it.
957 ///
958 /// Defaults to `Public` for snapshots written before this field existed -
959 /// the same value they were being restored with anyway, so nothing is worse
960 /// than it was, and new runs are correct from their first write.
961 #[serde(default)]
962 pub taint: crate::taint::TaintLevel,
963 /// The opaque provider token this turn has to be replayed with.
964 ///
965 /// Persisted for the same reason `taint` is: a restore that dropped it
966 /// would silently break reasoning continuity on a stateless backend, and
967 /// the run would look fine while paying to re-derive its chain of thought
968 /// every turn. Defaults to absent for snapshots written before the field,
969 /// which is what they were being restored with anyway.
970 #[serde(default, skip_serializing_if = "Option::is_none")]
971 pub reasoning: Option<String>,
972}
973
974/// Per-region token snapshot written by the background worker after each inference.
975#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
976pub struct RegionSnapshot {
977 /// The region's name, matching its key under `[context.regions]`.
978 pub name: String,
979 /// Stringified kind, spelled the way the blueprint spells it: `pinned`,
980 /// `temporary`, `clearable`, `sliding_window`, `compacting`,
981 /// `compact_history`, `hashmap`, `checklist`, `custom`.
982 ///
983 /// A snapshot written by an older build says `sliding` and `history` for
984 /// those two, and those files stay on disk, so a reader that renders this
985 /// accepts both spellings.
986 pub kind: String,
987 /// Tokens the region held when the snapshot was taken.
988 pub current_tokens: usize,
989 /// The region's ceiling at snapshot time, already resolved against the
990 /// model in front of it, so a percentage budget appears here as a number.
991 pub max_tokens: usize,
992 /// Actual content entries stored in this region (empty for zero-token regions).
993 #[serde(default, skip_serializing_if = "Vec::is_empty")]
994 pub entries: Vec<RegionEntrySnapshot>,
995 /// What the blueprint says this region is for, when it says.
996 ///
997 /// Carried on the snapshot so every reader of `context.json` can show it -
998 /// the dashboard, the history API, a console - rather than each having to
999 /// find and re-parse the manifest to explain a region it is already
1000 /// displaying.
1001 #[serde(default, skip_serializing_if = "Option::is_none")]
1002 pub description: Option<String>,
1003}
1004
1005/// Snapshot of the full context window, written to `context.json` alongside `meta.json`.
1006#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
1007pub struct ContextSnapshot {
1008 /// The stage the run was in when this was written.
1009 pub stage_name: String,
1010 /// Tokens held across every region, which is what the next request costs
1011 /// before the model's reply.
1012 pub total_tokens: usize,
1013 /// The whole window's budget, from the blueprint's `total_budget_tokens` or
1014 /// the model's own limit.
1015 pub max_tokens: usize,
1016 /// Every region, in layout order.
1017 pub regions: Vec<RegionSnapshot>,
1018}
1019
1020/// Current Unix time in seconds (saturating to 0 before the epoch).
1021#[cfg(test)]
1022mod tests {
1023 use super::*;
1024
1025 #[test]
1026 fn active_clock_banks_only_the_spans_it_was_running_for() {
1027 let mut clock = ActiveClock::default();
1028 clock.observe(100, true);
1029 assert_eq!(clock.since, Some(100));
1030 assert_eq!(clock.total_secs(140), 40, "the open span counts");
1031
1032 // Repeating the same answer is a no-op, so a per-tick call cannot
1033 // restart the span and lose the time already in it.
1034 clock.observe(120, true);
1035 assert_eq!(clock.since, Some(100));
1036
1037 clock.observe(140, false);
1038 assert_eq!(clock.banked_secs, 40);
1039 assert_eq!(clock.since, None);
1040 // Stopped: the reading no longer moves however long the caller waits.
1041 assert_eq!(clock.total_secs(9_999), 40);
1042
1043 clock.observe(1_000, true);
1044 assert_eq!(clock.total_secs(1_010), 50, "a resume adds to the bank");
1045 }
1046
1047 #[test]
1048 fn active_clock_ignores_a_clock_that_runs_backwards() {
1049 let mut clock = ActiveClock {
1050 banked_secs: 5,
1051 since: Some(500),
1052 };
1053 // An `as_of` before the span began (a corrected system clock) banks
1054 // nothing rather than wrapping the unsigned total.
1055 clock.settle(400);
1056 assert_eq!(clock.banked_secs, 5);
1057 assert_eq!(clock.since, None);
1058 assert_eq!(clock.total_secs(0), 5);
1059 }
1060
1061 #[test]
1062 fn settle_closes_an_open_span_at_the_moment_given() {
1063 let mut clock = ActiveClock {
1064 banked_secs: 10,
1065 since: Some(100),
1066 };
1067 // The daemon died at 130 and the run is reloaded much later: only the
1068 // 30 seconds it was actually up count.
1069 clock.settle(130);
1070 assert_eq!(clock.banked_secs, 40);
1071 assert_eq!(clock.total_secs(1_000_000), 40);
1072 }
1073
1074 #[test]
1075 fn clock_runs_while_working_or_held_for_the_runs_own_children() {
1076 // Working.
1077 assert!(clock_runs(&RunStatus::Starting, None));
1078 assert!(clock_runs(&RunStatus::Running, None));
1079 // Parked on work of its own: still the run taking as long as it takes.
1080 assert!(clock_runs(
1081 &RunStatus::WaitingInput,
1082 Some(&WaitReason::Children { outstanding: 2 })
1083 ));
1084 assert!(clock_runs(
1085 &RunStatus::WaitingInput,
1086 Some(&WaitReason::FanOutWorkers { outstanding: 5 })
1087 ));
1088 // Waiting on a person, in each of the ways it can happen.
1089 for reason in [
1090 WaitReason::ToolApproval,
1091 WaitReason::UserPrompt,
1092 WaitReason::TaintGate,
1093 WaitReason::InteractionPoint,
1094 WaitReason::NeedsSetup {
1095 blocker: SetupBlocker::CreditsExhausted,
1096 remedy: "top up".to_string(),
1097 },
1098 ] {
1099 assert!(
1100 !clock_runs(&RunStatus::WaitingInput, Some(&reason)),
1101 "{reason} should stop the clock"
1102 );
1103 }
1104 // An unclaimed wait is a wait: nothing is driving the run.
1105 assert!(!clock_runs(&RunStatus::WaitingInput, None));
1106 // Paused and every terminal state.
1107 for status in [
1108 RunStatus::Paused,
1109 RunStatus::Complete,
1110 RunStatus::CompleteInteractive,
1111 RunStatus::Error,
1112 RunStatus::Cancelled,
1113 ] {
1114 assert!(!clock_runs(&status, None), "{status} should stop the clock");
1115 }
1116 }
1117
1118 /// Age and working time are different questions, and a paused run is where
1119 /// they come apart.
1120 #[test]
1121 fn age_is_the_wall_clock_span_whatever_the_run_was_doing() {
1122 let mut meta = sample_meta();
1123 meta.started_at = 1_000;
1124 meta.active = Some(ActiveClock {
1125 banked_secs: 20,
1126 since: None,
1127 });
1128 assert_eq!(meta.age_secs(4_600), 3_600, "an hour old");
1129 assert_eq!(
1130 meta.active_runtime_secs(4_600),
1131 20,
1132 "twenty seconds of work"
1133 );
1134 // A clock corrected backwards reads as brand new, not as a huge age.
1135 assert_eq!(meta.age_secs(500), 0);
1136 }
1137
1138 #[test]
1139 fn active_runtime_secs_falls_back_to_the_wall_clock_span_when_unrecorded() {
1140 // A run written by a build that kept no clock: the wall-clock span is
1141 // a better answer than "this run took no time".
1142 let mut meta = sample_meta();
1143 meta.started_at = 1_000;
1144 meta.updated_at = 1_600;
1145 assert_eq!(meta.active_runtime_secs(9_000), 600);
1146
1147 // Once the clock exists it is the authority, and the wall-clock span
1148 // stops mattering.
1149 meta.active = Some(ActiveClock {
1150 banked_secs: 20,
1151 since: None,
1152 });
1153 assert_eq!(meta.active_runtime_secs(9_000), 20);
1154
1155 // And a run that genuinely took no time reports zero, rather than
1156 // falling back to a span that would invent one.
1157 meta.active = Some(ActiveClock::default());
1158 assert_eq!(meta.active_runtime_secs(9_000), 0);
1159 }
1160
1161 #[test]
1162 fn stage_active_runtime_secs_falls_back_the_same_way() {
1163 let mut rec = StageRecord::new("plan".to_string(), 0);
1164 // Never entered: no time, and no timestamp to guess from.
1165 assert_eq!(rec.active_runtime_secs(500), 0);
1166
1167 rec.started_at = Some(100);
1168 assert_eq!(rec.active_runtime_secs(160), 60, "still running: up to now");
1169 rec.ended_at = Some(130);
1170 assert_eq!(rec.active_runtime_secs(160), 30, "finished: up to the end");
1171
1172 rec.active = Some(ActiveClock {
1173 banked_secs: 7,
1174 since: None,
1175 });
1176 assert_eq!(rec.active_runtime_secs(160), 7);
1177 }
1178
1179 fn sample_meta() -> RunMeta {
1180 RunMeta::new(
1181 "run-1".to_string(),
1182 "agent".to_string(),
1183 "/agents/agent".to_string(),
1184 "do the thing".to_string(),
1185 Some("claude-sonnet-4-6".to_string()),
1186 "/work".to_string(),
1187 3,
1188 )
1189 }
1190
1191 /// `wire()` has to say exactly what serde says, because the two spellings
1192 /// reach the same client from different routes - one from a whole run
1193 /// serialized as JSON, the other from a route that builds its own `status`
1194 /// string. Derived from serde here rather than compared to a second hand
1195 /// written list, so a renamed variant fails this instead of shipping two
1196 /// words for one state.
1197 #[test]
1198 fn the_wire_word_is_the_word_serde_writes() {
1199 for status in [
1200 RunStatus::Starting,
1201 RunStatus::Running,
1202 RunStatus::WaitingInput,
1203 RunStatus::Complete,
1204 RunStatus::CompleteInteractive,
1205 RunStatus::Paused,
1206 RunStatus::Error,
1207 RunStatus::Cancelled,
1208 ] {
1209 let serde_word = serde_json::to_value(&status).unwrap();
1210 assert_eq!(serde_word, serde_json::json!(status.wire()));
1211 }
1212 }
1213
1214 /// The webhook signing secret must not survive into anything served over
1215 /// the API - an unredacted meta lets `GET /api/agents` hand it to any
1216 /// token holder.
1217 #[test]
1218 fn redacted_drops_the_callback_secret_and_keeps_everything_else() {
1219 let mut m = sample_meta();
1220 m.callback_secret = Some("shhh".to_string());
1221 m.callback_url = Some("https://example.com/hook".to_string());
1222
1223 let r = m.redacted();
1224 assert_eq!(r.callback_secret, None);
1225 // The URL is not a secret and stays: a caller needs to see where its own
1226 // webhook was pointed.
1227 assert_eq!(r.callback_url.as_deref(), Some("https://example.com/hook"));
1228 assert_eq!(r.run_id, m.run_id);
1229 assert_eq!(r.task, m.task);
1230
1231 // Serializing the redacted form must not mention it at all - a `None`
1232 // that still emitted `"callback_secret": null` would be fine, but an
1233 // assertion on the wire format is what a reviewer actually checks.
1234 let json = serde_json::to_string(&r).unwrap();
1235 assert!(!json.contains("shhh"), "{json}");
1236
1237 // ...and the original is untouched, because the daemon still needs it to
1238 // sign the webhook for a run it reloaded after a restart.
1239 assert_eq!(m.callback_secret.as_deref(), Some("shhh"));
1240 }
1241
1242 #[test]
1243 fn run_meta_new_sets_defaults() {
1244 let m = sample_meta();
1245 assert_eq!(m.run_id, "run-1");
1246 assert_eq!(m.agent_name, "agent");
1247 assert_eq!(m.agent_path, "/agents/agent");
1248 assert_eq!(m.task, "do the thing");
1249 assert_eq!(m.model.as_deref(), Some("claude-sonnet-4-6"));
1250 assert_eq!(m.workdir, "/work");
1251 assert_eq!(m.num_stages, 3);
1252 assert_eq!(m.pid, 0);
1253 assert_eq!(m.status, RunStatus::Starting);
1254 assert_eq!(m.stage_index, 0);
1255 assert_eq!(m.iteration, 0);
1256 assert_eq!(m.prompt_tokens, 0);
1257 assert_eq!(m.completion_tokens, 0);
1258 assert_eq!(m.cached_tokens, 0);
1259 assert_eq!(m.cache_write_tokens, 0);
1260 assert_eq!(m.tool_calls, 0);
1261 assert!(m.error.is_none());
1262 assert!(m.title.is_none());
1263 assert!(m.metadata.is_empty());
1264 assert!(m.callback_url.is_none());
1265 assert!(m.callback_secret.is_none());
1266 assert!(m.parent_run_id.is_none());
1267 assert!(m.children.is_empty());
1268 assert_eq!(m.depth, 0);
1269 assert_eq!(m.max_child_depth, 0);
1270 assert!(m.current_stage.is_empty());
1271 assert_eq!(m.started_at, m.updated_at);
1272 }
1273
1274 #[test]
1275 fn run_meta_touch_advances_updated_at() {
1276 let mut m = sample_meta();
1277 m.updated_at = 0;
1278 m.touch();
1279 assert!(m.updated_at > 0);
1280 }
1281
1282 /// A `meta.json` written before `waiting_on` existed still loads.
1283 ///
1284 /// This is the whole compatibility question for the field, and it is worth
1285 /// a test rather than a reading of the serde attributes: every run already
1286 /// on disk was written by a build that had never heard of it, and a
1287 /// deserialize that insisted on the key would make every one of them
1288 /// unreadable.
1289 #[test]
1290 fn a_run_written_before_waiting_on_existed_still_loads() {
1291 let mut original = sample_meta();
1292 original.status = RunStatus::WaitingInput;
1293 let mut value = serde_json::to_value(&original).unwrap();
1294 // Whatever the current build writes, an older file simply has no such
1295 // key. Removing it reproduces that exactly.
1296 value
1297 .as_object_mut()
1298 .expect("meta is an object")
1299 .remove("waiting_on");
1300 assert!(value.get("waiting_on").is_none(), "the old shape");
1301
1302 let back: RunMeta = serde_json::from_value(value).unwrap();
1303 assert_eq!(back.waiting_on, None);
1304 assert_eq!(back.status, RunStatus::WaitingInput);
1305 assert_eq!(back.run_id, original.run_id);
1306 }
1307
1308 /// A run that is not parked writes the file it always wrote, so an older
1309 /// build reading a newer run sees nothing it does not understand.
1310 #[test]
1311 fn a_run_that_is_not_parked_writes_no_waiting_on_key() {
1312 let mut m = sample_meta();
1313 m.status = RunStatus::Running;
1314 m.waiting_on = None;
1315 let json = serde_json::to_value(&m).unwrap();
1316 assert!(json.get("waiting_on").is_none(), "{json}");
1317
1318 m.waiting_on = Some(WaitReason::FanOutWorkers { outstanding: 3 });
1319 let json = serde_json::to_value(&m).unwrap();
1320 assert_eq!(
1321 json["waiting_on"],
1322 serde_json::json!({"reason": "fan_out_workers", "outstanding": 3})
1323 );
1324 }
1325
1326 /// Every variant is on the wire in snake_case, the way `RunStatus` is, and
1327 /// round-trips. The counted ones carry their number with them, which is
1328 /// what lets a client say "waiting on 3 of them" rather than "waiting".
1329 #[test]
1330 fn wait_reason_serializes_in_snake_case() {
1331 for (variant, wire) in [
1332 (WaitReason::ToolApproval, "tool_approval"),
1333 (WaitReason::UserPrompt, "user_prompt"),
1334 (WaitReason::TaintGate, "taint_gate"),
1335 (WaitReason::InteractionPoint, "interaction_point"),
1336 (
1337 WaitReason::FanOutWorkers { outstanding: 3 },
1338 "fan_out_workers",
1339 ),
1340 (WaitReason::Children { outstanding: 1 }, "children"),
1341 ] {
1342 let json = serde_json::to_value(&variant).unwrap();
1343 assert_eq!(json["reason"], serde_json::json!(wire));
1344 let back: WaitReason = serde_json::from_value(json).unwrap();
1345 assert_eq!(back, variant);
1346 }
1347 }
1348
1349 /// Each marker names its own reason, and the counted ones carry the count.
1350 ///
1351 /// The precedence is the specific claim first: a taint gate and a
1352 /// checkpoint each open a hub request of their own, so a generic-prompt
1353 /// answer would swallow both.
1354 #[test]
1355 fn each_marker_names_its_own_reason_specific_first() {
1356 let cases = [
1357 (
1358 WaitMarkers {
1359 gate_prompt: true,
1360 awaiting_interaction: true,
1361 ..Default::default()
1362 },
1363 WaitReason::TaintGate,
1364 ),
1365 (
1366 WaitMarkers {
1367 interaction_point: true,
1368 awaiting_interaction: true,
1369 ..Default::default()
1370 },
1371 WaitReason::InteractionPoint,
1372 ),
1373 (
1374 WaitMarkers {
1375 fan_out_outstanding: Some(5),
1376 ..Default::default()
1377 },
1378 WaitReason::FanOutWorkers { outstanding: 5 },
1379 ),
1380 (
1381 WaitMarkers {
1382 children_outstanding: Some(4),
1383 ..Default::default()
1384 },
1385 WaitReason::Children { outstanding: 4 },
1386 ),
1387 ];
1388 for (markers, expected) in cases {
1389 assert_eq!(
1390 wait_reason_from(true, &markers),
1391 Some(expected),
1392 "{markers:?}"
1393 );
1394 }
1395 // A parent holding both kinds of sub-work reports the more specific one.
1396 assert_eq!(
1397 wait_reason_from(
1398 true,
1399 &WaitMarkers {
1400 fan_out_outstanding: Some(2),
1401 children_outstanding: Some(9),
1402 ..Default::default()
1403 }
1404 ),
1405 Some(WaitReason::FanOutWorkers { outstanding: 2 })
1406 );
1407 }
1408
1409 /// A run parked until the machine is fixed says so before anything else.
1410 ///
1411 /// It outranks every other marker on purpose: answering a prompt does not
1412 /// help a run whose provider is not configured, so sending someone to the
1413 /// prompt would be sending them to the wrong screen.
1414 #[test]
1415 fn needing_setup_outranks_every_other_reason() {
1416 let need = SetupNeeded {
1417 blocker: SetupBlocker::ProviderMissing,
1418 remedy: "add it to config.toml".to_string(),
1419 };
1420 let reason = wait_reason_from(
1421 true,
1422 &WaitMarkers {
1423 needs_setup: Some(need.clone()),
1424 // Everything else at once, so precedence is being tested
1425 // rather than the absence of competition.
1426 gate_prompt: true,
1427 interaction_point: true,
1428 fan_out_outstanding: Some(2),
1429 children_outstanding: Some(3),
1430 awaiting_interaction: true,
1431 interaction: Some(crate::interaction::InteractionKind::ToolApproval),
1432 },
1433 );
1434 assert_eq!(
1435 reason,
1436 Some(WaitReason::NeedsSetup {
1437 blocker: SetupBlocker::ProviderMissing,
1438 remedy: "add it to config.toml".to_string(),
1439 })
1440 );
1441 assert!(
1442 reason.unwrap().needs_a_person(),
1443 "nothing resolves this without somebody"
1444 );
1445 }
1446
1447 /// Each blocker is its own value on the wire, so a console can offer the
1448 /// right remedy instead of matching on the sentence.
1449 #[test]
1450 fn every_blocker_has_its_own_wire_name_and_label() {
1451 for (blocker, wire, label) in [
1452 (
1453 SetupBlocker::ProviderMissing,
1454 "provider_missing",
1455 "provider",
1456 ),
1457 (
1458 SetupBlocker::CreditsExhausted,
1459 "credits_exhausted",
1460 "credits",
1461 ),
1462 (SetupBlocker::AuthFailed, "auth_failed", "key"),
1463 (SetupBlocker::Forbidden, "forbidden", "access"),
1464 (
1465 SetupBlocker::ProvidersUnavailable,
1466 "providers_unavailable",
1467 "providers",
1468 ),
1469 (
1470 SetupBlocker::ProviderUnreachable,
1471 "provider_unreachable",
1472 "unreachable",
1473 ),
1474 (
1475 SetupBlocker::ProviderTimedOut,
1476 "provider_timed_out",
1477 "timed out",
1478 ),
1479 (SetupBlocker::ProviderFailed, "provider_failed", "failed"),
1480 ] {
1481 assert_eq!(serde_json::to_value(blocker).unwrap(), wire);
1482 assert_eq!(blocker.to_string(), label);
1483 let back: SetupBlocker = serde_json::from_value(serde_json::json!(wire)).unwrap();
1484 assert_eq!(back, blocker);
1485 // The row renders the kind, not the sentence: a remedy is a
1486 // sentence and this is a table cell. A blocker that describes the
1487 // provider says what happened to it rather than what is needed.
1488 let lead = match blocker {
1489 SetupBlocker::ProviderUnreachable
1490 | SetupBlocker::ProviderTimedOut
1491 | SetupBlocker::ProviderFailed => "provider",
1492 _ => "needs",
1493 };
1494 assert_eq!(
1495 WaitReason::NeedsSetup {
1496 blocker,
1497 remedy: "a whole sentence that would not fit".to_string(),
1498 }
1499 .to_string(),
1500 format!("{lead} {label}")
1501 );
1502 }
1503 }
1504
1505 /// A generic hub block reports what kind of prompt it is, so "approve this
1506 /// tool call" and "answer this question" are not the same row.
1507 #[test]
1508 fn a_hub_block_reports_the_kind_of_prompt_holding_it() {
1509 let held = |kind| WaitMarkers {
1510 awaiting_interaction: true,
1511 interaction: kind,
1512 ..Default::default()
1513 };
1514 assert_eq!(
1515 wait_reason_from(
1516 true,
1517 &held(Some(crate::interaction::InteractionKind::ToolApproval))
1518 ),
1519 Some(WaitReason::ToolApproval)
1520 );
1521 // Anything else the agent asked for is a question for a person. The
1522 // kind can also be unknown while the block is real, which reads the
1523 // same way: somebody is being waited on.
1524 assert_eq!(
1525 wait_reason_from(
1526 true,
1527 &held(Some(crate::interaction::InteractionKind::FreeText))
1528 ),
1529 Some(WaitReason::UserPrompt)
1530 );
1531 assert_eq!(
1532 wait_reason_from(true, &held(None)),
1533 Some(WaitReason::UserPrompt)
1534 );
1535 }
1536
1537 /// Parked with nothing claiming it: the field is left off rather than
1538 /// filled with a guess, and a run that is not parked never has one.
1539 #[test]
1540 fn nothing_claiming_a_parked_run_reports_no_reason() {
1541 assert_eq!(wait_reason_from(true, &WaitMarkers::default()), None);
1542 assert_eq!(
1543 wait_reason_from(
1544 false,
1545 &WaitMarkers {
1546 gate_prompt: true,
1547 ..Default::default()
1548 }
1549 ),
1550 None,
1551 "a run that is not waiting is not waiting on anything"
1552 );
1553 }
1554
1555 /// The rendered form every text surface uses, counts included. Narrow
1556 /// enough for a table column, which is why it is not the variant name.
1557 #[test]
1558 fn every_reason_renders_for_a_narrow_column() {
1559 assert_eq!(WaitReason::ToolApproval.to_string(), "tool approval");
1560 assert_eq!(WaitReason::UserPrompt.to_string(), "user prompt");
1561 assert_eq!(WaitReason::TaintGate.to_string(), "taint gate");
1562 assert_eq!(WaitReason::InteractionPoint.to_string(), "checkpoint");
1563 assert_eq!(
1564 WaitReason::FanOutWorkers { outstanding: 3 }.to_string(),
1565 "workers(3)"
1566 );
1567 assert_eq!(
1568 WaitReason::Children { outstanding: 2 }.to_string(),
1569 "children(2)"
1570 );
1571 }
1572
1573 /// Only the two engine-side reasons resolve on their own; the rest are a
1574 /// person's to clear. This is the predicate a badge should be built on.
1575 #[test]
1576 fn only_the_engine_side_reasons_need_nobody() {
1577 assert!(WaitReason::ToolApproval.needs_a_person());
1578 assert!(WaitReason::UserPrompt.needs_a_person());
1579 assert!(WaitReason::TaintGate.needs_a_person());
1580 assert!(WaitReason::InteractionPoint.needs_a_person());
1581 assert!(!WaitReason::FanOutWorkers { outstanding: 2 }.needs_a_person());
1582 assert!(!WaitReason::Children { outstanding: 2 }.needs_a_person());
1583 }
1584
1585 #[test]
1586 fn run_meta_serde_roundtrip() {
1587 let mut m = sample_meta();
1588 m.status = RunStatus::Running;
1589 m.metadata.insert("k".to_string(), "v".to_string());
1590 m.title = Some("A title".to_string());
1591 m.callback_secret = Some("shh".to_string());
1592 m.parent_run_id = Some("parent-1".to_string());
1593 m.children = vec!["child-a".to_string(), "child-b".to_string()];
1594 m.depth = 2;
1595 m.max_child_depth = 5;
1596 let json = serde_json::to_string(&m).unwrap();
1597 let back: RunMeta = serde_json::from_str(&json).unwrap();
1598 assert_eq!(back.run_id, m.run_id);
1599 assert_eq!(back.status, RunStatus::Running);
1600 assert_eq!(back.metadata.get("k").map(String::as_str), Some("v"));
1601 assert_eq!(back.title.as_deref(), Some("A title"));
1602 assert_eq!(back.callback_secret.as_deref(), Some("shh"));
1603 assert_eq!(back.parent_run_id.as_deref(), Some("parent-1"));
1604 assert_eq!(
1605 back.children,
1606 vec!["child-a".to_string(), "child-b".to_string()]
1607 );
1608 assert_eq!(back.depth, 2);
1609 assert_eq!(back.max_child_depth, 5);
1610 }
1611
1612 #[test]
1613 fn run_status_display_all_variants() {
1614 assert_eq!(RunStatus::Starting.to_string(), "Starting");
1615 assert_eq!(RunStatus::Running.to_string(), "Running");
1616 assert_eq!(RunStatus::WaitingInput.to_string(), "WaitingInput");
1617 assert_eq!(RunStatus::Complete.to_string(), "Complete");
1618 assert_eq!(
1619 RunStatus::CompleteInteractive.to_string(),
1620 "CompleteInteractive"
1621 );
1622 assert_eq!(RunStatus::Paused.to_string(), "Paused");
1623 assert_eq!(RunStatus::Error.to_string(), "Error");
1624 assert_eq!(RunStatus::Cancelled.to_string(), "Cancelled");
1625 }
1626
1627 #[test]
1628 fn run_status_serde_snake_case_roundtrip() {
1629 for s in [
1630 RunStatus::Starting,
1631 RunStatus::Running,
1632 RunStatus::WaitingInput,
1633 RunStatus::Complete,
1634 RunStatus::CompleteInteractive,
1635 RunStatus::Paused,
1636 RunStatus::Error,
1637 RunStatus::Cancelled,
1638 ] {
1639 let json = serde_json::to_string(&s).unwrap();
1640 let back: RunStatus = serde_json::from_str(&json).unwrap();
1641 assert_eq!(back, s);
1642 }
1643 assert_eq!(
1644 serde_json::to_string(&RunStatus::WaitingInput).unwrap(),
1645 "\"waiting_input\""
1646 );
1647 assert_eq!(
1648 serde_json::to_string(&RunStatus::Paused).unwrap(),
1649 "\"paused\""
1650 );
1651 }
1652
1653 #[test]
1654 fn context_snapshot_serde_roundtrip() {
1655 let snap = ContextSnapshot {
1656 stage_name: "plan".to_string(),
1657 total_tokens: 42,
1658 max_tokens: 100,
1659 regions: vec![RegionSnapshot {
1660 name: "history".to_string(),
1661 kind: "sliding".to_string(),
1662 current_tokens: 10,
1663 max_tokens: 50,
1664 entries: vec![RegionEntrySnapshot {
1665 content: "hi".into(),
1666 tokens: 1,
1667 kind: crate::region::EntryKind::UserMessage,
1668 metadata: Some(serde_json::json!({"a": 1})),
1669 key: Some("k".to_string()),
1670 taint: Default::default(),
1671 reasoning: None,
1672 }],
1673 description: None,
1674 }],
1675 };
1676 let json = serde_json::to_string(&snap).unwrap();
1677 let back: ContextSnapshot = serde_json::from_str(&json).unwrap();
1678 assert_eq!(back.stage_name, "plan");
1679 assert_eq!(back.regions.len(), 1);
1680 assert_eq!(back.regions[0].entries.len(), 1);
1681 assert_eq!(back.regions[0].entries[0].content, "hi");
1682 assert_eq!(back.regions[0].entries[0].key.as_deref(), Some("k"));
1683 }
1684
1685 #[test]
1686 fn region_snapshot_skips_empty_entries_in_json() {
1687 let snap = RegionSnapshot {
1688 name: "r".to_string(),
1689 kind: "pinned".to_string(),
1690 current_tokens: 0,
1691 max_tokens: 0,
1692 entries: vec![],
1693 description: None,
1694 };
1695 let json = serde_json::to_string(&snap).unwrap();
1696 assert!(!json.contains("entries"));
1697 }
1698
1699 #[test]
1700 fn stage_run_status_display_all_variants() {
1701 assert_eq!(StageRunStatus::Pending.to_string(), "Pending");
1702 assert_eq!(StageRunStatus::Skipped.to_string(), "Skipped");
1703 assert_eq!(StageRunStatus::Active.to_string(), "Active");
1704 assert_eq!(StageRunStatus::WaitingInput.to_string(), "WaitingInput");
1705 assert_eq!(StageRunStatus::Complete.to_string(), "Complete");
1706 assert_eq!(StageRunStatus::Error.to_string(), "Error");
1707 }
1708
1709 #[test]
1710 fn run_flags_record_modification_dedups_paths_and_caps_the_list() {
1711 let mut flags = RunFlags::default();
1712 flags.record_modification("src/a.rs");
1713 flags.record_modification("src/a.rs");
1714 flags.record_modification("src/b.rs");
1715 assert_eq!(flags.modified_file_count, 3);
1716 assert_eq!(flags.modified_files, vec!["src/a.rs", "src/b.rs"]);
1717
1718 // Past the cap the count keeps rising but the list stops growing, so a
1719 // long run can't bloat meta.json.
1720 for i in 0..MAX_TRACKED_MODIFIED_FILES {
1721 flags.record_modification(&format!("f{i}.rs"));
1722 }
1723 assert_eq!(flags.modified_files.len(), MAX_TRACKED_MODIFIED_FILES);
1724 assert_eq!(flags.modified_file_count, 3 + MAX_TRACKED_MODIFIED_FILES);
1725 }
1726
1727 #[test]
1728 fn note_modified_path_joins_the_list_without_touching_the_call_count() {
1729 let mut flags = RunFlags::default();
1730 // A modifying tool call: counts and lists.
1731 flags.record_modification("plot_chart.py");
1732 // A shell creation, noted from a workdir scan: lists, but is not a
1733 // modifying tool call, so the count stays 1 - and a second scan that
1734 // finds it again neither re-adds nor re-counts.
1735 assert!(flags.note_modified_path("chart.png"));
1736 assert!(!flags.note_modified_path("chart.png"));
1737 assert_eq!(flags.modified_file_count, 1);
1738 assert_eq!(flags.modified_files, vec!["plot_chart.py", "chart.png"]);
1739
1740 // Past the cap it stops adding and says so.
1741 let mut full = RunFlags::default();
1742 for i in 0..MAX_TRACKED_MODIFIED_FILES {
1743 assert!(full.note_modified_path(&format!("f{i}.png")));
1744 }
1745 assert!(!full.note_modified_path("one-too-many.png"));
1746 assert_eq!(full.modified_files.len(), MAX_TRACKED_MODIFIED_FILES);
1747 assert_eq!(full.modified_file_count, 0);
1748 }
1749
1750 #[test]
1751 fn run_meta_flags_default_for_older_files() {
1752 // A meta.json written before `flags` existed has no such key at all.
1753 let mut meta = RunMeta::new(
1754 "r".to_string(),
1755 "a".to_string(),
1756 "/p".to_string(),
1757 "t".to_string(),
1758 None,
1759 "/w".to_string(),
1760 1,
1761 );
1762 meta.flags.empty_output = true;
1763 // Drop the key structurally rather than by string surgery: a literal
1764 // spelling of the serialized flags silently stops matching the moment a
1765 // field is added, and the test then passes for the wrong reason.
1766 let mut json = serde_json::to_value(&meta).unwrap();
1767 json.as_object_mut().unwrap().remove("flags").unwrap();
1768 assert!(!json.to_string().contains("flags"));
1769 let back: RunMeta = serde_json::from_value(json).unwrap();
1770 assert_eq!(back.flags, RunFlags::default());
1771 }
1772}