Skip to main content

agent_top_core/
model.rs

1//! The data model shared by discovery, the TUI and the JSON output.
2
3use serde::{Deserialize, Serialize};
4use std::path::PathBuf;
5use std::time::SystemTime;
6
7/// Which agent harness a process or transcript belongs to.
8#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
9#[serde(rename_all = "kebab-case")]
10pub enum Harness {
11    Claude,
12    Codex,
13    Gemini,
14    OpenCode,
15    Kodelet,
16    Aider,
17    Copilot,
18    Cursor,
19    Unknown,
20}
21
22impl Harness {
23    pub fn label(self) -> &'static str {
24        match self {
25            Harness::Claude => "claude",
26            Harness::Codex => "codex",
27            Harness::Gemini => "gemini",
28            Harness::OpenCode => "opencode",
29            Harness::Kodelet => "kodelet",
30            Harness::Aider => "aider",
31            Harness::Copilot => "copilot",
32            Harness::Cursor => "cursor",
33            Harness::Unknown => "unknown",
34        }
35    }
36
37    /// How much of a session id identifies it to `agent-top trace --session`,
38    /// which matches a prefix against the ids of every harness and fails when
39    /// one matches more than one session. `None` means the whole id is needed:
40    /// Kodelet's begin with a date that every session started that day shares.
41    pub fn session_id_prefix_len(self) -> Option<usize> {
42        match self {
43            Harness::Kodelet => None,
44            Harness::Claude
45            | Harness::Codex
46            | Harness::Gemini
47            | Harness::OpenCode
48            | Harness::Aider
49            | Harness::Copilot
50            | Harness::Cursor
51            | Harness::Unknown => Some(8),
52        }
53    }
54}
55
56/// Coarse lifecycle state, in the htop sense.
57///
58/// `Running` means the agent is mid-turn (inference or tool execution),
59/// `Idle` means the process is alive but waiting for a human, `Stopped` means
60/// the transcript exists and was recently active but no process owns it.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
62#[serde(rename_all = "kebab-case")]
63pub enum AgentState {
64    Running,
65    Idle,
66    Stopped,
67}
68
69impl AgentState {
70    pub fn label(self) -> &'static str {
71        match self {
72            AgentState::Running => "running",
73            AgentState::Idle => "idle",
74            AgentState::Stopped => "stopped",
75        }
76    }
77}
78
79/// What the transcript says the agent was last doing. Harness-neutral.
80#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
81#[serde(rename_all = "kebab-case")]
82pub enum Activity {
83    /// Mid-turn: a prompt or tool result was just submitted, or a tool call is pending.
84    Working,
85    /// The last thing that happened was the assistant ending its turn.
86    Waiting,
87    #[default]
88    Unknown,
89}
90
91/// Token counts, split the way the Anthropic and OpenAI usage objects split them.
92#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
93pub struct TokenUsage {
94    pub input: u64,
95    pub cache_write_5m: u64,
96    pub cache_write_1h: u64,
97    /// Cache creation a harness recorded without the TTL that decides its
98    /// rate. No price table can cover it: charging either TTL's rate would be
99    /// a guess about a lifetime nobody wrote down. Only a harness that reports
100    /// its own costs can put a number against these tokens.
101    #[serde(default)]
102    pub cache_write_unsplit: u64,
103    pub cache_read: u64,
104    pub output: u64,
105}
106
107impl TokenUsage {
108    pub fn cache_write(&self) -> u64 {
109        self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit
110    }
111
112    /// Everything the model consumed or produced. This is the "TOKENS" column.
113    pub fn total(&self) -> u64 {
114        self.input + self.cache_write() + self.cache_read + self.output
115    }
116
117    /// The input side: everything sent to the model as the prompt, fresh input
118    /// plus cache writes plus cache reads, but not the output it produced.
119    pub fn prompt(&self) -> u64 {
120        self.input + self.cache_write() + self.cache_read
121    }
122
123    /// The share of the prompt served from cache (the cheap reads), in `0..=1`.
124    /// A high number means most of the re-sent conversation was billed at the
125    /// cache-read rate rather than full input; a low one on a long session is
126    /// money left on the table. `None` when there was no prompt to judge, or
127    /// the model does not cache at all.
128    pub fn cache_hit_rate(&self) -> Option<f64> {
129        let p = self.prompt();
130        (p > 0).then(|| self.cache_read as f64 / p as f64)
131    }
132
133    pub fn add(&mut self, other: &TokenUsage) {
134        self.input += other.input;
135        self.cache_write_5m += other.cache_write_5m;
136        self.cache_write_1h += other.cache_write_1h;
137        self.cache_write_unsplit += other.cache_write_unsplit;
138        self.cache_read += other.cache_read;
139        self.output += other.output;
140    }
141
142    pub fn sub(&mut self, other: &TokenUsage) {
143        self.input = self.input.saturating_sub(other.input);
144        self.cache_write_5m = self.cache_write_5m.saturating_sub(other.cache_write_5m);
145        self.cache_write_1h = self.cache_write_1h.saturating_sub(other.cache_write_1h);
146        self.cache_write_unsplit = self.cache_write_unsplit.saturating_sub(other.cache_write_unsplit);
147        self.cache_read = self.cache_read.saturating_sub(other.cache_read);
148        self.output = self.output.saturating_sub(other.output);
149    }
150}
151
152/// What a span measures. Tool calls came first and gave the type its name;
153/// the other two label the time between them.
154#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, PartialOrd, Ord)]
155#[serde(rename_all = "kebab-case")]
156pub enum SpanKind {
157    /// One tool call, from the harness issuing it to the result coming back.
158    #[default]
159    Tool,
160    /// The model producing a response: from a prompt or tool results being
161    /// submitted to the last block of the reply being written. The gap in a
162    /// waterfall that is not a tool call is almost always one of these.
163    Inference,
164    /// One human turn: from the prompt to the model ending its reply.
165    /// Contains every tool call and inference span issued in between.
166    Turn,
167}
168
169impl SpanKind {
170    pub const ALL: [SpanKind; 3] = [SpanKind::Tool, SpanKind::Inference, SpanKind::Turn];
171
172    pub fn label(self) -> &'static str {
173        match self {
174            SpanKind::Tool => "tool",
175            SpanKind::Inference => "inference",
176            SpanKind::Turn => "turn",
177        }
178    }
179}
180
181/// One span of an agent trace: a tool call, an inference, or a turn.
182///
183/// Every harness writes the same shape in its own vocabulary — Claude pairs a
184/// `tool_use` block with a `tool_result` block by `tool_use_id`, Codex pairs a
185/// `function_call` with a `function_call_output` by `call_id` — and both stamp
186/// each line with a timestamp. That is a span: a name, a start and a duration.
187/// Only the call's metadata is kept; arguments and output are never read.
188///
189/// The name predates `kind`: the type carried only tool calls until 0.3.1 and
190/// is kept for the sake of the published API.
191#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
192pub struct ToolSpan {
193    /// The harness's own call id, so a span survives across refreshes. For
194    /// inference and turn spans, a counter the parser assigns.
195    pub id: String,
196    /// Tool name as the harness reports it (`Bash`, `exec_command`, ...), or
197    /// `inference` / `turn`.
198    pub name: String,
199    pub started_at: SystemTime,
200    /// Wall-clock duration, or `None` while the call is still in flight.
201    pub duration_ms: Option<u64>,
202    /// The call was issued by a subagent (Claude's `isSidechain`).
203    pub sidechain: bool,
204    /// The harness reported the result as an error.
205    pub error: bool,
206    /// Absent in snapshots written before 0.3.1, which held tool calls only.
207    #[serde(default)]
208    pub kind: SpanKind,
209}
210
211impl ToolSpan {
212    pub fn is_open(&self) -> bool {
213        self.duration_ms.is_none()
214    }
215
216    /// Duration if closed, otherwise how long it has been running as of `now`.
217    pub fn elapsed_ms(&self, now: SystemTime) -> u64 {
218        match self.duration_ms {
219            Some(ms) => ms,
220            None => now.duration_since(self.started_at).map(|d| d.as_millis() as u64).unwrap_or(0),
221        }
222    }
223
224    pub fn ended_at(&self, now: SystemTime) -> SystemTime {
225        self.started_at + std::time::Duration::from_millis(self.elapsed_ms(now))
226    }
227}
228
229/// One MCP server an agent uses, seen from either side or both: the process
230/// table has the server's pid, CPU and memory; the transcript has how often
231/// the agent called it. Claude Code names an MCP tool `mcp__<server>__<tool>`,
232/// which is where the server name and the call count come from.
233#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
234pub struct McpServer {
235    /// The server's name as the harness configured it, or, for a process the
236    /// transcript never named, the program it is running.
237    pub name: String,
238    pub pid: Option<u32>,
239    pub cmdline: Option<String>,
240    pub cpu_percent: f32,
241    pub rss_bytes: u64,
242    pub age_secs: Option<u64>,
243    /// Tool calls the agent made to this server, from the transcript.
244    pub calls: u64,
245    /// Of those, how many the harness reported as errors.
246    pub errors: u64,
247    pub last_call: Option<SystemTime>,
248    /// How the process and the transcript's server were put together, for
249    /// the UI to label a guess as one.
250    pub matched_by: McpMatch,
251}
252
253/// How an `McpServer` row was formed.
254#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
255#[serde(rename_all = "kebab-case")]
256pub enum McpMatch {
257    /// A process under the agent that no transcript server names. Either it
258    /// has not been called yet, or its name does not appear in its command.
259    ProcessOnly,
260    /// A server the transcript calls with no process under the agent: an HTTP
261    /// server, or one that has exited.
262    TranscriptOnly,
263    /// The server's name appears in the process's command line.
264    Name,
265    /// One unmatched process and one unmatched server were left; they are
266    /// taken to be the same. A guess.
267    Sole,
268}
269
270/// Where a stretch of a session's context came from.
271#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
272#[serde(rename_all = "kebab-case")]
273pub enum ContextOrigin {
274    /// A built-in tool of the harness (`Bash`, `Read`, `exec_command`, ...).
275    Tool,
276    /// An MCP server; `name` is the server, not the tool.
277    Mcp,
278    /// Everything that is not a tool result: the system prompt, the user's
279    /// own messages, a compaction summary.
280    #[default]
281    Other,
282}
283
284/// One source of a session's context: what its results have added to the
285/// prompt, and what sending that on every inference since has cost.
286///
287/// The tokens are the growth of the prompt between one response and the
288/// next, attributed to the tool results submitted in between; the cost is
289/// those tokens charged at the prompt rate of every response that carried
290/// them, the first read included. The rows sum to the session's prompt-side
291/// cost. See `docs/accounting.md`.
292#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
293pub struct ContextSource {
294    pub name: String,
295    pub origin: ContextOrigin,
296    /// Tool results attributed to this source. Zero for `Other`.
297    pub calls: u64,
298    /// Tokens the source added to the prompt over the session.
299    pub tokens: u64,
300    /// USD those tokens have cost across every response that read them.
301    /// An estimate; zero when the model has no price.
302    pub cost_usd: f64,
303}
304
305/// Role of a process inside an agent's tree.
306#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
307#[serde(rename_all = "kebab-case")]
308pub enum ProcKind {
309    /// A harness process, whether at the root or nested under another process.
310    /// Older snapshots called nested harness processes `subagent`.
311    #[serde(alias = "subagent")]
312    Agent,
313    /// A Model Context Protocol server or helper.
314    Mcp,
315    /// A shell spawned to run a tool call.
316    Shell,
317    /// Anything else the agent launched (test runners, sleep, caffeinate, ...).
318    Tool,
319}
320
321impl ProcKind {
322    pub fn label(self) -> &'static str {
323        match self {
324            ProcKind::Agent => "agent",
325            ProcKind::Mcp => "mcp",
326            ProcKind::Shell => "shell",
327            ProcKind::Tool => "tool",
328        }
329    }
330}
331
332/// One process, with its descendants.
333#[derive(Debug, Clone, Serialize, Deserialize)]
334pub struct ProcNode {
335    pub pid: u32,
336    pub ppid: Option<u32>,
337    pub name: String,
338    pub cmdline: String,
339    pub kind: ProcKind,
340    pub harness: Option<Harness>,
341    pub cpu_percent: f32,
342    pub rss_bytes: u64,
343    pub age_secs: u64,
344    pub cwd: Option<PathBuf>,
345    pub children: Vec<ProcNode>,
346}
347
348impl ProcNode {
349    /// CPU, RSS, process count and MCP server count summed over the whole
350    /// subtree. An MCP process's own children (an `npx` wrapper's `node`) are
351    /// the same server, so they add to the process count and not to the
352    /// server count.
353    pub fn totals(&self) -> (f32, u64, usize, usize) {
354        let mut cpu = self.cpu_percent;
355        let mut rss = self.rss_bytes;
356        let mut count = 1;
357        let mut mcp = usize::from(self.kind == ProcKind::Mcp);
358        for c in &self.children {
359            let (ccpu, crss, ccount, cmcp) = c.totals();
360            cpu += ccpu;
361            rss += crss;
362            count += ccount;
363            if self.kind != ProcKind::Mcp {
364                mcp += cmcp;
365            }
366        }
367        (cpu, rss, count, mcp)
368    }
369
370    /// The MCP servers in this tree: each `Mcp` node whose parent is not one.
371    /// A server started through `npx` or `uvx` is two or three processes; the
372    /// top one stands for the server.
373    pub fn mcp_roots(&self) -> Vec<&ProcNode> {
374        let mut out = Vec::new();
375        self.collect_mcp_roots(&mut out);
376        out
377    }
378
379    fn collect_mcp_roots<'a>(&'a self, out: &mut Vec<&'a ProcNode>) {
380        if self.kind == ProcKind::Mcp {
381            out.push(self);
382            return;
383        }
384        for c in &self.children {
385            c.collect_mcp_roots(out);
386        }
387    }
388
389    pub fn walk<'a>(&'a self, depth: usize, f: &mut dyn FnMut(&'a ProcNode, usize)) {
390        f(self, depth);
391        for c in &self.children {
392            c.walk(depth + 1, f);
393        }
394    }
395}
396
397/// A logical subagent relationship explicitly recorded by the harness.
398/// This is session lineage, not an OS process relationship or a history fork.
399#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
400pub struct SubagentInfo {
401    pub parent_session_id: String,
402    #[serde(default)]
403    pub nickname: Option<String>,
404    #[serde(default)]
405    pub role: Option<String>,
406}
407
408/// A single row in the agent table.
409#[derive(Debug, Clone, Serialize, Deserialize)]
410pub struct Agent {
411    /// Stable identity across refreshes: `pid:<n>` for live agents, `session:<id>` for stopped ones.
412    pub id: String,
413    pub name: String,
414    pub harness: Harness,
415    pub state: AgentState,
416    pub activity: Activity,
417    pub pid: Option<u32>,
418    pub session_id: Option<String>,
419    /// Explicit logical parent and identity, when the transcript records them.
420    /// Usage remains per session; adding hierarchy must not count it twice.
421    #[serde(default, skip_serializing_if = "Option::is_none")]
422    pub subagent: Option<SubagentInfo>,
423    pub session_path: Option<PathBuf>,
424    pub cwd: Option<PathBuf>,
425    pub model: Option<String>,
426    pub harness_version: Option<String>,
427    pub usage: TokenUsage,
428    /// USD spent on messages whose model had a known price.
429    pub cost_usd: f64,
430    /// `cost_usd` by kind of token, so a figure that differs from another
431    /// tool's can be traced to the one line that differs.
432    #[serde(default)]
433    pub cost_breakdown: CostBreakdown,
434    /// Where this row's costs came from; `None` when it has no known price.
435    #[serde(default)]
436    pub price_source: Option<PriceSource>,
437    /// Tokens on messages whose model had no known price (so `cost_usd` is a floor).
438    pub unpriced_tokens: u64,
439    pub turns: u64,
440    pub subagent_turns: u64,
441    /// See `SessionSummary::folds_child_usage`. Claude Code, Gemini CLI and
442    /// OpenCode fold a child's usage into its parent, the way those harnesses
443    /// bill it; Codex and Kodelet give each child its own row instead.
444    #[serde(default)]
445    pub folds_child_usage: bool,
446    pub tool_calls: u64,
447    /// The count is only known retained/observed calls, not an exact lifetime
448    /// total. Kodelet compaction discards old calls without a cumulative count.
449    #[serde(default)]
450    pub tool_calls_lower_bound: bool,
451    /// Server-side web searches the model ran, billed per search on top of
452    /// tokens. Counted for every harness; priced only where the price table
453    /// has a rate (Anthropic's, for Claude Code).
454    #[serde(default)]
455    pub web_searches: u64,
456    /// The most recent spans, oldest first: tool calls, inferences and turns.
457    /// Bounded; see `harness::MAX_SPANS`.
458    pub spans: Vec<ToolSpan>,
459    /// Seconds since the process started (live) or since the last transcript write (stopped).
460    pub age_secs: u64,
461    /// Seconds since the transcript was last written.
462    pub idle_secs: Option<u64>,
463    pub cpu_percent: f32,
464    pub rss_bytes: u64,
465    pub process_count: usize,
466    pub mcp_count: usize,
467    /// The MCP servers this agent uses, one row each, from the process tree
468    /// and the transcript. See `McpServer`.
469    #[serde(default)]
470    pub mcp_servers: Vec<McpServer>,
471    /// What each tool's results added to the prompt and what carrying it has
472    /// cost, largest first. See `ContextSource`.
473    #[serde(default)]
474    pub context: Vec<ContextSource>,
475    pub tree: Option<ProcNode>,
476    /// How the session was attributed to the process, for debugging attribution.
477    pub attribution: Attribution,
478    /// True when another row shares this pid and carries its CPU, memory and
479    /// process counts. One Codex app-server hosts many conversations, so its
480    /// threads each get a row while the process is only counted once.
481    /// Absent in snapshots written before 0.2.0.
482    #[serde(default)]
483    pub shares_process: bool,
484    /// Set when the transcript parsed but its usage records did not, which
485    /// means this row's tokens and cost are not to be believed. Almost always a
486    /// harness that changed its format under us.
487    pub parse_warning: Option<String>,
488    /// How close the session is to its rate limit, when the harness reports it.
489    #[serde(default)]
490    pub rate_limit: Option<RateLimit>,
491    /// The session's cost as the harness itself last recorded it, beside
492    /// `cost_usd` and never mixed into it. See `HarnessCost`.
493    #[serde(default, skip_serializing_if = "Option::is_none")]
494    pub harness_cost: Option<HarnessCost>,
495}
496
497/// A harness's own running total for a session, read from its transcript.
498/// Claude Code writes one; the other harnesses do not, or (OpenCode, Kodelet)
499/// their costs are already the row's `cost_usd`.
500///
501/// It is a different measurement from `cost_usd`, not a correction to it: it
502/// is priced at the harness's private table, and it includes requests the
503/// transcript never records. Claude Code writes it when a session exits, so
504/// while a session runs it is the figure from the last exit.
505#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
506pub struct HarnessCost {
507    pub usd: f64,
508    /// The harness said some of its usage had no price, so `usd` is a floor.
509    pub lower_bound: bool,
510    /// The timestamp of the last transcript line before the record, which
511    /// carries none of its own.
512    pub as_of: Option<SystemTime>,
513    /// No usage has been recorded since, in the transcript or a subagent's.
514    /// False for a live session that has done anything since it last exited:
515    /// `usd` leaves that out.
516    pub current: bool,
517}
518
519/// One rolling usage window a harness reports against a rate limit: how much
520/// of it is spent and when it rolls over. Codex writes two, a short window and
521/// a long one, on every usage record.
522#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
523pub struct RateWindow {
524    /// Percent of the window used, 0..=100.
525    pub used_percent: f64,
526    /// The window length in minutes (Codex: 300 for the short one, 10080 weekly).
527    pub window_minutes: u64,
528    /// When the window rolls over and the usage resets, if the harness says.
529    pub resets_at: Option<SystemTime>,
530}
531
532/// What a harness reports about how close a session is to its rate limit. Only
533/// the harnesses that write it (Codex today) populate this; the rest leave it
534/// `None`. Read-only, like everything else: a number the harness already wrote.
535#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
536pub struct RateLimit {
537    /// The short rolling window.
538    pub primary: Option<RateWindow>,
539    /// The long rolling window.
540    pub secondary: Option<RateWindow>,
541    /// The plan the limit is for (Codex: `plus`, `pro`, ...).
542    pub plan: Option<String>,
543    /// True when the harness says the limit is currently hit.
544    pub reached: bool,
545}
546
547impl RateLimit {
548    /// The window closest to its limit, for a one-glance figure.
549    pub fn tightest(&self) -> Option<&RateWindow> {
550        [self.primary.as_ref(), self.secondary.as_ref()]
551            .into_iter()
552            .flatten()
553            .max_by(|a, b| a.used_percent.partial_cmp(&b.used_percent).unwrap_or(std::cmp::Ordering::Equal))
554    }
555}
556
557/// USD by kind of token, accumulated message by message at each message's
558/// own model price, so a session that changed model part way is still exact.
559/// The lines sum to `Agent::cost_usd`.
560#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
561pub struct CostBreakdown {
562    pub input: f64,
563    pub cache_write_5m: f64,
564    pub cache_write_1h: f64,
565    /// What a harness recorded against `TokenUsage::cache_write_unsplit`.
566    /// Always zero when the cost came from a price table.
567    #[serde(default)]
568    pub cache_write_unsplit: f64,
569    pub cache_read: f64,
570    pub output: f64,
571    /// Server-side web searches, billed per search on top of the tokens.
572    pub web_search: f64,
573}
574
575impl CostBreakdown {
576    pub fn total(&self) -> f64 {
577        self.input + self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit + self.cache_read + self.output + self.web_search
578    }
579
580    pub fn add(&mut self, o: &CostBreakdown) {
581        self.input += o.input;
582        self.cache_write_5m += o.cache_write_5m;
583        self.cache_write_1h += o.cache_write_1h;
584        self.cache_write_unsplit += o.cache_write_unsplit;
585        self.cache_read += o.cache_read;
586        self.output += o.output;
587        self.web_search += o.web_search;
588    }
589
590    pub fn sub(&mut self, o: &CostBreakdown) {
591        self.input -= o.input;
592        self.cache_write_5m -= o.cache_write_5m;
593        self.cache_write_1h -= o.cache_write_1h;
594        self.cache_write_unsplit -= o.cache_write_unsplit;
595        self.cache_read -= o.cache_read;
596        self.output -= o.output;
597        self.web_search -= o.web_search;
598    }
599}
600
601/// Where costs came from: list prices, a user's price file, or the harness's
602/// own recorded accounting. The UI says which rather than implying a bill.
603#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
604#[serde(rename_all = "kebab-case")]
605pub enum PriceSource {
606    Builtin,
607    UserFile,
608    Harness,
609}
610
611#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
612#[serde(rename_all = "kebab-case")]
613pub enum Attribution {
614    /// The harness told us (Claude's `~/.claude/sessions/<pid>.json`).
615    HarnessRegistry,
616    /// A `--resume <id>` style argument on the command line.
617    CommandLine,
618    /// The process has the transcript open (Codex keeps every live thread's
619    /// rollout open); exact.
620    OpenFile,
621    /// Matched by working directory and start time; may be wrong with concurrent sessions.
622    CwdHeuristic,
623    /// No transcript found; process only.
624    None,
625    /// No process; transcript only.
626    TranscriptOnly,
627}
628
629#[derive(Debug, Clone, Default, Serialize, Deserialize)]
630pub struct HostStats {
631    pub hostname: Option<String>,
632    pub cpu_percent: f32,
633    pub cpu_count: usize,
634    pub mem_used_bytes: u64,
635    pub mem_total_bytes: u64,
636}
637
638#[derive(Debug, Clone, Default, Serialize, Deserialize)]
639pub struct Totals {
640    pub agents: usize,
641    pub running: usize,
642    pub idle: usize,
643    pub stopped: usize,
644    pub tokens: u64,
645    pub cost_usd: f64,
646    pub unpriced_tokens: u64,
647    pub processes: usize,
648    pub mcp_processes: usize,
649    pub orphaned_mcp: usize,
650    pub cpu_percent: f32,
651    pub rss_bytes: u64,
652}
653
654/// Version of the `--json` document. Bumped when a field changes meaning or
655/// disappears; new fields alone do not bump it.
656pub const SNAPSHOT_SCHEMA_VERSION: u32 = 1;
657
658/// What the collector remembers about an orphaned MCP process: when it first
659/// saw it, and, if it watched the process lose its parent, which agent that
660/// was. Memory lasts for the run; a process that was already an orphan when
661/// agent-top started has no parent on record.
662#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
663pub struct OrphanOrigin {
664    pub pid: u32,
665    pub first_seen: SystemTime,
666    /// When the process was first seen without its parent, if the parent was
667    /// seen before that.
668    pub orphaned_at: Option<SystemTime>,
669    pub parent: Option<OrphanParent>,
670}
671
672/// The agent an orphan used to belong to.
673#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
674pub struct OrphanParent {
675    pub pid: u32,
676    pub agent_id: String,
677    pub name: String,
678}
679
680/// Which rule a piece of advice came from. See `advice`.
681#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
682#[serde(rename_all = "kebab-case")]
683pub enum AdviceRule {
684    /// A tool or MCP server whose results are large per call and have been
685    /// re-read on every response since: the biggest line in the bill hiding
686    /// behind a few calls.
687    ExpensiveSource,
688    /// An MCP server attached to a live agent that has answered no calls in
689    /// a long while.
690    IdleMcpServer,
691    /// An MCP server whose memory has climbed steadily with no calls to
692    /// explain it.
693    GrowingMcpServer,
694}
695
696impl AdviceRule {
697    pub fn label(self) -> &'static str {
698        match self {
699            AdviceRule::ExpensiveSource => "expensive source",
700            AdviceRule::IdleMcpServer => "idle mcp server",
701            AdviceRule::GrowingMcpServer => "growing mcp server",
702        }
703    }
704}
705
706/// One sentence of advice about one agent, with the numbers that back it and
707/// the thing you could do. Nothing is done for you: agent-top only points.
708#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
709pub struct Advice {
710    pub agent_id: String,
711    pub agent_name: String,
712    pub rule: AdviceRule,
713    /// The tool, or the MCP server, the advice is about.
714    pub subject: String,
715    /// The server's process, when the advice is about one.
716    pub pid: Option<u32>,
717    /// What was seen, with its numbers.
718    pub headline: String,
719    /// What you might do about it.
720    pub action: String,
721    /// The evidence as numbers, for scripts: zero where a rule has none.
722    #[serde(default)]
723    pub cost_usd: f64,
724    #[serde(default)]
725    pub tokens: u64,
726    #[serde(default)]
727    pub calls: u64,
728    #[serde(default)]
729    pub rss_bytes: u64,
730}
731
732/// Everything the UI needs for one frame.
733#[derive(Debug, Clone, Serialize, Deserialize)]
734pub struct Snapshot {
735    pub schema_version: u32,
736    pub taken_at: SystemTime,
737    pub host: HostStats,
738    pub agents: Vec<Agent>,
739    /// MCP-looking processes with no live agent ancestor: leak candidates.
740    pub orphans: Vec<ProcNode>,
741    /// One entry per orphan, saying where it came from when that is known.
742    #[serde(default)]
743    pub orphan_origins: Vec<OrphanOrigin>,
744    /// What looks like a bad deal on this machine right now, and what could
745    /// be done about it. See `advice`.
746    #[serde(default)]
747    pub advice: Vec<Advice>,
748    pub totals: Totals,
749}
750
751impl Snapshot {
752    pub fn compute_totals(&mut self) {
753        let mut t = Totals::default();
754        for a in &self.agents {
755            t.agents += 1;
756            match a.state {
757                AgentState::Running => t.running += 1,
758                AgentState::Idle => t.idle += 1,
759                AgentState::Stopped => t.stopped += 1,
760            }
761            t.tokens += a.usage.total();
762            t.cost_usd += a.cost_usd;
763            t.unpriced_tokens += a.unpriced_tokens;
764            t.processes += a.process_count;
765            t.mcp_processes += a.mcp_count;
766            t.cpu_percent += a.cpu_percent;
767            t.rss_bytes += a.rss_bytes;
768        }
769        t.orphaned_mcp = self.orphans.len();
770        self.totals = t;
771    }
772}
773
774#[cfg(test)]
775mod usage_tests {
776    use super::TokenUsage;
777
778    #[test]
779    fn cache_hit_rate_is_reads_over_the_prompt() {
780        let u = TokenUsage { input: 200, cache_read: 800, output: 50, ..Default::default() };
781        // Prompt is 1000 (output excluded); 800 of it from cache.
782        assert_eq!(u.prompt(), 1000);
783        assert!((u.cache_hit_rate().unwrap() - 0.8).abs() < 1e-9);
784        // A cache write counts as prompt input, not as a hit.
785        let u = TokenUsage { input: 100, cache_write_5m: 900, ..Default::default() };
786        assert_eq!(u.cache_hit_rate(), Some(0.0));
787        // Nothing to judge.
788        assert_eq!(TokenUsage::default().cache_hit_rate(), None);
789    }
790}