Skip to main content

agent_top_core/
model.rs

1//! The data model shared by discovery, the TUI and the JSON output.
2
3use serde::{Deserialize, Serialize};
4use std::path::PathBuf;
5use std::time::SystemTime;
6
7/// Which agent harness a process or transcript belongs to.
8#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
9#[serde(rename_all = "kebab-case")]
10pub enum Harness {
11    Claude,
12    Codex,
13    Gemini,
14    OpenCode,
15    Kodelet,
16    Aider,
17    Copilot,
18    Cursor,
19    Unknown,
20}
21
22impl Harness {
23    pub fn label(self) -> &'static str {
24        match self {
25            Harness::Claude => "claude",
26            Harness::Codex => "codex",
27            Harness::Gemini => "gemini",
28            Harness::OpenCode => "opencode",
29            Harness::Kodelet => "kodelet",
30            Harness::Aider => "aider",
31            Harness::Copilot => "copilot",
32            Harness::Cursor => "cursor",
33            Harness::Unknown => "unknown",
34        }
35    }
36
37    /// How much of a session id identifies it to `agent-top trace --session`,
38    /// which matches a prefix against the ids of every harness and fails when
39    /// one matches more than one session. `None` means the whole id is needed:
40    /// Kodelet's begin with a date that every session started that day shares.
41    pub fn session_id_prefix_len(self) -> Option<usize> {
42        match self {
43            Harness::Kodelet => None,
44            Harness::Claude
45            | Harness::Codex
46            | Harness::Gemini
47            | Harness::OpenCode
48            | Harness::Aider
49            | Harness::Copilot
50            | Harness::Cursor
51            | Harness::Unknown => Some(8),
52        }
53    }
54}
55
56/// Coarse lifecycle state, in the htop sense.
57///
58/// `Running` means the agent is mid-turn (inference or tool execution),
59/// `Idle` means the process is alive but waiting for a human, `Stopped` means
60/// the transcript exists and was recently active but no process owns it.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
62#[serde(rename_all = "kebab-case")]
63pub enum AgentState {
64    Running,
65    Idle,
66    Stopped,
67}
68
69impl AgentState {
70    pub fn label(self) -> &'static str {
71        match self {
72            AgentState::Running => "running",
73            AgentState::Idle => "idle",
74            AgentState::Stopped => "stopped",
75        }
76    }
77}
78
79/// What the transcript says the agent was last doing. Harness-neutral.
80#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
81#[serde(rename_all = "kebab-case")]
82pub enum Activity {
83    /// Mid-turn: a prompt or tool result was just submitted, or a tool call is pending.
84    Working,
85    /// The last thing that happened was the assistant ending its turn.
86    Waiting,
87    #[default]
88    Unknown,
89}
90
91/// Token counts, split the way the Anthropic and OpenAI usage objects split them.
92#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
93pub struct TokenUsage {
94    pub input: u64,
95    pub cache_write_5m: u64,
96    pub cache_write_1h: u64,
97    /// Cache creation a harness recorded without the TTL that decides its
98    /// rate. No price table can cover it: charging either TTL's rate would be
99    /// a guess about a lifetime nobody wrote down. Only a harness that reports
100    /// its own costs can put a number against these tokens.
101    #[serde(default)]
102    pub cache_write_unsplit: u64,
103    pub cache_read: u64,
104    pub output: u64,
105}
106
107impl TokenUsage {
108    pub fn cache_write(&self) -> u64 {
109        self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit
110    }
111
112    /// Everything the model consumed or produced. This is the "TOKENS" column.
113    pub fn total(&self) -> u64 {
114        self.input + self.cache_write() + self.cache_read + self.output
115    }
116
117    /// The input side: everything sent to the model as the prompt, fresh input
118    /// plus cache writes plus cache reads, but not the output it produced.
119    pub fn prompt(&self) -> u64 {
120        self.input + self.cache_write() + self.cache_read
121    }
122
123    /// The share of the prompt served from cache (the cheap reads), in `0..=1`.
124    /// A high number means most of the re-sent conversation was billed at the
125    /// cache-read rate rather than full input; a low one on a long session is
126    /// money left on the table. `None` when there was no prompt to judge, or
127    /// the model does not cache at all.
128    pub fn cache_hit_rate(&self) -> Option<f64> {
129        let p = self.prompt();
130        (p > 0).then(|| self.cache_read as f64 / p as f64)
131    }
132
133    pub fn add(&mut self, other: &TokenUsage) {
134        self.input += other.input;
135        self.cache_write_5m += other.cache_write_5m;
136        self.cache_write_1h += other.cache_write_1h;
137        self.cache_write_unsplit += other.cache_write_unsplit;
138        self.cache_read += other.cache_read;
139        self.output += other.output;
140    }
141
142    pub fn sub(&mut self, other: &TokenUsage) {
143        self.input = self.input.saturating_sub(other.input);
144        self.cache_write_5m = self.cache_write_5m.saturating_sub(other.cache_write_5m);
145        self.cache_write_1h = self.cache_write_1h.saturating_sub(other.cache_write_1h);
146        self.cache_write_unsplit = self.cache_write_unsplit.saturating_sub(other.cache_write_unsplit);
147        self.cache_read = self.cache_read.saturating_sub(other.cache_read);
148        self.output = self.output.saturating_sub(other.output);
149    }
150}
151
152/// What a span measures. Tool calls came first and gave the type its name;
153/// the other two label the time between them.
154#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, PartialOrd, Ord)]
155#[serde(rename_all = "kebab-case")]
156pub enum SpanKind {
157    /// One tool call, from the harness issuing it to the result coming back.
158    #[default]
159    Tool,
160    /// The model producing a response: from a prompt or tool results being
161    /// submitted to the last block of the reply being written. The gap in a
162    /// waterfall that is not a tool call is almost always one of these.
163    Inference,
164    /// One human turn: from the prompt to the model ending its reply.
165    /// Contains every tool call and inference span issued in between.
166    Turn,
167}
168
169impl SpanKind {
170    pub const ALL: [SpanKind; 3] = [SpanKind::Tool, SpanKind::Inference, SpanKind::Turn];
171
172    pub fn label(self) -> &'static str {
173        match self {
174            SpanKind::Tool => "tool",
175            SpanKind::Inference => "inference",
176            SpanKind::Turn => "turn",
177        }
178    }
179}
180
181/// One span of an agent trace: a tool call, an inference, or a turn.
182///
183/// Every harness writes the same shape in its own vocabulary — Claude pairs a
184/// `tool_use` block with a `tool_result` block by `tool_use_id`, Codex pairs a
185/// `function_call` with a `function_call_output` by `call_id` — and both stamp
186/// each line with a timestamp. That is a span: a name, a start and a duration.
187/// Only the call's metadata is kept; arguments and output are never read.
188///
189/// The name predates `kind`: the type carried only tool calls until 0.3.1 and
190/// is kept for the sake of the published API.
191#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
192pub struct ToolSpan {
193    /// The harness's own call id, so a span survives across refreshes. For
194    /// inference and turn spans, a counter the parser assigns.
195    pub id: String,
196    /// Tool name as the harness reports it (`Bash`, `exec_command`, ...), or
197    /// `inference` / `turn`.
198    pub name: String,
199    pub started_at: SystemTime,
200    /// Wall-clock duration, or `None` while the call is still in flight.
201    pub duration_ms: Option<u64>,
202    /// The call was issued by a subagent (Claude's `isSidechain`).
203    pub sidechain: bool,
204    /// The harness reported the result as an error.
205    pub error: bool,
206    /// Absent in snapshots written before 0.3.1, which held tool calls only.
207    #[serde(default)]
208    pub kind: SpanKind,
209}
210
211impl ToolSpan {
212    pub fn is_open(&self) -> bool {
213        self.duration_ms.is_none()
214    }
215
216    /// Duration if closed, otherwise how long it has been running as of `now`.
217    pub fn elapsed_ms(&self, now: SystemTime) -> u64 {
218        match self.duration_ms {
219            Some(ms) => ms,
220            None => now.duration_since(self.started_at).map(|d| d.as_millis() as u64).unwrap_or(0),
221        }
222    }
223
224    pub fn ended_at(&self, now: SystemTime) -> SystemTime {
225        self.started_at + std::time::Duration::from_millis(self.elapsed_ms(now))
226    }
227}
228
229/// The last two components of a path, so `/Users/me/code/app` reads as
230/// `code/app` and two projects called `app` are still told apart.
231pub fn project_name(p: &std::path::Path) -> String {
232    let names: Vec<String> = p
233        .components()
234        .filter_map(|c| match c {
235            std::path::Component::Normal(s) => Some(s.to_string_lossy().into_owned()),
236            _ => None,
237        })
238        .collect();
239    let tail = names.iter().rev().take(2).rev().cloned().collect::<Vec<_>>().join("/");
240    if tail.is_empty() { p.to_string_lossy().into_owned() } else { tail }
241}
242
243/// The turn a span belongs to: the newest turn that started at or before it
244/// and had not ended when it started. Subagent spans prefer the subagent's
245/// own turn, which their transcript carries, and fall back to the main
246/// agent's. A turn has no parent.
247pub fn parent_turn<'a>(spans: &[&'a ToolSpan], i: usize) -> Option<&'a ToolSpan> {
248    let sp = spans[i];
249    if sp.kind == SpanKind::Turn {
250        return None;
251    }
252    let contains = |t: &ToolSpan| {
253        t.kind == SpanKind::Turn
254            && t.started_at <= sp.started_at
255            && t.duration_ms.map(|ms| t.started_at + std::time::Duration::from_millis(ms) >= sp.started_at).unwrap_or(true)
256    };
257    let own = spans[..i].iter().rev().find(|t| t.sidechain == sp.sidechain && contains(t));
258    own.or_else(|| spans[..i].iter().rev().find(|t| !t.sidechain && contains(t))).copied()
259}
260
261/// One MCP server an agent uses, seen from either side or both: the process
262/// table has the server's pid, CPU and memory; the transcript has how often
263/// the agent called it. Claude Code names an MCP tool `mcp__<server>__<tool>`,
264/// which is where the server name and the call count come from.
265#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
266pub struct McpServer {
267    /// The server's name as the harness configured it, or, for a process the
268    /// transcript never named, the program it is running.
269    pub name: String,
270    pub pid: Option<u32>,
271    pub cmdline: Option<String>,
272    pub cpu_percent: f32,
273    pub rss_bytes: u64,
274    pub age_secs: Option<u64>,
275    /// Tool calls the agent made to this server, from the transcript.
276    pub calls: u64,
277    /// Of those, how many the harness reported as errors.
278    pub errors: u64,
279    pub last_call: Option<SystemTime>,
280    /// How the process and the transcript's server were put together, for
281    /// the UI to label a guess as one.
282    pub matched_by: McpMatch,
283}
284
285/// How an `McpServer` row was formed.
286#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
287#[serde(rename_all = "kebab-case")]
288pub enum McpMatch {
289    /// A process under the agent that no transcript server names. Either it
290    /// has not been called yet, or its name does not appear in its command.
291    ProcessOnly,
292    /// A server the transcript calls with no process under the agent: an HTTP
293    /// server, or one that has exited.
294    TranscriptOnly,
295    /// The server's name appears in the process's command line.
296    Name,
297    /// One unmatched process and one unmatched server were left; they are
298    /// taken to be the same. A guess.
299    Sole,
300}
301
302/// Where a stretch of a session's context came from.
303#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
304#[serde(rename_all = "kebab-case")]
305pub enum ContextOrigin {
306    /// A built-in tool of the harness (`Bash`, `Read`, `exec_command`, ...).
307    Tool,
308    /// An MCP server; `name` is the server, not the tool.
309    Mcp,
310    /// Everything that is not a tool result: the system prompt, the user's
311    /// own messages, a compaction summary.
312    #[default]
313    Other,
314}
315
316/// One source of a session's context: what its results have added to the
317/// prompt, and what sending that on every inference since has cost.
318///
319/// The tokens are the growth of the prompt between one response and the
320/// next, attributed to the tool results submitted in between; the cost is
321/// those tokens charged at the prompt rate of every response that carried
322/// them, the first read included. The rows sum to the session's prompt-side
323/// cost. See `docs/accounting.md`.
324#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
325pub struct ContextSource {
326    pub name: String,
327    pub origin: ContextOrigin,
328    /// Tool results attributed to this source. Zero for `Other`.
329    pub calls: u64,
330    /// Tokens the source added to the prompt over the session.
331    pub tokens: u64,
332    /// USD those tokens have cost across every response that read them.
333    /// An estimate; zero when the model has no price.
334    pub cost_usd: f64,
335}
336
337/// Role of a process inside an agent's tree.
338#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
339#[serde(rename_all = "kebab-case")]
340pub enum ProcKind {
341    /// A harness process, whether at the root or nested under another process.
342    /// Older snapshots called nested harness processes `subagent`.
343    #[serde(alias = "subagent")]
344    Agent,
345    /// A Model Context Protocol server or helper.
346    Mcp,
347    /// A shell spawned to run a tool call.
348    Shell,
349    /// Anything else the agent launched (test runners, sleep, caffeinate, ...).
350    Tool,
351}
352
353impl ProcKind {
354    pub fn label(self) -> &'static str {
355        match self {
356            ProcKind::Agent => "agent",
357            ProcKind::Mcp => "mcp",
358            ProcKind::Shell => "shell",
359            ProcKind::Tool => "tool",
360        }
361    }
362}
363
364/// One process, with its descendants.
365#[derive(Debug, Clone, Serialize, Deserialize)]
366pub struct ProcNode {
367    pub pid: u32,
368    pub ppid: Option<u32>,
369    pub name: String,
370    pub cmdline: String,
371    pub kind: ProcKind,
372    pub harness: Option<Harness>,
373    pub cpu_percent: f32,
374    pub rss_bytes: u64,
375    pub age_secs: u64,
376    pub cwd: Option<PathBuf>,
377    pub children: Vec<ProcNode>,
378}
379
380impl ProcNode {
381    /// CPU, RSS, process count and MCP server count summed over the whole
382    /// subtree. An MCP process's own children (an `npx` wrapper's `node`) are
383    /// the same server, so they add to the process count and not to the
384    /// server count.
385    pub fn totals(&self) -> (f32, u64, usize, usize) {
386        let mut cpu = self.cpu_percent;
387        let mut rss = self.rss_bytes;
388        let mut count = 1;
389        let mut mcp = usize::from(self.kind == ProcKind::Mcp);
390        for c in &self.children {
391            let (ccpu, crss, ccount, cmcp) = c.totals();
392            cpu += ccpu;
393            rss += crss;
394            count += ccount;
395            if self.kind != ProcKind::Mcp {
396                mcp += cmcp;
397            }
398        }
399        (cpu, rss, count, mcp)
400    }
401
402    /// The MCP servers in this tree: each `Mcp` node whose parent is not one.
403    /// A server started through `npx` or `uvx` is two or three processes; the
404    /// top one stands for the server.
405    pub fn mcp_roots(&self) -> Vec<&ProcNode> {
406        let mut out = Vec::new();
407        self.collect_mcp_roots(&mut out);
408        out
409    }
410
411    fn collect_mcp_roots<'a>(&'a self, out: &mut Vec<&'a ProcNode>) {
412        if self.kind == ProcKind::Mcp {
413            out.push(self);
414            return;
415        }
416        for c in &self.children {
417            c.collect_mcp_roots(out);
418        }
419    }
420
421    pub fn walk<'a>(&'a self, depth: usize, f: &mut dyn FnMut(&'a ProcNode, usize)) {
422        f(self, depth);
423        for c in &self.children {
424            c.walk(depth + 1, f);
425        }
426    }
427}
428
429/// A logical subagent relationship explicitly recorded by the harness.
430/// This is session lineage, not an OS process relationship or a history fork.
431#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
432pub struct SubagentInfo {
433    pub parent_session_id: String,
434    #[serde(default)]
435    pub nickname: Option<String>,
436    #[serde(default)]
437    pub role: Option<String>,
438}
439
440/// A single row in the agent table.
441#[derive(Debug, Clone, Serialize, Deserialize)]
442pub struct Agent {
443    /// Stable identity across refreshes: `pid:<n>` for live agents, `session:<id>` for stopped ones.
444    pub id: String,
445    pub name: String,
446    pub harness: Harness,
447    pub state: AgentState,
448    pub activity: Activity,
449    pub pid: Option<u32>,
450    pub session_id: Option<String>,
451    /// Explicit logical parent and identity, when the transcript records them.
452    /// Usage remains per session; adding hierarchy must not count it twice.
453    #[serde(default, skip_serializing_if = "Option::is_none")]
454    pub subagent: Option<SubagentInfo>,
455    pub session_path: Option<PathBuf>,
456    pub cwd: Option<PathBuf>,
457    pub model: Option<String>,
458    pub harness_version: Option<String>,
459    pub usage: TokenUsage,
460    /// USD spent on messages whose model had a known price.
461    pub cost_usd: f64,
462    /// `cost_usd` by kind of token, so a figure that differs from another
463    /// tool's can be traced to the one line that differs.
464    #[serde(default)]
465    pub cost_breakdown: CostBreakdown,
466    /// Where this row's costs came from; `None` when it has no known price.
467    #[serde(default)]
468    pub price_source: Option<PriceSource>,
469    /// Tokens on messages whose model had no known price (so `cost_usd` is a floor).
470    pub unpriced_tokens: u64,
471    pub turns: u64,
472    pub subagent_turns: u64,
473    /// See `SessionSummary::folds_child_usage`. Claude Code, Gemini CLI and
474    /// OpenCode fold a child's usage into its parent, the way those harnesses
475    /// bill it; Codex and Kodelet give each child its own row instead.
476    #[serde(default)]
477    pub folds_child_usage: bool,
478    pub tool_calls: u64,
479    /// The count is only known retained/observed calls, not an exact lifetime
480    /// total. Kodelet compaction discards old calls without a cumulative count.
481    #[serde(default)]
482    pub tool_calls_lower_bound: bool,
483    /// Server-side web searches the model ran, billed per search on top of
484    /// tokens. Counted for every harness; priced only where the price table
485    /// has a rate (Anthropic's, for Claude Code).
486    #[serde(default)]
487    pub web_searches: u64,
488    /// The most recent spans, oldest first: tool calls, inferences and turns.
489    /// Bounded; see `harness::MAX_SPANS`.
490    pub spans: Vec<ToolSpan>,
491    /// Seconds since the process started (live) or since the last transcript write (stopped).
492    pub age_secs: u64,
493    /// Seconds since the transcript was last written.
494    pub idle_secs: Option<u64>,
495    pub cpu_percent: f32,
496    pub rss_bytes: u64,
497    pub process_count: usize,
498    pub mcp_count: usize,
499    /// The MCP servers this agent uses, one row each, from the process tree
500    /// and the transcript. See `McpServer`.
501    #[serde(default)]
502    pub mcp_servers: Vec<McpServer>,
503    /// What each tool's results added to the prompt and what carrying it has
504    /// cost, largest first. See `ContextSource`.
505    #[serde(default)]
506    pub context: Vec<ContextSource>,
507    pub tree: Option<ProcNode>,
508    /// How the session was attributed to the process, for debugging attribution.
509    pub attribution: Attribution,
510    /// True when another row shares this pid and carries its CPU, memory and
511    /// process counts. One Codex app-server hosts many conversations, so its
512    /// threads each get a row while the process is only counted once.
513    /// Absent in snapshots written before 0.2.0.
514    #[serde(default)]
515    pub shares_process: bool,
516    /// Set when the transcript parsed but its usage records did not, which
517    /// means this row's tokens and cost are not to be believed. Almost always a
518    /// harness that changed its format under us.
519    pub parse_warning: Option<String>,
520    /// How close the session is to its rate limit, when the harness reports it.
521    #[serde(default)]
522    pub rate_limit: Option<RateLimit>,
523    /// The session's cost as the harness itself last recorded it, beside
524    /// `cost_usd` and never mixed into it. See `HarnessCost`.
525    #[serde(default, skip_serializing_if = "Option::is_none")]
526    pub harness_cost: Option<HarnessCost>,
527}
528
529/// A harness's own running total for a session, read from its transcript.
530/// Claude Code writes one; the other harnesses do not, or (OpenCode, Kodelet)
531/// their costs are already the row's `cost_usd`.
532///
533/// It is a different measurement from `cost_usd`, not a correction to it: it
534/// is priced at the harness's private table, and it includes requests the
535/// transcript never records. Claude Code writes it when a session exits, so
536/// while a session runs it is the figure from the last exit.
537#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
538pub struct HarnessCost {
539    pub usd: f64,
540    /// The harness said some of its usage had no price, so `usd` is a floor.
541    pub lower_bound: bool,
542    /// The timestamp of the last transcript line before the record, which
543    /// carries none of its own.
544    pub as_of: Option<SystemTime>,
545    /// No usage has been recorded since, in the transcript or a subagent's.
546    /// False for a live session that has done anything since it last exited:
547    /// `usd` leaves that out.
548    pub current: bool,
549}
550
551/// One rolling usage window a harness reports against a rate limit: how much
552/// of it is spent and when it rolls over. Codex writes two, a short window and
553/// a long one, on every usage record.
554#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
555pub struct RateWindow {
556    /// Percent of the window used, 0..=100.
557    pub used_percent: f64,
558    /// The window length in minutes (Codex: 300 for the short one, 10080 weekly).
559    pub window_minutes: u64,
560    /// When the window rolls over and the usage resets, if the harness says.
561    pub resets_at: Option<SystemTime>,
562}
563
564/// What a harness reports about how close a session is to its rate limit. Only
565/// the harnesses that write it (Codex today) populate this; the rest leave it
566/// `None`. Read-only, like everything else: a number the harness already wrote.
567#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
568pub struct RateLimit {
569    /// The short rolling window.
570    pub primary: Option<RateWindow>,
571    /// The long rolling window.
572    pub secondary: Option<RateWindow>,
573    /// The plan the limit is for (Codex: `plus`, `pro`, ...).
574    pub plan: Option<String>,
575    /// True when the harness says the limit is currently hit.
576    pub reached: bool,
577}
578
579impl RateLimit {
580    /// The window closest to its limit, for a one-glance figure.
581    pub fn tightest(&self) -> Option<&RateWindow> {
582        [self.primary.as_ref(), self.secondary.as_ref()]
583            .into_iter()
584            .flatten()
585            .max_by(|a, b| a.used_percent.partial_cmp(&b.used_percent).unwrap_or(std::cmp::Ordering::Equal))
586    }
587}
588
589/// USD by kind of token, accumulated message by message at each message's
590/// own model price, so a session that changed model part way is still exact.
591/// The lines sum to `Agent::cost_usd`.
592#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
593pub struct CostBreakdown {
594    pub input: f64,
595    pub cache_write_5m: f64,
596    pub cache_write_1h: f64,
597    /// What a harness recorded against `TokenUsage::cache_write_unsplit`.
598    /// Always zero when the cost came from a price table.
599    #[serde(default)]
600    pub cache_write_unsplit: f64,
601    pub cache_read: f64,
602    pub output: f64,
603    /// Server-side web searches, billed per search on top of the tokens.
604    pub web_search: f64,
605}
606
607impl CostBreakdown {
608    pub fn total(&self) -> f64 {
609        self.input + self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit + self.cache_read + self.output + self.web_search
610    }
611
612    pub fn add(&mut self, o: &CostBreakdown) {
613        self.input += o.input;
614        self.cache_write_5m += o.cache_write_5m;
615        self.cache_write_1h += o.cache_write_1h;
616        self.cache_write_unsplit += o.cache_write_unsplit;
617        self.cache_read += o.cache_read;
618        self.output += o.output;
619        self.web_search += o.web_search;
620    }
621
622    pub fn sub(&mut self, o: &CostBreakdown) {
623        self.input -= o.input;
624        self.cache_write_5m -= o.cache_write_5m;
625        self.cache_write_1h -= o.cache_write_1h;
626        self.cache_write_unsplit -= o.cache_write_unsplit;
627        self.cache_read -= o.cache_read;
628        self.output -= o.output;
629        self.web_search -= o.web_search;
630    }
631}
632
633/// Where costs came from: list prices, a user's price file, or the harness's
634/// own recorded accounting. The UI says which rather than implying a bill.
635#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
636#[serde(rename_all = "kebab-case")]
637pub enum PriceSource {
638    Builtin,
639    UserFile,
640    Harness,
641}
642
643#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
644#[serde(rename_all = "kebab-case")]
645pub enum Attribution {
646    /// The harness told us (Claude's `~/.claude/sessions/<pid>.json`).
647    HarnessRegistry,
648    /// A `--resume <id>` style argument on the command line.
649    CommandLine,
650    /// The process has the transcript open (Codex keeps every live thread's
651    /// rollout open); exact.
652    OpenFile,
653    /// Matched by working directory and start time; may be wrong with concurrent sessions.
654    CwdHeuristic,
655    /// No transcript found; process only.
656    None,
657    /// No process; transcript only.
658    TranscriptOnly,
659}
660
661#[derive(Debug, Clone, Default, Serialize, Deserialize)]
662pub struct HostStats {
663    pub hostname: Option<String>,
664    pub cpu_percent: f32,
665    pub cpu_count: usize,
666    pub mem_used_bytes: u64,
667    pub mem_total_bytes: u64,
668}
669
670#[derive(Debug, Clone, Default, Serialize, Deserialize)]
671pub struct Totals {
672    pub agents: usize,
673    pub running: usize,
674    pub idle: usize,
675    pub stopped: usize,
676    pub tokens: u64,
677    pub cost_usd: f64,
678    pub unpriced_tokens: u64,
679    pub processes: usize,
680    pub mcp_processes: usize,
681    pub orphaned_mcp: usize,
682    pub cpu_percent: f32,
683    pub rss_bytes: u64,
684}
685
686/// Version of the `--json` document. Bumped when a field changes meaning or
687/// disappears; new fields alone do not bump it.
688pub const SNAPSHOT_SCHEMA_VERSION: u32 = 1;
689
690/// What the collector remembers about an orphaned MCP process: when it first
691/// saw it, and, if it watched the process lose its parent, which agent that
692/// was. Memory lasts for the run; a process that was already an orphan when
693/// agent-top started has no parent on record.
694#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
695pub struct OrphanOrigin {
696    pub pid: u32,
697    pub first_seen: SystemTime,
698    /// When the process was first seen without its parent, if the parent was
699    /// seen before that.
700    pub orphaned_at: Option<SystemTime>,
701    pub parent: Option<OrphanParent>,
702}
703
704/// The agent an orphan used to belong to.
705#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
706pub struct OrphanParent {
707    pub pid: u32,
708    pub agent_id: String,
709    pub name: String,
710}
711
712/// Which rule a piece of advice came from. See `advice`.
713#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
714#[serde(rename_all = "kebab-case")]
715pub enum AdviceRule {
716    /// A tool or MCP server whose results are large per call and have been
717    /// re-read on every response since: the biggest line in the bill hiding
718    /// behind a few calls.
719    ExpensiveSource,
720    /// An MCP server attached to a live agent that has answered no calls in
721    /// a long while.
722    IdleMcpServer,
723    /// An MCP server whose memory has climbed steadily with no calls to
724    /// explain it.
725    GrowingMcpServer,
726}
727
728impl AdviceRule {
729    pub fn label(self) -> &'static str {
730        match self {
731            AdviceRule::ExpensiveSource => "expensive source",
732            AdviceRule::IdleMcpServer => "idle mcp server",
733            AdviceRule::GrowingMcpServer => "growing mcp server",
734        }
735    }
736}
737
738/// One sentence of advice about one agent, with the numbers that back it and
739/// the thing you could do. Nothing is done for you: agent-top only points.
740#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
741pub struct Advice {
742    pub agent_id: String,
743    pub agent_name: String,
744    pub rule: AdviceRule,
745    /// The tool, or the MCP server, the advice is about.
746    pub subject: String,
747    /// The server's process, when the advice is about one.
748    pub pid: Option<u32>,
749    /// What was seen, with its numbers.
750    pub headline: String,
751    /// What you might do about it.
752    pub action: String,
753    /// The evidence as numbers, for scripts: zero where a rule has none.
754    #[serde(default)]
755    pub cost_usd: f64,
756    #[serde(default)]
757    pub tokens: u64,
758    #[serde(default)]
759    pub calls: u64,
760    #[serde(default)]
761    pub rss_bytes: u64,
762}
763
764/// Everything the UI needs for one frame.
765#[derive(Debug, Clone, Serialize, Deserialize)]
766pub struct Snapshot {
767    pub schema_version: u32,
768    pub taken_at: SystemTime,
769    pub host: HostStats,
770    pub agents: Vec<Agent>,
771    /// MCP-looking processes with no live agent ancestor: leak candidates.
772    pub orphans: Vec<ProcNode>,
773    /// One entry per orphan, saying where it came from when that is known.
774    #[serde(default)]
775    pub orphan_origins: Vec<OrphanOrigin>,
776    /// What looks like a bad deal on this machine right now, and what could
777    /// be done about it. See `advice`.
778    #[serde(default)]
779    pub advice: Vec<Advice>,
780    pub totals: Totals,
781}
782
783impl Snapshot {
784    pub fn compute_totals(&mut self) {
785        let mut t = Totals::default();
786        for a in &self.agents {
787            t.agents += 1;
788            match a.state {
789                AgentState::Running => t.running += 1,
790                AgentState::Idle => t.idle += 1,
791                AgentState::Stopped => t.stopped += 1,
792            }
793            t.tokens += a.usage.total();
794            t.cost_usd += a.cost_usd;
795            t.unpriced_tokens += a.unpriced_tokens;
796            t.processes += a.process_count;
797            t.mcp_processes += a.mcp_count;
798            t.cpu_percent += a.cpu_percent;
799            t.rss_bytes += a.rss_bytes;
800        }
801        t.orphaned_mcp = self.orphans.len();
802        self.totals = t;
803    }
804}
805
806#[cfg(test)]
807mod usage_tests {
808    use super::TokenUsage;
809
810    #[test]
811    fn cache_hit_rate_is_reads_over_the_prompt() {
812        let u = TokenUsage { input: 200, cache_read: 800, output: 50, ..Default::default() };
813        // Prompt is 1000 (output excluded); 800 of it from cache.
814        assert_eq!(u.prompt(), 1000);
815        assert!((u.cache_hit_rate().unwrap() - 0.8).abs() < 1e-9);
816        // A cache write counts as prompt input, not as a hit.
817        let u = TokenUsage { input: 100, cache_write_5m: 900, ..Default::default() };
818        assert_eq!(u.cache_hit_rate(), Some(0.0));
819        // Nothing to judge.
820        assert_eq!(TokenUsage::default().cache_hit_rate(), None);
821    }
822}
823
824#[cfg(test)]
825mod tests {
826    use super::*;
827    use std::path::Path;
828
829    #[test]
830    fn project_name_keeps_two_components() {
831        assert_eq!(project_name(Path::new("/Users/me/code/app")), "code/app");
832        assert_eq!(project_name(Path::new("/app")), "app");
833    }
834}