agent_top_core/model.rs
1//! The data model shared by discovery, the TUI and the JSON output.
2
3use serde::{Deserialize, Serialize};
4use std::path::PathBuf;
5use std::time::SystemTime;
6
7/// Which agent harness a process or transcript belongs to.
8#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
9#[serde(rename_all = "kebab-case")]
10pub enum Harness {
11 Claude,
12 Codex,
13 Gemini,
14 OpenCode,
15 Kodelet,
16 Aider,
17 Copilot,
18 Cursor,
19 Unknown,
20}
21
22impl Harness {
23 pub fn label(self) -> &'static str {
24 match self {
25 Harness::Claude => "claude",
26 Harness::Codex => "codex",
27 Harness::Gemini => "gemini",
28 Harness::OpenCode => "opencode",
29 Harness::Kodelet => "kodelet",
30 Harness::Aider => "aider",
31 Harness::Copilot => "copilot",
32 Harness::Cursor => "cursor",
33 Harness::Unknown => "unknown",
34 }
35 }
36
37 /// How much of a session id identifies it to `agent-top trace --session`,
38 /// which matches a prefix against the ids of every harness and fails when
39 /// one matches more than one session. `None` means the whole id is needed:
40 /// Kodelet's begin with a date that every session started that day shares.
41 pub fn session_id_prefix_len(self) -> Option<usize> {
42 match self {
43 Harness::Kodelet => None,
44 Harness::Claude
45 | Harness::Codex
46 | Harness::Gemini
47 | Harness::OpenCode
48 | Harness::Aider
49 | Harness::Copilot
50 | Harness::Cursor
51 | Harness::Unknown => Some(8),
52 }
53 }
54}
55
56/// Coarse lifecycle state, in the htop sense.
57///
58/// `Running` means the agent is mid-turn (inference or tool execution),
59/// `Idle` means the process is alive but waiting for a human, `Stopped` means
60/// the transcript exists and was recently active but no process owns it.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
62#[serde(rename_all = "kebab-case")]
63pub enum AgentState {
64 Running,
65 Idle,
66 Stopped,
67}
68
69impl AgentState {
70 pub fn label(self) -> &'static str {
71 match self {
72 AgentState::Running => "running",
73 AgentState::Idle => "idle",
74 AgentState::Stopped => "stopped",
75 }
76 }
77}
78
79/// What the transcript says the agent was last doing. Harness-neutral.
80#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
81#[serde(rename_all = "kebab-case")]
82pub enum Activity {
83 /// Mid-turn: a prompt or tool result was just submitted, or a tool call is pending.
84 Working,
85 /// The last thing that happened was the assistant ending its turn.
86 Waiting,
87 #[default]
88 Unknown,
89}
90
91/// Token counts, split the way the Anthropic and OpenAI usage objects split them.
92#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
93pub struct TokenUsage {
94 pub input: u64,
95 pub cache_write_5m: u64,
96 pub cache_write_1h: u64,
97 /// Cache creation a harness recorded without the TTL that decides its
98 /// rate. No price table can cover it: charging either TTL's rate would be
99 /// a guess about a lifetime nobody wrote down. Only a harness that reports
100 /// its own costs can put a number against these tokens.
101 #[serde(default)]
102 pub cache_write_unsplit: u64,
103 pub cache_read: u64,
104 pub output: u64,
105}
106
107impl TokenUsage {
108 pub fn cache_write(&self) -> u64 {
109 self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit
110 }
111
112 /// Everything the model consumed or produced. This is the "TOKENS" column.
113 pub fn total(&self) -> u64 {
114 self.input + self.cache_write() + self.cache_read + self.output
115 }
116
117 /// The input side: everything sent to the model as the prompt, fresh input
118 /// plus cache writes plus cache reads, but not the output it produced.
119 pub fn prompt(&self) -> u64 {
120 self.input + self.cache_write() + self.cache_read
121 }
122
123 /// The share of the prompt served from cache (the cheap reads), in `0..=1`.
124 /// A high number means most of the re-sent conversation was billed at the
125 /// cache-read rate rather than full input; a low one on a long session is
126 /// money left on the table. `None` when there was no prompt to judge, or
127 /// the model does not cache at all.
128 pub fn cache_hit_rate(&self) -> Option<f64> {
129 let p = self.prompt();
130 (p > 0).then(|| self.cache_read as f64 / p as f64)
131 }
132
133 pub fn add(&mut self, other: &TokenUsage) {
134 self.input += other.input;
135 self.cache_write_5m += other.cache_write_5m;
136 self.cache_write_1h += other.cache_write_1h;
137 self.cache_write_unsplit += other.cache_write_unsplit;
138 self.cache_read += other.cache_read;
139 self.output += other.output;
140 }
141
142 pub fn sub(&mut self, other: &TokenUsage) {
143 self.input = self.input.saturating_sub(other.input);
144 self.cache_write_5m = self.cache_write_5m.saturating_sub(other.cache_write_5m);
145 self.cache_write_1h = self.cache_write_1h.saturating_sub(other.cache_write_1h);
146 self.cache_write_unsplit = self.cache_write_unsplit.saturating_sub(other.cache_write_unsplit);
147 self.cache_read = self.cache_read.saturating_sub(other.cache_read);
148 self.output = self.output.saturating_sub(other.output);
149 }
150}
151
152/// What a span measures. Tool calls came first and gave the type its name;
153/// the other two label the time between them.
154#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, PartialOrd, Ord)]
155#[serde(rename_all = "kebab-case")]
156pub enum SpanKind {
157 /// One tool call, from the harness issuing it to the result coming back.
158 #[default]
159 Tool,
160 /// The model producing a response: from a prompt or tool results being
161 /// submitted to the last block of the reply being written. The gap in a
162 /// waterfall that is not a tool call is almost always one of these.
163 Inference,
164 /// One human turn: from the prompt to the model ending its reply.
165 /// Contains every tool call and inference span issued in between.
166 Turn,
167}
168
169impl SpanKind {
170 pub const ALL: [SpanKind; 3] = [SpanKind::Tool, SpanKind::Inference, SpanKind::Turn];
171
172 pub fn label(self) -> &'static str {
173 match self {
174 SpanKind::Tool => "tool",
175 SpanKind::Inference => "inference",
176 SpanKind::Turn => "turn",
177 }
178 }
179}
180
181/// One span of an agent trace: a tool call, an inference, or a turn.
182///
183/// Every harness writes the same shape in its own vocabulary — Claude pairs a
184/// `tool_use` block with a `tool_result` block by `tool_use_id`, Codex pairs a
185/// `function_call` with a `function_call_output` by `call_id` — and both stamp
186/// each line with a timestamp. That is a span: a name, a start and a duration.
187/// Only the call's metadata is kept; arguments and output are never read.
188///
189/// The name predates `kind`: the type carried only tool calls until 0.3.1 and
190/// is kept for the sake of the published API.
191#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
192pub struct ToolSpan {
193 /// The harness's own call id, so a span survives across refreshes. For
194 /// inference and turn spans, a counter the parser assigns.
195 pub id: String,
196 /// Tool name as the harness reports it (`Bash`, `exec_command`, ...), or
197 /// `inference` / `turn`.
198 pub name: String,
199 pub started_at: SystemTime,
200 /// Wall-clock duration, or `None` while the call is still in flight.
201 pub duration_ms: Option<u64>,
202 /// The call was issued by a subagent (Claude's `isSidechain`).
203 pub sidechain: bool,
204 /// The harness reported the result as an error.
205 pub error: bool,
206 /// Absent in snapshots written before 0.3.1, which held tool calls only.
207 #[serde(default)]
208 pub kind: SpanKind,
209}
210
211impl ToolSpan {
212 pub fn is_open(&self) -> bool {
213 self.duration_ms.is_none()
214 }
215
216 /// Duration if closed, otherwise how long it has been running as of `now`.
217 pub fn elapsed_ms(&self, now: SystemTime) -> u64 {
218 match self.duration_ms {
219 Some(ms) => ms,
220 None => now.duration_since(self.started_at).map(|d| d.as_millis() as u64).unwrap_or(0),
221 }
222 }
223
224 pub fn ended_at(&self, now: SystemTime) -> SystemTime {
225 self.started_at + std::time::Duration::from_millis(self.elapsed_ms(now))
226 }
227}
228
229/// The last two components of a path, so `/Users/me/code/app` reads as
230/// `code/app` and two projects called `app` are still told apart.
231pub fn project_name(p: &std::path::Path) -> String {
232 let names: Vec<String> = p
233 .components()
234 .filter_map(|c| match c {
235 std::path::Component::Normal(s) => Some(s.to_string_lossy().into_owned()),
236 _ => None,
237 })
238 .collect();
239 let tail = names.iter().rev().take(2).rev().cloned().collect::<Vec<_>>().join("/");
240 if tail.is_empty() { p.to_string_lossy().into_owned() } else { tail }
241}
242
243/// The turn a span belongs to: the newest turn that started at or before it
244/// and had not ended when it started. Subagent spans prefer the subagent's
245/// own turn, which their transcript carries, and fall back to the main
246/// agent's. A turn has no parent.
247pub fn parent_turn<'a>(spans: &[&'a ToolSpan], i: usize) -> Option<&'a ToolSpan> {
248 let sp = spans[i];
249 if sp.kind == SpanKind::Turn {
250 return None;
251 }
252 let contains = |t: &ToolSpan| {
253 t.kind == SpanKind::Turn
254 && t.started_at <= sp.started_at
255 && t.duration_ms.map(|ms| t.started_at + std::time::Duration::from_millis(ms) >= sp.started_at).unwrap_or(true)
256 };
257 let own = spans[..i].iter().rev().find(|t| t.sidechain == sp.sidechain && contains(t));
258 own.or_else(|| spans[..i].iter().rev().find(|t| !t.sidechain && contains(t))).copied()
259}
260
261/// One MCP server an agent uses, seen from either side or both: the process
262/// table has the server's pid, CPU and memory; the transcript has how often
263/// the agent called it. Claude Code names an MCP tool `mcp__<server>__<tool>`,
264/// which is where the server name and the call count come from.
265#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
266pub struct McpServer {
267 /// The server's name as the harness configured it, or, for a process the
268 /// transcript never named, the program it is running.
269 pub name: String,
270 pub pid: Option<u32>,
271 pub cmdline: Option<String>,
272 pub cpu_percent: f32,
273 pub rss_bytes: u64,
274 pub age_secs: Option<u64>,
275 /// Tool calls the agent made to this server, from the transcript.
276 pub calls: u64,
277 /// Of those, how many the harness reported as errors.
278 pub errors: u64,
279 pub last_call: Option<SystemTime>,
280 /// How the process and the transcript's server were put together, for
281 /// the UI to label a guess as one.
282 pub matched_by: McpMatch,
283}
284
285/// How an `McpServer` row was formed.
286#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
287#[serde(rename_all = "kebab-case")]
288pub enum McpMatch {
289 /// A process under the agent that no transcript server names. Either it
290 /// has not been called yet, or its name does not appear in its command.
291 ProcessOnly,
292 /// A server the transcript calls with no process under the agent: an HTTP
293 /// server, or one that has exited.
294 TranscriptOnly,
295 /// The server's name appears in the process's command line.
296 Name,
297 /// One unmatched process and one unmatched server were left; they are
298 /// taken to be the same. A guess.
299 Sole,
300}
301
302/// Where a stretch of a session's context came from.
303#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
304#[serde(rename_all = "kebab-case")]
305pub enum ContextOrigin {
306 /// A built-in tool of the harness (`Bash`, `Read`, `exec_command`, ...).
307 Tool,
308 /// An MCP server; `name` is the server, not the tool.
309 Mcp,
310 /// Everything that is not a tool result: the system prompt, the user's
311 /// own messages, a compaction summary.
312 #[default]
313 Other,
314}
315
316/// One source of a session's context: what its results have added to the
317/// prompt, and what sending that on every inference since has cost.
318///
319/// The tokens are the growth of the prompt between one response and the
320/// next, attributed to the tool results submitted in between; the cost is
321/// those tokens charged at the prompt rate of every response that carried
322/// them, the first read included. The rows sum to the session's prompt-side
323/// cost. See `docs/accounting.md`.
324#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
325pub struct ContextSource {
326 pub name: String,
327 pub origin: ContextOrigin,
328 /// Tool results attributed to this source. Zero for `Other`.
329 pub calls: u64,
330 /// Tokens the source added to the prompt over the session.
331 pub tokens: u64,
332 /// USD those tokens have cost across every response that read them.
333 /// An estimate; zero when the model has no price.
334 pub cost_usd: f64,
335}
336
337/// Role of a process inside an agent's tree.
338#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
339#[serde(rename_all = "kebab-case")]
340pub enum ProcKind {
341 /// A harness process, whether at the root or nested under another process.
342 /// Older snapshots called nested harness processes `subagent`.
343 #[serde(alias = "subagent")]
344 Agent,
345 /// A Model Context Protocol server or helper.
346 Mcp,
347 /// A shell spawned to run a tool call.
348 Shell,
349 /// Anything else the agent launched (test runners, sleep, caffeinate, ...).
350 Tool,
351}
352
353impl ProcKind {
354 pub fn label(self) -> &'static str {
355 match self {
356 ProcKind::Agent => "agent",
357 ProcKind::Mcp => "mcp",
358 ProcKind::Shell => "shell",
359 ProcKind::Tool => "tool",
360 }
361 }
362}
363
364/// One process, with its descendants.
365#[derive(Debug, Clone, Serialize, Deserialize)]
366pub struct ProcNode {
367 pub pid: u32,
368 pub ppid: Option<u32>,
369 pub name: String,
370 pub cmdline: String,
371 pub kind: ProcKind,
372 pub harness: Option<Harness>,
373 pub cpu_percent: f32,
374 pub rss_bytes: u64,
375 pub age_secs: u64,
376 pub cwd: Option<PathBuf>,
377 pub children: Vec<ProcNode>,
378}
379
380impl ProcNode {
381 /// CPU, RSS, process count and MCP server count summed over the whole
382 /// subtree. An MCP process's own children (an `npx` wrapper's `node`) are
383 /// the same server, so they add to the process count and not to the
384 /// server count.
385 pub fn totals(&self) -> (f32, u64, usize, usize) {
386 let mut cpu = self.cpu_percent;
387 let mut rss = self.rss_bytes;
388 let mut count = 1;
389 let mut mcp = usize::from(self.kind == ProcKind::Mcp);
390 for c in &self.children {
391 let (ccpu, crss, ccount, cmcp) = c.totals();
392 cpu += ccpu;
393 rss += crss;
394 count += ccount;
395 if self.kind != ProcKind::Mcp {
396 mcp += cmcp;
397 }
398 }
399 (cpu, rss, count, mcp)
400 }
401
402 /// The MCP servers in this tree: each `Mcp` node whose parent is not one.
403 /// A server started through `npx` or `uvx` is two or three processes; the
404 /// top one stands for the server.
405 pub fn mcp_roots(&self) -> Vec<&ProcNode> {
406 let mut out = Vec::new();
407 self.collect_mcp_roots(&mut out);
408 out
409 }
410
411 fn collect_mcp_roots<'a>(&'a self, out: &mut Vec<&'a ProcNode>) {
412 if self.kind == ProcKind::Mcp {
413 out.push(self);
414 return;
415 }
416 for c in &self.children {
417 c.collect_mcp_roots(out);
418 }
419 }
420
421 pub fn walk<'a>(&'a self, depth: usize, f: &mut dyn FnMut(&'a ProcNode, usize)) {
422 f(self, depth);
423 for c in &self.children {
424 c.walk(depth + 1, f);
425 }
426 }
427}
428
429/// A logical subagent relationship explicitly recorded by the harness.
430/// This is session lineage, not an OS process relationship or a history fork.
431#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
432pub struct SubagentInfo {
433 pub parent_session_id: String,
434 #[serde(default)]
435 pub nickname: Option<String>,
436 #[serde(default)]
437 pub role: Option<String>,
438}
439
440/// A single row in the agent table.
441#[derive(Debug, Clone, Serialize, Deserialize)]
442pub struct Agent {
443 /// Stable identity across refreshes: `pid:<n>` for live agents, `session:<id>` for stopped ones.
444 pub id: String,
445 pub name: String,
446 pub harness: Harness,
447 pub state: AgentState,
448 pub activity: Activity,
449 pub pid: Option<u32>,
450 pub session_id: Option<String>,
451 /// Explicit logical parent and identity, when the transcript records them.
452 /// Usage remains per session; adding hierarchy must not count it twice.
453 #[serde(default, skip_serializing_if = "Option::is_none")]
454 pub subagent: Option<SubagentInfo>,
455 pub session_path: Option<PathBuf>,
456 pub cwd: Option<PathBuf>,
457 pub model: Option<String>,
458 pub harness_version: Option<String>,
459 pub usage: TokenUsage,
460 /// USD spent on messages whose model had a known price.
461 pub cost_usd: f64,
462 /// `cost_usd` by kind of token, so a figure that differs from another
463 /// tool's can be traced to the one line that differs.
464 #[serde(default)]
465 pub cost_breakdown: CostBreakdown,
466 /// Where this row's costs came from; `None` when it has no known price.
467 #[serde(default)]
468 pub price_source: Option<PriceSource>,
469 /// Tokens on messages whose model had no known price (so `cost_usd` is a floor).
470 pub unpriced_tokens: u64,
471 pub turns: u64,
472 pub subagent_turns: u64,
473 /// See `SessionSummary::folds_child_usage`. Claude Code, Gemini CLI and
474 /// OpenCode fold a child's usage into its parent, the way those harnesses
475 /// bill it; Codex and Kodelet give each child its own row instead.
476 #[serde(default)]
477 pub folds_child_usage: bool,
478 pub tool_calls: u64,
479 /// The count is only known retained/observed calls, not an exact lifetime
480 /// total. Kodelet compaction discards old calls without a cumulative count.
481 #[serde(default)]
482 pub tool_calls_lower_bound: bool,
483 /// Server-side web searches the model ran, billed per search on top of
484 /// tokens. Counted for every harness; priced only where the price table
485 /// has a rate (Anthropic's, for Claude Code).
486 #[serde(default)]
487 pub web_searches: u64,
488 /// The most recent spans, oldest first: tool calls, inferences and turns.
489 /// Bounded; see `harness::MAX_SPANS`.
490 pub spans: Vec<ToolSpan>,
491 /// Seconds since the process started (live) or since the last transcript write (stopped).
492 pub age_secs: u64,
493 /// Seconds since the transcript was last written.
494 pub idle_secs: Option<u64>,
495 pub cpu_percent: f32,
496 pub rss_bytes: u64,
497 pub process_count: usize,
498 pub mcp_count: usize,
499 /// The MCP servers this agent uses, one row each, from the process tree
500 /// and the transcript. See `McpServer`.
501 #[serde(default)]
502 pub mcp_servers: Vec<McpServer>,
503 /// What each tool's results added to the prompt and what carrying it has
504 /// cost, largest first. See `ContextSource`.
505 #[serde(default)]
506 pub context: Vec<ContextSource>,
507 pub tree: Option<ProcNode>,
508 /// How the session was attributed to the process, for debugging attribution.
509 pub attribution: Attribution,
510 /// True when another row shares this pid and carries its CPU, memory and
511 /// process counts. One Codex app-server hosts many conversations, so its
512 /// threads each get a row while the process is only counted once.
513 /// Absent in snapshots written before 0.2.0.
514 #[serde(default)]
515 pub shares_process: bool,
516 /// Set when the transcript parsed but its usage records did not, which
517 /// means this row's tokens and cost are not to be believed. Almost always a
518 /// harness that changed its format under us.
519 pub parse_warning: Option<String>,
520 /// How close the session is to its rate limit, when the harness reports it.
521 #[serde(default)]
522 pub rate_limit: Option<RateLimit>,
523 /// The session's cost as the harness itself last recorded it, beside
524 /// `cost_usd` and never mixed into it. See `HarnessCost`.
525 #[serde(default, skip_serializing_if = "Option::is_none")]
526 pub harness_cost: Option<HarnessCost>,
527}
528
529/// A harness's own running total for a session, read from its transcript.
530/// Claude Code writes one; the other harnesses do not, or (OpenCode, Kodelet)
531/// their costs are already the row's `cost_usd`.
532///
533/// It is a different measurement from `cost_usd`, not a correction to it: it
534/// is priced at the harness's private table, and it includes requests the
535/// transcript never records. Claude Code writes it when a session exits, so
536/// while a session runs it is the figure from the last exit.
537#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
538pub struct HarnessCost {
539 pub usd: f64,
540 /// The harness said some of its usage had no price, so `usd` is a floor.
541 pub lower_bound: bool,
542 /// The timestamp of the last transcript line before the record, which
543 /// carries none of its own.
544 pub as_of: Option<SystemTime>,
545 /// No usage has been recorded since, in the transcript or a subagent's.
546 /// False for a live session that has done anything since it last exited:
547 /// `usd` leaves that out.
548 pub current: bool,
549}
550
551/// One rolling usage window a harness reports against a rate limit: how much
552/// of it is spent and when it rolls over. Codex writes two, a short window and
553/// a long one, on every usage record.
554#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
555pub struct RateWindow {
556 /// Percent of the window used, 0..=100.
557 pub used_percent: f64,
558 /// The window length in minutes (Codex: 300 for the short one, 10080 weekly).
559 pub window_minutes: u64,
560 /// When the window rolls over and the usage resets, if the harness says.
561 pub resets_at: Option<SystemTime>,
562}
563
564/// What a harness reports about how close a session is to its rate limit. Only
565/// the harnesses that write it (Codex today) populate this; the rest leave it
566/// `None`. Read-only, like everything else: a number the harness already wrote.
567#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
568pub struct RateLimit {
569 /// The short rolling window.
570 pub primary: Option<RateWindow>,
571 /// The long rolling window.
572 pub secondary: Option<RateWindow>,
573 /// The plan the limit is for (Codex: `plus`, `pro`, ...).
574 pub plan: Option<String>,
575 /// True when the harness says the limit is currently hit.
576 pub reached: bool,
577}
578
579impl RateLimit {
580 /// The window closest to its limit, for a one-glance figure.
581 pub fn tightest(&self) -> Option<&RateWindow> {
582 [self.primary.as_ref(), self.secondary.as_ref()]
583 .into_iter()
584 .flatten()
585 .max_by(|a, b| a.used_percent.partial_cmp(&b.used_percent).unwrap_or(std::cmp::Ordering::Equal))
586 }
587}
588
589/// USD by kind of token, accumulated message by message at each message's
590/// own model price, so a session that changed model part way is still exact.
591/// The lines sum to `Agent::cost_usd`.
592#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
593pub struct CostBreakdown {
594 pub input: f64,
595 pub cache_write_5m: f64,
596 pub cache_write_1h: f64,
597 /// What a harness recorded against `TokenUsage::cache_write_unsplit`.
598 /// Always zero when the cost came from a price table.
599 #[serde(default)]
600 pub cache_write_unsplit: f64,
601 pub cache_read: f64,
602 pub output: f64,
603 /// Server-side web searches, billed per search on top of the tokens.
604 pub web_search: f64,
605}
606
607impl CostBreakdown {
608 pub fn total(&self) -> f64 {
609 self.input + self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit + self.cache_read + self.output + self.web_search
610 }
611
612 pub fn add(&mut self, o: &CostBreakdown) {
613 self.input += o.input;
614 self.cache_write_5m += o.cache_write_5m;
615 self.cache_write_1h += o.cache_write_1h;
616 self.cache_write_unsplit += o.cache_write_unsplit;
617 self.cache_read += o.cache_read;
618 self.output += o.output;
619 self.web_search += o.web_search;
620 }
621
622 pub fn sub(&mut self, o: &CostBreakdown) {
623 self.input -= o.input;
624 self.cache_write_5m -= o.cache_write_5m;
625 self.cache_write_1h -= o.cache_write_1h;
626 self.cache_write_unsplit -= o.cache_write_unsplit;
627 self.cache_read -= o.cache_read;
628 self.output -= o.output;
629 self.web_search -= o.web_search;
630 }
631}
632
633/// Where costs came from: list prices, a user's price file, or the harness's
634/// own recorded accounting. The UI says which rather than implying a bill.
635#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
636#[serde(rename_all = "kebab-case")]
637pub enum PriceSource {
638 Builtin,
639 UserFile,
640 Harness,
641}
642
643#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
644#[serde(rename_all = "kebab-case")]
645pub enum Attribution {
646 /// The harness told us (Claude's `~/.claude/sessions/<pid>.json`).
647 HarnessRegistry,
648 /// A `--resume <id>` style argument on the command line.
649 CommandLine,
650 /// The process has the transcript open (Codex keeps every live thread's
651 /// rollout open); exact.
652 OpenFile,
653 /// Matched by working directory and start time; may be wrong with concurrent sessions.
654 CwdHeuristic,
655 /// No transcript found; process only.
656 None,
657 /// No process; transcript only.
658 TranscriptOnly,
659}
660
661#[derive(Debug, Clone, Default, Serialize, Deserialize)]
662pub struct HostStats {
663 pub hostname: Option<String>,
664 pub cpu_percent: f32,
665 pub cpu_count: usize,
666 pub mem_used_bytes: u64,
667 pub mem_total_bytes: u64,
668}
669
670#[derive(Debug, Clone, Default, Serialize, Deserialize)]
671pub struct Totals {
672 pub agents: usize,
673 pub running: usize,
674 pub idle: usize,
675 pub stopped: usize,
676 pub tokens: u64,
677 pub cost_usd: f64,
678 pub unpriced_tokens: u64,
679 pub processes: usize,
680 pub mcp_processes: usize,
681 pub orphaned_mcp: usize,
682 pub cpu_percent: f32,
683 pub rss_bytes: u64,
684}
685
686/// Version of the `--json` document. Bumped when a field changes meaning or
687/// disappears; new fields alone do not bump it.
688pub const SNAPSHOT_SCHEMA_VERSION: u32 = 1;
689
690/// What the collector remembers about an orphaned MCP process: when it first
691/// saw it, and, if it watched the process lose its parent, which agent that
692/// was. Memory lasts for the run; a process that was already an orphan when
693/// agent-top started has no parent on record.
694#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
695pub struct OrphanOrigin {
696 pub pid: u32,
697 pub first_seen: SystemTime,
698 /// When the process was first seen without its parent, if the parent was
699 /// seen before that.
700 pub orphaned_at: Option<SystemTime>,
701 pub parent: Option<OrphanParent>,
702}
703
704/// The agent an orphan used to belong to.
705#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
706pub struct OrphanParent {
707 pub pid: u32,
708 pub agent_id: String,
709 pub name: String,
710}
711
712/// Which rule a piece of advice came from. See `advice`.
713#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
714#[serde(rename_all = "kebab-case")]
715pub enum AdviceRule {
716 /// A tool or MCP server whose results are large per call and have been
717 /// re-read on every response since: the biggest line in the bill hiding
718 /// behind a few calls.
719 ExpensiveSource,
720 /// An MCP server attached to a live agent that has answered no calls in
721 /// a long while.
722 IdleMcpServer,
723 /// An MCP server whose memory has climbed steadily with no calls to
724 /// explain it.
725 GrowingMcpServer,
726}
727
728impl AdviceRule {
729 pub fn label(self) -> &'static str {
730 match self {
731 AdviceRule::ExpensiveSource => "expensive source",
732 AdviceRule::IdleMcpServer => "idle mcp server",
733 AdviceRule::GrowingMcpServer => "growing mcp server",
734 }
735 }
736}
737
738/// One sentence of advice about one agent, with the numbers that back it and
739/// the thing you could do. Nothing is done for you: agent-top only points.
740#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
741pub struct Advice {
742 pub agent_id: String,
743 pub agent_name: String,
744 pub rule: AdviceRule,
745 /// The tool, or the MCP server, the advice is about.
746 pub subject: String,
747 /// The server's process, when the advice is about one.
748 pub pid: Option<u32>,
749 /// What was seen, with its numbers.
750 pub headline: String,
751 /// What you might do about it.
752 pub action: String,
753 /// The evidence as numbers, for scripts: zero where a rule has none.
754 #[serde(default)]
755 pub cost_usd: f64,
756 #[serde(default)]
757 pub tokens: u64,
758 #[serde(default)]
759 pub calls: u64,
760 #[serde(default)]
761 pub rss_bytes: u64,
762}
763
764/// Everything the UI needs for one frame.
765#[derive(Debug, Clone, Serialize, Deserialize)]
766pub struct Snapshot {
767 pub schema_version: u32,
768 pub taken_at: SystemTime,
769 pub host: HostStats,
770 pub agents: Vec<Agent>,
771 /// MCP-looking processes with no live agent ancestor: leak candidates.
772 pub orphans: Vec<ProcNode>,
773 /// One entry per orphan, saying where it came from when that is known.
774 #[serde(default)]
775 pub orphan_origins: Vec<OrphanOrigin>,
776 /// What looks like a bad deal on this machine right now, and what could
777 /// be done about it. See `advice`.
778 #[serde(default)]
779 pub advice: Vec<Advice>,
780 pub totals: Totals,
781}
782
783impl Snapshot {
784 pub fn compute_totals(&mut self) {
785 let mut t = Totals::default();
786 for a in &self.agents {
787 t.agents += 1;
788 match a.state {
789 AgentState::Running => t.running += 1,
790 AgentState::Idle => t.idle += 1,
791 AgentState::Stopped => t.stopped += 1,
792 }
793 t.tokens += a.usage.total();
794 t.cost_usd += a.cost_usd;
795 t.unpriced_tokens += a.unpriced_tokens;
796 t.processes += a.process_count;
797 t.mcp_processes += a.mcp_count;
798 t.cpu_percent += a.cpu_percent;
799 t.rss_bytes += a.rss_bytes;
800 }
801 t.orphaned_mcp = self.orphans.len();
802 self.totals = t;
803 }
804}
805
806#[cfg(test)]
807mod usage_tests {
808 use super::TokenUsage;
809
810 #[test]
811 fn cache_hit_rate_is_reads_over_the_prompt() {
812 let u = TokenUsage { input: 200, cache_read: 800, output: 50, ..Default::default() };
813 // Prompt is 1000 (output excluded); 800 of it from cache.
814 assert_eq!(u.prompt(), 1000);
815 assert!((u.cache_hit_rate().unwrap() - 0.8).abs() < 1e-9);
816 // A cache write counts as prompt input, not as a hit.
817 let u = TokenUsage { input: 100, cache_write_5m: 900, ..Default::default() };
818 assert_eq!(u.cache_hit_rate(), Some(0.0));
819 // Nothing to judge.
820 assert_eq!(TokenUsage::default().cache_hit_rate(), None);
821 }
822}
823
824#[cfg(test)]
825mod tests {
826 use super::*;
827 use std::path::Path;
828
829 #[test]
830 fn project_name_keeps_two_components() {
831 assert_eq!(project_name(Path::new("/Users/me/code/app")), "code/app");
832 assert_eq!(project_name(Path::new("/app")), "app");
833 }
834}