agent_top_core/model.rs
1//! The data model shared by discovery, the TUI and the JSON output.
2
3use serde::{Deserialize, Serialize};
4use std::path::PathBuf;
5use std::time::SystemTime;
6
7/// Which agent harness a process or transcript belongs to.
8#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
9#[serde(rename_all = "kebab-case")]
10pub enum Harness {
11 Claude,
12 Codex,
13 Gemini,
14 OpenCode,
15 Kodelet,
16 Aider,
17 Copilot,
18 Cursor,
19 Unknown,
20}
21
22impl Harness {
23 pub fn label(self) -> &'static str {
24 match self {
25 Harness::Claude => "claude",
26 Harness::Codex => "codex",
27 Harness::Gemini => "gemini",
28 Harness::OpenCode => "opencode",
29 Harness::Kodelet => "kodelet",
30 Harness::Aider => "aider",
31 Harness::Copilot => "copilot",
32 Harness::Cursor => "cursor",
33 Harness::Unknown => "unknown",
34 }
35 }
36
37 /// How much of a session id identifies it to `agent-top trace --session`,
38 /// which matches a prefix against the ids of every harness and fails when
39 /// one matches more than one session. `None` means the whole id is needed:
40 /// Kodelet's begin with a date that every session started that day shares.
41 pub fn session_id_prefix_len(self) -> Option<usize> {
42 match self {
43 Harness::Kodelet => None,
44 Harness::Claude
45 | Harness::Codex
46 | Harness::Gemini
47 | Harness::OpenCode
48 | Harness::Aider
49 | Harness::Copilot
50 | Harness::Cursor
51 | Harness::Unknown => Some(8),
52 }
53 }
54}
55
56/// Coarse lifecycle state, in the htop sense.
57///
58/// `Running` means the agent is mid-turn (inference or tool execution),
59/// `Idle` means the process is alive but waiting for a human, `Stopped` means
60/// the transcript exists and was recently active but no process owns it.
61#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, PartialOrd, Ord)]
62#[serde(rename_all = "kebab-case")]
63pub enum AgentState {
64 Running,
65 Idle,
66 Stopped,
67}
68
69impl AgentState {
70 pub fn label(self) -> &'static str {
71 match self {
72 AgentState::Running => "running",
73 AgentState::Idle => "idle",
74 AgentState::Stopped => "stopped",
75 }
76 }
77}
78
79/// What the transcript says the agent was last doing. Harness-neutral.
80#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize, Default)]
81#[serde(rename_all = "kebab-case")]
82pub enum Activity {
83 /// Mid-turn: a prompt or tool result was just submitted, or a tool call is pending.
84 Working,
85 /// The last thing that happened was the assistant ending its turn.
86 Waiting,
87 #[default]
88 Unknown,
89}
90
91/// Token counts, split the way the Anthropic and OpenAI usage objects split them.
92#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
93pub struct TokenUsage {
94 pub input: u64,
95 pub cache_write_5m: u64,
96 pub cache_write_1h: u64,
97 /// Cache creation a harness recorded without the TTL that decides its
98 /// rate. No price table can cover it: charging either TTL's rate would be
99 /// a guess about a lifetime nobody wrote down. Only a harness that reports
100 /// its own costs can put a number against these tokens.
101 #[serde(default)]
102 pub cache_write_unsplit: u64,
103 pub cache_read: u64,
104 pub output: u64,
105}
106
107impl TokenUsage {
108 pub fn cache_write(&self) -> u64 {
109 self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit
110 }
111
112 /// Everything the model consumed or produced. This is the "TOKENS" column.
113 pub fn total(&self) -> u64 {
114 self.input + self.cache_write() + self.cache_read + self.output
115 }
116
117 /// The input side: everything sent to the model as the prompt, fresh input
118 /// plus cache writes plus cache reads, but not the output it produced.
119 pub fn prompt(&self) -> u64 {
120 self.input + self.cache_write() + self.cache_read
121 }
122
123 /// The share of the prompt served from cache (the cheap reads), in `0..=1`.
124 /// A high number means most of the re-sent conversation was billed at the
125 /// cache-read rate rather than full input; a low one on a long session is
126 /// money left on the table. `None` when there was no prompt to judge, or
127 /// the model does not cache at all.
128 pub fn cache_hit_rate(&self) -> Option<f64> {
129 let p = self.prompt();
130 (p > 0).then(|| self.cache_read as f64 / p as f64)
131 }
132
133 pub fn add(&mut self, other: &TokenUsage) {
134 self.input += other.input;
135 self.cache_write_5m += other.cache_write_5m;
136 self.cache_write_1h += other.cache_write_1h;
137 self.cache_write_unsplit += other.cache_write_unsplit;
138 self.cache_read += other.cache_read;
139 self.output += other.output;
140 }
141
142 pub fn sub(&mut self, other: &TokenUsage) {
143 self.input = self.input.saturating_sub(other.input);
144 self.cache_write_5m = self.cache_write_5m.saturating_sub(other.cache_write_5m);
145 self.cache_write_1h = self.cache_write_1h.saturating_sub(other.cache_write_1h);
146 self.cache_write_unsplit = self.cache_write_unsplit.saturating_sub(other.cache_write_unsplit);
147 self.cache_read = self.cache_read.saturating_sub(other.cache_read);
148 self.output = self.output.saturating_sub(other.output);
149 }
150}
151
152/// What a span measures. Tool calls came first and gave the type its name;
153/// the other two label the time between them.
154#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize, Default, PartialOrd, Ord)]
155#[serde(rename_all = "kebab-case")]
156pub enum SpanKind {
157 /// One tool call, from the harness issuing it to the result coming back.
158 #[default]
159 Tool,
160 /// The model producing a response: from a prompt or tool results being
161 /// submitted to the last block of the reply being written. The gap in a
162 /// waterfall that is not a tool call is almost always one of these.
163 Inference,
164 /// One human turn: from the prompt to the model ending its reply.
165 /// Contains every tool call and inference span issued in between.
166 Turn,
167}
168
169impl SpanKind {
170 pub const ALL: [SpanKind; 3] = [SpanKind::Tool, SpanKind::Inference, SpanKind::Turn];
171
172 pub fn label(self) -> &'static str {
173 match self {
174 SpanKind::Tool => "tool",
175 SpanKind::Inference => "inference",
176 SpanKind::Turn => "turn",
177 }
178 }
179}
180
181/// One span of an agent trace: a tool call, an inference, or a turn.
182///
183/// Every harness writes the same shape in its own vocabulary — Claude pairs a
184/// `tool_use` block with a `tool_result` block by `tool_use_id`, Codex pairs a
185/// `function_call` with a `function_call_output` by `call_id` — and both stamp
186/// each line with a timestamp. That is a span: a name, a start and a duration.
187/// Only the call's metadata is kept; arguments and output are never read.
188///
189/// The name predates `kind`: the type carried only tool calls until 0.3.1 and
190/// is kept for the sake of the published API.
191#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
192pub struct ToolSpan {
193 /// The harness's own call id, so a span survives across refreshes. For
194 /// inference and turn spans, a counter the parser assigns.
195 pub id: String,
196 /// Tool name as the harness reports it (`Bash`, `exec_command`, ...), or
197 /// `inference` / `turn`.
198 pub name: String,
199 pub started_at: SystemTime,
200 /// Wall-clock duration, or `None` while the call is still in flight.
201 pub duration_ms: Option<u64>,
202 /// The call was issued by a subagent (Claude's `isSidechain`).
203 pub sidechain: bool,
204 /// The harness reported the result as an error.
205 pub error: bool,
206 /// Absent in snapshots written before 0.3.1, which held tool calls only.
207 #[serde(default)]
208 pub kind: SpanKind,
209}
210
211impl ToolSpan {
212 pub fn is_open(&self) -> bool {
213 self.duration_ms.is_none()
214 }
215
216 /// Duration if closed, otherwise how long it has been running as of `now`.
217 pub fn elapsed_ms(&self, now: SystemTime) -> u64 {
218 match self.duration_ms {
219 Some(ms) => ms,
220 None => now.duration_since(self.started_at).map(|d| d.as_millis() as u64).unwrap_or(0),
221 }
222 }
223
224 pub fn ended_at(&self, now: SystemTime) -> SystemTime {
225 self.started_at + std::time::Duration::from_millis(self.elapsed_ms(now))
226 }
227}
228
229/// One MCP server an agent uses, seen from either side or both: the process
230/// table has the server's pid, CPU and memory; the transcript has how often
231/// the agent called it. Claude Code names an MCP tool `mcp__<server>__<tool>`,
232/// which is where the server name and the call count come from.
233#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
234pub struct McpServer {
235 /// The server's name as the harness configured it, or, for a process the
236 /// transcript never named, the program it is running.
237 pub name: String,
238 pub pid: Option<u32>,
239 pub cmdline: Option<String>,
240 pub cpu_percent: f32,
241 pub rss_bytes: u64,
242 pub age_secs: Option<u64>,
243 /// Tool calls the agent made to this server, from the transcript.
244 pub calls: u64,
245 /// Of those, how many the harness reported as errors.
246 pub errors: u64,
247 pub last_call: Option<SystemTime>,
248 /// How the process and the transcript's server were put together, for
249 /// the UI to label a guess as one.
250 pub matched_by: McpMatch,
251}
252
253/// How an `McpServer` row was formed.
254#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
255#[serde(rename_all = "kebab-case")]
256pub enum McpMatch {
257 /// A process under the agent that no transcript server names. Either it
258 /// has not been called yet, or its name does not appear in its command.
259 ProcessOnly,
260 /// A server the transcript calls with no process under the agent: an HTTP
261 /// server, or one that has exited.
262 TranscriptOnly,
263 /// The server's name appears in the process's command line.
264 Name,
265 /// One unmatched process and one unmatched server were left; they are
266 /// taken to be the same. A guess.
267 Sole,
268}
269
270/// Where a stretch of a session's context came from.
271#[derive(Debug, Clone, Copy, Default, PartialEq, Eq, PartialOrd, Ord, Hash, Serialize, Deserialize)]
272#[serde(rename_all = "kebab-case")]
273pub enum ContextOrigin {
274 /// A built-in tool of the harness (`Bash`, `Read`, `exec_command`, ...).
275 Tool,
276 /// An MCP server; `name` is the server, not the tool.
277 Mcp,
278 /// Everything that is not a tool result: the system prompt, the user's
279 /// own messages, a compaction summary.
280 #[default]
281 Other,
282}
283
284/// One source of a session's context: what its results have added to the
285/// prompt, and what sending that on every inference since has cost.
286///
287/// The tokens are the growth of the prompt between one response and the
288/// next, attributed to the tool results submitted in between; the cost is
289/// those tokens charged at the prompt rate of every response that carried
290/// them, the first read included. The rows sum to the session's prompt-side
291/// cost. See `docs/accounting.md`.
292#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
293pub struct ContextSource {
294 pub name: String,
295 pub origin: ContextOrigin,
296 /// Tool results attributed to this source. Zero for `Other`.
297 pub calls: u64,
298 /// Tokens the source added to the prompt over the session.
299 pub tokens: u64,
300 /// USD those tokens have cost across every response that read them.
301 /// An estimate; zero when the model has no price.
302 pub cost_usd: f64,
303}
304
305/// Role of a process inside an agent's tree.
306#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
307#[serde(rename_all = "kebab-case")]
308pub enum ProcKind {
309 /// A harness process, whether at the root or nested under another process.
310 /// Older snapshots called nested harness processes `subagent`.
311 #[serde(alias = "subagent")]
312 Agent,
313 /// A Model Context Protocol server or helper.
314 Mcp,
315 /// A shell spawned to run a tool call.
316 Shell,
317 /// Anything else the agent launched (test runners, sleep, caffeinate, ...).
318 Tool,
319}
320
321impl ProcKind {
322 pub fn label(self) -> &'static str {
323 match self {
324 ProcKind::Agent => "agent",
325 ProcKind::Mcp => "mcp",
326 ProcKind::Shell => "shell",
327 ProcKind::Tool => "tool",
328 }
329 }
330}
331
332/// One process, with its descendants.
333#[derive(Debug, Clone, Serialize, Deserialize)]
334pub struct ProcNode {
335 pub pid: u32,
336 pub ppid: Option<u32>,
337 pub name: String,
338 pub cmdline: String,
339 pub kind: ProcKind,
340 pub harness: Option<Harness>,
341 pub cpu_percent: f32,
342 pub rss_bytes: u64,
343 pub age_secs: u64,
344 pub cwd: Option<PathBuf>,
345 pub children: Vec<ProcNode>,
346}
347
348impl ProcNode {
349 /// CPU, RSS, process count and MCP server count summed over the whole
350 /// subtree. An MCP process's own children (an `npx` wrapper's `node`) are
351 /// the same server, so they add to the process count and not to the
352 /// server count.
353 pub fn totals(&self) -> (f32, u64, usize, usize) {
354 let mut cpu = self.cpu_percent;
355 let mut rss = self.rss_bytes;
356 let mut count = 1;
357 let mut mcp = usize::from(self.kind == ProcKind::Mcp);
358 for c in &self.children {
359 let (ccpu, crss, ccount, cmcp) = c.totals();
360 cpu += ccpu;
361 rss += crss;
362 count += ccount;
363 if self.kind != ProcKind::Mcp {
364 mcp += cmcp;
365 }
366 }
367 (cpu, rss, count, mcp)
368 }
369
370 /// The MCP servers in this tree: each `Mcp` node whose parent is not one.
371 /// A server started through `npx` or `uvx` is two or three processes; the
372 /// top one stands for the server.
373 pub fn mcp_roots(&self) -> Vec<&ProcNode> {
374 let mut out = Vec::new();
375 self.collect_mcp_roots(&mut out);
376 out
377 }
378
379 fn collect_mcp_roots<'a>(&'a self, out: &mut Vec<&'a ProcNode>) {
380 if self.kind == ProcKind::Mcp {
381 out.push(self);
382 return;
383 }
384 for c in &self.children {
385 c.collect_mcp_roots(out);
386 }
387 }
388
389 pub fn walk<'a>(&'a self, depth: usize, f: &mut dyn FnMut(&'a ProcNode, usize)) {
390 f(self, depth);
391 for c in &self.children {
392 c.walk(depth + 1, f);
393 }
394 }
395}
396
397/// A logical subagent relationship explicitly recorded by the harness.
398/// This is session lineage, not an OS process relationship or a history fork.
399#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
400pub struct SubagentInfo {
401 pub parent_session_id: String,
402 #[serde(default)]
403 pub nickname: Option<String>,
404 #[serde(default)]
405 pub role: Option<String>,
406}
407
408/// A single row in the agent table.
409#[derive(Debug, Clone, Serialize, Deserialize)]
410pub struct Agent {
411 /// Stable identity across refreshes: `pid:<n>` for live agents, `session:<id>` for stopped ones.
412 pub id: String,
413 pub name: String,
414 pub harness: Harness,
415 pub state: AgentState,
416 pub activity: Activity,
417 pub pid: Option<u32>,
418 pub session_id: Option<String>,
419 /// Explicit logical parent and identity, when the transcript records them.
420 /// Usage remains per session; adding hierarchy must not count it twice.
421 #[serde(default, skip_serializing_if = "Option::is_none")]
422 pub subagent: Option<SubagentInfo>,
423 pub session_path: Option<PathBuf>,
424 pub cwd: Option<PathBuf>,
425 pub model: Option<String>,
426 pub harness_version: Option<String>,
427 pub usage: TokenUsage,
428 /// USD spent on messages whose model had a known price.
429 pub cost_usd: f64,
430 /// `cost_usd` by kind of token, so a figure that differs from another
431 /// tool's can be traced to the one line that differs.
432 #[serde(default)]
433 pub cost_breakdown: CostBreakdown,
434 /// Where this row's costs came from; `None` when it has no known price.
435 #[serde(default)]
436 pub price_source: Option<PriceSource>,
437 /// Tokens on messages whose model had no known price (so `cost_usd` is a floor).
438 pub unpriced_tokens: u64,
439 pub turns: u64,
440 pub subagent_turns: u64,
441 /// See `SessionSummary::folds_child_usage`. Claude Code, Gemini CLI and
442 /// OpenCode fold a child's usage into its parent, the way those harnesses
443 /// bill it; Codex and Kodelet give each child its own row instead.
444 #[serde(default)]
445 pub folds_child_usage: bool,
446 pub tool_calls: u64,
447 /// The count is only known retained/observed calls, not an exact lifetime
448 /// total. Kodelet compaction discards old calls without a cumulative count.
449 #[serde(default)]
450 pub tool_calls_lower_bound: bool,
451 /// Server-side web searches the model ran, billed per search on top of
452 /// tokens. Counted for every harness; priced only where the price table
453 /// has a rate (Anthropic's, for Claude Code).
454 #[serde(default)]
455 pub web_searches: u64,
456 /// The most recent spans, oldest first: tool calls, inferences and turns.
457 /// Bounded; see `harness::MAX_SPANS`.
458 pub spans: Vec<ToolSpan>,
459 /// Seconds since the process started (live) or since the last transcript write (stopped).
460 pub age_secs: u64,
461 /// Seconds since the transcript was last written.
462 pub idle_secs: Option<u64>,
463 pub cpu_percent: f32,
464 pub rss_bytes: u64,
465 pub process_count: usize,
466 pub mcp_count: usize,
467 /// The MCP servers this agent uses, one row each, from the process tree
468 /// and the transcript. See `McpServer`.
469 #[serde(default)]
470 pub mcp_servers: Vec<McpServer>,
471 /// What each tool's results added to the prompt and what carrying it has
472 /// cost, largest first. See `ContextSource`.
473 #[serde(default)]
474 pub context: Vec<ContextSource>,
475 pub tree: Option<ProcNode>,
476 /// How the session was attributed to the process, for debugging attribution.
477 pub attribution: Attribution,
478 /// True when another row shares this pid and carries its CPU, memory and
479 /// process counts. One Codex app-server hosts many conversations, so its
480 /// threads each get a row while the process is only counted once.
481 /// Absent in snapshots written before 0.2.0.
482 #[serde(default)]
483 pub shares_process: bool,
484 /// Set when the transcript parsed but its usage records did not, which
485 /// means this row's tokens and cost are not to be believed. Almost always a
486 /// harness that changed its format under us.
487 pub parse_warning: Option<String>,
488 /// How close the session is to its rate limit, when the harness reports it.
489 #[serde(default)]
490 pub rate_limit: Option<RateLimit>,
491 /// The session's cost as the harness itself last recorded it, beside
492 /// `cost_usd` and never mixed into it. See `HarnessCost`.
493 #[serde(default, skip_serializing_if = "Option::is_none")]
494 pub harness_cost: Option<HarnessCost>,
495}
496
497/// A harness's own running total for a session, read from its transcript.
498/// Claude Code writes one; the other harnesses do not, or (OpenCode, Kodelet)
499/// their costs are already the row's `cost_usd`.
500///
501/// It is a different measurement from `cost_usd`, not a correction to it: it
502/// is priced at the harness's private table, and it includes requests the
503/// transcript never records. Claude Code writes it when a session exits, so
504/// while a session runs it is the figure from the last exit.
505#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
506pub struct HarnessCost {
507 pub usd: f64,
508 /// The harness said some of its usage had no price, so `usd` is a floor.
509 pub lower_bound: bool,
510 /// The timestamp of the last transcript line before the record, which
511 /// carries none of its own.
512 pub as_of: Option<SystemTime>,
513 /// No usage has been recorded since, in the transcript or a subagent's.
514 /// False for a live session that has done anything since it last exited:
515 /// `usd` leaves that out.
516 pub current: bool,
517}
518
519/// One rolling usage window a harness reports against a rate limit: how much
520/// of it is spent and when it rolls over. Codex writes two, a short window and
521/// a long one, on every usage record.
522#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
523pub struct RateWindow {
524 /// Percent of the window used, 0..=100.
525 pub used_percent: f64,
526 /// The window length in minutes (Codex: 300 for the short one, 10080 weekly).
527 pub window_minutes: u64,
528 /// When the window rolls over and the usage resets, if the harness says.
529 pub resets_at: Option<SystemTime>,
530}
531
532/// What a harness reports about how close a session is to its rate limit. Only
533/// the harnesses that write it (Codex today) populate this; the rest leave it
534/// `None`. Read-only, like everything else: a number the harness already wrote.
535#[derive(Debug, Clone, Default, Serialize, Deserialize, PartialEq)]
536pub struct RateLimit {
537 /// The short rolling window.
538 pub primary: Option<RateWindow>,
539 /// The long rolling window.
540 pub secondary: Option<RateWindow>,
541 /// The plan the limit is for (Codex: `plus`, `pro`, ...).
542 pub plan: Option<String>,
543 /// True when the harness says the limit is currently hit.
544 pub reached: bool,
545}
546
547impl RateLimit {
548 /// The window closest to its limit, for a one-glance figure.
549 pub fn tightest(&self) -> Option<&RateWindow> {
550 [self.primary.as_ref(), self.secondary.as_ref()]
551 .into_iter()
552 .flatten()
553 .max_by(|a, b| a.used_percent.partial_cmp(&b.used_percent).unwrap_or(std::cmp::Ordering::Equal))
554 }
555}
556
557/// USD by kind of token, accumulated message by message at each message's
558/// own model price, so a session that changed model part way is still exact.
559/// The lines sum to `Agent::cost_usd`.
560#[derive(Debug, Clone, Copy, Default, Serialize, Deserialize, PartialEq)]
561pub struct CostBreakdown {
562 pub input: f64,
563 pub cache_write_5m: f64,
564 pub cache_write_1h: f64,
565 /// What a harness recorded against `TokenUsage::cache_write_unsplit`.
566 /// Always zero when the cost came from a price table.
567 #[serde(default)]
568 pub cache_write_unsplit: f64,
569 pub cache_read: f64,
570 pub output: f64,
571 /// Server-side web searches, billed per search on top of the tokens.
572 pub web_search: f64,
573}
574
575impl CostBreakdown {
576 pub fn total(&self) -> f64 {
577 self.input + self.cache_write_5m + self.cache_write_1h + self.cache_write_unsplit + self.cache_read + self.output + self.web_search
578 }
579
580 pub fn add(&mut self, o: &CostBreakdown) {
581 self.input += o.input;
582 self.cache_write_5m += o.cache_write_5m;
583 self.cache_write_1h += o.cache_write_1h;
584 self.cache_write_unsplit += o.cache_write_unsplit;
585 self.cache_read += o.cache_read;
586 self.output += o.output;
587 self.web_search += o.web_search;
588 }
589
590 pub fn sub(&mut self, o: &CostBreakdown) {
591 self.input -= o.input;
592 self.cache_write_5m -= o.cache_write_5m;
593 self.cache_write_1h -= o.cache_write_1h;
594 self.cache_write_unsplit -= o.cache_write_unsplit;
595 self.cache_read -= o.cache_read;
596 self.output -= o.output;
597 self.web_search -= o.web_search;
598 }
599}
600
601/// Where costs came from: list prices, a user's price file, or the harness's
602/// own recorded accounting. The UI says which rather than implying a bill.
603#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
604#[serde(rename_all = "kebab-case")]
605pub enum PriceSource {
606 Builtin,
607 UserFile,
608 Harness,
609}
610
611#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
612#[serde(rename_all = "kebab-case")]
613pub enum Attribution {
614 /// The harness told us (Claude's `~/.claude/sessions/<pid>.json`).
615 HarnessRegistry,
616 /// A `--resume <id>` style argument on the command line.
617 CommandLine,
618 /// The process has the transcript open (Codex keeps every live thread's
619 /// rollout open); exact.
620 OpenFile,
621 /// Matched by working directory and start time; may be wrong with concurrent sessions.
622 CwdHeuristic,
623 /// No transcript found; process only.
624 None,
625 /// No process; transcript only.
626 TranscriptOnly,
627}
628
629#[derive(Debug, Clone, Default, Serialize, Deserialize)]
630pub struct HostStats {
631 pub hostname: Option<String>,
632 pub cpu_percent: f32,
633 pub cpu_count: usize,
634 pub mem_used_bytes: u64,
635 pub mem_total_bytes: u64,
636}
637
638#[derive(Debug, Clone, Default, Serialize, Deserialize)]
639pub struct Totals {
640 pub agents: usize,
641 pub running: usize,
642 pub idle: usize,
643 pub stopped: usize,
644 pub tokens: u64,
645 pub cost_usd: f64,
646 pub unpriced_tokens: u64,
647 pub processes: usize,
648 pub mcp_processes: usize,
649 pub orphaned_mcp: usize,
650 pub cpu_percent: f32,
651 pub rss_bytes: u64,
652}
653
654/// Version of the `--json` document. Bumped when a field changes meaning or
655/// disappears; new fields alone do not bump it.
656pub const SNAPSHOT_SCHEMA_VERSION: u32 = 1;
657
658/// What the collector remembers about an orphaned MCP process: when it first
659/// saw it, and, if it watched the process lose its parent, which agent that
660/// was. Memory lasts for the run; a process that was already an orphan when
661/// agent-top started has no parent on record.
662#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
663pub struct OrphanOrigin {
664 pub pid: u32,
665 pub first_seen: SystemTime,
666 /// When the process was first seen without its parent, if the parent was
667 /// seen before that.
668 pub orphaned_at: Option<SystemTime>,
669 pub parent: Option<OrphanParent>,
670}
671
672/// The agent an orphan used to belong to.
673#[derive(Debug, Clone, Serialize, Deserialize, PartialEq, Eq)]
674pub struct OrphanParent {
675 pub pid: u32,
676 pub agent_id: String,
677 pub name: String,
678}
679
680/// Which rule a piece of advice came from. See `advice`.
681#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Serialize, Deserialize)]
682#[serde(rename_all = "kebab-case")]
683pub enum AdviceRule {
684 /// A tool or MCP server whose results are large per call and have been
685 /// re-read on every response since: the biggest line in the bill hiding
686 /// behind a few calls.
687 ExpensiveSource,
688 /// An MCP server attached to a live agent that has answered no calls in
689 /// a long while.
690 IdleMcpServer,
691 /// An MCP server whose memory has climbed steadily with no calls to
692 /// explain it.
693 GrowingMcpServer,
694}
695
696impl AdviceRule {
697 pub fn label(self) -> &'static str {
698 match self {
699 AdviceRule::ExpensiveSource => "expensive source",
700 AdviceRule::IdleMcpServer => "idle mcp server",
701 AdviceRule::GrowingMcpServer => "growing mcp server",
702 }
703 }
704}
705
706/// One sentence of advice about one agent, with the numbers that back it and
707/// the thing you could do. Nothing is done for you: agent-top only points.
708#[derive(Debug, Clone, Serialize, Deserialize, PartialEq)]
709pub struct Advice {
710 pub agent_id: String,
711 pub agent_name: String,
712 pub rule: AdviceRule,
713 /// The tool, or the MCP server, the advice is about.
714 pub subject: String,
715 /// The server's process, when the advice is about one.
716 pub pid: Option<u32>,
717 /// What was seen, with its numbers.
718 pub headline: String,
719 /// What you might do about it.
720 pub action: String,
721 /// The evidence as numbers, for scripts: zero where a rule has none.
722 #[serde(default)]
723 pub cost_usd: f64,
724 #[serde(default)]
725 pub tokens: u64,
726 #[serde(default)]
727 pub calls: u64,
728 #[serde(default)]
729 pub rss_bytes: u64,
730}
731
732/// Everything the UI needs for one frame.
733#[derive(Debug, Clone, Serialize, Deserialize)]
734pub struct Snapshot {
735 pub schema_version: u32,
736 pub taken_at: SystemTime,
737 pub host: HostStats,
738 pub agents: Vec<Agent>,
739 /// MCP-looking processes with no live agent ancestor: leak candidates.
740 pub orphans: Vec<ProcNode>,
741 /// One entry per orphan, saying where it came from when that is known.
742 #[serde(default)]
743 pub orphan_origins: Vec<OrphanOrigin>,
744 /// What looks like a bad deal on this machine right now, and what could
745 /// be done about it. See `advice`.
746 #[serde(default)]
747 pub advice: Vec<Advice>,
748 pub totals: Totals,
749}
750
751impl Snapshot {
752 pub fn compute_totals(&mut self) {
753 let mut t = Totals::default();
754 for a in &self.agents {
755 t.agents += 1;
756 match a.state {
757 AgentState::Running => t.running += 1,
758 AgentState::Idle => t.idle += 1,
759 AgentState::Stopped => t.stopped += 1,
760 }
761 t.tokens += a.usage.total();
762 t.cost_usd += a.cost_usd;
763 t.unpriced_tokens += a.unpriced_tokens;
764 t.processes += a.process_count;
765 t.mcp_processes += a.mcp_count;
766 t.cpu_percent += a.cpu_percent;
767 t.rss_bytes += a.rss_bytes;
768 }
769 t.orphaned_mcp = self.orphans.len();
770 self.totals = t;
771 }
772}
773
774#[cfg(test)]
775mod usage_tests {
776 use super::TokenUsage;
777
778 #[test]
779 fn cache_hit_rate_is_reads_over_the_prompt() {
780 let u = TokenUsage { input: 200, cache_read: 800, output: 50, ..Default::default() };
781 // Prompt is 1000 (output excluded); 800 of it from cache.
782 assert_eq!(u.prompt(), 1000);
783 assert!((u.cache_hit_rate().unwrap() - 0.8).abs() < 1e-9);
784 // A cache write counts as prompt input, not as a hit.
785 let u = TokenUsage { input: 100, cache_write_5m: 900, ..Default::default() };
786 assert_eq!(u.cache_hit_rate(), Some(0.0));
787 // Nothing to judge.
788 assert_eq!(TokenUsage::default().cache_hit_rate(), None);
789 }
790}