vtcode_commons/runtime_diagnostics.rs
1//! Stable Tokio runtime diagnostics (no `tokio_unstable` required).
2//!
3//! Implements step 1 of the fast-Tokio principles: measure first. We capture
4//! the stable subset of [`tokio::runtime::RuntimeMetrics`] — worker count,
5//! alive tasks, global (injection) queue depth, and total worker busy time.
6//! The article's headline signal is the global queue staying deep ("in a
7//! healthy application it should generally stay close to empty"); that metric
8//! is stable, so it ships here.
9//!
10//! Per-worker local-queue depth, steal/overflow counts, blocking-pool depth,
11//! and the poll-time/schedule-latency histograms are all gated behind
12//! `RUSTFLAGS="--cfg tokio_unstable"` in current Tokio. Enabling that cfg
13//! changes the whole dependency graph build, so it stays opt-in follow-up work
14//! rather than a default here.
15
16use std::time::Duration;
17use tokio::runtime::Handle;
18
19/// Env var enabling periodic runtime snapshot logs.
20const RUNTIME_METRICS_ENV: &str = "VTCODE_RUNTIME_METRICS";
21
22/// Snapshot of stable runtime counters at one instant.
23#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)]
24pub struct RuntimeSnapshot {
25 /// Worker threads configured on the runtime.
26 pub workers_len: usize,
27 /// Currently alive tasks.
28 pub alive_tasks_len: usize,
29 /// Tasks pending in the global (injection) queue.
30 pub global_queue_depth_len: usize,
31 /// Sum of per-worker busy time since runtime creation.
32 pub total_busy: Duration,
33}
34
35/// Returns true when runtime diagnostics logging is enabled.
36///
37/// Enabled by `VTCODE_RUNTIME_METRICS=1|true|yes|on|debug` or when
38/// `VTCODE_STARTUP_TRACE=1` (startup trace already implies diagnostics).
39#[must_use]
40pub fn runtime_diagnostics_enabled() -> bool {
41 crate::utils::env_flag_enabled(RUNTIME_METRICS_ENV) || startup_trace_enabled()
42}
43
44fn startup_trace_enabled() -> bool {
45 std::env::var("VTCODE_STARTUP_TRACE").is_ok_and(|value| value == "1")
46}
47
48/// Env var that overrides the Tokio worker-thread count.
49const RUNTIME_WORKERS_ENV: &str = "VTCODE_RUNTIME_WORKERS";
50
51/// Resolve an explicit worker-thread count for the main runtime.
52///
53/// Returns `None` when unset or not a positive integer, so the default
54/// one-worker-per-core behaviour is preserved. Reserving cores for non-Tokio
55/// background work is the isolation lever the fast-Tokio guidance recommends
56/// ("you rarely need every core for Tokio"); operators that co-locate VT Code
57/// with other processes can lower this without rebuilding.
58#[must_use]
59pub fn configured_worker_threads() -> Option<usize> {
60 std::env::var(RUNTIME_WORKERS_ENV)
61 .ok()
62 .and_then(|value| value.trim().parse::<usize>().ok())
63 .filter(|workers| *workers > 0)
64}
65
66/// Capture stable counters from a runtime handle.
67#[must_use]
68pub fn snapshot(handle: &Handle) -> RuntimeSnapshot {
69 let metrics = handle.metrics();
70 RuntimeSnapshot {
71 workers_len: metrics.num_workers(),
72 alive_tasks_len: metrics.num_alive_tasks(),
73 global_queue_depth_len: metrics.global_queue_depth(),
74 total_busy: total_busy(&metrics),
75 }
76}
77
78#[cfg(target_has_atomic = "64")]
79fn total_busy(metrics: &tokio::runtime::RuntimeMetrics) -> Duration {
80 let mut busy = Duration::ZERO;
81 for worker in 0..metrics.num_workers() {
82 busy = busy.saturating_add(metrics.worker_total_busy_duration(worker));
83 }
84 busy
85}
86
87#[cfg(not(target_has_atomic = "64"))]
88fn total_busy(_metrics: &tokio::runtime::RuntimeMetrics) -> Duration {
89 // Worker busy duration requires 64-bit atomics; report zero otherwise.
90 Duration::ZERO
91}
92
93/// Log one snapshot at `DEBUG` level; no-op unless diagnostics are enabled.
94///
95/// Callers should gate periodic reporting on [`runtime_diagnostics_enabled`]
96/// to keep the hot path free when observability is off.
97pub fn log_snapshot(handle: &Handle, context: &str) {
98 if !runtime_diagnostics_enabled() {
99 return;
100 }
101 let snapshot = snapshot(handle);
102 let worker_busy_ms = u64::try_from(snapshot.total_busy.as_millis()).unwrap_or(u64::MAX);
103 tracing::debug!(
104 target = "vtcode.runtime",
105 context,
106 workers = snapshot.workers_len,
107 alive_tasks = snapshot.alive_tasks_len,
108 global_queue_depth = snapshot.global_queue_depth_len,
109 worker_busy_ms,
110 "tokio runtime snapshot"
111 );
112}
113
114/// Default interval between periodic runtime snapshots.
115const REPORT_INTERVAL: Duration = Duration::from_secs(60);
116
117/// Spawn a low-frequency reporter for long-lived processes.
118///
119/// The reporter ticks every 60s and logs via [`log_snapshot`]. Only spawns when
120/// [`runtime_diagnostics_enabled`] is true, otherwise returns `None`. Dropping
121/// the returned handle detaches the task; it is cancelled when the runtime
122/// drops.
123pub fn spawn_periodic_reporter(handle: &Handle) -> Option<tokio::task::JoinHandle<()>> {
124 if !runtime_diagnostics_enabled() {
125 return None;
126 }
127 Some(spawn_reporter(handle, REPORT_INTERVAL))
128}
129
130/// Spawn the reporter on the supplied runtime handle.
131///
132/// Uses [`Handle::spawn`] rather than [`tokio::spawn`]: bootstrap starts the
133/// reporter before `Runtime::block_on` establishes an ambient runtime context,
134/// and `tokio::spawn` panics when no runtime context is active (the same reason
135/// `agent::probe` uses `handle.spawn_blocking`).
136fn spawn_reporter(handle: &Handle, interval: Duration) -> tokio::task::JoinHandle<()> {
137 let metrics_handle = handle.clone();
138 handle.spawn(async move {
139 let mut ticker = tokio::time::interval(interval);
140 // Skip bursts of missed ticks rather than replaying them.
141 ticker.set_missed_tick_behavior(tokio::time::MissedTickBehavior::Skip);
142 // `interval` yields immediately on its first tick; consume it so the
143 // first periodic snapshot lands one interval after startup instead of
144 // duplicating the caller's boot snapshot.
145 let _ = ticker.tick().await;
146 loop {
147 let _ = ticker.tick().await;
148 log_snapshot(&metrics_handle, "periodic");
149 }
150 })
151}
152
153#[cfg(test)]
154mod tests {
155 use super::*;
156
157 #[test]
158 fn absent_env_flag_is_disabled() {
159 // `VTCODE_RUNTIME_METRICS_TEST_ABSENT_XYZ` is never set, so the flag
160 // parser must report disabled without mutating process env.
161 assert!(!crate::utils::env_flag_enabled("VTCODE_RUNTIME_METRICS_TEST_ABSENT_XYZ"));
162 }
163
164 #[test]
165 fn absent_worker_override_uses_default() {
166 // No override is configured in the isolated test env, so the default
167 // one-worker-per-core behaviour must be preserved.
168 if std::env::var("VTCODE_RUNTIME_WORKERS").is_err() {
169 assert_eq!(configured_worker_threads(), None);
170 }
171 }
172
173 #[tokio::test]
174 async fn snapshot_reflects_current_runtime() {
175 let snapshot = snapshot(&Handle::current());
176 // At least one worker exists on either runtime flavor.
177 assert!(snapshot.workers_len >= 1);
178 }
179
180 #[tokio::test]
181 async fn log_snapshot_is_quiet_when_disabled() {
182 // Must not panic and must stay a no-op when diagnostics are off.
183 log_snapshot(&Handle::current(), "test");
184 }
185
186 #[test]
187 fn reporter_spawns_outside_ambient_runtime_context() {
188 // Bootstrap starts the reporter before `Runtime::block_on`, so spawning
189 // must go through the supplied handle: `tokio::spawn` panics with
190 // "there is no reactor running" when no runtime context is active.
191 let runtime = tokio::runtime::Builder::new_current_thread()
192 .build()
193 .expect("current-thread runtime");
194 let reporter = spawn_reporter(runtime.handle(), Duration::from_secs(3600));
195 reporter.abort();
196 }
197}