toolkit/runtime/readiness.rs
1//! Framework-managed readiness state for the `OoP` bootstrap (`cpt-cf-fr-eventual-readiness`).
2//!
3//! An `OoP` gear becomes *live* the moment its HTTP server binds (`/healthz`),
4//! but only becomes *ready* (`/readyz`) once startup is complete, every critical
5//! dependency has been resolved, **and** the gear's registered healthchecks
6//! report it can serve traffic (Spring Boot-style health groups per
7//! `cpt-cf-fr-eventual-readiness`).
8//!
9//! Readiness reuses the framework's standard healthcheck mechanism
10//! ([`crate::healthcheck`]): a gear expresses readiness once, via
11//! [`RestApiCapability::healthcheck`](crate::contracts::RestApiCapability::healthcheck),
12//! and it is honored identically whether the gear is hosted in-process by the
13//! `api-gateway` or run `OoP`. This module layers the three `OoP`-only concerns
14//! the gateway path lacks — startup completion, critical-dependency resolution
15//! gating, and the graceful-drain readiness flip — on top of that shared
16//! healthcheck report.
17//!
18//! The healthcheck report itself is fanned out concurrently, timeout-bounded,
19//! panic-isolated, and cached inside the [`RestHealthcheckRegistry`], so a burst
20//! of probe traffic cannot storm the registered checks. The dependency and
21//! draining state layered on top are cheap live reads.
22
23use std::collections::BTreeSet;
24use std::sync::Arc;
25use std::sync::atomic::{AtomicBool, Ordering};
26use std::time::Duration;
27
28use parking_lot::Mutex;
29
30use crate::healthcheck::{HealthcheckReport, HealthcheckStatus, RestHealthcheckRegistry};
31
32/// Default per-check timeout for readiness healthchecks.
33///
34/// Matches the `api-gateway` `healthcheck_timeout_ms` default so a gear's
35/// healthcheck behaves identically in-process and `OoP`.
36pub const DEFAULT_HEALTHCHECK_TIMEOUT: Duration = Duration::from_millis(500);
37
38/// Lifecycle state reported on `/readyz` (`cpt-cf-adr-eventual-readiness`).
39///
40/// Serialized lowercase; the four variants are a stable wire contract.
41#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize)]
42#[serde(rename_all = "lowercase")]
43pub enum ReadinessLifecycle {
44 /// Not yet able to serve — startup is not complete, critical deps are
45 /// unresolved, or a healthcheck is `Unhealthy`. Maps to `503`.
46 Starting,
47 /// Fully serving traffic. Maps to `200`.
48 Ready,
49 /// Serving with reduced functionality — a healthcheck reported `Degraded`
50 /// (e.g. an optional backend is down but a fallback is acceptable). Kept in
51 /// rotation: maps to `200`.
52 Degraded,
53 /// Graceful shutdown in progress; upstreams should stop routing. Maps to
54 /// `503`.
55 Draining,
56}
57
58/// The aggregate readiness report rendered as the `/readyz` response body.
59///
60/// Readiness still depends on the aggregated [`HealthcheckReport`] internally,
61/// but the detailed per-component report is intentionally not echoed here; it
62/// belongs on the separate `/health` endpoint (see `oop_serve.rs`).
63#[derive(Debug, Clone, serde::Serialize)]
64pub struct ReadinessReport {
65 /// Lifecycle state — the primary readiness signal (`cpt-cf-adr-eventual-readiness`).
66 pub state: ReadinessLifecycle,
67 /// Whether the gear is ready to receive traffic (`true` → `200`, else `503`).
68 /// Convenience mirror of `state ∈ {ready, degraded}` for probes/clients that
69 /// do not want to know the `state → status` mapping.
70 pub ready: bool,
71 /// Critical dependencies not yet resolved via `DirectoryService` / DNS.
72 /// Non-empty only while `starting`. Omitted from the body when empty.
73 #[serde(skip_serializing_if = "Vec::is_empty")]
74 pub unresolved_deps: Vec<String>,
75}
76
77/// Shared, framework-owned readiness state for an `OoP` gear instance.
78///
79/// Created by the `OoP` bootstrap with the gear's critical dependency names and
80/// a shared [`RestHealthcheckRegistry`] (populated from each gear's
81/// [`RestApiCapability::healthcheck`](crate::contracts::RestApiCapability::healthcheck)).
82/// Cloned as an `Arc` into the probe router and the dependency-resolution task.
83pub struct ReadinessState {
84 /// Critical deps still awaiting resolution. Empty ⇒ deps satisfied.
85 unresolved_deps: Mutex<BTreeSet<String>>,
86 /// Graceful-shutdown flag; when set, `/readyz` reports `503` (readiness flip).
87 draining: AtomicBool,
88 /// Whether the gear has finished startup and is actually serving traffic.
89 /// Remains `false` until the bootstrap publishes the composed routes.
90 startup_complete: AtomicBool,
91 /// Shared gear healthcheck registry; supplies the "custom checks" dimension
92 /// of readiness (fan-out, timeout, panic isolation, and caching live here).
93 healthchecks: Arc<RestHealthcheckRegistry>,
94 /// Per-check timeout passed to [`RestHealthcheckRegistry::report`].
95 check_timeout: Duration,
96}
97
98impl std::fmt::Debug for ReadinessState {
99 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
100 f.debug_struct("ReadinessState")
101 .field("unresolved_deps", &self.unresolved_deps.lock())
102 .field("draining", &self.draining.load(Ordering::Relaxed))
103 .field(
104 "startup_complete",
105 &self.startup_complete.load(Ordering::Relaxed),
106 )
107 .field("check_timeout", &self.check_timeout)
108 .finish_non_exhaustive()
109 }
110}
111
112impl ReadinessState {
113 /// Create a new readiness state seeded with the gear's critical dependency
114 /// names and the shared healthcheck registry. All listed deps start
115 /// unresolved; the gear is not ready until startup is complete, each dep is
116 /// marked resolved via [`mark_dep_resolved`](Self::mark_dep_resolved), and
117 /// the healthchecks pass. Uses [`DEFAULT_HEALTHCHECK_TIMEOUT`] as the
118 /// per-check timeout.
119 #[must_use]
120 pub fn new<I, S>(critical_deps: I, healthchecks: Arc<RestHealthcheckRegistry>) -> Arc<Self>
121 where
122 I: IntoIterator<Item = S>,
123 S: Into<String>,
124 {
125 Self::with_check_timeout(critical_deps, healthchecks, DEFAULT_HEALTHCHECK_TIMEOUT)
126 }
127
128 /// Like [`new`](Self::new) but with an explicit per-check timeout.
129 #[must_use]
130 pub fn with_check_timeout<I, S>(
131 critical_deps: I,
132 healthchecks: Arc<RestHealthcheckRegistry>,
133 check_timeout: Duration,
134 ) -> Arc<Self>
135 where
136 I: IntoIterator<Item = S>,
137 S: Into<String>,
138 {
139 Arc::new(Self {
140 unresolved_deps: Mutex::new(critical_deps.into_iter().map(Into::into).collect()),
141 draining: AtomicBool::new(false),
142 startup_complete: AtomicBool::new(false),
143 healthchecks,
144 check_timeout,
145 })
146 }
147
148 /// Mark startup as complete. Idempotent; subsequent calls are ignored.
149 /// `/readyz` will not report `Ready` or `Degraded` until this is called.
150 pub fn mark_startup_complete(&self) {
151 self.startup_complete.store(true, Ordering::SeqCst);
152 }
153
154 /// Whether startup is complete.
155 #[must_use]
156 pub fn is_startup_complete(&self) -> bool {
157 self.startup_complete.load(Ordering::SeqCst)
158 }
159
160 /// Mark a critical dependency as resolved. Idempotent; unknown names are
161 /// ignored.
162 pub fn mark_dep_resolved(&self, name: &str) {
163 let removed = self.unresolved_deps.lock().remove(name);
164 if removed {
165 tracing::info!(dep = %name, "critical dependency resolved");
166 }
167 }
168
169 /// Whether all critical dependencies have been resolved.
170 #[must_use]
171 pub fn all_deps_resolved(&self) -> bool {
172 self.unresolved_deps.lock().is_empty()
173 }
174
175 /// Set (or clear) the draining flag. Setting it flips `/readyz` to `503`
176 /// immediately so upstreams pull the instance out of rotation while
177 /// in-flight requests drain.
178 pub fn set_draining(&self, draining: bool) {
179 self.draining.store(draining, Ordering::SeqCst);
180 }
181
182 /// Whether the gear is currently draining.
183 #[must_use]
184 pub fn is_draining(&self) -> bool {
185 self.draining.load(Ordering::SeqCst)
186 }
187
188 /// Run the registered healthchecks and return the aggregated report.
189 ///
190 /// Used by `/health` to expose full per-component detail and by
191 /// [`Self::evaluate`] to decide readiness state. The registry caches the
192 /// report, so repeated calls within the cache window do not re-run checks.
193 pub async fn health_report(&self) -> HealthcheckReport {
194 self.healthchecks.report(self.check_timeout).await
195 }
196
197 /// Evaluate the aggregate readiness.
198 ///
199 /// The gear is ready when it is not draining, startup is complete, all
200 /// critical deps are resolved, and the aggregated healthcheck report is not
201 /// `Unhealthy`. `Degraded` healthchecks keep the gear ready (`state =
202 /// degraded`, `ready = true`) but the detailed per-component messages belong
203 /// on `/health`, not `/readyz`. The healthcheck fan-out is cached inside the
204 /// registry, so repeated probe traffic does not re-run checks; dependency and
205 /// draining state are read live so transitions take effect immediately.
206 pub async fn evaluate(&self) -> ReadinessReport {
207 let health = self.health_report().await;
208 let draining = self.is_draining();
209 let startup_complete = self.is_startup_complete();
210 let unresolved_deps: Vec<String> = self.unresolved_deps.lock().iter().cloned().collect();
211
212 // Draining wins; then any not-ready condition (startup not complete,
213 // unresolved deps, or an Unhealthy check) is `Starting`; then `Degraded`;
214 // else `Ready`.
215 let state = if draining {
216 ReadinessLifecycle::Draining
217 } else if !startup_complete
218 || !unresolved_deps.is_empty()
219 || health.status == HealthcheckStatus::Unhealthy
220 {
221 ReadinessLifecycle::Starting
222 } else if health.status == HealthcheckStatus::Degraded {
223 ReadinessLifecycle::Degraded
224 } else {
225 ReadinessLifecycle::Ready
226 };
227
228 let ready = matches!(
229 state,
230 ReadinessLifecycle::Ready | ReadinessLifecycle::Degraded
231 );
232
233 ReadinessReport {
234 state,
235 ready,
236 unresolved_deps,
237 }
238 }
239}
240
241#[cfg(test)]
242#[cfg_attr(coverage_nightly, coverage(off))]
243#[path = "readiness_tests.rs"]
244mod tests;