Skip to main content

toolkit/runtime/
readiness.rs

1//! Framework-managed readiness state for the `OoP` bootstrap (`cpt-cf-fr-eventual-readiness`).
2//!
3//! An `OoP` gear becomes *live* the moment its HTTP server binds (`/healthz`),
4//! but only becomes *ready* (`/readyz`) once startup is complete, every critical
5//! dependency has been resolved, **and** the gear's registered healthchecks
6//! report it can serve traffic (Spring Boot-style health groups per
7//! `cpt-cf-fr-eventual-readiness`).
8//!
9//! Readiness reuses the framework's standard healthcheck mechanism
10//! ([`crate::healthcheck`]): a gear expresses readiness once, via
11//! [`RestApiCapability::healthcheck`](crate::contracts::RestApiCapability::healthcheck),
12//! and it is honored identically whether the gear is hosted in-process by the
13//! `api-gateway` or run `OoP`. This module layers the three `OoP`-only concerns
14//! the gateway path lacks — startup completion, critical-dependency resolution
15//! gating, and the graceful-drain readiness flip — on top of that shared
16//! healthcheck report.
17//!
18//! The healthcheck report itself is fanned out concurrently, timeout-bounded,
19//! panic-isolated, and cached inside the [`RestHealthcheckRegistry`], so a burst
20//! of probe traffic cannot storm the registered checks. The dependency and
21//! draining state layered on top are cheap live reads.
22
23use std::collections::BTreeSet;
24use std::sync::Arc;
25use std::sync::atomic::{AtomicBool, Ordering};
26use std::time::Duration;
27
28use parking_lot::Mutex;
29
30use crate::healthcheck::{HealthcheckReport, HealthcheckStatus, RestHealthcheckRegistry};
31
32/// Default per-check timeout for readiness healthchecks.
33///
34/// Matches the `api-gateway` `healthcheck_timeout_ms` default so a gear's
35/// healthcheck behaves identically in-process and `OoP`.
36pub const DEFAULT_HEALTHCHECK_TIMEOUT: Duration = Duration::from_millis(500);
37
38/// Lifecycle state reported on `/readyz` (`cpt-cf-adr-eventual-readiness`).
39///
40/// Serialized lowercase; the four variants are a stable wire contract.
41#[derive(Debug, Clone, Copy, PartialEq, Eq, serde::Serialize)]
42#[serde(rename_all = "lowercase")]
43pub enum ReadinessLifecycle {
44    /// Not yet able to serve — startup is not complete, critical deps are
45    /// unresolved, or a healthcheck is `Unhealthy`. Maps to `503`.
46    Starting,
47    /// Fully serving traffic. Maps to `200`.
48    Ready,
49    /// Serving with reduced functionality — a healthcheck reported `Degraded`
50    /// (e.g. an optional backend is down but a fallback is acceptable). Kept in
51    /// rotation: maps to `200`.
52    Degraded,
53    /// Graceful shutdown in progress; upstreams should stop routing. Maps to
54    /// `503`.
55    Draining,
56}
57
58/// The aggregate readiness report rendered as the `/readyz` response body.
59///
60/// Readiness still depends on the aggregated [`HealthcheckReport`] internally,
61/// but the detailed per-component report is intentionally not echoed here; it
62/// belongs on the separate `/health` endpoint (see `oop_serve.rs`).
63#[derive(Debug, Clone, serde::Serialize)]
64pub struct ReadinessReport {
65    /// Lifecycle state — the primary readiness signal (`cpt-cf-adr-eventual-readiness`).
66    pub state: ReadinessLifecycle,
67    /// Whether the gear is ready to receive traffic (`true` → `200`, else `503`).
68    /// Convenience mirror of `state ∈ {ready, degraded}` for probes/clients that
69    /// do not want to know the `state → status` mapping.
70    pub ready: bool,
71    /// Critical dependencies not yet resolved via `DirectoryService` / DNS.
72    /// Non-empty only while `starting`. Omitted from the body when empty.
73    #[serde(skip_serializing_if = "Vec::is_empty")]
74    pub unresolved_deps: Vec<String>,
75}
76
77/// Shared, framework-owned readiness state for an `OoP` gear instance.
78///
79/// Created by the `OoP` bootstrap with the gear's critical dependency names and
80/// a shared [`RestHealthcheckRegistry`] (populated from each gear's
81/// [`RestApiCapability::healthcheck`](crate::contracts::RestApiCapability::healthcheck)).
82/// Cloned as an `Arc` into the probe router and the dependency-resolution task.
83pub struct ReadinessState {
84    /// Critical deps still awaiting resolution. Empty ⇒ deps satisfied.
85    unresolved_deps: Mutex<BTreeSet<String>>,
86    /// Graceful-shutdown flag; when set, `/readyz` reports `503` (readiness flip).
87    draining: AtomicBool,
88    /// Whether the gear has finished startup and is actually serving traffic.
89    /// Remains `false` until the bootstrap publishes the composed routes.
90    startup_complete: AtomicBool,
91    /// Shared gear healthcheck registry; supplies the "custom checks" dimension
92    /// of readiness (fan-out, timeout, panic isolation, and caching live here).
93    healthchecks: Arc<RestHealthcheckRegistry>,
94    /// Per-check timeout passed to [`RestHealthcheckRegistry::report`].
95    check_timeout: Duration,
96}
97
98impl std::fmt::Debug for ReadinessState {
99    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
100        f.debug_struct("ReadinessState")
101            .field("unresolved_deps", &self.unresolved_deps.lock())
102            .field("draining", &self.draining.load(Ordering::Relaxed))
103            .field(
104                "startup_complete",
105                &self.startup_complete.load(Ordering::Relaxed),
106            )
107            .field("check_timeout", &self.check_timeout)
108            .finish_non_exhaustive()
109    }
110}
111
112impl ReadinessState {
113    /// Create a new readiness state seeded with the gear's critical dependency
114    /// names and the shared healthcheck registry. All listed deps start
115    /// unresolved; the gear is not ready until startup is complete, each dep is
116    /// marked resolved via [`mark_dep_resolved`](Self::mark_dep_resolved), and
117    /// the healthchecks pass. Uses [`DEFAULT_HEALTHCHECK_TIMEOUT`] as the
118    /// per-check timeout.
119    #[must_use]
120    pub fn new<I, S>(critical_deps: I, healthchecks: Arc<RestHealthcheckRegistry>) -> Arc<Self>
121    where
122        I: IntoIterator<Item = S>,
123        S: Into<String>,
124    {
125        Self::with_check_timeout(critical_deps, healthchecks, DEFAULT_HEALTHCHECK_TIMEOUT)
126    }
127
128    /// Like [`new`](Self::new) but with an explicit per-check timeout.
129    #[must_use]
130    pub fn with_check_timeout<I, S>(
131        critical_deps: I,
132        healthchecks: Arc<RestHealthcheckRegistry>,
133        check_timeout: Duration,
134    ) -> Arc<Self>
135    where
136        I: IntoIterator<Item = S>,
137        S: Into<String>,
138    {
139        Arc::new(Self {
140            unresolved_deps: Mutex::new(critical_deps.into_iter().map(Into::into).collect()),
141            draining: AtomicBool::new(false),
142            startup_complete: AtomicBool::new(false),
143            healthchecks,
144            check_timeout,
145        })
146    }
147
148    /// Mark startup as complete. Idempotent; subsequent calls are ignored.
149    /// `/readyz` will not report `Ready` or `Degraded` until this is called.
150    pub fn mark_startup_complete(&self) {
151        self.startup_complete.store(true, Ordering::SeqCst);
152    }
153
154    /// Whether startup is complete.
155    #[must_use]
156    pub fn is_startup_complete(&self) -> bool {
157        self.startup_complete.load(Ordering::SeqCst)
158    }
159
160    /// Mark a critical dependency as resolved. Idempotent; unknown names are
161    /// ignored.
162    pub fn mark_dep_resolved(&self, name: &str) {
163        let removed = self.unresolved_deps.lock().remove(name);
164        if removed {
165            tracing::info!(dep = %name, "critical dependency resolved");
166        }
167    }
168
169    /// Whether all critical dependencies have been resolved.
170    #[must_use]
171    pub fn all_deps_resolved(&self) -> bool {
172        self.unresolved_deps.lock().is_empty()
173    }
174
175    /// Set (or clear) the draining flag. Setting it flips `/readyz` to `503`
176    /// immediately so upstreams pull the instance out of rotation while
177    /// in-flight requests drain.
178    pub fn set_draining(&self, draining: bool) {
179        self.draining.store(draining, Ordering::SeqCst);
180    }
181
182    /// Whether the gear is currently draining.
183    #[must_use]
184    pub fn is_draining(&self) -> bool {
185        self.draining.load(Ordering::SeqCst)
186    }
187
188    /// Run the registered healthchecks and return the aggregated report.
189    ///
190    /// Used by `/health` to expose full per-component detail and by
191    /// [`Self::evaluate`] to decide readiness state. The registry caches the
192    /// report, so repeated calls within the cache window do not re-run checks.
193    pub async fn health_report(&self) -> HealthcheckReport {
194        self.healthchecks.report(self.check_timeout).await
195    }
196
197    /// Evaluate the aggregate readiness.
198    ///
199    /// The gear is ready when it is not draining, startup is complete, all
200    /// critical deps are resolved, and the aggregated healthcheck report is not
201    /// `Unhealthy`. `Degraded` healthchecks keep the gear ready (`state =
202    /// degraded`, `ready = true`) but the detailed per-component messages belong
203    /// on `/health`, not `/readyz`. The healthcheck fan-out is cached inside the
204    /// registry, so repeated probe traffic does not re-run checks; dependency and
205    /// draining state are read live so transitions take effect immediately.
206    pub async fn evaluate(&self) -> ReadinessReport {
207        let health = self.health_report().await;
208        let draining = self.is_draining();
209        let startup_complete = self.is_startup_complete();
210        let unresolved_deps: Vec<String> = self.unresolved_deps.lock().iter().cloned().collect();
211
212        // Draining wins; then any not-ready condition (startup not complete,
213        // unresolved deps, or an Unhealthy check) is `Starting`; then `Degraded`;
214        // else `Ready`.
215        let state = if draining {
216            ReadinessLifecycle::Draining
217        } else if !startup_complete
218            || !unresolved_deps.is_empty()
219            || health.status == HealthcheckStatus::Unhealthy
220        {
221            ReadinessLifecycle::Starting
222        } else if health.status == HealthcheckStatus::Degraded {
223            ReadinessLifecycle::Degraded
224        } else {
225            ReadinessLifecycle::Ready
226        };
227
228        let ready = matches!(
229            state,
230            ReadinessLifecycle::Ready | ReadinessLifecycle::Degraded
231        );
232
233        ReadinessReport {
234            state,
235            ready,
236            unresolved_deps,
237        }
238    }
239}
240
241#[cfg(test)]
242#[cfg_attr(coverage_nightly, coverage(off))]
243#[path = "readiness_tests.rs"]
244mod tests;