Skip to main content

turnframe_telemetry/
dashboard.rs

1//! The reliability dashboard of spec §26.3, as data.
2//!
3//! Spec §26.3 is a warning as much as a layout: *"a single 'agent accuracy'
4//! percentage hides the most important distinctions."* A response that gives
5//! the user the wrong wording and a response that rebooks the wrong flight are
6//! not the same failure, and averaging them produces a number nobody can act
7//! on. The seven panels below keep them apart.
8//!
9//! This module describes the dashboard; it does not draw one. [`Dashboard`] is
10//! plain serializable data, so the same description can generate a Grafana or
11//! Datadog board, a docs page ([`Dashboard::to_markdown`]) or an alert
12//! inventory, and stay in step with the metrics because it names them by the
13//! same constants the observers emit.
14
15use serde::{Deserialize, Serialize};
16use turnframe_core::observe::Signal;
17
18/// One panel of the reliability dashboard (spec §26.3).
19#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
20#[serde(rename_all = "snake_case")]
21#[non_exhaustive]
22pub enum PanelId {
23    /// Did the system do something to the world it should not have done, or
24    /// does it not know what it did?
25    SideEffectIntegrityFailures,
26    /// Did the assistant tell the user something no committed event backs?
27    ClaimIntegrityFailures,
28    /// Did the system understand the request?
29    SemanticUnderstandingFailures,
30    /// How often does the system have to ask before it can act?
31    ClarificationRate,
32    /// How often does the user walk away from what it asked?
33    AbandonmentRate,
34    /// How often does the model layer itself fail?
35    ProviderFailures,
36    /// How good does the result feel to the person who asked?
37    UserExperienceScores,
38}
39
40impl PanelId {
41    /// Every panel, in the order of spec §26.3 — severity first.
42    pub const ALL: [Self; 7] = [
43        Self::SideEffectIntegrityFailures,
44        Self::ClaimIntegrityFailures,
45        Self::SemanticUnderstandingFailures,
46        Self::ClarificationRate,
47        Self::AbandonmentRate,
48        Self::ProviderFailures,
49        Self::UserExperienceScores,
50    ];
51
52    /// Stable machine-readable key of the panel.
53    #[must_use]
54    pub const fn as_str(self) -> &'static str {
55        match self {
56            Self::SideEffectIntegrityFailures => "side_effect_integrity_failures",
57            Self::ClaimIntegrityFailures => "claim_integrity_failures",
58            Self::SemanticUnderstandingFailures => "semantic_understanding_failures",
59            Self::ClarificationRate => "clarification_rate",
60            Self::AbandonmentRate => "abandonment_rate",
61            Self::ProviderFailures => "provider_failures",
62            Self::UserExperienceScores => "user_experience_scores",
63        }
64    }
65}
66
67impl std::fmt::Display for PanelId {
68    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
69        f.write_str(self.as_str())
70    }
71}
72
73/// Where a panel's numbers come from.
74#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
75#[serde(rename_all = "snake_case")]
76#[non_exhaustive]
77pub enum PanelSource {
78    /// Every series is a metric this library emits.
79    LibraryMetrics,
80    /// The library supplies context but the score itself comes from the
81    /// application: a rating, a survey, an offline evaluation (spec §27.6).
82    ApplicationSupplied,
83}
84
85/// One panel: what it answers, which metrics feed it, and against what it is
86/// normalized.
87#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
88#[serde(deny_unknown_fields)]
89#[non_exhaustive]
90pub struct Panel {
91    /// Which panel this is.
92    pub id: PanelId,
93    /// Human title for the board.
94    pub title: String,
95    /// The operational question the panel answers.
96    pub question: String,
97    /// Metric names plotted on the panel, in the order they should be read.
98    pub series: Vec<String>,
99    /// Metric the series are divided by to become a rate, when the panel is a
100    /// rate rather than a count.
101    #[serde(default, skip_serializing_if = "Option::is_none")]
102    pub denominator: Option<String>,
103    /// Where the numbers come from.
104    pub source: PanelSource,
105    /// Whether a non-zero value on this panel is a defect that should page
106    /// somebody rather than a statistic to watch.
107    pub alerts: bool,
108}
109
110/// A data-only description of the reliability dashboard.
111#[derive(Debug, Clone, PartialEq, Eq, Serialize, Deserialize)]
112#[serde(deny_unknown_fields)]
113#[non_exhaustive]
114pub struct Dashboard {
115    /// The panels, in reading order.
116    pub panels: Vec<Panel>,
117}
118
119impl Dashboard {
120    /// The dashboard of spec §26.3.
121    ///
122    /// ```rust
123    /// use turnframe_telemetry::{Dashboard, PanelId};
124    ///
125    /// let dashboard = Dashboard::reliability();
126    /// let claims = dashboard.panel(PanelId::ClaimIntegrityFailures).expect("panel");
127    /// assert!(claims.series.contains(&String::from("turnframe.claim.receipt_emitted")));
128    /// assert!(claims.alerts);
129    /// ```
130    #[must_use]
131    pub fn reliability() -> Self {
132        Self {
133            panels: vec![
134                Panel {
135                    id: PanelId::SideEffectIntegrityFailures,
136                    title: String::from("Side-effect integrity failures"),
137                    question: String::from(
138                        "Did a command act on a state the user never saw, or did an effect leave \
139                         the system in a state nobody can describe?",
140                    ),
141                    series: names(&[
142                        Signal::CommandRevisionConflict,
143                        Signal::InteractionStale,
144                        Signal::ExternalOutcomeUnknown,
145                        Signal::ExternalReconciled,
146                        Signal::WorkflowInvariantViolation,
147                    ]),
148                    denominator: Some(name(Signal::CommandExecuted)),
149                    source: PanelSource::LibraryMetrics,
150                    alerts: true,
151                },
152                Panel {
153                    id: PanelId::ClaimIntegrityFailures,
154                    title: String::from("Claim integrity failures"),
155                    // The question used to be «did the assistant assert an
156                    // outcome no event backs», answered by a counter of blocks
157                    // a word matcher had refused. That matcher is gone — it
158                    // could not see a negation, so it withheld the true
159                    // sentence on exactly the turns that mattered — and the
160                    // panel now answers the question that has an honest
161                    // source: every receipt on screen came from a committed
162                    // event, because that is the only way one is rendered.
163                    question: String::from(
164                        "Are the operational receipts the user sees derived from committed \
165                         events?",
166                    ),
167                    series: names(&[Signal::ClaimReceiptEmitted]),
168                    denominator: Some(name(Signal::ClaimReceiptEmitted)),
169                    source: PanelSource::LibraryMetrics,
170                    alerts: true,
171                },
172                Panel {
173                    id: PanelId::SemanticUnderstandingFailures,
174                    title: String::from("Semantic understanding failures"),
175                    question: String::from(
176                        "Did an understanding task need a repair or disagree with itself, or \
177                         name a target that does not resolve?",
178                    ),
179                    series: names(&[
180                        Signal::TaskRepaired,
181                        Signal::TaskVoteDisagreement,
182                        Signal::TargetAmbiguous,
183                        Signal::TargetMissing,
184                    ]),
185                    denominator: Some(name(Signal::TurnReceived)),
186                    source: PanelSource::LibraryMetrics,
187                    alerts: false,
188                },
189                Panel {
190                    id: PanelId::ClarificationRate,
191                    title: String::from("Clarification rate"),
192                    question: String::from(
193                        "How often does a turn have to stop and ask instead of acting?",
194                    ),
195                    series: names(&[
196                        Signal::InteractionCreated,
197                        Signal::CommandConfirmationRequired,
198                        Signal::QuestionUnanswered,
199                    ]),
200                    denominator: Some(name(Signal::TurnReceived)),
201                    source: PanelSource::LibraryMetrics,
202                    alerts: false,
203                },
204                Panel {
205                    id: PanelId::AbandonmentRate,
206                    title: String::from("Abandonment rate"),
207                    question: String::from(
208                        "How often is a card the system opened never answered, answered too late, \
209                         or answered in a way that could not be accepted?",
210                    ),
211                    series: names(&[
212                        Signal::InteractionResolved,
213                        Signal::InteractionStale,
214                        Signal::InteractionFailed,
215                    ]),
216                    denominator: Some(name(Signal::InteractionCreated)),
217                    source: PanelSource::LibraryMetrics,
218                    alerts: false,
219                },
220                Panel {
221                    id: PanelId::ProviderFailures,
222                    title: String::from("Provider failures"),
223                    question: String::from(
224                        "How often does the model layer fail, fall back, or lack a capability the \
225                         turn required?",
226                    ),
227                    series: names(&[
228                        Signal::ProviderFallback,
229                        Signal::ProviderCapabilityMismatch,
230                        Signal::ProviderLatency,
231                        Signal::TurnFailed,
232                    ]),
233                    denominator: Some(name(Signal::TurnReceived)),
234                    source: PanelSource::LibraryMetrics,
235                    alerts: true,
236                },
237                Panel {
238                    id: PanelId::UserExperienceScores,
239                    title: String::from("User experience scores"),
240                    question: String::from(
241                        "Did the answer feel right and arrive quickly? The score itself comes \
242                         from the application; the library supplies the latency and the answer \
243                         coverage beside it.",
244                    ),
245                    series: names(&[
246                        Signal::TurnDuration,
247                        Signal::NarrationLatency,
248                        Signal::QuestionAnswered,
249                        Signal::QuestionUnanswered,
250                    ]),
251                    denominator: None,
252                    source: PanelSource::ApplicationSupplied,
253                    alerts: false,
254                },
255            ],
256        }
257    }
258
259    /// The panel with this identifier, if the dashboard has one.
260    #[must_use]
261    pub fn panel(&self, id: PanelId) -> Option<&Panel> {
262        self.panels.iter().find(|panel| panel.id == id)
263    }
264
265    /// Every metric name the dashboard plots, deduplicated, in reading order.
266    #[must_use]
267    pub fn series(&self) -> Vec<String> {
268        let mut out: Vec<String> = Vec::new();
269        for panel in &self.panels {
270            for series in panel.series.iter().chain(panel.denominator.iter()) {
271                if !out.contains(series) {
272                    out.push(series.clone());
273                }
274            }
275        }
276        out
277    }
278
279    /// Renders the dashboard as the markdown published in the documentation.
280    #[must_use]
281    pub fn to_markdown(&self) -> String {
282        let mut out = String::from("# Reliability dashboard (spec §26.3)\n\n");
283        out.push_str(
284            "A single \"agent accuracy\" percentage hides the most important distinctions, so \
285             these seven panels stay separate.\n",
286        );
287        for panel in &self.panels {
288            out.push_str(&format!("\n## {}\n\n{}\n\n", panel.title, panel.question));
289            out.push_str(&format!(
290                "- Source: {}\n",
291                match panel.source {
292                    PanelSource::LibraryMetrics => "library metrics",
293                    PanelSource::ApplicationSupplied => "application-supplied score",
294                }
295            ));
296            out.push_str(&format!(
297                "- Alerts: {}\n",
298                if panel.alerts { "yes" } else { "no" }
299            ));
300            if let Some(denominator) = &panel.denominator {
301                out.push_str(&format!("- Normalized by: `{denominator}`\n"));
302            }
303            out.push_str("- Series:\n");
304            for series in &panel.series {
305                out.push_str(&format!("  - `{series}`\n"));
306            }
307        }
308        out
309    }
310}
311
312impl Default for Dashboard {
313    fn default() -> Self {
314        Self::reliability()
315    }
316}
317
318fn name(signal: Signal) -> String {
319    signal.name().to_owned()
320}
321
322fn names(signals: &[Signal]) -> Vec<String> {
323    signals.iter().copied().map(name).collect()
324}
325
326#[cfg(test)]
327mod tests {
328    use super::*;
329
330    #[test]
331    fn the_dashboard_has_every_panel_of_26_3_once() {
332        let dashboard = Dashboard::reliability();
333        assert_eq!(dashboard.panels.len(), PanelId::ALL.len());
334        for id in PanelId::ALL {
335            let panel = dashboard
336                .panel(id)
337                .unwrap_or_else(|| panic!("{id} missing"));
338            assert_eq!(panel.id, id);
339            assert!(!panel.title.is_empty());
340            assert!(!panel.question.is_empty());
341            assert!(!panel.series.is_empty(), "{id} plots nothing");
342        }
343    }
344
345    #[test]
346    fn every_series_is_a_metric_this_crate_emits() {
347        let known: Vec<&str> = Signal::ALL.iter().map(Signal::name).collect();
348        for series in Dashboard::reliability().series() {
349            assert!(known.contains(&series.as_str()), "unknown metric {series}");
350        }
351    }
352
353    #[test]
354    fn every_safety_signal_appears_on_an_alerting_panel() {
355        let dashboard = Dashboard::reliability();
356        let alerting: Vec<String> = dashboard
357            .panels
358            .iter()
359            .filter(|panel| panel.alerts)
360            .flat_map(|panel| panel.series.clone())
361            .collect();
362        for signal in Signal::ALL.into_iter().filter(Signal::is_safety_signal) {
363            assert!(
364                alerting.contains(&signal.name().to_owned()),
365                "{signal:?} is a safety signal but no alerting panel plots it"
366            );
367        }
368    }
369
370    #[test]
371    fn integrity_and_semantics_are_never_merged() {
372        let dashboard = Dashboard::reliability();
373        let side_effects = dashboard
374            .panel(PanelId::SideEffectIntegrityFailures)
375            .expect("panel");
376        let semantics = dashboard
377            .panel(PanelId::SemanticUnderstandingFailures)
378            .expect("panel");
379        assert!(
380            side_effects
381                .series
382                .iter()
383                .all(|series| !semantics.series.contains(series)),
384            "a series feeds both the integrity and the semantics panel"
385        );
386        assert!(side_effects.alerts);
387        assert!(!semantics.alerts);
388    }
389
390    #[test]
391    fn the_dashboard_round_trips_through_json() {
392        let dashboard = Dashboard::reliability();
393        let json = serde_json::to_string(&dashboard).expect("serializable");
394        let back: Dashboard = serde_json::from_str(&json).expect("deserializable");
395        assert_eq!(back, dashboard);
396        assert!(json.contains("side_effect_integrity_failures"));
397        assert!(json.contains("application_supplied"));
398    }
399
400    #[test]
401    fn markdown_names_every_panel_and_series() {
402        let dashboard = Dashboard::reliability();
403        let markdown = dashboard.to_markdown();
404        for panel in &dashboard.panels {
405            assert!(markdown.contains(&panel.title), "{}", panel.title);
406        }
407        for series in dashboard.series() {
408            assert!(markdown.contains(&series), "{series}");
409        }
410        assert!(markdown.contains("application-supplied score"));
411    }
412
413    #[test]
414    fn the_default_dashboard_is_the_reliability_one() {
415        assert_eq!(Dashboard::default(), Dashboard::reliability());
416    }
417}