Skip to main content

areev_loop/analyzers/
outcome_review.rs

1//! Outcome review (T0). For applied recommendations past their `review_after`,
2//! the engine re-runs the stored metric query (it owns the `&mut` substrate)
3//! and hands the measured values in as `OutcomeInput`s; this analyzer makes the
4//! deterministic changed/regressed decision and proposes a revert on
5//! regression. Closes the honesty loop — makes approve and auto-apply
6//! accountable to measured history.
7//!
8//! Our built-in metrics are lower-is-better (e.g. `tool_error_rate`), so a
9//! regression is `current > baseline` beyond a small epsilon. An evalset
10//! metric can be higher-is-better and inverts that, which is why the direction
11//! travels on the input and the comparison lives in ONE place
12//! (`recommendation::is_regression`).
13
14use crate::analyzer::{AnalyzeCtx, Analyzer};
15use crate::error::Result;
16use crate::manifest::*;
17use crate::model::{ActionKind, Severity};
18use crate::recommendation::{Proposal, RecDraft, Summary};
19use serde_json::{json, Map};
20
21pub struct OutcomeReview {
22    manifest: AnalyzerManifest,
23}
24
25impl OutcomeReview {
26    pub fn new() -> Self {
27        OutcomeReview {
28            manifest: AnalyzerManifest {
29                id: "loop.outcome_review/1".into(),
30                title: "Outcome review".into(),
31                description:
32                    "Re-measures applied recommendations and proposes revert on regression.".into(),
33                tier: Tier::T0,
34                cadence: CadenceClass::Fast,
35                requires: vec![],
36                target_classes: vec![TargetClass::Memory, TargetClass::Query],
37                auto_apply: AutoApplyClass::Never,
38                trust_class: TrustClass::Builtin,
39                params: vec![],
40                default_on: true,
41            },
42        }
43    }
44}
45
46impl Default for OutcomeReview {
47    fn default() -> Self {
48        Self::new()
49    }
50}
51
52impl Analyzer for OutcomeReview {
53    fn manifest(&self) -> &AnalyzerManifest {
54        &self.manifest
55    }
56
57    fn analyze(&self, ctx: &AnalyzeCtx) -> Result<Vec<RecDraft>> {
58        let mut drafts = Vec::new();
59        for input in ctx.outcome_inputs() {
60            let regressed = crate::recommendation::is_regression(
61                input.baseline,
62                input.current,
63                input.higher_is_better,
64            );
65            if !regressed {
66                continue;
67            }
68            let mut args = Map::new();
69            args.insert("metric".into(), json!(input.metric));
70            args.insert("baseline".into(), json!(round4(input.baseline)));
71            args.insert("current".into(), json!(round4(input.current)));
72
73            let mut data = Map::new();
74            data.insert("revert_of".into(), json!(input.rec_hash));
75            data.insert("metric".into(), json!(input.metric));
76
77            // The gate asks two questions of an applied recommendation: did
78            // its metric hold, and does its premise still stand. The engine
79            // feeds both here as inputs; the summary says which one failed.
80            let key = if input.metric == crate::engine::PREMISE_DRIFT_METRIC {
81                "outcome.premise_drift"
82            } else {
83                "outcome.regression"
84            };
85            drafts.push(
86                RecDraft::new(
87                    input.target_ref.clone(),
88                    ActionKind::Revert,
89                    Summary::new(key, args),
90                    Proposal::Data { data },
91                )
92                .severity(Severity::High)
93                .evidence(vec![input.rec_hash.clone()]),
94            );
95        }
96        drafts.sort_by(|a, b| a.evidence.cmp(&b.evidence));
97        Ok(drafts)
98    }
99}
100
101fn round4(x: f64) -> f64 {
102    (x * 10_000.0).round() / 10_000.0
103}
104
105#[cfg(test)]
106mod tests {
107    use super::*;
108    use crate::analyzer::OutcomeInput;
109    use crate::testkit::TestSubstrate;
110
111    fn input(baseline: f64, current: f64) -> OutcomeInput {
112        OutcomeInput {
113            rec_hash: "ref-1".into(),
114            target_ref: "entity:lessons/stripe_refund".into(),
115            metric: "tool_error_rate".into(),
116            baseline,
117            current,
118            unit: "ratio".into(),
119            higher_is_better: false,
120        }
121    }
122
123    /// A higher-is-better metric (an evalset accuracy) must invert the verdict.
124    /// Reading it the built-in way would propose reverting exactly the changes
125    /// that worked.
126    fn rising(baseline: f64, current: f64) -> OutcomeInput {
127        OutcomeInput {
128            metric: "evalset:abc123:category_accuracy".into(),
129            higher_is_better: true,
130            ..input(baseline, current)
131        }
132    }
133
134    #[test]
135    fn proposes_revert_on_regression() {
136        let mut sub = TestSubstrate::new();
137        sub.set_outcome_inputs(vec![input(0.2, 0.5)]);
138        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
139        assert_eq!(drafts.len(), 1);
140        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
141    }
142
143    #[test]
144    fn silent_when_improved_or_unchanged() {
145        let mut sub = TestSubstrate::new();
146        sub.set_outcome_inputs(vec![input(0.5, 0.2), input(0.3, 0.3)]);
147        assert!(sub.analyze(&OutcomeReview::new(), 10_000).is_empty());
148    }
149
150    #[test]
151    fn a_higher_is_better_metric_regresses_when_it_falls() {
152        let mut sub = TestSubstrate::new();
153        // Accuracy dropped 0.92 -> 0.71: that IS the regression.
154        sub.set_outcome_inputs(vec![rising(0.92, 0.71)]);
155        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
156        assert_eq!(drafts.len(), 1, "a fall in accuracy must propose a revert");
157        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
158    }
159
160    #[test]
161    fn a_higher_is_better_metric_holds_when_it_rises() {
162        let mut sub = TestSubstrate::new();
163        // The failure this pins: reading accuracy with the lower-is-better
164        // rule would revert the recommendation that improved it.
165        sub.set_outcome_inputs(vec![rising(0.71, 0.92), rising(0.8, 0.8)]);
166        assert!(
167            sub.analyze(&OutcomeReview::new(), 10_000).is_empty(),
168            "rising accuracy is the receipt, not a regression"
169        );
170    }
171}