Skip to main content

areev_loop/analyzers/
outcome_review.rs

1//! Outcome review (T0). For applied recommendations past their `review_after`,
2//! the engine re-runs the stored metric query (it owns the `&mut` substrate)
3//! and hands the measured values in as `OutcomeInput`s; this analyzer makes the
4//! deterministic changed/regressed decision and proposes a revert on
5//! regression. Closes the honesty loop — makes approve and auto-apply
6//! accountable to measured history.
7//!
8//! Our built-in metrics are lower-is-better (e.g. `tool_error_rate`), so a
9//! regression is `current > baseline` beyond a small epsilon. An evalset
10//! metric can be higher-is-better and inverts that, which is why the direction
11//! travels on the input and the comparison lives in ONE place
12//! (`recommendation::is_regression`).
13
14use crate::analyzer::{AnalyzeCtx, Analyzer};
15use crate::error::Result;
16use crate::manifest::*;
17use crate::model::{ActionKind, Severity};
18use crate::recommendation::{Proposal, RecDraft, Summary};
19use serde_json::{json, Map};
20
21pub struct OutcomeReview {
22    manifest: AnalyzerManifest,
23}
24
25impl OutcomeReview {
26    pub fn new() -> Self {
27        OutcomeReview {
28            manifest: AnalyzerManifest {
29                id: "loop.outcome_review/1".into(),
30                title: "Outcome review".into(),
31                description:
32                    "Re-measures applied recommendations and proposes revert on regression.".into(),
33                tier: Tier::T0,
34                cadence: CadenceClass::Fast,
35                requires: vec![],
36                target_classes: vec![TargetClass::Memory, TargetClass::Query],
37                auto_apply: AutoApplyClass::Never,
38                trust_class: TrustClass::Builtin,
39                params: vec![],
40                default_on: true,
41            },
42        }
43    }
44}
45
46impl Default for OutcomeReview {
47    fn default() -> Self {
48        Self::new()
49    }
50}
51
52impl Analyzer for OutcomeReview {
53    fn manifest(&self) -> &AnalyzerManifest {
54        &self.manifest
55    }
56
57    fn analyze(&self, ctx: &AnalyzeCtx) -> Result<Vec<RecDraft>> {
58        let mut drafts = Vec::new();
59        for input in ctx.outcome_inputs() {
60            let regressed = crate::recommendation::is_regression(
61                input.baseline,
62                input.current,
63                input.higher_is_better,
64            );
65            if !regressed {
66                continue;
67            }
68            let mut args = Map::new();
69            args.insert("metric".into(), json!(input.metric));
70            args.insert("baseline".into(), json!(round4(input.baseline)));
71            args.insert("current".into(), json!(round4(input.current)));
72
73            let mut data = Map::new();
74            data.insert("revert_of".into(), json!(input.rec_hash));
75            data.insert("metric".into(), json!(input.metric));
76
77            drafts.push(
78                RecDraft::new(
79                    input.target_ref.clone(),
80                    ActionKind::Revert,
81                    Summary::new("outcome.regression", args),
82                    Proposal::Data { data },
83                )
84                .severity(Severity::High)
85                .evidence(vec![input.rec_hash.clone()]),
86            );
87        }
88        drafts.sort_by(|a, b| a.evidence.cmp(&b.evidence));
89        Ok(drafts)
90    }
91}
92
93fn round4(x: f64) -> f64 {
94    (x * 10_000.0).round() / 10_000.0
95}
96
97#[cfg(test)]
98mod tests {
99    use super::*;
100    use crate::analyzer::OutcomeInput;
101    use crate::testkit::TestSubstrate;
102
103    fn input(baseline: f64, current: f64) -> OutcomeInput {
104        OutcomeInput {
105            rec_hash: "ref-1".into(),
106            target_ref: "entity:lessons/stripe_refund".into(),
107            metric: "tool_error_rate".into(),
108            baseline,
109            current,
110            unit: "ratio".into(),
111            higher_is_better: false,
112        }
113    }
114
115    /// A higher-is-better metric (an evalset accuracy) must invert the verdict.
116    /// Reading it the built-in way would propose reverting exactly the changes
117    /// that worked.
118    fn rising(baseline: f64, current: f64) -> OutcomeInput {
119        OutcomeInput {
120            metric: "evalset:abc123:category_accuracy".into(),
121            higher_is_better: true,
122            ..input(baseline, current)
123        }
124    }
125
126    #[test]
127    fn proposes_revert_on_regression() {
128        let mut sub = TestSubstrate::new();
129        sub.set_outcome_inputs(vec![input(0.2, 0.5)]);
130        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
131        assert_eq!(drafts.len(), 1);
132        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
133    }
134
135    #[test]
136    fn silent_when_improved_or_unchanged() {
137        let mut sub = TestSubstrate::new();
138        sub.set_outcome_inputs(vec![input(0.5, 0.2), input(0.3, 0.3)]);
139        assert!(sub.analyze(&OutcomeReview::new(), 10_000).is_empty());
140    }
141
142    #[test]
143    fn a_higher_is_better_metric_regresses_when_it_falls() {
144        let mut sub = TestSubstrate::new();
145        // Accuracy dropped 0.92 -> 0.71: that IS the regression.
146        sub.set_outcome_inputs(vec![rising(0.92, 0.71)]);
147        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
148        assert_eq!(drafts.len(), 1, "a fall in accuracy must propose a revert");
149        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
150    }
151
152    #[test]
153    fn a_higher_is_better_metric_holds_when_it_rises() {
154        let mut sub = TestSubstrate::new();
155        // The failure this pins: reading accuracy with the lower-is-better
156        // rule would revert the recommendation that improved it.
157        sub.set_outcome_inputs(vec![rising(0.71, 0.92), rising(0.8, 0.8)]);
158        assert!(
159            sub.analyze(&OutcomeReview::new(), 10_000).is_empty(),
160            "rising accuracy is the receipt, not a regression"
161        );
162    }
163}