Skip to main content

areev_loop/analyzers/
outcome_review.rs

1//! Outcome review (T0). For applied recommendations past their `review_after`,
2//! the engine re-runs the stored metric query (it owns the `&mut` substrate)
3//! and hands the measured values in as `OutcomeInput`s; this analyzer makes the
4//! deterministic changed/regressed decision and proposes a revert on
5//! regression. Closes the honesty loop — makes approve and auto-apply
6//! accountable to measured history.
7//!
8//! Our built-in metrics are lower-is-better (e.g. `tool_error_rate`), so a
9//! regression is `current > baseline` beyond a small epsilon. An evalset
10//! metric can be higher-is-better and inverts that, which is why the direction
11//! travels on the input and the comparison lives in ONE place
12//! (`recommendation::is_regression`).
13
14use crate::analyzer::{AnalyzeCtx, Analyzer};
15use crate::error::Result;
16use crate::manifest::*;
17use crate::model::{ActionKind, Severity};
18use crate::recommendation::{Proposal, RecDraft, Summary};
19use serde_json::{json, Map};
20
21pub struct OutcomeReview {
22    manifest: AnalyzerManifest,
23}
24
25impl OutcomeReview {
26    pub fn new() -> Self {
27        OutcomeReview {
28            manifest: AnalyzerManifest {
29                id: "loop.outcome_review/1".into(),
30                title: "Outcome review".into(),
31                description:
32                    "Re-measures applied recommendations and proposes revert on regression.".into(),
33                tier: Tier::T0,
34                cadence: CadenceClass::Fast,
35                requires: vec![],
36                target_classes: vec![TargetClass::Memory, TargetClass::Query],
37                auto_apply: AutoApplyClass::Never,
38                trust_class: TrustClass::Builtin,
39                params: vec![],
40                default_on: true,
41            },
42        }
43    }
44}
45
46impl Default for OutcomeReview {
47    fn default() -> Self {
48        Self::new()
49    }
50}
51
52impl Analyzer for OutcomeReview {
53    fn manifest(&self) -> &AnalyzerManifest {
54        &self.manifest
55    }
56
57    fn analyze(&self, ctx: &AnalyzeCtx) -> Result<Vec<RecDraft>> {
58        let mut drafts = Vec::new();
59        for input in ctx.outcome_inputs() {
60            let regressed = crate::recommendation::is_regression(
61                input.baseline,
62                input.current,
63                input.higher_is_better,
64                input.tolerance,
65            );
66            let costlier = input.cost.as_ref().is_some_and(|c| c.breached());
67            if !regressed && !costlier {
68                continue;
69            }
70            let mut args = Map::new();
71            args.insert("metric".into(), json!(input.metric));
72            args.insert("baseline".into(), json!(round4(input.baseline)));
73            args.insert("current".into(), json!(round4(input.current)));
74            if let Some(run) = &input.baseline_run_id {
75                args.insert("baseline_run".into(), json!(run));
76            }
77            if let Some(run) = &input.current_run_id {
78                args.insert("current_run".into(), json!(run));
79            }
80            if let Some(best) = input.best_before {
81                args.insert("best_before".into(), json!(round4(best)));
82            }
83            if input.tolerance > 0.0 {
84                args.insert("tolerance".into(), json!(round4(input.tolerance)));
85            }
86            if let Some(c) = input.cost.as_ref().filter(|c| c.breached()) {
87                args.insert("cost_field".into(), json!(c.field));
88                args.insert("cost_baseline".into(), json!(c.baseline.map(round4)));
89                args.insert("cost_current".into(), json!(c.current.map(round4)));
90                args.insert("cost_ratio".into(), json!(c.max_increase_ratio));
91            }
92
93            if !regressed {
94                // Quality held, cost did not: advisory. A cost/quality trade
95                // is a human decision, so this is a Flag citing both runs —
96                // never a revert draft.
97                let mut data = Map::new();
98                data.insert("cost_of".into(), json!(input.rec_hash));
99                data.insert("metric".into(), json!(input.metric));
100                if let Some(c) = &input.cost {
101                    data.insert("cost".into(), serde_json::to_value(c).unwrap_or_default());
102                }
103                drafts.push(
104                    RecDraft::new(
105                        input.target_ref.clone(),
106                        ActionKind::Flag,
107                        Summary::new("outcome.held_costlier", args),
108                        Proposal::Data { data },
109                    )
110                    .severity(Severity::Medium)
111                    .evidence(vec![input.rec_hash.clone()]),
112                );
113                continue;
114            }
115
116            let mut data = Map::new();
117            data.insert("revert_of".into(), json!(input.rec_hash));
118            data.insert("metric".into(), json!(input.metric));
119            if let Some(c) = input.cost.as_ref().filter(|c| c.breached()) {
120                data.insert("cost".into(), serde_json::to_value(c).unwrap_or_default());
121            }
122
123            // The gate asks two questions of an applied recommendation: did
124            // its metric hold, and does its premise still stand. The engine
125            // feeds both here as inputs; the summary says which one failed.
126            let key = if input.metric == crate::engine::PREMISE_DRIFT_METRIC {
127                "outcome.premise_drift"
128            } else {
129                // Drafted against the peak: the summary names both figures so
130                // the reviewer judges whether this rule owns the whole fall.
131                // A breached cost bound rides on the revert as a second clause.
132                match (input.baseline_kind == "high_water", costlier) {
133                    (false, false) => "outcome.regression",
134                    (true, false) => "outcome.regression_high_water",
135                    (false, true) => "outcome.regression_costlier",
136                    (true, true) => "outcome.regression_high_water_costlier",
137                }
138            };
139            drafts.push(
140                RecDraft::new(
141                    input.target_ref.clone(),
142                    ActionKind::Revert,
143                    Summary::new(key, args),
144                    Proposal::Data { data },
145                )
146                .severity(Severity::High)
147                .evidence(vec![input.rec_hash.clone()]),
148            );
149        }
150        drafts.sort_by(|a, b| a.evidence.cmp(&b.evidence));
151        Ok(drafts)
152    }
153}
154
155fn round4(x: f64) -> f64 {
156    (x * 10_000.0).round() / 10_000.0
157}
158
159#[cfg(test)]
160mod tests {
161    use super::*;
162    use crate::analyzer::OutcomeInput;
163    use crate::testkit::TestSubstrate;
164
165    fn input(baseline: f64, current: f64) -> OutcomeInput {
166        OutcomeInput {
167            rec_hash: "ref-1".into(),
168            target_ref: "entity:lessons/stripe_refund".into(),
169            metric: "tool_error_rate".into(),
170            baseline,
171            current,
172            unit: "ratio".into(),
173            higher_is_better: false,
174            baseline_kind: "snapshot".into(),
175            baseline_run_id: None,
176            best_before: None,
177            tolerance: 0.0,
178            current_run_id: None,
179            cost: None,
180        }
181    }
182
183    fn breached(field: &str, baseline: f64, current: f64, ratio: f64) -> crate::recommendation::CostRead {
184        crate::recommendation::CostRead {
185            field: field.into(),
186            max_increase_ratio: ratio,
187            baseline: Some(baseline),
188            current: Some(current),
189            status: "breached".into(),
190        }
191    }
192
193    /// A higher-is-better metric (an evalset accuracy) must invert the verdict.
194    /// Reading it the built-in way would propose reverting exactly the changes
195    /// that worked.
196    fn rising(baseline: f64, current: f64) -> OutcomeInput {
197        OutcomeInput {
198            metric: "evalset:abc123:category_accuracy".into(),
199            higher_is_better: true,
200            ..input(baseline, current)
201        }
202    }
203
204    #[test]
205    fn proposes_revert_on_regression() {
206        let mut sub = TestSubstrate::new();
207        sub.set_outcome_inputs(vec![input(0.2, 0.5)]);
208        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
209        assert_eq!(drafts.len(), 1);
210        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
211    }
212
213    #[test]
214    fn silent_when_improved_or_unchanged() {
215        let mut sub = TestSubstrate::new();
216        sub.set_outcome_inputs(vec![input(0.5, 0.2), input(0.3, 0.3)]);
217        assert!(sub.analyze(&OutcomeReview::new(), 10_000).is_empty());
218    }
219
220    #[test]
221    fn a_higher_is_better_metric_regresses_when_it_falls() {
222        let mut sub = TestSubstrate::new();
223        // Accuracy dropped 0.92 -> 0.71: that IS the regression.
224        sub.set_outcome_inputs(vec![rising(0.92, 0.71)]);
225        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
226        assert_eq!(drafts.len(), 1, "a fall in accuracy must propose a revert");
227        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
228    }
229
230    /// A verdict drafted against the high-water mark names the run it fell
231    /// from: a reviewer deciding whether THIS rule owns the whole fall needs
232    /// the peak beside the current value, not a bare pair of numbers.
233    #[test]
234    fn a_high_water_regression_names_the_peak_run() {
235        let mut sub = TestSubstrate::new();
236        sub.set_outcome_inputs(vec![OutcomeInput {
237            baseline_kind: "high_water".into(),
238            baseline_run_id: Some("eval-peak".into()),
239            best_before: Some(238.0),
240            ..rising(238.0, 133.0)
241        }]);
242        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
243        assert_eq!(drafts.len(), 1);
244        let text = drafts[0].summary.render();
245        assert!(text.contains("eval-peak") && text.contains("238") && text.contains("133"), "{text}");
246        assert_eq!(drafts[0].summary.args["best_before"], 238.0);
247    }
248
249    /// The floor the engine judged under travels with the input: a dip inside
250    /// it drafts nothing, one past it drafts the revert and records the floor.
251    #[test]
252    fn the_revert_draft_applies_the_same_floor_as_the_verdict() {
253        let mut sub = TestSubstrate::new();
254        sub.set_outcome_inputs(vec![
255            OutcomeInput { tolerance: 5.0, ..rising(359.0, 355.0) },
256            OutcomeInput { tolerance: 5.0, rec_hash: "ref-2".into(), ..rising(359.0, 353.0) },
257        ]);
258        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
259        assert_eq!(drafts.len(), 1, "only the dip past the floor reverts");
260        assert_eq!(drafts[0].evidence, vec!["ref-2".to_string()]);
261        assert_eq!(drafts[0].summary.args["tolerance"], 5.0);
262    }
263
264    /// Quality held, cost breached: one advisory Flag citing both runs, and
265    /// no revert — the trade is the reviewer's to make.
266    #[test]
267    fn a_breached_cost_bound_under_a_held_score_is_a_flag_not_a_revert() {
268        let mut sub = TestSubstrate::new();
269        sub.set_outcome_inputs(vec![OutcomeInput {
270            baseline_run_id: Some("eval-before".into()),
271            current_run_id: Some("eval-after".into()),
272            cost: Some(breached("tokens", 1000.0, 1600.0, 1.5)),
273            ..rising(100.0, 102.0)
274        }]);
275        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
276        assert_eq!(drafts.len(), 1);
277        assert_eq!(drafts[0].action_kind, ActionKind::Flag);
278        assert_eq!(drafts[0].severity, Severity::Medium);
279        let text = drafts[0].summary.render();
280        assert!(text.contains("eval-before") && text.contains("eval-after"), "cites both runs: {text}");
281        assert!(text.contains("1000") && text.contains("1600") && text.contains("1.5"), "{text}");
282        assert_eq!(drafts[0].proposal, Proposal::Data { data: {
283            let mut m = Map::new();
284            m.insert("cost_of".into(), json!("ref-1"));
285            m.insert("metric".into(), json!("evalset:abc123:category_accuracy"));
286            m.insert("cost".into(), serde_json::to_value(breached("tokens", 1000.0, 1600.0, 1.5)).unwrap());
287            m
288        }});
289    }
290
291    /// Quality regressed AND cost breached: one revert (regressed dominates),
292    /// whose summary names the cost delta.
293    #[test]
294    fn a_regression_that_also_cost_more_is_one_revert_naming_both() {
295        let mut sub = TestSubstrate::new();
296        sub.set_outcome_inputs(vec![OutcomeInput {
297            cost: Some(breached("tokens", 1000.0, 1600.0, 1.5)),
298            ..rising(100.0, 90.0)
299        }]);
300        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
301        assert_eq!(drafts.len(), 1, "one verdict, one draft");
302        assert_eq!(drafts[0].action_kind, ActionKind::Revert);
303        let text = drafts[0].summary.render();
304        assert!(text.contains("regressed") && text.contains("tokens") && text.contains("1600"), "{text}");
305    }
306
307    /// A cost read that is within the bound, or not measurable, changes
308    /// nothing: held stays silent, regressed stays a plain revert.
309    #[test]
310    fn a_cost_within_bound_or_not_measurable_leaves_the_verdict_alone() {
311        let within = crate::recommendation::CostRead { status: "within".into(), ..breached("tokens", 1000.0, 1200.0, 1.5) };
312        let unmeasurable = crate::recommendation::CostRead {
313            status: "not_measurable".into(), baseline: Some(1000.0), current: None, ..breached("tokens", 0.0, 0.0, 1.5)
314        };
315        let mut sub = TestSubstrate::new();
316        sub.set_outcome_inputs(vec![
317            OutcomeInput { cost: Some(within.clone()), ..rising(100.0, 102.0) },
318            OutcomeInput { cost: Some(unmeasurable.clone()), rec_hash: "ref-2".into(), ..rising(100.0, 102.0) },
319            OutcomeInput { cost: Some(unmeasurable), rec_hash: "ref-3".into(), ..rising(100.0, 90.0) },
320        ]);
321        let drafts = sub.analyze(&OutcomeReview::new(), 10_000);
322        assert_eq!(drafts.len(), 1);
323        assert_eq!(drafts[0].evidence, vec!["ref-3".to_string()]);
324        assert_eq!(drafts[0].summary.template_id, "outcome.regression");
325    }
326
327    #[test]
328    fn a_higher_is_better_metric_holds_when_it_rises() {
329        let mut sub = TestSubstrate::new();
330        // The failure this pins: reading accuracy with the lower-is-better
331        // rule would revert the recommendation that improved it.
332        sub.set_outcome_inputs(vec![rising(0.71, 0.92), rising(0.8, 0.8)]);
333        assert!(
334            sub.analyze(&OutcomeReview::new(), 10_000).is_empty(),
335            "rising accuracy is the receipt, not a regression"
336        );
337    }
338}