Skip to main content

areev_loop/analyzers/
tool_failure.rs

1//! Tool-failure clustering (T0) — the flagship analyzer. Groups error Tool
2//! grains (captured tool calls) by (tool_name, normalized error signature) and
3//! fires when a cluster is frequent (≥ min_count) AND is either a meaningful
4//! share of that signature's OPPORTUNITIES (≥ min_rate) OR a large absolute
5//! count (≥ min_abs) — so high-volume, moderate-rate failures aren't hidden.
6//! Emits a memory lesson. Because the signature is derived from
7//! attacker-influenceable tool output, this analyzer never auto-applies
8//! (§6.3) — its manifest is `Never`.
9//!
10//! **Opportunities, not all calls.** The rate denominator is the tool's
11//! successful calls plus this cluster — NOT every call to the tool. A call
12//! that failed at an earlier check never reached the failure being scored, so
13//! counting it as an opportunity understates every mode. Using all calls made
14//! sibling failure modes mask each other: the more distinct ways a tool broke,
15//! the smaller each mode's share, so the tools failing in the most ways were
16//! the hardest to learn from — backwards. Measured on a real agent trace (150
17//! tasks, 772 calls, 139 failures across 5 modes) the old denominator put
18//! every mode between 9% and 30% and the analyzer proposed nothing at all.
19
20use crate::analyzer::{AnalyzeCtx, Analyzer};
21use crate::analyzers::bound_evidence;
22use crate::cal;
23use crate::error::Result;
24use crate::manifest::*;
25use crate::model::{ActionKind, Severity};
26use crate::recommendation::{MetricSnapshot, Proposal, RecDraft, Summary};
27use std::collections::BTreeMap;
28
29use serde_json::{json, Map};
30
31pub struct ToolFailureClustering {
32    manifest: AnalyzerManifest,
33}
34
35impl ToolFailureClustering {
36    pub fn new() -> Self {
37        ToolFailureClustering {
38            manifest: AnalyzerManifest {
39                id: "loop.tool_failure/1".into(),
40                title: "Tool-failure clustering".into(),
41                description: "Clusters recurring tool failures into a memory lesson.".into(),
42                tier: Tier::T0,
43                cadence: CadenceClass::Batch,
44                requires: vec![],
45                target_classes: vec![TargetClass::Memory],
46                auto_apply: AutoApplyClass::Never, // evidence-derived free text
47                trust_class: TrustClass::Builtin,
48                params: vec![
49                    ParamSpec::Int {
50                        name: "min_count".into(),
51                        default: 3,
52                        min: 1,
53                        max: 1000,
54                        description: "Minimum failures in a cluster to fire.".into(),
55                    },
56                    ParamSpec::Float {
57                        name: "min_rate".into(),
58                        default: 0.4,
59                        min: 0.0,
60                        max: 1.0,
61                        description: "Minimum share of this signature's opportunities \
62                                      (the tool's successes plus this cluster)."
63                            .into(),
64                    },
65                    ParamSpec::Int {
66                        name: "min_abs".into(),
67                        default: 50,
68                        min: 1,
69                        max: 100_000,
70                        description: "Absolute failure count that fires regardless of rate \
71                                      (so high-volume, moderate-rate failures aren't hidden)."
72                            .into(),
73                    },
74                    ParamSpec::Int {
75                        name: "window_days".into(),
76                        default: 30,
77                        min: 1,
78                        max: 365,
79                        description: "Lookback window.".into(),
80                    },
81                ],
82                default_on: true,
83            },
84        }
85    }
86}
87
88impl Default for ToolFailureClustering {
89    fn default() -> Self {
90        Self::new()
91    }
92}
93
94impl Analyzer for ToolFailureClustering {
95    fn manifest(&self) -> &AnalyzerManifest {
96        &self.manifest
97    }
98
99    fn analyze(&self, ctx: &AnalyzeCtx) -> Result<Vec<RecDraft>> {
100        let min_count = ctx.params().get_int("min_count").max(1) as usize;
101        let min_rate = ctx.params().get_float("min_rate");
102        let min_abs = ctx.params().get_int("min_abs").max(1) as usize;
103        let window_ms = ctx.params().get_int("window_days") * 86_400_000;
104        let since = Some(ctx.now_ms() - window_ms);
105
106        let tools = ctx.tools_since(since)?;
107
108        // Per-tool total calls, and per (tool, signature) error clusters —
109        // each member carries its namespace so the lesson can land where the
110        // evidence lives.
111        let mut tool_totals: BTreeMap<String, usize> = BTreeMap::new();
112        let mut tool_errors: BTreeMap<String, usize> = BTreeMap::new();
113        let mut clusters: BTreeMap<(String, String), Vec<(String, String)>> = BTreeMap::new();
114        for e in &tools {
115            let Some(tool) = e.tool_name() else {
116                continue;
117            };
118            *tool_totals.entry(tool.to_string()).or_default() += 1;
119            if e.is_error() {
120                *tool_errors.entry(tool.to_string()).or_default() += 1;
121                let sig = normalize_signature(e.tool_content().unwrap_or(""));
122                clusters
123                    .entry((tool.to_string(), sig))
124                    .or_default()
125                    .push((e.hash.clone(), e.namespace.clone()));
126            }
127        }
128
129        let mut drafts = Vec::new();
130        for ((tool, signature), mut members) in clusters {
131            // An empty/whitespace-only error body normalizes to "" — a lesson
132            // with object="" would trip the store's non-empty-object validation
133            // (VAL-E001) at apply time, so it can never be applied. Drop the
134            // cluster rather than queue an unappliable recommendation.
135            if signature.is_empty() {
136                continue;
137            }
138            let count = members.len();
139            // Opportunities = this tool's successful calls + this cluster.
140            // Calls that failed some OTHER way never reached this failure, so
141            // they are not opportunities to exhibit it (see the module docs).
142            let calls = tool_totals.get(&tool).copied().unwrap_or(count);
143            let errors = tool_errors.get(&tool).copied().unwrap_or(count);
144            let opportunities = calls.saturating_sub(errors).saturating_add(count).max(1);
145            let rate = count as f64 / opportunities as f64;
146            // Fire on a meaningful cluster that is EITHER a high share of the
147            // tool's calls OR a large absolute count — so a tool called 1000×
148            // at a 30% failure rate (300 real failures) isn't hidden by the
149            // rate gate alone.
150            if count < min_count || (rate < min_rate && count < min_abs) {
151                continue;
152            }
153            members.sort_by(|a, b| a.0.cmp(&b.0));
154            let evidence = bound_evidence(members.iter().map(|(h, _)| h.clone()).collect());
155            let rate_pct = (rate * 100.0).round() as i64;
156
157            let mut args = Map::new();
158            args.insert("tool".into(), json!(tool));
159            args.insert("count".into(), json!(count));
160            args.insert("rate".into(), json!(rate_pct));
161            args.insert("signature".into(), json!(signature));
162
163            // Proposed lesson: a fact recording the recurring failure. It
164            // lands in the DOMINANT namespace of the evidence tool calls (an
165            // ADD without a namespace would fall to the store default and be
166            // invisible to the ns-scoped recall the agent actually runs);
167            // the `entity:lessons/…` target_ref stays a stable grouping
168            // label, deliberately independent of where the grain lives.
169            let mut ns_counts: BTreeMap<&str, usize> = BTreeMap::new();
170            for (_, ns) in &members {
171                if !ns.is_empty() {
172                    *ns_counts.entry(ns.as_str()).or_default() += 1;
173                }
174            }
175            // Max count; ties break to the lexicographically smallest
176            // namespace — deterministic on any host.
177            let lesson_ns = ns_counts
178                .iter()
179                .max_by(|a, b| a.1.cmp(b.1).then_with(|| b.0.cmp(a.0)))
180                .map(|(ns, _)| ns.to_string());
181            let mut lesson = Map::new();
182            lesson.insert("subject".into(), json!(tool));
183            lesson.insert("relation".into(), json!("fails_with"));
184            lesson.insert("object".into(), json!(signature));
185            lesson.insert("confidence".into(), json!(rate));
186            if let Some(ns) = &lesson_ns {
187                lesson.insert("namespace".into(), json!(ns));
188            }
189
190            let severity = if rate >= 0.7 && count >= 5 {
191                Severity::High
192            } else if rate >= 0.5 {
193                Severity::Medium
194            } else {
195                Severity::Low
196            };
197
198            drafts.push(
199                RecDraft::new(
200                    format!("entity:lessons/{tool}"),
201                    ActionKind::ClusterFailure,
202                    Summary::new("tool_failure.cluster", args),
203                    Proposal::Cal {
204                        cal: cal::add("fact", &lesson),
205                    },
206                )
207                .severity(severity)
208                .evidence(evidence)
209                .confidence(rate)
210                .metric(MetricSnapshot {
211                    // After the lesson is applied, does this exact tool failure
212                    // recur? Baseline 0 = we expect zero recurrences if the
213                    // lesson worked; any recurrence is a regression → revert.
214                    metric: "tool_error_recurrence".into(),
215                    baseline: 0.0,
216                    unit: "count".into(),
217                    // Sample size is the set the rate was computed over — this
218                    // signature's opportunities, not every call to the tool.
219                    n: opportunities as u64,
220                    window: format!("{}d", ctx.params().get_int("window_days")),
221                    subject: Some(tool.clone()),
222                    namespace: None,
223                    // The metric is scoped to THIS failure signature, not the
224                    // whole tool — `relation` carries it so measure_metric can
225                    // count only recurrences of the same signature. Without it a
226                    // later, unrelated failure of the same tool reads as a
227                    // regression and reverts a still-valid lesson.
228                    relation: Some(signature.clone()),
229                    query: format!(
230                        "RECALL tools WHERE tool_name = \"{tool}\" AND is_error AND signature = \"{signature}\" SINCE <applied_at> | COUNT"
231                    ),
232                    review_after_ms: 86_400_000,
233                    // Re-measure at 1 day, 1 week, 1 month — a late recurrence
234                    // (held at 1d, regressed at 30d) is caught by the schedule.
235                    horizons_ms: vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000],
236                    checkpoints: Vec::new(),
237                    // A count of recurrences: fewer is better.
238                    higher_is_better: false,
239                }),
240            );
241        }
242        drafts.sort_by(|a, b| a.target_ref.cmp(&b.target_ref));
243        Ok(drafts)
244    }
245}
246
247/// Normalize an error message into a stable signature: lowercase, first ~80
248/// chars, digit runs → `#`, path-like tokens → `<path>`. Regex-free (std only).
249/// `pub(crate)` so the Verify gate (`measure_metric`) can re-derive the same
250/// signature when scoping a `tool_error_recurrence` re-measurement.
251pub(crate) fn normalize_signature(content: &str) -> String {
252    let lowered: String = content.trim().to_lowercase().chars().take(80).collect();
253    let mut out = String::with_capacity(lowered.len());
254    for token in lowered.split_whitespace() {
255        if !out.is_empty() {
256            out.push(' ');
257        }
258        if token.contains('/') || token.contains('\\') {
259            out.push_str("<path>");
260        } else {
261            let mut prev_digit = false;
262            for c in token.chars() {
263                if c.is_ascii_digit() {
264                    if !prev_digit {
265                        out.push('#');
266                    }
267                    prev_digit = true;
268                } else {
269                    out.push(c);
270                    prev_digit = false;
271                }
272            }
273        }
274    }
275    out
276}
277
278#[cfg(test)]
279mod tests {
280    use super::*;
281    use crate::testkit::TestSubstrate;
282
283    #[test]
284    fn signature_strips_digits_and_paths() {
285        assert_eq!(
286            normalize_signature("Rate limited after 4295 ms"),
287            "rate limited after # ms"
288        );
289        assert_eq!(
290            normalize_signature("open /etc/passwd failed"),
291            "open <path> failed"
292        );
293    }
294
295    #[test]
296    fn fires_on_frequent_and_dominant_cluster() {
297        let mut sub = TestSubstrate::new();
298        for _ in 0..5 {
299            sub.add_tool_call("stripe_refund", true, "rate_limited 429");
300        }
301        sub.add_tool_call("stripe_refund", false, "ok");
302        let drafts = sub.analyze(&ToolFailureClustering::new(), 10_000);
303        assert_eq!(drafts.len(), 1);
304        assert_eq!(drafts[0].action_kind, ActionKind::ClusterFailure);
305        assert_eq!(drafts[0].evidence.len(), 5);
306    }
307
308    #[test]
309    fn fires_on_high_volume_moderate_rate_via_absolute_count() {
310        let mut sub = TestSubstrate::new();
311        // 10 identical failures out of 40 calls = 25% (below the 40% rate) —
312        // but 10 ≥ min_abs, so it must still fire.
313        for _ in 0..10 {
314            sub.add_tool_call("search", true, "boom 500");
315        }
316        for _ in 0..30 {
317            sub.add_tool_call("search", false, "ok");
318        }
319        let drafts = sub.analyze_with(&ToolFailureClustering::new(), 10_000, &[("min_abs", serde_json::json!(10))]);
320        assert_eq!(drafts.len(), 1, "high-volume moderate-rate cluster fires via min_abs");
321    }
322
323    #[test]
324    fn sibling_failure_modes_do_not_mask_each_other() {
325        // The shape a real agent trace produced: one tool, several distinct
326        // failure modes, none of them a large absolute count. Scored against
327        // ALL calls every mode sat under the rate gate and the analyzer went
328        // silent on 93 real failures; scored against each mode's own
329        // opportunities (successes + that mode) the dominant one fires.
330        let mut sub = TestSubstrate::new();
331        for _ in 0..46 {
332            sub.add_tool_call("refund", true, "rate_limited 429");
333        }
334        for _ in 0..33 {
335            sub.add_tool_call("refund", true, "approval_required");
336        }
337        for _ in 0..14 {
338            sub.add_tool_call("refund", true, "cancel_before_refund");
339        }
340        for _ in 0..59 {
341            sub.add_tool_call("refund", false, "ok");
342        }
343        let drafts = sub.analyze(&ToolFailureClustering::new(), 10_000);
344        // 46 / (59 + 46) = 43.8% clears the 40% gate; the smaller siblings
345        // (33/92 = 36%, 14/73 = 19%) stay below it and stay silent.
346        assert_eq!(drafts.len(), 1, "the dominant mode fires, its siblings do not");
347        assert_eq!(drafts[0].evidence.len(), 46);
348    }
349
350    #[test]
351    fn a_mode_that_is_a_small_share_of_its_own_opportunities_stays_silent() {
352        // The denominator change must not turn every recurring error into a
353        // lesson: 5 failures against 95 successes is 5%, still silent.
354        let mut sub = TestSubstrate::new();
355        for _ in 0..5 {
356            sub.add_tool_call("search", true, "boom 500");
357        }
358        for _ in 0..95 {
359            sub.add_tool_call("search", false, "ok");
360        }
361        assert!(sub.analyze(&ToolFailureClustering::new(), 10_000).is_empty());
362    }
363
364    #[test]
365    fn silent_below_rate_threshold() {
366        let mut sub = TestSubstrate::new();
367        sub.add_tool_call("search", true, "boom 500");
368        sub.add_tool_call("search", true, "boom 500");
369        for _ in 0..20 {
370            sub.add_tool_call("search", false, "ok");
371        }
372        // 2 errors of 22 calls = 9% < 40%.
373        assert!(sub
374            .analyze(&ToolFailureClustering::new(), 10_000)
375            .is_empty());
376    }
377}