Skip to main content

areev_loop/analyzers/
tool_failure.rs

1//! Tool-failure clustering (T0) — the flagship analyzer. Groups error Tool
2//! grains (captured tool calls) by (tool_name, normalized error signature) and
3//! fires when a cluster is frequent (≥ min_count) AND is either a meaningful
4//! share of that tool's calls (≥ min_rate) OR a large absolute count
5//! (≥ min_abs) — so high-volume, moderate-rate failures aren't hidden. Emits a
6//! memory lesson. Because the signature is derived from
7//! attacker-influenceable tool output, this analyzer never auto-applies
8//! (§6.3) — its manifest is `Never`.
9
10use crate::analyzer::{AnalyzeCtx, Analyzer};
11use crate::analyzers::bound_evidence;
12use crate::cal;
13use crate::error::Result;
14use crate::manifest::*;
15use crate::model::{ActionKind, Severity};
16use crate::recommendation::{MetricSnapshot, Proposal, RecDraft, Summary};
17use std::collections::BTreeMap;
18
19use serde_json::{json, Map};
20
21pub struct ToolFailureClustering {
22    manifest: AnalyzerManifest,
23}
24
25impl ToolFailureClustering {
26    pub fn new() -> Self {
27        ToolFailureClustering {
28            manifest: AnalyzerManifest {
29                id: "loop.tool_failure/1".into(),
30                title: "Tool-failure clustering".into(),
31                description: "Clusters recurring tool failures into a memory lesson.".into(),
32                tier: Tier::T0,
33                cadence: CadenceClass::Batch,
34                requires: vec![],
35                target_classes: vec![TargetClass::Memory],
36                auto_apply: AutoApplyClass::Never, // evidence-derived free text
37                trust_class: TrustClass::Builtin,
38                params: vec![
39                    ParamSpec::Int {
40                        name: "min_count".into(),
41                        default: 3,
42                        min: 1,
43                        max: 1000,
44                        description: "Minimum failures in a cluster to fire.".into(),
45                    },
46                    ParamSpec::Float {
47                        name: "min_rate".into(),
48                        default: 0.4,
49                        min: 0.0,
50                        max: 1.0,
51                        description: "Minimum share of the tool's calls.".into(),
52                    },
53                    ParamSpec::Int {
54                        name: "min_abs".into(),
55                        default: 50,
56                        min: 1,
57                        max: 100_000,
58                        description: "Absolute failure count that fires regardless of rate \
59                                      (so high-volume, moderate-rate failures aren't hidden)."
60                            .into(),
61                    },
62                    ParamSpec::Int {
63                        name: "window_days".into(),
64                        default: 30,
65                        min: 1,
66                        max: 365,
67                        description: "Lookback window.".into(),
68                    },
69                ],
70                default_on: true,
71            },
72        }
73    }
74}
75
76impl Default for ToolFailureClustering {
77    fn default() -> Self {
78        Self::new()
79    }
80}
81
82impl Analyzer for ToolFailureClustering {
83    fn manifest(&self) -> &AnalyzerManifest {
84        &self.manifest
85    }
86
87    fn analyze(&self, ctx: &AnalyzeCtx) -> Result<Vec<RecDraft>> {
88        let min_count = ctx.params().get_int("min_count").max(1) as usize;
89        let min_rate = ctx.params().get_float("min_rate");
90        let min_abs = ctx.params().get_int("min_abs").max(1) as usize;
91        let window_ms = ctx.params().get_int("window_days") * 86_400_000;
92        let since = Some(ctx.now_ms() - window_ms);
93
94        let tools = ctx.tools_since(since)?;
95
96        // Per-tool total calls, and per (tool, signature) error clusters —
97        // each member carries its namespace so the lesson can land where the
98        // evidence lives.
99        let mut tool_totals: BTreeMap<String, usize> = BTreeMap::new();
100        let mut clusters: BTreeMap<(String, String), Vec<(String, String)>> = BTreeMap::new();
101        for e in &tools {
102            let Some(tool) = e.tool_name() else {
103                continue;
104            };
105            *tool_totals.entry(tool.to_string()).or_default() += 1;
106            if e.is_error() {
107                let sig = normalize_signature(e.tool_content().unwrap_or(""));
108                clusters
109                    .entry((tool.to_string(), sig))
110                    .or_default()
111                    .push((e.hash.clone(), e.namespace.clone()));
112            }
113        }
114
115        let mut drafts = Vec::new();
116        for ((tool, signature), mut members) in clusters {
117            // An empty/whitespace-only error body normalizes to "" — a lesson
118            // with object="" would trip the store's non-empty-object validation
119            // (VAL-E001) at apply time, so it can never be applied. Drop the
120            // cluster rather than queue an unappliable recommendation.
121            if signature.is_empty() {
122                continue;
123            }
124            let count = members.len();
125            let total = tool_totals.get(&tool).copied().unwrap_or(count).max(1);
126            let rate = count as f64 / total as f64;
127            // Fire on a meaningful cluster that is EITHER a high share of the
128            // tool's calls OR a large absolute count — so a tool called 1000×
129            // at a 30% failure rate (300 real failures) isn't hidden by the
130            // rate gate alone.
131            if count < min_count || (rate < min_rate && count < min_abs) {
132                continue;
133            }
134            members.sort_by(|a, b| a.0.cmp(&b.0));
135            let evidence = bound_evidence(members.iter().map(|(h, _)| h.clone()).collect());
136            let rate_pct = (rate * 100.0).round() as i64;
137
138            let mut args = Map::new();
139            args.insert("tool".into(), json!(tool));
140            args.insert("count".into(), json!(count));
141            args.insert("rate".into(), json!(rate_pct));
142            args.insert("signature".into(), json!(signature));
143
144            // Proposed lesson: a fact recording the recurring failure. It
145            // lands in the DOMINANT namespace of the evidence tool calls (an
146            // ADD without a namespace would fall to the store default and be
147            // invisible to the ns-scoped recall the agent actually runs);
148            // the `entity:lessons/…` target_ref stays a stable grouping
149            // label, deliberately independent of where the grain lives.
150            let mut ns_counts: BTreeMap<&str, usize> = BTreeMap::new();
151            for (_, ns) in &members {
152                if !ns.is_empty() {
153                    *ns_counts.entry(ns.as_str()).or_default() += 1;
154                }
155            }
156            // Max count; ties break to the lexicographically smallest
157            // namespace — deterministic on any host.
158            let lesson_ns = ns_counts
159                .iter()
160                .max_by(|a, b| a.1.cmp(b.1).then_with(|| b.0.cmp(a.0)))
161                .map(|(ns, _)| ns.to_string());
162            let mut lesson = Map::new();
163            lesson.insert("subject".into(), json!(tool));
164            lesson.insert("relation".into(), json!("fails_with"));
165            lesson.insert("object".into(), json!(signature));
166            lesson.insert("confidence".into(), json!(rate));
167            if let Some(ns) = &lesson_ns {
168                lesson.insert("namespace".into(), json!(ns));
169            }
170
171            let severity = if rate >= 0.7 && count >= 5 {
172                Severity::High
173            } else if rate >= 0.5 {
174                Severity::Medium
175            } else {
176                Severity::Low
177            };
178
179            drafts.push(
180                RecDraft::new(
181                    format!("entity:lessons/{tool}"),
182                    ActionKind::ClusterFailure,
183                    Summary::new("tool_failure.cluster", args),
184                    Proposal::Cal {
185                        cal: cal::add("fact", &lesson),
186                    },
187                )
188                .severity(severity)
189                .evidence(evidence)
190                .confidence(rate)
191                .metric(MetricSnapshot {
192                    // After the lesson is applied, does this exact tool failure
193                    // recur? Baseline 0 = we expect zero recurrences if the
194                    // lesson worked; any recurrence is a regression → revert.
195                    metric: "tool_error_recurrence".into(),
196                    baseline: 0.0,
197                    unit: "count".into(),
198                    n: total as u64,
199                    window: format!("{}d", ctx.params().get_int("window_days")),
200                    subject: Some(tool.clone()),
201                    namespace: None,
202                    // The metric is scoped to THIS failure signature, not the
203                    // whole tool — `relation` carries it so measure_metric can
204                    // count only recurrences of the same signature. Without it a
205                    // later, unrelated failure of the same tool reads as a
206                    // regression and reverts a still-valid lesson.
207                    relation: Some(signature.clone()),
208                    query: format!(
209                        "RECALL tools WHERE tool_name = \"{tool}\" AND is_error AND signature = \"{signature}\" SINCE <applied_at> | COUNT"
210                    ),
211                    review_after_ms: 86_400_000,
212                    // Re-measure at 1 day, 1 week, 1 month — a late recurrence
213                    // (held at 1d, regressed at 30d) is caught by the schedule.
214                    horizons_ms: vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000],
215                }),
216            );
217        }
218        drafts.sort_by(|a, b| a.target_ref.cmp(&b.target_ref));
219        Ok(drafts)
220    }
221}
222
223/// Normalize an error message into a stable signature: lowercase, first ~80
224/// chars, digit runs → `#`, path-like tokens → `<path>`. Regex-free (std only).
225/// `pub(crate)` so the Verify gate (`measure_metric`) can re-derive the same
226/// signature when scoping a `tool_error_recurrence` re-measurement.
227pub(crate) fn normalize_signature(content: &str) -> String {
228    let lowered: String = content.trim().to_lowercase().chars().take(80).collect();
229    let mut out = String::with_capacity(lowered.len());
230    for token in lowered.split_whitespace() {
231        if !out.is_empty() {
232            out.push(' ');
233        }
234        if token.contains('/') || token.contains('\\') {
235            out.push_str("<path>");
236        } else {
237            let mut prev_digit = false;
238            for c in token.chars() {
239                if c.is_ascii_digit() {
240                    if !prev_digit {
241                        out.push('#');
242                    }
243                    prev_digit = true;
244                } else {
245                    out.push(c);
246                    prev_digit = false;
247                }
248            }
249        }
250    }
251    out
252}
253
254#[cfg(test)]
255mod tests {
256    use super::*;
257    use crate::testkit::TestSubstrate;
258
259    #[test]
260    fn signature_strips_digits_and_paths() {
261        assert_eq!(
262            normalize_signature("Rate limited after 4295 ms"),
263            "rate limited after # ms"
264        );
265        assert_eq!(
266            normalize_signature("open /etc/passwd failed"),
267            "open <path> failed"
268        );
269    }
270
271    #[test]
272    fn fires_on_frequent_and_dominant_cluster() {
273        let mut sub = TestSubstrate::new();
274        for _ in 0..5 {
275            sub.add_tool_call("stripe_refund", true, "rate_limited 429");
276        }
277        sub.add_tool_call("stripe_refund", false, "ok");
278        let drafts = sub.analyze(&ToolFailureClustering::new(), 10_000);
279        assert_eq!(drafts.len(), 1);
280        assert_eq!(drafts[0].action_kind, ActionKind::ClusterFailure);
281        assert_eq!(drafts[0].evidence.len(), 5);
282    }
283
284    #[test]
285    fn fires_on_high_volume_moderate_rate_via_absolute_count() {
286        let mut sub = TestSubstrate::new();
287        // 10 identical failures out of 40 calls = 25% (below the 40% rate) —
288        // but 10 ≥ min_abs, so it must still fire.
289        for _ in 0..10 {
290            sub.add_tool_call("search", true, "boom 500");
291        }
292        for _ in 0..30 {
293            sub.add_tool_call("search", false, "ok");
294        }
295        let drafts = sub.analyze_with(&ToolFailureClustering::new(), 10_000, &[("min_abs", serde_json::json!(10))]);
296        assert_eq!(drafts.len(), 1, "high-volume moderate-rate cluster fires via min_abs");
297    }
298
299    #[test]
300    fn silent_below_rate_threshold() {
301        let mut sub = TestSubstrate::new();
302        sub.add_tool_call("search", true, "boom 500");
303        sub.add_tool_call("search", true, "boom 500");
304        for _ in 0..20 {
305            sub.add_tool_call("search", false, "ok");
306        }
307        // 2 errors of 22 calls = 9% < 40%.
308        assert!(sub
309            .analyze(&ToolFailureClustering::new(), 10_000)
310            .is_empty());
311    }
312}