areev-loop 1.7.3

Areev Loop: the governed self-improvement engine for AI-agent memory. Standalone engine over an OmsSubstrate (CAL + grains) — zero Areev dependencies.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
//! Optional LLM enrichment (proposal §9).
//!
//! The engine's deterministic output stays a pure function of `(store, params,
//! now)`. This layer is strictly **additive**: with a backend attached the
//! pipeline gains two optional stages —
//!
//! ```text
//! ANALYZE (deterministic) → DISCOVER (LLM) → ENRICH (LLM) → VALIDATE+DEDUP → STORE
//! ```
//!
//! and with no backend those stages are the identity function, so the no-LLM
//! path is byte-for-byte the deterministic path. The LLM can only:
//!   - **DISCOVER**: propose *new* draft recommendations, which enter through
//!     the ordinary candidate/dedup/store path stamped `origin = llm` — so they
//!     can **never auto-apply** and never target prompt/host surfaces. A draft
//!     MAY author a `lesson` (one capped imperative line); a lesson-bearing
//!     draft that survives GROUND + VERIFY stamps as an *applicable*,
//!     rollbackable `ADD fact` proposal instead of an advisory flag — still
//!     `origin = llm`, so applying it always takes a human review with a
//!     BECAUSE plus an explicit apply; and
//!   - **ENRICH**: add a whitelisted `guidance` note to a deterministic
//!     recommendation. The engine-templated summary is always kept; the model
//!     never rewrites it.
//!
//! Trust floor (enforced by the engine, not the backend): responses are parsed
//! to a fixed schema (unknown fields dropped, strings capped), DISCOVER drafts
//! must cite evidence present in the bundle (by bundle id, full hash, or an
//! unambiguous hash prefix), instructions never interleave with evidence, and
//! a failed/timed-out/garbled call drops the LLM contribution for the run
//! rather than failing it.
//!
//! `CommandLlm` mirrors the shipped `CommandEmbed`: whitespace-split argv (no
//! shell), one process per call, a JSON request on stdin and a JSON response on
//! stdout, and a construction-time probe that fails loud.

use crate::error::{Error, Result};
use serde::{Deserialize, Serialize};
use serde_json::Value;

/// Caps that bound what a single LLM contribution can inject (defense in depth;
/// the engine enforces them after parsing).
pub const MAX_LLM_DRAFTS: usize = 8;
pub const MAX_GUIDANCE_LEN: usize = 600;
pub const MAX_SUMMARY_LEN: usize = 200;
/// An authored lesson is one imperative line — anything longer is a document,
/// not a lesson, and a bound on what a single approved apply can put into
/// every future prompt.
pub const MAX_LESSON_LEN: usize = 240;
/// Caps on the rest of the proposal vocabulary. Each one bounds what a single
/// approved apply can put into every future run of the agent, so they are part
/// of the trust floor rather than tuning knobs.
pub const MAX_RELATION_LEN: usize = 64;
pub const MAX_OBJECT_LEN: usize = 480;
/// A skill step is one instruction line; a skill has at most this many.
pub const MAX_SKILL_STEP_LEN: usize = 240;
pub const MAX_SKILL_STEPS: usize = 20;
/// A skill name is an identifier a person types at `skill_view`, not a title.
pub const MAX_SKILL_NAME_LEN: usize = 64;
/// A plan has at most this many steps; a condition is one short line.
pub const MAX_PLAN_NODES: usize = 20;
pub const MAX_COND_LEN: usize = 200;
pub const MAX_QUERY_BODY_LEN: usize = 2_000;
pub const MAX_PLAN_EDITS: usize = 8;
pub const MAX_CODE_LEN: usize = 20_000;

/// A backend that answers one JSON request with one JSON response. Object-safe
/// so the engine can hold a `Box<dyn LlmBackend>`.
pub trait LlmBackend: Send + Sync {
    /// Model identifier, stamped as provenance on `origin = llm` grains.
    fn model(&self) -> &str;
    /// Run one request. `request` is a JSON string; the returned text is
    /// expected to be JSON and is validated by the caller.
    fn complete(&self, request: &str) -> Result<String>;
}

/// Boxed backends forward — lets decorators wrap `Box<dyn LlmBackend>`
/// without knowing the concrete type.
impl<T: LlmBackend + ?Sized> LlmBackend for Box<T> {
    fn model(&self) -> &str {
        (**self).model()
    }
    fn complete(&self, request: &str) -> Result<String> {
        (**self).complete(request)
    }
}

// ---- wire schema (request) -------------------------------------------------

/// One deterministic finding, handed to DISCOVER as context (never as an
/// instruction — see `LlmRequest`).
#[derive(Debug, Clone, Serialize)]
pub struct FindingBrief {
    pub analyzer: String,
    pub summary: String,
    pub target: String,
    pub severity: String,
}

/// One evidence grain, provenance-tagged. `id` is the bundle-local label
/// (`e1`, `e2`, …) a draft may cite instead of the 64-hex `hash`: the engine
/// resolves either back to the grain, so a citation is still checked against
/// the bundle — it is just no longer a transcription test for the model.
#[derive(Debug, Clone, Serialize)]
pub struct EvidenceItem {
    pub id: String,
    pub hash: String,
    pub grain_type: String,
    pub text: String,
}

/// The request envelope. `op` selects the stage; `instructions` is a fixed
/// engine string kept in its own field so it never interleaves with evidence.
#[derive(Debug, Clone, Serialize)]
pub struct LlmRequest<'a> {
    #[serde(rename = "loop")]
    pub loop_proto: u8,
    pub op: &'a str,
    pub instructions: &'a str,
    #[serde(skip_serializing_if = "Vec::is_empty")]
    pub findings: Vec<FindingBrief>,
    #[serde(skip_serializing_if = "Vec::is_empty")]
    pub evidence: Vec<EvidenceItem>,
    /// The operator's recent decisions — what they reject/approve — so the
    /// model learns this reviewer's taste. (Bounded by the engine.)
    #[serde(skip_serializing_if = "Vec::is_empty")]
    pub rejected: Vec<String>,
    #[serde(skip_serializing_if = "Vec::is_empty")]
    pub approved: Vec<String>,
}

// ---- wire schema (response) ------------------------------------------------

/// One DISCOVER draft as returned by the model. Unknown fields are dropped by
/// serde; the engine further validates (cite-check, caps, target class,
/// grounding, and independent verification before it is ever stored).
#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct LlmDraft {
    pub summary: String,
    pub target: String,
    pub guidance: String,
    pub evidence: Vec<String>,
    /// The model's self-reported confidence 0.0–1.0 that this finding is both
    /// correct and materially useful (§5.1). Missing/garbled → 0.0 (rejected by
    /// the confidence floor), a safe default.
    pub confidence: f64,
    /// Optional authored lesson: one imperative rule the model proposes to
    /// record as a Fact grain. Empty (the default) keeps the draft advisory.
    /// A non-empty lesson makes the surviving recommendation *applicable* —
    /// through human review + apply only, never auto-apply — and the lesson
    /// text is folded into the GROUND claim and VERIFY summary so both gates
    /// judge exactly what an apply would write.
    pub lesson: String,
    /// The generalized proposal vocabulary (§9.1): what change this draft asks
    /// a reviewer to make. Held as raw JSON so one draft naming an unknown or
    /// malformed `kind` degrades to advisory instead of dropping the whole
    /// response — read it through [`LlmDraft::parsed_proposal`].
    pub proposal: Option<Value>,
}

/// One field-level edit to a Workflow plan grain. `from` is a staleness check
/// (it must equal what the live plan holds at `path`), which is what stops a
/// proposal authored against a superseded plan from applying to a newer one —
/// the role `base_digest` plays on [`super::recommendation::Proposal::Edit`].
#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
#[serde(default)]
pub struct PlanEdit {
    /// A dotted path into the plan body, e.g. `edges.2.max_cycles`. The
    /// engine's allowlist decides which paths are editable at all.
    pub path: String,
    pub from: Value,
    pub to: Value,
}

/// What a DISCOVER draft proposes to change. Closed vocabulary: every variant
/// maps onto an apply path that already records an inverse, and anything the
/// model returns outside it leaves the draft advisory.
///
/// Note what each variant does NOT carry. The subject of a `Fact`, the name of
/// a `QueryRevision`, the hash of a `PlanRevision` and the tool of a
/// `CodeRevision` all come from the draft's `target`, and the evalset a
/// `CodeRevision` is gated against comes from the substrate — so the model
/// names the change but never names its own scope or its own grader.
#[derive(Debug, Clone, Deserialize, PartialEq)]
#[serde(tag = "kind", rename_all = "snake_case")]
pub enum DraftProposal {
    /// One imperative line recorded as a Fact with `relation = "lesson"`.
    /// The pre-vocabulary shape, and still the default one.
    Lesson {
        #[serde(default)]
        lesson: String,
    },
    /// A durable fact under a model-chosen relation — the "stop making a
    /// person re-supply this every time" proposal.
    Fact {
        #[serde(default)]
        relation: String,
        #[serde(default)]
        object: String,
    },
    /// A rewrite of the saved CAL query or template named by the target: the
    /// agent changing how it assembles its own context.
    QueryRevision {
        #[serde(default)]
        body: String,
    },
    /// Field-level edits to the Workflow plan named by the target. Node
    /// topology is not expressible here by construction — only the paths the
    /// engine's allowlist admits.
    PlanRevision {
        #[serde(default)]
        edits: Vec<PlanEdit>,
    },
    /// New source for the executable tool named by the target. Applies only
    /// through §7.4's recorded evalset-run edge (Rule E1).
    CodeRevision {
        #[serde(default)]
        source: String,
    },
    /// A reusable procedure — a Skill grain — derived from a trajectory that
    /// succeeded: what it does, when to reach for it, and the ordered steps.
    /// The skill's NAME comes from the target (`entity:<ns>/<name>`), like a
    /// fact's subject; when a live skill of that name exists the proposal
    /// supersedes it rather than adding a near-duplicate (PAST-Bench's own
    /// analysis names "splits into near-duplicate notes" as the procedural
    /// failure mode). Offered only when `Policy::skills.enabled`.
    Skill {
        #[serde(default)]
        description: String,
        #[serde(default)]
        when_to_use: String,
        #[serde(default)]
        steps: Vec<String>,
    },
    /// A reusable procedure as a PLAN: named steps, each bound to a tool the
    /// evidence shows was called, and edges with conditions in the runtime's
    /// frozen grammar. Applies as a Workflow grain (the structure the runtime
    /// validates and can execute) plus a Skill grain of the same name (the
    /// prose a model reads), in one batch; a live pair of that name is
    /// superseded. Offered only when `Policy::plans.enabled`.
    Plan {
        #[serde(default)]
        description: String,
        #[serde(default)]
        when_to_use: String,
        #[serde(default)]
        nodes: Vec<PlanNodeDraft>,
        #[serde(default)]
        edges: Vec<PlanEdgeDraft>,
    },
}

/// One step of a `plan` draft: an identifier, the tool it calls, and what it
/// does with it.
#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
#[serde(default)]
pub struct PlanNodeDraft {
    pub id: String,
    pub tool: String,
    pub step: String,
}

/// One edge of a `plan` draft. `cond` is in the runtime's frozen grammar
/// (`path == literal`, `path != literal`, `path exists`, `!path`).
#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
#[serde(default)]
pub struct PlanEdgeDraft {
    pub src: String,
    pub dst: String,
    pub cond: Option<String>,
    pub max_cycles: Option<u32>,
}

impl LlmDraft {
    /// The proposal this draft makes, or `None` when it is advisory.
    ///
    /// An explicit `proposal` decides on its own: if it names an unknown kind
    /// or fails to parse, the draft is advisory rather than being quietly
    /// re-read as something the model did not ask for. `lesson` is the
    /// fallback only when no `proposal` was sent at all, which is what keeps
    /// every transcript recorded before the vocabulary existed parsing — and
    /// therefore keeps the published runs comparable.
    pub fn parsed_proposal(&self) -> Option<DraftProposal> {
        if let Some(v) = &self.proposal {
            return serde_json::from_value::<DraftProposal>(v.clone()).ok();
        }
        if self.lesson.trim().is_empty() {
            None
        } else {
            Some(DraftProposal::Lesson {
                lesson: self.lesson.clone(),
            })
        }
    }
}

/// The DISCOVER response.
#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct DiscoverResponse {
    pub recommendations: Vec<LlmDraft>,
}

/// The ENRICH response: guidance keyed by target_ref of a deterministic rec.
#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct EnrichResponse {
    /// `[{ "target": "...", "guidance": "..." }]`
    pub notes: Vec<EnrichNote>,
}

#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct EnrichNote {
    pub target: String,
    pub guidance: String,
}

// ---- verifier stages (§5.2 GROUND, §5.3 VERIFY) ----------------------------

/// GROUND request: for each candidate draft, does its cited evidence actually
/// *entail* the claim? Decompose-then-entail is asked of the model here; a
/// stronger deployment can swap a dedicated entailment checker behind the same
/// shape. Kept a separate op/call from DISCOVER (proposer ≠ grounder).
#[derive(Debug, Clone, Serialize)]
pub struct GroundRequest<'a> {
    #[serde(rename = "loop")]
    pub loop_proto: u8,
    pub op: &'a str, // "ground"
    pub instructions: &'a str,
    pub claims: Vec<GroundItem>,
}

#[derive(Debug, Clone, Serialize)]
pub struct GroundItem {
    pub id: usize,
    pub claim: String,
    pub evidence: Vec<EvidenceItem>,
}

#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct GroundResponse {
    pub results: Vec<GroundResult>,
}

#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct GroundResult {
    pub id: usize,
    pub supported: bool,
    pub reason: String,
}

/// VERIFY request: an **independent** adversarial pass (a separate call from the
/// proposer — the anti-Goodhart rule) that tries to refute each grounded draft
/// on novelty / reality / out-of-context grounds and returns keep/kill + a
/// calibrated confidence. Deterministic findings are passed as context so the
/// verifier can reject drafts that merely restate them.
#[derive(Debug, Clone, Serialize)]
pub struct VerifyRequest<'a> {
    #[serde(rename = "loop")]
    pub loop_proto: u8,
    pub op: &'a str, // "verify"
    pub instructions: &'a str,
    pub findings: Vec<VerifyItem>,
}

#[derive(Debug, Clone, Serialize)]
pub struct VerifyItem {
    pub id: usize,
    pub summary: String,
    pub target: String,
    pub evidence: Vec<EvidenceItem>,
}

#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct VerifyResponse {
    pub results: Vec<VerifyResult>,
}

#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
pub struct VerifyResult {
    pub id: usize,
    pub keep: bool,
    pub confidence: f64,
    pub reason: String,
}

/// The probe response.
#[derive(Debug, Clone, Deserialize, Default)]
#[serde(default)]
struct ProbeResponse {
    model: String,
}

/// A subprocess LLM backend. One process per call; argv is whitespace-split
/// with no shell (identical rules to `CommandEmbed`).
pub struct CommandLlm {
    argv: Vec<String>,
    model: String,
}

impl CommandLlm {
    /// Construct and probe. The probe (`{"loop":1,"op":"probe"}`) must return
    /// JSON with a `model` (or one is supplied), so a misconfigured command
    /// fails at construction, not mid-run.
    pub fn new(cmd: &str, model: Option<&str>) -> Result<Self> {
        let argv: Vec<String> = cmd.split_whitespace().map(str::to_string).collect();
        if argv.is_empty() {
            return Err(Error::LlmBackend("--llm-cmd is empty".into()));
        }
        let mut me = CommandLlm {
            argv,
            model: model.unwrap_or("").to_string(),
        };
        let probe = me.run(r#"{"loop":1,"op":"probe"}"#)?;
        let parsed: ProbeResponse = serde_json::from_str(probe.trim()).map_err(|e| {
            Error::LlmBackend(format!("--llm-cmd probe did not return JSON with a model: {e}"))
        })?;
        if me.model.is_empty() {
            me.model = if parsed.model.is_empty() {
                "unspecified".to_string()
            } else {
                parsed.model
            };
        }
        Ok(me)
    }

    fn run(&self, request: &str) -> Result<String> {
        let out = crate::proc::run_argv(&self.argv, request, Some(crate::proc::DEFAULT_TIMEOUT))
            .map_err(|e| Error::LlmBackend(format!("spawn --llm-cmd {:?}: {e}", self.argv[0])))?;
        if let Some(why) = out.failure("--llm-cmd") {
            return Err(Error::LlmBackend(why));
        }
        String::from_utf8(out.stdout)
            .map_err(|e| Error::LlmBackend(format!("--llm-cmd stdout not UTF-8: {e}")))
    }
}

impl LlmBackend for CommandLlm {
    fn model(&self) -> &str {
        &self.model
    }
    fn complete(&self, request: &str) -> Result<String> {
        self.run(request)
    }
}

/// Parse a DISCOVER response, dropping anything malformed. Never errors on
/// model garbage — a bad response yields no drafts.
pub fn parse_discover(raw: &str) -> DiscoverResponse {
    serde_json::from_str(raw.trim()).unwrap_or_default()
}

/// Parse an ENRICH response, dropping anything malformed.
pub fn parse_enrich(raw: &str) -> EnrichResponse {
    serde_json::from_str(raw.trim()).unwrap_or_default()
}

/// Parse a GROUND response; garbage → no results (⇒ every draft is treated as
/// ungrounded and dropped, the safe default).
pub fn parse_ground(raw: &str) -> GroundResponse {
    serde_json::from_str(raw.trim()).unwrap_or_default()
}

/// Parse a VERIFY response; garbage → no results (⇒ every draft is dropped).
pub fn parse_verify(raw: &str) -> VerifyResponse {
    serde_json::from_str(raw.trim()).unwrap_or_default()
}

/// Truncate to a char cap without splitting a UTF-8 boundary.
pub fn cap(s: &str, max: usize) -> String {
    if s.chars().count() <= max {
        s.to_string()
    } else {
        s.chars().take(max).collect()
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn parse_discover_drops_garbage() {
        assert!(parse_discover("not json").recommendations.is_empty());
        let r = parse_discover(r#"{"recommendations":[{"summary":"s","target":"entity:x/y","evidence":["h1"],"junk":1}]}"#);
        assert_eq!(r.recommendations.len(), 1);
        assert_eq!(r.recommendations[0].summary, "s");
        assert_eq!(r.recommendations[0].evidence, vec!["h1"]);
    }

    #[test]
    fn parse_enrich_reads_notes() {
        let r = parse_enrich(r#"{"notes":[{"target":"entity:a/b","guidance":"g"}]}"#);
        assert_eq!(r.notes.len(), 1);
        assert_eq!(r.notes[0].guidance, "g");
    }

    #[test]
    fn lesson_desugars_when_no_proposal_is_sent() {
        // Every transcript recorded before the vocabulary existed must still
        // resolve to the same change, or the published runs stop being
        // comparable with anything measured after it.
        let r = parse_discover(
            r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["h"],"lesson":"Do the thing"}]}"#,
        );
        assert_eq!(
            r.recommendations[0].parsed_proposal(),
            Some(DraftProposal::Lesson { lesson: "Do the thing".into() })
        );
    }

    #[test]
    fn an_explicit_proposal_wins_over_lesson() {
        let r = parse_discover(
            r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["h"],
                "lesson":"ignored","proposal":{"kind":"fact","relation":"alias_of","object":"Cobalt Cloud"}}]}"#,
        );
        assert_eq!(
            r.recommendations[0].parsed_proposal(),
            Some(DraftProposal::Fact {
                relation: "alias_of".into(),
                object: "Cobalt Cloud".into()
            })
        );
    }

    #[test]
    fn an_unparseable_proposal_is_advisory_not_reinterpreted() {
        // A garbled proposal must NOT fall back to the lesson: applying an
        // `ADD fact` when the model asked for something we could not read is
        // doing something it never proposed.
        for body in [
            r#""proposal":{"kind":"teleport","x":1}"#,
            r#""proposal":{"kind":"plan_revision","edits":"not-a-list"}"#,
            r#""proposal":42"#,
        ] {
            let raw = format!(
                r#"{{"recommendations":[{{"summary":"s","target":"entity:a/b","evidence":["h"],"lesson":"L",{body}}}]}}"#
            );
            let r = parse_discover(&raw);
            assert_eq!(r.recommendations.len(), 1, "{body}");
            assert_eq!(r.recommendations[0].parsed_proposal(), None, "{body}");
        }
    }

    #[test]
    fn one_bad_proposal_does_not_drop_its_siblings() {
        // Per-draft tolerance: the whole response surviving is what keeps a
        // single malformed kind from silently costing a run its findings.
        let r = parse_discover(
            r#"{"recommendations":[
                {"summary":"a","target":"entity:a/b","evidence":["h"],"proposal":{"kind":"nope"}},
                {"summary":"b","target":"entity:a/c","evidence":["h"],"proposal":{"kind":"lesson","lesson":"Keep me"}}
            ]}"#,
        );
        assert_eq!(r.recommendations.len(), 2);
        assert_eq!(r.recommendations[0].parsed_proposal(), None);
        assert_eq!(
            r.recommendations[1].parsed_proposal(),
            Some(DraftProposal::Lesson { lesson: "Keep me".into() })
        );
    }

    #[test]
    fn plan_edits_carry_their_staleness_check() {
        let r = parse_discover(
            r#"{"recommendations":[{"summary":"s","target":"grain:abc","evidence":["h"],
                "proposal":{"kind":"plan_revision","edits":[{"path":"retries.fetch","from":null,"to":3}]}}]}"#,
        );
        let Some(DraftProposal::PlanRevision { edits }) =
            r.recommendations[0].parsed_proposal()
        else {
            panic!("expected a plan revision");
        };
        assert_eq!(edits.len(), 1);
        assert_eq!(edits[0].path, "retries.fetch");
        assert_eq!(edits[0].from, Value::Null);
        assert_eq!(edits[0].to, Value::from(3));
    }

    #[test]
    fn cap_respects_char_boundaries() {
        assert_eq!(cap("hello", 3), "hel");
        assert_eq!(cap("héllo", 2), "hé");
        assert_eq!(cap("hi", 5), "hi");
    }
    #[test]
    fn skill_proposal_parses_and_missing_fields_default_empty() {
        let raw = r#"{"recommendations":[{"summary":"s","target":"entity:ops/triage-batch","evidence":["e1"],
            "proposal":{"kind":"skill","description":"Triage an open ticket batch","when_to_use":"a batch of open helpdesk tickets arrives",
            "steps":["List open tickets with helpdesk_list_tickets","Fetch each with helpdesk_get_ticket","Update priority and tags"]}}]}"#;
        let d = &parse_discover(raw).recommendations[0];
        match d.parsed_proposal() {
            Some(DraftProposal::Skill { description, when_to_use, steps }) => {
                assert_eq!(description, "Triage an open ticket batch");
                assert!(when_to_use.starts_with("a batch"));
                assert_eq!(steps.len(), 3);
            }
            other => panic!("expected a skill, got {other:?}"),
        }
        let d = &parse_discover(r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["e1"],"proposal":{"kind":"skill"}}]}"#)
            .recommendations[0];
        assert!(
            matches!(d.parsed_proposal(), Some(DraftProposal::Skill { steps, .. }) if steps.is_empty()),
            "a bare skill parses to empties; the engine, not the parser, decides it is too thin"
        );
    }

}