Skip to main content

areev_loop/
llm.rs

1//! Optional LLM enrichment (proposal §9).
2//!
3//! The engine's deterministic output stays a pure function of `(store, params,
4//! now)`. This layer is strictly **additive**: with a backend attached the
5//! pipeline gains two optional stages —
6//!
7//! ```text
8//! ANALYZE (deterministic) → DISCOVER (LLM) → ENRICH (LLM) → VALIDATE+DEDUP → STORE
9//! ```
10//!
11//! and with no backend those stages are the identity function, so the no-LLM
12//! path is byte-for-byte the deterministic path. The LLM can only:
13//!   - **DISCOVER**: propose *new* draft recommendations, which enter through
14//!     the ordinary candidate/dedup/store path stamped `origin = llm` — so they
15//!     can **never auto-apply** and never target prompt/host surfaces. A draft
16//!     MAY author a `lesson` (one capped imperative line); a lesson-bearing
17//!     draft that survives GROUND + VERIFY stamps as an *applicable*,
18//!     rollbackable `ADD fact` proposal instead of an advisory flag — still
19//!     `origin = llm`, so applying it always takes a human review with a
20//!     BECAUSE plus an explicit apply; and
21//!   - **ENRICH**: add a whitelisted `guidance` note to a deterministic
22//!     recommendation. The engine-templated summary is always kept; the model
23//!     never rewrites it.
24//!
25//! Trust floor (enforced by the engine, not the backend): responses are parsed
26//! to a fixed schema (unknown fields dropped, strings capped), DISCOVER drafts
27//! must cite evidence present in the bundle (by bundle id, full hash, or an
28//! unambiguous hash prefix), instructions never interleave with evidence, and
29//! a failed/timed-out/garbled call drops the LLM contribution for the run
30//! rather than failing it.
31//!
32//! `CommandLlm` mirrors the shipped `CommandEmbed`: whitespace-split argv (no
33//! shell), one process per call, a JSON request on stdin and a JSON response on
34//! stdout, and a construction-time probe that fails loud.
35
36use crate::error::{Error, Result};
37use serde::{Deserialize, Serialize};
38use serde_json::Value;
39
40/// Caps that bound what a single LLM contribution can inject (defense in depth;
41/// the engine enforces them after parsing).
42pub const MAX_LLM_DRAFTS: usize = 8;
43pub const MAX_GUIDANCE_LEN: usize = 600;
44pub const MAX_SUMMARY_LEN: usize = 200;
45/// An authored lesson is one imperative line — anything longer is a document,
46/// not a lesson, and a bound on what a single approved apply can put into
47/// every future prompt.
48pub const MAX_LESSON_LEN: usize = 240;
49/// Caps on the rest of the proposal vocabulary. Each one bounds what a single
50/// approved apply can put into every future run of the agent, so they are part
51/// of the trust floor rather than tuning knobs.
52pub const MAX_RELATION_LEN: usize = 64;
53pub const MAX_OBJECT_LEN: usize = 480;
54/// A skill step is one instruction line; a skill has at most this many.
55pub const MAX_SKILL_STEP_LEN: usize = 240;
56pub const MAX_SKILL_STEPS: usize = 20;
57/// A skill name is an identifier a person types at `skill_view`, not a title.
58pub const MAX_SKILL_NAME_LEN: usize = 64;
59/// A plan has at most this many steps; a condition is one short line.
60pub const MAX_PLAN_NODES: usize = 20;
61pub const MAX_COND_LEN: usize = 200;
62pub const MAX_QUERY_BODY_LEN: usize = 2_000;
63pub const MAX_PLAN_EDITS: usize = 8;
64pub const MAX_CODE_LEN: usize = 20_000;
65
66/// A backend that answers one JSON request with one JSON response. Object-safe
67/// so the engine can hold a `Box<dyn LlmBackend>`.
68pub trait LlmBackend: Send + Sync {
69    /// Model identifier, stamped as provenance on `origin = llm` grains.
70    fn model(&self) -> &str;
71    /// Run one request. `request` is a JSON string; the returned text is
72    /// expected to be JSON and is validated by the caller.
73    fn complete(&self, request: &str) -> Result<String>;
74}
75
76/// Boxed backends forward — lets decorators wrap `Box<dyn LlmBackend>`
77/// without knowing the concrete type.
78impl<T: LlmBackend + ?Sized> LlmBackend for Box<T> {
79    fn model(&self) -> &str {
80        (**self).model()
81    }
82    fn complete(&self, request: &str) -> Result<String> {
83        (**self).complete(request)
84    }
85}
86
87// ---- wire schema (request) -------------------------------------------------
88
89/// One deterministic finding, handed to DISCOVER as context (never as an
90/// instruction — see `LlmRequest`).
91#[derive(Debug, Clone, Serialize)]
92pub struct FindingBrief {
93    pub analyzer: String,
94    pub summary: String,
95    pub target: String,
96    pub severity: String,
97}
98
99/// One evidence grain, provenance-tagged. `id` is the bundle-local label
100/// (`e1`, `e2`, …) a draft may cite instead of the 64-hex `hash`: the engine
101/// resolves either back to the grain, so a citation is still checked against
102/// the bundle — it is just no longer a transcription test for the model.
103#[derive(Debug, Clone, Serialize)]
104pub struct EvidenceItem {
105    pub id: String,
106    pub hash: String,
107    pub grain_type: String,
108    pub text: String,
109}
110
111/// The request envelope. `op` selects the stage; `instructions` is a fixed
112/// engine string kept in its own field so it never interleaves with evidence.
113#[derive(Debug, Clone, Serialize)]
114pub struct LlmRequest<'a> {
115    #[serde(rename = "loop")]
116    pub loop_proto: u8,
117    pub op: &'a str,
118    pub instructions: &'a str,
119    #[serde(skip_serializing_if = "Vec::is_empty")]
120    pub findings: Vec<FindingBrief>,
121    #[serde(skip_serializing_if = "Vec::is_empty")]
122    pub evidence: Vec<EvidenceItem>,
123    /// The operator's recent decisions — what they reject/approve — so the
124    /// model learns this reviewer's taste. (Bounded by the engine.)
125    #[serde(skip_serializing_if = "Vec::is_empty")]
126    pub rejected: Vec<String>,
127    #[serde(skip_serializing_if = "Vec::is_empty")]
128    pub approved: Vec<String>,
129}
130
131// ---- wire schema (response) ------------------------------------------------
132
133/// One DISCOVER draft as returned by the model. Unknown fields are dropped by
134/// serde; the engine further validates (cite-check, caps, target class,
135/// grounding, and independent verification before it is ever stored).
136#[derive(Debug, Clone, Deserialize, Default)]
137#[serde(default)]
138pub struct LlmDraft {
139    pub summary: String,
140    pub target: String,
141    pub guidance: String,
142    pub evidence: Vec<String>,
143    /// The model's self-reported confidence 0.0–1.0 that this finding is both
144    /// correct and materially useful (§5.1). Missing/garbled → 0.0 (rejected by
145    /// the confidence floor), a safe default.
146    pub confidence: f64,
147    /// Optional authored lesson: one imperative rule the model proposes to
148    /// record as a Fact grain. Empty (the default) keeps the draft advisory.
149    /// A non-empty lesson makes the surviving recommendation *applicable* —
150    /// through human review + apply only, never auto-apply — and the lesson
151    /// text is folded into the GROUND claim and VERIFY summary so both gates
152    /// judge exactly what an apply would write.
153    pub lesson: String,
154    /// The generalized proposal vocabulary (§9.1): what change this draft asks
155    /// a reviewer to make. Held as raw JSON so one draft naming an unknown or
156    /// malformed `kind` degrades to advisory instead of dropping the whole
157    /// response — read it through [`LlmDraft::parsed_proposal`].
158    pub proposal: Option<Value>,
159}
160
161/// One field-level edit to a Workflow plan grain. `from` is a staleness check
162/// (it must equal what the live plan holds at `path`), which is what stops a
163/// proposal authored against a superseded plan from applying to a newer one —
164/// the role `base_digest` plays on [`super::recommendation::Proposal::Edit`].
165#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
166#[serde(default)]
167pub struct PlanEdit {
168    /// A dotted path into the plan body, e.g. `edges.2.max_cycles`. The
169    /// engine's allowlist decides which paths are editable at all.
170    pub path: String,
171    pub from: Value,
172    pub to: Value,
173}
174
175/// What a DISCOVER draft proposes to change. Closed vocabulary: every variant
176/// maps onto an apply path that already records an inverse, and anything the
177/// model returns outside it leaves the draft advisory.
178///
179/// Note what each variant does NOT carry. The subject of a `Fact`, the name of
180/// a `QueryRevision`, the hash of a `PlanRevision` and the tool of a
181/// `CodeRevision` all come from the draft's `target`, and the evalset a
182/// `CodeRevision` is gated against comes from the substrate — so the model
183/// names the change but never names its own scope or its own grader.
184#[derive(Debug, Clone, Deserialize, PartialEq)]
185#[serde(tag = "kind", rename_all = "snake_case")]
186pub enum DraftProposal {
187    /// One imperative line recorded as a Fact with `relation = "lesson"`.
188    /// The pre-vocabulary shape, and still the default one.
189    Lesson {
190        #[serde(default)]
191        lesson: String,
192    },
193    /// A durable fact under a model-chosen relation — the "stop making a
194    /// person re-supply this every time" proposal.
195    Fact {
196        #[serde(default)]
197        relation: String,
198        #[serde(default)]
199        object: String,
200    },
201    /// A rewrite of the saved CAL query or template named by the target: the
202    /// agent changing how it assembles its own context.
203    QueryRevision {
204        #[serde(default)]
205        body: String,
206    },
207    /// Field-level edits to the Workflow plan named by the target. Node
208    /// topology is not expressible here by construction — only the paths the
209    /// engine's allowlist admits.
210    PlanRevision {
211        #[serde(default)]
212        edits: Vec<PlanEdit>,
213    },
214    /// New source for the executable tool named by the target. Applies only
215    /// through §7.4's recorded evalset-run edge (Rule E1).
216    CodeRevision {
217        #[serde(default)]
218        source: String,
219    },
220    /// A reusable procedure — a Skill grain — derived from a trajectory that
221    /// succeeded: what it does, when to reach for it, and the ordered steps.
222    /// The skill's NAME comes from the target (`entity:<ns>/<name>`), like a
223    /// fact's subject; when a live skill of that name exists the proposal
224    /// supersedes it rather than adding a near-duplicate (PAST-Bench's own
225    /// analysis names "splits into near-duplicate notes" as the procedural
226    /// failure mode). Offered only when `Policy::skills.enabled`.
227    Skill {
228        #[serde(default)]
229        description: String,
230        #[serde(default)]
231        when_to_use: String,
232        #[serde(default)]
233        steps: Vec<String>,
234    },
235    /// One lesson replacing a PILE of live lessons on the same entity — the
236    /// answer to a `lesson_pile` finding. `supersedes` must be exactly the
237    /// live lesson hashes that finding listed; the apply supersedes each and
238    /// adds the one line, and a rollback restores every member.
239    Consolidation {
240        #[serde(default)]
241        lesson: String,
242        #[serde(default)]
243        supersedes: Vec<String>,
244    },
245    /// A reusable procedure as a PLAN: named steps, each bound to a tool the
246    /// evidence shows was called, and edges with conditions in the runtime's
247    /// frozen grammar. Applies as a Workflow grain (the structure the runtime
248    /// validates and can execute) plus a Skill grain of the same name (the
249    /// prose a model reads), in one batch; a live pair of that name is
250    /// superseded. Offered only when `Policy::plans.enabled`.
251    Plan {
252        #[serde(default)]
253        description: String,
254        #[serde(default)]
255        when_to_use: String,
256        #[serde(default)]
257        nodes: Vec<PlanNodeDraft>,
258        #[serde(default)]
259        edges: Vec<PlanEdgeDraft>,
260    },
261}
262
263/// One step of a `plan` draft: an identifier, the tool it calls, and what it
264/// does with it.
265#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
266#[serde(default)]
267pub struct PlanNodeDraft {
268    pub id: String,
269    pub tool: String,
270    pub step: String,
271}
272
273/// One edge of a `plan` draft. `cond` is in the runtime's frozen grammar
274/// (`path == literal`, `path != literal`, `path exists`, `!path`).
275#[derive(Debug, Clone, Deserialize, Default, PartialEq)]
276#[serde(default)]
277pub struct PlanEdgeDraft {
278    pub src: String,
279    pub dst: String,
280    pub cond: Option<String>,
281    pub max_cycles: Option<u32>,
282}
283
284impl LlmDraft {
285    /// The proposal this draft makes, or `None` when it is advisory.
286    ///
287    /// An explicit `proposal` decides on its own: if it names an unknown kind
288    /// or fails to parse, the draft is advisory rather than being quietly
289    /// re-read as something the model did not ask for. `lesson` is the
290    /// fallback only when no `proposal` was sent at all, which is what keeps
291    /// every transcript recorded before the vocabulary existed parsing — and
292    /// therefore keeps the published runs comparable.
293    pub fn parsed_proposal(&self) -> Option<DraftProposal> {
294        if let Some(v) = &self.proposal {
295            return serde_json::from_value::<DraftProposal>(v.clone()).ok();
296        }
297        if self.lesson.trim().is_empty() {
298            None
299        } else {
300            Some(DraftProposal::Lesson {
301                lesson: self.lesson.clone(),
302            })
303        }
304    }
305}
306
307/// The DISCOVER response.
308#[derive(Debug, Clone, Deserialize, Default)]
309#[serde(default)]
310pub struct DiscoverResponse {
311    pub recommendations: Vec<LlmDraft>,
312}
313
314/// The ENRICH response: guidance keyed by target_ref of a deterministic rec.
315#[derive(Debug, Clone, Deserialize, Default)]
316#[serde(default)]
317pub struct EnrichResponse {
318    /// `[{ "target": "...", "guidance": "..." }]`
319    pub notes: Vec<EnrichNote>,
320}
321
322#[derive(Debug, Clone, Deserialize, Default)]
323#[serde(default)]
324pub struct EnrichNote {
325    pub target: String,
326    pub guidance: String,
327}
328
329// ---- verifier stages (§5.2 GROUND, §5.3 VERIFY) ----------------------------
330
331/// GROUND request: for each candidate draft, does its cited evidence actually
332/// *entail* the claim? Decompose-then-entail is asked of the model here; a
333/// stronger deployment can swap a dedicated entailment checker behind the same
334/// shape. Kept a separate op/call from DISCOVER (proposer ≠ grounder).
335#[derive(Debug, Clone, Serialize)]
336pub struct GroundRequest<'a> {
337    #[serde(rename = "loop")]
338    pub loop_proto: u8,
339    pub op: &'a str, // "ground"
340    pub instructions: &'a str,
341    pub claims: Vec<GroundItem>,
342}
343
344#[derive(Debug, Clone, Serialize)]
345pub struct GroundItem {
346    pub id: usize,
347    pub claim: String,
348    pub evidence: Vec<EvidenceItem>,
349}
350
351#[derive(Debug, Clone, Deserialize, Default)]
352#[serde(default)]
353pub struct GroundResponse {
354    pub results: Vec<GroundResult>,
355}
356
357#[derive(Debug, Clone, Deserialize, Default)]
358#[serde(default)]
359pub struct GroundResult {
360    pub id: usize,
361    pub supported: bool,
362    pub reason: String,
363}
364
365/// VERIFY request: an **independent** adversarial pass (a separate call from the
366/// proposer — the anti-Goodhart rule) that tries to refute each grounded draft
367/// on novelty / reality / out-of-context grounds and returns keep/kill + a
368/// calibrated confidence. Deterministic findings are passed as context so the
369/// verifier can reject drafts that merely restate them.
370#[derive(Debug, Clone, Serialize)]
371pub struct VerifyRequest<'a> {
372    #[serde(rename = "loop")]
373    pub loop_proto: u8,
374    pub op: &'a str, // "verify"
375    pub instructions: &'a str,
376    pub findings: Vec<VerifyItem>,
377}
378
379#[derive(Debug, Clone, Serialize)]
380pub struct VerifyItem {
381    pub id: usize,
382    pub summary: String,
383    pub target: String,
384    pub evidence: Vec<EvidenceItem>,
385}
386
387#[derive(Debug, Clone, Deserialize, Default)]
388#[serde(default)]
389pub struct VerifyResponse {
390    pub results: Vec<VerifyResult>,
391}
392
393#[derive(Debug, Clone, Deserialize, Default)]
394#[serde(default)]
395pub struct VerifyResult {
396    pub id: usize,
397    pub keep: bool,
398    pub confidence: f64,
399    pub reason: String,
400}
401
402/// The probe response.
403#[derive(Debug, Clone, Deserialize, Default)]
404#[serde(default)]
405struct ProbeResponse {
406    model: String,
407}
408
409/// A subprocess LLM backend. One process per call; argv is whitespace-split
410/// with no shell (identical rules to `CommandEmbed`).
411pub struct CommandLlm {
412    argv: Vec<String>,
413    model: String,
414}
415
416impl CommandLlm {
417    /// Construct and probe. The probe (`{"loop":1,"op":"probe"}`) must return
418    /// JSON with a `model` (or one is supplied), so a misconfigured command
419    /// fails at construction, not mid-run.
420    pub fn new(cmd: &str, model: Option<&str>) -> Result<Self> {
421        let argv: Vec<String> = cmd.split_whitespace().map(str::to_string).collect();
422        if argv.is_empty() {
423            return Err(Error::LlmBackend("--llm-cmd is empty".into()));
424        }
425        let mut me = CommandLlm {
426            argv,
427            model: model.unwrap_or("").to_string(),
428        };
429        let probe = me.run(r#"{"loop":1,"op":"probe"}"#)?;
430        let parsed: ProbeResponse = serde_json::from_str(probe.trim()).map_err(|e| {
431            Error::LlmBackend(format!("--llm-cmd probe did not return JSON with a model: {e}"))
432        })?;
433        if me.model.is_empty() {
434            me.model = if parsed.model.is_empty() {
435                "unspecified".to_string()
436            } else {
437                parsed.model
438            };
439        }
440        Ok(me)
441    }
442
443    fn run(&self, request: &str) -> Result<String> {
444        let out = crate::proc::run_argv(&self.argv, request, Some(crate::proc::DEFAULT_TIMEOUT))
445            .map_err(|e| Error::LlmBackend(format!("spawn --llm-cmd {:?}: {e}", self.argv[0])))?;
446        if let Some(why) = out.failure("--llm-cmd") {
447            return Err(Error::LlmBackend(why));
448        }
449        String::from_utf8(out.stdout)
450            .map_err(|e| Error::LlmBackend(format!("--llm-cmd stdout not UTF-8: {e}")))
451    }
452}
453
454impl LlmBackend for CommandLlm {
455    fn model(&self) -> &str {
456        &self.model
457    }
458    fn complete(&self, request: &str) -> Result<String> {
459        self.run(request)
460    }
461}
462
463/// Parse a DISCOVER response, dropping anything malformed. Never errors on
464/// model garbage — a bad response yields no drafts.
465pub fn parse_discover(raw: &str) -> DiscoverResponse {
466    serde_json::from_str(raw.trim()).unwrap_or_default()
467}
468
469/// Parse an ENRICH response, dropping anything malformed.
470pub fn parse_enrich(raw: &str) -> EnrichResponse {
471    serde_json::from_str(raw.trim()).unwrap_or_default()
472}
473
474/// Parse a GROUND response; garbage → no results (⇒ every draft is treated as
475/// ungrounded and dropped, the safe default).
476pub fn parse_ground(raw: &str) -> GroundResponse {
477    serde_json::from_str(raw.trim()).unwrap_or_default()
478}
479
480/// Parse a VERIFY response; garbage → no results (⇒ every draft is dropped).
481pub fn parse_verify(raw: &str) -> VerifyResponse {
482    serde_json::from_str(raw.trim()).unwrap_or_default()
483}
484
485/// Truncate to a char cap without splitting a UTF-8 boundary.
486pub fn cap(s: &str, max: usize) -> String {
487    if s.chars().count() <= max {
488        s.to_string()
489    } else {
490        s.chars().take(max).collect()
491    }
492}
493
494#[cfg(test)]
495mod tests {
496    use super::*;
497
498    #[test]
499    fn parse_discover_drops_garbage() {
500        assert!(parse_discover("not json").recommendations.is_empty());
501        let r = parse_discover(r#"{"recommendations":[{"summary":"s","target":"entity:x/y","evidence":["h1"],"junk":1}]}"#);
502        assert_eq!(r.recommendations.len(), 1);
503        assert_eq!(r.recommendations[0].summary, "s");
504        assert_eq!(r.recommendations[0].evidence, vec!["h1"]);
505    }
506
507    #[test]
508    fn parse_enrich_reads_notes() {
509        let r = parse_enrich(r#"{"notes":[{"target":"entity:a/b","guidance":"g"}]}"#);
510        assert_eq!(r.notes.len(), 1);
511        assert_eq!(r.notes[0].guidance, "g");
512    }
513
514    #[test]
515    fn lesson_desugars_when_no_proposal_is_sent() {
516        // Every transcript recorded before the vocabulary existed must still
517        // resolve to the same change, or the published runs stop being
518        // comparable with anything measured after it.
519        let r = parse_discover(
520            r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["h"],"lesson":"Do the thing"}]}"#,
521        );
522        assert_eq!(
523            r.recommendations[0].parsed_proposal(),
524            Some(DraftProposal::Lesson { lesson: "Do the thing".into() })
525        );
526    }
527
528    #[test]
529    fn an_explicit_proposal_wins_over_lesson() {
530        let r = parse_discover(
531            r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["h"],
532                "lesson":"ignored","proposal":{"kind":"fact","relation":"alias_of","object":"Cobalt Cloud"}}]}"#,
533        );
534        assert_eq!(
535            r.recommendations[0].parsed_proposal(),
536            Some(DraftProposal::Fact {
537                relation: "alias_of".into(),
538                object: "Cobalt Cloud".into()
539            })
540        );
541    }
542
543    #[test]
544    fn an_unparseable_proposal_is_advisory_not_reinterpreted() {
545        // A garbled proposal must NOT fall back to the lesson: applying an
546        // `ADD fact` when the model asked for something we could not read is
547        // doing something it never proposed.
548        for body in [
549            r#""proposal":{"kind":"teleport","x":1}"#,
550            r#""proposal":{"kind":"plan_revision","edits":"not-a-list"}"#,
551            r#""proposal":42"#,
552        ] {
553            let raw = format!(
554                r#"{{"recommendations":[{{"summary":"s","target":"entity:a/b","evidence":["h"],"lesson":"L",{body}}}]}}"#
555            );
556            let r = parse_discover(&raw);
557            assert_eq!(r.recommendations.len(), 1, "{body}");
558            assert_eq!(r.recommendations[0].parsed_proposal(), None, "{body}");
559        }
560    }
561
562    #[test]
563    fn one_bad_proposal_does_not_drop_its_siblings() {
564        // Per-draft tolerance: the whole response surviving is what keeps a
565        // single malformed kind from silently costing a run its findings.
566        let r = parse_discover(
567            r#"{"recommendations":[
568                {"summary":"a","target":"entity:a/b","evidence":["h"],"proposal":{"kind":"nope"}},
569                {"summary":"b","target":"entity:a/c","evidence":["h"],"proposal":{"kind":"lesson","lesson":"Keep me"}}
570            ]}"#,
571        );
572        assert_eq!(r.recommendations.len(), 2);
573        assert_eq!(r.recommendations[0].parsed_proposal(), None);
574        assert_eq!(
575            r.recommendations[1].parsed_proposal(),
576            Some(DraftProposal::Lesson { lesson: "Keep me".into() })
577        );
578    }
579
580    #[test]
581    fn plan_edits_carry_their_staleness_check() {
582        let r = parse_discover(
583            r#"{"recommendations":[{"summary":"s","target":"grain:abc","evidence":["h"],
584                "proposal":{"kind":"plan_revision","edits":[{"path":"retries.fetch","from":null,"to":3}]}}]}"#,
585        );
586        let Some(DraftProposal::PlanRevision { edits }) =
587            r.recommendations[0].parsed_proposal()
588        else {
589            panic!("expected a plan revision");
590        };
591        assert_eq!(edits.len(), 1);
592        assert_eq!(edits[0].path, "retries.fetch");
593        assert_eq!(edits[0].from, Value::Null);
594        assert_eq!(edits[0].to, Value::from(3));
595    }
596
597    #[test]
598    fn cap_respects_char_boundaries() {
599        assert_eq!(cap("hello", 3), "hel");
600        assert_eq!(cap("héllo", 2), "hé");
601        assert_eq!(cap("hi", 5), "hi");
602    }
603    #[test]
604    fn skill_proposal_parses_and_missing_fields_default_empty() {
605        let raw = r#"{"recommendations":[{"summary":"s","target":"entity:ops/triage-batch","evidence":["e1"],
606            "proposal":{"kind":"skill","description":"Triage an open ticket batch","when_to_use":"a batch of open helpdesk tickets arrives",
607            "steps":["List open tickets with helpdesk_list_tickets","Fetch each with helpdesk_get_ticket","Update priority and tags"]}}]}"#;
608        let d = &parse_discover(raw).recommendations[0];
609        match d.parsed_proposal() {
610            Some(DraftProposal::Skill { description, when_to_use, steps }) => {
611                assert_eq!(description, "Triage an open ticket batch");
612                assert!(when_to_use.starts_with("a batch"));
613                assert_eq!(steps.len(), 3);
614            }
615            other => panic!("expected a skill, got {other:?}"),
616        }
617        let d = &parse_discover(r#"{"recommendations":[{"summary":"s","target":"entity:a/b","evidence":["e1"],"proposal":{"kind":"skill"}}]}"#)
618            .recommendations[0];
619        assert!(
620            matches!(d.parsed_proposal(), Some(DraftProposal::Skill { steps, .. }) if steps.is_empty()),
621            "a bare skill parses to empties; the engine, not the parser, decides it is too thin"
622        );
623    }
624
625}