Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel.
9//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
10
11use crate::error::{Error, Result};
12use crate::model::{normalize_ident, ActionKind, Origin, Severity};
13use crate::substrate::GrainSpec;
14use serde::{Deserialize, Serialize};
15use serde_json::{Map, Value};
16
17/// A deterministic, template-rendered summary. The analyzer chooses a
18/// `template_id` and supplies `args`; the text is produced here, so an
19/// analyzer can never emit arbitrary prose into the queue.
20#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
21pub struct Summary {
22    pub template_id: String,
23    #[serde(default)]
24    pub args: Map<String, Value>,
25}
26
27impl Summary {
28    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
29        Summary {
30            template_id: template_id.into(),
31            args,
32        }
33    }
34
35    /// Render the summary. Unknown template ids fall back to a stable, honest
36    /// string (never a panic) so a manifest/template mismatch is visible, not
37    /// fatal.
38    pub fn render(&self) -> String {
39        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
40        interpolate(t, &self.args, &self.template_id)
41    }
42}
43
44/// The built-in template table. Deterministic; the only place summary prose
45/// lives. Keep placeholders in `{name}` form matching `args` keys.
46fn builtin_template(id: &str) -> Option<&'static str> {
47    Some(match id {
48        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
49        "duplicate.near" => {
50            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
51        }
52        "contradiction.functional" => {
53            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
54        }
55        "tool_failure.cluster" => {
56            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
57             fail this way): {signature}"
58        }
59        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
60        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
61        "skill.stall" => {
62            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
63        }
64        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
65        "cold.grain" => {
66            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
67        }
68        "coverage.gap" => {
69            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
70        }
71        "budget.pressure" => {
72            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
73        }
74        "retention.overdue" => {
75            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
76        }
77        "outcome.regression" => {
78            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
79        }
80        "run.failures" => {
81            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
82        }
83        "run.cost" => {
84            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
85        }
86        "adapter.candidate" => {
87            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
88        }
89        // origin=llm drafts carry free text (clearly marked llm) rather than a
90        // deterministic template — the model's proposed summary rides in {text}.
91        "llm.discover" => "{text}",
92        // An LLM-authored lesson shows the reviewer BOTH the finding and the
93        // exact line an apply would record — never one without the other.
94        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
95        // The rest of the LLM proposal vocabulary follows the same rule: the
96        // finding AND the exact change an apply would make, never one without
97        // the other. A reviewer approving blind is the failure this prevents.
98        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
99        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
100        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
101        "llm.code_revision" => {
102            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
103        }
104        // External command analyzers (trust class Command): free text from the
105        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
106        "command.finding" => "{text}",
107        _ => return None,
108    })
109}
110
111fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
112    let mut out = String::with_capacity(template.len());
113    let mut chars = template.chars().peekable();
114    while let Some(c) = chars.next() {
115        if c == '{' {
116            let mut key = String::new();
117            for k in chars.by_ref() {
118                if k == '}' {
119                    break;
120                }
121                key.push(k);
122            }
123            if key == "template_id" {
124                out.push_str(template_id);
125            } else {
126                match args.get(&key) {
127                    Some(Value::String(s)) => out.push_str(s),
128                    Some(v) => out.push_str(&v.to_string()),
129                    None => {
130                        // Missing arg — leave a visible marker rather than lie.
131                        out.push('{');
132                        out.push_str(&key);
133                        out.push('}');
134                    }
135                }
136            }
137        } else {
138            out.push(c);
139        }
140    }
141    out
142}
143
144/// A reproducible metric snapshot; powers outcome review.
145#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
146pub struct MetricSnapshot {
147    /// Metric kind — the engine knows how to re-measure a fixed set
148    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
149    pub metric: String,
150    pub baseline: f64,
151    pub unit: String,
152    pub n: u64,
153    pub window: String,
154    /// The subject the metric is about (e.g. the tool name), used by the
155    /// engine's typed re-measurement.
156    #[serde(default, skip_serializing_if = "Option::is_none")]
157    pub subject: Option<String>,
158    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
159    /// `None` = all namespaces. Additive: older snapshots deserialize to
160    /// `None` and existing metric kinds ignore it.
161    #[serde(default, skip_serializing_if = "Option::is_none")]
162    pub namespace: Option<String>,
163    /// Relation scope for fact-shaped metrics (e.g. which functional relation
164    /// a contradiction was resolved under), normalized. Additive like
165    /// `namespace`.
166    #[serde(default, skip_serializing_if = "Option::is_none")]
167    pub relation: Option<String>,
168    /// CAL that recomputes the metric at verify time (reproducibility /
169    /// documentation; the engine re-measures with typed reads).
170    pub query: String,
171    /// How long after apply to re-measure, in epoch-ms delta (the first / only
172    /// checkpoint when `horizons_ms` is empty).
173    pub review_after_ms: i64,
174    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
175    /// that `held` at an early checkpoint can `regress` at a later one, so a
176    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
177    #[serde(default, skip_serializing_if = "Vec::is_empty")]
178    pub horizons_ms: Vec<i64>,
179    /// Which direction is an improvement. The built-in metrics are all
180    /// recurrence counts, where lower is better — so this defaults to `false`
181    /// and every existing snapshot deserializes unchanged. An evalset accuracy
182    /// is the opposite, and getting it wrong does not merely misreport: the
183    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
184    /// propose reverting it.
185    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
186    pub higher_is_better: bool,
187}
188
189/// The one regression rule.
190///
191/// The verdict is computed in two places — once by the engine for the recorded
192/// `OutcomeResult`, once by `outcome_review` when it drafts the revert — and
193/// the two silently disagreeing would either revert a change that held or sit
194/// on one that regressed. So both call this.
195pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool) -> bool {
196    /// Minimum worsening to call a regression (avoids noise at n=1).
197    const EPSILON: f64 = 1e-9;
198    if higher_is_better {
199        current < baseline - EPSILON
200    } else {
201        current > baseline + EPSILON
202    }
203}
204
205impl MetricSnapshot {
206    /// The measurement schedule, sorted — `horizons_ms` if set, else the single
207    /// `review_after_ms`.
208    pub fn horizons(&self) -> Vec<i64> {
209        let mut h = if self.horizons_ms.is_empty() {
210            vec![self.review_after_ms]
211        } else {
212            self.horizons_ms.clone()
213        };
214        h.sort_unstable();
215        h
216    }
217}
218
219/// A measured outcome for an applied recommendation at one checkpoint — the
220/// Verify gate's output. `held` = the metric did not regress at this horizon;
221/// `regressed` = it got worse (a revert is proposed). A recommendation
222/// accumulates one of these per horizon, forming a time series.
223#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
224pub struct OutcomeResult {
225    pub rec_hash: String,
226    pub metric: String,
227    pub baseline: f64,
228    pub current: f64,
229    pub verdict: String,
230    /// Which checkpoint this measurement is for (ms after apply).
231    #[serde(default)]
232    pub horizon_ms: i64,
233    pub measured_at_ms: i64,
234}
235
236/// The proposed change. Exactly one variant per recommendation.
237#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
238#[serde(tag = "proposal", rename_all = "snake_case")]
239pub enum Proposal {
240    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
241    Cal { cal: String },
242    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
243    /// staleness check at apply).
244    Edit {
245        format: String,
246        base_digest: String,
247        diff: String,
248    },
249    /// An opaque map for host targets (applied by the host, §12.3).
250    Data { data: Map<String, Value> },
251}
252
253/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
254/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
255/// fields without breaking analyzers.
256#[derive(Debug, Clone, PartialEq)]
257#[non_exhaustive]
258pub struct RecDraft {
259    pub target_ref: String,
260    pub action_kind: ActionKind,
261    pub summary: Summary,
262    pub severity: Severity,
263    pub proposal: Proposal,
264    /// Bounded to ≤64 representative evidence hashes by the engine.
265    pub evidence: Vec<String>,
266    pub evidence_query: Option<String>,
267    pub metric: Option<MetricSnapshot>,
268    pub confidence: f64,
269    pub importance: f64,
270    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
271    pub evalset_hash: Option<String>,
272}
273
274impl RecDraft {
275    pub fn new(
276        target_ref: impl Into<String>,
277        action_kind: ActionKind,
278        summary: Summary,
279        proposal: Proposal,
280    ) -> Self {
281        RecDraft {
282            target_ref: target_ref.into(),
283            action_kind,
284            summary,
285            severity: Severity::Low,
286            proposal,
287            evidence: Vec::new(),
288            evidence_query: None,
289            metric: None,
290            confidence: 0.8,
291            importance: 0.5,
292            evalset_hash: None,
293        }
294    }
295
296    pub fn severity(mut self, s: Severity) -> Self {
297        self.severity = s;
298        self
299    }
300    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
301        self.evidence = hashes;
302        self
303    }
304    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
305        self.evidence_query = Some(q.into());
306        self
307    }
308    pub fn metric(mut self, m: MetricSnapshot) -> Self {
309        self.metric = Some(m);
310        self
311    }
312    pub fn confidence(mut self, c: f64) -> Self {
313        self.confidence = c;
314        self
315    }
316    pub fn importance(mut self, i: f64) -> Self {
317        self.importance = i;
318        self
319    }
320    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
321        self.evalset_hash = Some(h.into());
322        self
323    }
324}
325
326/// Maximum representative evidence hashes carried inline (proposal §7.1).
327pub const MAX_EVIDENCE: usize = 64;
328
329/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
330/// Excludes proposal content and evidence by construction.
331pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
332    // U+001F (unit separator) cannot appear in any of the inputs.
333    format!(
334        "{}\u{1f}{}\u{1f}{}",
335        normalize_ident(family),
336        normalize_ident(target_ref),
337        action.as_str()
338    )
339}
340
341/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
342/// content hash is stable for its whole life).
343#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
344#[serde(rename_all = "snake_case")]
345pub enum RecStatus {
346    #[default]
347    Pending,
348    Approved,
349    Rejected,
350    Applied,
351    RolledBack,
352    Expired,
353}
354
355impl RecStatus {
356    pub fn as_str(&self) -> &'static str {
357        match self {
358            RecStatus::Pending => "pending",
359            RecStatus::Approved => "approved",
360            RecStatus::Rejected => "rejected",
361            RecStatus::Applied => "applied",
362            RecStatus::RolledBack => "rolled_back",
363            RecStatus::Expired => "expired",
364        }
365    }
366
367    /// Is a transition to `to` allowed from this state? `by_policy` marks the
368    /// auto-apply actor, the only one permitted the reasonless
369    /// `pending → applied` jump.
370    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
371        use RecStatus::*;
372        match (self, to) {
373            (Pending, Approved) | (Pending, Rejected) => true,
374            (Pending, Applied) => by_policy, // auto-apply only
375            (Approved, Applied) => true,
376            (Applied, RolledBack) => true,
377            // `expired` is computed from valid_to, applied to still-open recs.
378            (Pending, Expired) | (Approved, Expired) => true,
379            _ => false,
380        }
381    }
382}
383
384/// Who/what performed a transition.
385#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
386#[serde(rename_all = "lowercase")]
387pub enum ObserverType {
388    Human,
389    Agent,
390    Policy,
391    System,
392}
393
394/// One immutable audit Observation per transition, hash-chained per
395/// recommendation.
396#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
397pub struct AuditRecord {
398    pub rec_hash: String,
399    #[serde(skip_serializing_if = "Option::is_none")]
400    pub from: Option<RecStatus>,
401    pub to: RecStatus,
402    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
403    /// `policy:auto`.
404    pub actor: String,
405    pub observer_type: ObserverType,
406    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
407    pub because: String,
408    #[serde(skip_serializing_if = "Option::is_none")]
409    pub previous_audit_hash: Option<String>,
410    /// §7.4: present exactly when this transition applied a code revision —
411    /// the evalset-run edge, recorded where the forensics look.
412    #[serde(default, skip_serializing_if = "Option::is_none")]
413    pub gating: Option<GatingEvidence>,
414    pub at_ms: i64,
415}
416
417/// Maximum length of a BECAUSE reason.
418pub const MAX_BECAUSE: usize = 500;
419
420impl AuditRecord {
421    /// Build the Observation grain that records this transition. `derived_from`
422    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
423    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
424        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
425        if let Some(prev) = &self.previous_audit_hash {
426            derived_from.push(Value::from(prev.clone()));
427        }
428        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
429            .with_field("observation_kind", "loop_audit")
430            .with_field("rec_hash", self.rec_hash.clone())
431            .with_field("to_status", self.to.as_str())
432            .with_field("actor", self.actor.clone())
433            .with_field(
434                "observer_type",
435                serde_json::to_value(self.observer_type).unwrap(),
436            )
437            .with_field("because", self.because.clone())
438            .with_field("at_ms", self.at_ms)
439            .with_field("derived_from", Value::Array(derived_from));
440        if let Some(from) = self.from {
441            spec.fields
442                .insert("from_status".into(), Value::from(from.as_str()));
443        }
444        if let Some(g) = &self.gating {
445            spec.fields
446                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
447            spec.fields
448                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
449            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
450            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
451        }
452        spec
453    }
454}
455
456/// A stored recommendation. `hash` (the content address) and `status` (the
457/// index-layer cache) are set by the engine, not serialized into the grain
458/// body — the body is immutable content, the lifecycle lives in the state
459/// index and the audit chain.
460#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
461pub struct Recommendation {
462    #[serde(skip)]
463    pub hash: String,
464    pub analyzer: String,
465    pub params_snapshot: Map<String, Value>,
466    pub origin: Origin,
467    pub target_ref: String,
468    pub action_kind: ActionKind,
469    pub dedup_key: String,
470    pub summary: Summary,
471    pub severity: Severity,
472    #[serde(flatten)]
473    pub proposal: Proposal,
474    pub destructive: bool,
475    pub rollbackable: bool,
476    #[serde(default)]
477    pub evidence: Vec<String>,
478    #[serde(default, skip_serializing_if = "Option::is_none")]
479    pub evidence_query: Option<String>,
480    #[serde(default, skip_serializing_if = "Option::is_none")]
481    pub metric: Option<MetricSnapshot>,
482    pub confidence: f64,
483    pub importance: f64,
484    pub created_at_ms: i64,
485    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
486    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
487    /// text — it never replaces the engine-templated `summary`.
488    #[serde(default, skip_serializing_if = "Option::is_none")]
489    pub guidance: Option<String>,
490    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
491    /// against — REQUIRED for code targets, forbidden elsewhere (a
492    /// recommendation targets code XOR an evalset, never both, and only
493    /// code carries the pin). Shown at review; checked live at apply
494    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
495    #[serde(default, skip_serializing_if = "Option::is_none")]
496    pub evalset_hash: Option<String>,
497    #[serde(skip)]
498    pub status: RecStatus,
499}
500
501/// The recorded evalset-run edge (§7.4): what an apply of a code revision
502/// must present, and what its audit Observation carries. "Trust me, it
503/// passed" is not an edge; (evalset, run, stats) is.
504#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
505pub struct GatingEvidence {
506    /// The evalset the gate ran — must equal the recommendation's pin.
507    pub evalset_hash: String,
508    /// The `areev run`/`areev eval` run id that executed the gate.
509    pub run_id: String,
510    pub passed: u64,
511    pub failed: u64,
512}
513
514/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
515/// re-checked when a stored grain is loaded (a hand-authored grain must not
516/// bypass the rule):
517/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
518/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
519///   seam inherits §7.4's gate wholesale).
520/// - An `evalset:` target is its own always-human-approved class: never a
521///   gated revision, and never itself pinned (the gate cannot gate itself).
522/// - No other target class carries a pin.
523pub fn validate_code_rules(
524    action: ActionKind,
525    target_class: &str,
526    evalset_hash: Option<&str>,
527) -> Result<()> {
528    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
529    match (action, target_class) {
530        (ActionKind::CodeRevision, "code") => {
531            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
532                return e1(
533                    "a code_revision must pin the evalset hash it was gated \
534                     against (evalset_hash)",
535                );
536            }
537        }
538        (ActionKind::CodeRevision, other) => {
539            return e1(&format!(
540                "code_revision requires a tool: target, got class '{other}'"
541            ));
542        }
543        (ActionKind::AdapterRevision, "model") => {
544            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
545                return e1(
546                    "an adapter_revision must pin the evalset hash it was \
547                     gated against (evalset_hash)",
548                );
549            }
550        }
551        (ActionKind::AdapterRevision, other) => {
552            return e1(&format!(
553                "adapter_revision requires a model: target, got class '{other}'"
554            ));
555        }
556        // A REVERT of an applied code revision targets the same tool: —
557        // it is the rollback inverse, not new code, so it needs no pin
558        // (blocking it here would leave the highest-stakes target class
559        // unable to propose its own measured rollback).
560        (ActionKind::Revert, "code") => {
561            if evalset_hash.is_some() {
562                return e1("a code revert carries no evalset pin");
563            }
564        }
565        (_, "code") => {
566            return e1("a tool: target requires action_kind code_revision");
567        }
568        // Same rollback-inverse escape hatch for the adapter class.
569        (ActionKind::Revert, "model") => {
570            if evalset_hash.is_some() {
571                return e1("an adapter revert carries no evalset pin");
572            }
573        }
574        (_, "model") => {
575            return e1("a model: target requires action_kind adapter_revision");
576        }
577        (_, "evalset") => {
578            if evalset_hash.is_some() {
579                return e1(
580                    "an evalset-target recommendation cannot pin an evalset — \
581                     the gate cannot gate itself",
582                );
583            }
584        }
585        _ => {
586            if evalset_hash.is_some() {
587                return e1(
588                    "evalset_hash is only valid on code_revision or \
589                     adapter_revision recommendations",
590                );
591            }
592        }
593    }
594    Ok(())
595}
596
597impl Recommendation {
598    /// Serialize the immutable body into grain fields (excludes hash/status).
599    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
600        let value = serde_json::to_value(self)
601            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
602        let obj = value
603            .as_object()
604            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
605            .clone();
606        Ok(GrainSpec {
607            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
608            namespace: namespace.to_string(),
609            fields: obj,
610        })
611    }
612
613    /// Reconstruct from a stored grain body. `hash` comes from the record's
614    /// address; `status` is supplied from the state index by the caller.
615    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
616    /// must not smuggle an unpinned code revision past the stamped path.
617    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
618        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
619            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
620        rec.hash = hash.to_string();
621        let class = crate::model::TargetRef::parse(&rec.target_ref)
622            .map(|t| t.target_class())
623            .unwrap_or("host");
624        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
625        Ok(rec)
626    }
627}
628
629#[cfg(test)]
630mod tests {
631    use super::*;
632    use serde_json::json;
633
634    #[test]
635    fn summary_renders_deterministically() {
636        let mut args = Map::new();
637        args.insert("count".into(), json!(3));
638        args.insert("subject".into(), json!("acme"));
639        let s = Summary::new("duplicate.exact", args);
640        assert_eq!(
641            s.render(),
642            "Consolidate 3 exact-duplicate grains for \"acme\""
643        );
644    }
645
646    #[test]
647    fn dedup_key_ignores_content_and_case() {
648        let a = dedup_key(
649            "loop.duplicate_sweep",
650            "entity:NS/John",
651            ActionKind::Consolidate,
652        );
653        let b = dedup_key(
654            "loop.duplicate_sweep",
655            "entity:ns/john",
656            ActionKind::Consolidate,
657        );
658        assert_eq!(a, b, "case-folded to one identity");
659    }
660
661    #[test]
662    fn dedup_key_distinguishes_action() {
663        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
664        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
665        assert_ne!(a, b);
666    }
667
668    #[test]
669    fn lifecycle_gates_pending_to_applied() {
670        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
671        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
672        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
673        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
674        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
675        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
676    }
677
678    #[test]
679    fn rule_e1_pins_code_and_only_code() {
680        use crate::model::ActionKind as A;
681        // code_revision on a tool target with a pin: OK.
682        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
683        // Unpinned code revision: refused.
684        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
685        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
686        // code_revision off a tool target: refused.
687        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
688        // A tool target with a non-code action: refused.
689        assert!(validate_code_rules(A::Flag, "code", None).is_err());
690        // The gate cannot gate itself.
691        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
692        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
693        // Pins don't leak onto ordinary targets.
694        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
695        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
696    }
697
698    #[test]
699    fn rule_e1_pins_adapter_revisions_like_code() {
700        use crate::model::ActionKind as A;
701        // adapter_revision on a model target with a pin: OK.
702        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
703        // Unpinned adapter revision: refused.
704        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
705        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
706        // adapter_revision off a model target: refused.
707        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
708        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
709        // A model target with a non-adapter action: refused.
710        assert!(validate_code_rules(A::Flag, "model", None).is_err());
711        // The rollback inverse needs no pin — and must carry none.
712        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
713        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
714    }
715
716    #[test]
717    fn from_fields_rechecks_rule_e1() {
718        // A hand-authored grain body carrying an unpinned code revision must
719        // not decode into a usable recommendation.
720        let mut rec = Recommendation {
721            hash: String::new(),
722            analyzer: "loop.codegen/1".into(),
723            params_snapshot: Map::new(),
724            origin: Origin::Builtin,
725            target_ref: "tool:abc123".into(),
726            action_kind: ActionKind::CodeRevision,
727            dedup_key: "k".into(),
728            summary: Summary::new("command.finding", Map::new()),
729            severity: Severity::Low,
730            proposal: Proposal::Data { data: Map::new() },
731            destructive: false,
732            rollbackable: false,
733            evidence: vec![],
734            evidence_query: None,
735            metric: None,
736            confidence: 0.5,
737            importance: 0.5,
738            created_at_ms: 0,
739            guidance: None,
740            evalset_hash: None, // the smuggle attempt
741            status: RecStatus::Pending,
742        };
743        let spec = rec.to_grain_spec("ns").unwrap();
744        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
745        rec.evalset_hash = Some("es-hash".into());
746        let spec = rec.to_grain_spec("ns").unwrap();
747        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
748        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
749    }
750
751    #[test]
752    fn recommendation_round_trips_through_fields() {
753        let rec = Recommendation {
754            hash: "ignored".into(),
755            analyzer: "loop.staleness/1".into(),
756            params_snapshot: Map::new(),
757            origin: Origin::Builtin,
758            target_ref: "grain:sha256:abc".into(),
759            action_kind: ActionKind::Expire,
760            dedup_key: "k".into(),
761            summary: Summary::new("staleness.expired", Map::new()),
762            severity: Severity::Low,
763            proposal: Proposal::Cal {
764                cal: "FORGET sha256:abc".into(),
765            },
766            destructive: true,
767            rollbackable: false,
768            evidence: vec!["sha256:abc".into()],
769            evidence_query: None,
770            metric: None,
771            confidence: 0.9,
772            importance: 0.4,
773            created_at_ms: 1000,
774            guidance: None,
775            evalset_hash: None,
776            status: RecStatus::Pending,
777        };
778        let spec = rec.to_grain_spec("ns").unwrap();
779        // hash and status are excluded from the immutable body.
780        assert!(!spec.fields.contains_key("hash"));
781        assert!(!spec.fields.contains_key("status"));
782        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
783        assert_eq!(back.hash, "realhash");
784        assert_eq!(back.analyzer, "loop.staleness/1");
785        assert!(back.destructive);
786        assert!(matches!(back.proposal, Proposal::Cal { .. }));
787    }
788}