Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel.
9//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
10
11use crate::error::{Error, Result};
12use crate::model::{normalize_ident, ActionKind, Origin, Severity};
13use crate::substrate::GrainSpec;
14use serde::{Deserialize, Serialize};
15use serde_json::{Map, Value};
16
17/// A deterministic, template-rendered summary. The analyzer chooses a
18/// `template_id` and supplies `args`; the text is produced here, so an
19/// analyzer can never emit arbitrary prose into the queue.
20#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
21pub struct Summary {
22    pub template_id: String,
23    #[serde(default)]
24    pub args: Map<String, Value>,
25}
26
27impl Summary {
28    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
29        Summary {
30            template_id: template_id.into(),
31            args,
32        }
33    }
34
35    /// Render the summary. Unknown template ids fall back to a stable, honest
36    /// string (never a panic) so a manifest/template mismatch is visible, not
37    /// fatal.
38    pub fn render(&self) -> String {
39        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
40        interpolate(t, &self.args, &self.template_id)
41    }
42}
43
44/// The built-in template table. Deterministic; the only place summary prose
45/// lives. Keep placeholders in `{name}` form matching `args` keys.
46fn builtin_template(id: &str) -> Option<&'static str> {
47    Some(match id {
48        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
49        "duplicate.near" => {
50            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
51        }
52        "contradiction.functional" => {
53            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
54        }
55        "tool_failure.cluster" => {
56            "Tool \"{tool}\" failed {count} times ({rate}% of calls): {signature}"
57        }
58        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
59        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
60        "skill.stall" => {
61            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
62        }
63        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
64        "cold.grain" => {
65            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
66        }
67        "coverage.gap" => {
68            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
69        }
70        "budget.pressure" => {
71            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
72        }
73        "retention.overdue" => {
74            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
75        }
76        "outcome.regression" => {
77            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
78        }
79        "run.failures" => {
80            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
81        }
82        "run.cost" => {
83            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
84        }
85        // origin=llm drafts carry free text (clearly marked llm) rather than a
86        // deterministic template — the model's proposed summary rides in {text}.
87        "llm.discover" => "{text}",
88        // External command analyzers (trust class Command): free text from the
89        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
90        "command.finding" => "{text}",
91        _ => return None,
92    })
93}
94
95fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
96    let mut out = String::with_capacity(template.len());
97    let mut chars = template.chars().peekable();
98    while let Some(c) = chars.next() {
99        if c == '{' {
100            let mut key = String::new();
101            for k in chars.by_ref() {
102                if k == '}' {
103                    break;
104                }
105                key.push(k);
106            }
107            if key == "template_id" {
108                out.push_str(template_id);
109            } else {
110                match args.get(&key) {
111                    Some(Value::String(s)) => out.push_str(s),
112                    Some(v) => out.push_str(&v.to_string()),
113                    None => {
114                        // Missing arg — leave a visible marker rather than lie.
115                        out.push('{');
116                        out.push_str(&key);
117                        out.push('}');
118                    }
119                }
120            }
121        } else {
122            out.push(c);
123        }
124    }
125    out
126}
127
128/// A reproducible metric snapshot; powers outcome review.
129#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
130pub struct MetricSnapshot {
131    /// Metric kind — the engine knows how to re-measure a fixed set
132    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
133    pub metric: String,
134    pub baseline: f64,
135    pub unit: String,
136    pub n: u64,
137    pub window: String,
138    /// The subject the metric is about (e.g. the tool name), used by the
139    /// engine's typed re-measurement.
140    #[serde(default, skip_serializing_if = "Option::is_none")]
141    pub subject: Option<String>,
142    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
143    /// `None` = all namespaces. Additive: older snapshots deserialize to
144    /// `None` and existing metric kinds ignore it.
145    #[serde(default, skip_serializing_if = "Option::is_none")]
146    pub namespace: Option<String>,
147    /// Relation scope for fact-shaped metrics (e.g. which functional relation
148    /// a contradiction was resolved under), normalized. Additive like
149    /// `namespace`.
150    #[serde(default, skip_serializing_if = "Option::is_none")]
151    pub relation: Option<String>,
152    /// CAL that recomputes the metric at verify time (reproducibility /
153    /// documentation; the engine re-measures with typed reads).
154    pub query: String,
155    /// How long after apply to re-measure, in epoch-ms delta (the first / only
156    /// checkpoint when `horizons_ms` is empty).
157    pub review_after_ms: i64,
158    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
159    /// that `held` at an early checkpoint can `regress` at a later one, so a
160    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
161    #[serde(default, skip_serializing_if = "Vec::is_empty")]
162    pub horizons_ms: Vec<i64>,
163    /// Which direction is an improvement. The built-in metrics are all
164    /// recurrence counts, where lower is better — so this defaults to `false`
165    /// and every existing snapshot deserializes unchanged. An evalset accuracy
166    /// is the opposite, and getting it wrong does not merely misreport: the
167    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
168    /// propose reverting it.
169    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
170    pub higher_is_better: bool,
171}
172
173/// The one regression rule.
174///
175/// The verdict is computed in two places — once by the engine for the recorded
176/// `OutcomeResult`, once by `outcome_review` when it drafts the revert — and
177/// the two silently disagreeing would either revert a change that held or sit
178/// on one that regressed. So both call this.
179pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool) -> bool {
180    /// Minimum worsening to call a regression (avoids noise at n=1).
181    const EPSILON: f64 = 1e-9;
182    if higher_is_better {
183        current < baseline - EPSILON
184    } else {
185        current > baseline + EPSILON
186    }
187}
188
189impl MetricSnapshot {
190    /// The measurement schedule, sorted — `horizons_ms` if set, else the single
191    /// `review_after_ms`.
192    pub fn horizons(&self) -> Vec<i64> {
193        let mut h = if self.horizons_ms.is_empty() {
194            vec![self.review_after_ms]
195        } else {
196            self.horizons_ms.clone()
197        };
198        h.sort_unstable();
199        h
200    }
201}
202
203/// A measured outcome for an applied recommendation at one checkpoint — the
204/// Verify gate's output. `held` = the metric did not regress at this horizon;
205/// `regressed` = it got worse (a revert is proposed). A recommendation
206/// accumulates one of these per horizon, forming a time series.
207#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
208pub struct OutcomeResult {
209    pub rec_hash: String,
210    pub metric: String,
211    pub baseline: f64,
212    pub current: f64,
213    pub verdict: String,
214    /// Which checkpoint this measurement is for (ms after apply).
215    #[serde(default)]
216    pub horizon_ms: i64,
217    pub measured_at_ms: i64,
218}
219
220/// The proposed change. Exactly one variant per recommendation.
221#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
222#[serde(tag = "proposal", rename_all = "snake_case")]
223pub enum Proposal {
224    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
225    Cal { cal: String },
226    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
227    /// staleness check at apply).
228    Edit {
229        format: String,
230        base_digest: String,
231        diff: String,
232    },
233    /// An opaque map for host targets (applied by the host, §12.3).
234    Data { data: Map<String, Value> },
235}
236
237/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
238/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
239/// fields without breaking analyzers.
240#[derive(Debug, Clone, PartialEq)]
241#[non_exhaustive]
242pub struct RecDraft {
243    pub target_ref: String,
244    pub action_kind: ActionKind,
245    pub summary: Summary,
246    pub severity: Severity,
247    pub proposal: Proposal,
248    /// Bounded to ≤64 representative evidence hashes by the engine.
249    pub evidence: Vec<String>,
250    pub evidence_query: Option<String>,
251    pub metric: Option<MetricSnapshot>,
252    pub confidence: f64,
253    pub importance: f64,
254    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
255    pub evalset_hash: Option<String>,
256}
257
258impl RecDraft {
259    pub fn new(
260        target_ref: impl Into<String>,
261        action_kind: ActionKind,
262        summary: Summary,
263        proposal: Proposal,
264    ) -> Self {
265        RecDraft {
266            target_ref: target_ref.into(),
267            action_kind,
268            summary,
269            severity: Severity::Low,
270            proposal,
271            evidence: Vec::new(),
272            evidence_query: None,
273            metric: None,
274            confidence: 0.8,
275            importance: 0.5,
276            evalset_hash: None,
277        }
278    }
279
280    pub fn severity(mut self, s: Severity) -> Self {
281        self.severity = s;
282        self
283    }
284    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
285        self.evidence = hashes;
286        self
287    }
288    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
289        self.evidence_query = Some(q.into());
290        self
291    }
292    pub fn metric(mut self, m: MetricSnapshot) -> Self {
293        self.metric = Some(m);
294        self
295    }
296    pub fn confidence(mut self, c: f64) -> Self {
297        self.confidence = c;
298        self
299    }
300    pub fn importance(mut self, i: f64) -> Self {
301        self.importance = i;
302        self
303    }
304    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
305        self.evalset_hash = Some(h.into());
306        self
307    }
308}
309
310/// Maximum representative evidence hashes carried inline (proposal §7.1).
311pub const MAX_EVIDENCE: usize = 64;
312
313/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
314/// Excludes proposal content and evidence by construction.
315pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
316    // U+001F (unit separator) cannot appear in any of the inputs.
317    format!(
318        "{}\u{1f}{}\u{1f}{}",
319        normalize_ident(family),
320        normalize_ident(target_ref),
321        action.as_str()
322    )
323}
324
325/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
326/// content hash is stable for its whole life).
327#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
328#[serde(rename_all = "snake_case")]
329pub enum RecStatus {
330    #[default]
331    Pending,
332    Approved,
333    Rejected,
334    Applied,
335    RolledBack,
336    Expired,
337}
338
339impl RecStatus {
340    pub fn as_str(&self) -> &'static str {
341        match self {
342            RecStatus::Pending => "pending",
343            RecStatus::Approved => "approved",
344            RecStatus::Rejected => "rejected",
345            RecStatus::Applied => "applied",
346            RecStatus::RolledBack => "rolled_back",
347            RecStatus::Expired => "expired",
348        }
349    }
350
351    /// Is a transition to `to` allowed from this state? `by_policy` marks the
352    /// auto-apply actor, the only one permitted the reasonless
353    /// `pending → applied` jump.
354    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
355        use RecStatus::*;
356        match (self, to) {
357            (Pending, Approved) | (Pending, Rejected) => true,
358            (Pending, Applied) => by_policy, // auto-apply only
359            (Approved, Applied) => true,
360            (Applied, RolledBack) => true,
361            // `expired` is computed from valid_to, applied to still-open recs.
362            (Pending, Expired) | (Approved, Expired) => true,
363            _ => false,
364        }
365    }
366}
367
368/// Who/what performed a transition.
369#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
370#[serde(rename_all = "lowercase")]
371pub enum ObserverType {
372    Human,
373    Agent,
374    Policy,
375    System,
376}
377
378/// One immutable audit Observation per transition, hash-chained per
379/// recommendation.
380#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
381pub struct AuditRecord {
382    pub rec_hash: String,
383    #[serde(skip_serializing_if = "Option::is_none")]
384    pub from: Option<RecStatus>,
385    pub to: RecStatus,
386    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
387    /// `policy:auto`.
388    pub actor: String,
389    pub observer_type: ObserverType,
390    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
391    pub because: String,
392    #[serde(skip_serializing_if = "Option::is_none")]
393    pub previous_audit_hash: Option<String>,
394    /// §7.4: present exactly when this transition applied a code revision —
395    /// the evalset-run edge, recorded where the forensics look.
396    #[serde(default, skip_serializing_if = "Option::is_none")]
397    pub gating: Option<GatingEvidence>,
398    pub at_ms: i64,
399}
400
401/// Maximum length of a BECAUSE reason.
402pub const MAX_BECAUSE: usize = 500;
403
404impl AuditRecord {
405    /// Build the Observation grain that records this transition. `derived_from`
406    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
407    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
408        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
409        if let Some(prev) = &self.previous_audit_hash {
410            derived_from.push(Value::from(prev.clone()));
411        }
412        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
413            .with_field("observation_kind", "loop_audit")
414            .with_field("rec_hash", self.rec_hash.clone())
415            .with_field("to_status", self.to.as_str())
416            .with_field("actor", self.actor.clone())
417            .with_field(
418                "observer_type",
419                serde_json::to_value(self.observer_type).unwrap(),
420            )
421            .with_field("because", self.because.clone())
422            .with_field("at_ms", self.at_ms)
423            .with_field("derived_from", Value::Array(derived_from));
424        if let Some(from) = self.from {
425            spec.fields
426                .insert("from_status".into(), Value::from(from.as_str()));
427        }
428        if let Some(g) = &self.gating {
429            spec.fields
430                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
431            spec.fields
432                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
433            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
434            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
435        }
436        spec
437    }
438}
439
440/// A stored recommendation. `hash` (the content address) and `status` (the
441/// index-layer cache) are set by the engine, not serialized into the grain
442/// body — the body is immutable content, the lifecycle lives in the state
443/// index and the audit chain.
444#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
445pub struct Recommendation {
446    #[serde(skip)]
447    pub hash: String,
448    pub analyzer: String,
449    pub params_snapshot: Map<String, Value>,
450    pub origin: Origin,
451    pub target_ref: String,
452    pub action_kind: ActionKind,
453    pub dedup_key: String,
454    pub summary: Summary,
455    pub severity: Severity,
456    #[serde(flatten)]
457    pub proposal: Proposal,
458    pub destructive: bool,
459    pub rollbackable: bool,
460    #[serde(default)]
461    pub evidence: Vec<String>,
462    #[serde(default, skip_serializing_if = "Option::is_none")]
463    pub evidence_query: Option<String>,
464    #[serde(default, skip_serializing_if = "Option::is_none")]
465    pub metric: Option<MetricSnapshot>,
466    pub confidence: f64,
467    pub importance: f64,
468    pub created_at_ms: i64,
469    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
470    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
471    /// text — it never replaces the engine-templated `summary`.
472    #[serde(default, skip_serializing_if = "Option::is_none")]
473    pub guidance: Option<String>,
474    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
475    /// against — REQUIRED for code targets, forbidden elsewhere (a
476    /// recommendation targets code XOR an evalset, never both, and only
477    /// code carries the pin). Shown at review; checked live at apply
478    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
479    #[serde(default, skip_serializing_if = "Option::is_none")]
480    pub evalset_hash: Option<String>,
481    #[serde(skip)]
482    pub status: RecStatus,
483}
484
485/// The recorded evalset-run edge (§7.4): what an apply of a code revision
486/// must present, and what its audit Observation carries. "Trust me, it
487/// passed" is not an edge; (evalset, run, stats) is.
488#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
489pub struct GatingEvidence {
490    /// The evalset the gate ran — must equal the recommendation's pin.
491    pub evalset_hash: String,
492    /// The `areev run`/`areev eval` run id that executed the gate.
493    pub run_id: String,
494    pub passed: u64,
495    pub failed: u64,
496}
497
498/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
499/// re-checked when a stored grain is loaded (a hand-authored grain must not
500/// bypass the rule):
501/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
502/// - An `evalset:` target is its own always-human-approved class: never a
503///   `code_revision`, and never itself pinned (the gate cannot gate itself).
504/// - No other target class carries a pin.
505pub fn validate_code_rules(
506    action: ActionKind,
507    target_class: &str,
508    evalset_hash: Option<&str>,
509) -> Result<()> {
510    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
511    match (action, target_class) {
512        (ActionKind::CodeRevision, "code") => {
513            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
514                return e1(
515                    "a code_revision must pin the evalset hash it was gated \
516                     against (evalset_hash)",
517                );
518            }
519        }
520        (ActionKind::CodeRevision, other) => {
521            return e1(&format!(
522                "code_revision requires a tool: target, got class '{other}'"
523            ));
524        }
525        // A REVERT of an applied code revision targets the same tool: —
526        // it is the rollback inverse, not new code, so it needs no pin
527        // (blocking it here would leave the highest-stakes target class
528        // unable to propose its own measured rollback).
529        (ActionKind::Revert, "code") => {
530            if evalset_hash.is_some() {
531                return e1("a code revert carries no evalset pin");
532            }
533        }
534        (_, "code") => {
535            return e1("a tool: target requires action_kind code_revision");
536        }
537        (_, "evalset") => {
538            if evalset_hash.is_some() {
539                return e1(
540                    "an evalset-target recommendation cannot pin an evalset — \
541                     the gate cannot gate itself",
542                );
543            }
544        }
545        _ => {
546            if evalset_hash.is_some() {
547                return e1("evalset_hash is only valid on code_revision recommendations");
548            }
549        }
550    }
551    Ok(())
552}
553
554impl Recommendation {
555    /// Serialize the immutable body into grain fields (excludes hash/status).
556    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
557        let value = serde_json::to_value(self)
558            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
559        let obj = value
560            .as_object()
561            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
562            .clone();
563        Ok(GrainSpec {
564            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
565            namespace: namespace.to_string(),
566            fields: obj,
567        })
568    }
569
570    /// Reconstruct from a stored grain body. `hash` comes from the record's
571    /// address; `status` is supplied from the state index by the caller.
572    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
573    /// must not smuggle an unpinned code revision past the stamped path.
574    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
575        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
576            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
577        rec.hash = hash.to_string();
578        let class = crate::model::TargetRef::parse(&rec.target_ref)
579            .map(|t| t.target_class())
580            .unwrap_or("host");
581        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
582        Ok(rec)
583    }
584}
585
586#[cfg(test)]
587mod tests {
588    use super::*;
589    use serde_json::json;
590
591    #[test]
592    fn summary_renders_deterministically() {
593        let mut args = Map::new();
594        args.insert("count".into(), json!(3));
595        args.insert("subject".into(), json!("acme"));
596        let s = Summary::new("duplicate.exact", args);
597        assert_eq!(
598            s.render(),
599            "Consolidate 3 exact-duplicate grains for \"acme\""
600        );
601    }
602
603    #[test]
604    fn dedup_key_ignores_content_and_case() {
605        let a = dedup_key(
606            "loop.duplicate_sweep",
607            "entity:NS/John",
608            ActionKind::Consolidate,
609        );
610        let b = dedup_key(
611            "loop.duplicate_sweep",
612            "entity:ns/john",
613            ActionKind::Consolidate,
614        );
615        assert_eq!(a, b, "case-folded to one identity");
616    }
617
618    #[test]
619    fn dedup_key_distinguishes_action() {
620        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
621        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
622        assert_ne!(a, b);
623    }
624
625    #[test]
626    fn lifecycle_gates_pending_to_applied() {
627        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
628        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
629        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
630        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
631        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
632        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
633    }
634
635    #[test]
636    fn rule_e1_pins_code_and_only_code() {
637        use crate::model::ActionKind as A;
638        // code_revision on a tool target with a pin: OK.
639        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
640        // Unpinned code revision: refused.
641        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
642        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
643        // code_revision off a tool target: refused.
644        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
645        // A tool target with a non-code action: refused.
646        assert!(validate_code_rules(A::Flag, "code", None).is_err());
647        // The gate cannot gate itself.
648        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
649        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
650        // Pins don't leak onto ordinary targets.
651        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
652        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
653    }
654
655    #[test]
656    fn from_fields_rechecks_rule_e1() {
657        // A hand-authored grain body carrying an unpinned code revision must
658        // not decode into a usable recommendation.
659        let mut rec = Recommendation {
660            hash: String::new(),
661            analyzer: "loop.codegen/1".into(),
662            params_snapshot: Map::new(),
663            origin: Origin::Builtin,
664            target_ref: "tool:abc123".into(),
665            action_kind: ActionKind::CodeRevision,
666            dedup_key: "k".into(),
667            summary: Summary::new("command.finding", Map::new()),
668            severity: Severity::Low,
669            proposal: Proposal::Data { data: Map::new() },
670            destructive: false,
671            rollbackable: false,
672            evidence: vec![],
673            evidence_query: None,
674            metric: None,
675            confidence: 0.5,
676            importance: 0.5,
677            created_at_ms: 0,
678            guidance: None,
679            evalset_hash: None, // the smuggle attempt
680            status: RecStatus::Pending,
681        };
682        let spec = rec.to_grain_spec("ns").unwrap();
683        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
684        rec.evalset_hash = Some("es-hash".into());
685        let spec = rec.to_grain_spec("ns").unwrap();
686        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
687        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
688    }
689
690    #[test]
691    fn recommendation_round_trips_through_fields() {
692        let rec = Recommendation {
693            hash: "ignored".into(),
694            analyzer: "loop.staleness/1".into(),
695            params_snapshot: Map::new(),
696            origin: Origin::Builtin,
697            target_ref: "grain:sha256:abc".into(),
698            action_kind: ActionKind::Expire,
699            dedup_key: "k".into(),
700            summary: Summary::new("staleness.expired", Map::new()),
701            severity: Severity::Low,
702            proposal: Proposal::Cal {
703                cal: "FORGET sha256:abc".into(),
704            },
705            destructive: true,
706            rollbackable: false,
707            evidence: vec!["sha256:abc".into()],
708            evidence_query: None,
709            metric: None,
710            confidence: 0.9,
711            importance: 0.4,
712            created_at_ms: 1000,
713            guidance: None,
714            evalset_hash: None,
715            status: RecStatus::Pending,
716        };
717        let spec = rec.to_grain_spec("ns").unwrap();
718        // hash and status are excluded from the immutable body.
719        assert!(!spec.fields.contains_key("hash"));
720        assert!(!spec.fields.contains_key("status"));
721        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
722        assert_eq!(back.hash, "realhash");
723        assert_eq!(back.analyzer, "loop.staleness/1");
724        assert!(back.destructive);
725        assert!(matches!(back.proposal, Proposal::Cal { .. }));
726    }
727}