Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel —
9//!   for ANALYZER findings. An authored (`origin = llm`) executable proposal
10//!   keys on a fingerprint of its content as well, because there the content
11//!   IS the finding: two different lessons on one entity are two findings,
12//!   and the same lesson re-authored is one.
13//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
14
15use crate::error::{Error, Result};
16use crate::model::{normalize_ident, ActionKind, Origin, Severity};
17use crate::substrate::GrainSpec;
18use serde::{Deserialize, Serialize};
19use serde_json::{Map, Value};
20
21/// A deterministic, template-rendered summary. The analyzer chooses a
22/// `template_id` and supplies `args`; the text is produced here, so an
23/// analyzer can never emit arbitrary prose into the queue.
24#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
25pub struct Summary {
26    pub template_id: String,
27    #[serde(default)]
28    pub args: Map<String, Value>,
29}
30
31impl Summary {
32    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
33        Summary {
34            template_id: template_id.into(),
35            args,
36        }
37    }
38
39    /// Render the summary. Unknown template ids fall back to a stable, honest
40    /// string (never a panic) so a manifest/template mismatch is visible, not
41    /// fatal.
42    pub fn render(&self) -> String {
43        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
44        interpolate(t, &self.args, &self.template_id)
45    }
46}
47
48/// The built-in template table. Deterministic; the only place summary prose
49/// lives. Keep placeholders in `{name}` form matching `args` keys.
50fn builtin_template(id: &str) -> Option<&'static str> {
51    Some(match id {
52        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
53        "duplicate.near" => {
54            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
55        }
56        "contradiction.functional" => {
57            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
58        }
59        "tool_failure.cluster" => {
60            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
61             fail this way): {signature}"
62        }
63        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
64        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
65        "skill.stall" => {
66            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
67        }
68        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
69        "cold.grain" => {
70            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
71        }
72        "coverage.gap" => {
73            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
74        }
75        "budget.pressure" => {
76            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
77        }
78        "retention.overdue" => {
79            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
80        }
81        "outcome.regression" => {
82            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
83        }
84        // The high-water baseline names both figures — the best run before
85        // the apply and the run that fell from it — because the reviewer has
86        // to judge whether THIS rule caused the whole fall from the peak.
87        "outcome.regression_high_water" => {
88            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}"
89        }
90        // A regression that also breached the cost bound: the revert names
91        // the cost delta so the reviewer sees both halves of the damage.
92        "outcome.regression_costlier" => {
93            "Applied recommendation regressed: {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
94        }
95        "outcome.regression_high_water_costlier" => {
96            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
97        }
98        // Quality held, cost did not: advisory only — a cost/quality trade
99        // is a human decision, so this is a Flag, never a revert draft.
100        "outcome.held_costlier" => {
101            "Applied recommendation held ({metric} {baseline} → {current}) but {cost_field} rose {cost_baseline} → {cost_current}, past ×{cost_ratio} of the baseline run ({baseline_run} → {current_run}) — a cost/quality trade to decide"
102        }
103        "outcome.premise_drift" => {
104            "{current} of the grains this recommendation cited have since been superseded by a different value or retracted — its premise moved; revert it"
105        }
106        "run.failures" => {
107            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
108        }
109        "run.cost" => {
110            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
111        }
112        "run.context_pressure" => {
113            "Workflow {workflow} summarized its own transcript {folds} time(s) across {runs} runs (avg {avg_folds}/run) — its nodes are outgrowing the model's window"
114        }
115        "adapter.candidate" => {
116            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
117        }
118        // origin=llm drafts carry free text (clearly marked llm) rather than a
119        // deterministic template — the model's proposed summary rides in {text}.
120        "llm.discover" => "{text}",
121        // An LLM-authored lesson shows the reviewer BOTH the finding and the
122        // exact line an apply would record — never one without the other.
123        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
124        // The same lesson, flagged as restating a live one: the reviewer sees
125        // that a rule of this meaning already exists before approving a
126        // second copy of it.
127        "llm.lesson_near_duplicate" => {
128            "{text} — record lesson: \"{lesson}\" — NEAR-DUPLICATE of {near_count} live lesson(s) on this entity (best match {near_score} by {near_method}, {near_hash})"
129        }
130        // A consolidation: one lesson replacing a pile. The apply supersedes
131        // every member and adds the one line; rollback restores them all.
132        "llm.consolidation" => {
133            "{text} — consolidate {count} lessons into one: \"{lesson}\""
134        }
135        "lesson.pile" => {
136            "{count} live lessons on \"{subject}\" exceed the budget of {max_active} — every rule competes for the model's attention; consolidate or retire: {members}"
137        }
138        // The rest of the LLM proposal vocabulary follows the same rule: the
139        // finding AND the exact change an apply would make, never one without
140        // the other. A reviewer approving blind is the failure this prevents.
141        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
142        "llm.skill" => "{text} — record skill: \"{name}\" ({steps} steps)",
143        "llm.plan" => "{text} — record plan: \"{name}\" ({nodes} steps, {edges} edges)",
144        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
145        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
146        // The rehearsal refused it: the edit is shown, the reason names the
147        // runs, and nothing is offered to apply.
148        "llm.plan_revision_refused" => {
149            "{text} — plan revision of {plan} ({edits}) NOT applicable: rehearsed against the plan's journaled runs and {reason}"
150        }
151        "llm.code_revision" => {
152            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
153        }
154        // External command analyzers (trust class Command): free text from the
155        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
156        "command.finding" => "{text}",
157        _ => return None,
158    })
159}
160
161fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
162    let mut out = String::with_capacity(template.len());
163    let mut chars = template.chars().peekable();
164    while let Some(c) = chars.next() {
165        if c == '{' {
166            let mut key = String::new();
167            for k in chars.by_ref() {
168                if k == '}' {
169                    break;
170                }
171                key.push(k);
172            }
173            if key == "template_id" {
174                out.push_str(template_id);
175            } else {
176                match args.get(&key) {
177                    Some(Value::String(s)) => out.push_str(s),
178                    Some(v) => out.push_str(&v.to_string()),
179                    None => {
180                        // Missing arg — leave a visible marker rather than lie.
181                        out.push('{');
182                        out.push_str(&key);
183                        out.push('}');
184                    }
185                }
186            }
187        } else {
188            out.push(c);
189        }
190    }
191    out
192}
193
194/// When an applied recommendation is re-measured — one checkpoint of the
195/// Verify gate's schedule, in the unit the deployment actually counts in.
196///
197/// Three units, because deployments count differently and a fixed one makes
198/// the gate inert everywhere else. A service evaluated nightly counts
199/// **time** (`after_ms`). A benchmark or a CI harness counts **graded runs**
200/// (`after_runs`): "measure at the next evaluation after the apply", however
201/// long that takes on the clock. A chat agent counts **turns** (`after_grains`):
202/// "after fifty more grains". The default schedule stayed ms-only for a year
203/// and was measured, on PAST-Bench, to fire exactly zero verdicts across 78
204/// governed runs — a family finishes in seven minutes and the first checkpoint
205/// was a day away (`crates/areev-bench/PERSIST.md`).
206///
207/// On the wire a bare integer is milliseconds, so every policy, metric
208/// snapshot and state blob written before this type existed reads back
209/// unchanged, and an all-ms schedule still serializes as `[86400000, …]`.
210#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
211pub enum Checkpoint {
212    /// Elapsed time since the apply.
213    AfterMs(i64),
214    /// Evalset runs journaled since the apply. Only an `evalset:` metric can
215    /// count these; on any other metric the checkpoint never comes due.
216    AfterRuns(u32),
217    /// Grains written since the apply — the loop's own notion of "activity".
218    AfterGrains(u32),
219}
220
221impl Checkpoint {
222    /// The label a person reads: `1d`, `12h`, `3 runs`, `50 grains`.
223    pub fn label(&self) -> String {
224        match self {
225            Checkpoint::AfterMs(ms) if ms % 86_400_000 == 0 => format!("{}d", ms / 86_400_000),
226            Checkpoint::AfterMs(ms) if ms % 3_600_000 == 0 => format!("{}h", ms / 3_600_000),
227            Checkpoint::AfterMs(ms) => format!("{ms}ms"),
228            Checkpoint::AfterRuns(n) => format!("{n} run{}", if *n == 1 { "" } else { "s" }),
229            Checkpoint::AfterGrains(n) => format!("{n} grain{}", if *n == 1 { "" } else { "s" }),
230        }
231    }
232
233    /// The ms value when this is a time checkpoint — what legacy consumers of
234    /// `OutcomeResult::horizon_ms` read.
235    pub fn as_ms(&self) -> Option<i64> {
236        match self {
237            Checkpoint::AfterMs(ms) => Some(*ms),
238            _ => None,
239        }
240    }
241}
242
243impl Serialize for Checkpoint {
244    fn serialize<S: serde::Serializer>(&self, ser: S) -> std::result::Result<S::Ok, S::Error> {
245        use serde::ser::SerializeMap;
246        match self {
247            // Bare integer: byte-identical to the pre-Checkpoint `Vec<i64>`.
248            Checkpoint::AfterMs(ms) => ser.serialize_i64(*ms),
249            Checkpoint::AfterRuns(n) => {
250                let mut m = ser.serialize_map(Some(1))?;
251                m.serialize_entry("after_runs", n)?;
252                m.end()
253            }
254            Checkpoint::AfterGrains(n) => {
255                let mut m = ser.serialize_map(Some(1))?;
256                m.serialize_entry("after_grains", n)?;
257                m.end()
258            }
259        }
260    }
261}
262
263impl<'de> Deserialize<'de> for Checkpoint {
264    fn deserialize<D: serde::Deserializer<'de>>(de: D) -> std::result::Result<Self, D::Error> {
265        use serde::de::Error as _;
266        let v = Value::deserialize(de)?;
267        match &v {
268            Value::Number(n) => n
269                .as_i64()
270                .filter(|ms| *ms >= 0)
271                .map(Checkpoint::AfterMs)
272                .ok_or_else(|| D::Error::custom("checkpoint: a bare number is non-negative milliseconds")),
273            Value::Object(m) if m.len() == 1 => {
274                let (k, val) = m.iter().next().unwrap();
275                let n = val.as_u64().ok_or_else(|| {
276                    D::Error::custom(format!("checkpoint: {k} takes a non-negative integer"))
277                })?;
278                match k.as_str() {
279                    "after_ms" => i64::try_from(n)
280                        .map(Checkpoint::AfterMs)
281                        .map_err(|_| D::Error::custom("checkpoint: after_ms out of range")),
282                    "after_runs" => u32::try_from(n)
283                        .map(Checkpoint::AfterRuns)
284                        .map_err(|_| D::Error::custom("checkpoint: after_runs out of range")),
285                    "after_grains" => u32::try_from(n)
286                        .map(Checkpoint::AfterGrains)
287                        .map_err(|_| D::Error::custom("checkpoint: after_grains out of range")),
288                    other => Err(D::Error::custom(format!(
289                        "checkpoint: unknown unit {other:?} (expected after_ms, after_runs or after_grains)"
290                    ))),
291                }
292            }
293            _ => Err(D::Error::custom(
294                "checkpoint: expected milliseconds or {\"after_ms\"|\"after_runs\"|\"after_grains\": n}",
295            )),
296        }
297    }
298}
299
300/// A reproducible metric snapshot; powers outcome review.
301#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
302pub struct MetricSnapshot {
303    /// Metric kind — the engine knows how to re-measure a fixed set
304    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
305    pub metric: String,
306    pub baseline: f64,
307    pub unit: String,
308    pub n: u64,
309    pub window: String,
310    /// The subject the metric is about (e.g. the tool name), used by the
311    /// engine's typed re-measurement.
312    #[serde(default, skip_serializing_if = "Option::is_none")]
313    pub subject: Option<String>,
314    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
315    /// `None` = all namespaces. Additive: older snapshots deserialize to
316    /// `None` and existing metric kinds ignore it.
317    #[serde(default, skip_serializing_if = "Option::is_none")]
318    pub namespace: Option<String>,
319    /// Relation scope for fact-shaped metrics (e.g. which functional relation
320    /// a contradiction was resolved under), normalized. Additive like
321    /// `namespace`.
322    #[serde(default, skip_serializing_if = "Option::is_none")]
323    pub relation: Option<String>,
324    /// CAL that recomputes the metric at verify time (reproducibility /
325    /// documentation; the engine re-measures with typed reads).
326    pub query: String,
327    /// How long after apply to re-measure, in epoch-ms delta (the first / only
328    /// checkpoint when `horizons_ms` is empty).
329    pub review_after_ms: i64,
330    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
331    /// that `held` at an early checkpoint can `regress` at a later one, so a
332    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
333    #[serde(default, skip_serializing_if = "Vec::is_empty")]
334    pub horizons_ms: Vec<i64>,
335    /// The schedule in the deployment's own unit ([`Checkpoint`]). When set it
336    /// is THE schedule and `horizons_ms`/`review_after_ms` are ignored; empty
337    /// (every snapshot written before it existed) falls back to them.
338    #[serde(default, skip_serializing_if = "Vec::is_empty")]
339    pub checkpoints: Vec<Checkpoint>,
340    /// Which direction is an improvement. The built-in metrics are all
341    /// recurrence counts, where lower is better — so this defaults to `false`
342    /// and every existing snapshot deserializes unchanged. An evalset accuracy
343    /// is the opposite, and getting it wrong does not merely misreport: the
344    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
345    /// propose reverting it.
346    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
347    pub higher_is_better: bool,
348}
349
350/// The one regression rule.
351///
352/// The verdict is computed in three places — by the engine for the recorded
353/// `OutcomeResult`, by `outcome_review` when it drafts the revert, and by
354/// `areev eval run --tolerance` when it re-accepts a model swap — and any two
355/// silently disagreeing would either revert a change that held or sit on one
356/// that regressed. So all of them call this.
357///
358/// `tolerance` is the policy's minimum effect size in the metric's own unit
359/// (`OutcomeEvalset::min_effect`, or `--tolerance` in percentage points of
360/// the pass rate): a worsening of at most that much is not a regression. It
361/// is a **floor, not a significance test** — no p-value, no interval; a host
362/// that needs statistics has the run counts to compute them. Zero is the
363/// pre-policy behaviour: any drop is a regression. `EPSILON` is only
364/// floating-point equality slack, so a value read twice that differs in the
365/// last bit does not flip the verdict; it is not a noise floor.
366pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool, tolerance: f64) -> bool {
367    const EPSILON: f64 = 1e-9;
368    let tolerance = if tolerance.is_finite() && tolerance > 0.0 { tolerance } else { 0.0 };
369    if higher_is_better {
370        current < baseline - tolerance - EPSILON
371    } else {
372        current > baseline + tolerance + EPSILON
373    }
374}
375
376impl MetricSnapshot {
377    /// The measurement schedule, sorted and deduplicated: `checkpoints` when
378    /// set, else `horizons_ms`, else the single `review_after_ms` — each older
379    /// spelling read as time, so a snapshot from before [`Checkpoint`] existed
380    /// measures exactly as it did.
381    pub fn schedule(&self) -> Vec<Checkpoint> {
382        let mut h: Vec<Checkpoint> = if !self.checkpoints.is_empty() {
383            self.checkpoints.clone()
384        } else if !self.horizons_ms.is_empty() {
385            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
386        } else {
387            vec![Checkpoint::AfterMs(self.review_after_ms)]
388        };
389        h.sort_unstable();
390        h.dedup();
391        h
392    }
393
394    /// The time checkpoints of the schedule, in ms — the pre-[`Checkpoint`]
395    /// view, kept for callers that only understand time.
396    pub fn horizons(&self) -> Vec<i64> {
397        self.schedule().iter().filter_map(Checkpoint::as_ms).collect()
398    }
399}
400
401fn is_zero(x: &f64) -> bool {
402    *x == 0.0
403}
404
405/// What a checkpoint read of the policy's cost bound, beside the quality
406/// verdict. `status` is `within`, `breached`, or `not_measurable` (the
407/// field was absent or malformed on either run — then `baseline`/`current`
408/// carry whichever side did measure, or nothing). Present on a record only
409/// when the policy set a bound.
410#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
411pub struct CostRead {
412    pub field: String,
413    pub max_increase_ratio: f64,
414    #[serde(default, skip_serializing_if = "Option::is_none")]
415    pub baseline: Option<f64>,
416    #[serde(default, skip_serializing_if = "Option::is_none")]
417    pub current: Option<f64>,
418    pub status: String,
419}
420
421impl CostRead {
422    pub fn breached(&self) -> bool {
423        self.status == "breached"
424    }
425}
426
427/// A measured outcome for an applied recommendation at one checkpoint — the
428/// Verify gate's output. `held` = the metric did not regress at this horizon;
429/// `regressed` = it got worse (a revert is proposed). A recommendation
430/// accumulates one of these per horizon, forming a time series.
431#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
432pub struct OutcomeResult {
433    pub rec_hash: String,
434    pub metric: String,
435    pub baseline: f64,
436    pub current: f64,
437    pub verdict: String,
438    /// Where `baseline` came from, so the receipt names the run it compared
439    /// against: `newest_before_apply` or `high_water` (an evalset run, named
440    /// in `baseline_run_id`), or `snapshot` — the number the proposal froze,
441    /// used when no run was journaled before the apply or the metric is not
442    /// evalset-backed. Empty on a record written before this field existed.
443    #[serde(default, skip_serializing_if = "String::is_empty")]
444    pub baseline_kind: String,
445    /// The evalset run `baseline` was read from, when it was read from one.
446    #[serde(default, skip_serializing_if = "Option::is_none")]
447    pub baseline_run_id: Option<String>,
448    /// Advisory, independent of the policy's baseline choice: the best value
449    /// the field attained on any run journaled before the apply. When the
450    /// verdict is `held` against a newer, lower baseline this is the lost
451    /// opportunity the marginal comparison cannot see.
452    #[serde(default, skip_serializing_if = "Option::is_none")]
453    pub best_before: Option<f64>,
454    /// The minimum effect size the verdict was made under, in the metric's
455    /// unit (`OutcomeEvalset::min_effect`, resolved at measure time). Absent
456    /// when zero, so a `held` under a floor is distinguishable from a `held`
457    /// at zero and a record without one reads as before.
458    #[serde(default, skip_serializing_if = "is_zero")]
459    pub tolerance: f64,
460    /// The evalset run `current` was read from, when it was read from one.
461    #[serde(default, skip_serializing_if = "Option::is_none")]
462    pub current_run_id: Option<String>,
463    /// The cost bound's reading at this checkpoint, when the policy set one.
464    /// A quality `held` with a breached bound records the verdict
465    /// `held_costlier`; `regressed` dominates.
466    #[serde(default, skip_serializing_if = "Option::is_none")]
467    pub cost: Option<CostRead>,
468    /// Which checkpoint this measurement is for (ms after apply). Zero for a
469    /// checkpoint counted in runs or grains — see `checkpoint`.
470    #[serde(default)]
471    pub horizon_ms: i64,
472    /// The checkpoint in its own unit, when that unit is not time. A time
473    /// checkpoint leaves this unset and speaks through `horizon_ms`, so every
474    /// consumer of the older shape reads an unchanged record.
475    #[serde(default, skip_serializing_if = "Option::is_none")]
476    pub checkpoint: Option<Checkpoint>,
477    pub measured_at_ms: i64,
478}
479
480/// The proposed change. Exactly one variant per recommendation.
481#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
482#[serde(tag = "proposal", rename_all = "snake_case")]
483pub enum Proposal {
484    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
485    Cal { cal: String },
486    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
487    /// staleness check at apply).
488    Edit {
489        format: String,
490        base_digest: String,
491        diff: String,
492    },
493    /// An opaque map for host targets (applied by the host, §12.3).
494    Data { data: Map<String, Value> },
495}
496
497/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
498/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
499/// fields without breaking analyzers.
500#[derive(Debug, Clone, PartialEq)]
501#[non_exhaustive]
502pub struct RecDraft {
503    pub target_ref: String,
504    pub action_kind: ActionKind,
505    pub summary: Summary,
506    pub severity: Severity,
507    pub proposal: Proposal,
508    /// Bounded to ≤64 representative evidence hashes by the engine.
509    pub evidence: Vec<String>,
510    pub evidence_query: Option<String>,
511    pub metric: Option<MetricSnapshot>,
512    pub confidence: f64,
513    pub importance: f64,
514    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
515    pub evalset_hash: Option<String>,
516}
517
518impl RecDraft {
519    pub fn new(
520        target_ref: impl Into<String>,
521        action_kind: ActionKind,
522        summary: Summary,
523        proposal: Proposal,
524    ) -> Self {
525        RecDraft {
526            target_ref: target_ref.into(),
527            action_kind,
528            summary,
529            severity: Severity::Low,
530            proposal,
531            evidence: Vec::new(),
532            evidence_query: None,
533            metric: None,
534            confidence: 0.8,
535            importance: 0.5,
536            evalset_hash: None,
537        }
538    }
539
540    pub fn severity(mut self, s: Severity) -> Self {
541        self.severity = s;
542        self
543    }
544    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
545        self.evidence = hashes;
546        self
547    }
548    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
549        self.evidence_query = Some(q.into());
550        self
551    }
552    pub fn metric(mut self, m: MetricSnapshot) -> Self {
553        self.metric = Some(m);
554        self
555    }
556    pub fn confidence(mut self, c: f64) -> Self {
557        self.confidence = c;
558        self
559    }
560    pub fn importance(mut self, i: f64) -> Self {
561        self.importance = i;
562        self
563    }
564    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
565        self.evalset_hash = Some(h.into());
566        self
567    }
568}
569
570/// Maximum representative evidence hashes carried inline (proposal §7.1).
571pub const MAX_EVIDENCE: usize = 64;
572
573/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
574/// Excludes proposal content and evidence by construction.
575pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
576    // U+001F (unit separator) cannot appear in any of the inputs.
577    format!(
578        "{}\u{1f}{}\u{1f}{}",
579        normalize_ident(family),
580        normalize_ident(target_ref),
581        action.as_str()
582    )
583}
584
585/// The dedup key of an AUTHORED executable proposal: [`dedup_key`] plus a
586/// fingerprint of the proposal's content. An analyzer finding is "this
587/// target has this kind of problem", so content is rightly excluded; an
588/// authored lesson is "do this", and two different lessons on the same
589/// entity must both reach the queue while the same lesson re-authored must
590/// not. The fingerprint is over the normalized text — case-folded, non-
591/// alphanumerics dropped, whitespace collapsed — so a rewording that changes
592/// no word is the same lesson and one that changes a word is a new one (a
593/// semantic near-duplicate is the reviewer's call, not this key's).
594pub fn authored_dedup_key(
595    family: &str,
596    target_ref: &str,
597    action: ActionKind,
598    content: &str,
599) -> String {
600    format!(
601        "{}\u{1f}{}",
602        dedup_key(family, target_ref, action),
603        content_fingerprint(content)
604    )
605}
606
607/// The dedup key of a REVERT: [`dedup_key`] plus the hash of the applied
608/// recommendation it retracts. An analyzer finding is "this target has this
609/// kind of problem", so two findings on one target rightly collapse — but a
610/// revert is about one specific applied recommendation, and two lessons on
611/// the same entity that both regressed need two reverts. Until 2026-09-06
612/// they shared a key and the second was dropped as a duplicate of the first
613/// (`crates/areev-bench/CURVE.md`, seed 3: two regressed, one revert).
614pub fn revert_dedup_key(family: &str, target_ref: &str, revert_of: &str) -> String {
615    format!(
616        "{}\u{1f}{}",
617        dedup_key(family, target_ref, ActionKind::Revert),
618        normalize_ident(revert_of)
619    )
620}
621
622/// Sixteen hex chars of FNV-1a (64-bit) over the normalized content. A dedup
623/// key needs stability and spread, not cryptographic strength — a collision
624/// here would merge two findings in a review queue, never grant anything —
625/// so this stays dependency-free, as the crate is by policy.
626pub fn content_fingerprint(content: &str) -> String {
627    let mut normalized = String::with_capacity(content.len());
628    let mut pending_space = false;
629    for c in content.chars() {
630        if c.is_alphanumeric() {
631            if pending_space && !normalized.is_empty() {
632                normalized.push(' ');
633            }
634            pending_space = false;
635            normalized.extend(c.to_lowercase());
636        } else if c.is_whitespace() || !c.is_alphanumeric() {
637            pending_space = true;
638        }
639    }
640    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
641    for b in normalized.as_bytes() {
642        h ^= u64::from(*b);
643        h = h.wrapping_mul(0x0000_0100_0000_01b3);
644    }
645    format!("{h:016x}")
646}
647
648/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
649/// content hash is stable for its whole life).
650#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
651#[serde(rename_all = "snake_case")]
652pub enum RecStatus {
653    #[default]
654    Pending,
655    Approved,
656    Rejected,
657    Applied,
658    RolledBack,
659    Expired,
660    /// The engine withdrew it: every grain it cited has moved (#317).
661    ///
662    /// Distinct from `Expired`, which says "time ran out", and from
663    /// `Rejected`, which is a human decision. A reviewer was being offered —
664    /// and could approve — findings whose entire evidence had been
665    /// superseded by a different value or retracted, and applying one then
666    /// produced a recommendation to revert it on the next pass.
667    ///
668    /// Reachable from `Pending` and `Approved`, by the ENGINE only.
669    Withdrawn,
670}
671
672impl RecStatus {
673    pub fn as_str(&self) -> &'static str {
674        match self {
675            RecStatus::Pending => "pending",
676            RecStatus::Approved => "approved",
677            RecStatus::Rejected => "rejected",
678            RecStatus::Applied => "applied",
679            RecStatus::RolledBack => "rolled_back",
680            RecStatus::Expired => "expired",
681            RecStatus::Withdrawn => "withdrawn",
682        }
683    }
684
685    /// Is a transition to `to` allowed from this state? `by_policy` marks the
686    /// auto-apply actor, the only one permitted the reasonless
687    /// `pending → applied` jump.
688    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
689        use RecStatus::*;
690        match (self, to) {
691            (Pending, Approved) | (Pending, Rejected) => true,
692            (Pending, Applied) => by_policy, // auto-apply only
693            (Approved, Applied) => true,
694            (Applied, RolledBack) => true,
695            // `expired` is computed from valid_to, applied to still-open recs.
696            (Pending, Expired) | (Approved, Expired) => true,
697            // `withdrawn` is computed from premise drift, applied to
698            // still-open recs (#317). Terminal: a withdrawn finding is not
699            // re-opened, because the evidence that is gone does not come
700            // back — a fresh finding on NEW evidence is proposed instead.
701            (Pending, Withdrawn) | (Approved, Withdrawn) => true,
702            _ => false,
703        }
704    }
705}
706
707/// Who/what performed a transition.
708#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
709#[serde(rename_all = "lowercase")]
710pub enum ObserverType {
711    Human,
712    Agent,
713    Policy,
714    System,
715}
716
717/// One immutable audit Observation per transition, hash-chained per
718/// recommendation.
719#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
720pub struct AuditRecord {
721    pub rec_hash: String,
722    #[serde(skip_serializing_if = "Option::is_none")]
723    pub from: Option<RecStatus>,
724    pub to: RecStatus,
725    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
726    /// `policy:auto`.
727    pub actor: String,
728    pub observer_type: ObserverType,
729    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
730    pub because: String,
731    #[serde(skip_serializing_if = "Option::is_none")]
732    pub previous_audit_hash: Option<String>,
733    /// §7.4: present exactly when this transition applied a code revision —
734    /// the evalset-run edge, recorded where the forensics look.
735    #[serde(default, skip_serializing_if = "Option::is_none")]
736    pub gating: Option<GatingEvidence>,
737    pub at_ms: i64,
738}
739
740/// Maximum length of a BECAUSE reason.
741pub const MAX_BECAUSE: usize = 500;
742
743impl AuditRecord {
744    /// Build the Observation grain that records this transition. `derived_from`
745    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
746    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
747        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
748        if let Some(prev) = &self.previous_audit_hash {
749            derived_from.push(Value::from(prev.clone()));
750        }
751        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
752            .with_field("observation_kind", "loop_audit")
753            .with_field("rec_hash", self.rec_hash.clone())
754            .with_field("to_status", self.to.as_str())
755            .with_field("actor", self.actor.clone())
756            .with_field(
757                "observer_type",
758                serde_json::to_value(self.observer_type).unwrap(),
759            )
760            .with_field("because", self.because.clone())
761            .with_field("at_ms", self.at_ms)
762            .with_field("derived_from", Value::Array(derived_from));
763        if let Some(from) = self.from {
764            spec.fields
765                .insert("from_status".into(), Value::from(from.as_str()));
766        }
767        if let Some(g) = &self.gating {
768            spec.fields
769                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
770            spec.fields
771                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
772            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
773            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
774        }
775        spec
776    }
777}
778
779/// A stored recommendation. `hash` (the content address) and `status` (the
780/// index-layer cache) are set by the engine, not serialized into the grain
781/// body — the body is immutable content, the lifecycle lives in the state
782/// index and the audit chain.
783/// One live lesson an authored lesson restates. `method` is `cosine`
784/// (the substrate's embedder, T1) or `jaccard` (normalized token sets, the
785/// T0 floor — weak, and honest about it).
786#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
787pub struct NearDuplicate {
788    pub hash: String,
789    pub score: f64,
790    pub method: String,
791}
792
793#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
794pub struct Recommendation {
795    #[serde(skip)]
796    pub hash: String,
797    pub analyzer: String,
798    pub params_snapshot: Map<String, Value>,
799    pub origin: Origin,
800    pub target_ref: String,
801    pub action_kind: ActionKind,
802    pub dedup_key: String,
803    pub summary: Summary,
804    pub severity: Severity,
805    #[serde(flatten)]
806    pub proposal: Proposal,
807    pub destructive: bool,
808    pub rollbackable: bool,
809    #[serde(default)]
810    pub evidence: Vec<String>,
811    #[serde(default, skip_serializing_if = "Option::is_none")]
812    pub evidence_query: Option<String>,
813    #[serde(default, skip_serializing_if = "Option::is_none")]
814    pub metric: Option<MetricSnapshot>,
815    pub confidence: f64,
816    pub importance: f64,
817    pub created_at_ms: i64,
818    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
819    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
820    /// text — it never replaces the engine-templated `summary`.
821    #[serde(default, skip_serializing_if = "Option::is_none")]
822    pub guidance: Option<String>,
823    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
824    /// against — REQUIRED for code targets, forbidden elsewhere (a
825    /// recommendation targets code XOR an evalset, never both, and only
826    /// code carries the pin). Shown at review; checked live at apply
827    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
828    #[serde(default, skip_serializing_if = "Option::is_none")]
829    pub evalset_hash: Option<String>,
830    /// Live lessons on the same entity this authored lesson restates in
831    /// other words (`Policy::near_duplicate = flag`). The reviewer's call
832    /// stays theirs — the card shows the existing rule beside the proposal.
833    #[serde(default, skip_serializing_if = "Vec::is_empty")]
834    pub near_duplicate_of: Vec<NearDuplicate>,
835    /// A `plan_revision`'s rehearsal against the journaled runs of the live
836    /// plan (the runtime's shadow report: per-run outcome under incumbent vs
837    /// candidate, effects out of support, spend) — computed at proposal
838    /// time when the substrate can, so the reviewer approves from evidence,
839    /// not prose. Absent when the substrate has no runtime or the plan has
840    /// no journaled runs.
841    #[serde(default, skip_serializing_if = "Option::is_none")]
842    pub replay: Option<Value>,
843    /// The namespaces the producing analyzer was RUN OVER (#312) —
844    /// normalized and sorted, stamped by the engine.
845    ///
846    /// The loop's inputs are namespace-grant-gated; its OUTPUTS were not. A
847    /// recommendation's summary, proposal, guidance and evidence hashes are
848    /// derived content, so one `read ON areev-loop` grant disclosed all of it
849    /// for every namespace in the memory, and one `loop.review` grant decided
850    /// all of it — including striking a memory-wide cooldown on a finding the
851    /// reviewer could not read the evidence for.
852    ///
853    /// Stamped where `dedup_key` and `origin` are, so an analyzer, an
854    /// external command or a model draft cannot set it.
855    /// `skip_serializing_if` empty, so stored recommendations read back
856    /// unchanged and an UNSCOPED pass produces an empty scope — which is
857    /// covered only by a whole-queue grant (fail closed).
858    #[serde(default, skip_serializing_if = "Vec::is_empty")]
859    pub scope: Vec<String>,
860    #[serde(skip)]
861    pub status: RecStatus,
862}
863
864/// Normalize a namespace list for [`Recommendation::scope`]: trimmed,
865/// non-empty, deduplicated, sorted — so two passes over the same set stamp
866/// the same scope and the coverage check is order-independent.
867pub fn normalize_scope(namespaces: &[String]) -> Vec<String> {
868    let mut out: Vec<String> = namespaces
869        .iter()
870        .map(|s| s.trim().to_string())
871        .filter(|s| !s.is_empty())
872        .collect();
873    out.sort();
874    out.dedup();
875    out
876}
877
878/// The recorded evalset-run edge (§7.4): what an apply of a code revision
879/// must present, and what its audit Observation carries. "Trust me, it
880/// passed" is not an edge; (evalset, run, stats) is.
881#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
882pub struct GatingEvidence {
883    /// The evalset the gate ran — must equal the recommendation's pin.
884    pub evalset_hash: String,
885    /// The `areev run`/`areev eval` run id that executed the gate.
886    pub run_id: String,
887    pub passed: u64,
888    pub failed: u64,
889}
890
891/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
892/// re-checked when a stored grain is loaded (a hand-authored grain must not
893/// bypass the rule):
894/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
895/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
896///   seam inherits §7.4's gate wholesale).
897/// - An `evalset:` target is its own always-human-approved class: never a
898///   gated revision, and never itself pinned (the gate cannot gate itself).
899/// - No other target class carries a pin.
900pub fn validate_code_rules(
901    action: ActionKind,
902    target_class: &str,
903    evalset_hash: Option<&str>,
904) -> Result<()> {
905    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
906    match (action, target_class) {
907        (ActionKind::CodeRevision, "code") => {
908            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
909                return e1(
910                    "a code_revision must pin the evalset hash it was gated \
911                     against (evalset_hash)",
912                );
913            }
914        }
915        (ActionKind::CodeRevision, other) => {
916            return e1(&format!(
917                "code_revision requires a tool: target, got class '{other}'"
918            ));
919        }
920        (ActionKind::AdapterRevision, "model") => {
921            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
922                return e1(
923                    "an adapter_revision must pin the evalset hash it was \
924                     gated against (evalset_hash)",
925                );
926            }
927        }
928        (ActionKind::AdapterRevision, other) => {
929            return e1(&format!(
930                "adapter_revision requires a model: target, got class '{other}'"
931            ));
932        }
933        // A REVERT of an applied code revision targets the same tool: —
934        // it is the rollback inverse, not new code, so it needs no pin
935        // (blocking it here would leave the highest-stakes target class
936        // unable to propose its own measured rollback).
937        (ActionKind::Revert, "code") => {
938            if evalset_hash.is_some() {
939                return e1("a code revert carries no evalset pin");
940            }
941        }
942        (_, "code") => {
943            return e1("a tool: target requires action_kind code_revision");
944        }
945        // Same rollback-inverse escape hatch for the adapter class.
946        (ActionKind::Revert, "model") => {
947            if evalset_hash.is_some() {
948                return e1("an adapter revert carries no evalset pin");
949            }
950        }
951        (_, "model") => {
952            return e1("a model: target requires action_kind adapter_revision");
953        }
954        (_, "evalset") => {
955            if evalset_hash.is_some() {
956                return e1(
957                    "an evalset-target recommendation cannot pin an evalset — \
958                     the gate cannot gate itself",
959                );
960            }
961        }
962        _ => {
963            if evalset_hash.is_some() {
964                return e1(
965                    "evalset_hash is only valid on code_revision or \
966                     adapter_revision recommendations",
967                );
968            }
969        }
970    }
971    Ok(())
972}
973
974impl Recommendation {
975    /// Serialize the immutable body into grain fields (excludes hash/status).
976    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
977        let value = serde_json::to_value(self)
978            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
979        let obj = value
980            .as_object()
981            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
982            .clone();
983        Ok(GrainSpec {
984            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
985            namespace: namespace.to_string(),
986            fields: obj,
987        })
988    }
989
990    /// Reconstruct from a stored grain body. `hash` comes from the record's
991    /// address; `status` is supplied from the state index by the caller.
992    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
993    /// must not smuggle an unpinned code revision past the stamped path.
994    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
995        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
996            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
997        rec.hash = hash.to_string();
998        let class = crate::model::TargetRef::parse(&rec.target_ref)
999            .map(|t| t.target_class())
1000            .unwrap_or("host");
1001        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
1002        Ok(rec)
1003    }
1004}
1005
1006#[cfg(test)]
1007mod tests {
1008    /// The gate and `areev eval run --tolerance` call this one function, so
1009    /// the boundary is the same for both readers by construction: a drop of
1010    /// exactly the tolerance holds, a hair past it regresses, and a tolerance
1011    /// that is not a positive finite number is zero.
1012    #[test]
1013    fn the_regression_rule_holds_at_the_tolerance_and_regresses_past_it() {
1014        use super::is_regression;
1015        // Percentage points, the way `eval run --tolerance` reads it.
1016        assert!(!is_regression(100.0, 0.0, true, 100.0), "a 100-point drop under a 100-point tolerance is accepted");
1017        assert!(is_regression(100.0, 0.0, true, 99.9));
1018        assert!(!is_regression(92.0, 91.5, true, 1.0));
1019        assert!(is_regression(92.0, 90.5, true, 1.0));
1020        // Counts, the way the gate reads `min_effect.count`.
1021        assert!(!is_regression(359.0, 355.0, true, 5.0));
1022        assert!(!is_regression(359.0, 354.0, true, 5.0), "exactly at the floor holds");
1023        assert!(is_regression(359.0, 353.0, true, 5.0));
1024        assert!(!is_regression(28.0, 33.0, false, 5.0));
1025        assert!(is_regression(28.0, 34.0, false, 5.0));
1026        // Zero is the pre-policy rule; garbage is zero, never a wider floor.
1027        assert!(is_regression(359.0, 355.0, true, 0.0));
1028        assert!(is_regression(359.0, 355.0, true, -5.0));
1029        assert!(is_regression(359.0, 355.0, true, f64::NAN));
1030        assert!(!is_regression(1.0, 1.0, true, 0.0), "equal is never a regression");
1031    }
1032
1033    use super::*;
1034    use serde_json::json;
1035
1036    #[test]
1037    fn summary_renders_deterministically() {
1038        let mut args = Map::new();
1039        args.insert("count".into(), json!(3));
1040        args.insert("subject".into(), json!("acme"));
1041        let s = Summary::new("duplicate.exact", args);
1042        assert_eq!(
1043            s.render(),
1044            "Consolidate 3 exact-duplicate grains for \"acme\""
1045        );
1046    }
1047
1048    #[test]
1049    fn dedup_key_ignores_content_and_case() {
1050        let a = dedup_key(
1051            "loop.duplicate_sweep",
1052            "entity:NS/John",
1053            ActionKind::Consolidate,
1054        );
1055        let b = dedup_key(
1056            "loop.duplicate_sweep",
1057            "entity:ns/john",
1058            ActionKind::Consolidate,
1059        );
1060        assert_eq!(a, b, "case-folded to one identity");
1061    }
1062
1063    #[test]
1064    fn authored_dedup_key_distinguishes_content_but_not_wording_noise() {
1065        let a = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the vendor name.");
1066        let same = authored_dedup_key("llm", "entity:NS/X", ActionKind::Record, "  record THE vendor  name ");
1067        let other = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the amount.");
1068        assert_eq!(a, same, "case, punctuation and spacing are not a new lesson");
1069        assert_ne!(a, other, "a different lesson on the same entity is a different finding");
1070        assert!(a.starts_with(&dedup_key("llm", "entity:ns/x", ActionKind::Record)));
1071        assert_eq!(content_fingerprint("A b"), content_fingerprint("a-b"));
1072        assert_ne!(content_fingerprint("ab"), content_fingerprint("a b"));
1073    }
1074
1075    #[test]
1076    fn a_revert_is_keyed_by_what_it_reverts() {
1077        let a = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "aaaa");
1078        let b = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "bbbb");
1079        let a2 = revert_dedup_key("loop.outcome_review", "entity:NS/Capture", "AAAA");
1080        assert_ne!(a, b, "two reverts on one target are two findings");
1081        assert_eq!(a, a2, "the same revert, case-folded, is one");
1082        assert!(a.starts_with(&dedup_key("loop.outcome_review", "entity:ns/capture", ActionKind::Revert)));
1083    }
1084
1085    #[test]
1086    fn dedup_key_distinguishes_action() {
1087        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
1088        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
1089        assert_ne!(a, b);
1090    }
1091
1092    #[test]
1093    fn lifecycle_gates_pending_to_applied() {
1094        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
1095        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
1096        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
1097        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
1098        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
1099        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
1100    }
1101
1102    #[test]
1103    fn rule_e1_pins_code_and_only_code() {
1104        use crate::model::ActionKind as A;
1105        // code_revision on a tool target with a pin: OK.
1106        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
1107        // Unpinned code revision: refused.
1108        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
1109        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
1110        // code_revision off a tool target: refused.
1111        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
1112        // A tool target with a non-code action: refused.
1113        assert!(validate_code_rules(A::Flag, "code", None).is_err());
1114        // The gate cannot gate itself.
1115        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
1116        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
1117        // Pins don't leak onto ordinary targets.
1118        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
1119        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
1120    }
1121
1122    #[test]
1123    fn rule_e1_pins_adapter_revisions_like_code() {
1124        use crate::model::ActionKind as A;
1125        // adapter_revision on a model target with a pin: OK.
1126        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
1127        // Unpinned adapter revision: refused.
1128        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
1129        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
1130        // adapter_revision off a model target: refused.
1131        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
1132        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
1133        // A model target with a non-adapter action: refused.
1134        assert!(validate_code_rules(A::Flag, "model", None).is_err());
1135        // The rollback inverse needs no pin — and must carry none.
1136        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
1137        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
1138    }
1139
1140    #[test]
1141    fn from_fields_rechecks_rule_e1() {
1142        // A hand-authored grain body carrying an unpinned code revision must
1143        // not decode into a usable recommendation.
1144        let mut rec = Recommendation {
1145            hash: String::new(),
1146            analyzer: "loop.codegen/1".into(),
1147            params_snapshot: Map::new(),
1148            origin: Origin::Builtin,
1149            target_ref: "tool:abc123".into(),
1150            action_kind: ActionKind::CodeRevision,
1151            dedup_key: "k".into(),
1152            summary: Summary::new("command.finding", Map::new()),
1153            severity: Severity::Low,
1154            proposal: Proposal::Data { data: Map::new() },
1155            destructive: false,
1156            rollbackable: false,
1157            evidence: vec![],
1158            evidence_query: None,
1159            metric: None,
1160            confidence: 0.5,
1161            importance: 0.5,
1162            created_at_ms: 0,
1163            guidance: None,
1164            evalset_hash: None, // the smuggle attempt
1165            near_duplicate_of: Vec::new(),
1166            replay: None,
1167            status: RecStatus::Pending,
1168            scope: Vec::new(),
1169        };
1170        let spec = rec.to_grain_spec("ns").unwrap();
1171        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
1172        rec.evalset_hash = Some("es-hash".into());
1173        let spec = rec.to_grain_spec("ns").unwrap();
1174        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
1175        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
1176    }
1177
1178    #[test]
1179    fn recommendation_round_trips_through_fields() {
1180        let rec = Recommendation {
1181            hash: "ignored".into(),
1182            analyzer: "loop.staleness/1".into(),
1183            params_snapshot: Map::new(),
1184            origin: Origin::Builtin,
1185            target_ref: "grain:sha256:abc".into(),
1186            action_kind: ActionKind::Expire,
1187            dedup_key: "k".into(),
1188            summary: Summary::new("staleness.expired", Map::new()),
1189            severity: Severity::Low,
1190            proposal: Proposal::Cal {
1191                cal: "FORGET sha256:abc".into(),
1192            },
1193            destructive: true,
1194            rollbackable: false,
1195            evidence: vec!["sha256:abc".into()],
1196            evidence_query: None,
1197            metric: None,
1198            confidence: 0.9,
1199            importance: 0.4,
1200            created_at_ms: 1000,
1201            guidance: None,
1202            evalset_hash: None,
1203            near_duplicate_of: Vec::new(),
1204            replay: None,
1205            status: RecStatus::Pending,
1206            scope: Vec::new(),
1207        };
1208        let spec = rec.to_grain_spec("ns").unwrap();
1209        // hash and status are excluded from the immutable body.
1210        assert!(!spec.fields.contains_key("hash"));
1211        assert!(!spec.fields.contains_key("status"));
1212        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
1213        assert_eq!(back.hash, "realhash");
1214        assert_eq!(back.analyzer, "loop.staleness/1");
1215        assert!(back.destructive);
1216        assert!(matches!(back.proposal, Proposal::Cal { .. }));
1217    }
1218}