Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel —
9//!   for ANALYZER findings. An authored (`origin = llm`) executable proposal
10//!   keys on a fingerprint of its content as well, because there the content
11//!   IS the finding: two different lessons on one entity are two findings,
12//!   and the same lesson re-authored is one.
13//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
14
15use crate::error::{Error, Result};
16use crate::model::{normalize_ident, ActionKind, Origin, Severity};
17use crate::substrate::GrainSpec;
18use serde::{Deserialize, Serialize};
19use serde_json::{Map, Value};
20
21/// A deterministic, template-rendered summary. The analyzer chooses a
22/// `template_id` and supplies `args`; the text is produced here, so an
23/// analyzer can never emit arbitrary prose into the queue.
24#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
25pub struct Summary {
26    pub template_id: String,
27    #[serde(default)]
28    pub args: Map<String, Value>,
29}
30
31impl Summary {
32    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
33        Summary {
34            template_id: template_id.into(),
35            args,
36        }
37    }
38
39    /// Render the summary. Unknown template ids fall back to a stable, honest
40    /// string (never a panic) so a manifest/template mismatch is visible, not
41    /// fatal.
42    pub fn render(&self) -> String {
43        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
44        interpolate(t, &self.args, &self.template_id)
45    }
46}
47
48/// The built-in template table. Deterministic; the only place summary prose
49/// lives. Keep placeholders in `{name}` form matching `args` keys.
50fn builtin_template(id: &str) -> Option<&'static str> {
51    Some(match id {
52        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
53        "duplicate.near" => {
54            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
55        }
56        "contradiction.functional" => {
57            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
58        }
59        "tool_failure.cluster" => {
60            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
61             fail this way): {signature}"
62        }
63        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
64        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
65        "skill.stall" => {
66            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
67        }
68        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
69        "cold.grain" => {
70            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
71        }
72        "coverage.gap" => {
73            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
74        }
75        "budget.pressure" => {
76            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
77        }
78        "retention.overdue" => {
79            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
80        }
81        "outcome.regression" => {
82            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
83        }
84        // The high-water baseline names both figures — the best run before
85        // the apply and the run that fell from it — because the reviewer has
86        // to judge whether THIS rule caused the whole fall from the peak.
87        "outcome.regression_high_water" => {
88            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}"
89        }
90        // A regression that also breached the cost bound: the revert names
91        // the cost delta so the reviewer sees both halves of the damage.
92        "outcome.regression_costlier" => {
93            "Applied recommendation regressed: {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
94        }
95        "outcome.regression_high_water_costlier" => {
96            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
97        }
98        // Quality held, cost did not: advisory only — a cost/quality trade
99        // is a human decision, so this is a Flag, never a revert draft.
100        "outcome.held_costlier" => {
101            "Applied recommendation held ({metric} {baseline} → {current}) but {cost_field} rose {cost_baseline} → {cost_current}, past ×{cost_ratio} of the baseline run ({baseline_run} → {current_run}) — a cost/quality trade to decide"
102        }
103        "outcome.premise_drift" => {
104            "{current} of the grains this recommendation cited have since been superseded by a different value or retracted — its premise moved; revert it"
105        }
106        "run.failures" => {
107            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
108        }
109        "run.cost" => {
110            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
111        }
112        "run.context_pressure" => {
113            "Workflow {workflow} summarized its own transcript {folds} time(s) across {runs} runs (avg {avg_folds}/run) — its nodes are outgrowing the model's window"
114        }
115        "adapter.candidate" => {
116            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
117        }
118        // origin=llm drafts carry free text (clearly marked llm) rather than a
119        // deterministic template — the model's proposed summary rides in {text}.
120        "llm.discover" => "{text}",
121        // An LLM-authored lesson shows the reviewer BOTH the finding and the
122        // exact line an apply would record — never one without the other.
123        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
124        // The same lesson, flagged as restating a live one: the reviewer sees
125        // that a rule of this meaning already exists before approving a
126        // second copy of it.
127        "llm.lesson_near_duplicate" => {
128            "{text} — record lesson: \"{lesson}\" — NEAR-DUPLICATE of {near_count} live lesson(s) on this entity (best match {near_score} by {near_method}, {near_hash})"
129        }
130        // A consolidation: one lesson replacing a pile. The apply supersedes
131        // every member and adds the one line; rollback restores them all.
132        "llm.consolidation" => {
133            "{text} — consolidate {count} lessons into one: \"{lesson}\""
134        }
135        "lesson.pile" => {
136            "{count} live lessons on \"{subject}\" exceed the budget of {max_active} — every rule competes for the model's attention; consolidate or retire: {members}"
137        }
138        // The rest of the LLM proposal vocabulary follows the same rule: the
139        // finding AND the exact change an apply would make, never one without
140        // the other. A reviewer approving blind is the failure this prevents.
141        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
142        "llm.skill" => "{text} — record skill: \"{name}\" ({steps} steps)",
143        "llm.plan" => "{text} — record plan: \"{name}\" ({nodes} steps, {edges} edges)",
144        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
145        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
146        // The rehearsal refused it: the edit is shown, the reason names the
147        // runs, and nothing is offered to apply.
148        "llm.plan_revision_refused" => {
149            "{text} — plan revision of {plan} ({edits}) NOT applicable: rehearsed against the plan's journaled runs and {reason}"
150        }
151        "llm.code_revision" => {
152            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
153        }
154        // External command analyzers (trust class Command): free text from the
155        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
156        "command.finding" => "{text}",
157        _ => return None,
158    })
159}
160
161fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
162    let mut out = String::with_capacity(template.len());
163    let mut chars = template.chars().peekable();
164    while let Some(c) = chars.next() {
165        if c == '{' {
166            let mut key = String::new();
167            for k in chars.by_ref() {
168                if k == '}' {
169                    break;
170                }
171                key.push(k);
172            }
173            if key == "template_id" {
174                out.push_str(template_id);
175            } else {
176                match args.get(&key) {
177                    Some(Value::String(s)) => out.push_str(s),
178                    Some(v) => out.push_str(&v.to_string()),
179                    None => {
180                        // Missing arg — leave a visible marker rather than lie.
181                        out.push('{');
182                        out.push_str(&key);
183                        out.push('}');
184                    }
185                }
186            }
187        } else {
188            out.push(c);
189        }
190    }
191    out
192}
193
194/// When an applied recommendation is re-measured — one checkpoint of the
195/// Verify gate's schedule, in the unit the deployment actually counts in.
196///
197/// Three units, because deployments count differently and a fixed one makes
198/// the gate inert everywhere else. A service evaluated nightly counts
199/// **time** (`after_ms`). A benchmark or a CI harness counts **graded runs**
200/// (`after_runs`): "measure at the next evaluation after the apply", however
201/// long that takes on the clock. A chat agent counts **turns** (`after_grains`):
202/// "after fifty more grains". The default schedule stayed ms-only for a year
203/// and was measured, on PAST-Bench, to fire exactly zero verdicts across 78
204/// governed runs — a family finishes in seven minutes and the first checkpoint
205/// was a day away (`crates/areev-bench/PERSIST.md`).
206///
207/// On the wire a bare integer is milliseconds, so every policy, metric
208/// snapshot and state blob written before this type existed reads back
209/// unchanged, and an all-ms schedule still serializes as `[86400000, …]`.
210#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
211pub enum Checkpoint {
212    /// Elapsed time since the apply.
213    AfterMs(i64),
214    /// Evalset runs journaled since the apply. Only an `evalset:` metric can
215    /// count these; on any other metric the checkpoint never comes due.
216    AfterRuns(u32),
217    /// Grains written since the apply — the loop's own notion of "activity".
218    AfterGrains(u32),
219}
220
221impl Checkpoint {
222    /// The label a person reads: `1d`, `12h`, `3 runs`, `50 grains`.
223    pub fn label(&self) -> String {
224        match self {
225            Checkpoint::AfterMs(ms) if ms % 86_400_000 == 0 => format!("{}d", ms / 86_400_000),
226            Checkpoint::AfterMs(ms) if ms % 3_600_000 == 0 => format!("{}h", ms / 3_600_000),
227            Checkpoint::AfterMs(ms) => format!("{ms}ms"),
228            Checkpoint::AfterRuns(n) => format!("{n} run{}", if *n == 1 { "" } else { "s" }),
229            Checkpoint::AfterGrains(n) => format!("{n} grain{}", if *n == 1 { "" } else { "s" }),
230        }
231    }
232
233    /// The ms value when this is a time checkpoint — what legacy consumers of
234    /// `OutcomeResult::horizon_ms` read.
235    pub fn as_ms(&self) -> Option<i64> {
236        match self {
237            Checkpoint::AfterMs(ms) => Some(*ms),
238            _ => None,
239        }
240    }
241}
242
243impl Serialize for Checkpoint {
244    fn serialize<S: serde::Serializer>(&self, ser: S) -> std::result::Result<S::Ok, S::Error> {
245        use serde::ser::SerializeMap;
246        match self {
247            // Bare integer: byte-identical to the pre-Checkpoint `Vec<i64>`.
248            Checkpoint::AfterMs(ms) => ser.serialize_i64(*ms),
249            Checkpoint::AfterRuns(n) => {
250                let mut m = ser.serialize_map(Some(1))?;
251                m.serialize_entry("after_runs", n)?;
252                m.end()
253            }
254            Checkpoint::AfterGrains(n) => {
255                let mut m = ser.serialize_map(Some(1))?;
256                m.serialize_entry("after_grains", n)?;
257                m.end()
258            }
259        }
260    }
261}
262
263impl<'de> Deserialize<'de> for Checkpoint {
264    fn deserialize<D: serde::Deserializer<'de>>(de: D) -> std::result::Result<Self, D::Error> {
265        use serde::de::Error as _;
266        let v = Value::deserialize(de)?;
267        match &v {
268            Value::Number(n) => n
269                .as_i64()
270                .filter(|ms| *ms >= 0)
271                .map(Checkpoint::AfterMs)
272                .ok_or_else(|| D::Error::custom("checkpoint: a bare number is non-negative milliseconds")),
273            Value::Object(m) if m.len() == 1 => {
274                let (k, val) = m.iter().next().unwrap();
275                let n = val.as_u64().ok_or_else(|| {
276                    D::Error::custom(format!("checkpoint: {k} takes a non-negative integer"))
277                })?;
278                match k.as_str() {
279                    "after_ms" => i64::try_from(n)
280                        .map(Checkpoint::AfterMs)
281                        .map_err(|_| D::Error::custom("checkpoint: after_ms out of range")),
282                    "after_runs" => u32::try_from(n)
283                        .map(Checkpoint::AfterRuns)
284                        .map_err(|_| D::Error::custom("checkpoint: after_runs out of range")),
285                    "after_grains" => u32::try_from(n)
286                        .map(Checkpoint::AfterGrains)
287                        .map_err(|_| D::Error::custom("checkpoint: after_grains out of range")),
288                    other => Err(D::Error::custom(format!(
289                        "checkpoint: unknown unit {other:?} (expected after_ms, after_runs or after_grains)"
290                    ))),
291                }
292            }
293            _ => Err(D::Error::custom(
294                "checkpoint: expected milliseconds or {\"after_ms\"|\"after_runs\"|\"after_grains\": n}",
295            )),
296        }
297    }
298}
299
300/// A reproducible metric snapshot; powers outcome review.
301#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
302pub struct MetricSnapshot {
303    /// Metric kind — the engine knows how to re-measure a fixed set
304    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
305    pub metric: String,
306    pub baseline: f64,
307    pub unit: String,
308    pub n: u64,
309    pub window: String,
310    /// The subject the metric is about (e.g. the tool name), used by the
311    /// engine's typed re-measurement.
312    #[serde(default, skip_serializing_if = "Option::is_none")]
313    pub subject: Option<String>,
314    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
315    /// `None` = all namespaces. Additive: older snapshots deserialize to
316    /// `None` and existing metric kinds ignore it.
317    #[serde(default, skip_serializing_if = "Option::is_none")]
318    pub namespace: Option<String>,
319    /// Relation scope for fact-shaped metrics (e.g. which functional relation
320    /// a contradiction was resolved under), normalized. Additive like
321    /// `namespace`.
322    #[serde(default, skip_serializing_if = "Option::is_none")]
323    pub relation: Option<String>,
324    /// CAL that recomputes the metric at verify time (reproducibility /
325    /// documentation; the engine re-measures with typed reads).
326    pub query: String,
327    /// How long after apply to re-measure, in epoch-ms delta (the first / only
328    /// checkpoint when `horizons_ms` is empty).
329    pub review_after_ms: i64,
330    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
331    /// that `held` at an early checkpoint can `regress` at a later one, so a
332    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
333    #[serde(default, skip_serializing_if = "Vec::is_empty")]
334    pub horizons_ms: Vec<i64>,
335    /// The schedule in the deployment's own unit ([`Checkpoint`]). When set it
336    /// is THE schedule and `horizons_ms`/`review_after_ms` are ignored; empty
337    /// (every snapshot written before it existed) falls back to them.
338    #[serde(default, skip_serializing_if = "Vec::is_empty")]
339    pub checkpoints: Vec<Checkpoint>,
340    /// Which direction is an improvement. The built-in metrics are all
341    /// recurrence counts, where lower is better — so this defaults to `false`
342    /// and every existing snapshot deserializes unchanged. An evalset accuracy
343    /// is the opposite, and getting it wrong does not merely misreport: the
344    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
345    /// propose reverting it.
346    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
347    pub higher_is_better: bool,
348}
349
350/// The one regression rule.
351///
352/// The verdict is computed in three places — by the engine for the recorded
353/// `OutcomeResult`, by `outcome_review` when it drafts the revert, and by
354/// `areev eval run --tolerance` when it re-accepts a model swap — and any two
355/// silently disagreeing would either revert a change that held or sit on one
356/// that regressed. So all of them call this.
357///
358/// `tolerance` is the policy's minimum effect size in the metric's own unit
359/// (`OutcomeEvalset::min_effect`, or `--tolerance` in percentage points of
360/// the pass rate): a worsening of at most that much is not a regression. It
361/// is a **floor, not a significance test** — no p-value, no interval; a host
362/// that needs statistics has the run counts to compute them. Zero is the
363/// pre-policy behaviour: any drop is a regression. `EPSILON` is only
364/// floating-point equality slack, so a value read twice that differs in the
365/// last bit does not flip the verdict; it is not a noise floor.
366pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool, tolerance: f64) -> bool {
367    const EPSILON: f64 = 1e-9;
368    let tolerance = if tolerance.is_finite() && tolerance > 0.0 { tolerance } else { 0.0 };
369    if higher_is_better {
370        current < baseline - tolerance - EPSILON
371    } else {
372        current > baseline + tolerance + EPSILON
373    }
374}
375
376impl MetricSnapshot {
377    /// The measurement schedule, sorted and deduplicated: `checkpoints` when
378    /// set, else `horizons_ms`, else the single `review_after_ms` — each older
379    /// spelling read as time, so a snapshot from before [`Checkpoint`] existed
380    /// measures exactly as it did.
381    pub fn schedule(&self) -> Vec<Checkpoint> {
382        let mut h: Vec<Checkpoint> = if !self.checkpoints.is_empty() {
383            self.checkpoints.clone()
384        } else if !self.horizons_ms.is_empty() {
385            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
386        } else {
387            vec![Checkpoint::AfterMs(self.review_after_ms)]
388        };
389        h.sort_unstable();
390        h.dedup();
391        h
392    }
393
394    /// The time checkpoints of the schedule, in ms — the pre-[`Checkpoint`]
395    /// view, kept for callers that only understand time.
396    pub fn horizons(&self) -> Vec<i64> {
397        self.schedule().iter().filter_map(Checkpoint::as_ms).collect()
398    }
399}
400
401fn is_zero(x: &f64) -> bool {
402    *x == 0.0
403}
404
405/// What a checkpoint read of the policy's cost bound, beside the quality
406/// verdict. `status` is `within`, `breached`, or `not_measurable` (the
407/// field was absent or malformed on either run — then `baseline`/`current`
408/// carry whichever side did measure, or nothing). Present on a record only
409/// when the policy set a bound.
410#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
411pub struct CostRead {
412    pub field: String,
413    pub max_increase_ratio: f64,
414    #[serde(default, skip_serializing_if = "Option::is_none")]
415    pub baseline: Option<f64>,
416    #[serde(default, skip_serializing_if = "Option::is_none")]
417    pub current: Option<f64>,
418    pub status: String,
419}
420
421impl CostRead {
422    pub fn breached(&self) -> bool {
423        self.status == "breached"
424    }
425}
426
427/// A measured outcome for an applied recommendation at one checkpoint — the
428/// Verify gate's output. `held` = the metric did not regress at this horizon;
429/// `regressed` = it got worse (a revert is proposed). A recommendation
430/// accumulates one of these per horizon, forming a time series.
431#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
432pub struct OutcomeResult {
433    pub rec_hash: String,
434    pub metric: String,
435    pub baseline: f64,
436    pub current: f64,
437    pub verdict: String,
438    /// Where `baseline` came from, so the receipt names the run it compared
439    /// against: `newest_before_apply` or `high_water` (an evalset run, named
440    /// in `baseline_run_id`), or `snapshot` — the number the proposal froze,
441    /// used when no run was journaled before the apply or the metric is not
442    /// evalset-backed. Empty on a record written before this field existed.
443    #[serde(default, skip_serializing_if = "String::is_empty")]
444    pub baseline_kind: String,
445    /// The evalset run `baseline` was read from, when it was read from one.
446    #[serde(default, skip_serializing_if = "Option::is_none")]
447    pub baseline_run_id: Option<String>,
448    /// Advisory, independent of the policy's baseline choice: the best value
449    /// the field attained on any run journaled before the apply. When the
450    /// verdict is `held` against a newer, lower baseline this is the lost
451    /// opportunity the marginal comparison cannot see.
452    #[serde(default, skip_serializing_if = "Option::is_none")]
453    pub best_before: Option<f64>,
454    /// The minimum effect size the verdict was made under, in the metric's
455    /// unit (`OutcomeEvalset::min_effect`, resolved at measure time). Absent
456    /// when zero, so a `held` under a floor is distinguishable from a `held`
457    /// at zero and a record without one reads as before.
458    #[serde(default, skip_serializing_if = "is_zero")]
459    pub tolerance: f64,
460    /// The evalset run `current` was read from, when it was read from one.
461    #[serde(default, skip_serializing_if = "Option::is_none")]
462    pub current_run_id: Option<String>,
463    /// The cost bound's reading at this checkpoint, when the policy set one.
464    /// A quality `held` with a breached bound records the verdict
465    /// `held_costlier`; `regressed` dominates.
466    #[serde(default, skip_serializing_if = "Option::is_none")]
467    pub cost: Option<CostRead>,
468    /// Which checkpoint this measurement is for (ms after apply). Zero for a
469    /// checkpoint counted in runs or grains — see `checkpoint`.
470    #[serde(default)]
471    pub horizon_ms: i64,
472    /// The checkpoint in its own unit, when that unit is not time. A time
473    /// checkpoint leaves this unset and speaks through `horizon_ms`, so every
474    /// consumer of the older shape reads an unchanged record.
475    #[serde(default, skip_serializing_if = "Option::is_none")]
476    pub checkpoint: Option<Checkpoint>,
477    pub measured_at_ms: i64,
478}
479
480/// The proposed change. Exactly one variant per recommendation.
481#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
482#[serde(tag = "proposal", rename_all = "snake_case")]
483pub enum Proposal {
484    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
485    Cal { cal: String },
486    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
487    /// staleness check at apply).
488    Edit {
489        format: String,
490        base_digest: String,
491        diff: String,
492    },
493    /// An opaque map for host targets (applied by the host, §12.3).
494    Data { data: Map<String, Value> },
495}
496
497/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
498/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
499/// fields without breaking analyzers.
500#[derive(Debug, Clone, PartialEq)]
501#[non_exhaustive]
502pub struct RecDraft {
503    pub target_ref: String,
504    pub action_kind: ActionKind,
505    pub summary: Summary,
506    pub severity: Severity,
507    pub proposal: Proposal,
508    /// Bounded to ≤64 representative evidence hashes by the engine.
509    pub evidence: Vec<String>,
510    pub evidence_query: Option<String>,
511    pub metric: Option<MetricSnapshot>,
512    pub confidence: f64,
513    pub importance: f64,
514    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
515    pub evalset_hash: Option<String>,
516}
517
518impl RecDraft {
519    pub fn new(
520        target_ref: impl Into<String>,
521        action_kind: ActionKind,
522        summary: Summary,
523        proposal: Proposal,
524    ) -> Self {
525        RecDraft {
526            target_ref: target_ref.into(),
527            action_kind,
528            summary,
529            severity: Severity::Low,
530            proposal,
531            evidence: Vec::new(),
532            evidence_query: None,
533            metric: None,
534            confidence: 0.8,
535            importance: 0.5,
536            evalset_hash: None,
537        }
538    }
539
540    pub fn severity(mut self, s: Severity) -> Self {
541        self.severity = s;
542        self
543    }
544    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
545        self.evidence = hashes;
546        self
547    }
548    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
549        self.evidence_query = Some(q.into());
550        self
551    }
552    pub fn metric(mut self, m: MetricSnapshot) -> Self {
553        self.metric = Some(m);
554        self
555    }
556    pub fn confidence(mut self, c: f64) -> Self {
557        self.confidence = c;
558        self
559    }
560    pub fn importance(mut self, i: f64) -> Self {
561        self.importance = i;
562        self
563    }
564    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
565        self.evalset_hash = Some(h.into());
566        self
567    }
568}
569
570/// Maximum representative evidence hashes carried inline (proposal §7.1).
571pub const MAX_EVIDENCE: usize = 64;
572
573/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
574/// Excludes proposal content and evidence by construction.
575pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
576    // U+001F (unit separator) cannot appear in any of the inputs.
577    format!(
578        "{}\u{1f}{}\u{1f}{}",
579        normalize_ident(family),
580        normalize_ident(target_ref),
581        action.as_str()
582    )
583}
584
585/// The dedup key of an AUTHORED executable proposal: [`dedup_key`] plus a
586/// fingerprint of the proposal's content. An analyzer finding is "this
587/// target has this kind of problem", so content is rightly excluded; an
588/// authored lesson is "do this", and two different lessons on the same
589/// entity must both reach the queue while the same lesson re-authored must
590/// not. The fingerprint is over the normalized text — case-folded, non-
591/// alphanumerics dropped, whitespace collapsed — so a rewording that changes
592/// no word is the same lesson and one that changes a word is a new one (a
593/// semantic near-duplicate is the reviewer's call, not this key's).
594pub fn authored_dedup_key(
595    family: &str,
596    target_ref: &str,
597    action: ActionKind,
598    content: &str,
599) -> String {
600    format!(
601        "{}\u{1f}{}",
602        dedup_key(family, target_ref, action),
603        content_fingerprint(content)
604    )
605}
606
607/// The dedup key of a REVERT: [`dedup_key`] plus the hash of the applied
608/// recommendation it retracts. An analyzer finding is "this target has this
609/// kind of problem", so two findings on one target rightly collapse — but a
610/// revert is about one specific applied recommendation, and two lessons on
611/// the same entity that both regressed need two reverts. Until 2026-09-06
612/// they shared a key and the second was dropped as a duplicate of the first
613/// (`crates/areev-bench/CURVE.md`, seed 3: two regressed, one revert).
614pub fn revert_dedup_key(family: &str, target_ref: &str, revert_of: &str) -> String {
615    format!(
616        "{}\u{1f}{}",
617        dedup_key(family, target_ref, ActionKind::Revert),
618        normalize_ident(revert_of)
619    )
620}
621
622/// Sixteen hex chars of FNV-1a (64-bit) over the normalized content. A dedup
623/// key needs stability and spread, not cryptographic strength — a collision
624/// here would merge two findings in a review queue, never grant anything —
625/// so this stays dependency-free, as the crate is by policy.
626pub fn content_fingerprint(content: &str) -> String {
627    let mut normalized = String::with_capacity(content.len());
628    let mut pending_space = false;
629    for c in content.chars() {
630        if c.is_alphanumeric() {
631            if pending_space && !normalized.is_empty() {
632                normalized.push(' ');
633            }
634            pending_space = false;
635            normalized.extend(c.to_lowercase());
636        } else if c.is_whitespace() || !c.is_alphanumeric() {
637            pending_space = true;
638        }
639    }
640    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
641    for b in normalized.as_bytes() {
642        h ^= u64::from(*b);
643        h = h.wrapping_mul(0x0000_0100_0000_01b3);
644    }
645    format!("{h:016x}")
646}
647
648/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
649/// content hash is stable for its whole life).
650#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
651#[serde(rename_all = "snake_case")]
652pub enum RecStatus {
653    #[default]
654    Pending,
655    Approved,
656    Rejected,
657    Applied,
658    RolledBack,
659    Expired,
660}
661
662impl RecStatus {
663    pub fn as_str(&self) -> &'static str {
664        match self {
665            RecStatus::Pending => "pending",
666            RecStatus::Approved => "approved",
667            RecStatus::Rejected => "rejected",
668            RecStatus::Applied => "applied",
669            RecStatus::RolledBack => "rolled_back",
670            RecStatus::Expired => "expired",
671        }
672    }
673
674    /// Is a transition to `to` allowed from this state? `by_policy` marks the
675    /// auto-apply actor, the only one permitted the reasonless
676    /// `pending → applied` jump.
677    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
678        use RecStatus::*;
679        match (self, to) {
680            (Pending, Approved) | (Pending, Rejected) => true,
681            (Pending, Applied) => by_policy, // auto-apply only
682            (Approved, Applied) => true,
683            (Applied, RolledBack) => true,
684            // `expired` is computed from valid_to, applied to still-open recs.
685            (Pending, Expired) | (Approved, Expired) => true,
686            _ => false,
687        }
688    }
689}
690
691/// Who/what performed a transition.
692#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
693#[serde(rename_all = "lowercase")]
694pub enum ObserverType {
695    Human,
696    Agent,
697    Policy,
698    System,
699}
700
701/// One immutable audit Observation per transition, hash-chained per
702/// recommendation.
703#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
704pub struct AuditRecord {
705    pub rec_hash: String,
706    #[serde(skip_serializing_if = "Option::is_none")]
707    pub from: Option<RecStatus>,
708    pub to: RecStatus,
709    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
710    /// `policy:auto`.
711    pub actor: String,
712    pub observer_type: ObserverType,
713    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
714    pub because: String,
715    #[serde(skip_serializing_if = "Option::is_none")]
716    pub previous_audit_hash: Option<String>,
717    /// §7.4: present exactly when this transition applied a code revision —
718    /// the evalset-run edge, recorded where the forensics look.
719    #[serde(default, skip_serializing_if = "Option::is_none")]
720    pub gating: Option<GatingEvidence>,
721    pub at_ms: i64,
722}
723
724/// Maximum length of a BECAUSE reason.
725pub const MAX_BECAUSE: usize = 500;
726
727impl AuditRecord {
728    /// Build the Observation grain that records this transition. `derived_from`
729    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
730    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
731        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
732        if let Some(prev) = &self.previous_audit_hash {
733            derived_from.push(Value::from(prev.clone()));
734        }
735        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
736            .with_field("observation_kind", "loop_audit")
737            .with_field("rec_hash", self.rec_hash.clone())
738            .with_field("to_status", self.to.as_str())
739            .with_field("actor", self.actor.clone())
740            .with_field(
741                "observer_type",
742                serde_json::to_value(self.observer_type).unwrap(),
743            )
744            .with_field("because", self.because.clone())
745            .with_field("at_ms", self.at_ms)
746            .with_field("derived_from", Value::Array(derived_from));
747        if let Some(from) = self.from {
748            spec.fields
749                .insert("from_status".into(), Value::from(from.as_str()));
750        }
751        if let Some(g) = &self.gating {
752            spec.fields
753                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
754            spec.fields
755                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
756            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
757            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
758        }
759        spec
760    }
761}
762
763/// A stored recommendation. `hash` (the content address) and `status` (the
764/// index-layer cache) are set by the engine, not serialized into the grain
765/// body — the body is immutable content, the lifecycle lives in the state
766/// index and the audit chain.
767/// One live lesson an authored lesson restates. `method` is `cosine`
768/// (the substrate's embedder, T1) or `jaccard` (normalized token sets, the
769/// T0 floor — weak, and honest about it).
770#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
771pub struct NearDuplicate {
772    pub hash: String,
773    pub score: f64,
774    pub method: String,
775}
776
777#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
778pub struct Recommendation {
779    #[serde(skip)]
780    pub hash: String,
781    pub analyzer: String,
782    pub params_snapshot: Map<String, Value>,
783    pub origin: Origin,
784    pub target_ref: String,
785    pub action_kind: ActionKind,
786    pub dedup_key: String,
787    pub summary: Summary,
788    pub severity: Severity,
789    #[serde(flatten)]
790    pub proposal: Proposal,
791    pub destructive: bool,
792    pub rollbackable: bool,
793    #[serde(default)]
794    pub evidence: Vec<String>,
795    #[serde(default, skip_serializing_if = "Option::is_none")]
796    pub evidence_query: Option<String>,
797    #[serde(default, skip_serializing_if = "Option::is_none")]
798    pub metric: Option<MetricSnapshot>,
799    pub confidence: f64,
800    pub importance: f64,
801    pub created_at_ms: i64,
802    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
803    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
804    /// text — it never replaces the engine-templated `summary`.
805    #[serde(default, skip_serializing_if = "Option::is_none")]
806    pub guidance: Option<String>,
807    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
808    /// against — REQUIRED for code targets, forbidden elsewhere (a
809    /// recommendation targets code XOR an evalset, never both, and only
810    /// code carries the pin). Shown at review; checked live at apply
811    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
812    #[serde(default, skip_serializing_if = "Option::is_none")]
813    pub evalset_hash: Option<String>,
814    /// Live lessons on the same entity this authored lesson restates in
815    /// other words (`Policy::near_duplicate = flag`). The reviewer's call
816    /// stays theirs — the card shows the existing rule beside the proposal.
817    #[serde(default, skip_serializing_if = "Vec::is_empty")]
818    pub near_duplicate_of: Vec<NearDuplicate>,
819    /// A `plan_revision`'s rehearsal against the journaled runs of the live
820    /// plan (the runtime's shadow report: per-run outcome under incumbent vs
821    /// candidate, effects out of support, spend) — computed at proposal
822    /// time when the substrate can, so the reviewer approves from evidence,
823    /// not prose. Absent when the substrate has no runtime or the plan has
824    /// no journaled runs.
825    #[serde(default, skip_serializing_if = "Option::is_none")]
826    pub replay: Option<Value>,
827    #[serde(skip)]
828    pub status: RecStatus,
829}
830
831/// The recorded evalset-run edge (§7.4): what an apply of a code revision
832/// must present, and what its audit Observation carries. "Trust me, it
833/// passed" is not an edge; (evalset, run, stats) is.
834#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
835pub struct GatingEvidence {
836    /// The evalset the gate ran — must equal the recommendation's pin.
837    pub evalset_hash: String,
838    /// The `areev run`/`areev eval` run id that executed the gate.
839    pub run_id: String,
840    pub passed: u64,
841    pub failed: u64,
842}
843
844/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
845/// re-checked when a stored grain is loaded (a hand-authored grain must not
846/// bypass the rule):
847/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
848/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
849///   seam inherits §7.4's gate wholesale).
850/// - An `evalset:` target is its own always-human-approved class: never a
851///   gated revision, and never itself pinned (the gate cannot gate itself).
852/// - No other target class carries a pin.
853pub fn validate_code_rules(
854    action: ActionKind,
855    target_class: &str,
856    evalset_hash: Option<&str>,
857) -> Result<()> {
858    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
859    match (action, target_class) {
860        (ActionKind::CodeRevision, "code") => {
861            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
862                return e1(
863                    "a code_revision must pin the evalset hash it was gated \
864                     against (evalset_hash)",
865                );
866            }
867        }
868        (ActionKind::CodeRevision, other) => {
869            return e1(&format!(
870                "code_revision requires a tool: target, got class '{other}'"
871            ));
872        }
873        (ActionKind::AdapterRevision, "model") => {
874            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
875                return e1(
876                    "an adapter_revision must pin the evalset hash it was \
877                     gated against (evalset_hash)",
878                );
879            }
880        }
881        (ActionKind::AdapterRevision, other) => {
882            return e1(&format!(
883                "adapter_revision requires a model: target, got class '{other}'"
884            ));
885        }
886        // A REVERT of an applied code revision targets the same tool: —
887        // it is the rollback inverse, not new code, so it needs no pin
888        // (blocking it here would leave the highest-stakes target class
889        // unable to propose its own measured rollback).
890        (ActionKind::Revert, "code") => {
891            if evalset_hash.is_some() {
892                return e1("a code revert carries no evalset pin");
893            }
894        }
895        (_, "code") => {
896            return e1("a tool: target requires action_kind code_revision");
897        }
898        // Same rollback-inverse escape hatch for the adapter class.
899        (ActionKind::Revert, "model") => {
900            if evalset_hash.is_some() {
901                return e1("an adapter revert carries no evalset pin");
902            }
903        }
904        (_, "model") => {
905            return e1("a model: target requires action_kind adapter_revision");
906        }
907        (_, "evalset") => {
908            if evalset_hash.is_some() {
909                return e1(
910                    "an evalset-target recommendation cannot pin an evalset — \
911                     the gate cannot gate itself",
912                );
913            }
914        }
915        _ => {
916            if evalset_hash.is_some() {
917                return e1(
918                    "evalset_hash is only valid on code_revision or \
919                     adapter_revision recommendations",
920                );
921            }
922        }
923    }
924    Ok(())
925}
926
927impl Recommendation {
928    /// Serialize the immutable body into grain fields (excludes hash/status).
929    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
930        let value = serde_json::to_value(self)
931            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
932        let obj = value
933            .as_object()
934            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
935            .clone();
936        Ok(GrainSpec {
937            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
938            namespace: namespace.to_string(),
939            fields: obj,
940        })
941    }
942
943    /// Reconstruct from a stored grain body. `hash` comes from the record's
944    /// address; `status` is supplied from the state index by the caller.
945    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
946    /// must not smuggle an unpinned code revision past the stamped path.
947    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
948        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
949            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
950        rec.hash = hash.to_string();
951        let class = crate::model::TargetRef::parse(&rec.target_ref)
952            .map(|t| t.target_class())
953            .unwrap_or("host");
954        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
955        Ok(rec)
956    }
957}
958
959#[cfg(test)]
960mod tests {
961    /// The gate and `areev eval run --tolerance` call this one function, so
962    /// the boundary is the same for both readers by construction: a drop of
963    /// exactly the tolerance holds, a hair past it regresses, and a tolerance
964    /// that is not a positive finite number is zero.
965    #[test]
966    fn the_regression_rule_holds_at_the_tolerance_and_regresses_past_it() {
967        use super::is_regression;
968        // Percentage points, the way `eval run --tolerance` reads it.
969        assert!(!is_regression(100.0, 0.0, true, 100.0), "a 100-point drop under a 100-point tolerance is accepted");
970        assert!(is_regression(100.0, 0.0, true, 99.9));
971        assert!(!is_regression(92.0, 91.5, true, 1.0));
972        assert!(is_regression(92.0, 90.5, true, 1.0));
973        // Counts, the way the gate reads `min_effect.count`.
974        assert!(!is_regression(359.0, 355.0, true, 5.0));
975        assert!(!is_regression(359.0, 354.0, true, 5.0), "exactly at the floor holds");
976        assert!(is_regression(359.0, 353.0, true, 5.0));
977        assert!(!is_regression(28.0, 33.0, false, 5.0));
978        assert!(is_regression(28.0, 34.0, false, 5.0));
979        // Zero is the pre-policy rule; garbage is zero, never a wider floor.
980        assert!(is_regression(359.0, 355.0, true, 0.0));
981        assert!(is_regression(359.0, 355.0, true, -5.0));
982        assert!(is_regression(359.0, 355.0, true, f64::NAN));
983        assert!(!is_regression(1.0, 1.0, true, 0.0), "equal is never a regression");
984    }
985
986    use super::*;
987    use serde_json::json;
988
989    #[test]
990    fn summary_renders_deterministically() {
991        let mut args = Map::new();
992        args.insert("count".into(), json!(3));
993        args.insert("subject".into(), json!("acme"));
994        let s = Summary::new("duplicate.exact", args);
995        assert_eq!(
996            s.render(),
997            "Consolidate 3 exact-duplicate grains for \"acme\""
998        );
999    }
1000
1001    #[test]
1002    fn dedup_key_ignores_content_and_case() {
1003        let a = dedup_key(
1004            "loop.duplicate_sweep",
1005            "entity:NS/John",
1006            ActionKind::Consolidate,
1007        );
1008        let b = dedup_key(
1009            "loop.duplicate_sweep",
1010            "entity:ns/john",
1011            ActionKind::Consolidate,
1012        );
1013        assert_eq!(a, b, "case-folded to one identity");
1014    }
1015
1016    #[test]
1017    fn authored_dedup_key_distinguishes_content_but_not_wording_noise() {
1018        let a = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the vendor name.");
1019        let same = authored_dedup_key("llm", "entity:NS/X", ActionKind::Record, "  record THE vendor  name ");
1020        let other = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the amount.");
1021        assert_eq!(a, same, "case, punctuation and spacing are not a new lesson");
1022        assert_ne!(a, other, "a different lesson on the same entity is a different finding");
1023        assert!(a.starts_with(&dedup_key("llm", "entity:ns/x", ActionKind::Record)));
1024        assert_eq!(content_fingerprint("A b"), content_fingerprint("a-b"));
1025        assert_ne!(content_fingerprint("ab"), content_fingerprint("a b"));
1026    }
1027
1028    #[test]
1029    fn a_revert_is_keyed_by_what_it_reverts() {
1030        let a = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "aaaa");
1031        let b = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "bbbb");
1032        let a2 = revert_dedup_key("loop.outcome_review", "entity:NS/Capture", "AAAA");
1033        assert_ne!(a, b, "two reverts on one target are two findings");
1034        assert_eq!(a, a2, "the same revert, case-folded, is one");
1035        assert!(a.starts_with(&dedup_key("loop.outcome_review", "entity:ns/capture", ActionKind::Revert)));
1036    }
1037
1038    #[test]
1039    fn dedup_key_distinguishes_action() {
1040        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
1041        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
1042        assert_ne!(a, b);
1043    }
1044
1045    #[test]
1046    fn lifecycle_gates_pending_to_applied() {
1047        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
1048        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
1049        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
1050        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
1051        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
1052        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
1053    }
1054
1055    #[test]
1056    fn rule_e1_pins_code_and_only_code() {
1057        use crate::model::ActionKind as A;
1058        // code_revision on a tool target with a pin: OK.
1059        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
1060        // Unpinned code revision: refused.
1061        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
1062        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
1063        // code_revision off a tool target: refused.
1064        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
1065        // A tool target with a non-code action: refused.
1066        assert!(validate_code_rules(A::Flag, "code", None).is_err());
1067        // The gate cannot gate itself.
1068        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
1069        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
1070        // Pins don't leak onto ordinary targets.
1071        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
1072        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
1073    }
1074
1075    #[test]
1076    fn rule_e1_pins_adapter_revisions_like_code() {
1077        use crate::model::ActionKind as A;
1078        // adapter_revision on a model target with a pin: OK.
1079        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
1080        // Unpinned adapter revision: refused.
1081        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
1082        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
1083        // adapter_revision off a model target: refused.
1084        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
1085        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
1086        // A model target with a non-adapter action: refused.
1087        assert!(validate_code_rules(A::Flag, "model", None).is_err());
1088        // The rollback inverse needs no pin — and must carry none.
1089        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
1090        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
1091    }
1092
1093    #[test]
1094    fn from_fields_rechecks_rule_e1() {
1095        // A hand-authored grain body carrying an unpinned code revision must
1096        // not decode into a usable recommendation.
1097        let mut rec = Recommendation {
1098            hash: String::new(),
1099            analyzer: "loop.codegen/1".into(),
1100            params_snapshot: Map::new(),
1101            origin: Origin::Builtin,
1102            target_ref: "tool:abc123".into(),
1103            action_kind: ActionKind::CodeRevision,
1104            dedup_key: "k".into(),
1105            summary: Summary::new("command.finding", Map::new()),
1106            severity: Severity::Low,
1107            proposal: Proposal::Data { data: Map::new() },
1108            destructive: false,
1109            rollbackable: false,
1110            evidence: vec![],
1111            evidence_query: None,
1112            metric: None,
1113            confidence: 0.5,
1114            importance: 0.5,
1115            created_at_ms: 0,
1116            guidance: None,
1117            evalset_hash: None, // the smuggle attempt
1118            near_duplicate_of: Vec::new(),
1119            replay: None,
1120            status: RecStatus::Pending,
1121        };
1122        let spec = rec.to_grain_spec("ns").unwrap();
1123        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
1124        rec.evalset_hash = Some("es-hash".into());
1125        let spec = rec.to_grain_spec("ns").unwrap();
1126        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
1127        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
1128    }
1129
1130    #[test]
1131    fn recommendation_round_trips_through_fields() {
1132        let rec = Recommendation {
1133            hash: "ignored".into(),
1134            analyzer: "loop.staleness/1".into(),
1135            params_snapshot: Map::new(),
1136            origin: Origin::Builtin,
1137            target_ref: "grain:sha256:abc".into(),
1138            action_kind: ActionKind::Expire,
1139            dedup_key: "k".into(),
1140            summary: Summary::new("staleness.expired", Map::new()),
1141            severity: Severity::Low,
1142            proposal: Proposal::Cal {
1143                cal: "FORGET sha256:abc".into(),
1144            },
1145            destructive: true,
1146            rollbackable: false,
1147            evidence: vec!["sha256:abc".into()],
1148            evidence_query: None,
1149            metric: None,
1150            confidence: 0.9,
1151            importance: 0.4,
1152            created_at_ms: 1000,
1153            guidance: None,
1154            evalset_hash: None,
1155            near_duplicate_of: Vec::new(),
1156            replay: None,
1157            status: RecStatus::Pending,
1158        };
1159        let spec = rec.to_grain_spec("ns").unwrap();
1160        // hash and status are excluded from the immutable body.
1161        assert!(!spec.fields.contains_key("hash"));
1162        assert!(!spec.fields.contains_key("status"));
1163        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
1164        assert_eq!(back.hash, "realhash");
1165        assert_eq!(back.analyzer, "loop.staleness/1");
1166        assert!(back.destructive);
1167        assert!(matches!(back.proposal, Proposal::Cal { .. }));
1168    }
1169}