Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel —
9//!   for ANALYZER findings. An authored (`origin = llm`) executable proposal
10//!   keys on a fingerprint of its content as well, because there the content
11//!   IS the finding: two different lessons on one entity are two findings,
12//!   and the same lesson re-authored is one.
13//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
14
15use crate::error::{Error, Result};
16use crate::model::{normalize_ident, ActionKind, Origin, Severity};
17use crate::substrate::GrainSpec;
18use serde::{Deserialize, Serialize};
19use serde_json::{Map, Value};
20
21/// A deterministic, template-rendered summary. The analyzer chooses a
22/// `template_id` and supplies `args`; the text is produced here, so an
23/// analyzer can never emit arbitrary prose into the queue.
24#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
25pub struct Summary {
26    pub template_id: String,
27    #[serde(default)]
28    pub args: Map<String, Value>,
29}
30
31impl Summary {
32    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
33        Summary {
34            template_id: template_id.into(),
35            args,
36        }
37    }
38
39    /// Render the summary. Unknown template ids fall back to a stable, honest
40    /// string (never a panic) so a manifest/template mismatch is visible, not
41    /// fatal.
42    pub fn render(&self) -> String {
43        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
44        interpolate(t, &self.args, &self.template_id)
45    }
46}
47
48/// The built-in template table. Deterministic; the only place summary prose
49/// lives. Keep placeholders in `{name}` form matching `args` keys.
50fn builtin_template(id: &str) -> Option<&'static str> {
51    Some(match id {
52        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
53        "duplicate.near" => {
54            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
55        }
56        "duplicate.judged" => {
57            "Consolidate {count} observations a decision model judged to state the same claim (p = {p}; token similarity {similarity})"
58        }
59        "contradiction.functional" => {
60            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
61        }
62        // E2: a relation outside the functional set, judged by a decision
63        // model — the summary says both, so the reviewer knows no one
64        // declared the relation single-valued.
65        "contradiction.judged" => {
66            "\"{subject}\" holds {count} live values for relation \"{relation}\" that a decision model judged cannot both be true (p = {p}); \"{relation}\" is not a seeded functional relation"
67        }
68        "tool_failure.cluster" => {
69            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
70             fail this way): {signature}"
71        }
72        "tool_failure.cluster_cause" => {
73            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
74             fail this way): {signature} (cause: {cause})"
75        }
76        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
77        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
78        "skill.stall" => {
79            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
80        }
81        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
82        "cold.grain" => {
83            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
84        }
85        "coverage.gap" => {
86            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
87        }
88        "budget.pressure" => {
89            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
90        }
91        "retention.overdue" => {
92            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
93        }
94        "outcome.regression" => {
95            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
96        }
97        // The high-water baseline names both figures — the best run before
98        // the apply and the run that fell from it — because the reviewer has
99        // to judge whether THIS rule caused the whole fall from the peak.
100        "outcome.regression_high_water" => {
101            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}"
102        }
103        // A regression that also breached the cost bound: the revert names
104        // the cost delta so the reviewer sees both halves of the damage.
105        "outcome.regression_costlier" => {
106            "Applied recommendation regressed: {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
107        }
108        "outcome.regression_high_water_costlier" => {
109            "Applied recommendation regressed against the best run before the apply ({baseline_run}): {metric} moved {baseline} → {current}, and {cost_field} rose {cost_baseline} → {cost_current} (bound ×{cost_ratio})"
110        }
111        // Quality held, cost did not: advisory only — a cost/quality trade
112        // is a human decision, so this is a Flag, never a revert draft.
113        "outcome.held_costlier" => {
114            "Applied recommendation held ({metric} {baseline} → {current}) but {cost_field} rose {cost_baseline} → {cost_current}, past ×{cost_ratio} of the baseline run ({baseline_run} → {current_run}) — a cost/quality trade to decide"
115        }
116        "outcome.premise_drift" => {
117            "{current} of the grains this recommendation cited have since been superseded by a different value or retracted — its premise moved; revert it"
118        }
119        "run.failures" => {
120            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
121        }
122        "run.cost" => {
123            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
124        }
125        "run.context_pressure" => {
126            "Workflow {workflow} summarized its own transcript {folds} time(s) across {runs} runs (avg {avg_folds}/run) — its nodes are outgrowing the model's window"
127        }
128        "adapter.candidate" => {
129            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
130        }
131        // origin=llm drafts carry free text (clearly marked llm) rather than a
132        // deterministic template — the model's proposed summary rides in {text}.
133        "llm.discover" => "{text}",
134        // An LLM-authored lesson shows the reviewer BOTH the finding and the
135        // exact line an apply would record — never one without the other.
136        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
137        // The same lesson, flagged as restating a live one: the reviewer sees
138        // that a rule of this meaning already exists before approving a
139        // second copy of it.
140        "llm.lesson_near_duplicate" => {
141            "{text} — record lesson: \"{lesson}\" — NEAR-DUPLICATE of {near_count} live lesson(s) on this entity (best match {near_score} by {near_method}, {near_hash})"
142        }
143        // A consolidation: one lesson replacing a pile. The apply supersedes
144        // every member and adds the one line; rollback restores them all.
145        "llm.consolidation" => {
146            "{text} — consolidate {count} lessons into one: \"{lesson}\""
147        }
148        "lesson.pile" => {
149            "{count} live lessons on \"{subject}\" exceed the budget of {max_active} — every rule competes for the model's attention; consolidate or retire: {members}"
150        }
151        // The rest of the LLM proposal vocabulary follows the same rule: the
152        // finding AND the exact change an apply would make, never one without
153        // the other. A reviewer approving blind is the failure this prevents.
154        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
155        "llm.skill" => "{text} — record skill: \"{name}\" ({steps} steps)",
156        "llm.plan" => "{text} — record plan: \"{name}\" ({nodes} steps, {edges} edges)",
157        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
158        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
159        // The rehearsal refused it: the edit is shown, the reason names the
160        // runs, and nothing is offered to apply.
161        "llm.plan_revision_refused" => {
162            "{text} — plan revision of {plan} ({edits}) NOT applicable: rehearsed against the plan's journaled runs and {reason}"
163        }
164        "llm.code_revision" => {
165            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
166        }
167        // External command analyzers (trust class Command): free text from the
168        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
169        "command.finding" => "{text}",
170        _ => return None,
171    })
172}
173
174fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
175    let mut out = String::with_capacity(template.len());
176    let mut chars = template.chars().peekable();
177    while let Some(c) = chars.next() {
178        if c == '{' {
179            let mut key = String::new();
180            for k in chars.by_ref() {
181                if k == '}' {
182                    break;
183                }
184                key.push(k);
185            }
186            if key == "template_id" {
187                out.push_str(template_id);
188            } else {
189                match args.get(&key) {
190                    Some(Value::String(s)) => out.push_str(s),
191                    Some(v) => out.push_str(&v.to_string()),
192                    None => {
193                        // Missing arg — leave a visible marker rather than lie.
194                        out.push('{');
195                        out.push_str(&key);
196                        out.push('}');
197                    }
198                }
199            }
200        } else {
201            out.push(c);
202        }
203    }
204    out
205}
206
207/// When an applied recommendation is re-measured — one checkpoint of the
208/// Verify gate's schedule, in the unit the deployment actually counts in.
209///
210/// Three units, because deployments count differently and a fixed one makes
211/// the gate inert everywhere else. A service evaluated nightly counts
212/// **time** (`after_ms`). A benchmark or a CI harness counts **graded runs**
213/// (`after_runs`): "measure at the next evaluation after the apply", however
214/// long that takes on the clock. A chat agent counts **turns** (`after_grains`):
215/// "after fifty more grains". The default schedule stayed ms-only for a year
216/// and was measured, on PAST-Bench, to fire exactly zero verdicts across 78
217/// governed runs — a family finishes in seven minutes and the first checkpoint
218/// was a day away (`crates/areev-bench/PERSIST.md`).
219///
220/// On the wire a bare integer is milliseconds, so every policy, metric
221/// snapshot and state blob written before this type existed reads back
222/// unchanged, and an all-ms schedule still serializes as `[86400000, …]`.
223#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
224pub enum Checkpoint {
225    /// Elapsed time since the apply.
226    AfterMs(i64),
227    /// Evalset runs journaled since the apply. Only an `evalset:` metric can
228    /// count these; on any other metric the checkpoint never comes due.
229    AfterRuns(u32),
230    /// Grains written since the apply — the loop's own notion of "activity".
231    AfterGrains(u32),
232}
233
234impl Checkpoint {
235    /// The label a person reads: `1d`, `12h`, `3 runs`, `50 grains`.
236    pub fn label(&self) -> String {
237        match self {
238            Checkpoint::AfterMs(ms) if ms % 86_400_000 == 0 => format!("{}d", ms / 86_400_000),
239            Checkpoint::AfterMs(ms) if ms % 3_600_000 == 0 => format!("{}h", ms / 3_600_000),
240            Checkpoint::AfterMs(ms) => format!("{ms}ms"),
241            Checkpoint::AfterRuns(n) => format!("{n} run{}", if *n == 1 { "" } else { "s" }),
242            Checkpoint::AfterGrains(n) => format!("{n} grain{}", if *n == 1 { "" } else { "s" }),
243        }
244    }
245
246    /// The ms value when this is a time checkpoint — what legacy consumers of
247    /// `OutcomeResult::horizon_ms` read.
248    pub fn as_ms(&self) -> Option<i64> {
249        match self {
250            Checkpoint::AfterMs(ms) => Some(*ms),
251            _ => None,
252        }
253    }
254}
255
256impl Serialize for Checkpoint {
257    fn serialize<S: serde::Serializer>(&self, ser: S) -> std::result::Result<S::Ok, S::Error> {
258        use serde::ser::SerializeMap;
259        match self {
260            // Bare integer: byte-identical to the pre-Checkpoint `Vec<i64>`.
261            Checkpoint::AfterMs(ms) => ser.serialize_i64(*ms),
262            Checkpoint::AfterRuns(n) => {
263                let mut m = ser.serialize_map(Some(1))?;
264                m.serialize_entry("after_runs", n)?;
265                m.end()
266            }
267            Checkpoint::AfterGrains(n) => {
268                let mut m = ser.serialize_map(Some(1))?;
269                m.serialize_entry("after_grains", n)?;
270                m.end()
271            }
272        }
273    }
274}
275
276impl<'de> Deserialize<'de> for Checkpoint {
277    fn deserialize<D: serde::Deserializer<'de>>(de: D) -> std::result::Result<Self, D::Error> {
278        use serde::de::Error as _;
279        let v = Value::deserialize(de)?;
280        match &v {
281            Value::Number(n) => n
282                .as_i64()
283                .filter(|ms| *ms >= 0)
284                .map(Checkpoint::AfterMs)
285                .ok_or_else(|| D::Error::custom("checkpoint: a bare number is non-negative milliseconds")),
286            Value::Object(m) if m.len() == 1 => {
287                let (k, val) = m.iter().next().unwrap();
288                let n = val.as_u64().ok_or_else(|| {
289                    D::Error::custom(format!("checkpoint: {k} takes a non-negative integer"))
290                })?;
291                match k.as_str() {
292                    "after_ms" => i64::try_from(n)
293                        .map(Checkpoint::AfterMs)
294                        .map_err(|_| D::Error::custom("checkpoint: after_ms out of range")),
295                    "after_runs" => u32::try_from(n)
296                        .map(Checkpoint::AfterRuns)
297                        .map_err(|_| D::Error::custom("checkpoint: after_runs out of range")),
298                    "after_grains" => u32::try_from(n)
299                        .map(Checkpoint::AfterGrains)
300                        .map_err(|_| D::Error::custom("checkpoint: after_grains out of range")),
301                    other => Err(D::Error::custom(format!(
302                        "checkpoint: unknown unit {other:?} (expected after_ms, after_runs or after_grains)"
303                    ))),
304                }
305            }
306            _ => Err(D::Error::custom(
307                "checkpoint: expected milliseconds or {\"after_ms\"|\"after_runs\"|\"after_grains\": n}",
308            )),
309        }
310    }
311}
312
313/// A reproducible metric snapshot; powers outcome review.
314#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
315pub struct MetricSnapshot {
316    /// Metric kind — the engine knows how to re-measure a fixed set
317    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
318    pub metric: String,
319    pub baseline: f64,
320    pub unit: String,
321    pub n: u64,
322    pub window: String,
323    /// The subject the metric is about (e.g. the tool name), used by the
324    /// engine's typed re-measurement.
325    #[serde(default, skip_serializing_if = "Option::is_none")]
326    pub subject: Option<String>,
327    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
328    /// `None` = all namespaces. Additive: older snapshots deserialize to
329    /// `None` and existing metric kinds ignore it.
330    #[serde(default, skip_serializing_if = "Option::is_none")]
331    pub namespace: Option<String>,
332    /// Relation scope for fact-shaped metrics (e.g. which functional relation
333    /// a contradiction was resolved under), normalized. Additive like
334    /// `namespace`.
335    #[serde(default, skip_serializing_if = "Option::is_none")]
336    pub relation: Option<String>,
337    /// CAL that recomputes the metric at verify time (reproducibility /
338    /// documentation; the engine re-measures with typed reads).
339    pub query: String,
340    /// How long after apply to re-measure, in epoch-ms delta (the first / only
341    /// checkpoint when `horizons_ms` is empty).
342    pub review_after_ms: i64,
343    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
344    /// that `held` at an early checkpoint can `regress` at a later one, so a
345    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
346    #[serde(default, skip_serializing_if = "Vec::is_empty")]
347    pub horizons_ms: Vec<i64>,
348    /// The schedule in the deployment's own unit ([`Checkpoint`]). When set it
349    /// is THE schedule and `horizons_ms`/`review_after_ms` are ignored; empty
350    /// (every snapshot written before it existed) falls back to them.
351    #[serde(default, skip_serializing_if = "Vec::is_empty")]
352    pub checkpoints: Vec<Checkpoint>,
353    /// Which direction is an improvement. The built-in metrics are all
354    /// recurrence counts, where lower is better — so this defaults to `false`
355    /// and every existing snapshot deserializes unchanged. An evalset accuracy
356    /// is the opposite, and getting it wrong does not merely misreport: the
357    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
358    /// propose reverting it.
359    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
360    pub higher_is_better: bool,
361}
362
363/// The one regression rule.
364///
365/// The verdict is computed in three places — by the engine for the recorded
366/// `OutcomeResult`, by `outcome_review` when it drafts the revert, and by
367/// `areev eval run --tolerance` when it re-accepts a model swap — and any two
368/// silently disagreeing would either revert a change that held or sit on one
369/// that regressed. So all of them call this.
370///
371/// `tolerance` is the policy's minimum effect size in the metric's own unit
372/// (`OutcomeEvalset::min_effect`, or `--tolerance` in percentage points of
373/// the pass rate): a worsening of at most that much is not a regression. It
374/// is a **floor, not a significance test** — no p-value, no interval; a host
375/// that needs statistics has the run counts to compute them. Zero is the
376/// pre-policy behaviour: any drop is a regression. `EPSILON` is only
377/// floating-point equality slack, so a value read twice that differs in the
378/// last bit does not flip the verdict; it is not a noise floor.
379pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool, tolerance: f64) -> bool {
380    const EPSILON: f64 = 1e-9;
381    let tolerance = if tolerance.is_finite() && tolerance > 0.0 { tolerance } else { 0.0 };
382    if higher_is_better {
383        current < baseline - tolerance - EPSILON
384    } else {
385        current > baseline + tolerance + EPSILON
386    }
387}
388
389impl MetricSnapshot {
390    /// The measurement schedule, sorted and deduplicated: `checkpoints` when
391    /// set, else `horizons_ms`, else the single `review_after_ms` — each older
392    /// spelling read as time, so a snapshot from before [`Checkpoint`] existed
393    /// measures exactly as it did.
394    pub fn schedule(&self) -> Vec<Checkpoint> {
395        let mut h: Vec<Checkpoint> = if !self.checkpoints.is_empty() {
396            self.checkpoints.clone()
397        } else if !self.horizons_ms.is_empty() {
398            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
399        } else {
400            vec![Checkpoint::AfterMs(self.review_after_ms)]
401        };
402        h.sort_unstable();
403        h.dedup();
404        h
405    }
406
407    /// The time checkpoints of the schedule, in ms — the pre-[`Checkpoint`]
408    /// view, kept for callers that only understand time.
409    pub fn horizons(&self) -> Vec<i64> {
410        self.schedule().iter().filter_map(Checkpoint::as_ms).collect()
411    }
412}
413
414fn is_zero(x: &f64) -> bool {
415    *x == 0.0
416}
417
418/// What a checkpoint read of the policy's cost bound, beside the quality
419/// verdict. `status` is `within`, `breached`, or `not_measurable` (the
420/// field was absent or malformed on either run — then `baseline`/`current`
421/// carry whichever side did measure, or nothing). Present on a record only
422/// when the policy set a bound.
423#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
424pub struct CostRead {
425    pub field: String,
426    pub max_increase_ratio: f64,
427    #[serde(default, skip_serializing_if = "Option::is_none")]
428    pub baseline: Option<f64>,
429    #[serde(default, skip_serializing_if = "Option::is_none")]
430    pub current: Option<f64>,
431    pub status: String,
432}
433
434impl CostRead {
435    pub fn breached(&self) -> bool {
436        self.status == "breached"
437    }
438}
439
440/// A measured outcome for an applied recommendation at one checkpoint — the
441/// Verify gate's output. `held` = the metric did not regress at this horizon;
442/// `regressed` = it got worse (a revert is proposed). A recommendation
443/// accumulates one of these per horizon, forming a time series.
444#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
445pub struct OutcomeResult {
446    pub rec_hash: String,
447    pub metric: String,
448    pub baseline: f64,
449    pub current: f64,
450    pub verdict: String,
451    /// Where `baseline` came from, so the receipt names the run it compared
452    /// against: `newest_before_apply` or `high_water` (an evalset run, named
453    /// in `baseline_run_id`), or `snapshot` — the number the proposal froze,
454    /// used when no run was journaled before the apply or the metric is not
455    /// evalset-backed. Empty on a record written before this field existed.
456    #[serde(default, skip_serializing_if = "String::is_empty")]
457    pub baseline_kind: String,
458    /// The evalset run `baseline` was read from, when it was read from one.
459    #[serde(default, skip_serializing_if = "Option::is_none")]
460    pub baseline_run_id: Option<String>,
461    /// Advisory, independent of the policy's baseline choice: the best value
462    /// the field attained on any run journaled before the apply. When the
463    /// verdict is `held` against a newer, lower baseline this is the lost
464    /// opportunity the marginal comparison cannot see.
465    #[serde(default, skip_serializing_if = "Option::is_none")]
466    pub best_before: Option<f64>,
467    /// The minimum effect size the verdict was made under, in the metric's
468    /// unit (`OutcomeEvalset::min_effect`, resolved at measure time). Absent
469    /// when zero, so a `held` under a floor is distinguishable from a `held`
470    /// at zero and a record without one reads as before.
471    #[serde(default, skip_serializing_if = "is_zero")]
472    pub tolerance: f64,
473    /// The evalset run `current` was read from, when it was read from one.
474    #[serde(default, skip_serializing_if = "Option::is_none")]
475    pub current_run_id: Option<String>,
476    /// The cost bound's reading at this checkpoint, when the policy set one.
477    /// A quality `held` with a breached bound records the verdict
478    /// `held_costlier`; `regressed` dominates.
479    #[serde(default, skip_serializing_if = "Option::is_none")]
480    pub cost: Option<CostRead>,
481    /// Which checkpoint this measurement is for (ms after apply). Zero for a
482    /// checkpoint counted in runs or grains — see `checkpoint`.
483    #[serde(default)]
484    pub horizon_ms: i64,
485    /// The checkpoint in its own unit, when that unit is not time. A time
486    /// checkpoint leaves this unset and speaks through `horizon_ms`, so every
487    /// consumer of the older shape reads an unchanged record.
488    #[serde(default, skip_serializing_if = "Option::is_none")]
489    pub checkpoint: Option<Checkpoint>,
490    pub measured_at_ms: i64,
491}
492
493/// The proposed change. Exactly one variant per recommendation.
494#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
495#[serde(tag = "proposal", rename_all = "snake_case")]
496pub enum Proposal {
497    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
498    Cal { cal: String },
499    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
500    /// staleness check at apply).
501    Edit {
502        format: String,
503        base_digest: String,
504        diff: String,
505    },
506    /// An opaque map for host targets (applied by the host, §12.3).
507    Data { data: Map<String, Value> },
508}
509
510/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
511/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
512/// fields without breaking analyzers.
513#[derive(Debug, Clone, PartialEq)]
514#[non_exhaustive]
515pub struct RecDraft {
516    pub target_ref: String,
517    pub action_kind: ActionKind,
518    pub summary: Summary,
519    pub severity: Severity,
520    pub proposal: Proposal,
521    /// Bounded to ≤64 representative evidence hashes by the engine.
522    pub evidence: Vec<String>,
523    pub evidence_query: Option<String>,
524    pub metric: Option<MetricSnapshot>,
525    pub confidence: f64,
526    pub importance: f64,
527    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
528    pub evalset_hash: Option<String>,
529    /// Set when a decision backend's probability shaped this draft (a sweep
530    /// pair judged the same claim / not both true, a classified tool cause).
531    /// Carried onto the recommendation; a judged recommendation never
532    /// auto-applies.
533    pub judged_by: Option<crate::decide::JudgedBy>,
534}
535
536impl RecDraft {
537    pub fn new(
538        target_ref: impl Into<String>,
539        action_kind: ActionKind,
540        summary: Summary,
541        proposal: Proposal,
542    ) -> Self {
543        RecDraft {
544            target_ref: target_ref.into(),
545            action_kind,
546            summary,
547            severity: Severity::Low,
548            proposal,
549            evidence: Vec::new(),
550            evidence_query: None,
551            metric: None,
552            confidence: 0.8,
553            importance: 0.5,
554            evalset_hash: None,
555            judged_by: None,
556        }
557    }
558
559    pub fn judged_by(mut self, j: crate::decide::JudgedBy) -> Self {
560        self.judged_by = Some(j);
561        self
562    }
563
564    pub fn severity(mut self, s: Severity) -> Self {
565        self.severity = s;
566        self
567    }
568    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
569        self.evidence = hashes;
570        self
571    }
572    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
573        self.evidence_query = Some(q.into());
574        self
575    }
576    pub fn metric(mut self, m: MetricSnapshot) -> Self {
577        self.metric = Some(m);
578        self
579    }
580    pub fn confidence(mut self, c: f64) -> Self {
581        self.confidence = c;
582        self
583    }
584    pub fn importance(mut self, i: f64) -> Self {
585        self.importance = i;
586        self
587    }
588    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
589        self.evalset_hash = Some(h.into());
590        self
591    }
592}
593
594/// Maximum representative evidence hashes carried inline (proposal §7.1).
595pub const MAX_EVIDENCE: usize = 64;
596
597/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
598/// Excludes proposal content and evidence by construction.
599pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
600    // U+001F (unit separator) cannot appear in any of the inputs.
601    format!(
602        "{}\u{1f}{}\u{1f}{}",
603        normalize_ident(family),
604        normalize_ident(target_ref),
605        action.as_str()
606    )
607}
608
609/// The dedup key of an AUTHORED executable proposal: [`dedup_key`] plus a
610/// fingerprint of the proposal's content. An analyzer finding is "this
611/// target has this kind of problem", so content is rightly excluded; an
612/// authored lesson is "do this", and two different lessons on the same
613/// entity must both reach the queue while the same lesson re-authored must
614/// not. The fingerprint is over the normalized text — case-folded, non-
615/// alphanumerics dropped, whitespace collapsed — so a rewording that changes
616/// no word is the same lesson and one that changes a word is a new one (a
617/// semantic near-duplicate is the reviewer's call, not this key's).
618pub fn authored_dedup_key(
619    family: &str,
620    target_ref: &str,
621    action: ActionKind,
622    content: &str,
623) -> String {
624    format!(
625        "{}\u{1f}{}",
626        dedup_key(family, target_ref, action),
627        content_fingerprint(content)
628    )
629}
630
631/// The dedup key of a REVERT: [`dedup_key`] plus the hash of the applied
632/// recommendation it retracts. An analyzer finding is "this target has this
633/// kind of problem", so two findings on one target rightly collapse — but a
634/// revert is about one specific applied recommendation, and two lessons on
635/// the same entity that both regressed need two reverts. Until 2026-09-06
636/// they shared a key and the second was dropped as a duplicate of the first
637/// (`crates/areev-bench/CURVE.md`, seed 3: two regressed, one revert).
638pub fn revert_dedup_key(family: &str, target_ref: &str, revert_of: &str) -> String {
639    format!(
640        "{}\u{1f}{}",
641        dedup_key(family, target_ref, ActionKind::Revert),
642        normalize_ident(revert_of)
643    )
644}
645
646/// Sixteen hex chars of FNV-1a (64-bit) over the normalized content. A dedup
647/// key needs stability and spread, not cryptographic strength — a collision
648/// here would merge two findings in a review queue, never grant anything —
649/// so this stays dependency-free, as the crate is by policy.
650pub fn content_fingerprint(content: &str) -> String {
651    let mut normalized = String::with_capacity(content.len());
652    let mut pending_space = false;
653    for c in content.chars() {
654        if c.is_alphanumeric() {
655            if pending_space && !normalized.is_empty() {
656                normalized.push(' ');
657            }
658            pending_space = false;
659            normalized.extend(c.to_lowercase());
660        } else if c.is_whitespace() || !c.is_alphanumeric() {
661            pending_space = true;
662        }
663    }
664    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
665    for b in normalized.as_bytes() {
666        h ^= u64::from(*b);
667        h = h.wrapping_mul(0x0000_0100_0000_01b3);
668    }
669    format!("{h:016x}")
670}
671
672/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
673/// content hash is stable for its whole life).
674#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
675#[serde(rename_all = "snake_case")]
676pub enum RecStatus {
677    #[default]
678    Pending,
679    Approved,
680    Rejected,
681    Applied,
682    RolledBack,
683    Expired,
684    /// The engine withdrew it: every grain it cited has moved (#317).
685    ///
686    /// Distinct from `Expired`, which says "time ran out", and from
687    /// `Rejected`, which is a human decision. A reviewer was being offered —
688    /// and could approve — findings whose entire evidence had been
689    /// superseded by a different value or retracted, and applying one then
690    /// produced a recommendation to revert it on the next pass.
691    ///
692    /// Reachable from `Pending` and `Approved`, by the ENGINE only.
693    Withdrawn,
694}
695
696impl RecStatus {
697    pub fn as_str(&self) -> &'static str {
698        match self {
699            RecStatus::Pending => "pending",
700            RecStatus::Approved => "approved",
701            RecStatus::Rejected => "rejected",
702            RecStatus::Applied => "applied",
703            RecStatus::RolledBack => "rolled_back",
704            RecStatus::Expired => "expired",
705            RecStatus::Withdrawn => "withdrawn",
706        }
707    }
708
709    /// Is a transition to `to` allowed from this state? `by_policy` marks the
710    /// auto-apply actor, the only one permitted the reasonless
711    /// `pending → applied` jump.
712    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
713        use RecStatus::*;
714        match (self, to) {
715            (Pending, Approved) | (Pending, Rejected) => true,
716            (Pending, Applied) => by_policy, // auto-apply only
717            (Approved, Applied) => true,
718            (Applied, RolledBack) => true,
719            // `expired` is computed from valid_to, applied to still-open recs.
720            (Pending, Expired) | (Approved, Expired) => true,
721            // `withdrawn` is computed from premise drift, applied to
722            // still-open recs (#317). Terminal: a withdrawn finding is not
723            // re-opened, because the evidence that is gone does not come
724            // back — a fresh finding on NEW evidence is proposed instead.
725            (Pending, Withdrawn) | (Approved, Withdrawn) => true,
726            _ => false,
727        }
728    }
729}
730
731/// Who/what performed a transition.
732#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
733#[serde(rename_all = "lowercase")]
734pub enum ObserverType {
735    Human,
736    Agent,
737    Policy,
738    System,
739}
740
741/// One immutable audit Observation per transition, hash-chained per
742/// recommendation.
743#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
744pub struct AuditRecord {
745    pub rec_hash: String,
746    #[serde(skip_serializing_if = "Option::is_none")]
747    pub from: Option<RecStatus>,
748    pub to: RecStatus,
749    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
750    /// `policy:auto`.
751    pub actor: String,
752    pub observer_type: ObserverType,
753    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
754    pub because: String,
755    #[serde(skip_serializing_if = "Option::is_none")]
756    pub previous_audit_hash: Option<String>,
757    /// §7.4: present exactly when this transition applied a code revision —
758    /// the evalset-run edge, recorded where the forensics look.
759    #[serde(default, skip_serializing_if = "Option::is_none")]
760    pub gating: Option<GatingEvidence>,
761    pub at_ms: i64,
762}
763
764/// Maximum length of a BECAUSE reason.
765pub const MAX_BECAUSE: usize = 500;
766
767impl AuditRecord {
768    /// Build the Observation grain that records this transition. `derived_from`
769    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
770    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
771        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
772        if let Some(prev) = &self.previous_audit_hash {
773            derived_from.push(Value::from(prev.clone()));
774        }
775        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
776            .with_field("observation_kind", "loop_audit")
777            .with_field("rec_hash", self.rec_hash.clone())
778            .with_field("to_status", self.to.as_str())
779            .with_field("actor", self.actor.clone())
780            .with_field(
781                "observer_type",
782                serde_json::to_value(self.observer_type).unwrap(),
783            )
784            .with_field("because", self.because.clone())
785            .with_field("at_ms", self.at_ms)
786            .with_field("derived_from", Value::Array(derived_from));
787        if let Some(from) = self.from {
788            spec.fields
789                .insert("from_status".into(), Value::from(from.as_str()));
790        }
791        if let Some(g) = &self.gating {
792            spec.fields
793                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
794            spec.fields
795                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
796            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
797            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
798        }
799        spec
800    }
801}
802
803/// A stored recommendation. `hash` (the content address) and `status` (the
804/// index-layer cache) are set by the engine, not serialized into the grain
805/// body — the body is immutable content, the lifecycle lives in the state
806/// index and the audit chain.
807/// One live lesson an authored lesson restates. `method` is `cosine`
808/// (the substrate's embedder, T1) or `jaccard` (normalized token sets, the
809/// T0 floor — weak, and honest about it).
810#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
811pub struct NearDuplicate {
812    pub hash: String,
813    pub score: f64,
814    pub method: String,
815}
816
817#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
818pub struct Recommendation {
819    #[serde(skip)]
820    pub hash: String,
821    pub analyzer: String,
822    pub params_snapshot: Map<String, Value>,
823    pub origin: Origin,
824    pub target_ref: String,
825    pub action_kind: ActionKind,
826    pub dedup_key: String,
827    pub summary: Summary,
828    pub severity: Severity,
829    #[serde(flatten)]
830    pub proposal: Proposal,
831    pub destructive: bool,
832    pub rollbackable: bool,
833    #[serde(default)]
834    pub evidence: Vec<String>,
835    #[serde(default, skip_serializing_if = "Option::is_none")]
836    pub evidence_query: Option<String>,
837    #[serde(default, skip_serializing_if = "Option::is_none")]
838    pub metric: Option<MetricSnapshot>,
839    pub confidence: f64,
840    pub importance: f64,
841    pub created_at_ms: i64,
842    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
843    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
844    /// text — it never replaces the engine-templated `summary`.
845    #[serde(default, skip_serializing_if = "Option::is_none")]
846    pub guidance: Option<String>,
847    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
848    /// against — REQUIRED for code targets, forbidden elsewhere (a
849    /// recommendation targets code XOR an evalset, never both, and only
850    /// code carries the pin). Shown at review; checked live at apply
851    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
852    #[serde(default, skip_serializing_if = "Option::is_none")]
853    pub evalset_hash: Option<String>,
854    /// Live lessons on the same entity this authored lesson restates in
855    /// other words (`Policy::near_duplicate = flag`). The reviewer's call
856    /// stays theirs — the card shows the existing rule beside the proposal.
857    #[serde(default, skip_serializing_if = "Vec::is_empty")]
858    pub near_duplicate_of: Vec<NearDuplicate>,
859    /// A `plan_revision`'s rehearsal against the journaled runs of the live
860    /// plan (the runtime's shadow report: per-run outcome under incumbent vs
861    /// candidate, effects out of support, spend) — computed at proposal
862    /// time when the substrate can, so the reviewer approves from evidence,
863    /// not prose. Absent when the substrate has no runtime or the plan has
864    /// no journaled runs.
865    #[serde(default, skip_serializing_if = "Option::is_none")]
866    pub replay: Option<Value>,
867    /// The namespaces the producing analyzer was RUN OVER (#312) —
868    /// normalized and sorted, stamped by the engine.
869    ///
870    /// The loop's inputs are namespace-grant-gated; its OUTPUTS were not. A
871    /// recommendation's summary, proposal, guidance and evidence hashes are
872    /// derived content, so one `read ON areev-loop` grant disclosed all of it
873    /// for every namespace in the memory, and one `loop.review` grant decided
874    /// all of it — including striking a memory-wide cooldown on a finding the
875    /// reviewer could not read the evidence for.
876    ///
877    /// Stamped where `dedup_key` and `origin` are, so an analyzer, an
878    /// external command or a model draft cannot set it.
879    /// `skip_serializing_if` empty, so stored recommendations read back
880    /// unchanged and an UNSCOPED pass produces an empty scope — which is
881    /// covered only by a whole-queue grant (fail closed).
882    #[serde(default, skip_serializing_if = "Vec::is_empty")]
883    pub scope: Vec<String>,
884    /// The decision backend that shaped this recommendation, with the
885    /// probabilities it answered (`docs/decision-model-proposal.md` rule 4).
886    /// A recommendation carrying one is NEVER auto-applied: a decision model
887    /// may score, only code — and a human — gates. Absent (and so absent from
888    /// the stored body) when no decision backend was involved.
889    #[serde(default, skip_serializing_if = "Option::is_none")]
890    pub judged_by: Option<crate::decide::JudgedBy>,
891    /// An LLM draft's verifier self-reported confidence, kept beside
892    /// `confidence` when a calibrated decision backend supplied the routing
893    /// number instead — so a reviewer sees any disagreement between the two.
894    #[serde(default, skip_serializing_if = "Option::is_none")]
895    pub llm_confidence: Option<f64>,
896    #[serde(skip)]
897    pub status: RecStatus,
898}
899
900/// Normalize a namespace list for [`Recommendation::scope`]: trimmed,
901/// non-empty, deduplicated, sorted — so two passes over the same set stamp
902/// the same scope and the coverage check is order-independent.
903pub fn normalize_scope(namespaces: &[String]) -> Vec<String> {
904    let mut out: Vec<String> = namespaces
905        .iter()
906        .map(|s| s.trim().to_string())
907        .filter(|s| !s.is_empty())
908        .collect();
909    out.sort();
910    out.dedup();
911    out
912}
913
914/// The recorded evalset-run edge (§7.4): what an apply of a code revision
915/// must present, and what its audit Observation carries. "Trust me, it
916/// passed" is not an edge; (evalset, run, stats) is.
917#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
918pub struct GatingEvidence {
919    /// The evalset the gate ran — must equal the recommendation's pin.
920    pub evalset_hash: String,
921    /// The `areev run`/`areev eval` run id that executed the gate.
922    pub run_id: String,
923    pub passed: u64,
924    pub failed: u64,
925}
926
927/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
928/// re-checked when a stored grain is loaded (a hand-authored grain must not
929/// bypass the rule):
930/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
931/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
932///   seam inherits §7.4's gate wholesale).
933/// - An `evalset:` target is its own always-human-approved class: never a
934///   gated revision, and never itself pinned (the gate cannot gate itself).
935/// - No other target class carries a pin.
936pub fn validate_code_rules(
937    action: ActionKind,
938    target_class: &str,
939    evalset_hash: Option<&str>,
940) -> Result<()> {
941    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
942    match (action, target_class) {
943        (ActionKind::CodeRevision, "code") => {
944            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
945                return e1(
946                    "a code_revision must pin the evalset hash it was gated \
947                     against (evalset_hash)",
948                );
949            }
950        }
951        (ActionKind::CodeRevision, other) => {
952            return e1(&format!(
953                "code_revision requires a tool: target, got class '{other}'"
954            ));
955        }
956        (ActionKind::AdapterRevision, "model") => {
957            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
958                return e1(
959                    "an adapter_revision must pin the evalset hash it was \
960                     gated against (evalset_hash)",
961                );
962            }
963        }
964        (ActionKind::AdapterRevision, other) => {
965            return e1(&format!(
966                "adapter_revision requires a model: target, got class '{other}'"
967            ));
968        }
969        // A REVERT of an applied code revision targets the same tool: —
970        // it is the rollback inverse, not new code, so it needs no pin
971        // (blocking it here would leave the highest-stakes target class
972        // unable to propose its own measured rollback).
973        (ActionKind::Revert, "code") => {
974            if evalset_hash.is_some() {
975                return e1("a code revert carries no evalset pin");
976            }
977        }
978        (_, "code") => {
979            return e1("a tool: target requires action_kind code_revision");
980        }
981        // Same rollback-inverse escape hatch for the adapter class.
982        (ActionKind::Revert, "model") => {
983            if evalset_hash.is_some() {
984                return e1("an adapter revert carries no evalset pin");
985            }
986        }
987        (_, "model") => {
988            return e1("a model: target requires action_kind adapter_revision");
989        }
990        (_, "evalset") => {
991            if evalset_hash.is_some() {
992                return e1(
993                    "an evalset-target recommendation cannot pin an evalset — \
994                     the gate cannot gate itself",
995                );
996            }
997        }
998        _ => {
999            if evalset_hash.is_some() {
1000                return e1(
1001                    "evalset_hash is only valid on code_revision or \
1002                     adapter_revision recommendations",
1003                );
1004            }
1005        }
1006    }
1007    Ok(())
1008}
1009
1010impl Recommendation {
1011    /// Serialize the immutable body into grain fields (excludes hash/status).
1012    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
1013        let value = serde_json::to_value(self)
1014            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
1015        let obj = value
1016            .as_object()
1017            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
1018            .clone();
1019        Ok(GrainSpec {
1020            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
1021            namespace: namespace.to_string(),
1022            fields: obj,
1023        })
1024    }
1025
1026    /// Reconstruct from a stored grain body. `hash` comes from the record's
1027    /// address; `status` is supplied from the state index by the caller.
1028    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
1029    /// must not smuggle an unpinned code revision past the stamped path.
1030    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
1031        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
1032            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
1033        rec.hash = hash.to_string();
1034        let class = crate::model::TargetRef::parse(&rec.target_ref)
1035            .map(|t| t.target_class())
1036            .unwrap_or("host");
1037        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
1038        Ok(rec)
1039    }
1040}
1041
1042#[cfg(test)]
1043mod tests {
1044    /// The gate and `areev eval run --tolerance` call this one function, so
1045    /// the boundary is the same for both readers by construction: a drop of
1046    /// exactly the tolerance holds, a hair past it regresses, and a tolerance
1047    /// that is not a positive finite number is zero.
1048    #[test]
1049    fn the_regression_rule_holds_at_the_tolerance_and_regresses_past_it() {
1050        use super::is_regression;
1051        // Percentage points, the way `eval run --tolerance` reads it.
1052        assert!(!is_regression(100.0, 0.0, true, 100.0), "a 100-point drop under a 100-point tolerance is accepted");
1053        assert!(is_regression(100.0, 0.0, true, 99.9));
1054        assert!(!is_regression(92.0, 91.5, true, 1.0));
1055        assert!(is_regression(92.0, 90.5, true, 1.0));
1056        // Counts, the way the gate reads `min_effect.count`.
1057        assert!(!is_regression(359.0, 355.0, true, 5.0));
1058        assert!(!is_regression(359.0, 354.0, true, 5.0), "exactly at the floor holds");
1059        assert!(is_regression(359.0, 353.0, true, 5.0));
1060        assert!(!is_regression(28.0, 33.0, false, 5.0));
1061        assert!(is_regression(28.0, 34.0, false, 5.0));
1062        // Zero is the pre-policy rule; garbage is zero, never a wider floor.
1063        assert!(is_regression(359.0, 355.0, true, 0.0));
1064        assert!(is_regression(359.0, 355.0, true, -5.0));
1065        assert!(is_regression(359.0, 355.0, true, f64::NAN));
1066        assert!(!is_regression(1.0, 1.0, true, 0.0), "equal is never a regression");
1067    }
1068
1069    use super::*;
1070    use serde_json::json;
1071
1072    #[test]
1073    fn summary_renders_deterministically() {
1074        let mut args = Map::new();
1075        args.insert("count".into(), json!(3));
1076        args.insert("subject".into(), json!("acme"));
1077        let s = Summary::new("duplicate.exact", args);
1078        assert_eq!(
1079            s.render(),
1080            "Consolidate 3 exact-duplicate grains for \"acme\""
1081        );
1082    }
1083
1084    #[test]
1085    fn dedup_key_ignores_content_and_case() {
1086        let a = dedup_key(
1087            "loop.duplicate_sweep",
1088            "entity:NS/John",
1089            ActionKind::Consolidate,
1090        );
1091        let b = dedup_key(
1092            "loop.duplicate_sweep",
1093            "entity:ns/john",
1094            ActionKind::Consolidate,
1095        );
1096        assert_eq!(a, b, "case-folded to one identity");
1097    }
1098
1099    #[test]
1100    fn authored_dedup_key_distinguishes_content_but_not_wording_noise() {
1101        let a = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the vendor name.");
1102        let same = authored_dedup_key("llm", "entity:NS/X", ActionKind::Record, "  record THE vendor  name ");
1103        let other = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the amount.");
1104        assert_eq!(a, same, "case, punctuation and spacing are not a new lesson");
1105        assert_ne!(a, other, "a different lesson on the same entity is a different finding");
1106        assert!(a.starts_with(&dedup_key("llm", "entity:ns/x", ActionKind::Record)));
1107        assert_eq!(content_fingerprint("A b"), content_fingerprint("a-b"));
1108        assert_ne!(content_fingerprint("ab"), content_fingerprint("a b"));
1109    }
1110
1111    #[test]
1112    fn a_revert_is_keyed_by_what_it_reverts() {
1113        let a = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "aaaa");
1114        let b = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "bbbb");
1115        let a2 = revert_dedup_key("loop.outcome_review", "entity:NS/Capture", "AAAA");
1116        assert_ne!(a, b, "two reverts on one target are two findings");
1117        assert_eq!(a, a2, "the same revert, case-folded, is one");
1118        assert!(a.starts_with(&dedup_key("loop.outcome_review", "entity:ns/capture", ActionKind::Revert)));
1119    }
1120
1121    #[test]
1122    fn dedup_key_distinguishes_action() {
1123        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
1124        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
1125        assert_ne!(a, b);
1126    }
1127
1128    #[test]
1129    fn lifecycle_gates_pending_to_applied() {
1130        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
1131        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
1132        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
1133        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
1134        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
1135        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
1136    }
1137
1138    #[test]
1139    fn rule_e1_pins_code_and_only_code() {
1140        use crate::model::ActionKind as A;
1141        // code_revision on a tool target with a pin: OK.
1142        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
1143        // Unpinned code revision: refused.
1144        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
1145        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
1146        // code_revision off a tool target: refused.
1147        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
1148        // A tool target with a non-code action: refused.
1149        assert!(validate_code_rules(A::Flag, "code", None).is_err());
1150        // The gate cannot gate itself.
1151        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
1152        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
1153        // Pins don't leak onto ordinary targets.
1154        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
1155        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
1156    }
1157
1158    #[test]
1159    fn rule_e1_pins_adapter_revisions_like_code() {
1160        use crate::model::ActionKind as A;
1161        // adapter_revision on a model target with a pin: OK.
1162        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
1163        // Unpinned adapter revision: refused.
1164        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
1165        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
1166        // adapter_revision off a model target: refused.
1167        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
1168        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
1169        // A model target with a non-adapter action: refused.
1170        assert!(validate_code_rules(A::Flag, "model", None).is_err());
1171        // The rollback inverse needs no pin — and must carry none.
1172        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
1173        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
1174    }
1175
1176    #[test]
1177    fn from_fields_rechecks_rule_e1() {
1178        // A hand-authored grain body carrying an unpinned code revision must
1179        // not decode into a usable recommendation.
1180        let mut rec = Recommendation {
1181            hash: String::new(),
1182            analyzer: "loop.codegen/1".into(),
1183            params_snapshot: Map::new(),
1184            origin: Origin::Builtin,
1185            target_ref: "tool:abc123".into(),
1186            action_kind: ActionKind::CodeRevision,
1187            dedup_key: "k".into(),
1188            summary: Summary::new("command.finding", Map::new()),
1189            severity: Severity::Low,
1190            proposal: Proposal::Data { data: Map::new() },
1191            destructive: false,
1192            rollbackable: false,
1193            evidence: vec![],
1194            evidence_query: None,
1195            metric: None,
1196            confidence: 0.5,
1197            importance: 0.5,
1198            created_at_ms: 0,
1199            guidance: None,
1200            evalset_hash: None, // the smuggle attempt
1201            near_duplicate_of: Vec::new(),
1202            replay: None,
1203            status: RecStatus::Pending,
1204            scope: Vec::new(),
1205            judged_by: None,
1206            llm_confidence: None,
1207        };
1208        let spec = rec.to_grain_spec("ns").unwrap();
1209        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
1210        rec.evalset_hash = Some("es-hash".into());
1211        let spec = rec.to_grain_spec("ns").unwrap();
1212        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
1213        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
1214    }
1215
1216    #[test]
1217    fn recommendation_round_trips_through_fields() {
1218        let rec = Recommendation {
1219            hash: "ignored".into(),
1220            analyzer: "loop.staleness/1".into(),
1221            params_snapshot: Map::new(),
1222            origin: Origin::Builtin,
1223            target_ref: "grain:sha256:abc".into(),
1224            action_kind: ActionKind::Expire,
1225            dedup_key: "k".into(),
1226            summary: Summary::new("staleness.expired", Map::new()),
1227            severity: Severity::Low,
1228            proposal: Proposal::Cal {
1229                cal: "FORGET sha256:abc".into(),
1230            },
1231            destructive: true,
1232            rollbackable: false,
1233            evidence: vec!["sha256:abc".into()],
1234            evidence_query: None,
1235            metric: None,
1236            confidence: 0.9,
1237            importance: 0.4,
1238            created_at_ms: 1000,
1239            guidance: None,
1240            evalset_hash: None,
1241            near_duplicate_of: Vec::new(),
1242            replay: None,
1243            status: RecStatus::Pending,
1244            scope: Vec::new(),
1245            judged_by: None,
1246            llm_confidence: None,
1247        };
1248        let spec = rec.to_grain_spec("ns").unwrap();
1249        // hash and status are excluded from the immutable body.
1250        assert!(!spec.fields.contains_key("hash"));
1251        assert!(!spec.fields.contains_key("status"));
1252        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
1253        assert_eq!(back.hash, "realhash");
1254        assert_eq!(back.analyzer, "loop.staleness/1");
1255        assert!(back.destructive);
1256        assert!(matches!(back.proposal, Proposal::Cal { .. }));
1257    }
1258}