Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel —
9//!   for ANALYZER findings. An authored (`origin = llm`) executable proposal
10//!   keys on a fingerprint of its content as well, because there the content
11//!   IS the finding: two different lessons on one entity are two findings,
12//!   and the same lesson re-authored is one.
13//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
14
15use crate::error::{Error, Result};
16use crate::model::{normalize_ident, ActionKind, Origin, Severity};
17use crate::substrate::GrainSpec;
18use serde::{Deserialize, Serialize};
19use serde_json::{Map, Value};
20
21/// A deterministic, template-rendered summary. The analyzer chooses a
22/// `template_id` and supplies `args`; the text is produced here, so an
23/// analyzer can never emit arbitrary prose into the queue.
24#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
25pub struct Summary {
26    pub template_id: String,
27    #[serde(default)]
28    pub args: Map<String, Value>,
29}
30
31impl Summary {
32    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
33        Summary {
34            template_id: template_id.into(),
35            args,
36        }
37    }
38
39    /// Render the summary. Unknown template ids fall back to a stable, honest
40    /// string (never a panic) so a manifest/template mismatch is visible, not
41    /// fatal.
42    pub fn render(&self) -> String {
43        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
44        interpolate(t, &self.args, &self.template_id)
45    }
46}
47
48/// The built-in template table. Deterministic; the only place summary prose
49/// lives. Keep placeholders in `{name}` form matching `args` keys.
50fn builtin_template(id: &str) -> Option<&'static str> {
51    Some(match id {
52        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
53        "duplicate.near" => {
54            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
55        }
56        "contradiction.functional" => {
57            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
58        }
59        "tool_failure.cluster" => {
60            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
61             fail this way): {signature}"
62        }
63        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
64        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
65        "skill.stall" => {
66            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
67        }
68        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
69        "cold.grain" => {
70            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
71        }
72        "coverage.gap" => {
73            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
74        }
75        "budget.pressure" => {
76            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
77        }
78        "retention.overdue" => {
79            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
80        }
81        "outcome.regression" => {
82            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
83        }
84        "outcome.premise_drift" => {
85            "{current} of the grains this recommendation cited have since been superseded by a different value or retracted — its premise moved; revert it"
86        }
87        "run.failures" => {
88            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
89        }
90        "run.cost" => {
91            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
92        }
93        "run.context_pressure" => {
94            "Workflow {workflow} summarized its own transcript {folds} time(s) across {runs} runs (avg {avg_folds}/run) — its nodes are outgrowing the model's window"
95        }
96        "adapter.candidate" => {
97            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
98        }
99        // origin=llm drafts carry free text (clearly marked llm) rather than a
100        // deterministic template — the model's proposed summary rides in {text}.
101        "llm.discover" => "{text}",
102        // An LLM-authored lesson shows the reviewer BOTH the finding and the
103        // exact line an apply would record — never one without the other.
104        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
105        // The rest of the LLM proposal vocabulary follows the same rule: the
106        // finding AND the exact change an apply would make, never one without
107        // the other. A reviewer approving blind is the failure this prevents.
108        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
109        "llm.skill" => "{text} — record skill: \"{name}\" ({steps} steps)",
110        "llm.plan" => "{text} — record plan: \"{name}\" ({nodes} steps, {edges} edges)",
111        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
112        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
113        "llm.code_revision" => {
114            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
115        }
116        // External command analyzers (trust class Command): free text from the
117        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
118        "command.finding" => "{text}",
119        _ => return None,
120    })
121}
122
123fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
124    let mut out = String::with_capacity(template.len());
125    let mut chars = template.chars().peekable();
126    while let Some(c) = chars.next() {
127        if c == '{' {
128            let mut key = String::new();
129            for k in chars.by_ref() {
130                if k == '}' {
131                    break;
132                }
133                key.push(k);
134            }
135            if key == "template_id" {
136                out.push_str(template_id);
137            } else {
138                match args.get(&key) {
139                    Some(Value::String(s)) => out.push_str(s),
140                    Some(v) => out.push_str(&v.to_string()),
141                    None => {
142                        // Missing arg — leave a visible marker rather than lie.
143                        out.push('{');
144                        out.push_str(&key);
145                        out.push('}');
146                    }
147                }
148            }
149        } else {
150            out.push(c);
151        }
152    }
153    out
154}
155
156/// When an applied recommendation is re-measured — one checkpoint of the
157/// Verify gate's schedule, in the unit the deployment actually counts in.
158///
159/// Three units, because deployments count differently and a fixed one makes
160/// the gate inert everywhere else. A service evaluated nightly counts
161/// **time** (`after_ms`). A benchmark or a CI harness counts **graded runs**
162/// (`after_runs`): "measure at the next evaluation after the apply", however
163/// long that takes on the clock. A chat agent counts **turns** (`after_grains`):
164/// "after fifty more grains". The default schedule stayed ms-only for a year
165/// and was measured, on PAST-Bench, to fire exactly zero verdicts across 78
166/// governed runs — a family finishes in seven minutes and the first checkpoint
167/// was a day away (`crates/areev-bench/PERSIST.md`).
168///
169/// On the wire a bare integer is milliseconds, so every policy, metric
170/// snapshot and state blob written before this type existed reads back
171/// unchanged, and an all-ms schedule still serializes as `[86400000, …]`.
172#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
173pub enum Checkpoint {
174    /// Elapsed time since the apply.
175    AfterMs(i64),
176    /// Evalset runs journaled since the apply. Only an `evalset:` metric can
177    /// count these; on any other metric the checkpoint never comes due.
178    AfterRuns(u32),
179    /// Grains written since the apply — the loop's own notion of "activity".
180    AfterGrains(u32),
181}
182
183impl Checkpoint {
184    /// The label a person reads: `1d`, `12h`, `3 runs`, `50 grains`.
185    pub fn label(&self) -> String {
186        match self {
187            Checkpoint::AfterMs(ms) if ms % 86_400_000 == 0 => format!("{}d", ms / 86_400_000),
188            Checkpoint::AfterMs(ms) if ms % 3_600_000 == 0 => format!("{}h", ms / 3_600_000),
189            Checkpoint::AfterMs(ms) => format!("{ms}ms"),
190            Checkpoint::AfterRuns(n) => format!("{n} run{}", if *n == 1 { "" } else { "s" }),
191            Checkpoint::AfterGrains(n) => format!("{n} grain{}", if *n == 1 { "" } else { "s" }),
192        }
193    }
194
195    /// The ms value when this is a time checkpoint — what legacy consumers of
196    /// `OutcomeResult::horizon_ms` read.
197    pub fn as_ms(&self) -> Option<i64> {
198        match self {
199            Checkpoint::AfterMs(ms) => Some(*ms),
200            _ => None,
201        }
202    }
203}
204
205impl Serialize for Checkpoint {
206    fn serialize<S: serde::Serializer>(&self, ser: S) -> std::result::Result<S::Ok, S::Error> {
207        use serde::ser::SerializeMap;
208        match self {
209            // Bare integer: byte-identical to the pre-Checkpoint `Vec<i64>`.
210            Checkpoint::AfterMs(ms) => ser.serialize_i64(*ms),
211            Checkpoint::AfterRuns(n) => {
212                let mut m = ser.serialize_map(Some(1))?;
213                m.serialize_entry("after_runs", n)?;
214                m.end()
215            }
216            Checkpoint::AfterGrains(n) => {
217                let mut m = ser.serialize_map(Some(1))?;
218                m.serialize_entry("after_grains", n)?;
219                m.end()
220            }
221        }
222    }
223}
224
225impl<'de> Deserialize<'de> for Checkpoint {
226    fn deserialize<D: serde::Deserializer<'de>>(de: D) -> std::result::Result<Self, D::Error> {
227        use serde::de::Error as _;
228        let v = Value::deserialize(de)?;
229        match &v {
230            Value::Number(n) => n
231                .as_i64()
232                .filter(|ms| *ms >= 0)
233                .map(Checkpoint::AfterMs)
234                .ok_or_else(|| D::Error::custom("checkpoint: a bare number is non-negative milliseconds")),
235            Value::Object(m) if m.len() == 1 => {
236                let (k, val) = m.iter().next().unwrap();
237                let n = val.as_u64().ok_or_else(|| {
238                    D::Error::custom(format!("checkpoint: {k} takes a non-negative integer"))
239                })?;
240                match k.as_str() {
241                    "after_ms" => i64::try_from(n)
242                        .map(Checkpoint::AfterMs)
243                        .map_err(|_| D::Error::custom("checkpoint: after_ms out of range")),
244                    "after_runs" => u32::try_from(n)
245                        .map(Checkpoint::AfterRuns)
246                        .map_err(|_| D::Error::custom("checkpoint: after_runs out of range")),
247                    "after_grains" => u32::try_from(n)
248                        .map(Checkpoint::AfterGrains)
249                        .map_err(|_| D::Error::custom("checkpoint: after_grains out of range")),
250                    other => Err(D::Error::custom(format!(
251                        "checkpoint: unknown unit {other:?} (expected after_ms, after_runs or after_grains)"
252                    ))),
253                }
254            }
255            _ => Err(D::Error::custom(
256                "checkpoint: expected milliseconds or {\"after_ms\"|\"after_runs\"|\"after_grains\": n}",
257            )),
258        }
259    }
260}
261
262/// A reproducible metric snapshot; powers outcome review.
263#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
264pub struct MetricSnapshot {
265    /// Metric kind — the engine knows how to re-measure a fixed set
266    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
267    pub metric: String,
268    pub baseline: f64,
269    pub unit: String,
270    pub n: u64,
271    pub window: String,
272    /// The subject the metric is about (e.g. the tool name), used by the
273    /// engine's typed re-measurement.
274    #[serde(default, skip_serializing_if = "Option::is_none")]
275    pub subject: Option<String>,
276    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
277    /// `None` = all namespaces. Additive: older snapshots deserialize to
278    /// `None` and existing metric kinds ignore it.
279    #[serde(default, skip_serializing_if = "Option::is_none")]
280    pub namespace: Option<String>,
281    /// Relation scope for fact-shaped metrics (e.g. which functional relation
282    /// a contradiction was resolved under), normalized. Additive like
283    /// `namespace`.
284    #[serde(default, skip_serializing_if = "Option::is_none")]
285    pub relation: Option<String>,
286    /// CAL that recomputes the metric at verify time (reproducibility /
287    /// documentation; the engine re-measures with typed reads).
288    pub query: String,
289    /// How long after apply to re-measure, in epoch-ms delta (the first / only
290    /// checkpoint when `horizons_ms` is empty).
291    pub review_after_ms: i64,
292    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
293    /// that `held` at an early checkpoint can `regress` at a later one, so a
294    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
295    #[serde(default, skip_serializing_if = "Vec::is_empty")]
296    pub horizons_ms: Vec<i64>,
297    /// The schedule in the deployment's own unit ([`Checkpoint`]). When set it
298    /// is THE schedule and `horizons_ms`/`review_after_ms` are ignored; empty
299    /// (every snapshot written before it existed) falls back to them.
300    #[serde(default, skip_serializing_if = "Vec::is_empty")]
301    pub checkpoints: Vec<Checkpoint>,
302    /// Which direction is an improvement. The built-in metrics are all
303    /// recurrence counts, where lower is better — so this defaults to `false`
304    /// and every existing snapshot deserializes unchanged. An evalset accuracy
305    /// is the opposite, and getting it wrong does not merely misreport: the
306    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
307    /// propose reverting it.
308    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
309    pub higher_is_better: bool,
310}
311
312/// The one regression rule.
313///
314/// The verdict is computed in two places — once by the engine for the recorded
315/// `OutcomeResult`, once by `outcome_review` when it drafts the revert — and
316/// the two silently disagreeing would either revert a change that held or sit
317/// on one that regressed. So both call this.
318pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool) -> bool {
319    /// Minimum worsening to call a regression (avoids noise at n=1).
320    const EPSILON: f64 = 1e-9;
321    if higher_is_better {
322        current < baseline - EPSILON
323    } else {
324        current > baseline + EPSILON
325    }
326}
327
328impl MetricSnapshot {
329    /// The measurement schedule, sorted and deduplicated: `checkpoints` when
330    /// set, else `horizons_ms`, else the single `review_after_ms` — each older
331    /// spelling read as time, so a snapshot from before [`Checkpoint`] existed
332    /// measures exactly as it did.
333    pub fn schedule(&self) -> Vec<Checkpoint> {
334        let mut h: Vec<Checkpoint> = if !self.checkpoints.is_empty() {
335            self.checkpoints.clone()
336        } else if !self.horizons_ms.is_empty() {
337            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
338        } else {
339            vec![Checkpoint::AfterMs(self.review_after_ms)]
340        };
341        h.sort_unstable();
342        h.dedup();
343        h
344    }
345
346    /// The time checkpoints of the schedule, in ms — the pre-[`Checkpoint`]
347    /// view, kept for callers that only understand time.
348    pub fn horizons(&self) -> Vec<i64> {
349        self.schedule().iter().filter_map(Checkpoint::as_ms).collect()
350    }
351}
352
353/// A measured outcome for an applied recommendation at one checkpoint — the
354/// Verify gate's output. `held` = the metric did not regress at this horizon;
355/// `regressed` = it got worse (a revert is proposed). A recommendation
356/// accumulates one of these per horizon, forming a time series.
357#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
358pub struct OutcomeResult {
359    pub rec_hash: String,
360    pub metric: String,
361    pub baseline: f64,
362    pub current: f64,
363    pub verdict: String,
364    /// Which checkpoint this measurement is for (ms after apply). Zero for a
365    /// checkpoint counted in runs or grains — see `checkpoint`.
366    #[serde(default)]
367    pub horizon_ms: i64,
368    /// The checkpoint in its own unit, when that unit is not time. A time
369    /// checkpoint leaves this unset and speaks through `horizon_ms`, so every
370    /// consumer of the older shape reads an unchanged record.
371    #[serde(default, skip_serializing_if = "Option::is_none")]
372    pub checkpoint: Option<Checkpoint>,
373    pub measured_at_ms: i64,
374}
375
376/// The proposed change. Exactly one variant per recommendation.
377#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
378#[serde(tag = "proposal", rename_all = "snake_case")]
379pub enum Proposal {
380    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
381    Cal { cal: String },
382    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
383    /// staleness check at apply).
384    Edit {
385        format: String,
386        base_digest: String,
387        diff: String,
388    },
389    /// An opaque map for host targets (applied by the host, §12.3).
390    Data { data: Map<String, Value> },
391}
392
393/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
394/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
395/// fields without breaking analyzers.
396#[derive(Debug, Clone, PartialEq)]
397#[non_exhaustive]
398pub struct RecDraft {
399    pub target_ref: String,
400    pub action_kind: ActionKind,
401    pub summary: Summary,
402    pub severity: Severity,
403    pub proposal: Proposal,
404    /// Bounded to ≤64 representative evidence hashes by the engine.
405    pub evidence: Vec<String>,
406    pub evidence_query: Option<String>,
407    pub metric: Option<MetricSnapshot>,
408    pub confidence: f64,
409    pub importance: f64,
410    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
411    pub evalset_hash: Option<String>,
412}
413
414impl RecDraft {
415    pub fn new(
416        target_ref: impl Into<String>,
417        action_kind: ActionKind,
418        summary: Summary,
419        proposal: Proposal,
420    ) -> Self {
421        RecDraft {
422            target_ref: target_ref.into(),
423            action_kind,
424            summary,
425            severity: Severity::Low,
426            proposal,
427            evidence: Vec::new(),
428            evidence_query: None,
429            metric: None,
430            confidence: 0.8,
431            importance: 0.5,
432            evalset_hash: None,
433        }
434    }
435
436    pub fn severity(mut self, s: Severity) -> Self {
437        self.severity = s;
438        self
439    }
440    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
441        self.evidence = hashes;
442        self
443    }
444    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
445        self.evidence_query = Some(q.into());
446        self
447    }
448    pub fn metric(mut self, m: MetricSnapshot) -> Self {
449        self.metric = Some(m);
450        self
451    }
452    pub fn confidence(mut self, c: f64) -> Self {
453        self.confidence = c;
454        self
455    }
456    pub fn importance(mut self, i: f64) -> Self {
457        self.importance = i;
458        self
459    }
460    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
461        self.evalset_hash = Some(h.into());
462        self
463    }
464}
465
466/// Maximum representative evidence hashes carried inline (proposal §7.1).
467pub const MAX_EVIDENCE: usize = 64;
468
469/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
470/// Excludes proposal content and evidence by construction.
471pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
472    // U+001F (unit separator) cannot appear in any of the inputs.
473    format!(
474        "{}\u{1f}{}\u{1f}{}",
475        normalize_ident(family),
476        normalize_ident(target_ref),
477        action.as_str()
478    )
479}
480
481/// The dedup key of an AUTHORED executable proposal: [`dedup_key`] plus a
482/// fingerprint of the proposal's content. An analyzer finding is "this
483/// target has this kind of problem", so content is rightly excluded; an
484/// authored lesson is "do this", and two different lessons on the same
485/// entity must both reach the queue while the same lesson re-authored must
486/// not. The fingerprint is over the normalized text — case-folded, non-
487/// alphanumerics dropped, whitespace collapsed — so a rewording that changes
488/// no word is the same lesson and one that changes a word is a new one (a
489/// semantic near-duplicate is the reviewer's call, not this key's).
490pub fn authored_dedup_key(
491    family: &str,
492    target_ref: &str,
493    action: ActionKind,
494    content: &str,
495) -> String {
496    format!(
497        "{}\u{1f}{}",
498        dedup_key(family, target_ref, action),
499        content_fingerprint(content)
500    )
501}
502
503/// The dedup key of a REVERT: [`dedup_key`] plus the hash of the applied
504/// recommendation it retracts. An analyzer finding is "this target has this
505/// kind of problem", so two findings on one target rightly collapse — but a
506/// revert is about one specific applied recommendation, and two lessons on
507/// the same entity that both regressed need two reverts. Until 2026-09-06
508/// they shared a key and the second was dropped as a duplicate of the first
509/// (`crates/areev-bench/CURVE.md`, seed 3: two regressed, one revert).
510pub fn revert_dedup_key(family: &str, target_ref: &str, revert_of: &str) -> String {
511    format!(
512        "{}\u{1f}{}",
513        dedup_key(family, target_ref, ActionKind::Revert),
514        normalize_ident(revert_of)
515    )
516}
517
518/// Sixteen hex chars of FNV-1a (64-bit) over the normalized content. A dedup
519/// key needs stability and spread, not cryptographic strength — a collision
520/// here would merge two findings in a review queue, never grant anything —
521/// so this stays dependency-free, as the crate is by policy.
522pub fn content_fingerprint(content: &str) -> String {
523    let mut normalized = String::with_capacity(content.len());
524    let mut pending_space = false;
525    for c in content.chars() {
526        if c.is_alphanumeric() {
527            if pending_space && !normalized.is_empty() {
528                normalized.push(' ');
529            }
530            pending_space = false;
531            normalized.extend(c.to_lowercase());
532        } else if c.is_whitespace() || !c.is_alphanumeric() {
533            pending_space = true;
534        }
535    }
536    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
537    for b in normalized.as_bytes() {
538        h ^= u64::from(*b);
539        h = h.wrapping_mul(0x0000_0100_0000_01b3);
540    }
541    format!("{h:016x}")
542}
543
544/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
545/// content hash is stable for its whole life).
546#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
547#[serde(rename_all = "snake_case")]
548pub enum RecStatus {
549    #[default]
550    Pending,
551    Approved,
552    Rejected,
553    Applied,
554    RolledBack,
555    Expired,
556}
557
558impl RecStatus {
559    pub fn as_str(&self) -> &'static str {
560        match self {
561            RecStatus::Pending => "pending",
562            RecStatus::Approved => "approved",
563            RecStatus::Rejected => "rejected",
564            RecStatus::Applied => "applied",
565            RecStatus::RolledBack => "rolled_back",
566            RecStatus::Expired => "expired",
567        }
568    }
569
570    /// Is a transition to `to` allowed from this state? `by_policy` marks the
571    /// auto-apply actor, the only one permitted the reasonless
572    /// `pending → applied` jump.
573    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
574        use RecStatus::*;
575        match (self, to) {
576            (Pending, Approved) | (Pending, Rejected) => true,
577            (Pending, Applied) => by_policy, // auto-apply only
578            (Approved, Applied) => true,
579            (Applied, RolledBack) => true,
580            // `expired` is computed from valid_to, applied to still-open recs.
581            (Pending, Expired) | (Approved, Expired) => true,
582            _ => false,
583        }
584    }
585}
586
587/// Who/what performed a transition.
588#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
589#[serde(rename_all = "lowercase")]
590pub enum ObserverType {
591    Human,
592    Agent,
593    Policy,
594    System,
595}
596
597/// One immutable audit Observation per transition, hash-chained per
598/// recommendation.
599#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
600pub struct AuditRecord {
601    pub rec_hash: String,
602    #[serde(skip_serializing_if = "Option::is_none")]
603    pub from: Option<RecStatus>,
604    pub to: RecStatus,
605    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
606    /// `policy:auto`.
607    pub actor: String,
608    pub observer_type: ObserverType,
609    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
610    pub because: String,
611    #[serde(skip_serializing_if = "Option::is_none")]
612    pub previous_audit_hash: Option<String>,
613    /// §7.4: present exactly when this transition applied a code revision —
614    /// the evalset-run edge, recorded where the forensics look.
615    #[serde(default, skip_serializing_if = "Option::is_none")]
616    pub gating: Option<GatingEvidence>,
617    pub at_ms: i64,
618}
619
620/// Maximum length of a BECAUSE reason.
621pub const MAX_BECAUSE: usize = 500;
622
623impl AuditRecord {
624    /// Build the Observation grain that records this transition. `derived_from`
625    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
626    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
627        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
628        if let Some(prev) = &self.previous_audit_hash {
629            derived_from.push(Value::from(prev.clone()));
630        }
631        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
632            .with_field("observation_kind", "loop_audit")
633            .with_field("rec_hash", self.rec_hash.clone())
634            .with_field("to_status", self.to.as_str())
635            .with_field("actor", self.actor.clone())
636            .with_field(
637                "observer_type",
638                serde_json::to_value(self.observer_type).unwrap(),
639            )
640            .with_field("because", self.because.clone())
641            .with_field("at_ms", self.at_ms)
642            .with_field("derived_from", Value::Array(derived_from));
643        if let Some(from) = self.from {
644            spec.fields
645                .insert("from_status".into(), Value::from(from.as_str()));
646        }
647        if let Some(g) = &self.gating {
648            spec.fields
649                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
650            spec.fields
651                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
652            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
653            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
654        }
655        spec
656    }
657}
658
659/// A stored recommendation. `hash` (the content address) and `status` (the
660/// index-layer cache) are set by the engine, not serialized into the grain
661/// body — the body is immutable content, the lifecycle lives in the state
662/// index and the audit chain.
663#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
664pub struct Recommendation {
665    #[serde(skip)]
666    pub hash: String,
667    pub analyzer: String,
668    pub params_snapshot: Map<String, Value>,
669    pub origin: Origin,
670    pub target_ref: String,
671    pub action_kind: ActionKind,
672    pub dedup_key: String,
673    pub summary: Summary,
674    pub severity: Severity,
675    #[serde(flatten)]
676    pub proposal: Proposal,
677    pub destructive: bool,
678    pub rollbackable: bool,
679    #[serde(default)]
680    pub evidence: Vec<String>,
681    #[serde(default, skip_serializing_if = "Option::is_none")]
682    pub evidence_query: Option<String>,
683    #[serde(default, skip_serializing_if = "Option::is_none")]
684    pub metric: Option<MetricSnapshot>,
685    pub confidence: f64,
686    pub importance: f64,
687    pub created_at_ms: i64,
688    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
689    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
690    /// text — it never replaces the engine-templated `summary`.
691    #[serde(default, skip_serializing_if = "Option::is_none")]
692    pub guidance: Option<String>,
693    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
694    /// against — REQUIRED for code targets, forbidden elsewhere (a
695    /// recommendation targets code XOR an evalset, never both, and only
696    /// code carries the pin). Shown at review; checked live at apply
697    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
698    #[serde(default, skip_serializing_if = "Option::is_none")]
699    pub evalset_hash: Option<String>,
700    #[serde(skip)]
701    pub status: RecStatus,
702}
703
704/// The recorded evalset-run edge (§7.4): what an apply of a code revision
705/// must present, and what its audit Observation carries. "Trust me, it
706/// passed" is not an edge; (evalset, run, stats) is.
707#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
708pub struct GatingEvidence {
709    /// The evalset the gate ran — must equal the recommendation's pin.
710    pub evalset_hash: String,
711    /// The `areev run`/`areev eval` run id that executed the gate.
712    pub run_id: String,
713    pub passed: u64,
714    pub failed: u64,
715}
716
717/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
718/// re-checked when a stored grain is loaded (a hand-authored grain must not
719/// bypass the rule):
720/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
721/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
722///   seam inherits §7.4's gate wholesale).
723/// - An `evalset:` target is its own always-human-approved class: never a
724///   gated revision, and never itself pinned (the gate cannot gate itself).
725/// - No other target class carries a pin.
726pub fn validate_code_rules(
727    action: ActionKind,
728    target_class: &str,
729    evalset_hash: Option<&str>,
730) -> Result<()> {
731    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
732    match (action, target_class) {
733        (ActionKind::CodeRevision, "code") => {
734            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
735                return e1(
736                    "a code_revision must pin the evalset hash it was gated \
737                     against (evalset_hash)",
738                );
739            }
740        }
741        (ActionKind::CodeRevision, other) => {
742            return e1(&format!(
743                "code_revision requires a tool: target, got class '{other}'"
744            ));
745        }
746        (ActionKind::AdapterRevision, "model") => {
747            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
748                return e1(
749                    "an adapter_revision must pin the evalset hash it was \
750                     gated against (evalset_hash)",
751                );
752            }
753        }
754        (ActionKind::AdapterRevision, other) => {
755            return e1(&format!(
756                "adapter_revision requires a model: target, got class '{other}'"
757            ));
758        }
759        // A REVERT of an applied code revision targets the same tool: —
760        // it is the rollback inverse, not new code, so it needs no pin
761        // (blocking it here would leave the highest-stakes target class
762        // unable to propose its own measured rollback).
763        (ActionKind::Revert, "code") => {
764            if evalset_hash.is_some() {
765                return e1("a code revert carries no evalset pin");
766            }
767        }
768        (_, "code") => {
769            return e1("a tool: target requires action_kind code_revision");
770        }
771        // Same rollback-inverse escape hatch for the adapter class.
772        (ActionKind::Revert, "model") => {
773            if evalset_hash.is_some() {
774                return e1("an adapter revert carries no evalset pin");
775            }
776        }
777        (_, "model") => {
778            return e1("a model: target requires action_kind adapter_revision");
779        }
780        (_, "evalset") => {
781            if evalset_hash.is_some() {
782                return e1(
783                    "an evalset-target recommendation cannot pin an evalset — \
784                     the gate cannot gate itself",
785                );
786            }
787        }
788        _ => {
789            if evalset_hash.is_some() {
790                return e1(
791                    "evalset_hash is only valid on code_revision or \
792                     adapter_revision recommendations",
793                );
794            }
795        }
796    }
797    Ok(())
798}
799
800impl Recommendation {
801    /// Serialize the immutable body into grain fields (excludes hash/status).
802    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
803        let value = serde_json::to_value(self)
804            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
805        let obj = value
806            .as_object()
807            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
808            .clone();
809        Ok(GrainSpec {
810            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
811            namespace: namespace.to_string(),
812            fields: obj,
813        })
814    }
815
816    /// Reconstruct from a stored grain body. `hash` comes from the record's
817    /// address; `status` is supplied from the state index by the caller.
818    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
819    /// must not smuggle an unpinned code revision past the stamped path.
820    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
821        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
822            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
823        rec.hash = hash.to_string();
824        let class = crate::model::TargetRef::parse(&rec.target_ref)
825            .map(|t| t.target_class())
826            .unwrap_or("host");
827        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
828        Ok(rec)
829    }
830}
831
832#[cfg(test)]
833mod tests {
834    use super::*;
835    use serde_json::json;
836
837    #[test]
838    fn summary_renders_deterministically() {
839        let mut args = Map::new();
840        args.insert("count".into(), json!(3));
841        args.insert("subject".into(), json!("acme"));
842        let s = Summary::new("duplicate.exact", args);
843        assert_eq!(
844            s.render(),
845            "Consolidate 3 exact-duplicate grains for \"acme\""
846        );
847    }
848
849    #[test]
850    fn dedup_key_ignores_content_and_case() {
851        let a = dedup_key(
852            "loop.duplicate_sweep",
853            "entity:NS/John",
854            ActionKind::Consolidate,
855        );
856        let b = dedup_key(
857            "loop.duplicate_sweep",
858            "entity:ns/john",
859            ActionKind::Consolidate,
860        );
861        assert_eq!(a, b, "case-folded to one identity");
862    }
863
864    #[test]
865    fn authored_dedup_key_distinguishes_content_but_not_wording_noise() {
866        let a = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the vendor name.");
867        let same = authored_dedup_key("llm", "entity:NS/X", ActionKind::Record, "  record THE vendor  name ");
868        let other = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the amount.");
869        assert_eq!(a, same, "case, punctuation and spacing are not a new lesson");
870        assert_ne!(a, other, "a different lesson on the same entity is a different finding");
871        assert!(a.starts_with(&dedup_key("llm", "entity:ns/x", ActionKind::Record)));
872        assert_eq!(content_fingerprint("A b"), content_fingerprint("a-b"));
873        assert_ne!(content_fingerprint("ab"), content_fingerprint("a b"));
874    }
875
876    #[test]
877    fn a_revert_is_keyed_by_what_it_reverts() {
878        let a = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "aaaa");
879        let b = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "bbbb");
880        let a2 = revert_dedup_key("loop.outcome_review", "entity:NS/Capture", "AAAA");
881        assert_ne!(a, b, "two reverts on one target are two findings");
882        assert_eq!(a, a2, "the same revert, case-folded, is one");
883        assert!(a.starts_with(&dedup_key("loop.outcome_review", "entity:ns/capture", ActionKind::Revert)));
884    }
885
886    #[test]
887    fn dedup_key_distinguishes_action() {
888        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
889        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
890        assert_ne!(a, b);
891    }
892
893    #[test]
894    fn lifecycle_gates_pending_to_applied() {
895        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
896        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
897        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
898        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
899        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
900        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
901    }
902
903    #[test]
904    fn rule_e1_pins_code_and_only_code() {
905        use crate::model::ActionKind as A;
906        // code_revision on a tool target with a pin: OK.
907        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
908        // Unpinned code revision: refused.
909        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
910        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
911        // code_revision off a tool target: refused.
912        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
913        // A tool target with a non-code action: refused.
914        assert!(validate_code_rules(A::Flag, "code", None).is_err());
915        // The gate cannot gate itself.
916        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
917        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
918        // Pins don't leak onto ordinary targets.
919        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
920        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
921    }
922
923    #[test]
924    fn rule_e1_pins_adapter_revisions_like_code() {
925        use crate::model::ActionKind as A;
926        // adapter_revision on a model target with a pin: OK.
927        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
928        // Unpinned adapter revision: refused.
929        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
930        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
931        // adapter_revision off a model target: refused.
932        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
933        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
934        // A model target with a non-adapter action: refused.
935        assert!(validate_code_rules(A::Flag, "model", None).is_err());
936        // The rollback inverse needs no pin — and must carry none.
937        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
938        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
939    }
940
941    #[test]
942    fn from_fields_rechecks_rule_e1() {
943        // A hand-authored grain body carrying an unpinned code revision must
944        // not decode into a usable recommendation.
945        let mut rec = Recommendation {
946            hash: String::new(),
947            analyzer: "loop.codegen/1".into(),
948            params_snapshot: Map::new(),
949            origin: Origin::Builtin,
950            target_ref: "tool:abc123".into(),
951            action_kind: ActionKind::CodeRevision,
952            dedup_key: "k".into(),
953            summary: Summary::new("command.finding", Map::new()),
954            severity: Severity::Low,
955            proposal: Proposal::Data { data: Map::new() },
956            destructive: false,
957            rollbackable: false,
958            evidence: vec![],
959            evidence_query: None,
960            metric: None,
961            confidence: 0.5,
962            importance: 0.5,
963            created_at_ms: 0,
964            guidance: None,
965            evalset_hash: None, // the smuggle attempt
966            status: RecStatus::Pending,
967        };
968        let spec = rec.to_grain_spec("ns").unwrap();
969        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
970        rec.evalset_hash = Some("es-hash".into());
971        let spec = rec.to_grain_spec("ns").unwrap();
972        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
973        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
974    }
975
976    #[test]
977    fn recommendation_round_trips_through_fields() {
978        let rec = Recommendation {
979            hash: "ignored".into(),
980            analyzer: "loop.staleness/1".into(),
981            params_snapshot: Map::new(),
982            origin: Origin::Builtin,
983            target_ref: "grain:sha256:abc".into(),
984            action_kind: ActionKind::Expire,
985            dedup_key: "k".into(),
986            summary: Summary::new("staleness.expired", Map::new()),
987            severity: Severity::Low,
988            proposal: Proposal::Cal {
989                cal: "FORGET sha256:abc".into(),
990            },
991            destructive: true,
992            rollbackable: false,
993            evidence: vec!["sha256:abc".into()],
994            evidence_query: None,
995            metric: None,
996            confidence: 0.9,
997            importance: 0.4,
998            created_at_ms: 1000,
999            guidance: None,
1000            evalset_hash: None,
1001            status: RecStatus::Pending,
1002        };
1003        let spec = rec.to_grain_spec("ns").unwrap();
1004        // hash and status are excluded from the immutable body.
1005        assert!(!spec.fields.contains_key("hash"));
1006        assert!(!spec.fields.contains_key("status"));
1007        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
1008        assert_eq!(back.hash, "realhash");
1009        assert_eq!(back.analyzer, "loop.staleness/1");
1010        assert!(back.destructive);
1011        assert!(matches!(back.proposal, Proposal::Cal { .. }));
1012    }
1013}