Skip to main content

areev_loop/
recommendation.rs

1//! The recommendation object (OMS 0x0C), the deterministic summary renderer,
2//! the dedup key, and the lifecycle state machine + audit records.
3//!
4//! Invariants enforced here:
5//! - Analyzers emit a `(template_id, args)` summary, never free prose.
6//! - `dedup_key`, `origin`, and the params snapshot are engine-stamped.
7//! - `dedup_key` excludes proposal content and the `/major` version, so a
8//!   growing cluster or an analyzer upgrade does not re-propose as novel —
9//!   for ANALYZER findings. An authored (`origin = llm`) executable proposal
10//!   keys on a fingerprint of its content as well, because there the content
11//!   IS the finding: two different lessons on one entity are two findings,
12//!   and the same lesson re-authored is one.
13//! - Lifecycle transitions are gated; `pending → applied` is policy-only.
14
15use crate::error::{Error, Result};
16use crate::model::{normalize_ident, ActionKind, Origin, Severity};
17use crate::substrate::GrainSpec;
18use serde::{Deserialize, Serialize};
19use serde_json::{Map, Value};
20
21/// A deterministic, template-rendered summary. The analyzer chooses a
22/// `template_id` and supplies `args`; the text is produced here, so an
23/// analyzer can never emit arbitrary prose into the queue.
24#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
25pub struct Summary {
26    pub template_id: String,
27    #[serde(default)]
28    pub args: Map<String, Value>,
29}
30
31impl Summary {
32    pub fn new(template_id: impl Into<String>, args: Map<String, Value>) -> Self {
33        Summary {
34            template_id: template_id.into(),
35            args,
36        }
37    }
38
39    /// Render the summary. Unknown template ids fall back to a stable, honest
40    /// string (never a panic) so a manifest/template mismatch is visible, not
41    /// fatal.
42    pub fn render(&self) -> String {
43        let t = builtin_template(&self.template_id).unwrap_or("{template_id}: {summary}");
44        interpolate(t, &self.args, &self.template_id)
45    }
46}
47
48/// The built-in template table. Deterministic; the only place summary prose
49/// lives. Keep placeholders in `{name}` form matching `args` keys.
50fn builtin_template(id: &str) -> Option<&'static str> {
51    Some(match id {
52        "duplicate.exact" => "Consolidate {count} exact-duplicate grains for \"{subject}\"",
53        "duplicate.near" => {
54            "Consolidate {count} near-duplicate observations (similarity ≥ {threshold})"
55        }
56        "contradiction.functional" => {
57            "\"{subject}\" holds {count} live values for functional relation \"{relation}\""
58        }
59        "tool_failure.cluster" => {
60            "Tool \"{tool}\" failed {count} times ({rate}% of the calls that could \
61             fail this way): {signature}"
62        }
63        "staleness.expired" => "Expire \"{subject}\": past its declared valid_to ({age_days}d ago)",
64        "fork.multi_head" => "Entity \"{entity}\" has {count} competing heads",
65        "skill.stall" => {
66            "Skill \"{skill}\" isn't improving: practiced {practice_count}× but proficiency is still {proficiency}"
67        }
68        "goal.stagnation" => "Goal \"{goal}\" is stalled: active {age_days}d with {progress} progress",
69        "cold.grain" => {
70            "Cold memory: \"{subject}\" ({age_days}d old) has never been recalled — retire candidate"
71        }
72        "coverage.gap" => {
73            "Recurring question with no matching memory: \"{query}\" asked {count}× ({empty_rate}% empty)"
74        }
75        "budget.pressure" => {
76            "Assembly budget overflowed on {overflow_rate}% of {samples} recalls — raise the budget or curate memory"
77        }
78        "retention.overdue" => {
79            "Retention: {count} grain(s) in \"{namespace}\" exceed the {max_age_days}d policy (oldest {oldest_days}d); {remaining} more await a later pass"
80        }
81        "outcome.regression" => {
82            "Applied recommendation regressed: {metric} moved {baseline} → {current}"
83        }
84        "outcome.premise_drift" => {
85            "{current} of the grains this recommendation cited have since been superseded by a different value or retracted — its premise moved; revert it"
86        }
87        "run.failures" => {
88            "Workflow {workflow} failed {failed}/{runs} recent runs ({rate}%): {last_error}"
89        }
90        "run.cost" => {
91            "Workflow {workflow} spent ${usd} across {runs} runs (avg ${avg_usd}/run)"
92        }
93        "adapter.candidate" => {
94            "Promote adapter for \"{model}\" (base {base_model}) — gated on its pinned evalset"
95        }
96        // origin=llm drafts carry free text (clearly marked llm) rather than a
97        // deterministic template — the model's proposed summary rides in {text}.
98        "llm.discover" => "{text}",
99        // An LLM-authored lesson shows the reviewer BOTH the finding and the
100        // exact line an apply would record — never one without the other.
101        "llm.lesson" => "{text} — record lesson: \"{lesson}\"",
102        // The rest of the LLM proposal vocabulary follows the same rule: the
103        // finding AND the exact change an apply would make, never one without
104        // the other. A reviewer approving blind is the failure this prevents.
105        "llm.fact" => "{text} — record fact: {relation} = \"{object}\"",
106        "llm.skill" => "{text} — record skill: \"{name}\" ({steps} steps)",
107        "llm.plan" => "{text} — record plan: \"{name}\" ({nodes} steps, {edges} edges)",
108        "llm.query_revision" => "{text} — redefine \"{name}\" as: {body}",
109        "llm.plan_revision" => "{text} — revise plan {plan}: {edits}",
110        "llm.code_revision" => {
111            "{text} — new source for tool \"{tool}\" ({bytes} bytes), pending its evalset gate"
112        }
113        // External command analyzers (trust class Command): free text from the
114        // subprocess, rendered as-is (the trust badge, not the prose, marks it).
115        "command.finding" => "{text}",
116        _ => return None,
117    })
118}
119
120fn interpolate(template: &str, args: &Map<String, Value>, template_id: &str) -> String {
121    let mut out = String::with_capacity(template.len());
122    let mut chars = template.chars().peekable();
123    while let Some(c) = chars.next() {
124        if c == '{' {
125            let mut key = String::new();
126            for k in chars.by_ref() {
127                if k == '}' {
128                    break;
129                }
130                key.push(k);
131            }
132            if key == "template_id" {
133                out.push_str(template_id);
134            } else {
135                match args.get(&key) {
136                    Some(Value::String(s)) => out.push_str(s),
137                    Some(v) => out.push_str(&v.to_string()),
138                    None => {
139                        // Missing arg — leave a visible marker rather than lie.
140                        out.push('{');
141                        out.push_str(&key);
142                        out.push('}');
143                    }
144                }
145            }
146        } else {
147            out.push(c);
148        }
149    }
150    out
151}
152
153/// When an applied recommendation is re-measured — one checkpoint of the
154/// Verify gate's schedule, in the unit the deployment actually counts in.
155///
156/// Three units, because deployments count differently and a fixed one makes
157/// the gate inert everywhere else. A service evaluated nightly counts
158/// **time** (`after_ms`). A benchmark or a CI harness counts **graded runs**
159/// (`after_runs`): "measure at the next evaluation after the apply", however
160/// long that takes on the clock. A chat agent counts **turns** (`after_grains`):
161/// "after fifty more grains". The default schedule stayed ms-only for a year
162/// and was measured, on PAST-Bench, to fire exactly zero verdicts across 78
163/// governed runs — a family finishes in seven minutes and the first checkpoint
164/// was a day away (`crates/areev-bench/PERSIST.md`).
165///
166/// On the wire a bare integer is milliseconds, so every policy, metric
167/// snapshot and state blob written before this type existed reads back
168/// unchanged, and an all-ms schedule still serializes as `[86400000, …]`.
169#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord)]
170pub enum Checkpoint {
171    /// Elapsed time since the apply.
172    AfterMs(i64),
173    /// Evalset runs journaled since the apply. Only an `evalset:` metric can
174    /// count these; on any other metric the checkpoint never comes due.
175    AfterRuns(u32),
176    /// Grains written since the apply — the loop's own notion of "activity".
177    AfterGrains(u32),
178}
179
180impl Checkpoint {
181    /// The label a person reads: `1d`, `12h`, `3 runs`, `50 grains`.
182    pub fn label(&self) -> String {
183        match self {
184            Checkpoint::AfterMs(ms) if ms % 86_400_000 == 0 => format!("{}d", ms / 86_400_000),
185            Checkpoint::AfterMs(ms) if ms % 3_600_000 == 0 => format!("{}h", ms / 3_600_000),
186            Checkpoint::AfterMs(ms) => format!("{ms}ms"),
187            Checkpoint::AfterRuns(n) => format!("{n} run{}", if *n == 1 { "" } else { "s" }),
188            Checkpoint::AfterGrains(n) => format!("{n} grain{}", if *n == 1 { "" } else { "s" }),
189        }
190    }
191
192    /// The ms value when this is a time checkpoint — what legacy consumers of
193    /// `OutcomeResult::horizon_ms` read.
194    pub fn as_ms(&self) -> Option<i64> {
195        match self {
196            Checkpoint::AfterMs(ms) => Some(*ms),
197            _ => None,
198        }
199    }
200}
201
202impl Serialize for Checkpoint {
203    fn serialize<S: serde::Serializer>(&self, ser: S) -> std::result::Result<S::Ok, S::Error> {
204        use serde::ser::SerializeMap;
205        match self {
206            // Bare integer: byte-identical to the pre-Checkpoint `Vec<i64>`.
207            Checkpoint::AfterMs(ms) => ser.serialize_i64(*ms),
208            Checkpoint::AfterRuns(n) => {
209                let mut m = ser.serialize_map(Some(1))?;
210                m.serialize_entry("after_runs", n)?;
211                m.end()
212            }
213            Checkpoint::AfterGrains(n) => {
214                let mut m = ser.serialize_map(Some(1))?;
215                m.serialize_entry("after_grains", n)?;
216                m.end()
217            }
218        }
219    }
220}
221
222impl<'de> Deserialize<'de> for Checkpoint {
223    fn deserialize<D: serde::Deserializer<'de>>(de: D) -> std::result::Result<Self, D::Error> {
224        use serde::de::Error as _;
225        let v = Value::deserialize(de)?;
226        match &v {
227            Value::Number(n) => n
228                .as_i64()
229                .filter(|ms| *ms >= 0)
230                .map(Checkpoint::AfterMs)
231                .ok_or_else(|| D::Error::custom("checkpoint: a bare number is non-negative milliseconds")),
232            Value::Object(m) if m.len() == 1 => {
233                let (k, val) = m.iter().next().unwrap();
234                let n = val.as_u64().ok_or_else(|| {
235                    D::Error::custom(format!("checkpoint: {k} takes a non-negative integer"))
236                })?;
237                match k.as_str() {
238                    "after_ms" => i64::try_from(n)
239                        .map(Checkpoint::AfterMs)
240                        .map_err(|_| D::Error::custom("checkpoint: after_ms out of range")),
241                    "after_runs" => u32::try_from(n)
242                        .map(Checkpoint::AfterRuns)
243                        .map_err(|_| D::Error::custom("checkpoint: after_runs out of range")),
244                    "after_grains" => u32::try_from(n)
245                        .map(Checkpoint::AfterGrains)
246                        .map_err(|_| D::Error::custom("checkpoint: after_grains out of range")),
247                    other => Err(D::Error::custom(format!(
248                        "checkpoint: unknown unit {other:?} (expected after_ms, after_runs or after_grains)"
249                    ))),
250                }
251            }
252            _ => Err(D::Error::custom(
253                "checkpoint: expected milliseconds or {\"after_ms\"|\"after_runs\"|\"after_grains\": n}",
254            )),
255        }
256    }
257}
258
259/// A reproducible metric snapshot; powers outcome review.
260#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
261pub struct MetricSnapshot {
262    /// Metric kind — the engine knows how to re-measure a fixed set
263    /// (e.g. `tool_error_recurrence`); unknown kinds are skipped, not faked.
264    pub metric: String,
265    pub baseline: f64,
266    pub unit: String,
267    pub n: u64,
268    pub window: String,
269    /// The subject the metric is about (e.g. the tool name), used by the
270    /// engine's typed re-measurement.
271    #[serde(default, skip_serializing_if = "Option::is_none")]
272    pub subject: Option<String>,
273    /// Namespace scope for the re-measurement, normalized (case-fold/trim).
274    /// `None` = all namespaces. Additive: older snapshots deserialize to
275    /// `None` and existing metric kinds ignore it.
276    #[serde(default, skip_serializing_if = "Option::is_none")]
277    pub namespace: Option<String>,
278    /// Relation scope for fact-shaped metrics (e.g. which functional relation
279    /// a contradiction was resolved under), normalized. Additive like
280    /// `namespace`.
281    #[serde(default, skip_serializing_if = "Option::is_none")]
282    pub relation: Option<String>,
283    /// CAL that recomputes the metric at verify time (reproducibility /
284    /// documentation; the engine re-measures with typed reads).
285    pub query: String,
286    /// How long after apply to re-measure, in epoch-ms delta (the first / only
287    /// checkpoint when `horizons_ms` is empty).
288    pub review_after_ms: i64,
289    /// A schedule of checkpoints (ms after apply) to re-measure at. An outcome
290    /// that `held` at an early checkpoint can `regress` at a later one, so a
291    /// verdict is never final until the last horizon. Empty → `[review_after_ms]`.
292    #[serde(default, skip_serializing_if = "Vec::is_empty")]
293    pub horizons_ms: Vec<i64>,
294    /// The schedule in the deployment's own unit ([`Checkpoint`]). When set it
295    /// is THE schedule and `horizons_ms`/`review_after_ms` are ignored; empty
296    /// (every snapshot written before it existed) falls back to them.
297    #[serde(default, skip_serializing_if = "Vec::is_empty")]
298    pub checkpoints: Vec<Checkpoint>,
299    /// Which direction is an improvement. The built-in metrics are all
300    /// recurrence counts, where lower is better — so this defaults to `false`
301    /// and every existing snapshot deserializes unchanged. An evalset accuracy
302    /// is the opposite, and getting it wrong does not merely misreport: the
303    /// Verify gate would read a rule that IMPROVED accuracy as a regression and
304    /// propose reverting it.
305    #[serde(default, skip_serializing_if = "std::ops::Not::not")]
306    pub higher_is_better: bool,
307}
308
309/// The one regression rule.
310///
311/// The verdict is computed in two places — once by the engine for the recorded
312/// `OutcomeResult`, once by `outcome_review` when it drafts the revert — and
313/// the two silently disagreeing would either revert a change that held or sit
314/// on one that regressed. So both call this.
315pub fn is_regression(baseline: f64, current: f64, higher_is_better: bool) -> bool {
316    /// Minimum worsening to call a regression (avoids noise at n=1).
317    const EPSILON: f64 = 1e-9;
318    if higher_is_better {
319        current < baseline - EPSILON
320    } else {
321        current > baseline + EPSILON
322    }
323}
324
325impl MetricSnapshot {
326    /// The measurement schedule, sorted and deduplicated: `checkpoints` when
327    /// set, else `horizons_ms`, else the single `review_after_ms` — each older
328    /// spelling read as time, so a snapshot from before [`Checkpoint`] existed
329    /// measures exactly as it did.
330    pub fn schedule(&self) -> Vec<Checkpoint> {
331        let mut h: Vec<Checkpoint> = if !self.checkpoints.is_empty() {
332            self.checkpoints.clone()
333        } else if !self.horizons_ms.is_empty() {
334            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
335        } else {
336            vec![Checkpoint::AfterMs(self.review_after_ms)]
337        };
338        h.sort_unstable();
339        h.dedup();
340        h
341    }
342
343    /// The time checkpoints of the schedule, in ms — the pre-[`Checkpoint`]
344    /// view, kept for callers that only understand time.
345    pub fn horizons(&self) -> Vec<i64> {
346        self.schedule().iter().filter_map(Checkpoint::as_ms).collect()
347    }
348}
349
350/// A measured outcome for an applied recommendation at one checkpoint — the
351/// Verify gate's output. `held` = the metric did not regress at this horizon;
352/// `regressed` = it got worse (a revert is proposed). A recommendation
353/// accumulates one of these per horizon, forming a time series.
354#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
355pub struct OutcomeResult {
356    pub rec_hash: String,
357    pub metric: String,
358    pub baseline: f64,
359    pub current: f64,
360    pub verdict: String,
361    /// Which checkpoint this measurement is for (ms after apply). Zero for a
362    /// checkpoint counted in runs or grains — see `checkpoint`.
363    #[serde(default)]
364    pub horizon_ms: i64,
365    /// The checkpoint in its own unit, when that unit is not time. A time
366    /// checkpoint leaves this unset and speaks through `horizon_ms`, so every
367    /// consumer of the older shape reads an unchanged record.
368    #[serde(default, skip_serializing_if = "Option::is_none")]
369    pub checkpoint: Option<Checkpoint>,
370    pub measured_at_ms: i64,
371}
372
373/// The proposed change. Exactly one variant per recommendation.
374#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
375#[serde(tag = "proposal", rename_all = "snake_case")]
376pub enum Proposal {
377    /// A batch of CAL Tier-1 evolve writes (ADD/SUPERSEDE), MAY contain FORGET.
378    Cal { cal: String },
379    /// A doc-target edit: `{format, base_digest, diff}` (base_digest enables a
380    /// staleness check at apply).
381    Edit {
382        format: String,
383        base_digest: String,
384        diff: String,
385    },
386    /// An opaque map for host targets (applied by the host, §12.3).
387    Data { data: Map<String, Value> },
388}
389
390/// What an analyzer emits. `dedup_key`, `origin`, and the params snapshot are
391/// **not** here — the engine stamps them. Non-exhaustive so the engine can add
392/// fields without breaking analyzers.
393#[derive(Debug, Clone, PartialEq)]
394#[non_exhaustive]
395pub struct RecDraft {
396    pub target_ref: String,
397    pub action_kind: ActionKind,
398    pub summary: Summary,
399    pub severity: Severity,
400    pub proposal: Proposal,
401    /// Bounded to ≤64 representative evidence hashes by the engine.
402    pub evidence: Vec<String>,
403    pub evidence_query: Option<String>,
404    pub metric: Option<MetricSnapshot>,
405    pub confidence: f64,
406    pub importance: f64,
407    /// Rule E1 pin for `code_revision` drafts (see [`validate_code_rules`]).
408    pub evalset_hash: Option<String>,
409}
410
411impl RecDraft {
412    pub fn new(
413        target_ref: impl Into<String>,
414        action_kind: ActionKind,
415        summary: Summary,
416        proposal: Proposal,
417    ) -> Self {
418        RecDraft {
419            target_ref: target_ref.into(),
420            action_kind,
421            summary,
422            severity: Severity::Low,
423            proposal,
424            evidence: Vec::new(),
425            evidence_query: None,
426            metric: None,
427            confidence: 0.8,
428            importance: 0.5,
429            evalset_hash: None,
430        }
431    }
432
433    pub fn severity(mut self, s: Severity) -> Self {
434        self.severity = s;
435        self
436    }
437    pub fn evidence(mut self, hashes: Vec<String>) -> Self {
438        self.evidence = hashes;
439        self
440    }
441    pub fn evidence_query(mut self, q: impl Into<String>) -> Self {
442        self.evidence_query = Some(q.into());
443        self
444    }
445    pub fn metric(mut self, m: MetricSnapshot) -> Self {
446        self.metric = Some(m);
447        self
448    }
449    pub fn confidence(mut self, c: f64) -> Self {
450        self.confidence = c;
451        self
452    }
453    pub fn importance(mut self, i: f64) -> Self {
454        self.importance = i;
455        self
456    }
457    pub fn evalset_hash(mut self, h: impl Into<String>) -> Self {
458        self.evalset_hash = Some(h.into());
459        self
460    }
461}
462
463/// Maximum representative evidence hashes carried inline (proposal §7.1).
464pub const MAX_EVIDENCE: usize = 64;
465
466/// Compute the dedup key: `family ⟂ target_ref ⟂ action_kind`, case-folded.
467/// Excludes proposal content and evidence by construction.
468pub fn dedup_key(family: &str, target_ref: &str, action: ActionKind) -> String {
469    // U+001F (unit separator) cannot appear in any of the inputs.
470    format!(
471        "{}\u{1f}{}\u{1f}{}",
472        normalize_ident(family),
473        normalize_ident(target_ref),
474        action.as_str()
475    )
476}
477
478/// The dedup key of an AUTHORED executable proposal: [`dedup_key`] plus a
479/// fingerprint of the proposal's content. An analyzer finding is "this
480/// target has this kind of problem", so content is rightly excluded; an
481/// authored lesson is "do this", and two different lessons on the same
482/// entity must both reach the queue while the same lesson re-authored must
483/// not. The fingerprint is over the normalized text — case-folded, non-
484/// alphanumerics dropped, whitespace collapsed — so a rewording that changes
485/// no word is the same lesson and one that changes a word is a new one (a
486/// semantic near-duplicate is the reviewer's call, not this key's).
487pub fn authored_dedup_key(
488    family: &str,
489    target_ref: &str,
490    action: ActionKind,
491    content: &str,
492) -> String {
493    format!(
494        "{}\u{1f}{}",
495        dedup_key(family, target_ref, action),
496        content_fingerprint(content)
497    )
498}
499
500/// The dedup key of a REVERT: [`dedup_key`] plus the hash of the applied
501/// recommendation it retracts. An analyzer finding is "this target has this
502/// kind of problem", so two findings on one target rightly collapse — but a
503/// revert is about one specific applied recommendation, and two lessons on
504/// the same entity that both regressed need two reverts. Until 2026-09-06
505/// they shared a key and the second was dropped as a duplicate of the first
506/// (`crates/areev-bench/CURVE.md`, seed 3: two regressed, one revert).
507pub fn revert_dedup_key(family: &str, target_ref: &str, revert_of: &str) -> String {
508    format!(
509        "{}\u{1f}{}",
510        dedup_key(family, target_ref, ActionKind::Revert),
511        normalize_ident(revert_of)
512    )
513}
514
515/// Sixteen hex chars of FNV-1a (64-bit) over the normalized content. A dedup
516/// key needs stability and spread, not cryptographic strength — a collision
517/// here would merge two findings in a review queue, never grant anything —
518/// so this stays dependency-free, as the crate is by policy.
519pub fn content_fingerprint(content: &str) -> String {
520    let mut normalized = String::with_capacity(content.len());
521    let mut pending_space = false;
522    for c in content.chars() {
523        if c.is_alphanumeric() {
524            if pending_space && !normalized.is_empty() {
525                normalized.push(' ');
526            }
527            pending_space = false;
528            normalized.extend(c.to_lowercase());
529        } else if c.is_whitespace() || !c.is_alphanumeric() {
530            pending_space = true;
531        }
532    }
533    let mut h: u64 = 0xcbf2_9ce4_8422_2325;
534    for b in normalized.as_bytes() {
535        h ^= u64::from(*b);
536        h = h.wrapping_mul(0x0000_0100_0000_01b3);
537    }
538    format!("{h:016x}")
539}
540
541/// Lifecycle status — a rebuildable index-layer cache (the recommendation's
542/// content hash is stable for its whole life).
543#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
544#[serde(rename_all = "snake_case")]
545pub enum RecStatus {
546    #[default]
547    Pending,
548    Approved,
549    Rejected,
550    Applied,
551    RolledBack,
552    Expired,
553}
554
555impl RecStatus {
556    pub fn as_str(&self) -> &'static str {
557        match self {
558            RecStatus::Pending => "pending",
559            RecStatus::Approved => "approved",
560            RecStatus::Rejected => "rejected",
561            RecStatus::Applied => "applied",
562            RecStatus::RolledBack => "rolled_back",
563            RecStatus::Expired => "expired",
564        }
565    }
566
567    /// Is a transition to `to` allowed from this state? `by_policy` marks the
568    /// auto-apply actor, the only one permitted the reasonless
569    /// `pending → applied` jump.
570    pub fn can_transition_to(&self, to: RecStatus, by_policy: bool) -> bool {
571        use RecStatus::*;
572        match (self, to) {
573            (Pending, Approved) | (Pending, Rejected) => true,
574            (Pending, Applied) => by_policy, // auto-apply only
575            (Approved, Applied) => true,
576            (Applied, RolledBack) => true,
577            // `expired` is computed from valid_to, applied to still-open recs.
578            (Pending, Expired) | (Approved, Expired) => true,
579            _ => false,
580        }
581    }
582}
583
584/// Who/what performed a transition.
585#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
586#[serde(rename_all = "lowercase")]
587pub enum ObserverType {
588    Human,
589    Agent,
590    Policy,
591    System,
592}
593
594/// One immutable audit Observation per transition, hash-chained per
595/// recommendation.
596#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
597pub struct AuditRecord {
598    pub rec_hash: String,
599    #[serde(skip_serializing_if = "Option::is_none")]
600    pub from: Option<RecStatus>,
601    pub to: RecStatus,
602    /// Host-asserted actor label, e.g. `user:alice`, `agent:worker-3`,
603    /// `policy:auto`.
604    pub actor: String,
605    pub observer_type: ObserverType,
606    /// Mandatory written reason (≤500 chars), the review statement's BECAUSE.
607    pub because: String,
608    #[serde(skip_serializing_if = "Option::is_none")]
609    pub previous_audit_hash: Option<String>,
610    /// §7.4: present exactly when this transition applied a code revision —
611    /// the evalset-run edge, recorded where the forensics look.
612    #[serde(default, skip_serializing_if = "Option::is_none")]
613    pub gating: Option<GatingEvidence>,
614    pub at_ms: i64,
615}
616
617/// Maximum length of a BECAUSE reason.
618pub const MAX_BECAUSE: usize = 500;
619
620impl AuditRecord {
621    /// Build the Observation grain that records this transition. `derived_from`
622    /// chains `[rec_hash, previous_audit_hash]` per the lifecycle spec.
623    pub fn to_grain_spec(&self, namespace: &str) -> GrainSpec {
624        let mut derived_from = vec![Value::from(self.rec_hash.clone())];
625        if let Some(prev) = &self.previous_audit_hash {
626            derived_from.push(Value::from(prev.clone()));
627        }
628        let mut spec = GrainSpec::new(crate::model::grain_type::OBSERVATION, namespace)
629            .with_field("observation_kind", "loop_audit")
630            .with_field("rec_hash", self.rec_hash.clone())
631            .with_field("to_status", self.to.as_str())
632            .with_field("actor", self.actor.clone())
633            .with_field(
634                "observer_type",
635                serde_json::to_value(self.observer_type).unwrap(),
636            )
637            .with_field("because", self.because.clone())
638            .with_field("at_ms", self.at_ms)
639            .with_field("derived_from", Value::Array(derived_from));
640        if let Some(from) = self.from {
641            spec.fields
642                .insert("from_status".into(), Value::from(from.as_str()));
643        }
644        if let Some(g) = &self.gating {
645            spec.fields
646                .insert("gating_evalset".into(), Value::from(g.evalset_hash.clone()));
647            spec.fields
648                .insert("gating_run_id".into(), Value::from(g.run_id.clone()));
649            spec.fields.insert("gating_passed".into(), Value::from(g.passed));
650            spec.fields.insert("gating_failed".into(), Value::from(g.failed));
651        }
652        spec
653    }
654}
655
656/// A stored recommendation. `hash` (the content address) and `status` (the
657/// index-layer cache) are set by the engine, not serialized into the grain
658/// body — the body is immutable content, the lifecycle lives in the state
659/// index and the audit chain.
660#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
661pub struct Recommendation {
662    #[serde(skip)]
663    pub hash: String,
664    pub analyzer: String,
665    pub params_snapshot: Map<String, Value>,
666    pub origin: Origin,
667    pub target_ref: String,
668    pub action_kind: ActionKind,
669    pub dedup_key: String,
670    pub summary: Summary,
671    pub severity: Severity,
672    #[serde(flatten)]
673    pub proposal: Proposal,
674    pub destructive: bool,
675    pub rollbackable: bool,
676    #[serde(default)]
677    pub evidence: Vec<String>,
678    #[serde(default, skip_serializing_if = "Option::is_none")]
679    pub evidence_query: Option<String>,
680    #[serde(default, skip_serializing_if = "Option::is_none")]
681    pub metric: Option<MetricSnapshot>,
682    pub confidence: f64,
683    pub importance: f64,
684    pub created_at_ms: i64,
685    /// Optional LLM guidance: an ENRICH note on a deterministic recommendation,
686    /// or an `origin = llm` draft's own rationale (§9). Whitelisted, capped
687    /// text — it never replaces the engine-templated `summary`.
688    #[serde(default, skip_serializing_if = "Option::is_none")]
689    pub guidance: Option<String>,
690    /// Rule E1's pin (§7.4): the evalset hash a `code_revision` was gated
691    /// against — REQUIRED for code targets, forbidden elsewhere (a
692    /// recommendation targets code XOR an evalset, never both, and only
693    /// code carries the pin). Shown at review; checked live at apply
694    /// (superseded evalset ⇒ the pin is stale ⇒ re-gate).
695    #[serde(default, skip_serializing_if = "Option::is_none")]
696    pub evalset_hash: Option<String>,
697    #[serde(skip)]
698    pub status: RecStatus,
699}
700
701/// The recorded evalset-run edge (§7.4): what an apply of a code revision
702/// must present, and what its audit Observation carries. "Trust me, it
703/// passed" is not an edge; (evalset, run, stats) is.
704#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
705pub struct GatingEvidence {
706    /// The evalset the gate ran — must equal the recommendation's pin.
707    pub evalset_hash: String,
708    /// The `areev run`/`areev eval` run id that executed the gate.
709    pub run_id: String,
710    pub passed: u64,
711    pub failed: u64,
712}
713
714/// Rule E1's structural checks (§7.4), applied when a draft is stamped and
715/// re-checked when a stored grain is loaded (a hand-authored grain must not
716/// bypass the rule):
717/// - `code_revision` ⇔ a `tool:` target, and it MUST pin an evalset hash.
718/// - `adapter_revision` ⇔ a `model:` target, same mandatory pin (the tuning
719///   seam inherits §7.4's gate wholesale).
720/// - An `evalset:` target is its own always-human-approved class: never a
721///   gated revision, and never itself pinned (the gate cannot gate itself).
722/// - No other target class carries a pin.
723pub fn validate_code_rules(
724    action: ActionKind,
725    target_class: &str,
726    evalset_hash: Option<&str>,
727) -> Result<()> {
728    let e1 = |why: &str| Err(Error::InvalidRecommendation(format!("Rule E1: {why}")));
729    match (action, target_class) {
730        (ActionKind::CodeRevision, "code") => {
731            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
732                return e1(
733                    "a code_revision must pin the evalset hash it was gated \
734                     against (evalset_hash)",
735                );
736            }
737        }
738        (ActionKind::CodeRevision, other) => {
739            return e1(&format!(
740                "code_revision requires a tool: target, got class '{other}'"
741            ));
742        }
743        (ActionKind::AdapterRevision, "model") => {
744            if evalset_hash.is_none_or(|h| h.trim().is_empty()) {
745                return e1(
746                    "an adapter_revision must pin the evalset hash it was \
747                     gated against (evalset_hash)",
748                );
749            }
750        }
751        (ActionKind::AdapterRevision, other) => {
752            return e1(&format!(
753                "adapter_revision requires a model: target, got class '{other}'"
754            ));
755        }
756        // A REVERT of an applied code revision targets the same tool: —
757        // it is the rollback inverse, not new code, so it needs no pin
758        // (blocking it here would leave the highest-stakes target class
759        // unable to propose its own measured rollback).
760        (ActionKind::Revert, "code") => {
761            if evalset_hash.is_some() {
762                return e1("a code revert carries no evalset pin");
763            }
764        }
765        (_, "code") => {
766            return e1("a tool: target requires action_kind code_revision");
767        }
768        // Same rollback-inverse escape hatch for the adapter class.
769        (ActionKind::Revert, "model") => {
770            if evalset_hash.is_some() {
771                return e1("an adapter revert carries no evalset pin");
772            }
773        }
774        (_, "model") => {
775            return e1("a model: target requires action_kind adapter_revision");
776        }
777        (_, "evalset") => {
778            if evalset_hash.is_some() {
779                return e1(
780                    "an evalset-target recommendation cannot pin an evalset — \
781                     the gate cannot gate itself",
782                );
783            }
784        }
785        _ => {
786            if evalset_hash.is_some() {
787                return e1(
788                    "evalset_hash is only valid on code_revision or \
789                     adapter_revision recommendations",
790                );
791            }
792        }
793    }
794    Ok(())
795}
796
797impl Recommendation {
798    /// Serialize the immutable body into grain fields (excludes hash/status).
799    pub fn to_grain_spec(&self, namespace: &str) -> Result<GrainSpec> {
800        let value = serde_json::to_value(self)
801            .map_err(|e| Error::Internal(format!("serialize recommendation: {e}")))?;
802        let obj = value
803            .as_object()
804            .ok_or_else(|| Error::Internal("recommendation did not serialize to object".into()))?
805            .clone();
806        Ok(GrainSpec {
807            grain_type: crate::model::grain_type::RECOMMENDATION.to_string(),
808            namespace: namespace.to_string(),
809            fields: obj,
810        })
811    }
812
813    /// Reconstruct from a stored grain body. `hash` comes from the record's
814    /// address; `status` is supplied from the state index by the caller.
815    /// Rule E1 is RE-CHECKED here — a hand-authored recommendation grain
816    /// must not smuggle an unpinned code revision past the stamped path.
817    pub fn from_fields(hash: &str, fields: &Map<String, Value>) -> Result<Self> {
818        let mut rec: Recommendation = serde_json::from_value(Value::Object(fields.clone()))
819            .map_err(|e| Error::InvalidRecommendation(format!("decode {hash}: {e}")))?;
820        rec.hash = hash.to_string();
821        let class = crate::model::TargetRef::parse(&rec.target_ref)
822            .map(|t| t.target_class())
823            .unwrap_or("host");
824        validate_code_rules(rec.action_kind, class, rec.evalset_hash.as_deref())?;
825        Ok(rec)
826    }
827}
828
829#[cfg(test)]
830mod tests {
831    use super::*;
832    use serde_json::json;
833
834    #[test]
835    fn summary_renders_deterministically() {
836        let mut args = Map::new();
837        args.insert("count".into(), json!(3));
838        args.insert("subject".into(), json!("acme"));
839        let s = Summary::new("duplicate.exact", args);
840        assert_eq!(
841            s.render(),
842            "Consolidate 3 exact-duplicate grains for \"acme\""
843        );
844    }
845
846    #[test]
847    fn dedup_key_ignores_content_and_case() {
848        let a = dedup_key(
849            "loop.duplicate_sweep",
850            "entity:NS/John",
851            ActionKind::Consolidate,
852        );
853        let b = dedup_key(
854            "loop.duplicate_sweep",
855            "entity:ns/john",
856            ActionKind::Consolidate,
857        );
858        assert_eq!(a, b, "case-folded to one identity");
859    }
860
861    #[test]
862    fn authored_dedup_key_distinguishes_content_but_not_wording_noise() {
863        let a = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the vendor name.");
864        let same = authored_dedup_key("llm", "entity:NS/X", ActionKind::Record, "  record THE vendor  name ");
865        let other = authored_dedup_key("llm", "entity:ns/x", ActionKind::Record, "Record the amount.");
866        assert_eq!(a, same, "case, punctuation and spacing are not a new lesson");
867        assert_ne!(a, other, "a different lesson on the same entity is a different finding");
868        assert!(a.starts_with(&dedup_key("llm", "entity:ns/x", ActionKind::Record)));
869        assert_eq!(content_fingerprint("A b"), content_fingerprint("a-b"));
870        assert_ne!(content_fingerprint("ab"), content_fingerprint("a b"));
871    }
872
873    #[test]
874    fn a_revert_is_keyed_by_what_it_reverts() {
875        let a = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "aaaa");
876        let b = revert_dedup_key("loop.outcome_review", "entity:ns/capture", "bbbb");
877        let a2 = revert_dedup_key("loop.outcome_review", "entity:NS/Capture", "AAAA");
878        assert_ne!(a, b, "two reverts on one target are two findings");
879        assert_eq!(a, a2, "the same revert, case-folded, is one");
880        assert!(a.starts_with(&dedup_key("loop.outcome_review", "entity:ns/capture", ActionKind::Revert)));
881    }
882
883    #[test]
884    fn dedup_key_distinguishes_action() {
885        let a = dedup_key("f", "entity:ns/x", ActionKind::Consolidate);
886        let b = dedup_key("f", "entity:ns/x", ActionKind::FlagContradiction);
887        assert_ne!(a, b);
888    }
889
890    #[test]
891    fn lifecycle_gates_pending_to_applied() {
892        assert!(!RecStatus::Pending.can_transition_to(RecStatus::Applied, false));
893        assert!(RecStatus::Pending.can_transition_to(RecStatus::Applied, true)); // policy
894        assert!(RecStatus::Pending.can_transition_to(RecStatus::Approved, false));
895        assert!(RecStatus::Approved.can_transition_to(RecStatus::Applied, false));
896        assert!(RecStatus::Applied.can_transition_to(RecStatus::RolledBack, false));
897        assert!(!RecStatus::Rejected.can_transition_to(RecStatus::Applied, true));
898    }
899
900    #[test]
901    fn rule_e1_pins_code_and_only_code() {
902        use crate::model::ActionKind as A;
903        // code_revision on a tool target with a pin: OK.
904        assert!(validate_code_rules(A::CodeRevision, "code", Some("abc")).is_ok());
905        // Unpinned code revision: refused.
906        assert!(validate_code_rules(A::CodeRevision, "code", None).is_err());
907        assert!(validate_code_rules(A::CodeRevision, "code", Some("  ")).is_err());
908        // code_revision off a tool target: refused.
909        assert!(validate_code_rules(A::CodeRevision, "memory", Some("abc")).is_err());
910        // A tool target with a non-code action: refused.
911        assert!(validate_code_rules(A::Flag, "code", None).is_err());
912        // The gate cannot gate itself.
913        assert!(validate_code_rules(A::Flag, "evalset", Some("abc")).is_err());
914        assert!(validate_code_rules(A::Flag, "evalset", None).is_ok());
915        // Pins don't leak onto ordinary targets.
916        assert!(validate_code_rules(A::Consolidate, "memory", Some("abc")).is_err());
917        assert!(validate_code_rules(A::Consolidate, "memory", None).is_ok());
918    }
919
920    #[test]
921    fn rule_e1_pins_adapter_revisions_like_code() {
922        use crate::model::ActionKind as A;
923        // adapter_revision on a model target with a pin: OK.
924        assert!(validate_code_rules(A::AdapterRevision, "model", Some("abc")).is_ok());
925        // Unpinned adapter revision: refused.
926        assert!(validate_code_rules(A::AdapterRevision, "model", None).is_err());
927        assert!(validate_code_rules(A::AdapterRevision, "model", Some("  ")).is_err());
928        // adapter_revision off a model target: refused.
929        assert!(validate_code_rules(A::AdapterRevision, "memory", Some("abc")).is_err());
930        assert!(validate_code_rules(A::AdapterRevision, "code", Some("abc")).is_err());
931        // A model target with a non-adapter action: refused.
932        assert!(validate_code_rules(A::Flag, "model", None).is_err());
933        // The rollback inverse needs no pin — and must carry none.
934        assert!(validate_code_rules(A::Revert, "model", None).is_ok());
935        assert!(validate_code_rules(A::Revert, "model", Some("abc")).is_err());
936    }
937
938    #[test]
939    fn from_fields_rechecks_rule_e1() {
940        // A hand-authored grain body carrying an unpinned code revision must
941        // not decode into a usable recommendation.
942        let mut rec = Recommendation {
943            hash: String::new(),
944            analyzer: "loop.codegen/1".into(),
945            params_snapshot: Map::new(),
946            origin: Origin::Builtin,
947            target_ref: "tool:abc123".into(),
948            action_kind: ActionKind::CodeRevision,
949            dedup_key: "k".into(),
950            summary: Summary::new("command.finding", Map::new()),
951            severity: Severity::Low,
952            proposal: Proposal::Data { data: Map::new() },
953            destructive: false,
954            rollbackable: false,
955            evidence: vec![],
956            evidence_query: None,
957            metric: None,
958            confidence: 0.5,
959            importance: 0.5,
960            created_at_ms: 0,
961            guidance: None,
962            evalset_hash: None, // the smuggle attempt
963            status: RecStatus::Pending,
964        };
965        let spec = rec.to_grain_spec("ns").unwrap();
966        assert!(Recommendation::from_fields("h", &spec.fields).is_err());
967        rec.evalset_hash = Some("es-hash".into());
968        let spec = rec.to_grain_spec("ns").unwrap();
969        let back = Recommendation::from_fields("h", &spec.fields).unwrap();
970        assert_eq!(back.evalset_hash.as_deref(), Some("es-hash"));
971    }
972
973    #[test]
974    fn recommendation_round_trips_through_fields() {
975        let rec = Recommendation {
976            hash: "ignored".into(),
977            analyzer: "loop.staleness/1".into(),
978            params_snapshot: Map::new(),
979            origin: Origin::Builtin,
980            target_ref: "grain:sha256:abc".into(),
981            action_kind: ActionKind::Expire,
982            dedup_key: "k".into(),
983            summary: Summary::new("staleness.expired", Map::new()),
984            severity: Severity::Low,
985            proposal: Proposal::Cal {
986                cal: "FORGET sha256:abc".into(),
987            },
988            destructive: true,
989            rollbackable: false,
990            evidence: vec!["sha256:abc".into()],
991            evidence_query: None,
992            metric: None,
993            confidence: 0.9,
994            importance: 0.4,
995            created_at_ms: 1000,
996            guidance: None,
997            evalset_hash: None,
998            status: RecStatus::Pending,
999        };
1000        let spec = rec.to_grain_spec("ns").unwrap();
1001        // hash and status are excluded from the immutable body.
1002        assert!(!spec.fields.contains_key("hash"));
1003        assert!(!spec.fields.contains_key("status"));
1004        let back = Recommendation::from_fields("realhash", &spec.fields).unwrap();
1005        assert_eq!(back.hash, "realhash");
1006        assert_eq!(back.analyzer, "loop.staleness/1");
1007        assert!(back.destructive);
1008        assert!(matches!(back.proposal, Proposal::Cal { .. }));
1009    }
1010}