Skip to main content

areev_loop/
policy.rs

1//! Host policy — the optional `loop-policy.json` (proposal §6.2). It is the
2//! **only** place auto-apply is granted, and it is host config (per-process,
3//! never persisted in a memory file). All fields default-closed; the whole
4//! struct rejects unknown keys, so a policy that tries to register an
5//! executable (`--analyzer-cmd`) or touch a trust-floor field fails to load —
6//! a stolen or committed policy file must be inert.
7//!
8//! Precedence (enforced by the engine): engine ceilings > host CLI flags >
9//! this policy file > memory-file config. "The file selects and restricts;
10//! only the host grants."
11
12use crate::error::{Error, Result};
13use crate::model::Severity;
14use crate::recommendation::Checkpoint;
15use serde::{Deserialize, Serialize};
16use std::collections::BTreeMap;
17
18/// Telemetry sidecar mode (host-only).
19#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
20#[serde(rename_all = "lowercase")]
21pub enum TelemetryMode {
22    Off,
23    #[default]
24    Aggregate,
25    Full,
26}
27
28/// What DISCOVER optimizes for (`docs/loop-reflection.md` §5.1). Host config
29/// like everything else here: it changes the scoring rule the proposer is
30/// given, never the gates — every draft still has to survive GROUND, VERIFY,
31/// the confidence floor and a human review with a BECAUSE.
32#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
33#[serde(rename_all = "snake_case")]
34pub enum DiscoverObjective {
35    /// The review-queue objective: "nothing to report" is a zero-penalty
36    /// answer and a wrong finding costs twice a right one. Right for a queue
37    /// a person triages — it keeps the queue clean at the price of drafts
38    /// the model was not sure enough about.
39    #[default]
40    ReviewQueue,
41    /// The learner objective: the agent has to improve from THIS pass, so
42    /// abstaining in the face of a recurring failure, repeated rejections or
43    /// a person's instruction is penalized like a wrong lesson. Measured
44    /// need: under the review-queue rule a cheap model authored a lesson on
45    /// fewer than half of its passes over evidence that plainly held one.
46    Learner,
47}
48
49/// One auto-apply grant: an analyzer family may auto-apply to these target
50/// classes up to (and including) `max_severity`.
51#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
52#[serde(deny_unknown_fields)]
53pub struct AutoApplyGrant {
54    /// Analyzer family (e.g. `loop.duplicate_sweep`) or full id; matched by
55    /// family so a version bump keeps the grant.
56    pub analyzer: String,
57    /// Eligible target classes: `memory` and/or `query` only (prompt/host are
58    /// never auto-appliable and are rejected at eval time regardless).
59    pub targets: Vec<String>,
60    /// Highest severity this grant covers.
61    pub max_severity: Severity,
62}
63
64/// How an Observation is attributed in the evidence bundle handed to the LLM
65/// (`docs/loop.md`). `Named` renders `<observer> (a person) said of
66/// <subject>: <text>`; `Anonymous` renders the bare text, which is what the
67/// engine did before 2026-09-04.
68///
69/// It is host policy for two independent reasons. An operator may not want
70/// observer identities rendered into a model prompt at all — an observer id
71/// can be a person's name or account — and that is a privacy decision only
72/// the host can make. And it is the one variable in the receipts ablation
73/// (`crates/areev-bench/RECEIPTS.md`), where naming the speaker is what
74/// stopped one model reading a correction as a request.
75#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
76#[serde(rename_all = "snake_case")]
77pub enum EvidenceAttribution {
78    /// Name the observer on an Observation that records one.
79    #[default]
80    Named,
81    /// Render the bare text, attributing nothing.
82    Anonymous,
83}
84
85/// Which run journaled before the apply an evalset verdict compares
86/// against (`docs/loop.md`, "Evalset-backed outcomes").
87///
88/// `newest_before_apply` is the marginal question — did THIS apply make the
89/// agent worse than it was the moment before — and it is the default because
90/// it never proposes reverting a rule for a drop an earlier rule caused.
91/// `high_water` asks the question an operator actually has — is the agent
92/// worse than the best it has been — and it is a policy choice, not the
93/// default, because on a noisy evalset (or a deployment that does not
94/// journal a run between applies) it attributes the whole fall from the
95/// peak to whichever rule was applied last. That confounding is the trade;
96/// measured need: on the ad-buy corpus (`crates/areev-bench/ADBUY.md`, seed
97/// 3) an agent that reached 238 of 280 and then fell to 128 measured `held`
98/// against the day-one run of 35, because day one was the only run before
99/// the apply.
100#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
101#[serde(rename_all = "snake_case")]
102pub enum BaselineKind {
103    /// The newest run journaled strictly before the apply.
104    #[default]
105    NewestBeforeApply,
106    /// The best run journaled strictly before the apply — max for a
107    /// higher-is-better field, min otherwise.
108    HighWater,
109}
110
111impl BaselineKind {
112    /// The spelling the receipt records in `OutcomeResult::baseline_kind`.
113    pub fn as_str(self) -> &'static str {
114        match self {
115            BaselineKind::NewestBeforeApply => "newest_before_apply",
116            BaselineKind::HighWater => "high_water",
117        }
118    }
119}
120
121/// The Verify gate's minimum effect size — a **floor, not a significance
122/// test**. A worsening of at most this much is `held`; no p-value, no
123/// interval, and the docs say so (`docs/loop-proposal.md` §18 forbids
124/// invented precision). Exactly one form:
125///
126/// - `{"count": n}` — absolute, in the field's own unit (`passed: 359 →
127///   355` under `count: 5` holds);
128/// - `{"points": p}` — percentage points. For the promoted count fields
129///   (`passed`, `failed`, `total`) it is scaled by the baseline run's
130///   `total`; for `error_rate`, and for any host-written field, it is read
131///   as `p / 100` — a host field under `points` is assumed to be a ratio in
132///   `0..1`, so a count-valued host field wants `count`.
133///
134/// Measured need (`docs/loop.md`): a 359 → 355 dip on 387 trials — within
135/// what one adapter read twice can differ by — proposed a revert.
136#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)]
137#[serde(deny_unknown_fields)]
138pub struct MinEffect {
139    #[serde(default, skip_serializing_if = "Option::is_none")]
140    pub count: Option<f64>,
141    #[serde(default, skip_serializing_if = "Option::is_none")]
142    pub points: Option<f64>,
143}
144
145impl MinEffect {
146    fn validate(&self) -> Result<()> {
147        let bad = |what: &str| Err(Error::InvalidProposal(format!("policy: outcome_evalset.min_effect: {what}")));
148        match (self.count, self.points) {
149            (None, None) => bad("give {\"count\": n} or {\"points\": p}"),
150            (Some(_), Some(_)) => bad("give count or points, not both"),
151            (Some(v), None) | (None, Some(v)) if !(v.is_finite() && v >= 0.0) => {
152                bad("must be a finite number ≥ 0")
153            }
154            _ => Ok(()),
155        }
156    }
157
158    /// The tolerance in the field's unit, given the run total the `points`
159    /// form scales a count field by.
160    pub fn resolve(&self, field: &str, total: u64) -> f64 {
161        match (self.count, self.points) {
162            (Some(c), _) => c,
163            (None, Some(p)) => match field {
164                "passed" | "failed" | "total" => p / 100.0 * total as f64,
165                _ => p / 100.0,
166            },
167            (None, None) => 0.0,
168        }
169    }
170}
171
172/// A cost bound beside the quality metric. The verdict on the quality
173/// field is unchanged; when quality held but `field` on the run after the
174/// apply exceeds `max_increase_ratio` × its value on the baseline run, the
175/// checkpoint records `held_costlier` and `outcome_review` emits an
176/// advisory Flag citing both runs — never a revert draft, because a
177/// cost/quality trade is a human decision. `regressed` dominates: one
178/// verdict per checkpoint. `field` is one of the promoted cost fields
179/// (`effects`, `tokens`, `usd`, `wall_ms`, `cost_per_pass`) or any integer
180/// key the harness writes; a run on which it is not measurable records no
181/// cost figures, and the quality verdict still records.
182#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
183#[serde(deny_unknown_fields)]
184pub struct CostBound {
185    pub field: String,
186    /// Current may be at most this many times the baseline (`1.0` = no
187    /// increase at all); finite and > 0.
188    pub max_increase_ratio: f64,
189}
190
191impl CostBound {
192    fn validate(&self) -> Result<()> {
193        let bad = |what: &str| Err(Error::InvalidProposal(format!("policy: outcome_evalset.cost: {what}")));
194        if self.field.trim().is_empty() {
195            return bad("field must name a cost field (effects, tokens, usd, wall_ms, cost_per_pass, or a harness key)");
196        }
197        if !(self.max_increase_ratio.is_finite() && self.max_increase_ratio > 0.0) {
198            return bad("max_increase_ratio must be a finite number > 0");
199        }
200        Ok(())
201    }
202}
203
204/// The evalset every LLM-authored, applicable proposal is measured against
205/// after apply (`docs/loop.md`, "Evalset-backed outcomes"). An authored
206/// lesson carries no built-in recurrence metric — nothing errors when a
207/// lesson is merely useless — so without this the Verify gate has nothing
208/// to re-measure for exactly the proposals a human was least able to judge.
209/// The host names the evalset and the field; the engine takes the baseline
210/// from the newest run journaled BEFORE the proposal and reads the current
211/// value from runs journaled AFTER the apply. No baseline run → no metric,
212/// never a fabricated one.
213#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
214#[serde(deny_unknown_fields)]
215pub struct OutcomeEvalset {
216    /// The evalset hash (the subject is `evalset:<hash>` in `agent:harness`).
217    pub hash: String,
218    /// The summary field to read: `passed`, `failed`, `total`, `error_rate`,
219    /// or any numeric field the host's harness writes into the summary.
220    pub field: String,
221    /// Which direction is an improvement — `passed` and an accuracy are
222    /// higher-is-better, `failed` and `error_rate` are not. Stated by the
223    /// host because getting it wrong would revert an improvement.
224    pub higher_is_better: bool,
225    /// Checkpoints after apply, in ms. Default 1d / 7d / 30d. The older
226    /// spelling; `checkpoints` wins when both are given.
227    #[serde(default = "default_horizons")]
228    pub horizons_ms: Vec<i64>,
229    /// The schedule in the deployment's own unit — `{"after_ms": n}`,
230    /// `{"after_runs": n}` or `{"after_grains": n}` (a bare integer is ms).
231    /// A benchmark or CI harness wants `[{"after_runs": 1}]`: measure at the
232    /// next graded run after the apply, however soon that is. Empty (the
233    /// default) means `horizons_ms`.
234    #[serde(default, skip_serializing_if = "Vec::is_empty")]
235    pub checkpoints: Vec<Checkpoint>,
236    /// Which run before the apply the verdict compares against (default
237    /// `newest_before_apply`; see [`BaselineKind`] for why `high_water` is
238    /// opt-in).
239    #[serde(default)]
240    pub baseline: BaselineKind,
241    /// The minimum effect size a verdict needs to call a regression (default
242    /// none: any drop past floating-point slack is a regression).
243    #[serde(default, skip_serializing_if = "Option::is_none")]
244    pub min_effect: Option<MinEffect>,
245    /// A cost bound read beside the quality field (default none).
246    #[serde(default, skip_serializing_if = "Option::is_none")]
247    pub cost: Option<CostBound>,
248}
249
250fn default_horizons() -> Vec<i64> {
251    vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]
252}
253
254impl OutcomeEvalset {
255    /// The effective schedule: `checkpoints` when set, else `horizons_ms` as
256    /// time checkpoints.
257    pub fn schedule(&self) -> Vec<Checkpoint> {
258        let mut h: Vec<Checkpoint> = if self.checkpoints.is_empty() {
259            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
260        } else {
261            self.checkpoints.clone()
262        };
263        h.sort_unstable();
264        h.dedup();
265        h
266    }
267}
268
269/// When a loop pass is due — the loop's cadence, as host policy.
270///
271/// The engine has no clock and no scheduler of its own (ARCHITECTURE.md:
272/// cadence is data, evaluation is a command); a host calls `run` and the
273/// engine decides whether there is anything to do. Until now that decision
274/// was only expressible as per-call flags (`--min-new`, `--if-stale`), so
275/// every surface that can trigger a run — CLI, MCP, the console — had to be
276/// told separately, and none of them could count the units a chat deployment
277/// actually thinks in. This block is the same gate as the flags, set once in
278/// the policy file, with two more units.
279///
280/// Each field is a threshold; the pass is due when **any** set one is met
281/// (whichever comes first). Nothing set — the default — means a pass is due
282/// whenever it is called, which is what every deployment had before. Explicit
283/// per-call flags override the block (host CLI flags > policy file), and a
284/// full sweep (`areev loop reflect`) is a command, not a tick: it always runs.
285#[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)]
286#[serde(deny_unknown_fields)]
287pub struct Cadence {
288    /// Due when this long has passed since the last run (or it never ran).
289    #[serde(default, skip_serializing_if = "Option::is_none")]
290    pub every_ms: Option<i64>,
291    /// Due when this many grains of any kind landed since the last run.
292    #[serde(default, skip_serializing_if = "Option::is_none")]
293    pub every_grains: Option<u64>,
294    /// Due when this many Event grains — turns, in a chat deployment — landed
295    /// since the last run. Hermes's post-turn review fires every ten.
296    #[serde(default, skip_serializing_if = "Option::is_none")]
297    pub every_events: Option<u64>,
298    /// Due when Events from this many distinct sessions landed since the last
299    /// run: "reflect once per conversation" is `1`.
300    #[serde(default, skip_serializing_if = "Option::is_none")]
301    pub every_sessions: Option<u64>,
302}
303
304impl Cadence {
305    /// Whether any threshold is configured at all.
306    pub fn is_set(&self) -> bool {
307        self.every_ms.is_some()
308            || self.every_grains.is_some()
309            || self.every_events.is_some()
310            || self.every_sessions.is_some()
311    }
312}
313
314/// Whether, and how, DISCOVER may author a **Skill** — a reusable procedure
315/// with an applicability condition and ordered steps, derived from a
316/// trajectory that succeeded.
317///
318/// This exists because of a measured gap. On PAST-Bench the agent performed
319/// the procedure correctly in the learn episode on every seed and then, asked
320/// at session end whether there was anything to save, answered "nothing to
321/// save" — so the store was empty at evaluation and the memory scored below
322/// having none (`crates/areev-bench/PERSIST.md`). Every Skill in the memory
323/// depended on the model volunteering one mid-task. Hermes does not depend on
324/// that: a separate review pass writes its skills. This is Areev's equivalent,
325/// and it runs through the same gates as every other draft — GROUND, VERIFY,
326/// the confidence floor, a review with a BECAUSE — and is never auto-applied.
327#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
328#[serde(deny_unknown_fields)]
329pub struct SkillAuthoring {
330    /// Offer the `skill` proposal kind to the proposer at all (default: yes,
331    /// under LLM enrichment; no LLM, no skills).
332    #[serde(default = "default_true")]
333    pub enabled: bool,
334    /// Fewer ordered steps than this is a lesson, not a procedure (default 2).
335    #[serde(default = "default_min_steps")]
336    pub min_steps: u32,
337}
338
339pub(crate) fn default_true() -> bool {
340    true
341}
342fn default_min_steps() -> u32 {
343    2
344}
345fn default_min_evidence() -> u32 {
346    1
347}
348
349impl Default for SkillAuthoring {
350    fn default() -> Self {
351        SkillAuthoring { enabled: true, min_steps: 2 }
352    }
353}
354
355/// Whether, and how, DISCOVER may author a **plan** — a Workflow grain: named
356/// steps, edges with conditions in the runtime's frozen grammar, validated
357/// before a reviewer sees it — beside the Skill that carries the prose.
358///
359/// A skill is what a model reads; a plan is what the runtime can check and
360/// run. PAST-Bench's own labels call every procedural family "ordered steps,
361/// tools, conditions… a patched v2 supersedes v1" — which is a Workflow, and
362/// its patch is the `plan_revision` this engine already has. Storing the
363/// procedure as a plan buys structural validation (unique, reachable nodes;
364/// conditions that parse; bounded cycles) at author time, and puts the
365/// procedure where `areev run`, the run journal and `run_outcome` can reach
366/// it. Governed like every draft; never auto-applied.
367#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
368#[serde(deny_unknown_fields)]
369pub struct PlanAuthoring {
370    /// Offer the `plan` proposal kind (default: yes, under LLM enrichment).
371    #[serde(default = "default_true")]
372    pub enabled: bool,
373    /// Fewer steps than this is a lesson, not a procedure (default 2).
374    #[serde(default = "default_min_steps")]
375    pub min_nodes: u32,
376}
377
378impl Default for PlanAuthoring {
379    fn default() -> Self {
380        PlanAuthoring { enabled: true, min_nodes: 2 }
381    }
382}
383
384/// What DISCOVER does with a lesson that says, in other words, what a live
385/// lesson on the same entity already says. `authored_dedup_key` collapses
386/// the same text; a rewording is the reviewer's call — so `flag` (default)
387/// lets it reach the queue carrying `near_duplicate_of`, and `suppress`
388/// drops it before the queue and counts it in the funnel as
389/// `dropped_near_duplicate`. Measured need (`crates/areev-bench/ADBUY.md`,
390/// seed 3): ten approved rules stated four facts, each approvable alone.
391#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
392#[serde(rename_all = "snake_case")]
393pub enum NearDuplicateMode {
394    #[default]
395    Flag,
396    Suppress,
397}
398
399/// The pre-apply gate on a `plan_revision`: when the substrate can rehearse
400/// the candidate against the journaled runs of the live plan
401/// (`SubstrateRead::plan_replay` — `areev run shadow --plan-file`), refuse
402/// to stamp the revision *applicable* when the rehearsal says it is worse
403/// than the incumbent on the same runs, or when too many runs fall outside
404/// the journal's support. Dream-RSI's monotone selection (arXiv 2609.14858
405/// §3) as a gate rather than an auto-deploy: applying stays human, with a
406/// BECAUSE. Default none — a revision is rehearsed when it can be and the
407/// report rides on the card, but nothing is refused.
408#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
409#[serde(deny_unknown_fields)]
410pub struct PlanReplayPolicy {
411    /// Fewer rehearsed runs than this and the gate abstains: two runs are an
412    /// anecdote (default 3).
413    #[serde(default = "default_min_runs")]
414    pub min_runs: u32,
415    /// Refuse a candidate that completes fewer of the same runs than the
416    /// incumbent did (default true).
417    #[serde(default = "default_true")]
418    pub require_no_worse: bool,
419    /// Refuse when more than this fraction of the runs could not be scored
420    /// because the candidate asked for effects the journal never recorded
421    /// (default 0.5).
422    #[serde(default = "default_max_out_of_support")]
423    pub max_out_of_support: f64,
424}
425
426fn default_min_runs() -> u32 {
427    3
428}
429fn default_max_out_of_support() -> f64 {
430    0.5
431}
432
433impl PlanReplayPolicy {
434    fn validate(&self) -> Result<()> {
435        if !(self.max_out_of_support.is_finite() && (0.0..=1.0).contains(&self.max_out_of_support)) {
436            return Err(Error::InvalidProposal(
437                "policy: plan_replay.max_out_of_support must be a fraction in 0..=1".into(),
438            ));
439        }
440        Ok(())
441    }
442
443    /// The reason a rehearsal refuses the revision, or `None` to admit it.
444    /// `report` is the runtime's `ShadowPlanReport` as JSON.
445    pub fn refusal(&self, report: &serde_json::Value) -> Option<String> {
446        let runs = report["totals"]["runs"].as_u64().unwrap_or(0);
447        if runs < u64::from(self.min_runs) {
448            return None;
449        }
450        let oos = report["out_of_support_fraction"].as_f64().unwrap_or(0.0);
451        if oos > self.max_out_of_support {
452            let ids: Vec<&str> = report["runs"]
453                .as_array()
454                .map(|a| {
455                    a.iter()
456                        .filter(|r| r["verdict"] == "out_of_support")
457                        .filter_map(|r| r["run_id"].as_str())
458                        .collect()
459                })
460                .unwrap_or_default();
461            return Some(format!(
462                "{:.0}% of {runs} rehearsed runs fall outside the journal's support (limit {:.0}%): {}",
463                oos * 100.0,
464                self.max_out_of_support * 100.0,
465                ids.join(", ")
466            ));
467        }
468        if self.require_no_worse && report["no_worse"].as_bool() != Some(true) {
469            let worse: Vec<String> = report["runs"]
470                .as_array()
471                .map(|a| {
472                    a.iter()
473                        .filter(|r| r["verdict"] == "worse")
474                        .filter_map(|r| {
475                            Some(format!(
476                                "{} ({} → {})",
477                                r["run_id"].as_str()?,
478                                r["incumbent_outcome"].as_str().unwrap_or("?"),
479                                r["candidate_outcome"].as_str().unwrap_or("?")
480                            ))
481                        })
482                        .collect()
483                })
484                .unwrap_or_default();
485            let scored = runs - report["totals"]["out_of_support"].as_u64().unwrap_or(0);
486            return Some(if worse.is_empty() {
487                format!("no rehearsed run could be scored ({scored} of {runs})")
488            } else {
489                format!("worse than the incumbent on {} of {scored} rehearsed runs: {}", worse.len(), worse.join(", "))
490            });
491        }
492        None
493    }
494}
495
496/// The parsed host policy. Everything default-closed — the two fields whose
497/// closed state is not the zero value (`skills`, `min_evidence`) say so in
498/// their own `Default`.
499#[derive(Debug, Clone, Serialize, Deserialize)]
500#[serde(deny_unknown_fields)]
501pub struct Policy {
502    /// Master opt-in (same posture as `allow_destructive_ops`: default off).
503    /// Auto-apply never fires unless this is true AND a grant matches.
504    #[serde(default)]
505    pub auto_apply_enabled: bool,
506    /// Auto-apply grants (default: none).
507    #[serde(default)]
508    pub auto_apply: Vec<AutoApplyGrant>,
509    /// Analyzer families the host disables entirely.
510    #[serde(default)]
511    pub deny: Vec<String>,
512    /// Per-analyzer severity floors (family → floor); combined with the
513    /// file's floors by taking the stricter of the two.
514    #[serde(default)]
515    pub severity_floors: BTreeMap<String, Severity>,
516    #[serde(default)]
517    pub telemetry: TelemetryMode,
518    /// The DISCOVER scoring rule (default: the review-queue objective).
519    #[serde(default)]
520    pub discover_objective: DiscoverObjective,
521    /// Measure every applicable LLM-authored proposal against this evalset
522    /// after apply (default: none — authored lessons carry no metric).
523    #[serde(default, skip_serializing_if = "Option::is_none")]
524    pub outcome_evalset: Option<OutcomeEvalset>,
525    /// Whether an Observation names its observer in the evidence bundle
526    /// (default: named).
527    #[serde(default)]
528    pub evidence_attribution: EvidenceAttribution,
529    /// When a pass is due (default: whenever it is called).
530    #[serde(default, skip_serializing_if = "is_default_cadence")]
531    pub cadence: Cadence,
532    /// Skill authoring by the LLM proposer (default: on, two steps minimum).
533    #[serde(default)]
534    pub skills: SkillAuthoring,
535    /// The fewest distinct evidence grains an LLM draft must cite to be
536    /// offered as a change rather than an advisory finding (default 1 — a
537    /// single instance may become a rule). An independent audit of 88 governed
538    /// decisions found that 15 of 28 approvals had generalised one instance
539    /// into standing policy; `2` is the setting that audit argues for. A draft
540    /// under the threshold is still stored and still reviewable — it simply
541    /// carries nothing a reviewer could apply.
542    #[serde(default = "default_min_evidence")]
543    pub min_evidence: u32,
544    /// Plan authoring by the LLM proposer (default: on, two steps minimum).
545    #[serde(default)]
546    pub plans: PlanAuthoring,
547    /// The Verify gate's second question (default: on). An applied
548    /// recommendation cites the grains it was derived from; when one of them
549    /// is later superseded by a DIFFERENT value, or retracted, the premise
550    /// the reviewer approved no longer holds. A lesson that outlives its
551    /// premise is measured harm: on PAST-Bench a rule encoding the old
552    /// regime's flag cost the governed arm 0.32 on the migration family it
553    /// was learned in (`crates/areev-bench/PERSIST.md`). With this on, the
554    /// gate records `drifted` and proposes the revert; a value-identical
555    /// supersession (consolidation) is not drift.
556    #[serde(default = "default_true")]
557    pub premise_drift: bool,
558    /// Whether EVERY cited grain must have moved before the engine withdraws
559    /// an open recommendation (#317), or any one is enough.
560    ///
561    /// `true` (the default, `"all"`) because a finding derived from six
562    /// grains of which one changed is weakened, not baseless — deciding that
563    /// is a reviewer's job. `false` is `"any"`, for hosts that want the
564    /// stricter sweep.
565    #[serde(default = "crate::policy::default_true")]
566    pub premise_drift_open_all: bool,
567    /// What to do with an authored lesson that near-duplicates a live one on
568    /// the same entity (default `flag`: queue it, marked).
569    #[serde(default, skip_serializing_if = "is_default_near_duplicate")]
570    pub near_duplicate: NearDuplicateMode,
571    /// The pre-apply rehearsal gate on plan revisions (default none).
572    #[serde(default, skip_serializing_if = "Option::is_none")]
573    pub plan_replay: Option<PlanReplayPolicy>,
574}
575
576fn is_default_near_duplicate(m: &NearDuplicateMode) -> bool {
577    *m == NearDuplicateMode::default()
578}
579
580fn is_default_cadence(c: &Cadence) -> bool {
581    !c.is_set()
582}
583
584impl Default for Policy {
585    fn default() -> Self {
586        Policy {
587            auto_apply_enabled: false,
588            auto_apply: Vec::new(),
589            deny: Vec::new(),
590            severity_floors: BTreeMap::new(),
591            telemetry: TelemetryMode::default(),
592            discover_objective: DiscoverObjective::default(),
593            outcome_evalset: None,
594            evidence_attribution: EvidenceAttribution::default(),
595            cadence: Cadence::default(),
596            skills: SkillAuthoring::default(),
597            min_evidence: 1,
598            plans: PlanAuthoring::default(),
599            premise_drift: true,
600            premise_drift_open_all: true,
601            near_duplicate: NearDuplicateMode::default(),
602            plan_replay: None,
603        }
604    }
605}
606
607impl Policy {
608    /// Parse a policy JSON string. Unknown keys are rejected (fail-closed).
609    pub fn from_json(s: &str) -> Result<Self> {
610        let p: Policy =
611            serde_json::from_str(s).map_err(|e| Error::InvalidProposal(format!("policy: {e}")))?;
612        if let Some(m) = p.outcome_evalset.as_ref().and_then(|e| e.min_effect.as_ref()) {
613            m.validate()?;
614        }
615        if let Some(c) = p.outcome_evalset.as_ref().and_then(|e| e.cost.as_ref()) {
616            c.validate()?;
617        }
618        if let Some(r) = p.plan_replay.as_ref() {
619            r.validate()?;
620        }
621        Ok(p)
622    }
623
624    /// Is this analyzer family denied by the host?
625    pub fn denies(&self, family: &str) -> bool {
626        self.deny.iter().any(|d| crate::manifest::analyzer_family(d) == family)
627    }
628
629    /// The host severity floor for a family, if any.
630    pub fn severity_floor(&self, family: &str) -> Option<Severity> {
631        self.severity_floors
632            .iter()
633            .find(|(k, _)| crate::manifest::analyzer_family(k) == family)
634            .map(|(_, v)| *v)
635    }
636
637    /// Does a grant permit auto-applying this family to `target_class` at
638    /// `severity`? Only the `memory` class is ever eligible.
639    ///
640    /// `query` was eligible until definition rewrites became executable
641    /// (issue #28). A grain edit changes one remembered value; a saved-query
642    /// or template rewrite changes what EVERY future context contains — the
643    /// blast radius is every turn from now on, not one fact. So a definition
644    /// rewrite always requires a human APPROVE + APPLY with `BECAUSE`, and
645    /// the class is excluded here by name, exactly as `code`/`evalset` are.
646    pub fn grants_auto_apply(&self, family: &str, target_class: &str, severity: Severity) -> bool {
647        if !self.auto_apply_enabled || target_class != "memory" {
648            return false;
649        }
650        self.auto_apply.iter().any(|g| {
651            crate::manifest::analyzer_family(&g.analyzer) == family
652                && g.targets.iter().any(|t| t == target_class)
653                && severity <= g.max_severity
654        })
655    }
656}
657
658#[cfg(test)]
659mod tests {
660    use super::*;
661
662    /// §7.4's stated invariant, pinned: code and evalset targets are
663    /// excluded from auto-apply BY NAME — even a policy that explicitly
664    /// names those classes in a grant is inert, because
665    /// `grants_auto_apply` hard-codes memory|query.
666    #[test]
667    fn code_targets_never_auto_apply_even_when_granted() {
668        let p = Policy::from_json(
669            r#"{"auto_apply_enabled": true,
670                "auto_apply": [{"analyzer": "loop.codegen", "targets": ["code", "evalset", "memory"], "max_severity": "high"}]}"#,
671        )
672        .unwrap();
673        assert!(!p.grants_auto_apply("loop.codegen", "code", Severity::Info));
674        assert!(!p.grants_auto_apply("loop.codegen", "evalset", Severity::Info));
675        assert!(
676            p.grants_auto_apply("loop.codegen", "memory", Severity::Low),
677            "the same grant's memory leg still works — the exclusion is by class"
678        );
679    }
680
681    #[test]
682    fn default_policy_grants_nothing() {
683        let p = Policy::default();
684        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Info));
685        assert!(!p.denies("loop.staleness"));
686        assert_eq!(p.telemetry, TelemetryMode::Aggregate);
687    }
688
689    #[test]
690    fn parses_and_grants() {
691        let p = Policy::from_json(
692            r#"{"auto_apply_enabled": true,
693                "auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}],
694                "deny": ["loop.staleness"],
695                "severity_floors": {"loop.contradiction_sweep": "high"}}"#,
696        )
697        .unwrap();
698        assert!(p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Low));
699        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::High), "above max_severity");
700        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "query", Severity::Low), "query not granted");
701        assert!(p.denies("loop.staleness"));
702        assert_eq!(p.severity_floor("loop.contradiction_sweep"), Some(Severity::High));
703    }
704
705    #[test]
706    fn prompt_and_host_targets_never_granted() {
707        let p = Policy::from_json(
708            r#"{"auto_apply_enabled": true,
709                "auto_apply": [{"analyzer": "x", "targets": ["prompt", "host"], "max_severity": "high"}]}"#,
710        )
711        .unwrap();
712        assert!(!p.grants_auto_apply("x", "prompt", Severity::Info));
713        assert!(!p.grants_auto_apply("x", "host", Severity::Info));
714    }
715
716    #[test]
717    fn discover_objective_defaults_to_the_review_queue_rule() {
718        assert_eq!(Policy::default().discover_objective, DiscoverObjective::ReviewQueue);
719        let p = Policy::from_json(r#"{"discover_objective": "learner"}"#).unwrap();
720        assert_eq!(p.discover_objective, DiscoverObjective::Learner);
721        assert!(
722            Policy::from_json(r#"{"discover_objective": "eager"}"#).is_err(),
723            "an unknown objective must not load as the default"
724        );
725    }
726
727    #[test]
728    fn outcome_evalset_baseline_is_a_named_choice() {
729        let p = Policy::from_json(
730            r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true, "baseline": "high_water"}}"#,
731        )
732        .unwrap();
733        assert_eq!(p.outcome_evalset.unwrap().baseline, BaselineKind::HighWater);
734        // Absent → the marginal comparison every existing verdict was made under.
735        let p = Policy::from_json(r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true}}"#).unwrap();
736        assert_eq!(p.outcome_evalset.unwrap().baseline, BaselineKind::NewestBeforeApply);
737        // A misspelling is a policy error that names the two accepted values,
738        // never a silent fall-through to the default.
739        let err = Policy::from_json(
740            r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true, "baseline": "best_ever"}}"#,
741        )
742        .expect_err("unknown baseline kind");
743        let msg = err.to_string();
744        assert!(msg.contains("newest_before_apply") && msg.contains("high_water"), "{msg}");
745    }
746
747    #[test]
748    fn plan_replay_gate_reads_the_report_the_way_the_ticket_says() {
749        let p = Policy::from_json(r#"{"plan_replay": {"min_runs": 3, "require_no_worse": true}}"#).unwrap();
750        let g = p.plan_replay.unwrap();
751        assert_eq!((g.min_runs, g.require_no_worse, g.max_out_of_support), (3, true, 0.5));
752        let report = |runs: u64, worse: u64, oos: u64, no_worse: bool| {
753            let rows: Vec<serde_json::Value> = (0..runs)
754                .map(|i| {
755                    let verdict = if i < worse { "worse" } else if i < worse + oos { "out_of_support" } else { "same" };
756                    serde_json::json!({"run_id": format!("r{i}"), "verdict": verdict, "incumbent_outcome": "completed", "candidate_outcome": if verdict == "worse" { "failed" } else { "completed" }})
757                })
758                .collect();
759            serde_json::json!({"totals": {"runs": runs, "out_of_support": oos}, "no_worse": no_worse,
760                               "out_of_support_fraction": oos as f64 / runs.max(1) as f64, "runs": rows})
761        };
762        assert_eq!(g.refusal(&report(2, 2, 0, false)), None, "under min_runs the gate abstains");
763        assert_eq!(g.refusal(&report(3, 0, 0, true)), None);
764        let why = g.refusal(&report(3, 1, 0, false)).expect("worse is refused");
765        assert!(why.contains("r0") && why.contains("completed → failed"), "{why}");
766        let why = g.refusal(&report(4, 0, 3, false)).expect("out of support beyond the limit is refused");
767        assert!(why.contains("75%") && why.contains("r0, r1, r2"), "{why}");
768        assert!(Policy::from_json(r#"{"plan_replay": {"max_out_of_support": 1.5}}"#).is_err());
769        assert!(Policy::from_json(r#"{"plan_replay": {"min_runs": 3, "strict": true}}"#).is_err());
770    }
771
772    #[test]
773    fn near_duplicate_mode_parses_and_rejects_unknown() {
774        assert_eq!(Policy::default().near_duplicate, NearDuplicateMode::Flag);
775        let p = Policy::from_json(r#"{"near_duplicate": "suppress"}"#).unwrap();
776        assert_eq!(p.near_duplicate, NearDuplicateMode::Suppress);
777        let err = Policy::from_json(r#"{"near_duplicate": "drop"}"#).expect_err("unknown mode");
778        let msg = err.to_string();
779        assert!(msg.contains("flag") && msg.contains("suppress"), "{msg}");
780    }
781
782    #[test]
783    fn cost_bound_needs_a_field_and_a_positive_ratio() {
784        let base = |extra: &str| {
785            format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "passed", "higher_is_better": true, "cost": {extra}}}}}"#)
786        };
787        let c = Policy::from_json(&base(r#"{"field": "tokens", "max_increase_ratio": 1.5}"#))
788            .unwrap().outcome_evalset.unwrap().cost.unwrap();
789        assert_eq!((c.field.as_str(), c.max_increase_ratio), ("tokens", 1.5));
790        for bad in [
791            r#"{"field": "tokens"}"#,
792            r#"{"max_increase_ratio": 1.5}"#,
793            r#"{"field": "", "max_increase_ratio": 1.5}"#,
794            r#"{"field": "tokens", "max_increase_ratio": 0}"#,
795            r#"{"field": "tokens", "max_increase_ratio": -1}"#,
796            r#"{"field": "tokens", "max_increase_ratio": "1.5"}"#,
797            r#"{"field": "tokens", "max_increase_ratio": 1.5, "hard": true}"#,
798        ] {
799            assert!(Policy::from_json(&base(bad)).is_err(), "{bad} must be a policy error");
800        }
801    }
802
803    #[test]
804    fn min_effect_is_one_non_negative_number() {
805        let base = |extra: &str| {
806            format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "passed", "higher_is_better": true, "min_effect": {extra}}}}}"#)
807        };
808        let m = Policy::from_json(&base(r#"{"count": 5}"#)).unwrap().outcome_evalset.unwrap().min_effect.unwrap();
809        assert_eq!(m.resolve("passed", 387), 5.0);
810        let m = Policy::from_json(&base(r#"{"points": 1.0}"#)).unwrap().outcome_evalset.unwrap().min_effect.unwrap();
811        assert_eq!(m.resolve("passed", 200), 2.0, "points scale a count field by the run total");
812        assert!((m.resolve("error_rate", 200) - 0.01).abs() < 1e-12, "a ratio field reads points as a fraction");
813        assert!((m.resolve("category_accuracy", 200) - 0.01).abs() < 1e-12, "a host field is assumed a ratio");
814        for bad in [r#"{"count": -1}"#, r#"{"points": "1"}"#, r#"{}"#, r#"{"count": 1, "points": 1}"#, r#"{"count": null}"#, r#"{"width": 2}"#] {
815            assert!(Policy::from_json(&base(bad)).is_err(), "{bad} must be a policy error");
816        }
817        // Absent → zero: today's verdicts.
818        assert!(Policy::from_json(&base("null")).unwrap().outcome_evalset.unwrap().min_effect.is_none());
819    }
820
821    #[test]
822    fn outcome_evalset_parses_with_default_horizons() {
823        let p = Policy::from_json(
824            r#"{"outcome_evalset": {"hash": "abc123", "field": "exact", "higher_is_better": true}}"#,
825        )
826        .unwrap();
827        let e = p.outcome_evalset.expect("parsed");
828        assert_eq!((e.hash.as_str(), e.field.as_str(), e.higher_is_better), ("abc123", "exact", true));
829        assert_eq!(e.horizons_ms, vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]);
830        assert!(Policy::default().outcome_evalset.is_none());
831        assert!(
832            Policy::from_json(r#"{"outcome_evalset": {"hash": "abc123", "field": "exact"}}"#).is_err(),
833            "the direction is not optional — a guessed one could revert an improvement"
834        );
835    }
836
837    #[test]
838    fn checkpoints_take_the_deployments_unit_and_a_bare_integer_stays_ms() {
839        let p = Policy::from_json(
840            r#"{"outcome_evalset": {"hash": "f", "field": "task_score", "higher_is_better": true,
841                "checkpoints": [{"after_runs": 1}, 3600000, {"after_grains": 50}, {"after_ms": 86400000}]}}"#,
842        )
843        .unwrap();
844        let e = p.outcome_evalset.unwrap();
845        assert_eq!(
846            e.schedule(),
847            vec![
848                Checkpoint::AfterMs(3_600_000),
849                Checkpoint::AfterMs(86_400_000),
850                Checkpoint::AfterRuns(1),
851                Checkpoint::AfterGrains(50),
852            ],
853            "sorted, deduplicated, and the bare integer read as milliseconds"
854        );
855        // Nothing set: the ms defaults, as time checkpoints — the schedule
856        // every deployment had before checkpoints had units.
857        let p = Policy::from_json(r#"{"outcome_evalset": {"hash": "f", "field": "x", "higher_is_better": true}}"#).unwrap();
858        assert_eq!(
859            p.outcome_evalset.unwrap().schedule(),
860            vec![
861                Checkpoint::AfterMs(86_400_000),
862                Checkpoint::AfterMs(7 * 86_400_000),
863                Checkpoint::AfterMs(30 * 86_400_000)
864            ]
865        );
866        for bad in [
867            r#"[{"after_turns": 3}]"#,
868            r#"[{"after_runs": -1}]"#,
869            r#"["1d"]"#,
870            r#"[{"after_runs": 1, "after_ms": 2}]"#,
871        ] {
872            let js = format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "x", "higher_is_better": true, "checkpoints": {bad}}}}}"#);
873            assert!(Policy::from_json(&js).is_err(), "{bad} must not load");
874        }
875    }
876
877    #[test]
878    fn cadence_defaults_to_always_due_and_parses_every_unit() {
879        let p = Policy::default();
880        assert!(!p.cadence.is_set());
881        let p = Policy::from_json(
882            r#"{"cadence": {"every_ms": 3600000, "every_events": 10, "every_sessions": 1, "every_grains": 50}}"#,
883        )
884        .unwrap();
885        assert!(p.cadence.is_set());
886        assert_eq!(p.cadence.every_events, Some(10));
887        assert!(
888            Policy::from_json(r#"{"cadence": {"every_turns": 10}}"#).is_err(),
889            "an unknown unit must not load as always-due"
890        );
891        // An unset cadence does not appear in the effective policy print.
892        assert!(!serde_json::to_string(&Policy::default()).unwrap().contains("cadence"));
893    }
894
895    #[test]
896    fn skills_default_on_with_two_steps_and_min_evidence_defaults_to_one() {
897        let p = Policy::default();
898        assert!(p.skills.enabled);
899        assert_eq!(p.skills.min_steps, 2);
900        assert_eq!(p.min_evidence, 1, "one instance may become a rule — today's behaviour");
901        let p = Policy::from_json(r#"{"skills": {"enabled": false}, "min_evidence": 2}"#).unwrap();
902        assert!(!p.skills.enabled);
903        assert_eq!(p.skills.min_steps, 2, "the unset field keeps its default, not zero");
904        assert_eq!(p.min_evidence, 2);
905        assert!(Policy::from_json(r#"{"skills": {"auto_apply": true}}"#).is_err(), "no back door");
906        // The JSON default round-trips through from_json identically.
907        let round = Policy::from_json(&serde_json::to_string(&Policy::default()).unwrap()).unwrap();
908        assert_eq!(round.min_evidence, 1);
909        assert!(round.skills.enabled);
910    }
911
912    #[test]
913    fn plans_and_premise_drift_default_on_and_are_switchable() {
914        let p = Policy::default();
915        assert!(p.plans.enabled);
916        assert_eq!(p.plans.min_nodes, 2);
917        assert!(p.premise_drift);
918        let p = Policy::from_json(r#"{"plans": {"enabled": false}, "premise_drift": false}"#).unwrap();
919        assert!(!p.plans.enabled);
920        assert_eq!(p.plans.min_nodes, 2);
921        assert!(!p.premise_drift);
922        assert!(Policy::from_json(r#"{"plans": {"auto_apply": true}}"#).is_err(), "no back door");
923    }
924
925    #[test]
926    fn evidence_attribution_defaults_to_named() {
927        assert_eq!(Policy::default().evidence_attribution, EvidenceAttribution::Named);
928        let p = Policy::from_json(r#"{"evidence_attribution": "anonymous"}"#).unwrap();
929        assert_eq!(p.evidence_attribution, EvidenceAttribution::Anonymous);
930        assert!(
931            Policy::from_json(r#"{"evidence_attribution": "redacted"}"#).is_err(),
932            "an unknown mode must not load as the default"
933        );
934    }
935
936    #[test]
937    fn unknown_keys_rejected() {
938        // A trust-floor field or an executable registration must not load.
939        assert!(Policy::from_json(r#"{"analyzer_cmd": "evil"}"#).is_err());
940        assert!(Policy::from_json(r#"{"auto_apply_free_text": true}"#).is_err());
941    }
942}