Skip to main content

areev_loop/
policy.rs

1//! Host policy — the optional `loop-policy.json` (proposal §6.2). It is the
2//! **only** place auto-apply is granted, and it is host config (per-process,
3//! never persisted in a memory file). All fields default-closed; the whole
4//! struct rejects unknown keys, so a policy that tries to register an
5//! executable (`--analyzer-cmd`) or touch a trust-floor field fails to load —
6//! a stolen or committed policy file must be inert.
7//!
8//! Precedence (enforced by the engine): engine ceilings > host CLI flags >
9//! this policy file > memory-file config. "The file selects and restricts;
10//! only the host grants."
11
12use crate::error::{Error, Result};
13use crate::model::Severity;
14use crate::recommendation::Checkpoint;
15use serde::{Deserialize, Serialize};
16use std::collections::BTreeMap;
17
18/// Telemetry sidecar mode (host-only).
19#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
20#[serde(rename_all = "lowercase")]
21pub enum TelemetryMode {
22    Off,
23    #[default]
24    Aggregate,
25    Full,
26}
27
28/// What DISCOVER optimizes for (`docs/loop-reflection.md` §5.1). Host config
29/// like everything else here: it changes the scoring rule the proposer is
30/// given, never the gates — every draft still has to survive GROUND, VERIFY,
31/// the confidence floor and a human review with a BECAUSE.
32#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
33#[serde(rename_all = "snake_case")]
34pub enum DiscoverObjective {
35    /// The review-queue objective: "nothing to report" is a zero-penalty
36    /// answer and a wrong finding costs twice a right one. Right for a queue
37    /// a person triages — it keeps the queue clean at the price of drafts
38    /// the model was not sure enough about.
39    #[default]
40    ReviewQueue,
41    /// The learner objective: the agent has to improve from THIS pass, so
42    /// abstaining in the face of a recurring failure, repeated rejections or
43    /// a person's instruction is penalized like a wrong lesson. Measured
44    /// need: under the review-queue rule a cheap model authored a lesson on
45    /// fewer than half of its passes over evidence that plainly held one.
46    Learner,
47}
48
49/// One auto-apply grant: an analyzer family may auto-apply to these target
50/// classes up to (and including) `max_severity`.
51#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
52#[serde(deny_unknown_fields)]
53pub struct AutoApplyGrant {
54    /// Analyzer family (e.g. `loop.duplicate_sweep`) or full id; matched by
55    /// family so a version bump keeps the grant.
56    pub analyzer: String,
57    /// Eligible target classes: `memory` and/or `query` only (prompt/host are
58    /// never auto-appliable and are rejected at eval time regardless).
59    pub targets: Vec<String>,
60    /// Highest severity this grant covers.
61    pub max_severity: Severity,
62}
63
64/// How an Observation is attributed in the evidence bundle handed to the LLM
65/// (`docs/loop.md`). `Named` renders `<observer> (a person) said of
66/// <subject>: <text>`; `Anonymous` renders the bare text, which is what the
67/// engine did before 2026-09-04.
68///
69/// It is host policy for two independent reasons. An operator may not want
70/// observer identities rendered into a model prompt at all — an observer id
71/// can be a person's name or account — and that is a privacy decision only
72/// the host can make. And it is the one variable in the receipts ablation
73/// (`crates/areev-bench/RECEIPTS.md`), where naming the speaker is what
74/// stopped one model reading a correction as a request.
75#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
76#[serde(rename_all = "snake_case")]
77pub enum EvidenceAttribution {
78    /// Name the observer on an Observation that records one.
79    #[default]
80    Named,
81    /// Render the bare text, attributing nothing.
82    Anonymous,
83}
84
85/// The evalset every LLM-authored, applicable proposal is measured against
86/// after apply (`docs/loop.md`, "Evalset-backed outcomes"). An authored
87/// lesson carries no built-in recurrence metric — nothing errors when a
88/// lesson is merely useless — so without this the Verify gate has nothing
89/// to re-measure for exactly the proposals a human was least able to judge.
90/// The host names the evalset and the field; the engine takes the baseline
91/// from the newest run journaled BEFORE the proposal and reads the current
92/// value from runs journaled AFTER the apply. No baseline run → no metric,
93/// never a fabricated one.
94#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
95#[serde(deny_unknown_fields)]
96pub struct OutcomeEvalset {
97    /// The evalset hash (the subject is `evalset:<hash>` in `agent:harness`).
98    pub hash: String,
99    /// The summary field to read: `passed`, `failed`, `total`, `error_rate`,
100    /// or any numeric field the host's harness writes into the summary.
101    pub field: String,
102    /// Which direction is an improvement — `passed` and an accuracy are
103    /// higher-is-better, `failed` and `error_rate` are not. Stated by the
104    /// host because getting it wrong would revert an improvement.
105    pub higher_is_better: bool,
106    /// Checkpoints after apply, in ms. Default 1d / 7d / 30d. The older
107    /// spelling; `checkpoints` wins when both are given.
108    #[serde(default = "default_horizons")]
109    pub horizons_ms: Vec<i64>,
110    /// The schedule in the deployment's own unit — `{"after_ms": n}`,
111    /// `{"after_runs": n}` or `{"after_grains": n}` (a bare integer is ms).
112    /// A benchmark or CI harness wants `[{"after_runs": 1}]`: measure at the
113    /// next graded run after the apply, however soon that is. Empty (the
114    /// default) means `horizons_ms`.
115    #[serde(default, skip_serializing_if = "Vec::is_empty")]
116    pub checkpoints: Vec<Checkpoint>,
117}
118
119fn default_horizons() -> Vec<i64> {
120    vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]
121}
122
123impl OutcomeEvalset {
124    /// The effective schedule: `checkpoints` when set, else `horizons_ms` as
125    /// time checkpoints.
126    pub fn schedule(&self) -> Vec<Checkpoint> {
127        let mut h: Vec<Checkpoint> = if self.checkpoints.is_empty() {
128            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
129        } else {
130            self.checkpoints.clone()
131        };
132        h.sort_unstable();
133        h.dedup();
134        h
135    }
136}
137
138/// When a loop pass is due — the loop's cadence, as host policy.
139///
140/// The engine has no clock and no scheduler of its own (ARCHITECTURE.md:
141/// cadence is data, evaluation is a command); a host calls `run` and the
142/// engine decides whether there is anything to do. Until now that decision
143/// was only expressible as per-call flags (`--min-new`, `--if-stale`), so
144/// every surface that can trigger a run — CLI, MCP, the console — had to be
145/// told separately, and none of them could count the units a chat deployment
146/// actually thinks in. This block is the same gate as the flags, set once in
147/// the policy file, with two more units.
148///
149/// Each field is a threshold; the pass is due when **any** set one is met
150/// (whichever comes first). Nothing set — the default — means a pass is due
151/// whenever it is called, which is what every deployment had before. Explicit
152/// per-call flags override the block (host CLI flags > policy file), and a
153/// full sweep (`areev loop reflect`) is a command, not a tick: it always runs.
154#[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)]
155#[serde(deny_unknown_fields)]
156pub struct Cadence {
157    /// Due when this long has passed since the last run (or it never ran).
158    #[serde(default, skip_serializing_if = "Option::is_none")]
159    pub every_ms: Option<i64>,
160    /// Due when this many grains of any kind landed since the last run.
161    #[serde(default, skip_serializing_if = "Option::is_none")]
162    pub every_grains: Option<u64>,
163    /// Due when this many Event grains — turns, in a chat deployment — landed
164    /// since the last run. Hermes's post-turn review fires every ten.
165    #[serde(default, skip_serializing_if = "Option::is_none")]
166    pub every_events: Option<u64>,
167    /// Due when Events from this many distinct sessions landed since the last
168    /// run: "reflect once per conversation" is `1`.
169    #[serde(default, skip_serializing_if = "Option::is_none")]
170    pub every_sessions: Option<u64>,
171}
172
173impl Cadence {
174    /// Whether any threshold is configured at all.
175    pub fn is_set(&self) -> bool {
176        self.every_ms.is_some()
177            || self.every_grains.is_some()
178            || self.every_events.is_some()
179            || self.every_sessions.is_some()
180    }
181}
182
183/// Whether, and how, DISCOVER may author a **Skill** — a reusable procedure
184/// with an applicability condition and ordered steps, derived from a
185/// trajectory that succeeded.
186///
187/// This exists because of a measured gap. On PAST-Bench the agent performed
188/// the procedure correctly in the learn episode on every seed and then, asked
189/// at session end whether there was anything to save, answered "nothing to
190/// save" — so the store was empty at evaluation and the memory scored below
191/// having none (`crates/areev-bench/PERSIST.md`). Every Skill in the memory
192/// depended on the model volunteering one mid-task. Hermes does not depend on
193/// that: a separate review pass writes its skills. This is Areev's equivalent,
194/// and it runs through the same gates as every other draft — GROUND, VERIFY,
195/// the confidence floor, a review with a BECAUSE — and is never auto-applied.
196#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
197#[serde(deny_unknown_fields)]
198pub struct SkillAuthoring {
199    /// Offer the `skill` proposal kind to the proposer at all (default: yes,
200    /// under LLM enrichment; no LLM, no skills).
201    #[serde(default = "default_true")]
202    pub enabled: bool,
203    /// Fewer ordered steps than this is a lesson, not a procedure (default 2).
204    #[serde(default = "default_min_steps")]
205    pub min_steps: u32,
206}
207
208fn default_true() -> bool {
209    true
210}
211fn default_min_steps() -> u32 {
212    2
213}
214fn default_min_evidence() -> u32 {
215    1
216}
217
218impl Default for SkillAuthoring {
219    fn default() -> Self {
220        SkillAuthoring { enabled: true, min_steps: 2 }
221    }
222}
223
224/// Whether, and how, DISCOVER may author a **plan** — a Workflow grain: named
225/// steps, edges with conditions in the runtime's frozen grammar, validated
226/// before a reviewer sees it — beside the Skill that carries the prose.
227///
228/// A skill is what a model reads; a plan is what the runtime can check and
229/// run. PAST-Bench's own labels call every procedural family "ordered steps,
230/// tools, conditions… a patched v2 supersedes v1" — which is a Workflow, and
231/// its patch is the `plan_revision` this engine already has. Storing the
232/// procedure as a plan buys structural validation (unique, reachable nodes;
233/// conditions that parse; bounded cycles) at author time, and puts the
234/// procedure where `areev run`, the run journal and `run_outcome` can reach
235/// it. Governed like every draft; never auto-applied.
236#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
237#[serde(deny_unknown_fields)]
238pub struct PlanAuthoring {
239    /// Offer the `plan` proposal kind (default: yes, under LLM enrichment).
240    #[serde(default = "default_true")]
241    pub enabled: bool,
242    /// Fewer steps than this is a lesson, not a procedure (default 2).
243    #[serde(default = "default_min_steps")]
244    pub min_nodes: u32,
245}
246
247impl Default for PlanAuthoring {
248    fn default() -> Self {
249        PlanAuthoring { enabled: true, min_nodes: 2 }
250    }
251}
252
253/// The parsed host policy. Everything default-closed — the two fields whose
254/// closed state is not the zero value (`skills`, `min_evidence`) say so in
255/// their own `Default`.
256#[derive(Debug, Clone, Serialize, Deserialize)]
257#[serde(deny_unknown_fields)]
258pub struct Policy {
259    /// Master opt-in (same posture as `allow_destructive_ops`: default off).
260    /// Auto-apply never fires unless this is true AND a grant matches.
261    #[serde(default)]
262    pub auto_apply_enabled: bool,
263    /// Auto-apply grants (default: none).
264    #[serde(default)]
265    pub auto_apply: Vec<AutoApplyGrant>,
266    /// Analyzer families the host disables entirely.
267    #[serde(default)]
268    pub deny: Vec<String>,
269    /// Per-analyzer severity floors (family → floor); combined with the
270    /// file's floors by taking the stricter of the two.
271    #[serde(default)]
272    pub severity_floors: BTreeMap<String, Severity>,
273    #[serde(default)]
274    pub telemetry: TelemetryMode,
275    /// The DISCOVER scoring rule (default: the review-queue objective).
276    #[serde(default)]
277    pub discover_objective: DiscoverObjective,
278    /// Measure every applicable LLM-authored proposal against this evalset
279    /// after apply (default: none — authored lessons carry no metric).
280    #[serde(default, skip_serializing_if = "Option::is_none")]
281    pub outcome_evalset: Option<OutcomeEvalset>,
282    /// Whether an Observation names its observer in the evidence bundle
283    /// (default: named).
284    #[serde(default)]
285    pub evidence_attribution: EvidenceAttribution,
286    /// When a pass is due (default: whenever it is called).
287    #[serde(default, skip_serializing_if = "is_default_cadence")]
288    pub cadence: Cadence,
289    /// Skill authoring by the LLM proposer (default: on, two steps minimum).
290    #[serde(default)]
291    pub skills: SkillAuthoring,
292    /// The fewest distinct evidence grains an LLM draft must cite to be
293    /// offered as a change rather than an advisory finding (default 1 — a
294    /// single instance may become a rule). An independent audit of 88 governed
295    /// decisions found that 15 of 28 approvals had generalised one instance
296    /// into standing policy; `2` is the setting that audit argues for. A draft
297    /// under the threshold is still stored and still reviewable — it simply
298    /// carries nothing a reviewer could apply.
299    #[serde(default = "default_min_evidence")]
300    pub min_evidence: u32,
301    /// Plan authoring by the LLM proposer (default: on, two steps minimum).
302    #[serde(default)]
303    pub plans: PlanAuthoring,
304    /// The Verify gate's second question (default: on). An applied
305    /// recommendation cites the grains it was derived from; when one of them
306    /// is later superseded by a DIFFERENT value, or retracted, the premise
307    /// the reviewer approved no longer holds. A lesson that outlives its
308    /// premise is measured harm: on PAST-Bench a rule encoding the old
309    /// regime's flag cost the governed arm 0.32 on the migration family it
310    /// was learned in (`crates/areev-bench/PERSIST.md`). With this on, the
311    /// gate records `drifted` and proposes the revert; a value-identical
312    /// supersession (consolidation) is not drift.
313    #[serde(default = "default_true")]
314    pub premise_drift: bool,
315}
316
317fn is_default_cadence(c: &Cadence) -> bool {
318    !c.is_set()
319}
320
321impl Default for Policy {
322    fn default() -> Self {
323        Policy {
324            auto_apply_enabled: false,
325            auto_apply: Vec::new(),
326            deny: Vec::new(),
327            severity_floors: BTreeMap::new(),
328            telemetry: TelemetryMode::default(),
329            discover_objective: DiscoverObjective::default(),
330            outcome_evalset: None,
331            evidence_attribution: EvidenceAttribution::default(),
332            cadence: Cadence::default(),
333            skills: SkillAuthoring::default(),
334            min_evidence: 1,
335            plans: PlanAuthoring::default(),
336            premise_drift: true,
337        }
338    }
339}
340
341impl Policy {
342    /// Parse a policy JSON string. Unknown keys are rejected (fail-closed).
343    pub fn from_json(s: &str) -> Result<Self> {
344        serde_json::from_str(s).map_err(|e| Error::InvalidProposal(format!("policy: {e}")))
345    }
346
347    /// Is this analyzer family denied by the host?
348    pub fn denies(&self, family: &str) -> bool {
349        self.deny.iter().any(|d| crate::manifest::analyzer_family(d) == family)
350    }
351
352    /// The host severity floor for a family, if any.
353    pub fn severity_floor(&self, family: &str) -> Option<Severity> {
354        self.severity_floors
355            .iter()
356            .find(|(k, _)| crate::manifest::analyzer_family(k) == family)
357            .map(|(_, v)| *v)
358    }
359
360    /// Does a grant permit auto-applying this family to `target_class` at
361    /// `severity`? Only the `memory` class is ever eligible.
362    ///
363    /// `query` was eligible until definition rewrites became executable
364    /// (issue #28). A grain edit changes one remembered value; a saved-query
365    /// or template rewrite changes what EVERY future context contains — the
366    /// blast radius is every turn from now on, not one fact. So a definition
367    /// rewrite always requires a human APPROVE + APPLY with `BECAUSE`, and
368    /// the class is excluded here by name, exactly as `code`/`evalset` are.
369    pub fn grants_auto_apply(&self, family: &str, target_class: &str, severity: Severity) -> bool {
370        if !self.auto_apply_enabled || target_class != "memory" {
371            return false;
372        }
373        self.auto_apply.iter().any(|g| {
374            crate::manifest::analyzer_family(&g.analyzer) == family
375                && g.targets.iter().any(|t| t == target_class)
376                && severity <= g.max_severity
377        })
378    }
379}
380
381#[cfg(test)]
382mod tests {
383    use super::*;
384
385    /// §7.4's stated invariant, pinned: code and evalset targets are
386    /// excluded from auto-apply BY NAME — even a policy that explicitly
387    /// names those classes in a grant is inert, because
388    /// `grants_auto_apply` hard-codes memory|query.
389    #[test]
390    fn code_targets_never_auto_apply_even_when_granted() {
391        let p = Policy::from_json(
392            r#"{"auto_apply_enabled": true,
393                "auto_apply": [{"analyzer": "loop.codegen", "targets": ["code", "evalset", "memory"], "max_severity": "high"}]}"#,
394        )
395        .unwrap();
396        assert!(!p.grants_auto_apply("loop.codegen", "code", Severity::Info));
397        assert!(!p.grants_auto_apply("loop.codegen", "evalset", Severity::Info));
398        assert!(
399            p.grants_auto_apply("loop.codegen", "memory", Severity::Low),
400            "the same grant's memory leg still works — the exclusion is by class"
401        );
402    }
403
404    #[test]
405    fn default_policy_grants_nothing() {
406        let p = Policy::default();
407        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Info));
408        assert!(!p.denies("loop.staleness"));
409        assert_eq!(p.telemetry, TelemetryMode::Aggregate);
410    }
411
412    #[test]
413    fn parses_and_grants() {
414        let p = Policy::from_json(
415            r#"{"auto_apply_enabled": true,
416                "auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}],
417                "deny": ["loop.staleness"],
418                "severity_floors": {"loop.contradiction_sweep": "high"}}"#,
419        )
420        .unwrap();
421        assert!(p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Low));
422        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::High), "above max_severity");
423        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "query", Severity::Low), "query not granted");
424        assert!(p.denies("loop.staleness"));
425        assert_eq!(p.severity_floor("loop.contradiction_sweep"), Some(Severity::High));
426    }
427
428    #[test]
429    fn prompt_and_host_targets_never_granted() {
430        let p = Policy::from_json(
431            r#"{"auto_apply_enabled": true,
432                "auto_apply": [{"analyzer": "x", "targets": ["prompt", "host"], "max_severity": "high"}]}"#,
433        )
434        .unwrap();
435        assert!(!p.grants_auto_apply("x", "prompt", Severity::Info));
436        assert!(!p.grants_auto_apply("x", "host", Severity::Info));
437    }
438
439    #[test]
440    fn discover_objective_defaults_to_the_review_queue_rule() {
441        assert_eq!(Policy::default().discover_objective, DiscoverObjective::ReviewQueue);
442        let p = Policy::from_json(r#"{"discover_objective": "learner"}"#).unwrap();
443        assert_eq!(p.discover_objective, DiscoverObjective::Learner);
444        assert!(
445            Policy::from_json(r#"{"discover_objective": "eager"}"#).is_err(),
446            "an unknown objective must not load as the default"
447        );
448    }
449
450    #[test]
451    fn outcome_evalset_parses_with_default_horizons() {
452        let p = Policy::from_json(
453            r#"{"outcome_evalset": {"hash": "abc123", "field": "exact", "higher_is_better": true}}"#,
454        )
455        .unwrap();
456        let e = p.outcome_evalset.expect("parsed");
457        assert_eq!((e.hash.as_str(), e.field.as_str(), e.higher_is_better), ("abc123", "exact", true));
458        assert_eq!(e.horizons_ms, vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]);
459        assert!(Policy::default().outcome_evalset.is_none());
460        assert!(
461            Policy::from_json(r#"{"outcome_evalset": {"hash": "abc123", "field": "exact"}}"#).is_err(),
462            "the direction is not optional — a guessed one could revert an improvement"
463        );
464    }
465
466    #[test]
467    fn checkpoints_take_the_deployments_unit_and_a_bare_integer_stays_ms() {
468        let p = Policy::from_json(
469            r#"{"outcome_evalset": {"hash": "f", "field": "task_score", "higher_is_better": true,
470                "checkpoints": [{"after_runs": 1}, 3600000, {"after_grains": 50}, {"after_ms": 86400000}]}}"#,
471        )
472        .unwrap();
473        let e = p.outcome_evalset.unwrap();
474        assert_eq!(
475            e.schedule(),
476            vec![
477                Checkpoint::AfterMs(3_600_000),
478                Checkpoint::AfterMs(86_400_000),
479                Checkpoint::AfterRuns(1),
480                Checkpoint::AfterGrains(50),
481            ],
482            "sorted, deduplicated, and the bare integer read as milliseconds"
483        );
484        // Nothing set: the ms defaults, as time checkpoints — the schedule
485        // every deployment had before checkpoints had units.
486        let p = Policy::from_json(r#"{"outcome_evalset": {"hash": "f", "field": "x", "higher_is_better": true}}"#).unwrap();
487        assert_eq!(
488            p.outcome_evalset.unwrap().schedule(),
489            vec![
490                Checkpoint::AfterMs(86_400_000),
491                Checkpoint::AfterMs(7 * 86_400_000),
492                Checkpoint::AfterMs(30 * 86_400_000)
493            ]
494        );
495        for bad in [
496            r#"[{"after_turns": 3}]"#,
497            r#"[{"after_runs": -1}]"#,
498            r#"["1d"]"#,
499            r#"[{"after_runs": 1, "after_ms": 2}]"#,
500        ] {
501            let js = format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "x", "higher_is_better": true, "checkpoints": {bad}}}}}"#);
502            assert!(Policy::from_json(&js).is_err(), "{bad} must not load");
503        }
504    }
505
506    #[test]
507    fn cadence_defaults_to_always_due_and_parses_every_unit() {
508        let p = Policy::default();
509        assert!(!p.cadence.is_set());
510        let p = Policy::from_json(
511            r#"{"cadence": {"every_ms": 3600000, "every_events": 10, "every_sessions": 1, "every_grains": 50}}"#,
512        )
513        .unwrap();
514        assert!(p.cadence.is_set());
515        assert_eq!(p.cadence.every_events, Some(10));
516        assert!(
517            Policy::from_json(r#"{"cadence": {"every_turns": 10}}"#).is_err(),
518            "an unknown unit must not load as always-due"
519        );
520        // An unset cadence does not appear in the effective policy print.
521        assert!(!serde_json::to_string(&Policy::default()).unwrap().contains("cadence"));
522    }
523
524    #[test]
525    fn skills_default_on_with_two_steps_and_min_evidence_defaults_to_one() {
526        let p = Policy::default();
527        assert!(p.skills.enabled);
528        assert_eq!(p.skills.min_steps, 2);
529        assert_eq!(p.min_evidence, 1, "one instance may become a rule — today's behaviour");
530        let p = Policy::from_json(r#"{"skills": {"enabled": false}, "min_evidence": 2}"#).unwrap();
531        assert!(!p.skills.enabled);
532        assert_eq!(p.skills.min_steps, 2, "the unset field keeps its default, not zero");
533        assert_eq!(p.min_evidence, 2);
534        assert!(Policy::from_json(r#"{"skills": {"auto_apply": true}}"#).is_err(), "no back door");
535        // The JSON default round-trips through from_json identically.
536        let round = Policy::from_json(&serde_json::to_string(&Policy::default()).unwrap()).unwrap();
537        assert_eq!(round.min_evidence, 1);
538        assert!(round.skills.enabled);
539    }
540
541    #[test]
542    fn plans_and_premise_drift_default_on_and_are_switchable() {
543        let p = Policy::default();
544        assert!(p.plans.enabled);
545        assert_eq!(p.plans.min_nodes, 2);
546        assert!(p.premise_drift);
547        let p = Policy::from_json(r#"{"plans": {"enabled": false}, "premise_drift": false}"#).unwrap();
548        assert!(!p.plans.enabled);
549        assert_eq!(p.plans.min_nodes, 2);
550        assert!(!p.premise_drift);
551        assert!(Policy::from_json(r#"{"plans": {"auto_apply": true}}"#).is_err(), "no back door");
552    }
553
554    #[test]
555    fn evidence_attribution_defaults_to_named() {
556        assert_eq!(Policy::default().evidence_attribution, EvidenceAttribution::Named);
557        let p = Policy::from_json(r#"{"evidence_attribution": "anonymous"}"#).unwrap();
558        assert_eq!(p.evidence_attribution, EvidenceAttribution::Anonymous);
559        assert!(
560            Policy::from_json(r#"{"evidence_attribution": "redacted"}"#).is_err(),
561            "an unknown mode must not load as the default"
562        );
563    }
564
565    #[test]
566    fn unknown_keys_rejected() {
567        // A trust-floor field or an executable registration must not load.
568        assert!(Policy::from_json(r#"{"analyzer_cmd": "evil"}"#).is_err());
569        assert!(Policy::from_json(r#"{"auto_apply_free_text": true}"#).is_err());
570    }
571}