areev-loop 1.10.1

Areev Loop: the governed self-improvement engine for AI-agent memory. Standalone engine over an OmsSubstrate (CAL + grains) — zero Areev dependencies.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
//! Host policy — the optional `loop-policy.json` (proposal §6.2). It is the
//! **only** place auto-apply is granted, and it is host config (per-process,
//! never persisted in a memory file). All fields default-closed; the whole
//! struct rejects unknown keys, so a policy that tries to register an
//! executable (`--analyzer-cmd`) or touch a trust-floor field fails to load —
//! a stolen or committed policy file must be inert.
//!
//! Precedence (enforced by the engine): engine ceilings > host CLI flags >
//! this policy file > memory-file config. "The file selects and restricts;
//! only the host grants."

use crate::error::{Error, Result};
use crate::model::Severity;
use crate::recommendation::Checkpoint;
use serde::{Deserialize, Serialize};
use std::collections::BTreeMap;

/// Telemetry sidecar mode (host-only).
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "lowercase")]
pub enum TelemetryMode {
    Off,
    #[default]
    Aggregate,
    Full,
}

/// What DISCOVER optimizes for (`docs/loop-reflection.md` §5.1). Host config
/// like everything else here: it changes the scoring rule the proposer is
/// given, never the gates — every draft still has to survive GROUND, VERIFY,
/// the confidence floor and a human review with a BECAUSE.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum DiscoverObjective {
    /// The review-queue objective: "nothing to report" is a zero-penalty
    /// answer and a wrong finding costs twice a right one. Right for a queue
    /// a person triages — it keeps the queue clean at the price of drafts
    /// the model was not sure enough about.
    #[default]
    ReviewQueue,
    /// The learner objective: the agent has to improve from THIS pass, so
    /// abstaining in the face of a recurring failure, repeated rejections or
    /// a person's instruction is penalized like a wrong lesson. Measured
    /// need: under the review-queue rule a cheap model authored a lesson on
    /// fewer than half of its passes over evidence that plainly held one.
    Learner,
}

/// One auto-apply grant: an analyzer family may auto-apply to these target
/// classes up to (and including) `max_severity`.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct AutoApplyGrant {
    /// Analyzer family (e.g. `loop.duplicate_sweep`) or full id; matched by
    /// family so a version bump keeps the grant.
    pub analyzer: String,
    /// Eligible target classes: `memory` and/or `query` only (prompt/host are
    /// never auto-appliable and are rejected at eval time regardless).
    pub targets: Vec<String>,
    /// Highest severity this grant covers.
    pub max_severity: Severity,
}

/// How an Observation is attributed in the evidence bundle handed to the LLM
/// (`docs/loop.md`). `Named` renders `<observer> (a person) said of
/// <subject>: <text>`; `Anonymous` renders the bare text, which is what the
/// engine did before 2026-09-04.
///
/// It is host policy for two independent reasons. An operator may not want
/// observer identities rendered into a model prompt at all — an observer id
/// can be a person's name or account — and that is a privacy decision only
/// the host can make. And it is the one variable in the receipts ablation
/// (`crates/areev-bench/RECEIPTS.md`), where naming the speaker is what
/// stopped one model reading a correction as a request.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum EvidenceAttribution {
    /// Name the observer on an Observation that records one.
    #[default]
    Named,
    /// Render the bare text, attributing nothing.
    Anonymous,
}

/// Which run journaled before the apply an evalset verdict compares
/// against (`docs/loop.md`, "Evalset-backed outcomes").
///
/// `newest_before_apply` is the marginal question — did THIS apply make the
/// agent worse than it was the moment before — and it is the default because
/// it never proposes reverting a rule for a drop an earlier rule caused.
/// `high_water` asks the question an operator actually has — is the agent
/// worse than the best it has been — and it is a policy choice, not the
/// default, because on a noisy evalset (or a deployment that does not
/// journal a run between applies) it attributes the whole fall from the
/// peak to whichever rule was applied last. That confounding is the trade;
/// measured need: on the ad-buy corpus (`crates/areev-bench/ADBUY.md`, seed
/// 3) an agent that reached 238 of 280 and then fell to 128 measured `held`
/// against the day-one run of 35, because day one was the only run before
/// the apply.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum BaselineKind {
    /// The newest run journaled strictly before the apply.
    #[default]
    NewestBeforeApply,
    /// The best run journaled strictly before the apply — max for a
    /// higher-is-better field, min otherwise.
    HighWater,
}

impl BaselineKind {
    /// The spelling the receipt records in `OutcomeResult::baseline_kind`.
    pub fn as_str(self) -> &'static str {
        match self {
            BaselineKind::NewestBeforeApply => "newest_before_apply",
            BaselineKind::HighWater => "high_water",
        }
    }
}

/// The Verify gate's minimum effect size — a **floor, not a significance
/// test**. A worsening of at most this much is `held`; no p-value, no
/// interval, and the docs say so (`docs/loop-proposal.md` §18 forbids
/// invented precision). Exactly one form:
///
/// - `{"count": n}` — absolute, in the field's own unit (`passed: 359 →
///   355` under `count: 5` holds);
/// - `{"points": p}` — percentage points. For the promoted count fields
///   (`passed`, `failed`, `total`) it is scaled by the baseline run's
///   `total`; for `error_rate`, and for any host-written field, it is read
///   as `p / 100` — a host field under `points` is assumed to be a ratio in
///   `0..1`, so a count-valued host field wants `count`.
///
/// Measured need (`docs/loop.md`): a 359 → 355 dip on 387 trials — within
/// what one adapter read twice can differ by — proposed a revert.
#[derive(Debug, Clone, Copy, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct MinEffect {
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub count: Option<f64>,
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub points: Option<f64>,
}

impl MinEffect {
    fn validate(&self) -> Result<()> {
        let bad = |what: &str| Err(Error::InvalidProposal(format!("policy: outcome_evalset.min_effect: {what}")));
        match (self.count, self.points) {
            (None, None) => bad("give {\"count\": n} or {\"points\": p}"),
            (Some(_), Some(_)) => bad("give count or points, not both"),
            (Some(v), None) | (None, Some(v)) if !(v.is_finite() && v >= 0.0) => {
                bad("must be a finite number ≥ 0")
            }
            _ => Ok(()),
        }
    }

    /// The tolerance in the field's unit, given the run total the `points`
    /// form scales a count field by.
    pub fn resolve(&self, field: &str, total: u64) -> f64 {
        match (self.count, self.points) {
            (Some(c), _) => c,
            (None, Some(p)) => match field {
                "passed" | "failed" | "total" => p / 100.0 * total as f64,
                _ => p / 100.0,
            },
            (None, None) => 0.0,
        }
    }
}

/// A cost bound beside the quality metric. The verdict on the quality
/// field is unchanged; when quality held but `field` on the run after the
/// apply exceeds `max_increase_ratio` × its value on the baseline run, the
/// checkpoint records `held_costlier` and `outcome_review` emits an
/// advisory Flag citing both runs — never a revert draft, because a
/// cost/quality trade is a human decision. `regressed` dominates: one
/// verdict per checkpoint. `field` is one of the promoted cost fields
/// (`effects`, `tokens`, `usd`, `wall_ms`, `cost_per_pass`) or any integer
/// key the harness writes; a run on which it is not measurable records no
/// cost figures, and the quality verdict still records.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct CostBound {
    pub field: String,
    /// Current may be at most this many times the baseline (`1.0` = no
    /// increase at all); finite and > 0.
    pub max_increase_ratio: f64,
}

impl CostBound {
    fn validate(&self) -> Result<()> {
        let bad = |what: &str| Err(Error::InvalidProposal(format!("policy: outcome_evalset.cost: {what}")));
        if self.field.trim().is_empty() {
            return bad("field must name a cost field (effects, tokens, usd, wall_ms, cost_per_pass, or a harness key)");
        }
        if !(self.max_increase_ratio.is_finite() && self.max_increase_ratio > 0.0) {
            return bad("max_increase_ratio must be a finite number > 0");
        }
        Ok(())
    }
}

/// The evalset every LLM-authored, applicable proposal is measured against
/// after apply (`docs/loop.md`, "Evalset-backed outcomes"). An authored
/// lesson carries no built-in recurrence metric — nothing errors when a
/// lesson is merely useless — so without this the Verify gate has nothing
/// to re-measure for exactly the proposals a human was least able to judge.
/// The host names the evalset and the field; the engine takes the baseline
/// from the newest run journaled BEFORE the proposal and reads the current
/// value from runs journaled AFTER the apply. No baseline run → no metric,
/// never a fabricated one.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct OutcomeEvalset {
    /// The evalset hash (the subject is `evalset:<hash>` in `agent:harness`).
    pub hash: String,
    /// The summary field to read: `passed`, `failed`, `total`, `error_rate`,
    /// or any numeric field the host's harness writes into the summary.
    pub field: String,
    /// Which direction is an improvement — `passed` and an accuracy are
    /// higher-is-better, `failed` and `error_rate` are not. Stated by the
    /// host because getting it wrong would revert an improvement.
    pub higher_is_better: bool,
    /// Checkpoints after apply, in ms. Default 1d / 7d / 30d. The older
    /// spelling; `checkpoints` wins when both are given.
    #[serde(default = "default_horizons")]
    pub horizons_ms: Vec<i64>,
    /// The schedule in the deployment's own unit — `{"after_ms": n}`,
    /// `{"after_runs": n}` or `{"after_grains": n}` (a bare integer is ms).
    /// A benchmark or CI harness wants `[{"after_runs": 1}]`: measure at the
    /// next graded run after the apply, however soon that is. Empty (the
    /// default) means `horizons_ms`.
    #[serde(default, skip_serializing_if = "Vec::is_empty")]
    pub checkpoints: Vec<Checkpoint>,
    /// Which run before the apply the verdict compares against (default
    /// `newest_before_apply`; see [`BaselineKind`] for why `high_water` is
    /// opt-in).
    #[serde(default)]
    pub baseline: BaselineKind,
    /// The minimum effect size a verdict needs to call a regression (default
    /// none: any drop past floating-point slack is a regression).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub min_effect: Option<MinEffect>,
    /// A cost bound read beside the quality field (default none).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub cost: Option<CostBound>,
}

fn default_horizons() -> Vec<i64> {
    vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]
}

impl OutcomeEvalset {
    /// The effective schedule: `checkpoints` when set, else `horizons_ms` as
    /// time checkpoints.
    pub fn schedule(&self) -> Vec<Checkpoint> {
        let mut h: Vec<Checkpoint> = if self.checkpoints.is_empty() {
            self.horizons_ms.iter().map(|ms| Checkpoint::AfterMs(*ms)).collect()
        } else {
            self.checkpoints.clone()
        };
        h.sort_unstable();
        h.dedup();
        h
    }
}

/// When a loop pass is due — the loop's cadence, as host policy.
///
/// The engine has no clock and no scheduler of its own (ARCHITECTURE.md:
/// cadence is data, evaluation is a command); a host calls `run` and the
/// engine decides whether there is anything to do. Until now that decision
/// was only expressible as per-call flags (`--min-new`, `--if-stale`), so
/// every surface that can trigger a run — CLI, MCP, the console — had to be
/// told separately, and none of them could count the units a chat deployment
/// actually thinks in. This block is the same gate as the flags, set once in
/// the policy file, with two more units.
///
/// Each field is a threshold; the pass is due when **any** set one is met
/// (whichever comes first). Nothing set — the default — means a pass is due
/// whenever it is called, which is what every deployment had before. Explicit
/// per-call flags override the block (host CLI flags > policy file), and a
/// full sweep (`areev loop reflect`) is a command, not a tick: it always runs.
#[derive(Debug, Clone, Default, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Cadence {
    /// Due when this long has passed since the last run (or it never ran).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub every_ms: Option<i64>,
    /// Due when this many grains of any kind landed since the last run.
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub every_grains: Option<u64>,
    /// Due when this many Event grains — turns, in a chat deployment — landed
    /// since the last run. Hermes's post-turn review fires every ten.
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub every_events: Option<u64>,
    /// Due when Events from this many distinct sessions landed since the last
    /// run: "reflect once per conversation" is `1`.
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub every_sessions: Option<u64>,
}

impl Cadence {
    /// Whether any threshold is configured at all.
    pub fn is_set(&self) -> bool {
        self.every_ms.is_some()
            || self.every_grains.is_some()
            || self.every_events.is_some()
            || self.every_sessions.is_some()
    }
}

/// Whether, and how, DISCOVER may author a **Skill** — a reusable procedure
/// with an applicability condition and ordered steps, derived from a
/// trajectory that succeeded.
///
/// This exists because of a measured gap. On PAST-Bench the agent performed
/// the procedure correctly in the learn episode on every seed and then, asked
/// at session end whether there was anything to save, answered "nothing to
/// save" — so the store was empty at evaluation and the memory scored below
/// having none (`crates/areev-bench/PERSIST.md`). Every Skill in the memory
/// depended on the model volunteering one mid-task. Hermes does not depend on
/// that: a separate review pass writes its skills. This is Areev's equivalent,
/// and it runs through the same gates as every other draft — GROUND, VERIFY,
/// the confidence floor, a review with a BECAUSE — and is never auto-applied.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct SkillAuthoring {
    /// Offer the `skill` proposal kind to the proposer at all (default: yes,
    /// under LLM enrichment; no LLM, no skills).
    #[serde(default = "default_true")]
    pub enabled: bool,
    /// Fewer ordered steps than this is a lesson, not a procedure (default 2).
    #[serde(default = "default_min_steps")]
    pub min_steps: u32,
}

pub(crate) fn default_true() -> bool {
    true
}
fn default_min_steps() -> u32 {
    2
}
fn default_min_evidence() -> u32 {
    1
}

impl Default for SkillAuthoring {
    fn default() -> Self {
        SkillAuthoring { enabled: true, min_steps: 2 }
    }
}

/// Whether, and how, DISCOVER may author a **plan** — a Workflow grain: named
/// steps, edges with conditions in the runtime's frozen grammar, validated
/// before a reviewer sees it — beside the Skill that carries the prose.
///
/// A skill is what a model reads; a plan is what the runtime can check and
/// run. PAST-Bench's own labels call every procedural family "ordered steps,
/// tools, conditions… a patched v2 supersedes v1" — which is a Workflow, and
/// its patch is the `plan_revision` this engine already has. Storing the
/// procedure as a plan buys structural validation (unique, reachable nodes;
/// conditions that parse; bounded cycles) at author time, and puts the
/// procedure where `areev run`, the run journal and `run_outcome` can reach
/// it. Governed like every draft; never auto-applied.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct PlanAuthoring {
    /// Offer the `plan` proposal kind (default: yes, under LLM enrichment).
    #[serde(default = "default_true")]
    pub enabled: bool,
    /// Fewer steps than this is a lesson, not a procedure (default 2).
    #[serde(default = "default_min_steps")]
    pub min_nodes: u32,
}

impl Default for PlanAuthoring {
    fn default() -> Self {
        PlanAuthoring { enabled: true, min_nodes: 2 }
    }
}

/// What DISCOVER does with a lesson that says, in other words, what a live
/// lesson on the same entity already says. `authored_dedup_key` collapses
/// the same text; a rewording is the reviewer's call — so `flag` (default)
/// lets it reach the queue carrying `near_duplicate_of`, and `suppress`
/// drops it before the queue and counts it in the funnel as
/// `dropped_near_duplicate`. Measured need (`crates/areev-bench/ADBUY.md`,
/// seed 3): ten approved rules stated four facts, each approvable alone.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Default, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum NearDuplicateMode {
    #[default]
    Flag,
    Suppress,
}

/// The pre-apply gate on a `plan_revision`: when the substrate can rehearse
/// the candidate against the journaled runs of the live plan
/// (`SubstrateRead::plan_replay` — `areev run shadow --plan-file`), refuse
/// to stamp the revision *applicable* when the rehearsal says it is worse
/// than the incumbent on the same runs, or when too many runs fall outside
/// the journal's support. Dream-RSI's monotone selection (arXiv 2609.14858
/// §3) as a gate rather than an auto-deploy: applying stays human, with a
/// BECAUSE. Default none — a revision is rehearsed when it can be and the
/// report rides on the card, but nothing is refused.
#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct PlanReplayPolicy {
    /// Fewer rehearsed runs than this and the gate abstains: two runs are an
    /// anecdote (default 3).
    #[serde(default = "default_min_runs")]
    pub min_runs: u32,
    /// Refuse a candidate that completes fewer of the same runs than the
    /// incumbent did (default true).
    #[serde(default = "default_true")]
    pub require_no_worse: bool,
    /// Refuse when more than this fraction of the runs could not be scored
    /// because the candidate asked for effects the journal never recorded
    /// (default 0.5).
    #[serde(default = "default_max_out_of_support")]
    pub max_out_of_support: f64,
}

fn default_min_runs() -> u32 {
    3
}
fn default_max_out_of_support() -> f64 {
    0.5
}

impl PlanReplayPolicy {
    fn validate(&self) -> Result<()> {
        if !(self.max_out_of_support.is_finite() && (0.0..=1.0).contains(&self.max_out_of_support)) {
            return Err(Error::InvalidProposal(
                "policy: plan_replay.max_out_of_support must be a fraction in 0..=1".into(),
            ));
        }
        Ok(())
    }

    /// The reason a rehearsal refuses the revision, or `None` to admit it.
    /// `report` is the runtime's `ShadowPlanReport` as JSON.
    pub fn refusal(&self, report: &serde_json::Value) -> Option<String> {
        let runs = report["totals"]["runs"].as_u64().unwrap_or(0);
        if runs < u64::from(self.min_runs) {
            return None;
        }
        let oos = report["out_of_support_fraction"].as_f64().unwrap_or(0.0);
        if oos > self.max_out_of_support {
            let ids: Vec<&str> = report["runs"]
                .as_array()
                .map(|a| {
                    a.iter()
                        .filter(|r| r["verdict"] == "out_of_support")
                        .filter_map(|r| r["run_id"].as_str())
                        .collect()
                })
                .unwrap_or_default();
            return Some(format!(
                "{:.0}% of {runs} rehearsed runs fall outside the journal's support (limit {:.0}%): {}",
                oos * 100.0,
                self.max_out_of_support * 100.0,
                ids.join(", ")
            ));
        }
        if self.require_no_worse && report["no_worse"].as_bool() != Some(true) {
            let worse: Vec<String> = report["runs"]
                .as_array()
                .map(|a| {
                    a.iter()
                        .filter(|r| r["verdict"] == "worse")
                        .filter_map(|r| {
                            Some(format!(
                                "{} ({} → {})",
                                r["run_id"].as_str()?,
                                r["incumbent_outcome"].as_str().unwrap_or("?"),
                                r["candidate_outcome"].as_str().unwrap_or("?")
                            ))
                        })
                        .collect()
                })
                .unwrap_or_default();
            let scored = runs - report["totals"]["out_of_support"].as_u64().unwrap_or(0);
            return Some(if worse.is_empty() {
                format!("no rehearsed run could be scored ({scored} of {runs})")
            } else {
                format!("worse than the incumbent on {} of {scored} rehearsed runs: {}", worse.len(), worse.join(", "))
            });
        }
        None
    }
}

/// The parsed host policy. Everything default-closed — the two fields whose
/// closed state is not the zero value (`skills`, `min_evidence`) say so in
/// their own `Default`.
#[derive(Debug, Clone, Serialize, Deserialize)]
#[serde(deny_unknown_fields)]
pub struct Policy {
    /// Master opt-in (same posture as `allow_destructive_ops`: default off).
    /// Auto-apply never fires unless this is true AND a grant matches.
    #[serde(default)]
    pub auto_apply_enabled: bool,
    /// Auto-apply grants (default: none).
    #[serde(default)]
    pub auto_apply: Vec<AutoApplyGrant>,
    /// Analyzer families the host disables entirely.
    #[serde(default)]
    pub deny: Vec<String>,
    /// Per-analyzer severity floors (family → floor); combined with the
    /// file's floors by taking the stricter of the two.
    #[serde(default)]
    pub severity_floors: BTreeMap<String, Severity>,
    #[serde(default)]
    pub telemetry: TelemetryMode,
    /// The DISCOVER scoring rule (default: the review-queue objective).
    #[serde(default)]
    pub discover_objective: DiscoverObjective,
    /// Measure every applicable LLM-authored proposal against this evalset
    /// after apply (default: none — authored lessons carry no metric).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub outcome_evalset: Option<OutcomeEvalset>,
    /// Whether an Observation names its observer in the evidence bundle
    /// (default: named).
    #[serde(default)]
    pub evidence_attribution: EvidenceAttribution,
    /// When a pass is due (default: whenever it is called).
    #[serde(default, skip_serializing_if = "is_default_cadence")]
    pub cadence: Cadence,
    /// Skill authoring by the LLM proposer (default: on, two steps minimum).
    #[serde(default)]
    pub skills: SkillAuthoring,
    /// The fewest distinct evidence grains an LLM draft must cite to be
    /// offered as a change rather than an advisory finding (default 1 — a
    /// single instance may become a rule). An independent audit of 88 governed
    /// decisions found that 15 of 28 approvals had generalised one instance
    /// into standing policy; `2` is the setting that audit argues for. A draft
    /// under the threshold is still stored and still reviewable — it simply
    /// carries nothing a reviewer could apply.
    #[serde(default = "default_min_evidence")]
    pub min_evidence: u32,
    /// Plan authoring by the LLM proposer (default: on, two steps minimum).
    #[serde(default)]
    pub plans: PlanAuthoring,
    /// The Verify gate's second question (default: on). An applied
    /// recommendation cites the grains it was derived from; when one of them
    /// is later superseded by a DIFFERENT value, or retracted, the premise
    /// the reviewer approved no longer holds. A lesson that outlives its
    /// premise is measured harm: on PAST-Bench a rule encoding the old
    /// regime's flag cost the governed arm 0.32 on the migration family it
    /// was learned in (`crates/areev-bench/PERSIST.md`). With this on, the
    /// gate records `drifted` and proposes the revert; a value-identical
    /// supersession (consolidation) is not drift.
    #[serde(default = "default_true")]
    pub premise_drift: bool,
    /// Whether EVERY cited grain must have moved before the engine withdraws
    /// an open recommendation (#317), or any one is enough.
    ///
    /// `true` (the default, `"all"`) because a finding derived from six
    /// grains of which one changed is weakened, not baseless — deciding that
    /// is a reviewer's job. `false` is `"any"`, for hosts that want the
    /// stricter sweep.
    #[serde(default = "crate::policy::default_true")]
    pub premise_drift_open_all: bool,
    /// What to do with an authored lesson that near-duplicates a live one on
    /// the same entity (default `flag`: queue it, marked).
    #[serde(default, skip_serializing_if = "is_default_near_duplicate")]
    pub near_duplicate: NearDuplicateMode,
    /// The pre-apply rehearsal gate on plan revisions (default none).
    #[serde(default, skip_serializing_if = "Option::is_none")]
    pub plan_replay: Option<PlanReplayPolicy>,
}

fn is_default_near_duplicate(m: &NearDuplicateMode) -> bool {
    *m == NearDuplicateMode::default()
}

fn is_default_cadence(c: &Cadence) -> bool {
    !c.is_set()
}

impl Default for Policy {
    fn default() -> Self {
        Policy {
            auto_apply_enabled: false,
            auto_apply: Vec::new(),
            deny: Vec::new(),
            severity_floors: BTreeMap::new(),
            telemetry: TelemetryMode::default(),
            discover_objective: DiscoverObjective::default(),
            outcome_evalset: None,
            evidence_attribution: EvidenceAttribution::default(),
            cadence: Cadence::default(),
            skills: SkillAuthoring::default(),
            min_evidence: 1,
            plans: PlanAuthoring::default(),
            premise_drift: true,
            premise_drift_open_all: true,
            near_duplicate: NearDuplicateMode::default(),
            plan_replay: None,
        }
    }
}

impl Policy {
    /// Parse a policy JSON string. Unknown keys are rejected (fail-closed).
    pub fn from_json(s: &str) -> Result<Self> {
        let p: Policy =
            serde_json::from_str(s).map_err(|e| Error::InvalidProposal(format!("policy: {e}")))?;
        if let Some(m) = p.outcome_evalset.as_ref().and_then(|e| e.min_effect.as_ref()) {
            m.validate()?;
        }
        if let Some(c) = p.outcome_evalset.as_ref().and_then(|e| e.cost.as_ref()) {
            c.validate()?;
        }
        if let Some(r) = p.plan_replay.as_ref() {
            r.validate()?;
        }
        Ok(p)
    }

    /// Is this analyzer family denied by the host?
    pub fn denies(&self, family: &str) -> bool {
        self.deny.iter().any(|d| crate::manifest::analyzer_family(d) == family)
    }

    /// The host severity floor for a family, if any.
    pub fn severity_floor(&self, family: &str) -> Option<Severity> {
        self.severity_floors
            .iter()
            .find(|(k, _)| crate::manifest::analyzer_family(k) == family)
            .map(|(_, v)| *v)
    }

    /// Does a grant permit auto-applying this family to `target_class` at
    /// `severity`? Only the `memory` class is ever eligible.
    ///
    /// `query` was eligible until definition rewrites became executable
    /// (issue #28). A grain edit changes one remembered value; a saved-query
    /// or template rewrite changes what EVERY future context contains — the
    /// blast radius is every turn from now on, not one fact. So a definition
    /// rewrite always requires a human APPROVE + APPLY with `BECAUSE`, and
    /// the class is excluded here by name, exactly as `code`/`evalset` are.
    pub fn grants_auto_apply(&self, family: &str, target_class: &str, severity: Severity) -> bool {
        if !self.auto_apply_enabled || target_class != "memory" {
            return false;
        }
        self.auto_apply.iter().any(|g| {
            crate::manifest::analyzer_family(&g.analyzer) == family
                && g.targets.iter().any(|t| t == target_class)
                && severity <= g.max_severity
        })
    }
}

#[cfg(test)]
mod tests {
    use super::*;

    /// §7.4's stated invariant, pinned: code and evalset targets are
    /// excluded from auto-apply BY NAME — even a policy that explicitly
    /// names those classes in a grant is inert, because
    /// `grants_auto_apply` hard-codes memory|query.
    #[test]
    fn code_targets_never_auto_apply_even_when_granted() {
        let p = Policy::from_json(
            r#"{"auto_apply_enabled": true,
                "auto_apply": [{"analyzer": "loop.codegen", "targets": ["code", "evalset", "memory"], "max_severity": "high"}]}"#,
        )
        .unwrap();
        assert!(!p.grants_auto_apply("loop.codegen", "code", Severity::Info));
        assert!(!p.grants_auto_apply("loop.codegen", "evalset", Severity::Info));
        assert!(
            p.grants_auto_apply("loop.codegen", "memory", Severity::Low),
            "the same grant's memory leg still works — the exclusion is by class"
        );
    }

    #[test]
    fn default_policy_grants_nothing() {
        let p = Policy::default();
        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Info));
        assert!(!p.denies("loop.staleness"));
        assert_eq!(p.telemetry, TelemetryMode::Aggregate);
    }

    #[test]
    fn parses_and_grants() {
        let p = Policy::from_json(
            r#"{"auto_apply_enabled": true,
                "auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}],
                "deny": ["loop.staleness"],
                "severity_floors": {"loop.contradiction_sweep": "high"}}"#,
        )
        .unwrap();
        assert!(p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::Low));
        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "memory", Severity::High), "above max_severity");
        assert!(!p.grants_auto_apply("loop.duplicate_sweep", "query", Severity::Low), "query not granted");
        assert!(p.denies("loop.staleness"));
        assert_eq!(p.severity_floor("loop.contradiction_sweep"), Some(Severity::High));
    }

    #[test]
    fn prompt_and_host_targets_never_granted() {
        let p = Policy::from_json(
            r#"{"auto_apply_enabled": true,
                "auto_apply": [{"analyzer": "x", "targets": ["prompt", "host"], "max_severity": "high"}]}"#,
        )
        .unwrap();
        assert!(!p.grants_auto_apply("x", "prompt", Severity::Info));
        assert!(!p.grants_auto_apply("x", "host", Severity::Info));
    }

    #[test]
    fn discover_objective_defaults_to_the_review_queue_rule() {
        assert_eq!(Policy::default().discover_objective, DiscoverObjective::ReviewQueue);
        let p = Policy::from_json(r#"{"discover_objective": "learner"}"#).unwrap();
        assert_eq!(p.discover_objective, DiscoverObjective::Learner);
        assert!(
            Policy::from_json(r#"{"discover_objective": "eager"}"#).is_err(),
            "an unknown objective must not load as the default"
        );
    }

    #[test]
    fn outcome_evalset_baseline_is_a_named_choice() {
        let p = Policy::from_json(
            r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true, "baseline": "high_water"}}"#,
        )
        .unwrap();
        assert_eq!(p.outcome_evalset.unwrap().baseline, BaselineKind::HighWater);
        // Absent → the marginal comparison every existing verdict was made under.
        let p = Policy::from_json(r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true}}"#).unwrap();
        assert_eq!(p.outcome_evalset.unwrap().baseline, BaselineKind::NewestBeforeApply);
        // A misspelling is a policy error that names the two accepted values,
        // never a silent fall-through to the default.
        let err = Policy::from_json(
            r#"{"outcome_evalset": {"hash": "f", "field": "passed", "higher_is_better": true, "baseline": "best_ever"}}"#,
        )
        .expect_err("unknown baseline kind");
        let msg = err.to_string();
        assert!(msg.contains("newest_before_apply") && msg.contains("high_water"), "{msg}");
    }

    #[test]
    fn plan_replay_gate_reads_the_report_the_way_the_ticket_says() {
        let p = Policy::from_json(r#"{"plan_replay": {"min_runs": 3, "require_no_worse": true}}"#).unwrap();
        let g = p.plan_replay.unwrap();
        assert_eq!((g.min_runs, g.require_no_worse, g.max_out_of_support), (3, true, 0.5));
        let report = |runs: u64, worse: u64, oos: u64, no_worse: bool| {
            let rows: Vec<serde_json::Value> = (0..runs)
                .map(|i| {
                    let verdict = if i < worse { "worse" } else if i < worse + oos { "out_of_support" } else { "same" };
                    serde_json::json!({"run_id": format!("r{i}"), "verdict": verdict, "incumbent_outcome": "completed", "candidate_outcome": if verdict == "worse" { "failed" } else { "completed" }})
                })
                .collect();
            serde_json::json!({"totals": {"runs": runs, "out_of_support": oos}, "no_worse": no_worse,
                               "out_of_support_fraction": oos as f64 / runs.max(1) as f64, "runs": rows})
        };
        assert_eq!(g.refusal(&report(2, 2, 0, false)), None, "under min_runs the gate abstains");
        assert_eq!(g.refusal(&report(3, 0, 0, true)), None);
        let why = g.refusal(&report(3, 1, 0, false)).expect("worse is refused");
        assert!(why.contains("r0") && why.contains("completed → failed"), "{why}");
        let why = g.refusal(&report(4, 0, 3, false)).expect("out of support beyond the limit is refused");
        assert!(why.contains("75%") && why.contains("r0, r1, r2"), "{why}");
        assert!(Policy::from_json(r#"{"plan_replay": {"max_out_of_support": 1.5}}"#).is_err());
        assert!(Policy::from_json(r#"{"plan_replay": {"min_runs": 3, "strict": true}}"#).is_err());
    }

    #[test]
    fn near_duplicate_mode_parses_and_rejects_unknown() {
        assert_eq!(Policy::default().near_duplicate, NearDuplicateMode::Flag);
        let p = Policy::from_json(r#"{"near_duplicate": "suppress"}"#).unwrap();
        assert_eq!(p.near_duplicate, NearDuplicateMode::Suppress);
        let err = Policy::from_json(r#"{"near_duplicate": "drop"}"#).expect_err("unknown mode");
        let msg = err.to_string();
        assert!(msg.contains("flag") && msg.contains("suppress"), "{msg}");
    }

    #[test]
    fn cost_bound_needs_a_field_and_a_positive_ratio() {
        let base = |extra: &str| {
            format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "passed", "higher_is_better": true, "cost": {extra}}}}}"#)
        };
        let c = Policy::from_json(&base(r#"{"field": "tokens", "max_increase_ratio": 1.5}"#))
            .unwrap().outcome_evalset.unwrap().cost.unwrap();
        assert_eq!((c.field.as_str(), c.max_increase_ratio), ("tokens", 1.5));
        for bad in [
            r#"{"field": "tokens"}"#,
            r#"{"max_increase_ratio": 1.5}"#,
            r#"{"field": "", "max_increase_ratio": 1.5}"#,
            r#"{"field": "tokens", "max_increase_ratio": 0}"#,
            r#"{"field": "tokens", "max_increase_ratio": -1}"#,
            r#"{"field": "tokens", "max_increase_ratio": "1.5"}"#,
            r#"{"field": "tokens", "max_increase_ratio": 1.5, "hard": true}"#,
        ] {
            assert!(Policy::from_json(&base(bad)).is_err(), "{bad} must be a policy error");
        }
    }

    #[test]
    fn min_effect_is_one_non_negative_number() {
        let base = |extra: &str| {
            format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "passed", "higher_is_better": true, "min_effect": {extra}}}}}"#)
        };
        let m = Policy::from_json(&base(r#"{"count": 5}"#)).unwrap().outcome_evalset.unwrap().min_effect.unwrap();
        assert_eq!(m.resolve("passed", 387), 5.0);
        let m = Policy::from_json(&base(r#"{"points": 1.0}"#)).unwrap().outcome_evalset.unwrap().min_effect.unwrap();
        assert_eq!(m.resolve("passed", 200), 2.0, "points scale a count field by the run total");
        assert!((m.resolve("error_rate", 200) - 0.01).abs() < 1e-12, "a ratio field reads points as a fraction");
        assert!((m.resolve("category_accuracy", 200) - 0.01).abs() < 1e-12, "a host field is assumed a ratio");
        for bad in [r#"{"count": -1}"#, r#"{"points": "1"}"#, r#"{}"#, r#"{"count": 1, "points": 1}"#, r#"{"count": null}"#, r#"{"width": 2}"#] {
            assert!(Policy::from_json(&base(bad)).is_err(), "{bad} must be a policy error");
        }
        // Absent → zero: today's verdicts.
        assert!(Policy::from_json(&base("null")).unwrap().outcome_evalset.unwrap().min_effect.is_none());
    }

    #[test]
    fn outcome_evalset_parses_with_default_horizons() {
        let p = Policy::from_json(
            r#"{"outcome_evalset": {"hash": "abc123", "field": "exact", "higher_is_better": true}}"#,
        )
        .unwrap();
        let e = p.outcome_evalset.expect("parsed");
        assert_eq!((e.hash.as_str(), e.field.as_str(), e.higher_is_better), ("abc123", "exact", true));
        assert_eq!(e.horizons_ms, vec![86_400_000, 7 * 86_400_000, 30 * 86_400_000]);
        assert!(Policy::default().outcome_evalset.is_none());
        assert!(
            Policy::from_json(r#"{"outcome_evalset": {"hash": "abc123", "field": "exact"}}"#).is_err(),
            "the direction is not optional — a guessed one could revert an improvement"
        );
    }

    #[test]
    fn checkpoints_take_the_deployments_unit_and_a_bare_integer_stays_ms() {
        let p = Policy::from_json(
            r#"{"outcome_evalset": {"hash": "f", "field": "task_score", "higher_is_better": true,
                "checkpoints": [{"after_runs": 1}, 3600000, {"after_grains": 50}, {"after_ms": 86400000}]}}"#,
        )
        .unwrap();
        let e = p.outcome_evalset.unwrap();
        assert_eq!(
            e.schedule(),
            vec![
                Checkpoint::AfterMs(3_600_000),
                Checkpoint::AfterMs(86_400_000),
                Checkpoint::AfterRuns(1),
                Checkpoint::AfterGrains(50),
            ],
            "sorted, deduplicated, and the bare integer read as milliseconds"
        );
        // Nothing set: the ms defaults, as time checkpoints — the schedule
        // every deployment had before checkpoints had units.
        let p = Policy::from_json(r#"{"outcome_evalset": {"hash": "f", "field": "x", "higher_is_better": true}}"#).unwrap();
        assert_eq!(
            p.outcome_evalset.unwrap().schedule(),
            vec![
                Checkpoint::AfterMs(86_400_000),
                Checkpoint::AfterMs(7 * 86_400_000),
                Checkpoint::AfterMs(30 * 86_400_000)
            ]
        );
        for bad in [
            r#"[{"after_turns": 3}]"#,
            r#"[{"after_runs": -1}]"#,
            r#"["1d"]"#,
            r#"[{"after_runs": 1, "after_ms": 2}]"#,
        ] {
            let js = format!(r#"{{"outcome_evalset": {{"hash": "f", "field": "x", "higher_is_better": true, "checkpoints": {bad}}}}}"#);
            assert!(Policy::from_json(&js).is_err(), "{bad} must not load");
        }
    }

    #[test]
    fn cadence_defaults_to_always_due_and_parses_every_unit() {
        let p = Policy::default();
        assert!(!p.cadence.is_set());
        let p = Policy::from_json(
            r#"{"cadence": {"every_ms": 3600000, "every_events": 10, "every_sessions": 1, "every_grains": 50}}"#,
        )
        .unwrap();
        assert!(p.cadence.is_set());
        assert_eq!(p.cadence.every_events, Some(10));
        assert!(
            Policy::from_json(r#"{"cadence": {"every_turns": 10}}"#).is_err(),
            "an unknown unit must not load as always-due"
        );
        // An unset cadence does not appear in the effective policy print.
        assert!(!serde_json::to_string(&Policy::default()).unwrap().contains("cadence"));
    }

    #[test]
    fn skills_default_on_with_two_steps_and_min_evidence_defaults_to_one() {
        let p = Policy::default();
        assert!(p.skills.enabled);
        assert_eq!(p.skills.min_steps, 2);
        assert_eq!(p.min_evidence, 1, "one instance may become a rule — today's behaviour");
        let p = Policy::from_json(r#"{"skills": {"enabled": false}, "min_evidence": 2}"#).unwrap();
        assert!(!p.skills.enabled);
        assert_eq!(p.skills.min_steps, 2, "the unset field keeps its default, not zero");
        assert_eq!(p.min_evidence, 2);
        assert!(Policy::from_json(r#"{"skills": {"auto_apply": true}}"#).is_err(), "no back door");
        // The JSON default round-trips through from_json identically.
        let round = Policy::from_json(&serde_json::to_string(&Policy::default()).unwrap()).unwrap();
        assert_eq!(round.min_evidence, 1);
        assert!(round.skills.enabled);
    }

    #[test]
    fn plans_and_premise_drift_default_on_and_are_switchable() {
        let p = Policy::default();
        assert!(p.plans.enabled);
        assert_eq!(p.plans.min_nodes, 2);
        assert!(p.premise_drift);
        let p = Policy::from_json(r#"{"plans": {"enabled": false}, "premise_drift": false}"#).unwrap();
        assert!(!p.plans.enabled);
        assert_eq!(p.plans.min_nodes, 2);
        assert!(!p.premise_drift);
        assert!(Policy::from_json(r#"{"plans": {"auto_apply": true}}"#).is_err(), "no back door");
    }

    #[test]
    fn evidence_attribution_defaults_to_named() {
        assert_eq!(Policy::default().evidence_attribution, EvidenceAttribution::Named);
        let p = Policy::from_json(r#"{"evidence_attribution": "anonymous"}"#).unwrap();
        assert_eq!(p.evidence_attribution, EvidenceAttribution::Anonymous);
        assert!(
            Policy::from_json(r#"{"evidence_attribution": "redacted"}"#).is_err(),
            "an unknown mode must not load as the default"
        );
    }

    #[test]
    fn unknown_keys_rejected() {
        // A trust-floor field or an executable registration must not load.
        assert!(Policy::from_json(r#"{"analyzer_cmd": "evil"}"#).is_err());
        assert!(Policy::from_json(r#"{"auto_apply_free_text": true}"#).is_err());
    }
}