use crate::engine::{Decision, Engine, RunOptions, RunOutcome, Scope, ScopeSet, SkipReason, LOOP_NS};
use crate::error::Error;
use crate::policy::Policy;
use crate::recommendation::{ObserverType, RecStatus, Recommendation};
use crate::testkit::TestSubstrate;
struct CountingSubstrate<'a> {
inner: &'a mut crate::reference::ReferenceSubstrate,
effect_calls: usize,
unpinned_specs: usize,
}
impl CountingSubstrate<'_> {
fn note_spec(&mut self, spec: &crate::substrate::GrainSpec) {
self.effect_calls += 1;
if !spec.fields.contains_key("created_at_ms") && !spec.fields.contains_key("at_ms") {
self.unpinned_specs += 1;
}
}
}
impl crate::substrate::SubstrateRead for CountingSubstrate<'_> {
fn capabilities(&self) -> crate::substrate::Capabilities {
self.inner.capabilities()
}
fn grains_of_type(
&self,
grain_type: &str,
namespace: Option<&str>,
opts: crate::substrate::ReadOpts,
) -> crate::error::Result<Vec<crate::model::GrainRecord>> {
self.inner.grains_of_type(grain_type, namespace, opts)
}
fn grain(&self, hash: &str) -> crate::error::Result<Option<crate::model::GrainRecord>> {
self.inner.grain(hash)
}
fn heads(
&self,
namespace: Option<&str>,
) -> crate::error::Result<Vec<crate::substrate::HeadGroup>> {
self.inner.heads(namespace)
}
fn telemetry(
&self,
namespace: Option<&str>,
) -> crate::error::Result<Option<crate::substrate::TelemetryView>> {
self.inner.telemetry(namespace)
}
}
impl crate::substrate::OmsSubstrate for CountingSubstrate<'_> {
fn put_grain(
&mut self,
spec: &crate::substrate::GrainSpec,
) -> crate::error::Result<String> {
self.note_spec(spec);
self.inner.put_grain(spec)
}
fn supersede(
&mut self,
target_hash: &str,
spec: &crate::substrate::GrainSpec,
justification: &str,
) -> crate::error::Result<String> {
self.note_spec(spec);
self.inner.supersede(target_hash, spec, justification)
}
fn retract(&mut self, hash: &str, reason: &str) -> crate::error::Result<()> {
self.effect_calls += 1;
self.inner.retract(hash, reason)
}
fn execute_cal(&mut self, cal: &str) -> crate::error::Result<Vec<serde_json::Value>> {
self.effect_calls += 1;
self.inner.execute_cal(cal)
}
fn validate_cal(&self, cal: &str) -> crate::error::Result<()> {
self.inner.validate_cal(cal)
}
fn load_state(&self) -> crate::error::Result<serde_json::Value> {
self.inner.load_state()
}
fn store_state(&mut self, state: &serde_json::Value) -> crate::error::Result<()> {
self.effect_calls += 1;
self.inner.store_state(state)
}
}
fn seed_all(sub: &mut TestSubstrate) {
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "deploy_target", "us-east-1");
sub.add_fact("acme", "deploy_target", "eu-west-1");
sub.add_fact_valid_to("promo", "active", "true", 500);
for _ in 0..5 {
sub.add_tool_call("stripe_refund", true, "rate_limited 429");
}
sub.add_tool_call("stripe_refund", false, "ok");
}
#[test]
fn run_proposes_across_analyzers_and_is_idempotent() {
let mut sub = TestSubstrate::new();
seed_all(&mut sub);
let e = Engine::with_builtins();
let r1 = e
.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
assert!(r1.ran());
assert!(
r1.stored >= 4,
"expected duplicate + contradiction + staleness + tool-failure, got {}",
r1.stored
);
let r2 = e
.run(&mut sub.inner, &RunOptions::default(), 20_000)
.unwrap();
assert!(r2.ran());
assert_eq!(r2.stored, 0, "no re-proposals");
}
#[test]
fn analyze_only_is_side_effect_free_and_matches_production_decisions() {
use crate::substrate::{OmsSubstrate, ReadOpts, SubstrateRead};
use std::collections::BTreeMap;
let mut sub = TestSubstrate::new();
seed_all(&mut sub);
let e = Engine::with_builtins();
let opts = RunOptions::default();
let now = 10_000;
let mut counted = CountingSubstrate {
inner: &mut sub.inner,
effect_calls: 0,
unpinned_specs: 0,
};
let state_before = counted.load_state().unwrap();
let count_before: usize = [
"fact",
"event",
"tool",
"observation",
"recommendation",
"audit",
]
.iter()
.map(|ty| {
counted
.grains_of_type(
ty,
None,
ReadOpts {
live_only: false,
since_ms: None,
},
)
.unwrap()
.len()
})
.sum();
let replay = e
.analyze_only(&counted, &opts, &BTreeMap::new(), now)
.unwrap();
assert_eq!(counted.effect_calls, 0, "replay must invoke no effect executor");
assert_eq!(counted.load_state().unwrap(), state_before, "state is immutable");
let count_after: usize = [
"fact",
"event",
"tool",
"observation",
"recommendation",
"audit",
]
.iter()
.map(|ty| {
counted
.grains_of_type(
ty,
None,
ReadOpts {
live_only: false,
since_ms: None,
},
)
.unwrap()
.len()
})
.sum();
assert_eq!(count_after, count_before, "analysis must append no grains or audit rows");
e.run(&mut counted, &opts, now).unwrap();
assert!(counted.effect_calls > 0, "production control must exercise effects");
assert_eq!(counted.unpinned_specs, 0, "engine writes must carry an injected clock");
let stored = counted
.grains_of_type(
crate::model::grain_type::RECOMMENDATION,
Some(LOOP_NS),
ReadOpts {
live_only: false,
since_ms: None,
},
)
.unwrap()
.into_iter()
.map(|g| Recommendation::from_fields(&g.hash, &g.fields).unwrap())
.collect::<Vec<_>>();
let replay_values: Vec<_> = replay
.iter()
.map(|r| serde_json::to_value(r).unwrap())
.collect();
let stored_values: Vec<_> = stored
.iter()
.map(|r| serde_json::to_value(r).unwrap())
.collect();
assert_eq!(replay_values, stored_values, "replay preserves production decision order");
}
#[test]
fn analyze_only_overrides_are_validated_and_applied() {
use std::collections::BTreeMap;
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("stripe_refund", true, "rate_limited 429");
}
let e = Engine::with_builtins();
let baseline = e
.analyze_only(&sub.inner, &RunOptions::default(), &BTreeMap::new(), 10_000)
.unwrap();
assert!(baseline.iter().any(|r| r.analyzer.starts_with("loop.tool_failure")));
let mut params = serde_json::Map::new();
params.insert("min_count".into(), serde_json::json!(10));
let overrides = BTreeMap::from([("loop.tool_failure/1".to_string(), params)]);
let replay = e
.analyze_only(&sub.inner, &RunOptions::default(), &overrides, 10_000)
.unwrap();
assert!(replay.iter().all(|r| !r.analyzer.starts_with("loop.tool_failure")));
}
struct MockLlm {
discover: String,
ground: String,
verify: String,
enrich: String,
}
impl crate::llm::LlmBackend for MockLlm {
fn model(&self) -> &str {
"mock-llm"
}
fn complete(&self, request: &str) -> crate::error::Result<String> {
Ok(if request.contains("\"op\":\"discover\"") {
self.discover.clone()
} else if request.contains("\"op\":\"ground\"") {
self.ground.clone()
} else if request.contains("\"op\":\"verify\"") {
self.verify.clone()
} else {
self.enrich.clone()
})
}
}
#[test]
fn llm_discover_verified_rec_is_stamped_with_confidence_and_enrich_adds_guidance() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "deploy_target", "us-east-1");
let _h2 = sub.add_fact("acme", "deploy_target", "eu-west-1");
let discover = format!(
r#"{{"recommendations":[
{{"summary":"prod region is ambiguous","target":"entity:test/acme","guidance":"pick one","evidence":["{h1}"],"confidence":0.9}},
{{"summary":"uncited nonsense","target":"entity:test/acme","evidence":["deadbeef"],"confidence":0.9}}
]}}"#
);
let ground = r#"{"results":[{"id":0,"supported":true,"reason":"entailed"}]}"#.to_string();
let verify =
r#"{"results":[{"id":0,"keep":true,"confidence":0.88,"reason":"novel and real"}]}"#.to_string();
let enrich =
r#"{"notes":[{"target":"entity:test/acme","guidance":"resolve to latest"}]}"#.to_string();
let e = Engine::with_builtins().with_llm(Box::new(MockLlm { discover, ground, verify, enrich }));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
let llm: Vec<_> = recs
.iter()
.filter(|r| matches!(r.origin, Origin::Llm { .. }))
.collect();
assert_eq!(llm.len(), 1, "only the cited+grounded+verified draft survives");
assert!(llm[0].summary.render().contains("ambiguous"));
assert_eq!(llm[0].evidence, vec![h1.clone()]);
assert!((llm[0].confidence - 0.88).abs() < 1e-9, "conf {}", llm[0].confidence);
assert!(!llm[0].destructive);
assert_eq!(llm[0].status, RecStatus::Pending);
let det = recs
.iter()
.find(|r| r.analyzer.starts_with("loop.contradiction"))
.expect("a contradiction recommendation");
assert_eq!(det.guidance.as_deref(), Some("resolve to latest"));
assert!(det.summary.render().contains("deploy_target"));
let m = e.llm_metrics(&sub.inner).unwrap();
assert_eq!(m.proposed, 1);
assert_eq!(m.pending, 1);
assert_eq!(m.approval_rate, None);
}
#[test]
fn verifier_drops_ungrounded_and_low_confidence_drafts() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "deploy_target", "us-east-1");
let _h2 = sub.add_fact("acme", "deploy_target", "eu-west-1");
let discover = format!(
r#"{{"recommendations":[
{{"summary":"grounded but the verifier is unsure","target":"entity:test/acme","evidence":["{h1}"],"confidence":0.9}},
{{"summary":"an ungrounded claim","target":"entity:test/acme","evidence":["{h1}"],"confidence":0.9}}
]}}"#
);
let ground =
r#"{"results":[{"id":0,"supported":true},{"id":1,"supported":false}]}"#.to_string();
let verify = r#"{"results":[{"id":0,"keep":true,"confidence":0.5}]}"#.to_string();
let enrich = r#"{"notes":[]}"#.to_string();
let e = Engine::with_builtins().with_llm(Box::new(MockLlm { discover, ground, verify, enrich }));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
assert!(
recs.iter().all(|r| !matches!(r.origin, Origin::Llm { .. })),
"ungrounded (id 1) and below-floor (id 0) drafts never reach the queue"
);
}
#[test]
fn separate_ground_backend_is_consulted_for_grounding() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "deploy_target", "us-east-1");
let _h2 = sub.add_fact("acme", "deploy_target", "eu-west-1");
let discover = format!(
r#"{{"recommendations":[
{{"summary":"grounded per the main model","target":"entity:test/acme","evidence":["{h1}"],"confidence":0.9}}
]}}"#
);
let main = MockLlm {
discover,
ground: r#"{"results":[{"id":0,"supported":true}]}"#.to_string(),
verify: r#"{"results":[{"id":0,"keep":true,"confidence":0.9}]}"#.to_string(),
enrich: r#"{"notes":[]}"#.to_string(),
};
let ground = MockLlm {
discover: String::new(),
ground: r#"{"results":[{"id":0,"supported":false}]}"#.to_string(),
verify: String::new(),
enrich: String::new(),
};
let e = Engine::with_builtins()
.with_llm(Box::new(main))
.with_ground_llm(Box::new(ground));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
assert!(
recs.iter().all(|r| !matches!(r.origin, Origin::Llm { .. })),
"the separate ground backend's rejection gates the draft"
);
}
#[cfg(unix)]
#[test]
fn external_command_analyzer_surfaces_advisory_findings() {
use crate::analyzer::Analyzer; use std::os::unix::fs::PermissionsExt;
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "country", "germany");
let script = std::env::temp_dir().join(format!("loop_ext_{}.sh", std::process::id()));
std::fs::write(
&script,
"#!/bin/sh\ncat >/dev/null\nprintf '%s' '{\"id\":\"acme.ext/1\",\"title\":\"ext\",\
\"findings\":[{\"target\":\"entity:test/acme\",\"summary\":\"external flags acme\",\
\"severity\":\"medium\",\"evidence\":[\"deadbeef\"]}]}'\n",
)
.unwrap();
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(0o755)).unwrap();
let analyzer = crate::external::CommandAnalyzer::new(script.to_str().unwrap()).unwrap();
assert_eq!(analyzer.manifest().id, "acme.ext/1");
assert_eq!(analyzer.manifest().trust_class, crate::manifest::TrustClass::Command);
assert_eq!(analyzer.manifest().auto_apply, crate::manifest::AutoApplyClass::Never);
let mut e = Engine::with_builtins();
e.register(Box::new(analyzer));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
let ext: Vec<_> = recs.iter().filter(|r| r.analyzer == "acme.ext/1").collect();
assert_eq!(ext.len(), 1, "the external finding is surfaced");
assert_eq!(ext[0].summary.render(), "external flags acme");
assert_eq!(ext[0].severity, crate::model::Severity::Medium);
assert!(!ext[0].destructive, "advisory flag, not a mutation");
std::fs::remove_file(&script).ok();
}
#[test]
fn config_edit_toggles_analyzer_and_is_admin_gated() {
use crate::config::AnalyzerConfigUpdate;
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "deploy_target", "us-east-1");
sub.add_fact("acme", "deploy_target", "eu-west-1");
let e = Engine::with_builtins();
let cid = e
.analyzer_settings(&sub.inner)
.unwrap()
.into_iter()
.find(|s| s.id.starts_with("loop.contradiction"))
.expect("contradiction analyzer present")
.id;
let denied = e.set_analyzer_config(
&mut sub.inner,
&cid,
AnalyzerConfigUpdate { enabled: Some(false), ..Default::default() },
&ScopeSet::of(&[Scope::Review]),
);
assert!(matches!(denied, Err(Error::ScopeDenied(_))), "config edit needs admin");
assert!(e
.set_analyzer_config(
&mut sub.inner,
"nope.x/1",
AnalyzerConfigUpdate::default(),
&ScopeSet::all(),
)
.is_err());
e.set_analyzer_config(
&mut sub.inner,
&cid,
AnalyzerConfigUpdate { enabled: Some(false), ..Default::default() },
&ScopeSet::all(),
)
.unwrap();
assert!(
!e.analyzer_settings(&sub.inner).unwrap().iter().find(|s| s.id == cid).unwrap().enabled,
"disabled in the effective settings"
);
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
assert!(
recs.iter().all(|r| !r.analyzer.starts_with("loop.contradiction")),
"the disabled analyzer produced no findings"
);
}
#[test]
fn full_sweep_reconsiders_grains_before_the_watermark() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "country", "germany");
Engine::with_builtins()
.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let discover = format!(
r#"{{"recommendations":[{{"summary":"semantic issue on acme","target":"entity:test/acme","evidence":["{h1}"],"confidence":0.9}}]}}"#
);
let e = Engine::with_builtins().with_llm(Box::new(MockLlm {
discover,
ground: r#"{"results":[{"id":0,"supported":true}]}"#.to_string(),
verify: r#"{"results":[{"id":0,"keep":true,"confidence":0.9}]}"#.to_string(),
enrich: r#"{"notes":[]}"#.to_string(),
}));
e.run(&mut sub.inner, &RunOptions::default(), 20_000).unwrap();
let incremental = e
.recommendations(&sub.inner, None)
.unwrap()
.into_iter()
.filter(|r| matches!(r.origin, Origin::Llm { .. }))
.count();
assert_eq!(incremental, 0, "an incremental run skips pre-watermark grains");
let sweep = RunOptions { full_sweep: true, ..Default::default() };
e.run(&mut sub.inner, &sweep, 30_000).unwrap();
let swept = e
.recommendations(&sub.inner, None)
.unwrap()
.into_iter()
.filter(|r| matches!(r.origin, Origin::Llm { .. }))
.count();
assert_eq!(swept, 1, "a full sweep reconsiders pre-watermark grains");
}
#[test]
fn no_llm_backend_is_the_identity() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "deploy_target", "us-east-1");
sub.add_fact("acme", "deploy_target", "eu-west-1");
let e = Engine::with_builtins(); e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
assert!(
recs.iter().all(|r| !matches!(r.origin, Origin::Llm { .. })),
"no llm-origin recs without a backend"
);
}
#[test]
fn review_apply_rollback_on_nondestructive() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("stripe_refund", true, "rate_limited 429");
}
sub.add_tool_call("stripe_refund", false, "ok");
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let recs = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap();
let tf = recs
.iter()
.find(|r| r.analyzer.starts_with("loop.tool_failure"))
.expect("a tool-failure recommendation");
let hash = tf.hash.clone();
assert!(!tf.destructive);
let scopes = ScopeSet::all();
e.review(
&mut sub.inner,
&hash,
Decision::Approve,
"user:alice",
ObserverType::Human,
&scopes,
"retries belong in the client",
11_000,
)
.unwrap();
let applied = e
.apply(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&scopes,
"applying the lesson",
false,
12_000,
)
.unwrap();
assert!(applied.rollbackable);
assert_eq!(
applied.created_hashes.len(),
1,
"the ADD created one lesson grain"
);
assert_eq!(status_of(&e, &sub, &hash), RecStatus::Applied);
e.rollback(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&scopes,
"undo",
13_000,
)
.unwrap();
assert_eq!(status_of(&e, &sub, &hash), RecStatus::RolledBack);
}
#[test]
fn destructive_apply_requires_admin_and_flag() {
let mut sub = TestSubstrate::new();
sub.add_fact_valid_to("promo", "active", "true", 500);
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let recs = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap();
let st = recs
.iter()
.find(|r| r.analyzer.starts_with("loop.staleness"))
.expect("a staleness recommendation");
let hash = st.hash.clone();
assert!(st.destructive);
let scopes = ScopeSet::all();
e.review(
&mut sub.inner,
&hash,
Decision::Approve,
"user:alice",
ObserverType::Human,
&scopes,
"expired",
11_000,
)
.unwrap();
let denied = e.apply(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&scopes,
"apply",
false,
12_000,
);
assert!(matches!(denied, Err(Error::DestructiveGated(_))));
let ok = e
.apply(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&scopes,
"apply",
true,
12_000,
)
.unwrap();
assert!(!ok.rollbackable, "FORGET has no inverse");
}
#[test]
fn purge_proposal_is_stamped_destructive_and_gated() {
use crate::analyzer::{AnalyzeCtx, Analyzer};
use crate::manifest::{
AnalyzerManifest, AutoApplyClass, CadenceClass, TargetClass, Tier, TrustClass,
};
use crate::model::ActionKind;
use crate::recommendation::{Proposal, RecDraft, Summary};
struct PurgeProposer {
manifest: AnalyzerManifest,
}
impl Analyzer for PurgeProposer {
fn manifest(&self) -> &AnalyzerManifest {
&self.manifest
}
fn analyze(&self, _ctx: &AnalyzeCtx) -> crate::error::Result<Vec<RecDraft>> {
Ok(vec![RecDraft::new(
"grain:deadbeef",
ActionKind::Expire,
Summary::new("test.purge", serde_json::Map::new()),
Proposal::Cal {
cal: r#"PURGE OLDER THAN 90d BECAUSE "retention""#.into(),
},
)])
}
}
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
let mut e = Engine::with_builtins();
e.register(Box::new(PurgeProposer {
manifest: AnalyzerManifest {
id: "test.purge/1".into(),
title: "Purge proposer".into(),
description: "test-only".into(),
tier: Tier::T0,
cadence: CadenceClass::Fast,
requires: vec![],
target_classes: vec![TargetClass::Memory],
auto_apply: AutoApplyClass::Never,
trust_class: TrustClass::Builtin,
params: vec![],
default_on: true,
},
}));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let recs = e.recommendations(&sub.inner, None).unwrap();
let purge = recs
.iter()
.find(|r| r.analyzer == "test.purge/1")
.expect("the purge recommendation");
assert!(purge.destructive, "PURGE must stamp destructive");
assert!(!purge.rollbackable, "bulk erasure has no inverse");
let scopes = ScopeSet::all();
let hash = purge.hash.clone();
e.review(
&mut sub.inner,
&hash,
Decision::Approve,
"user:alice",
ObserverType::Human,
&scopes,
"retention",
11_000,
)
.unwrap();
let denied = e.apply(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&scopes,
"apply",
false,
12_000,
);
assert!(
matches!(denied, Err(Error::DestructiveGated(_))),
"destructive gate must hold for PURGE, got {denied:?}"
);
}
#[test]
fn apply_on_pending_is_rejected() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("s", true, "boom 1");
}
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()[0]
.hash
.clone();
let res = e.apply(
&mut sub.inner,
&hash,
"user:alice",
ObserverType::Human,
&ScopeSet::all(),
"x",
false,
11_000,
);
assert!(matches!(res, Err(Error::LifecycleViolation(_))));
}
#[test]
fn self_approval_blocked_against_creator() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("s", true, "boom 1");
}
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let rec = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()[0]
.clone();
let creator = format!("engine:{}", rec.analyzer);
let blocked = e.review(
&mut sub.inner,
&rec.hash,
Decision::Approve,
&creator,
ObserverType::System,
&ScopeSet::all(),
"self",
11_000,
);
assert!(matches!(blocked, Err(Error::SelfApproval(_))));
assert!(e
.review(
&mut sub.inner,
&rec.hash,
Decision::Approve,
"user:alice",
ObserverType::Human,
&ScopeSet::all(),
"ok",
11_000
)
.is_ok());
}
#[test]
fn llm_rec_blocks_approval_by_triggering_actor() {
use crate::model::Origin;
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "deploy_target", "us-east-1");
let _h2 = sub.add_fact("acme", "deploy_target", "eu-west-1");
let discover = format!(
r#"{{"recommendations":[
{{"summary":"prod region is ambiguous","target":"entity:test/acme","guidance":"pick one","evidence":["{h1}"],"confidence":0.9}}
]}}"#
);
let ground = r#"{"results":[{"id":0,"supported":true,"reason":"entailed"}]}"#.to_string();
let verify =
r#"{"results":[{"id":0,"keep":true,"confidence":0.88,"reason":"real"}]}"#.to_string();
let enrich = r#"{"notes":[]}"#.to_string();
let e = Engine::with_builtins().with_llm(Box::new(MockLlm { discover, ground, verify, enrich }));
let opts = RunOptions {
triggering_actor: Some("user:sam".into()),
..Default::default()
};
e.run(&mut sub.inner, &opts, 10_000).unwrap();
let recs = e.recommendations(&sub.inner, Some(RecStatus::Pending)).unwrap();
let llm = recs
.iter()
.find(|r| matches!(r.origin, Origin::Llm { .. }))
.expect("an llm-origin recommendation");
let det = recs
.iter()
.find(|r| matches!(r.origin, Origin::Builtin))
.expect("a deterministic recommendation");
let blocked = e.review(
&mut sub.inner,
&llm.hash,
Decision::Approve,
"user:sam",
ObserverType::Human,
&ScopeSet::all(),
"looks right to me",
11_000,
);
assert!(matches!(blocked, Err(Error::SelfApproval(_))));
assert!(e
.review(
&mut sub.inner,
&llm.hash,
Decision::Approve,
"user:review",
ObserverType::Human,
&ScopeSet::all(),
"verified against the fork",
11_000
)
.is_ok());
assert!(e
.review(
&mut sub.inner,
&det.hash,
Decision::Approve,
"user:sam",
ObserverType::Human,
&ScopeSet::all(),
"resolve to latest",
11_000
)
.is_ok());
}
#[test]
fn review_requires_review_scope() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("s", true, "boom 1");
}
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()[0]
.hash
.clone();
let write_only = ScopeSet::of(&[Scope::Read, Scope::Write]);
let res = e.review(
&mut sub.inner,
&hash,
Decision::Approve,
"user:bob",
ObserverType::Human,
&write_only,
"x",
11_000,
);
assert!(matches!(res, Err(Error::ScopeDenied(_))), "write ⊉ review");
}
#[test]
fn empty_because_is_rejected() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("s", true, "boom 1");
}
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 10_000)
.unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()[0]
.hash
.clone();
let res = e.review(
&mut sub.inner,
&hash,
Decision::Approve,
"user:bob",
ObserverType::Human,
&ScopeSet::all(),
" ",
11_000,
);
assert!(
matches!(res, Err(Error::InvalidProposal(_))),
"BECAUSE is mandatory"
);
}
#[test]
fn gating_min_new_skips_but_stale_runs_first() {
let mut sub = TestSubstrate::new();
sub.add_fact("a", "b", "c");
let e = Engine::with_builtins();
let opts = RunOptions {
min_new: Some(100),
..Default::default()
};
let r = e.run(&mut sub.inner, &opts, 10_000).unwrap();
assert_eq!(r.outcome, RunOutcome::Skipped);
assert_eq!(r.skip_reason, Some(SkipReason::MinNewNotMet));
let mut sub2 = TestSubstrate::new();
sub2.add_fact("a", "b", "c");
let stale = RunOptions {
if_stale_ms: Some(3_600_000),
..Default::default()
};
assert!(e.run(&mut sub2.inner, &stale, 10_000).unwrap().ran());
}
#[test]
fn min_new_errors_wakes_a_run() {
let mut sub = TestSubstrate::new();
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 1_000)
.unwrap();
for _ in 0..4 {
sub.add_tool_call("s", true, "boom 1");
}
let opts = RunOptions {
min_new: Some(1000),
min_new_errors: Some(3),
..Default::default()
};
assert!(
e.run(&mut sub.inner, &opts, 2_000).unwrap().ran(),
"error gate wakes the run"
);
}
#[test]
fn default_policy_auto_applies_nothing() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise");
let e = Engine::with_builtins();
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 0, "a closed policy applies nothing");
assert!(e
.recommendations(&sub.inner, None)
.unwrap()
.iter()
.all(|x| x.status == RecStatus::Pending));
}
#[test]
fn policy_grant_auto_applies_structural_consolidation() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise"); let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 1, "the consolidation is auto-applied");
let applied = e.recommendations(&sub.inner, Some(RecStatus::Applied)).unwrap();
assert_eq!(applied.len(), 1);
assert!(applied[0].analyzer.starts_with("loop.duplicate_sweep"));
}
#[test]
fn fork_merge_never_auto_applies_even_when_granted() {
let mut sub = TestSubstrate::new();
sub.add_fork("caller/john", &["ref-a", "ref-b"]);
let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.fork_surfacing", "targets": ["memory"], "max_severity": "high"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 0, "a lossy fork merge is never auto-applied");
assert!(
e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|x| x.analyzer.starts_with("loop.fork_surfacing")),
"it is proposed for human review instead"
);
}
#[test]
fn auto_apply_never_touches_free_text_add() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("s", true, "boom 1");
}
let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.tool_failure", "targets": ["memory"], "max_severity": "high"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 0, "an ADD-with-text proposal never auto-applies");
}
#[test]
fn near_duplicate_consolidation_never_auto_applies() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "enterprise"); sub.add_observation("caller", "user asked about pricing tiers refunds billing invoices today");
sub.add_observation(
"caller",
"user asked about pricing tiers refunds billing invoices today please", );
let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 1, "only the value-identical consolidation auto-applies");
let recs = e.recommendations(&sub.inner, None).unwrap();
let near = recs
.iter()
.find(|x| x.summary.template_id == "duplicate.near")
.expect("the near-dup consolidation is proposed");
assert_eq!(
near.status,
RecStatus::Pending,
"a body-rewriting consolidation waits for a human"
);
}
#[test]
fn auto_applied_consolidation_preserves_namespace() {
use crate::substrate::{ReadOpts, SubstrateRead};
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise");
let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 1);
let live = sub
.inner
.grains_of_type("fact", None, ReadOpts { live_only: true, since_ms: None })
.unwrap();
let acme: Vec<_> = live
.iter()
.filter(|g| g.fact_subject() == Some("acme"))
.collect();
assert!(!acme.is_empty());
assert!(
acme.iter().all(|g| g.namespace == "test"),
"no replacement grain escaped to the store default namespace"
);
}
#[test]
fn applied_lesson_lands_in_evidence_namespace() {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call("stripe_refund", true, "rate_limited 429");
}
sub.add_tool_call("stripe_refund", false, "ok");
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 1_000_000).unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.analyzer.starts_with("loop.tool_failure"))
.expect("a tool-failure lesson")
.hash;
let scopes = ScopeSet::all();
e.review(&mut sub.inner, &hash, Decision::Approve, "user:a", ObserverType::Human, &scopes, "codify", 1_000_100).unwrap();
let applied = e
.apply(&mut sub.inner, &hash, "user:a", ObserverType::Human, &scopes, "apply", false, 1_000_200)
.unwrap();
let created = applied.created_hashes.first().expect("the lesson grain");
use crate::substrate::SubstrateRead;
let grain = sub.inner.grain(created).unwrap().expect("stored");
assert_eq!(
grain.namespace, "test",
"the lesson inherits the evidence tool calls' namespace"
);
}
#[test]
fn policy_deny_disables_an_analyzer() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise");
let policy = Policy::from_json(r#"{"deny": ["loop.duplicate_sweep"]}"#).unwrap();
let e = Engine::with_builtins().with_policy(policy);
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert!(
e.recommendations(&sub.inner, None)
.unwrap()
.iter()
.all(|x| !x.analyzer.starts_with("loop.duplicate_sweep")),
"a denied analyzer produces nothing"
);
}
const DAY: i64 = 86_400_000;
fn apply_lesson(now: i64) -> (Engine, TestSubstrate, String) {
let mut sub = TestSubstrate::new();
for _ in 0..5 {
sub.add_tool_call_at("stripe_refund", true, "rate_limited 429", 1_000);
}
sub.add_tool_call_at("stripe_refund", false, "ok", 1_100);
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 1_000_000).unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.analyzer.starts_with("loop.tool_failure"))
.expect("a tool-failure lesson")
.hash;
let scopes = ScopeSet::all();
e.review(&mut sub.inner, &hash, Decision::Approve, "user:a", ObserverType::Human, &scopes, "codify", now).unwrap();
e.apply(&mut sub.inner, &hash, "user:a", ObserverType::Human, &scopes, "apply the rule", false, now).unwrap();
(e, sub, hash)
}
#[test]
fn outcome_time_series_catches_a_late_regression() {
let t = 2_000_000;
let (e, mut sub, hash) = apply_lesson(t);
e.run(&mut sub.inner, &RunOptions::default(), t + 2 * DAY).unwrap();
e.run(&mut sub.inner, &RunOptions::default(), t + 8 * DAY).unwrap();
for _ in 0..2 {
sub.add_tool_call_at("stripe_refund", true, "rate_limited 429", t + 20 * DAY);
}
e.run(&mut sub.inner, &RunOptions::default(), t + 31 * DAY).unwrap();
let series: Vec<_> = e
.outcomes(&sub.inner)
.unwrap()
.into_iter()
.filter(|o| o.rec_hash == hash)
.collect();
let verdict_at = |h: i64| series.iter().find(|o| o.horizon_ms == h).map(|o| o.verdict.as_str());
assert_eq!(verdict_at(DAY), Some("held"), "no recurrence at day 1");
assert_eq!(verdict_at(7 * DAY), Some("held"), "still held at day 7");
assert_eq!(verdict_at(30 * DAY), Some("regressed"), "the late recurrence is caught at day 30");
assert!(
e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|r| r.analyzer.starts_with("loop.outcome_review")),
"a revert is proposed once the regression appears"
);
use crate::substrate::{ReadOpts, SubstrateRead};
let lesson = sub
.inner
.grains_of_type("fact", Some("test"), ReadOpts::default())
.unwrap()
.into_iter()
.find(|g| {
g.fields.get("subject").and_then(serde_json::Value::as_str)
== Some("stripe_refund")
&& g.fields.get("relation").and_then(serde_json::Value::as_str)
== Some("fails_with")
})
.expect("applied lesson grain");
let revert = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.analyzer.starts_with("loop.outcome_review"))
.expect("revert recommendation");
let scopes = ScopeSet::all();
e.review(
&mut sub.inner,
&revert.hash,
Decision::Approve,
"user:a",
ObserverType::Human,
&scopes,
"regression confirmed",
t + 31 * DAY + 1,
)
.unwrap();
e.apply(
&mut sub.inner,
&revert.hash,
"user:a",
ObserverType::Human,
&scopes,
"revert regressed lesson",
false,
t + 31 * DAY + 2,
)
.unwrap();
assert_eq!(status_of(&e, &sub, &hash), RecStatus::RolledBack);
assert_eq!(status_of(&e, &sub, &revert.hash), RecStatus::Applied);
assert!(
!sub.inner.grain(&lesson.hash).unwrap().unwrap().is_live(),
"the lesson created by the original apply must be retracted"
);
}
#[test]
fn outcome_time_series_holds_when_fix_works() {
let t = 2_000_000;
let (e, mut sub, hash) = apply_lesson(t);
sub.add_tool_call_at("stripe_refund", false, "ok", t + 10 * DAY); e.run(&mut sub.inner, &RunOptions::default(), t + 31 * DAY).unwrap();
let series: Vec<_> = e
.outcomes(&sub.inner)
.unwrap()
.into_iter()
.filter(|o| o.rec_hash == hash)
.collect();
assert_eq!(series.len(), 3, "all three checkpoints measured");
assert!(series.iter().all(|o| o.verdict == "held"), "held throughout");
assert!(
!e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|r| r.analyzer.starts_with("loop.outcome_review")),
"no revert when the fix held"
);
}
fn apply_resolution(now: i64) -> (Engine, TestSubstrate, String) {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "deploy_target", "us-east-1");
sub.add_fact("acme", "deploy_target", "eu-west-1");
let e = Engine::with_builtins();
e.run(&mut sub.inner, &RunOptions::default(), 1_000_000).unwrap();
let hash = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.analyzer.starts_with("loop.contradiction_sweep"))
.expect("a contradiction resolution")
.hash;
let scopes = ScopeSet::all();
e.review(&mut sub.inner, &hash, Decision::Approve, "user:a", ObserverType::Human, &scopes, "latest wins", now).unwrap();
e.apply(&mut sub.inner, &hash, "user:a", ObserverType::Human, &scopes, "resolve", false, now).unwrap();
(e, sub, hash)
}
#[test]
fn contradiction_outcome_regresses_when_conflict_returns() {
let t = 2_000_000;
let (e, mut sub, hash) = apply_resolution(t);
e.run(&mut sub.inner, &RunOptions::default(), t + 2 * DAY).unwrap();
sub.add_fact("acme", "deploy_target", "ap-south-1");
e.run(&mut sub.inner, &RunOptions::default(), t + 8 * DAY).unwrap();
let series: Vec<_> = e
.outcomes(&sub.inner)
.unwrap()
.into_iter()
.filter(|o| o.rec_hash == hash)
.collect();
let verdict_at = |h: i64| series.iter().find(|o| o.horizon_ms == h).map(|o| o.verdict.as_str());
assert_eq!(verdict_at(DAY), Some("held"), "one live value at day 1");
assert_eq!(verdict_at(7 * DAY), Some("regressed"), "the returned conflict is caught at day 7");
assert!(
e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|r| r.analyzer.starts_with("loop.outcome_review")),
"a revert is proposed for the regressed resolution"
);
}
#[test]
fn contradiction_outcome_holds_when_resolution_sticks() {
let t = 2_000_000;
let (e, mut sub, hash) = apply_resolution(t);
e.run(&mut sub.inner, &RunOptions::default(), t + 31 * DAY).unwrap();
let series: Vec<_> = e
.outcomes(&sub.inner)
.unwrap()
.into_iter()
.filter(|o| o.rec_hash == hash)
.collect();
assert_eq!(series.len(), 3, "all three checkpoints measured");
assert!(series.iter().all(|o| o.verdict == "held"), "held throughout");
}
fn status_of(e: &Engine, sub: &TestSubstrate, hash: &str) -> RecStatus {
e.recommendations(&sub.inner, None)
.unwrap()
.into_iter()
.find(|r| r.hash == hash)
.unwrap()
.status
}
#[test]
fn tool_lesson_holds_on_unrelated_same_tool_failure() {
let t = 2_000_000;
let (e, mut sub, hash) = apply_lesson(t); for _ in 0..3 {
sub.add_tool_call_at("stripe_refund", true, "insufficient_funds 402", t + 2 * DAY);
}
e.run(&mut sub.inner, &RunOptions::default(), t + 31 * DAY).unwrap();
let series: Vec<_> = e
.outcomes(&sub.inner)
.unwrap()
.into_iter()
.filter(|o| o.rec_hash == hash)
.collect();
assert!(!series.is_empty(), "the metric was measured");
assert!(
series.iter().all(|o| o.verdict == "held"),
"an unrelated-signature failure must not regress the lesson: {:?}",
series.iter().map(|o| (o.horizon_ms, o.verdict.clone())).collect::<Vec<_>>()
);
assert!(
!e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|r| r.analyzer.starts_with("loop.outcome_review")),
"no revert proposed for an unrelated failure"
);
}
#[test]
fn auto_apply_blocks_dropped_expiry() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise"); sub.add_fact_valid_to("acme", "tier", "Enterprise", 999_999); let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 0, "a dup carrying a valid_to must not auto-apply (expiry would be lost)");
assert!(
e.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.iter()
.any(|r| r.analyzer.starts_with("loop.duplicate_sweep")),
"it stays pending for human review"
);
}
#[test]
fn auto_apply_still_consolidates_plain_duplicates() {
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "tier", "Enterprise");
sub.add_fact("acme", "tier", "Enterprise");
let policy = Policy::from_json(
r#"{"auto_apply_enabled": true,
"auto_apply": [{"analyzer": "loop.duplicate_sweep", "targets": ["memory"], "max_severity": "low"}]}"#,
)
.unwrap();
let e = Engine::with_builtins().with_policy(policy);
let r = e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
assert_eq!(r.auto_applied, 1, "plain value-identical duplicates still consolidate");
}
#[test]
fn empty_signature_cluster_not_proposed() {
let mut sub = TestSubstrate::new();
for _ in 0..6 {
sub.add_tool_call("flaky", true, " "); }
let drafts = sub.analyze(&crate::analyzers::tool_failure::ToolFailureClustering::new(), 10_000);
assert!(drafts.is_empty(), "an empty-signature cluster must not be proposed, got {}", drafts.len());
}
#[test]
fn rejection_cooldown_doubles() {
let e = Engine::with_builtins();
let scopes = ScopeSet::all();
let reject_at = |sub: &mut TestSubstrate, now: i64| -> String {
e.run(&mut sub.inner, &RunOptions::default(), now).unwrap();
let rec = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.analyzer.starts_with("loop.contradiction"))
.expect("a contradiction recommendation");
let dk = rec.dedup_key.clone();
e.review(&mut sub.inner, &rec.hash, Decision::Reject, "user:a", ObserverType::Human, &scopes, "no", now)
.unwrap();
dk
};
let mut sub = TestSubstrate::new();
sub.add_fact("acme", "deploy_target", "us-east-1");
sub.add_fact("acme", "deploy_target", "eu-west-1"); let t0 = 10_000;
let dk = reject_at(&mut sub, t0);
let cooldown = |sub: &TestSubstrate, dk: &str| -> i64 {
use crate::substrate::OmsSubstrate;
crate::config::LoopPersisted::from_value(sub.inner.load_state().unwrap())
.unwrap()
.cooldowns
.get(dk)
.copied()
.unwrap()
};
assert_eq!(cooldown(&sub, &dk) - t0, 7 * DAY, "first rejection = 7d");
let t1 = cooldown(&sub, &dk) + 1;
reject_at(&mut sub, t1);
assert_eq!(cooldown(&sub, &dk) - t1, 14 * DAY, "second rejection doubles to 14d");
}
#[test]
fn advisory_findings_are_refused_by_preflight_not_after_approval() {
use crate::engine::ensure_executable;
use crate::model::Origin;
use crate::recommendation::Proposal;
use serde_json::{json, Map};
assert!(ensure_executable(&Proposal::Cal { cal: "ADD fact …".into() }).is_ok());
assert!(matches!(
ensure_executable(&Proposal::Edit {
format: "md".into(),
base_digest: "d".into(),
diff: "-a\n+b".into(),
}),
Err(Error::InvalidProposal(_))
));
let mut advisory = Map::new();
advisory.insert("note".into(), json!("go look at this"));
assert!(matches!(
ensure_executable(&Proposal::Data { data: advisory }),
Err(Error::InvalidProposal(_))
));
let mut revert = Map::new();
revert.insert("revert_of".into(), json!("abc123"));
assert!(ensure_executable(&Proposal::Data { data: revert }).is_ok());
let mut sub = TestSubstrate::new();
let h1 = sub.add_fact("acme", "deploy_target", "us-east-1");
let _h2 = sub.add_fact("acme", "deploy_target", "eu-west-1");
let discover = format!(
r#"{{"recommendations":[
{{"summary":"prod region is ambiguous","target":"entity:test/acme","guidance":"pick one","evidence":["{h1}"],"confidence":0.9}}
]}}"#
);
let e = Engine::with_builtins().with_llm(Box::new(MockLlm {
discover,
ground: r#"{"results":[{"id":0,"supported":true,"reason":"entailed"}]}"#.into(),
verify: r#"{"results":[{"id":0,"keep":true,"confidence":0.88,"reason":"real"}]}"#.into(),
enrich: r#"{"notes":[]}"#.into(),
}));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let llm = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| matches!(r.origin, Origin::Llm { .. }))
.expect("an llm-origin recommendation");
let refused = e.preflight_apply(&sub.inner, &llm.hash, &ScopeSet::all(), true);
assert!(
matches!(refused, Err(Error::InvalidProposal(_))),
"preflight must refuse an advisory finding, got {refused:?}"
);
assert_eq!(
status_of(&e, &sub, &llm.hash),
RecStatus::Pending,
"a refused preflight must leave the finding dismissible"
);
e.review(
&mut sub.inner,
&llm.hash,
Decision::Approve,
"user:reviewer",
ObserverType::Human,
&ScopeSet::all(),
"real issue, handling it in the host",
11_000,
)
.expect("acknowledging an advisory finding is legal");
match e.apply(
&mut sub.inner,
&llm.hash,
"user:reviewer",
ObserverType::Human,
&ScopeSet::all(),
"try to apply",
true,
12_000,
) {
Err(Error::InvalidProposal(msg)) => {
assert!(msg.contains("advisory"), "message should explain, got {msg:?}")
}
other => panic!("expected an advisory refusal, got {other:?}"),
}
}
#[test]
fn governed_code_change_applies_only_through_the_gate() {
use crate::analyzer::{AnalyzeCtx, Analyzer};
use crate::manifest::{
AnalyzerManifest, AutoApplyClass, CadenceClass, TargetClass, Tier, TrustClass,
};
use crate::model::ActionKind;
use crate::recommendation::{GatingEvidence, Proposal, RecDraft, Summary};
struct CodeProposer {
manifest: AnalyzerManifest,
evalset: String,
}
impl Analyzer for CodeProposer {
fn manifest(&self) -> &AnalyzerManifest {
&self.manifest
}
fn analyze(&self, _ctx: &AnalyzeCtx) -> crate::error::Result<Vec<RecDraft>> {
let mut args = serde_json::Map::new();
args.insert("text".into(), serde_json::Value::from("faster retry backoff"));
let mut data = serde_json::Map::new();
data.insert("code_blob".into(), serde_json::Value::from("cas://sha256:feed"));
Ok(vec![RecDraft::new(
"tool:cafe0123",
ActionKind::CodeRevision,
Summary::new("command.finding", args),
Proposal::Data { data },
)
.evalset_hash(self.evalset.clone())])
}
}
let mut sub = TestSubstrate::new();
let evalset_hash = sub.add_fact("evalset:retry", "mg:evalset", "{\"cases\":[]}");
let mut e = Engine::with_builtins();
e.register(Box::new(CodeProposer {
manifest: AnalyzerManifest {
id: "test.codegen/1".into(),
title: "Code proposer".into(),
description: "test-only".into(),
tier: Tier::T0,
cadence: CadenceClass::Fast,
requires: vec![],
target_classes: vec![TargetClass::Host],
auto_apply: AutoApplyClass::Never,
trust_class: TrustClass::Builtin,
params: vec![],
default_on: true,
},
evalset: evalset_hash.clone(),
}));
e.run(&mut sub.inner, &RunOptions::default(), 10_000).unwrap();
let rec = e
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.action_kind == ActionKind::CodeRevision)
.expect("the code revision");
assert_eq!(rec.evalset_hash.as_deref(), Some(evalset_hash.as_str()));
let hash = rec.hash.clone();
let scopes = ScopeSet::all();
e.review(
&mut sub.inner, &hash, Decision::Approve, "user:reviewer",
ObserverType::Human, &scopes, "diff reviewed", 11_000,
)
.unwrap();
match e.apply(&mut sub.inner, &hash, "user:reviewer", ObserverType::Human, &scopes, "ship it", false, 12_000) {
Err(Error::InvalidProposal(m)) => assert!(m.contains("gating run"), "{m}"),
other => panic!("ungated apply must refuse, got {other:?}"),
}
let wrong = GatingEvidence { evalset_hash: "not-the-pin".into(), run_id: "eval-1".into(), passed: 3, failed: 0 };
match e.apply_gated(&mut sub.inner, &hash, "user:reviewer", ObserverType::Human, &scopes, "ship", false, &wrong, 12_100) {
Err(Error::InvalidProposal(m)) => assert!(m.contains("pinned"), "{m}"),
other => panic!("wrong-evalset apply must refuse, got {other:?}"),
}
let failing = GatingEvidence { evalset_hash: evalset_hash.clone(), run_id: "eval-2".into(), passed: 2, failed: 1 };
match e.apply_gated(&mut sub.inner, &hash, "user:reviewer", ObserverType::Human, &scopes, "ship", false, &failing, 12_200) {
Err(Error::InvalidProposal(m)) => assert!(m.contains("failing gate"), "{m}"),
other => panic!("failing-gate apply must refuse, got {other:?}"),
}
let clean = GatingEvidence { evalset_hash: evalset_hash.clone(), run_id: "eval-3".into(), passed: 3, failed: 0 };
e.apply_gated(&mut sub.inner, &hash, "user:reviewer", ObserverType::Human, &scopes, "gated and green", false, &clean, 12_300)
.unwrap();
let audits = {
use crate::substrate::SubstrateRead;
sub.inner
.grains_of_type("observation", Some("areev-loop"), Default::default())
.unwrap()
};
let applied_audit = audits
.iter()
.find(|a| {
a.str_field("rec_hash") == Some(hash.as_str())
&& a.str_field("to_status") == Some("applied")
})
.expect("the applied audit record");
assert_eq!(applied_audit.str_field("gating_evalset"), Some(evalset_hash.as_str()));
assert_eq!(applied_audit.str_field("gating_run_id"), Some("eval-3"));
e.run(&mut sub.inner, &RunOptions::default(), 20_000).unwrap(); {
use crate::substrate::{GrainSpec, OmsSubstrate};
let spec = GrainSpec::new("fact", "test")
.with_field("subject", "evalset:retry")
.with_field("relation", "mg:evalset")
.with_field("object", "{\"cases\":[1]}");
sub.inner
.supersede(&evalset_hash, &spec, "evalset v2")
.unwrap();
}
let mut e2 = Engine::with_builtins();
e2.register(Box::new(CodeProposer {
manifest: AnalyzerManifest {
id: "test.codegen2/1".into(),
title: "Code proposer 2".into(),
description: "test-only".into(),
tier: Tier::T0,
cadence: CadenceClass::Fast,
requires: vec![],
target_classes: vec![TargetClass::Host],
auto_apply: AutoApplyClass::Never,
trust_class: TrustClass::Builtin,
params: vec![],
default_on: true,
},
evalset: evalset_hash.clone(),
}));
e2.run(&mut sub.inner, &RunOptions::default(), 21_000).unwrap();
let rec2 = e2
.recommendations(&sub.inner, Some(RecStatus::Pending))
.unwrap()
.into_iter()
.find(|r| r.action_kind == ActionKind::CodeRevision && r.analyzer.starts_with("test.codegen2"))
.expect("second code revision");
e2.review(&mut sub.inner, &rec2.hash, Decision::Approve, "user:reviewer", ObserverType::Human, &scopes, "ok", 22_000).unwrap();
let stale = GatingEvidence { evalset_hash: evalset_hash.clone(), run_id: "eval-4".into(), passed: 3, failed: 0 };
match e2.apply_gated(&mut sub.inner, &rec2.hash, "user:reviewer", ObserverType::Human, &scopes, "ship", false, &stale, 23_000) {
Err(Error::InvalidProposal(m)) => assert!(m.contains("superseded"), "{m}"),
other => panic!("stale-pin apply must refuse (re-gate), got {other:?}"),
}
}
mod evalset_outcome {
use super::*;
use crate::recommendation::MetricSnapshot;
const EVALSET: &str = "abc123";
fn snapshot(field: &str, baseline: f64, higher_is_better: bool) -> MetricSnapshot {
MetricSnapshot {
metric: format!("evalset:{EVALSET}:{field}"),
baseline,
unit: "ratio".into(),
n: 184,
window: "evalset".into(),
subject: None,
namespace: None,
relation: None,
query: String::new(),
review_after_ms: 86_400_000,
horizons_ms: vec![],
higher_is_better,
}
}
fn journal_run(sub: &mut TestSubstrate, run_id: &str, at_ms: i64, extra: serde_json::Value) {
let mut summary = serde_json::json!({"run_id": run_id, "passed": 1, "failed": 0});
if let (Some(o), Some(e)) = (summary.as_object_mut(), extra.as_object()) {
for (k, v) in e {
o.insert(k.clone(), v.clone());
}
}
sub.add_fact_at(
"agent:harness",
&format!("evalset:{EVALSET}"),
"mg:eval_run",
&summary.to_string(),
at_ms,
);
}
fn measure(sub: &TestSubstrate, m: &MetricSnapshot, since: i64) -> Option<f64> {
crate::engine::measure_metric(&sub.inner, m, since).unwrap()
}
#[test]
fn resolves_a_host_field_from_the_newest_run() {
let mut sub = TestSubstrate::new();
journal_run(&mut sub, "eval-1", 1_000, serde_json::json!({"category_accuracy": 0.71}));
journal_run(&mut sub, "eval-2", 2_000, serde_json::json!({"category_accuracy": 0.92}));
let m = snapshot("category_accuracy", 0.71, true);
assert_eq!(measure(&sub, &m, 500), Some(0.92), "the newest run is the current value");
}
#[test]
fn a_run_that_predates_the_apply_is_not_evidence() {
let mut sub = TestSubstrate::new();
journal_run(&mut sub, "eval-1", 1_000, serde_json::json!({"category_accuracy": 0.71}));
let m = snapshot("category_accuracy", 0.71, true);
assert_eq!(
measure(&sub, &m, 5_000),
None,
"no run since the apply means NOT YET MEASURABLE, not 'held'"
);
journal_run(&mut sub, "eval-2", 6_000, serde_json::json!({"category_accuracy": 0.92}));
assert_eq!(measure(&sub, &m, 5_000), Some(0.92));
}
#[test]
fn promoted_fields_work_without_the_host_adding_any() {
let mut sub = TestSubstrate::new();
sub.add_fact_at(
"agent:harness",
&format!("evalset:{EVALSET}"),
"mg:eval_run",
&serde_json::json!({"run_id": "eval-1", "passed": 150, "failed": 34}).to_string(),
1_000,
);
assert_eq!(measure(&sub, &snapshot("failed", 0.0, false), 0), Some(34.0));
assert_eq!(measure(&sub, &snapshot("passed", 0.0, true), 0), Some(150.0));
assert_eq!(measure(&sub, &snapshot("total", 0.0, false), 0), Some(184.0));
let rate = measure(&sub, &snapshot("error_rate", 0.0, false), 0).unwrap();
assert!((rate - 34.0 / 184.0).abs() < 1e-9, "error_rate = failed/total, got {rate}");
}
#[test]
fn degenerate_and_malformed_inputs_measure_nothing_rather_than_guessing() {
let mut sub = TestSubstrate::new();
sub.add_fact_at(
"agent:harness",
&format!("evalset:{EVALSET}"),
"mg:eval_run",
&serde_json::json!({"run_id": "eval-0", "passed": 0, "failed": 0}).to_string(),
1_000,
);
assert_eq!(measure(&sub, &snapshot("error_rate", 0.0, false), 0), None);
assert_eq!(measure(&sub, &snapshot("category_accuracy", 0.0, true), 0), None);
let mut bad = snapshot("x", 0.0, false);
bad.metric = "evalset:onlyhash".into();
assert_eq!(measure(&sub, &bad, 0), None);
let mut other = snapshot("failed", 0.0, false);
other.metric = "evalset:deadbeef:failed".into();
assert_eq!(measure(&sub, &other, 0), None);
}
#[test]
fn a_summary_missing_its_counts_is_dropped_not_defaulted() {
let mut sub = TestSubstrate::new();
sub.add_fact_at(
"agent:harness",
&format!("evalset:{EVALSET}"),
"mg:eval_run",
&serde_json::json!({"run_id": "eval-x", "category_accuracy": 0.99}).to_string(),
1_000,
);
assert_eq!(measure(&sub, &snapshot("category_accuracy", 0.5, true), 0), None);
assert!(crate::eval::eval_runs(&sub.inner, EVALSET, None).unwrap().is_empty());
}
#[test]
fn the_gating_and_outcome_edges_read_the_same_runs() {
let mut sub = TestSubstrate::new();
journal_run(&mut sub, "eval-1", 1_000, serde_json::json!({}));
journal_run(&mut sub, "eval-2", 2_000, serde_json::json!({}));
let by_id = crate::eval::eval_run_by_id(&sub.inner, EVALSET, "eval-1")
.unwrap()
.expect("gating edge finds the named run");
assert_eq!(by_id.run_id, "eval-1");
let newest = crate::eval::newest_eval_run(&sub.inner, EVALSET, None)
.unwrap()
.expect("outcome edge finds the newest run");
assert_eq!(newest.run_id, "eval-2");
assert!(crate::eval::eval_run_by_id(&sub.inner, EVALSET, "eval-nope").unwrap().is_none());
}
}