supercode-harness 0.5.45

The optional native Volter Harness agent and tool harness
Documentation
//! P5-1 (COMPOSABLE-HARNESS-DESIGN.md §2.2 conflict C5, §5.3 risk 1, §4.4
//! oc-parity): oc/cx import translators. When importing/emulating a
//! last-match-wins rule set (opencode's native algebra — oc§4:239-241), this
//! module translates it into the engine's first-match deny→ask→allow form
//! and EMITS A WARNING on every pattern whose translated fixed point differs
//! from the source's — the risk-1 mitigation verbatim ("import-time
//! translators emit warnings on any rule whose translated fixed point
//! differs").
//!
//! **Algorithm.** Last-match-wins over an ordered list `[r1, r2, …, rn]`
//! means: for a given probe, the LAST rule in the list whose pattern matches
//! it determines the outcome. The target engine has no notion of "position"
//! at all — it has three FIXED-PRIORITY tiers (deny, then ask, then allow).
//! There is no general position-preserving translation between the two
//! algebras (§4.4's own gap ledger: "adversarial user rule sets exploiting
//! last-match order … have no first-match equivalent"), so this module does
//! the next best thing, and does it HONESTLY:
//!
//! 1. Resolve each DISTINCT pattern to its own last-match answer (identical
//!    patterns repeated: the last occurrence wins; this is just last-match
//!    applied to the degenerate single-pattern case).
//! 2. Bucket every resolved pattern into the target's deny/ask/allow tier by
//!    that answer — a first, "naive" `RuleSet`.
//! 3. For every DISTINCT pattern, simulate what the ORIGINAL source list
//!    would decide for a probe equal to that pattern (a true last-match
//!    walk, so cross-pattern glob overlap — e.g. a `"*"` rule interacting
//!    with a more specific one — is honored, not just same-literal-pattern
//!    repeats) and what the naive target `RuleSet` from step 2 decides for
//!    the SAME probe (a true first-match deny→ask→allow evaluation).
//! 4. Where they agree: no warning, nothing to fix.
//! 5. Where they disagree: ALWAYS warn (recording the divergence — the
//!    risk-1 mitigation's actual requirement). If the target's answer is
//!    STRICTER (or equal) than the source's, the divergence is safe-direction
//!    (exactly oc-parity's named `.env.example` case, §4.4) and the target's
//!    stricter reading is KEPT. If the target's answer is MORE PERMISSIVE
//!    than the source's — an unsafe divergence — the pattern is forcibly
//!    reassigned to the source's (stricter) tier before returning, so
//!    [`translate_last_match_to_first_match`] can never silently hand back a
//!    ruleset more permissive than the one it was asked to translate.

use std::collections::HashMap;

use super::rules::{Decision, RuleSet};
use crate::config::glob_match;

/// One rule in the SOURCE (last-match-wins) ordering.
#[derive(Debug, Clone)]
pub struct SourceRule {
    /// The pattern text, in this engine's own `tool`/`tool(subject)`/`*`
    /// syntax (see `crate::permissions::rules::RuleSet`'s doc comment) —
    /// this module translates the ALGEBRA (evaluation order), not a
    /// foreign harness's pattern grammar; a caller importing opencode's own
    /// config format is responsible for first rendering its patterns into
    /// this syntax (see [`opencode_default_policy`] for the worked example
    /// this build ships).
    pub pattern: String,
    /// What this rule resolves to when it's the one that matches.
    pub decision: Decision,
}

/// The result of a translation: the target [`RuleSet`] (safe by
/// construction — see the module doc's step 5) plus one warning string per
/// pattern whose fixed point differed from the source, safe or not.
#[derive(Debug, Clone)]
pub struct Translated {
    /// The target first-match deny→ask→allow rule set.
    pub rules: RuleSet,
    /// One entry per pattern whose translated fixed point differed from the
    /// source (the risk-1 mitigation: never silent).
    pub warnings: Vec<String>,
}

/// Translate `source` (last-match-wins order, first rule = lowest priority)
/// into a first-match deny→ask→allow [`RuleSet`] — see the module doc for
/// the algorithm and its safety guarantee.
pub fn translate_last_match_to_first_match(source: &[SourceRule]) -> Translated {
    // Step 1: last-occurrence-wins per distinct pattern, order-preserving
    // for readability (not semantically load-bearing — tier bucketing
    // doesn't care about relative order within a tier, C5).
    let mut order: Vec<String> = Vec::new();
    let mut last: HashMap<String, Decision> = HashMap::new();
    for r in source {
        if !last.contains_key(&r.pattern) {
            order.push(r.pattern.clone());
        }
        last.insert(r.pattern.clone(), r.decision);
    }

    // Step 2: naive bucketing by each pattern's own resolved decision.
    let mut rules = RuleSet::default();
    for pattern in &order {
        match last[pattern] {
            Decision::Deny => rules.deny.push(pattern.clone()),
            Decision::Ask => rules.ask.push(pattern.clone()),
            Decision::Allow => rules.allow.push(pattern.clone()),
        }
    }

    // Step 3-5: fixed-point comparison per distinct pattern, probing with
    // the pattern's own text as a stand-in command/subject (the same
    // "self-probe" a golden-vector suite would use to pin one named
    // pattern's behavior) — good enough to catch genuine cross-pattern
    // overlap (e.g. an earlier `"*"` vs a later specific rule) without
    // requiring the caller to hand this generic algorithm a full sample
    // corpus of real commands.
    let mut warnings = Vec::new();
    for pattern in &order {
        let probe = pattern.as_str();
        let source_decision = simulate_last_match(source, probe);
        let target_decision = evaluate_self(&rules, pattern);
        let (Some(sd), Some(td)) = (source_decision, target_decision) else {
            continue;
        };
        if sd != td {
            let safe = sd.stricter(td);
            if safe != td {
                // Target was MORE PERMISSIVE than source — an unsafe
                // divergence. Force the pattern into the safe tier.
                reassign(&mut rules, pattern, safe);
            }
            warnings.push(format!(
                "C5 translation: pattern `{pattern}` — source (last-match) resolves to \
                 {sd:?}, first-match-translated resolves to {td:?}; kept {safe:?} \
                 ({} divergence)",
                if safe == td {
                    "safe-direction"
                } else {
                    "unsafe, corrected"
                }
            ));
        }
    }

    Translated { rules, warnings }
}

/// True last-match walk of the ORIGINAL source list for `probe` — the last
/// rule (by source position) whose pattern glob-matches `probe`'s own text
/// wins. This is the semantics [`translate_last_match_to_first_match`]
/// exists to reproduce (subject to the target engine's tier-priority
/// limitation).
fn simulate_last_match(source: &[SourceRule], probe: &str) -> Option<Decision> {
    let mut result = None;
    for r in source {
        if glob_match(&r.pattern, probe) || r.pattern == probe {
            result = Some(r.decision);
        }
    }
    result
}

/// Self-referential fixed-point probe: does `pattern`, considered as its own
/// subject, resolve under `rules`' deny→ask→allow first-match evaluation?
/// Tries the bare-glob form first (bare tool-name-style patterns like
/// `"question"`/`"plan_enter"`), then the `tool(subject)` form using
/// `pattern`'s own tool-prefix against its own command-glob, so a pattern
/// like `"read(*.env)"` can be asked "does `read(*.env)` match itself" in a
/// way that actually exercises the tool+subject matcher.
fn evaluate_self(rules: &RuleSet, pattern: &str) -> Option<Decision> {
    if let Some(open) = pattern.find('(') {
        if let Some(subject) = pattern.strip_suffix(')').and_then(|p| p.get(open + 1..)) {
            let tool = &pattern[..open];
            return rules.evaluate(tool, Some(subject));
        }
    }
    // A bare tool-name-shaped pattern (including the literal wildcard
    // `"*"`): probe with the pattern text AS the tool name, no subject —
    // this exercises bare-glob rules (`"tool"`/`"tool*"`/`"*"`) correctly.
    // `tool(cmdglob)`-shaped rules elsewhere in `rules` never match a
    // `None` subject (`rule_matches`'s contract), so they cannot spuriously
    // fire here.
    rules.evaluate(pattern, None)
}

/// Remove `pattern` from every tier of `rules` and reinsert it in
/// `target_tier` — the step-5 safety correction.
fn reassign(rules: &mut RuleSet, pattern: &str, target_tier: Decision) {
    rules.deny.retain(|p| p != pattern);
    rules.ask.retain(|p| p != pattern);
    rules.allow.retain(|p| p != pattern);
    match target_tier {
        Decision::Deny => rules.deny.push(pattern.to_string()),
        Decision::Ask => rules.ask.push(pattern.to_string()),
        Decision::Allow => rules.allow.push(pattern.to_string()),
    }
}

/// opencode's documented DEFAULT policy (oc§4 "Default policy": `{"*":
/// allow}` with carve-outs `doom_loop: ask`, `external_directory: ask`,
/// `question: deny`, `plan_enter`/`plan_exit: deny`, `read {*.env: ask,
/// *.env.*: ask, *.env.example: allow}`), rendered into this engine's
/// pattern syntax and run through [`translate_last_match_to_first_match`] —
/// the worked example design §4.4 specifies and this build reproduces.
///
/// **Two of oc's five carve-outs are deliberately EXCLUDED from the rule
/// list itself** (§4.4's own text, reproduced here as the three named
/// deviations):
///
/// 1. `doom_loop: ask` is not a rule-language pattern at all — it's a
///    repetition TRIGGER (same call repeated), not a tool/path match.
///    Routed to its real mechanism instead: `Config::doom_loop_threshold`
///    (the P4 doom-loop breaker). No rule entry; a warning names the
///    routing decision explicitly (never silently dropped).
/// 2. `external_directory: ask` is an oc PERMISSION CATEGORY (any tool
///    touching paths outside the worktree), not a tool name. Routed to its
///    real mechanism: `Config::additional_dirs` — paths outside `cwd` and
///    outside `additional_dirs` are simply unreachable, a stricter (not
///    equivalent) reading. No rule entry; a warning names the routing
///    decision.
/// 3. `.env.example` → **ASK, not ALLOW** — this one DOES fall out of the
///    generic algorithm above (a genuine, detected, safe-direction fixed-
///    point divergence): under first-match deny→ask→allow, a read of
///    `.env.example` matches the ask-rule `read(*.env.*)` BEFORE the allow
///    list (`"*"`) is ever consulted, so it asks where stock opencode
///    allows. Recorded by [`translate_last_match_to_first_match`]'s own
///    warning, not hidden.
pub fn opencode_default_policy() -> Translated {
    let source = vec![
        SourceRule {
            pattern: "*".to_string(),
            decision: Decision::Allow,
        },
        SourceRule {
            pattern: "tools_question".to_string(),
            decision: Decision::Deny,
        },
        SourceRule {
            pattern: "plan_enter".to_string(),
            decision: Decision::Deny,
        },
        SourceRule {
            pattern: "plan_exit".to_string(),
            decision: Decision::Deny,
        },
        SourceRule {
            pattern: "read(*.env)".to_string(),
            decision: Decision::Ask,
        },
        SourceRule {
            pattern: "read(*.env.*)".to_string(),
            decision: Decision::Ask,
        },
        SourceRule {
            pattern: "read(*.env.example)".to_string(),
            decision: Decision::Allow,
        },
    ];
    let mut translated = translate_last_match_to_first_match(&source);
    translated.warnings.push(
        "C5/S4 deviation 2 (doom_loop): opencode's `doom_loop: ask` carve-out is a repetition \
         TRIGGER, not a rule-language pattern — routed to `Config::doom_loop_threshold` (the P4 \
         doom-loop breaker) instead of a rule entry; not silently dropped."
            .to_string(),
    );
    translated.warnings.push(
        "C5/S4 deviation 3 (external_directory): opencode's `external_directory: ask` carve-out \
         is a PERMISSION CATEGORY (any tool touching paths outside the worktree), not a tool \
         name — routed to `Config::additional_dirs` instead of a rule entry (paths outside cwd \
         and outside additional_dirs are simply unreachable, a stricter reading); not silently \
         dropped."
            .to_string(),
    );
    translated
}