supercode-runtime 0.5.49

Optional native model and tool runtime for Volter Harness
Documentation
//! The shared token estimator (SPEC.md C9). The repo has no tokenizer — only
//! provider-reported `Usage` (`provider.rs:153-164`) and the turn/output
//! counters `agent.rs` already tracks. Every other UX figure (the banner,
//! `/status`, `/tokens`, `show-reductions`, notices) derives from one
//! documented heuristic here, and is always printed with a `~` prefix so it
//! reads as an estimate, never a measurement. Acceptance criteria never
//! assert against these numbers directly — ACs assert exact byte counts
//! (`SessionInfo::full_bytes`/`view_bytes`, C9/D15) and only *derive* a
//! token/dollar figure from them for the test log.
//!
//! Provider-reported figures (`Usage.completion_tokens`, B7's optional
//! `cached_tokens`, `agent.total_output_tokens()`) are real counts and are
//! never routed through this module — they print un-tilded.

use supercode_interchange::{format_commas, ChatMessage};

use crate::ToolSchema;
pub use supercode_interchange::{estimate_tokens, estimate_view_tokens};

/// Deterministic token estimate: `ceil(utf8_bytes / 4)`. Documented
/// heuristic — all UX figures derived from it are printed with a `~`
/// prefix (see [`fmt_approx_tokens`]). Never used in acceptance criteria
/// (ACs assert exact byte counts).
/// PARITY-18 D2 — conservative safety margin folded into the context-guard
/// boundary only (never into the plain `~`-prefixed UX estimates
/// themselves, which stay the documented `ceil(bytes/4)` heuristic
/// unmodified). `ceil(utf8_bytes/4)` under-counts real tokenizer output on
/// CJK text, base64/binary-ish blobs, and dense code — informally by 25%+
/// against common tokenizers for those corpora, since multi-byte UTF-8
/// sequences and non-whitespace-delimited runs pack more real tokens per
/// byte than the heuristic assumes. 25% is a round number comfortably above
/// that observed skew: applying it can only make the guard MORE
/// conservative (refuse sooner), never let an over-context request through
/// that a real tokenizer would also have refused.
const GUARD_MARGIN_NUM: u64 = 5;
const GUARD_MARGIN_DEN: u64 = 4;

/// Apply the runtime's 5/4 context-guard safety margin to a raw token
/// estimate. Rounds up (`div_ceil`), never down —
/// the margin only ever pushes the boundary check to be more cautious.
pub fn with_guard_margin(tokens: u64) -> u64 {
    tokens
        .saturating_mul(GUARD_MARGIN_NUM)
        .div_ceil(GUARD_MARGIN_DEN)
}

/// PARITY-18 — headroom reserved for the model's own completion, folded
/// into the context-guard boundary alongside [`with_guard_margin`]. Neither
/// [`estimate_view_tokens`] nor [`estimate_request_tokens`] counts anything
/// for the reply the model is about to generate — this is a flat token
/// budget carved out of the model's context window for it, since the
/// completion shares the same window as the request on every provider this
/// crate targets.
///
/// PARITY-18 v3 NOTE-AND-DECIDE (NF6, owner-recorded, not fixed here): this
/// is a FLAT reserve — it does not read `Config::max_tokens`
/// (`crates/harness/src/config.rs`, user-settable via `--max-tokens`, wired
/// into the actual provider request at `provider.rs`'s
/// `ChatRequest::max_tokens`). A user who passes `--max-tokens` greater than
/// 16,384 can still pass this guard (`projected + 16_384 <= context_limit`)
/// and then draw a provider-side `input_tokens + max_tokens > context_window`
/// rejection the guard never anticipated — i.e. the guard's margin can be
/// smaller than what the user actually asked the provider to reserve for the
/// completion. Making the reserve `max(CONTEXT_RESPONSE_RESERVE_TOKENS,
/// config.max_tokens)` would close this, but `context_guard` doesn't
/// currently receive `Config` at all (only `messages`/`tools`/
/// `context_limit`) — threading it through is a small but real signature
/// change touching every call site (`resume_cmd`, `Agent::run_loop`, and
/// this pass's new `reduce_to_fit` `fits` closures) that's out of scope for
/// this pass's reducer/guard boundary fix. Recorded for the owner.
pub const CONTEXT_RESPONSE_RESERVE_TOKENS: u64 = 16_384;

/// PARITY-18 D1 — the full wire-request token estimate: every message in
/// `messages` (including the system prompt at index 0) plus the serialized
/// `tools` schema array, which is a real part of the provider request but — before
/// PARITY-18's re-fix — was never counted by the preflight guard at all.
/// A session whose messages alone fit comfortably could still carry a fat
/// builtin/MCP tool-schema array that blows the real wire request; this is
/// the fix.
pub fn estimate_request_tokens(messages: &[ChatMessage], tools: &[ToolSchema]) -> u64 {
    let tools_wire = serde_json::to_string(tools).unwrap_or_default();
    estimate_view_tokens(messages).saturating_add(estimate_tokens(&tools_wire))
}

/// PARITY-18 D1/D2/D4 — the single context-guard decision, shared by every
/// call site that must decide whether a request is safe to send: the CLI's
/// `resume_cmd` preflight check AND `Agent::run_loop`'s per-send check
/// (D4 — the guard is a session invariant, not a one-shot preflight, so
/// turn 2+ and `/expand all` are covered too). Because both call through
/// this one function, a request can never pass one gate and fail the
/// other — there is only one formula.
///
/// `fits` is true iff the [`with_guard_margin`]-adjusted
/// [`estimate_request_tokens`] estimate, plus the
/// [`CONTEXT_RESPONSE_RESERVE_TOKENS`] completion reserve, is still within
/// `context_limit` — i.e. gates on the reduce TARGET
/// (`context_limit - CONTEXT_RESPONSE_RESERVE_TOKENS`), not the raw limit,
/// closing the "blind band between reduce target and pass/fail boundary"
/// gap. Returns the margin-adjusted projected total either way so callers
/// can report it (dev/02) regardless of verdict.
pub fn context_guard(
    messages: &[ChatMessage],
    tools: &[ToolSchema],
    context_limit: u64,
) -> (bool, u64) {
    let raw = estimate_request_tokens(messages, tools);
    let projected = with_guard_margin(raw);
    let fits = projected.saturating_add(CONTEXT_RESPONSE_RESERVE_TOKENS) <= context_limit;
    (fits, projected)
}

/// Schema-token estimate for the B6 "tools" banner line: the estimate over
/// the serialized `full` [`ToolSchema`] list minus the estimate over the
/// serialized `advertised` list — i.e. the token cost of what's currently
/// deferred (hidden behind `tool_search`) rather than eagerly advertised.
/// Saturates to `0` rather than underflow if `advertised` somehow estimates
/// larger than `full` (e.g. formatting differences), since "negative
/// deferred tokens" has no meaning for the banner.
pub fn estimate_deferred_schema_tokens(full: &[ToolSchema], advertised: &[ToolSchema]) -> u64 {
    let full_tokens = estimate_tokens(&serde_json::to_string(full).unwrap_or_default());
    let advertised_tokens = estimate_tokens(&serde_json::to_string(advertised).unwrap_or_default());
    full_tokens.saturating_sub(advertised_tokens)
}

/// Render an estimated token count in the shared UX style: `~21,904 tok`
/// (tilde prefix + comma-grouped thousands, matching the stub-line comma
/// style in `reduce.rs`). Every figure that flows through
/// [`estimate_tokens`]/[`estimate_view_tokens`]/[`estimate_deferred_schema_tokens`]
/// should be rendered through this helper so the `~` discipline (D11) is
/// applied uniformly rather than ad hoc at each call site.
pub fn fmt_approx_tokens(n: u64) -> String {
    format!("~{} tok", format_commas(n as usize))
}