Skip to main content

supercode_runtime/
tokens.rs

1//! The shared token estimator (SPEC.md C9). The repo has no tokenizer — only
2//! provider-reported `Usage` (`provider.rs:153-164`) and the turn/output
3//! counters `agent.rs` already tracks. Every other UX figure (the banner,
4//! `/status`, `/tokens`, `show-reductions`, notices) derives from one
5//! documented heuristic here, and is always printed with a `~` prefix so it
6//! reads as an estimate, never a measurement. Acceptance criteria never
7//! assert against these numbers directly — ACs assert exact byte counts
8//! (`SessionInfo::full_bytes`/`view_bytes`, C9/D15) and only *derive* a
9//! token/dollar figure from them for the test log.
10//!
11//! Provider-reported figures (`Usage.completion_tokens`, B7's optional
12//! `cached_tokens`, `agent.total_output_tokens()`) are real counts and are
13//! never routed through this module — they print un-tilded.
14
15use supercode_interchange::{format_commas, ChatMessage};
16
17use crate::ToolSchema;
18pub use supercode_interchange::{estimate_tokens, estimate_view_tokens};
19
20/// Deterministic token estimate: `ceil(utf8_bytes / 4)`. Documented
21/// heuristic — all UX figures derived from it are printed with a `~`
22/// prefix (see [`fmt_approx_tokens`]). Never used in acceptance criteria
23/// (ACs assert exact byte counts).
24/// PARITY-18 D2 — conservative safety margin folded into the context-guard
25/// boundary only (never into the plain `~`-prefixed UX estimates
26/// themselves, which stay the documented `ceil(bytes/4)` heuristic
27/// unmodified). `ceil(utf8_bytes/4)` under-counts real tokenizer output on
28/// CJK text, base64/binary-ish blobs, and dense code — informally by 25%+
29/// against common tokenizers for those corpora, since multi-byte UTF-8
30/// sequences and non-whitespace-delimited runs pack more real tokens per
31/// byte than the heuristic assumes. 25% is a round number comfortably above
32/// that observed skew: applying it can only make the guard MORE
33/// conservative (refuse sooner), never let an over-context request through
34/// that a real tokenizer would also have refused.
35const GUARD_MARGIN_NUM: u64 = 5;
36const GUARD_MARGIN_DEN: u64 = 4;
37
38/// Apply the runtime's 5/4 context-guard safety margin to a raw token
39/// estimate. Rounds up (`div_ceil`), never down —
40/// the margin only ever pushes the boundary check to be more cautious.
41pub fn with_guard_margin(tokens: u64) -> u64 {
42    tokens
43        .saturating_mul(GUARD_MARGIN_NUM)
44        .div_ceil(GUARD_MARGIN_DEN)
45}
46
47/// PARITY-18 — headroom reserved for the model's own completion, folded
48/// into the context-guard boundary alongside [`with_guard_margin`]. Neither
49/// [`estimate_view_tokens`] nor [`estimate_request_tokens`] counts anything
50/// for the reply the model is about to generate — this is a flat token
51/// budget carved out of the model's context window for it, since the
52/// completion shares the same window as the request on every provider this
53/// crate targets.
54///
55/// PARITY-18 v3 NOTE-AND-DECIDE (NF6, owner-recorded, not fixed here): this
56/// is a FLAT reserve — it does not read `Config::max_tokens`
57/// (`crates/harness/src/config.rs`, user-settable via `--max-tokens`, wired
58/// into the actual provider request at `provider.rs`'s
59/// `ChatRequest::max_tokens`). A user who passes `--max-tokens` greater than
60/// 16,384 can still pass this guard (`projected + 16_384 <= context_limit`)
61/// and then draw a provider-side `input_tokens + max_tokens > context_window`
62/// rejection the guard never anticipated — i.e. the guard's margin can be
63/// smaller than what the user actually asked the provider to reserve for the
64/// completion. Making the reserve `max(CONTEXT_RESPONSE_RESERVE_TOKENS,
65/// config.max_tokens)` would close this, but `context_guard` doesn't
66/// currently receive `Config` at all (only `messages`/`tools`/
67/// `context_limit`) — threading it through is a small but real signature
68/// change touching every call site (`resume_cmd`, `Agent::run_loop`, and
69/// this pass's new `reduce_to_fit` `fits` closures) that's out of scope for
70/// this pass's reducer/guard boundary fix. Recorded for the owner.
71pub const CONTEXT_RESPONSE_RESERVE_TOKENS: u64 = 16_384;
72
73/// PARITY-18 D1 — the full wire-request token estimate: every message in
74/// `messages` (including the system prompt at index 0) plus the serialized
75/// `tools` schema array, which is a real part of the provider request but — before
76/// PARITY-18's re-fix — was never counted by the preflight guard at all.
77/// A session whose messages alone fit comfortably could still carry a fat
78/// builtin/MCP tool-schema array that blows the real wire request; this is
79/// the fix.
80pub fn estimate_request_tokens(messages: &[ChatMessage], tools: &[ToolSchema]) -> u64 {
81    let tools_wire = serde_json::to_string(tools).unwrap_or_default();
82    estimate_view_tokens(messages).saturating_add(estimate_tokens(&tools_wire))
83}
84
85/// PARITY-18 D1/D2/D4 — the single context-guard decision, shared by every
86/// call site that must decide whether a request is safe to send: the CLI's
87/// `resume_cmd` preflight check AND `Agent::run_loop`'s per-send check
88/// (D4 — the guard is a session invariant, not a one-shot preflight, so
89/// turn 2+ and `/expand all` are covered too). Because both call through
90/// this one function, a request can never pass one gate and fail the
91/// other — there is only one formula.
92///
93/// `fits` is true iff the [`with_guard_margin`]-adjusted
94/// [`estimate_request_tokens`] estimate, plus the
95/// [`CONTEXT_RESPONSE_RESERVE_TOKENS`] completion reserve, is still within
96/// `context_limit` — i.e. gates on the reduce TARGET
97/// (`context_limit - CONTEXT_RESPONSE_RESERVE_TOKENS`), not the raw limit,
98/// closing the "blind band between reduce target and pass/fail boundary"
99/// gap. Returns the margin-adjusted projected total either way so callers
100/// can report it (dev/02) regardless of verdict.
101pub fn context_guard(
102    messages: &[ChatMessage],
103    tools: &[ToolSchema],
104    context_limit: u64,
105) -> (bool, u64) {
106    let raw = estimate_request_tokens(messages, tools);
107    let projected = with_guard_margin(raw);
108    let fits = projected.saturating_add(CONTEXT_RESPONSE_RESERVE_TOKENS) <= context_limit;
109    (fits, projected)
110}
111
112/// Schema-token estimate for the B6 "tools" banner line: the estimate over
113/// the serialized `full` [`ToolSchema`] list minus the estimate over the
114/// serialized `advertised` list — i.e. the token cost of what's currently
115/// deferred (hidden behind `tool_search`) rather than eagerly advertised.
116/// Saturates to `0` rather than underflow if `advertised` somehow estimates
117/// larger than `full` (e.g. formatting differences), since "negative
118/// deferred tokens" has no meaning for the banner.
119pub fn estimate_deferred_schema_tokens(full: &[ToolSchema], advertised: &[ToolSchema]) -> u64 {
120    let full_tokens = estimate_tokens(&serde_json::to_string(full).unwrap_or_default());
121    let advertised_tokens = estimate_tokens(&serde_json::to_string(advertised).unwrap_or_default());
122    full_tokens.saturating_sub(advertised_tokens)
123}
124
125/// Render an estimated token count in the shared UX style: `~21,904 tok`
126/// (tilde prefix + comma-grouped thousands, matching the stub-line comma
127/// style in `reduce.rs`). Every figure that flows through
128/// [`estimate_tokens`]/[`estimate_view_tokens`]/[`estimate_deferred_schema_tokens`]
129/// should be rendered through this helper so the `~` discipline (D11) is
130/// applied uniformly rather than ad hoc at each call site.
131pub fn fmt_approx_tokens(n: u64) -> String {
132    format!("~{} tok", format_commas(n as usize))
133}