harn-stdlib 0.10.148

Embedded Harn standard library source catalog
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
import "std/agent/loop_call_resolution"
import "std/agent/loop_support"

/**
 * Route agent-loop budget timing through the harness clock for deterministic replay and tests.
 *
 * @effects: [agent]
 * @errors: []
 */
pub fn __agent_loop_clock_now_ms(clock: HarnessClock) {
  return clock.monotonic_ms()
}

pub fn __agent_loop_clock_sleep_ms(clock: HarnessClock, delay_ms: any) {
  clock.sleep_ms(delay_ms)
}

pub type AgentLoopDeadlineContract = {
  deadline_at_ms: int,
  duration_ms: int,
  source: "deadline_ms" | "iteration_budget.wall_clock_ms",
}

/**
 * Resolve the earliest configured loop deadline in the caller's monotonic clock domain.
 *
 * @effects: []
 * @errors: []
 */
pub fn __agent_loop_deadline_contract(
  now_ms: int,
  deadline_ms: int?,
  wall_clock_ms: int?,
) -> AgentLoopDeadlineContract? {
  const configured = to_int(deadline_ms)
  const budget = to_int(wall_clock_ms)
  const configured_valid = configured != nil && configured > 0
  const budget_valid = budget != nil && budget > 0
  if !configured_valid && !budget_valid {
    return nil
  }
  const budget_is_earliest = budget_valid && (!configured_valid || budget < configured)
  const duration_ms = if budget_is_earliest {
    budget
  } else {
    configured
  }
  const source = if budget_is_earliest {
    "iteration_budget.wall_clock_ms"
  } else {
    "deadline_ms"
  }
  return {deadline_at_ms: to_int(now_ms) + duration_ms, duration_ms: duration_ms, source: source}
}

pub fn __agent_loop_apply_deadline(opts: dict, budget: dict, now_ms: int) -> dict {
  const contract = __agent_loop_deadline_contract(now_ms, opts?.deadline_ms, budget?.wall_clock_ms)
  return if contract == nil {
    opts + {_deadline_at_ms: nil, _deadline_source: nil}
  } else {
    opts + {_deadline_at_ms: contract.deadline_at_ms, _deadline_source: contract.source}
  }
}

pub fn __agent_loop_deadline_reached(opts: dict, clock: HarnessClock) -> bool {
  return opts?._deadline_source == "deadline_ms"
    && opts?._deadline_at_ms != nil
    && __agent_loop_clock_now_ms(clock) >= opts._deadline_at_ms
}

pub const __AGENT_LOOP_CHECKPOINT_ITERATION_START = "iteration_start"

pub const __AGENT_LOOP_CHECKPOINT_POST_TOOL_DISPATCH = "post_tool_dispatch"

pub const __AGENT_LOOP_HOST_INJECTION_TURN_BOUNDARY = "turn_boundary"

pub const __AGENT_LOOP_HOST_INJECTION_AFTER_NEXT_TOOL_CALL = "after_next_tool_call"

// -------------------------------------------------------------------------------------------------

// Length-truncation auto-continue.
//
// When a value model hits the output-token cap mid-emit, the provider returns a
// length-truncation stop_reason (`length` for OpenAI/OpenRouter/Ollama,
// `max_tokens` for Anthropic) and the partial output often holds a TRUNCATED,
// unparseable tool call. Treating that as a malformed/missing call burns the
// turn on parse-guidance even though the model was mid-correct-action — a
// silent-corruption class that hurts even capable models.
//
// Instead we detect the specific condition deterministically (no model
// cooperation, no abuse surface) and re-issue the completion with a RAISED cap
// so the model can finish the call. The retry is invisible to the rest of the
// loop body: it does not consume an iteration or run stall/parse accounting.
// Bounded to a small number of continuations; if still truncated after the cap
// we return the last result and the existing parse-guidance path takes over.
//
// The tool-call gate (`__host_agent_truncated_tool_call`) fires ONLY on a real
// length truncation with zero usable calls AND a partial-call signal. A second
// gate covers hidden-reasoning-only truncation: strict providers can return no
// visible text/tool calls, `stop_reason: length`, and non-empty reasoning. That
// is still a budget exhaustion, so retry with a larger output cap instead of
// handing an empty turn to the normal loop body.

// -------------------------------------------------------------------------------------------------

pub fn __autocontinue_max_continuations(llm_opts: dict) {
  const configured = llm_opts?.truncation_auto_continue_max
  if type_of(configured) == "int" && configured >= 0 {
    return configured
  }
  return 2
}

/**
 * Compute the raised output-token cap for an auto-continue retry. An unset cap
 * (`<= 0`) means the provider's own default truncated us, so jump to an
 * explicit base. A set cap doubles. Both are clamped to a ceiling so a
 * misconfigured model can't request an unbounded body.
 *
 * @effects: []
 * @errors: []
 */
pub fn __autocontinue_raised_cap(current: int, llm_opts: dict) {
  const base = if type_of(llm_opts?.truncation_auto_continue_base) == "int"
    && llm_opts.truncation_auto_continue_base
      > 0 {
    llm_opts.truncation_auto_continue_base
  } else {
    16384
  }
  const ceiling = if type_of(llm_opts?.truncation_auto_continue_ceiling) == "int"
    && llm_opts.truncation_auto_continue_ceiling
      > 0 {
    llm_opts.truncation_auto_continue_ceiling
  } else {
    32768
  }
  const raised = if type_of(current) == "int" && current > 0 {
    current * 2
  } else {
    base
  }
  if raised > ceiling {
    return ceiling
  }
  return raised
}

pub fn __autocontinue_is_length_stop(stop_reason: string?) {
  const normalized = lowercase(trim(to_string(stop_reason ?? "")))
  return normalized == "length" || normalized == "max_tokens"
}

pub fn __autocontinue_hidden_reasoning_truncated(llm_result: dict, tool_call_count: int) {
  if !__autocontinue_is_length_stop(llm_result?.stop_reason) {
    return false
  }
  if tool_call_count > 0 {
    return false
  }
  if trim(to_string(llm_result?.text ?? "")) != "" {
    return false
  }
  const thinking = trim(to_string(llm_result?.thinking ?? ""))
  const summary = trim(to_string(llm_result?.thinking_summary ?? ""))
  if thinking != "" || summary != "" {
    return true
  }
  return (to_int(llm_result?.usage?.output_tokens ?? 0) ?? 0) > 0
}

/**
 * Detect whether `call` is a length-truncated turn that resolved no usable
 * tool call but looks like it was mid-call — the one condition where
 * re-issuing with a raised cap is the right move.
 *
 * @effects: [host]
 * @errors: []
 */
pub fn __autocontinue_should_continue(agent: HarnessAgent, call: dict, turn_opts: dict) {
  if !(call?.ok ?? false) {
    return false
  }
  const llm_result = call?.value
  if type_of(llm_result) != "dict" {
    return false
  }
  const raw_text = llm_result?.raw_text ?? llm_result?.text ?? ""
  const parsed = agent_parse_tool_calls(agent, raw_text, turn_opts?.tools, turn_opts?.tool_format)
  const tool_calls = __resolve_tool_calls(llm_result, parsed)
  const has_parse_errors = len(parsed?.tool_parse_errors ?? []) > 0
  if __host_agent_truncated_tool_call(
    llm_result?.stop_reason,
    raw_text,
    len(tool_calls),
    has_parse_errors,
  ) {
    return true
  }
  return __autocontinue_hidden_reasoning_truncated(llm_result, len(tool_calls))
}

/**
 * Call the LLM, and on a length-truncated turn with an incomplete tool call,
 * AUTO-CONTINUE with a bounded raised output cap, then return the same
 * `{ok, value, ...}` shape callers already parse and account for. Falls back
 * to the truncated result at the cap so existing parse guidance still fires.
 *
 * @effects: []
 * @errors: []
 */
pub fn __invoke_llm_with_autocontinue(
  harness: Harness,
  message: any,
  turn_system: any,
  llm_opts: dict,
  turn_opts: dict,
  session_id: any,
  iteration_index: int,
) {
  const retry_llm_opts = llm_clear_one_shot_prefill(llm_opts)
  agent_session_flush(harness.agent, session_id)
  let call = agent_invoke_llm(
    harness,
    message,
    turn_system,
    __agent_loop_provider_request_options(llm_opts),
  )
  const max_continuations = __autocontinue_max_continuations(llm_opts)
  let attempts = 0
  let cap = llm_opts?.max_tokens ?? 0
  while attempts < max_continuations
    && __autocontinue_should_continue(harness.agent, call, turn_opts) {
    const raised = __autocontinue_raised_cap(cap, llm_opts)
    // No headroom left to raise — re-issuing would just truncate again at the
    // same cap, so stop and let parse-guidance take the turn.
    if type_of(cap) == "int" && cap > 0 && raised <= cap {
      break
    }
    attempts = attempts + 1
    agent_emit_event(
      harness.agent,
      session_id,
      "llm_auto_continue",
      {
        iteration: iteration_index + 1,
        attempt: attempts,
        max_continuations: max_continuations,
        previous_max_tokens: cap,
        raised_max_tokens: raised,
        stop_reason: call?.value?.stop_reason ?? "",
      },
    )
    cap = raised
    const retry_opts = retry_llm_opts
      + {max_tokens: raised, _truncation_auto_continue_attempt: attempts}
    call = agent_invoke_llm(
      harness,
      message,
      turn_system,
      __agent_loop_provider_request_options(retry_opts),
    )
  }
  return call
}

/**
 * Dispatch one main-loop logical request after pre-turn controls have passed.
 * The claim receipt lives here so it fires once for the request lifecycle,
 * outside transport auto-continue and context-overflow retries.
 *
 * @effects: [agent, llm]
 * @errors: []
 */
pub fn __agent_loop_invoke_main_request(
  harness: Harness,
  message: any,
  turn_system: any,
  llm_opts: dict,
  turn_opts: dict,
  session_id: any,
  iteration_index: int,
) -> dict {
  __agent_loop_next_tool_claim_receipt(harness.agent, session_id, llm_opts, iteration_index)
  // The main request is the turn a person is waiting to read, so its visible
  // text streams to the host as it arrives. Side calls (classifiers, judges,
  // titles) leave this unset and stream nothing visible.
  return __invoke_llm_with_autocontinue(
    harness,
    message,
    turn_system,
    llm_opts + {_user_visible: true},
    turn_opts,
    session_id,
    iteration_index,
  )
}

/**
 * Context-overflow recovery.
 *
 * A provider can reject a turn with a `context_overflow` error when the
 * assembled prompt exceeds the model's real context window — typically on a
 * large repo where tool observations accreted past the budget, OR when the
 * model's window is mis/under-cataloged so auto-compaction never fired. The
 * agent must NOT die on this: it is a recoverable, self-inflicted condition.
 *
 * Recovery is a bounded loop: emergency-compact the live transcript
 * (deterministic observation masking — never an LLM call, which would itself
 * overflow), then re-issue the SAME turn. Each attempt compacts more
 * aggressively (shorter preserved tail). We stop when either the retry
 * succeeds, or emergency compaction can no longer shrink the transcript
 * (`archived == 0`, i.e. an irreducible system-prompt + single oversized
 * message), at which point the overflow is genuinely terminal.
 *
 * @effects: [agent]
 * @errors: []
 */
pub fn __agent_loop_is_context_overflow(error: unknown) {
  if type_of(error) != "dict" {
    return false
  }
  const reason = to_string(error?.reason ?? "")
  if reason == "context_overflow" {
    return true
  }
  // Defensive fallback: some routes surface the condition only in the message
  // text (uncataloged provider, no structured `reason`). Match the canonical
  // classifier tag the Rust LLM layer stamps onto the error string.
  const message = to_string(error?.message ?? "")
  return contains(message, "[context_overflow]")
}

pub fn __agent_loop_context_overflow_max_recoveries(llm_opts: dict) {
  const configured = llm_opts?.context_overflow_recover_max
  if type_of(configured) == "int" && configured >= 0 {
    return configured
  }
  return 3
}

/**
 * On a `context_overflow` provider error, emergency-compact + retry (bounded).
 * Returns a fresh `{ok, ...}` call result: `ok:true` when a retry succeeded,
 * or the last `ok:false` result (so the caller's terminal path runs unchanged)
 * when recovery is exhausted or the transcript is irreducible.
 *
 * @effects: [agent]
 * @errors: []
 */
pub fn __agent_loop_recover_context_overflow(
  harness: Harness,
  call: any,
  message: any,
  turn_system: unknown,
  llm_opts: any,
  turn_opts: dict,
  session: dict,
  iteration_index: int,
) {
  const max_recoveries = __agent_loop_context_overflow_max_recoveries(llm_opts)
  let current = call
  let attempt = 0
  while attempt < max_recoveries && !current.ok
    && __agent_loop_is_context_overflow(current?.error) {
    attempt = attempt + 1
    const archived = agent_emergency_compact(harness.agent, session, llm_opts, attempt)
    agent_emit_event(
      harness.agent,
      session.session_id,
      "context_overflow_recovery",
      {
        iteration: iteration_index + 1,
        attempt: attempt,
        max_recoveries: max_recoveries,
        archived_messages: archived,
        provider_error: current?.error ?? {},
      },
    )
    // Nothing left to shed: the prompt is irreducible (system prompt + a single
    // oversized message). Re-issuing would overflow identically, so surface the
    // original terminal error rather than spin.
    if archived <= 0 {
      break
    }
    // Rebuild the turn prompt against the now-compacted transcript and re-issue.
    const turn_prompt = __agent_loop_build_turn_prompt(harness, session, llm_opts, iteration_index)
    const retry_opts = llm_clear_one_shot_prefill(llm_opts)
      + {
        messages: turn_prompt.messages,
        _system_fragments: turn_prompt.fragments,
        _context_overflow_recovery_attempt: attempt,
      }
    current = __invoke_llm_with_autocontinue(
      harness,
      message,
      turn_prompt.system,
      retry_opts,
      turn_opts,
      session.session_id,
      iteration_index,
    )
  }
  return current
}