harn-stdlib 0.10.147

Embedded Harn standard library source catalog
Documentation
import { __judge_classify_verdict, __judge_run_checkpoint } from "std/agent/judge_internals"
import { step_judge_system_prompt, step_judge_user_prompt } from "std/agent/prompts"
import { agent_emit_event, agent_session_messages } from "std/agent/state"
import { estimate_call_cost } from "std/llm/economics"

fn __step_judge_schema() {
  return {
    type: "object",
    properties: {
      verdict: {
        type: "string",
        enum: ["pass", "revise"],
        description:
          "Use `pass` if the response advances the task acceptably, `revise` to request regeneration.",
      },
      reasoning: {type: "string"},
      critique: {type: "string"},
      confidence: {type: "number"},
    },
    required: ["verdict"],
  }
}

fn __serialize_response(llm_result: dict?) {
  const {text = "", tool_calls = []} = llm_result ?? {}
  let parts = []
  if text != "" {
    parts = parts.appending("text: " + text)
  }
  if len(tool_calls) > 0 {
    parts = parts.appending("tool_calls: " + json_stringify(tool_calls))
  }
  if len(parts) == 0 {
    return "<empty response>"
  }
  return join(parts, "\n")
}

fn __payload(agent: HarnessAgent, session: dict, llm_result: dict?, iteration: int) {
  const messages = agent_session_messages(agent, session.session_id)
  return {
    session_id: session.session_id,
    iteration: iteration,
    task: session?.task ?? "",
    transcript: json_stringify(messages),
    latest_response: __serialize_response(llm_result),
  }
}

/**
 * `replace` removes the vetoed assistant turn, so it only applies while that
 * turn is still the last message. Parse repair can append a paired
 * `tool_result` after it before the judge runs; removing the assistant turn
 * then would orphan that result, and the removal refuses. Such a veto is
 * applied as `retain` instead: the turn and its pairing stay, and the critique
 * follows them.
 */
fn __effective_on_veto(
  agent: HarnessAgent,
  session_id: any,
  on_veto: string,
  vetoed: bool,
) -> string {
  if !vetoed || on_veto != "replace" {
    return on_veto
  }
  const messages = agent_session_messages(agent, session_id)
  if len(messages) == 0 || messages[len(messages) - 1]?.role != "assistant" {
    return "retain"
  }
  return on_veto
}

fn __should_skip(
  judge_cfg: dict,
  llm_result: dict,
  attempts: int,
  stall_warning: any,
  remaining_iterations: any,
) {
  const max_attempts = judge_cfg?.max_attempts ?? 3
  if attempts >= max_attempts {
    return {skip: true, reason: "max_attempts_reached"}
  }
  const skip_when_iterations_remaining = judge_cfg?.skip_when_iterations_remaining ?? 1
  if remaining_iterations != nil && remaining_iterations <= skip_when_iterations_remaining {
    return {skip: true, reason: "low_iteration_budget"}
  }
  const skip_empty = judge_cfg?.skip_when_empty ?? true
  const text = llm_result?.text ?? ""
  const tools = llm_result?.tool_calls ?? []
  if skip_empty && text == "" && len(tools) == 0 {
    return {skip: true, reason: "empty_response"}
  }
  const skip_stalled = judge_cfg?.skip_when_stalled ?? true
  if skip_stalled && stall_warning != nil {
    return {skip: true, reason: "stall_already_detected"}
  }
  return {skip: false}
}

/**
 * Why a configured step judge could not review a turn. A closed set, so a host
 * can say the reason in its own words instead of parsing error text:
 * `schema_unsupported` (the model cannot return the verdict schema, e.g. a
 * label-only decision model), `model_unconfigured`, `admission_refused`, and
 * `provider_error` (the call was made and failed).
 */
type StepJudgeUnavailableReason = "schema_unsupported" \
  | "model_unconfigured" \
  | "admission_refused" \
  | "provider_error"

fn __unavailable_reason_for(checkpoint: dict) -> StepJudgeUnavailableReason {
  const status = to_string(checkpoint?.status ?? "")
  if status == "model_unconfigured" {
    return "model_unconfigured"
  }
  if status == "schema_unsupported" {
    return "schema_unsupported"
  }
  if status == "admission_refused" {
    return "admission_refused"
  }
  return "provider_error"
}

/**
 * The unavailable record. The turn still proceeds (a reviewer outage must not
 * block the agent), but the verdict is `unavailable`, never `pass`: an inert
 * reviewer must not read as an approval anywhere downstream.
 */
fn __unavailable(
  reason: StepJudgeUnavailableReason,
  detail: string,
  duration_ms: int | float,
  checkpoint: dict?,
) {
  return {
    vetoed: false,
    skipped: true,
    judge_error: true,
    reason: "judge_unavailable",
    unavailable_reason: reason,
    verdict: "unavailable",
    critique: "",
    reasoning: detail,
    confidence: 0,
    judge_duration_ms: duration_ms,
    typed_checkpoint: checkpoint,
  }
}

fn __invoke(harness: Harness, judge_cfg: dict, opts: dict, payload: any) {
  const system = step_judge_system_prompt(harness.fs, {rubric: judge_cfg?.rubric ?? "default"})
  const user = step_judge_user_prompt(harness.fs, payload)
  const schema = __step_judge_schema()
  const ran = __judge_run_checkpoint(
    harness,
    {
      judge_cfg: judge_cfg,
      opts: opts,
      payload: payload,
      checkpoint_name: "agent.step_judge",
      system: system,
      user: user,
      schema: schema,
    },
  )
  const checkpoint = ran.checkpoint
  const duration_ms = ran.duration_ms
  if !checkpoint.ok {
    const fail_open = judge_cfg?.fail_open_on_error ?? true
    if fail_open {
      // The turn PROCEEDS (we deliberately do not block on the judge's own
      // flakiness), but as `unavailable`, never as a pass.
      return __unavailable(
        __unavailable_reason_for(checkpoint),
        to_string(checkpoint.error),
        duration_ms,
        checkpoint,
      )
    }
    return {
      vetoed: true,
      verdict: "revise",
      critique: checkpoint.error,
      reasoning: checkpoint.error,
      confidence: 0,
      judge_duration_ms: duration_ms,
      typed_checkpoint: checkpoint,
    }
  }
  const result = checkpoint.data
  const {reasoning = "", critique = "", confidence = 1.0} = result ?? {}
  const pass_verdicts = ["pass", "yes", "ok", "advance", "proceed", "approve"]
  const feedback_default = judge_cfg?.feedback_fallback
    ?? "The previous response should be revised."
  const outcome = __judge_classify_verdict(
    result?.verdict ?? "revise",
    pass_verdicts,
    [critique, reasoning],
    feedback_default,
  )
  return outcome
    + {
      reasoning: reasoning,
      critique: critique,
      confidence: confidence,
      judge_duration_ms: duration_ms,
      typed_checkpoint: checkpoint,
    }
}

fn __emit_decision(
  agent: HarnessAgent,
  session_id: any,
  iteration: int,
  on_veto: any,
  verdict: dict,
) {
  const {usage = {}, provider = "", model = ""} = verdict?.typed_checkpoint ?? {}
  const cost_est = if provider != "" && model != "" {
    estimate_call_cost(
      {
        provider: provider,
        model: model,
        input_tokens: usage?.input_tokens ?? 0,
        output_tokens: usage?.output_tokens ?? 0,
        cache_read_tokens: usage?.cache_read_tokens ?? 0,
        cache_write_tokens: usage?.cache_write_tokens ?? 0,
        calls: 1,
      },
    )
  } else {
    {cost_usd: 0}
  }
  agent_emit_event(
    agent,
    session_id,
    "step_judge_decision",
    {
      iteration: iteration,
      verdict: verdict?.verdict
        ?? if verdict.vetoed {
          "revise"
        } else {
          "pass"
        },
      reasoning: verdict?.reasoning ?? "",
      critique: verdict?.critique ?? "",
      confidence: verdict?.confidence ?? 1.0,
      judge_duration_ms: verdict?.judge_duration_ms ?? 0,
      on_veto: on_veto,
      vetoed: verdict.vetoed,
      skipped: verdict?.skipped ?? false,
      reason: verdict?.reason,
      judge_error: verdict?.judge_error ?? false,
      unavailable_reason: verdict?.unavailable_reason,
      unavailable_count: verdict?.unavailable_count ?? 0,
      input_tokens: usage?.input_tokens ?? 0,
      output_tokens: usage?.output_tokens ?? 0,
      cost_usd: cost_est?.cost_usd ?? 0,
      provider: provider,
      model: model,
    },
  )
}

/**
 * agent_step_judge — per-turn critique hook. Sibling of
 * `agent_evaluate_completion` but fires AFTER each assistant turn and
 * BEFORE tool dispatch. Returns
 * `{vetoed, verdict, critique, reasoning, confidence, judge_duration_ms, ...}`.
 * The loop is responsible for acting on `vetoed`: inject feedback,
 * optionally pop the trailing assistant turn (replace mode), and
 * continue to next iteration. See `agent_loop` documentation for the
 * `step_judge` opts shape.
 *
 * Skip-path returns `{vetoed: false, skipped: true, reason: ...}` so
 * the caller can treat skip and pass uniformly. Reasons:
 * `not_configured`, `max_attempts_reached`, `low_iteration_budget`,
 * `empty_response`, `stall_already_detected`, `judge_unavailable` (the
 * judge could not review the turn). An unavailable judge lets the turn
 * proceed but reports `verdict: "unavailable"`, `judge_error: true`, a typed
 * `unavailable_reason` (`schema_unsupported`, `model_unconfigured`,
 * `admission_refused`, `provider_error`) and the session's running
 * `unavailable_count`, so a host can show the outage instead of a silent pass.
 * The caller passes the count it has seen so far as `unavailable_count`.
 * `skip_when_iterations_remaining` defaults to `1`, so a configured step
 * judge does not veto a turn when no regeneration turn remains. Configured
 * skip paths emit `step_judge_decision` with `skipped: true`.
 *
 * @effects: [host]
 * @errors: []
 * @api_stability: experimental
 */
pub fn agent_step_judge(
  harness: Harness,
  session: dict,
  llm_result: any,
  opts: dict,
  iteration: int = 0,
  stall_warning: any = nil,
  attempts: int = 0,
  remaining_iterations: any = nil,
  unavailable_count: int = 0,
) {
  const judge_cfg = opts?.step_judge
  if judge_cfg == nil {
    return {vetoed: false, skipped: true, reason: "not_configured"}
  }
  const on_veto = judge_cfg?.on_veto ?? "replace"
  const skip = __should_skip(judge_cfg, llm_result, attempts, stall_warning, remaining_iterations)
  if skip.skip {
    const verdict = {
      vetoed: false,
      skipped: true,
      reason: skip.reason,
      verdict: "pass",
      critique: "",
      reasoning: "",
      confidence: 1.0,
      judge_duration_ms: 0,
    }
    __emit_decision(harness.agent, session.session_id, iteration, on_veto, verdict)
    return verdict + {on_veto: on_veto, attempts: attempts}
  }
  const payload = __payload(harness.agent, session, llm_result, iteration)
  const invoked = __invoke(harness, judge_cfg, opts, payload)
  const verdict = if invoked?.judge_error ?? false {
    invoked + {unavailable_count: unavailable_count + 1}
  } else {
    invoked + {unavailable_count: unavailable_count}
  }
  const applied_on_veto = __effective_on_veto(
    harness.agent,
    session.session_id,
    on_veto,
    verdict?.vetoed ?? false,
  )
  __emit_decision(harness.agent, session.session_id, iteration, applied_on_veto, verdict)
  return verdict + {on_veto: applied_on_veto, attempts: attempts + 1}
}