rho-coding-agent 2.19.0

A fast Rust agent harness with a small footprint and opinionated defaults
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
use std::{path::Path, sync::Arc};

use anyhow::{anyhow, bail};
use rho_providers::{
    model::{
        models_dev::{cached_model_metadata, ModelMetadata},
        Message,
    },
    provider::ProviderId,
    reasoning::ReasoningLevel,
};
use rho_sdk::model::context::estimate_text_tokens;
use rho_sdk::{
    provider::ModelProvider, ApprovalRequest, CancellationToken, ProviderRequestUsageRecording,
    SessionId,
};

use crate::{
    agent::{effective_internal_agent_reasoning, PERMISSION_CLASSIFIER_AGENT_ID},
    config::{
        Config, InternalAgentModelConfig, InternalAgentTarget, ModelKind, RhoInternalAgentModel,
    },
    credential_store::build_provider_on,
};

use super::{
    review_verdict, screen_allow_probability, screen_verdict, transcript::render_with_pending_call,
    ClassifierVerdict, ScreenVerdict, TranscriptBudget, TranscriptOverBudget, CLASSIFIER_POLICY,
    DEFAULT_SCREEN_ALLOW_PERCENT, REVIEW_QUESTION, SCREEN_ALLOW_PERCENT_RANGE, SCREEN_QUESTION,
};
use crate::decision::{self, EntryModel, TextModel};
use rho_sdk::decision::{
    text::{questions_block, system_prompt, AnswerStyle},
    Answer, DecisionModel, DecisionRequest, Question,
};

/// Config entry, `[internal_agents.permission-classifier-screen]`, naming a
/// decision model or another chat model that answers the screen in place of
/// the classifier's own model. It is not an agent: it has no prompt or tools.
pub(crate) const DECISION_SCREEN_ID: &str = "permission-classifier-screen";

/// Context reserved for the classifier's own output when sizing the transcript.
///
/// Usage ledger receipt: 398 review calls peaked at 5,348 output tokens
/// (p99 954); screens peaked at 924. 8,192 covers the observed peak.
const CLASSIFIER_OUTPUT_RESERVE_TOKENS: u64 = 8_192;

pub(crate) struct ClassifyRequest<'a> {
    pub history: &'a [Message],
    pub pending: &'a ApprovalRequest,
    pub cancellation: CancellationToken,
    pub session_id: &'a SessionId,
    pub workspace_path: &'a Path,
    pub usage_recording: ProviderRequestUsageRecording,
}

impl ClassifyRequest<'_> {
    pub(super) fn scope(&self) -> CallScope<'_> {
        CallScope {
            cancellation: &self.cancellation,
            session_id: self.session_id,
            workspace_path: self.workspace_path,
            usage_recording: &self.usage_recording,
        }
    }
}

/// How a classifier model call is cancelled and where it records usage. Every
/// request of a batch shares one.
pub(super) struct CallScope<'a> {
    pub cancellation: &'a CancellationToken,
    pub session_id: &'a SessionId,
    pub workspace_path: &'a Path,
    pub usage_recording: &'a ProviderRequestUsageRecording,
}

pub(crate) async fn classify_capability_request(
    config: &Config,
    request: ClassifyRequest<'_>,
) -> ClassifierVerdict {
    let result = match ClassifierModel::resolve(config).await {
        Ok(model) => model.classify(request).await.result,
        Err(error) => Err(error),
    };
    result.unwrap_or_else(classifier_unavailable)
}

/// The configured classifier model, ready to classify requests.
pub(crate) struct ClassifierModel {
    pub(super) provider: Arc<dyn ModelProvider>,
    pub(super) reasoning: ReasoningLevel,
    pub(super) budget: TranscriptBudget,
    pub(super) screen: Screen,
}

/// What answers the screen, from [`DECISION_SCREEN_ID`].
pub(super) enum Screen {
    /// No entry: the classifier's own model, reading the review's transcript.
    Classifier,
    /// Another chat model, reading a transcript fitted to its own window.
    /// `budget` is `None` when the window it serves is unknown, so it could
    /// silently drop part of the transcript; it then escalates every request.
    Text {
        provider: Arc<dyn ModelProvider>,
        budget: Option<TranscriptBudget>,
    },
    /// A decision model, allowing at P(allow) of `allow_percent` or more.
    Decision {
        model: Box<dyn DecisionModel>,
        allow_percent: u8,
    },
}

/// Fails when [`DECISION_SCREEN_ID`] is configured but unusable, so a
/// headless run can refuse to start instead of denying every request.
pub(crate) fn check_screen_config(config: &Config) -> anyhow::Result<()> {
    // A text model is built at classification, so only its kind and provider
    // are checked here.
    match decision::resolve(config, DECISION_SCREEN_ID)? {
        Some(EntryModel::Decision(_)) => {
            decision_allow_percent(config)?;
        }
        Some(EntryModel::Text(_)) | None => {}
    }
    Ok(())
}

/// The allow percent of a screen entry that resolved to a decision model.
fn decision_allow_percent(config: &Config) -> Result<u8, decision::ConfigError> {
    config
        .internal_agent_model(DECISION_SCREEN_ID)
        .and_then(InternalAgentModelConfig::rho)
        .map_or(Ok(DEFAULT_SCREEN_ALLOW_PERCENT), screen_allow_percent)
}

/// The P(allow) percent at which the screen entry `selection` allows: its
/// `allow_threshold_percent`, else [`DEFAULT_SCREEN_ALLOW_PERCENT`]. Only a
/// decision model reports probabilities; a text screen ignores it.
pub(crate) fn screen_allow_percent(
    selection: &RhoInternalAgentModel,
) -> Result<u8, decision::ConfigError> {
    let percent = selection
        .allow_threshold_percent
        .unwrap_or(DEFAULT_SCREEN_ALLOW_PERCENT);
    if SCREEN_ALLOW_PERCENT_RANGE.contains(&percent) {
        Ok(percent)
    } else {
        Err(decision::ConfigError::AllowThresholdOutOfRange {
            entry: DECISION_SCREEN_ID,
            percent,
            min: *SCREEN_ALLOW_PERCENT_RANGE.start(),
            max: *SCREEN_ALLOW_PERCENT_RANGE.end(),
        })
    }
}

impl ClassifierModel {
    /// Builds the `[internal_agents.permission-classifier]` model.
    pub(crate) async fn resolve(config: &Config) -> anyhow::Result<Self> {
        let screen = resolve_screen(config).await?;
        let model = config
            .internal_agent_model(PERMISSION_CLASSIFIER_AGENT_ID)
            .ok_or_else(|| anyhow!("{PERMISSION_CLASSIFIER_AGENT_ID} model is not configured"))?;
        let reasoning = effective_internal_agent_reasoning(PERMISSION_CLASSIFIER_AGENT_ID, model);
        let InternalAgentTarget::Rho(selection) = &model.target else {
            bail!("{PERMISSION_CLASSIFIER_AGENT_ID} cannot run on Claude Code runtime");
        };
        let provider = build_provider_on(
        config,
            &selection.provider,
            &selection.model,
            reasoning,
            &selection.auth,
        )
        .await
        .map_err(|_| {
            anyhow!(
                "failed to build {PERMISSION_CLASSIFIER_AGENT_ID} provider; check configured credentials"
            )
        })?;
        // Read the window after building the provider: catalog-driven
        // providers hydrate the model catalog during construction.
        let budget = transcript_budget(
            cached_model_metadata(&selection.provider, &selection.model)
                .and_then(|metadata| metadata.display_context_window()),
        );
        Ok(Self {
            provider,
            reasoning,
            budget,
            screen,
        })
    }

    pub(crate) fn provider(&self) -> &dyn ModelProvider {
        self.provider.as_ref()
    }

    /// Review-stage reasoning; a text-model screen always runs at
    /// [`ReasoningLevel::Low`].
    pub(crate) fn reasoning(&self) -> ReasoningLevel {
        self.reasoning
    }

    pub(crate) async fn classify(&self, request: ClassifyRequest<'_>) -> ClassifierTrace {
        let pending_call_id = request.pending.tool_call_id().map(|id| id.as_str());
        run_pipeline(
            self.provider.as_ref(),
            &self.screen,
            self.reasoning,
            self.budget,
            pending_call_id,
            &request,
        )
        .await
    }

    /// [`Self::classify`] for a request the SDK did not build, such as a
    /// replayed call, whose tool call ID cannot be attached to it.
    pub(crate) async fn classify_with_pending_call(
        &self,
        request: ClassifyRequest<'_>,
        pending_call_id: &str,
    ) -> ClassifierTrace {
        run_pipeline(
            self.provider.as_ref(),
            &self.screen,
            self.reasoning,
            self.budget,
            Some(pending_call_id),
            &request,
        )
        .await
    }
}

/// Builds the screen [`DECISION_SCREEN_ID`] names. A text model's build
/// failure names the entry, since its credentials may be what is missing.
async fn resolve_screen(config: &Config) -> anyhow::Result<Screen> {
    let selection = match decision::resolve(config, DECISION_SCREEN_ID)? {
        None => return Ok(Screen::Classifier),
        Some(EntryModel::Decision(model)) => {
            return Ok(Screen::Decision {
                model,
                allow_percent: decision_allow_percent(config)?,
            });
        }
        Some(EntryModel::Text(selection)) => selection,
    };
    let provider = build_provider_on(
        config,
        &selection.provider,
        &selection.model,
        ReasoningLevel::Low,
        &selection.auth,
    )
    .await
    .map_err(|_| decision::ConfigError::TextModelUnavailable {
        entry: DECISION_SCREEN_ID,
        configured: rho_providers::provider::model_reference(&selection.provider, &selection.model),
    })?;
    let budget = text_screen_budget(
        &selection.provider,
        cached_model_metadata(&selection.provider, &selection.model),
    );
    Ok(Screen::Text { provider, budget })
}

/// Why the screen entry `selection` likely will not work as configured: its
/// kind does not fit what was discovered, or it is a text model whose served
/// window is unknown, so it escalates every request. For `/config` and
/// `/doctor`.
pub(crate) fn screen_warning(selection: &RhoInternalAgentModel) -> Option<String> {
    decision::kind_mismatch(selection).or_else(|| {
        let unknown_window = decision::entry_kind(selection) == ModelKind::Text
            && text_screen_budget(
                &selection.provider,
                cached_model_metadata(&selection.provider, &selection.model),
            )
            .is_none();
        unknown_window.then(|| {
            format!(
                "{} has no usable_context_window, so the screen escalates every request",
                selection.model
            )
        })
    })
}

/// The transcript budget a text screen on `provider` is fitted to, or `None`
/// when the window it serves is unknown.
///
/// Ollama serves a model at the server's `num_ctx`, often far below the
/// window the model advertises, and drops the front of a longer prompt
/// without an error, which could leave a screen allowing a request it never
/// read whole. So on Ollama only a measured `usable_context_window` counts.
/// Hosted providers reject an oversize prompt, and that error escalates.
pub(super) fn text_screen_budget(
    provider: &str,
    metadata: Option<ModelMetadata>,
) -> Option<TranscriptBudget> {
    let ollama = rho_providers::provider::provider_descriptor(provider)
        .is_some_and(|descriptor| descriptor.id == ProviderId::Ollama);
    if ollama {
        metadata
            .and_then(|metadata| metadata.usable_context_window)
            .map(|window| transcript_budget(Some(window)))
    } else {
        Some(transcript_budget(
            metadata.and_then(|metadata| metadata.display_context_window()),
        ))
    }
}

/// What one classification did at each stage. Production acts only on
/// `result`; the classifier eval also reports the screen outcome.
pub(crate) struct ClassifierTrace {
    pub screen: ScreenOutcome,
    /// A decision-model screen's P(allow), when it answered.
    pub screen_allow_probability: Option<f64>,
    /// `Err` fails closed in production, as "classifier unavailable" or as
    /// the over-budget reason.
    pub result: anyhow::Result<ClassifierVerdict>,
}

#[derive(Clone, Debug, PartialEq, Eq)]
pub(crate) enum ScreenOutcome {
    /// The transcript could not be rendered, so no stage ran.
    Skipped,
    Allowed,
    Escalated,
    /// The screen call or its parse failed with this error; the review
    /// decided.
    Failed(String),
}

#[cfg(test)]
pub(super) async fn classify_capability_request_with_provider(
    provider: &dyn ModelProvider,
    screen: &Screen,
    reasoning: ReasoningLevel,
    budget: TranscriptBudget,
    request: ClassifyRequest<'_>,
) -> ClassifierVerdict {
    let pending_call_id = request.pending.tool_call_id().map(|id| id.as_str());
    run_pipeline(
        provider,
        screen,
        reasoning,
        budget,
        pending_call_id,
        &request,
    )
    .await
    .result
    .unwrap_or_else(classifier_unavailable)
}

/// Transcript budget for a classifier model with `context_window` tokens.
///
/// Subtracts the shared system prompt, the longer stage questions block, and
/// [`CLASSIFIER_OUTPUT_RESERVE_TOKENS`]. An unknown window leaves the
/// transcript unbounded; the provider still rejects oversize requests.
pub(super) fn transcript_budget(context_window: Option<u64>) -> TranscriptBudget {
    let Some(window) = context_window else {
        return TranscriptBudget::Unbounded;
    };
    let questions_tokens = [SCREEN_STAGE, REVIEW_STAGE]
        .iter()
        .map(|stage| estimate_text_tokens(&questions_block(stage.questions, stage.style)))
        .max()
        .unwrap_or_default();
    let overhead = estimate_text_tokens(&system_prompt(CLASSIFIER_POLICY))
        .saturating_add(questions_tokens)
        .saturating_add(CLASSIFIER_OUTPUT_RESERVE_TOKENS);
    TranscriptBudget::Tokens(window.saturating_sub(overhead))
}

/// Runs the two-stage pipeline: a cheap screen, then a reasoned review.
///
/// Both stages are decision requests over the rendered transcript. Stage 1
/// answers `allow` or `escalate` from the [`Screen`]: a decision model, or a
/// text model at [`ReasoningLevel::Low`]. Only an escalation (or a stage 1 failure) pays
/// for stage 2, which the text model reasons through at the configured level.
///
/// Cache-prefix layout: both stages send the same system prompt and the same
/// rendered transcript as the first user text block. The stage's questions are
/// a second user text block so the last byte-identical block can be the cache
/// breakpoint. Never move a stage's questions into the system prompt.
///
/// That layout can reuse stage 1's message-cache prefix only when thinking and
/// effort stay the same, which is the default Low classifier reasoning. Raising
/// review reasoning keeps the common-path screen cheap and forgoes that cache
/// hit: Anthropic invalidates message-block cache when thinking or effort change.
async fn run_pipeline(
    provider: &dyn ModelProvider,
    screen: &Screen,
    reasoning: ReasoningLevel,
    budget: TranscriptBudget,
    pending_call_id: Option<&str>,
    request: &ClassifyRequest<'_>,
) -> ClassifierTrace {
    match screen_step(provider, screen, budget, pending_call_id, request).await {
        ScreenStep::Done(trace) => trace,
        ScreenStep::Review {
            screen,
            screen_allow_probability,
            transcript,
        } => ClassifierTrace {
            screen,
            screen_allow_probability,
            result: review(provider, reasoning, &request.scope(), &transcript).await,
        },
    }
}

/// Where one request stands after stage 1.
pub(crate) enum ScreenStep {
    /// No review is needed: the screen allowed the request, or its
    /// transcript could not be rendered.
    Done(ClassifierTrace),
    /// The screen escalated or failed, so the review decides, reading
    /// `transcript`.
    Review {
        screen: ScreenOutcome,
        screen_allow_probability: Option<f64>,
        transcript: String,
    },
}

/// Stage 1 of [`run_pipeline`] for one request.
pub(super) async fn screen_step(
    provider: &dyn ModelProvider,
    screen: &Screen,
    budget: TranscriptBudget,
    pending_call_id: Option<&str>,
    request: &ClassifyRequest<'_>,
) -> ScreenStep {
    let transcript =
        match render_with_pending_call(request.history, request.pending, pending_call_id, budget) {
            Ok(transcript) => transcript,
            Err(error) => {
                return ScreenStep::Done(ClassifierTrace {
                    screen: ScreenOutcome::Skipped,
                    screen_allow_probability: None,
                    result: Err(error),
                })
            }
        };

    let (screen, screen_allow_probability) = match run_screen(
        provider,
        screen,
        request,
        pending_call_id,
        &transcript,
    )
    .await
    {
        Ok((ScreenVerdict::Allow, allow_probability)) => {
            return ScreenStep::Done(ClassifierTrace {
                screen: ScreenOutcome::Allowed,
                screen_allow_probability: allow_probability,
                result: Ok(ClassifierVerdict::Allow),
            })
        }
        Ok((ScreenVerdict::Escalate, allow_probability)) => {
            (ScreenOutcome::Escalated, allow_probability)
        }
        Err(error) => {
            // A broken screen must not decide anything; stage 2 still runs
            // and fails closed on its own if it also breaks.
            tracing::warn!(error = %error, "permission classifier screen failed; running review");
            (ScreenOutcome::Failed(format!("{error:#}")), None)
        }
    };
    ScreenStep::Review {
        screen,
        screen_allow_probability,
        transcript,
    }
}

/// Stage 2 for one request: the reasoned review of `transcript`.
pub(super) async fn review(
    provider: &dyn ModelProvider,
    reasoning: ReasoningLevel,
    scope: &CallScope<'_>,
    transcript: &str,
) -> anyhow::Result<ClassifierVerdict> {
    let model = text_model(
        provider,
        scope,
        REVIEW_STAGE.usage_purpose,
        REVIEW_STAGE.style,
        reasoning,
    );
    let decision = DecisionRequest::new(CLASSIFIER_POLICY, transcript, REVIEW_STAGE.questions);
    review_verdict(&ask(&model, scope.cancellation, decision).await?)
}

/// Stage 1: the screen's verdict, with the model's P(allow) when it reports
/// probabilities.
///
/// The classifier's own model reads the review's `transcript`. Another
/// model with a known window, or a decision model with a state budget, gets
/// its own transcript fitted to it, with the oldest tool calls left out
/// first; the review keeps its own budget, so a screen that cannot fit only
/// escalates.
async fn run_screen(
    provider: &dyn ModelProvider,
    screen: &Screen,
    request: &ClassifyRequest<'_>,
    pending_call_id: Option<&str>,
    transcript: &str,
) -> anyhow::Result<(ScreenVerdict, Option<f64>)> {
    let text_screen;
    // A text model reports no probabilities, so its allow percent is moot.
    let (model, budget, allow_percent): (&dyn DecisionModel, _, _) = match screen {
        Screen::Classifier => {
            text_screen = text_model(
                provider,
                &request.scope(),
                SCREEN_STAGE.usage_purpose,
                SCREEN_STAGE.style,
                ReasoningLevel::Low,
            );
            (&text_screen, None, DEFAULT_SCREEN_ALLOW_PERCENT)
        }
        Screen::Text { provider, budget } => {
            let Some(budget) = budget else {
                let identity = provider.identity();
                bail!(
                    "text screen {} has no known served context window, so it may truncate the transcript; set usable_context_window for it in ~/.rho/models.toml",
                    rho_providers::provider::model_reference(&identity.provider, &identity.model)
                );
            };
            text_screen = text_model(
                provider.as_ref(),
                &request.scope(),
                SCREEN_STAGE.usage_purpose,
                SCREEN_STAGE.style,
                ReasoningLevel::Low,
            );
            let own = match budget {
                TranscriptBudget::Unbounded => None,
                TranscriptBudget::Tokens(_) => Some(*budget),
            };
            (&text_screen, own, DEFAULT_SCREEN_ALLOW_PERCENT)
        }
        Screen::Decision {
            model,
            allow_percent,
        } => (
            model.as_ref(),
            model.state_budget().map(TranscriptBudget::Tokens),
            *allow_percent,
        ),
    };
    let fitted;
    let state = match budget {
        None => transcript,
        Some(budget) => {
            fitted = render_with_pending_call(
                request.history,
                request.pending,
                pending_call_id,
                budget,
            )?;
            &fitted
        }
    };
    let decision = DecisionRequest::new(CLASSIFIER_POLICY, state, SCREEN_STAGE.questions);
    let answers = ask(model, &request.cancellation, decision).await?;
    Ok((
        screen_verdict(&answers, allow_percent),
        screen_allow_probability(&answers),
    ))
}

/// One classifier stage: the questions it asks and how the model answers.
pub(super) struct Stage {
    pub usage_purpose: &'static str,
    pub questions: &'static [Question<'static>],
    pub style: AnswerStyle,
}

const SCREEN_STAGE: Stage = Stage {
    usage_purpose: "permission-classifier-screen",
    questions: std::slice::from_ref(&SCREEN_QUESTION),
    style: AnswerStyle::Direct,
};

pub(super) const REVIEW_STAGE: Stage = Stage {
    usage_purpose: "permission-classifier-review",
    questions: std::slice::from_ref(&REVIEW_QUESTION),
    style: AnswerStyle::Reasoned,
};

/// The classifier's text model, recording usage under `usage_purpose`.
pub(super) fn text_model<'a>(
    provider: &'a dyn ModelProvider,
    scope: &CallScope<'a>,
    usage_purpose: &'static str,
    style: AnswerStyle,
    reasoning: ReasoningLevel,
) -> TextModel<'a> {
    TextModel {
        provider,
        agent_id: PERMISSION_CLASSIFIER_AGENT_ID,
        usage_purpose,
        reasoning,
        style,
        session_id: scope.session_id,
        workspace_path: scope.workspace_path,
        usage_recording: scope.usage_recording.clone(),
    }
}

/// Asks `decision` and checks the answers fit its questions.
pub(super) async fn ask(
    model: &dyn DecisionModel,
    cancellation: &CancellationToken,
    decision: DecisionRequest<'_>,
) -> anyhow::Result<Vec<Answer>> {
    let answers = model.decide(decision, cancellation).await?;
    decision.check_answers(&answers)?;
    Ok(answers)
}

fn classifier_unavailable(error: anyhow::Error) -> ClassifierVerdict {
    // An over-budget transcript carries only token counts, so the limit and
    // the asked size can be shown instead of hidden behind "unavailable".
    if let Some(over_budget) = error.downcast_ref::<TranscriptOverBudget>() {
        tracing::warn!(error = %over_budget, "permission classifier transcript over budget");
        return ClassifierVerdict::Deny {
            reason: over_budget.to_string(),
        };
    }
    // A misconfigured screen names only configured values, so the fix can
    // be shown too.
    if let Some(screen) = error.downcast_ref::<decision::ConfigError>() {
        tracing::warn!(error = %screen, "permission classifier screen misconfigured");
        return ClassifierVerdict::Deny {
            reason: screen.to_string(),
        };
    }
    // Keep other details out of the executor-facing deny reason; credential
    // and provider response bodies can show up in Display output.
    tracing::warn!(error = %error, "permission classifier unavailable");
    ClassifierVerdict::Deny {
        reason: "classifier unavailable".into(),
    }
}