Skip to main content

turnframe_provider/
response.rs

1//! The normalized model response (spec §20.1).
2//!
3//! A [`ModelResponse`] is what every adapter returns, whatever the vendor sent:
4//! the same [`ContentPart`] vocabulary the request uses, a
5//! [`FinishReason`], reported [`TokenUsage`], and the warnings the adapter
6//! wants the caller to know about.
7//!
8//! Three helpers cover the ways the runtime reads a response:
9//!
10//! * [`text`](ModelResponse::text) concatenates the prose;
11//! * [`tool_calls`](ModelResponse::tool_calls) lists the calls in order;
12//! * [`single_json`](ModelResponse::single_json) extracts the **one** JSON
13//!   document a structured stage expects, and fails rather than choosing when
14//!   there is more than one candidate (spec I18).
15//!
16//! A response is untrusted input (I9). Nothing here validates it against a
17//! schema — that is [`structured`](crate::structured)'s job, and it is
18//! all-or-nothing.
19
20use std::fmt;
21use std::time::Duration;
22
23use serde::{Deserialize, Serialize};
24
25use crate::ids::{CallId, ModelKey, ModelRef, ProviderKey, RequestId};
26use crate::request::{ContentPart, ToolCall};
27use crate::structured::StructuredOutputError;
28
29/// Why the model stopped generating.
30#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, PartialOrd, Ord, Serialize, Deserialize)]
31#[serde(rename_all = "snake_case")]
32#[non_exhaustive]
33pub enum FinishReason {
34    /// The model finished on its own or hit a stop sequence.
35    Stop,
36    /// The output token cap was reached. The answer is truncated, so a
37    /// structured stage must reject it rather than parse the prefix.
38    MaxTokens,
39    /// The model ended its turn with tool calls.
40    ToolCalls,
41    /// A safety filter stopped the generation.
42    ContentFilter,
43    /// The model declined. A semantic outcome, distinguishable from a
44    /// transport failure or a malformed body.
45    Refusal,
46    /// The provider reported something this crate does not model. Recorded as
47    /// a [`ResponseWarning::UnknownFinishReason`] too.
48    Other,
49}
50
51impl FinishReason {
52    /// Stable snake-case label.
53    #[must_use]
54    pub const fn as_str(self) -> &'static str {
55        match self {
56            Self::Stop => "stop",
57            Self::MaxTokens => "max_tokens",
58            Self::ToolCalls => "tool_calls",
59            Self::ContentFilter => "content_filter",
60            Self::Refusal => "refusal",
61            Self::Other => "other",
62        }
63    }
64
65    /// Returns `true` when the answer is complete enough to be parsed.
66    ///
67    /// A truncated, filtered or refused answer never is: parsing its prefix is
68    /// exactly the partial-execution failure I18 forbids.
69    #[must_use]
70    pub const fn is_complete(self) -> bool {
71        matches!(self, Self::Stop | Self::ToolCalls)
72    }
73}
74
75impl fmt::Display for FinishReason {
76    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
77        f.write_str(self.as_str())
78    }
79}
80
81/// Tokens the provider reported for a call.
82///
83/// Zero means "not reported": no provider distinguishes a real zero from a
84/// missing count, and a cost estimate built on a guess is worse than one built
85/// on a zero.
86#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash, Default, Serialize, Deserialize)]
87#[serde(deny_unknown_fields)]
88pub struct TokenUsage {
89    /// Prompt tokens.
90    pub input: u64,
91    /// Generated tokens.
92    pub output: u64,
93    /// Prompt tokens served from the provider's cache, already counted in
94    /// `input`.
95    #[serde(default)]
96    pub cached_input: u64,
97    /// Reasoning tokens, already counted in `output` where the provider bills
98    /// them that way.
99    #[serde(default)]
100    pub reasoning: u64,
101}
102
103impl TokenUsage {
104    /// Usage with input and output counts only.
105    #[must_use]
106    pub const fn new(input: u64, output: u64) -> Self {
107        Self {
108            input,
109            output,
110            cached_input: 0,
111            reasoning: 0,
112        }
113    }
114
115    /// Nothing reported.
116    #[must_use]
117    pub const fn none() -> Self {
118        Self::new(0, 0)
119    }
120
121    /// Records cached prompt tokens.
122    #[must_use]
123    pub const fn with_cached_input(mut self, cached: u64) -> Self {
124        self.cached_input = cached;
125        self
126    }
127
128    /// Records reasoning tokens.
129    #[must_use]
130    pub const fn with_reasoning(mut self, reasoning: u64) -> Self {
131        self.reasoning = reasoning;
132        self
133    }
134
135    /// Input plus output.
136    #[must_use]
137    pub const fn total(self) -> u64 {
138        self.input.saturating_add(self.output)
139    }
140
141    /// Returns `true` when the provider reported nothing.
142    #[must_use]
143    pub const fn is_unreported(self) -> bool {
144        self.input == 0 && self.output == 0
145    }
146}
147
148/// Something an adapter wants the caller to know without failing the call.
149///
150/// Warnings are the honest channel for "I did the job, but not exactly the way
151/// you asked": they belong in a replay record, and a stage that cares may
152/// refuse a response that carries one.
153#[derive(Debug, Clone, PartialEq, Eq, Hash, Serialize, Deserialize)]
154#[serde(tag = "kind", rename_all = "snake_case")]
155#[non_exhaustive]
156pub enum ResponseWarning {
157    /// The provider did not return usable tool-call ids and the adapter
158    /// synthesized them. The profile must declare
159    /// [`preserves_call_ids = false`](crate::capabilities::ProviderCapabilities::preserves_call_ids).
160    SynthesizedCallIds,
161    /// The provider reported a finish reason this crate does not model.
162    UnknownFinishReason {
163        /// The provider's own label, sanitized to a short code.
164        reported: String,
165    },
166    /// The provider did not report token usage.
167    UsageUnreported,
168    /// The adapter dropped a request feature the provider cannot express
169    /// (a stop sequence beyond the vendor limit, a cache hint, a tool-choice
170    /// mode). Never used for structured output: an adapter that cannot enforce
171    /// a schema declares a weaker capability instead of warning about it.
172    FeatureDropped {
173        /// Which feature, as a short code.
174        feature: String,
175    },
176    /// The answer was rebuilt from a stream rather than received whole.
177    Reconstructed,
178}
179
180/// One model answer (spec §20.1).
181#[derive(Debug, Clone, PartialEq, Serialize, Deserialize)]
182#[serde(deny_unknown_fields)]
183pub struct ModelResponse {
184    /// The request this answers. Equal to
185    /// [`ModelRequest::request_id`](crate::request::ModelRequest::request_id)
186    /// even after a retry.
187    pub request_id: RequestId,
188    /// Which provider answered.
189    pub provider: ProviderKey,
190    /// Which model answered. May differ from the requested key when the
191    /// provider resolved an alias; the *reported* model is recorded.
192    pub model: ModelKey,
193    /// The answer, in order.
194    pub content: Vec<ContentPart>,
195    /// Why generation stopped.
196    pub finish: FinishReason,
197    /// Reported tokens.
198    #[serde(default)]
199    pub usage: TokenUsage,
200    /// The provider's own response identifier, for support tickets. Never a
201    /// body and never a header.
202    #[serde(default, skip_serializing_if = "Option::is_none")]
203    pub raw_id: Option<String>,
204    /// Wall-clock duration of the call, measured by the adapter.
205    pub latency: Duration,
206    /// Adapter warnings.
207    #[serde(default, skip_serializing_if = "Vec::is_empty")]
208    pub warnings: Vec<ResponseWarning>,
209}
210
211impl ModelResponse {
212    /// A response with no content, for adapters to build on.
213    #[must_use]
214    pub fn new(
215        request_id: RequestId,
216        provider: impl Into<ProviderKey>,
217        model: impl Into<ModelKey>,
218    ) -> Self {
219        Self {
220            request_id,
221            provider: provider.into(),
222            model: model.into(),
223            content: Vec::new(),
224            finish: FinishReason::Stop,
225            usage: TokenUsage::none(),
226            raw_id: None,
227            latency: Duration::ZERO,
228            warnings: Vec::new(),
229        }
230    }
231
232    /// Appends a content part.
233    #[must_use]
234    pub fn with_part(mut self, part: ContentPart) -> Self {
235        self.content.push(part);
236        self
237    }
238
239    /// Appends a text part.
240    #[must_use]
241    pub fn with_text(self, text: impl Into<String>) -> Self {
242        self.with_part(ContentPart::text(text))
243    }
244
245    /// Appends a tool call.
246    #[must_use]
247    pub fn with_tool_call(self, call: ToolCall) -> Self {
248        self.with_part(ContentPart::ToolCall(call))
249    }
250
251    /// Sets the finish reason.
252    #[must_use]
253    pub const fn with_finish(mut self, finish: FinishReason) -> Self {
254        self.finish = finish;
255        self
256    }
257
258    /// Sets the reported usage.
259    #[must_use]
260    pub const fn with_usage(mut self, usage: TokenUsage) -> Self {
261        self.usage = usage;
262        self
263    }
264
265    /// Sets the provider's response identifier.
266    #[must_use]
267    pub fn with_raw_id(mut self, raw_id: impl Into<String>) -> Self {
268        self.raw_id = Some(raw_id.into());
269        self
270    }
271
272    /// Sets the measured latency.
273    #[must_use]
274    pub const fn with_latency(mut self, latency: Duration) -> Self {
275        self.latency = latency;
276        self
277    }
278
279    /// Adds a warning.
280    #[must_use]
281    pub fn with_warning(mut self, warning: ResponseWarning) -> Self {
282        self.warnings.push(warning);
283        self
284    }
285
286    /// The provider-model pair that answered.
287    #[must_use]
288    pub fn reference(&self) -> ModelRef {
289        ModelRef {
290            provider: self.provider.clone(),
291            model: self.model.clone(),
292        }
293    }
294
295    /// Every text part, concatenated in order.
296    ///
297    /// ```
298    /// use turnframe_provider::prelude::*;
299    ///
300    /// let response = ModelResponse::new(RequestId::nil(), "openai", "gpt-4o")
301    ///     .with_text("Ho preparato ")
302    ///     .with_text("la modifica.");
303    /// assert_eq!(response.text(), "Ho preparato la modifica.");
304    /// ```
305    #[must_use]
306    pub fn text(&self) -> String {
307        let mut out = String::new();
308        for part in &self.content {
309            if let Some(text) = part.as_text() {
310                out.push_str(text);
311            }
312        }
313        out
314    }
315
316    /// Every tool call, in the order the model produced them.
317    #[must_use]
318    pub fn tool_calls(&self) -> Vec<&ToolCall> {
319        self.content
320            .iter()
321            .filter_map(ContentPart::as_tool_call)
322            .collect()
323    }
324
325    /// The call with this id, when present.
326    #[must_use]
327    pub fn tool_call(&self, id: &CallId) -> Option<&ToolCall> {
328        self.tool_calls().into_iter().find(|call| &call.id == id)
329    }
330
331    /// Returns `true` when the model produced neither text nor a tool call.
332    #[must_use]
333    pub fn is_empty(&self) -> bool {
334        self.tool_calls().is_empty() && self.text().trim().is_empty()
335    }
336
337    /// Extracts the single JSON document a structured stage expects.
338    ///
339    /// The rule is deterministic and refuses to guess:
340    ///
341    /// 1. A [`FinishReason::Refusal`] is a
342    ///    [`Refusal`](StructuredOutputError::Refusal), never a parse attempt.
343    /// 2. A finish reason that is not
344    ///    [complete](FinishReason::is_complete) — truncation, a content filter —
345    ///    is a [`NoOutput`](StructuredOutputError::NoOutput): the bytes that
346    ///    arrived are a prefix, and parsing a prefix is what I18 forbids.
347    /// 3. **Exactly one** tool call: its `arguments` are the document. The
348    ///    surrounding prose is a preamble and is ignored, because a provider
349    ///    using a function schema as a transport routinely emits both.
350    /// 4. **More than one** tool call:
351    ///    [`MultipleCandidates`](StructuredOutputError::MultipleCandidates). A
352    ///    stage that expects one document never picks one of several.
353    /// 5. **No tool call**: the concatenated text is parsed. A single
354    ///    ```` ```json ```` fence around it is stripped first — a fence is
355    ///    framing a prompt-only transport adds, not content.
356    ///
357    /// # Errors
358    ///
359    /// See [`StructuredOutputError`].
360    pub fn single_json(&self) -> Result<serde_json::Value, StructuredOutputError> {
361        if self.finish == FinishReason::Refusal {
362            return Err(StructuredOutputError::Refusal);
363        }
364        let calls = self.tool_calls();
365        if calls.len() > 1 {
366            return Err(StructuredOutputError::MultipleCandidates {
367                candidates: calls.len(),
368            });
369        }
370        if !self.finish.is_complete() {
371            return Err(StructuredOutputError::NoOutput);
372        }
373        if let Some(call) = calls.first() {
374            return Ok(call.arguments.clone());
375        }
376        let text = self.text();
377        let payload = strip_code_fence(text.trim());
378        if payload.is_empty() {
379            return Err(StructuredOutputError::NoOutput);
380        }
381        serde_json::from_str(payload).map_err(StructuredOutputError::not_json)
382    }
383}
384
385/// Removes a single Markdown code fence around `text`, if that is all it is.
386///
387/// Only a fence that opens on the first line and closes on the last is removed,
388/// so JSON containing a fenced string is untouched.
389fn strip_code_fence(text: &str) -> &str {
390    let Some(rest) = text.strip_prefix("```") else {
391        return text;
392    };
393    let Some(body_start) = rest.find('\n') else {
394        return text;
395    };
396    // The opening line may carry a language tag and nothing else.
397    if rest[..body_start]
398        .chars()
399        .any(|ch| !ch.is_ascii_alphanumeric())
400    {
401        return text;
402    }
403    let body = &rest[body_start + 1..];
404    match body.trim_end().strip_suffix("```") {
405        Some(inner) => inner.trim(),
406        None => text,
407    }
408}
409
410#[cfg(test)]
411mod tests {
412    use super::*;
413    use serde_json::json;
414
415    fn response() -> ModelResponse {
416        ModelResponse::new(RequestId::nil(), "openai", "gpt-4o")
417    }
418
419    #[test]
420    fn helpers_read_text_and_calls_in_order() {
421        let built = response()
422            .with_text("first ")
423            .with_tool_call(ToolCall::new("call_a", "plan", json!({"n": 1})))
424            .with_text("second")
425            .with_tool_call(ToolCall::new("call_b", "plan", json!({"n": 2})));
426        assert_eq!(built.text(), "first second");
427        let calls = built.tool_calls();
428        assert_eq!(calls.len(), 2);
429        assert_eq!(calls[0].id.as_str(), "call_a");
430        assert_eq!(calls[1].id.as_str(), "call_b");
431        assert!(built.tool_call(&CallId::from("call_b")).is_some());
432        assert!(built.tool_call(&CallId::from("call_z")).is_none());
433        assert!(!built.is_empty());
434        assert_eq!(built.reference().to_string(), "openai/gpt-4o");
435    }
436
437    #[test]
438    fn single_json_takes_the_only_tool_call_and_ignores_the_preamble() {
439        let built = response()
440            .with_text("Certo, ecco il piano.")
441            .with_tool_call(ToolCall::new("call_a", "plan", json!({"acts": []})))
442            .with_finish(FinishReason::ToolCalls);
443        assert_eq!(built.single_json().unwrap(), json!({"acts": []}));
444    }
445
446    #[test]
447    fn single_json_refuses_to_choose_between_two_calls() {
448        let built = response()
449            .with_tool_call(ToolCall::new("a", "plan", json!({"n": 1})))
450            .with_tool_call(ToolCall::new("b", "plan", json!({"n": 2})))
451            .with_finish(FinishReason::ToolCalls);
452        assert!(matches!(
453            built.single_json(),
454            Err(StructuredOutputError::MultipleCandidates { candidates: 2 })
455        ));
456    }
457
458    #[test]
459    fn single_json_parses_text_and_strips_one_fence() {
460        let plain = response().with_text(" {\"a\": 1} ");
461        assert_eq!(plain.single_json().unwrap(), json!({"a": 1}));
462
463        let fenced = response().with_text("```json\n{\"a\": 1}\n```");
464        assert_eq!(fenced.single_json().unwrap(), json!({"a": 1}));
465
466        let bare_fence = response().with_text("```\n{\"a\": 1}\n```");
467        assert_eq!(bare_fence.single_json().unwrap(), json!({"a": 1}));
468
469        // A fence-looking prefix that is not a fence is left alone and fails.
470        let not_a_fence = response().with_text("```json {\"a\": 1}");
471        assert!(matches!(
472            not_a_fence.single_json(),
473            Err(StructuredOutputError::NotJson { .. })
474        ));
475    }
476
477    #[test]
478    fn single_json_never_parses_an_incomplete_answer() {
479        let truncated = response()
480            .with_text("{\"a\": 1")
481            .with_finish(FinishReason::MaxTokens);
482        assert!(matches!(
483            truncated.single_json(),
484            Err(StructuredOutputError::NoOutput)
485        ));
486
487        let filtered = response()
488            .with_text("{\"a\": 1}")
489            .with_finish(FinishReason::ContentFilter);
490        assert!(matches!(
491            filtered.single_json(),
492            Err(StructuredOutputError::NoOutput)
493        ));
494
495        let refused = response()
496            .with_text("I cannot help with that.")
497            .with_finish(FinishReason::Refusal);
498        assert!(matches!(
499            refused.single_json(),
500            Err(StructuredOutputError::Refusal)
501        ));
502
503        let empty = response();
504        assert!(empty.is_empty());
505        assert!(matches!(
506            empty.single_json(),
507            Err(StructuredOutputError::NoOutput)
508        ));
509    }
510
511    #[test]
512    fn finish_reasons_say_whether_output_is_complete() {
513        assert!(FinishReason::Stop.is_complete());
514        assert!(FinishReason::ToolCalls.is_complete());
515        for incomplete in [
516            FinishReason::MaxTokens,
517            FinishReason::ContentFilter,
518            FinishReason::Refusal,
519            FinishReason::Other,
520        ] {
521            assert!(!incomplete.is_complete(), "{incomplete}");
522        }
523    }
524
525    #[test]
526    fn usage_totals_and_reports_absence() {
527        assert!(TokenUsage::none().is_unreported());
528        let usage = TokenUsage::new(100, 20)
529            .with_cached_input(80)
530            .with_reasoning(5);
531        assert_eq!(usage.total(), 120);
532        assert_eq!(usage.cached_input, 80);
533        assert_eq!(usage.reasoning, 5);
534        assert!(!usage.is_unreported());
535    }
536
537    #[test]
538    fn responses_round_trip_through_serde() {
539        let built = response()
540            .with_text("hi")
541            .with_tool_call(ToolCall::new("c1", "plan", json!({})))
542            .with_finish(FinishReason::ToolCalls)
543            .with_usage(TokenUsage::new(10, 3))
544            .with_raw_id("resp_123")
545            .with_latency(Duration::from_millis(420))
546            .with_warning(ResponseWarning::Reconstructed);
547        let json = serde_json::to_string(&built).unwrap();
548        let back: ModelResponse = serde_json::from_str(&json).unwrap();
549        assert_eq!(back, built);
550    }
551}