Skip to main content

openai_interface/chat/create/
response.rs

1//! This module provides structures for streaming and non-streaming
2//! chat completion responses.
3
4pub mod streaming {
5    //! Streaming chat completion response.
6
7    use serde::Deserialize;
8
9    use crate::chat::ServiceTier;
10
11    #[derive(Debug, Deserialize, Clone)]
12    pub struct ChatCompletionChunk {
13        /// A unique identifier for the chat completion.
14        pub id: String,
15        /// A list of chat completion choices. Can be more than one
16        /// if `n` is greater than 1. Can also be empty for the last chunk if you set
17        /// `stream_options: {"include_usage": true}`.
18        pub choices: Vec<CompletionChunkChoice>,
19        /// The Unix timestamp (in seconds) of when the chat completion was created.
20        /// Each chunk has the same timestamp.
21        pub created: u64,
22        /// The model used for the chat completion.
23        pub model: String,
24        /// The object type, which is always `chat.completion.chunk`
25        pub object: ChatCompletionChunkObject,
26        /// Specifies the processing type used for serving the request.
27        ///
28        /// When the `service_tier` parameter is set, the response body will include the
29        /// `service_tier` value based on the processing mode actually used to serve the
30        /// request. This response value may be different from the value set in the
31        /// request parameter.
32        pub service_tier: Option<ServiceTier>,
33        /// This fingerprint represents the backend configuration that the model runs with.
34        /// Can be used in conjunction with the `seed` request parameter to understand when
35        /// backend changes have been made that might impact determinism.
36        pub system_fingerprint: Option<String>,
37        /// An optional field that will only be present when you set
38        /// `stream_options: {"include_usage": true}` in your request. When present, it
39        /// contains a null value **except for the last chunk** which contains the token
40        /// usage statistics for the entire request.
41        ///
42        /// **NOTE:** If the stream is interrupted or cancelled, you may not receive the
43        /// final usage chunk which contains the total token usage for the request.
44        pub usage: Option<CompletionUsage>,
45        /// Moderation results for the request input and generated output.
46        ///
47        /// Present on the moderation chunk when moderated completions are
48        /// requested via the `moderation` request parameter.
49        pub moderation: Option<crate::chat::ChatModeration>,
50    }
51
52    #[derive(Debug, Deserialize, Clone)]
53    pub enum ChatCompletionChunkObject {
54        #[serde(rename = "chat.completion.chunk")]
55        ChatCompletionChunk,
56    }
57
58    #[derive(Debug, Deserialize, Clone)]
59    pub struct CompletionChunkChoice {
60        /// A chat completion delta generated by streamed model responses.
61        pub delta: ChoiceDelta,
62        /// The index of the choice in the list of choices.
63        pub index: u32,
64        /// Log probability information for the choice.
65        pub logprobs: Option<ChoiceLogprobs>,
66        /// The reason the model stopped generating tokens.
67        ///
68        /// This will be `stop` if the model hit a natural stop point or a provided stop
69        /// sequence, `length` if the maximum number of tokens specified in the request was
70        /// reached, `content_filter` if content was omitted due to a flag from our content
71        /// filters, `tool_calls` if the model called a tool, or `function_call`
72        /// (deprecated) if the model called a function.
73        pub finish_reason: Option<FinishReason>,
74    }
75
76    #[derive(Debug, Deserialize, Clone)]
77    #[serde(rename_all = "snake_case")]
78    pub enum FinishReason {
79        /// The maximum number of tokens specified in the request was reached.
80        Length,
81        /// The model hit a natural stop point or a provided stop sequence.
82        Stop,
83        /// Content was omitted due to a flag from our content filters.
84        ContentFilter,
85        /// The model called a function (deprecated).
86        FunctionCall,
87        /// The model called a tool.
88        ToolCalls,
89        /// DeepSeek: the request is interrupted due to insufficient resource
90        /// of the inference system.
91        #[cfg(feature = "deepseek")]
92        InsufficientSystemResource,
93    }
94
95    #[derive(Debug, Deserialize, Clone)]
96    pub struct ChoiceDelta {
97        /// The contents of the chunk message.
98        pub content: Option<String>,
99        /// DeepSeek: the reasoning contents of the chunk message. Only
100        /// present for thinking models.
101        #[cfg(feature = "deepseek")]
102        pub reasoning_content: Option<String>,
103        /// Deprecated and replaced by `tool_calls`.
104        ///
105        /// The name and arguments of a function that should be called, as generated by the
106        /// model.
107        pub function_call: Option<ChoiceDeltaFunctionCall>,
108        /// The refusal message generated by the model.
109        pub refusal: Option<String>,
110        /// The role of the author of this message.
111        pub role: Option<CompletionRole>,
112        /// A list of tool calls generated by the model, such as function calls.
113        pub tool_calls: Option<Vec<ChoiceDeltaToolCall>>,
114        /// Annotations for the chunk message, such as URL citations emitted
115        /// by search deployments.
116        ///
117        /// Not part of the official OpenAI chunk schema, but Azure OpenAI
118        /// ("on your data" / search deployments) does stream `annotations`
119        /// inside `delta`, and dropping them silently loses the citations.
120        /// Gated behind the `azure` cargo feature, following the crate's
121        /// convention for provider-specific fields.
122        #[cfg(feature = "azure")]
123        pub annotations: Option<Vec<crate::chat::Annotation>>,
124        /// Data about a streamed audio response from the model.
125        ///
126        /// Not part of the official OpenAI chunk schema, but audio-capable
127        /// Azure OpenAI deployments stream `audio` inside `delta`. Gated
128        /// behind the `azure` cargo feature, following the crate's
129        /// convention for provider-specific fields.
130        #[cfg(feature = "azure")]
131        pub audio: Option<crate::chat::ChatCompletionAudio>,
132    }
133
134    #[derive(Debug, Deserialize, Clone)]
135    pub struct ChoiceDeltaToolCallFunction {
136        /// The arguments to call the function with, as generated by the model in JSON
137        /// format. Note that the model does not always generate valid JSON, and may
138        /// hallucinate parameters not defined by your function schema. Validate the
139        /// arguments in your code before calling your function.
140        pub arguments: Option<String>,
141        /// The name of the function to call.
142        pub name: Option<String>,
143    }
144
145    #[derive(Debug, Deserialize, Clone)]
146    pub struct ChoiceDeltaFunctionCall {
147        /// The arguments to call the function with, as generated by the model in JSON
148        /// format. Note that the model does not always generate valid JSON, and may
149        /// hallucinate parameters not defined by your function schema. Validate the
150        /// arguments in your code before calling your function.
151        pub arguments: Option<String>,
152        /// The name of the function to call.
153        pub name: Option<String>,
154    }
155
156    #[derive(Debug, Deserialize, Clone)]
157    pub struct ChoiceDeltaToolCall {
158        /// The index of the tool call in the list of tool calls.
159        pub index: usize,
160        /// The ID of the tool call.
161        pub id: Option<String>,
162        /// The function that the model called.
163        pub function: Option<ChoiceDeltaToolCallFunction>,
164        /// The type of the tool. Currently, only `function` is supported.
165        #[serde(rename = "type")]
166        pub type_: Option<ChoiceDeltaToolCallType>,
167    }
168
169    #[derive(Debug, Deserialize, Clone)]
170    #[serde(rename_all = "snake_case")]
171    pub enum ChoiceDeltaToolCallType {
172        Function,
173    }
174
175    #[derive(Debug, Deserialize, Clone)]
176    #[serde(rename_all = "snake_case")]
177    pub enum CompletionRole {
178        Assistant,
179        Developer,
180        System,
181        Tool,
182        User,
183    }
184
185    /// Log probability information for a choice.
186    #[derive(Debug, Deserialize, Clone)]
187    pub struct ChoiceLogprobs {
188        /// A list of message content tokens with log probability information.
189        pub content: Option<Vec<LogprobeContent>>,
190        /// DeepSeek: a list of reasoning content tokens with log probability
191        /// information. Only present for thinking models.
192        #[cfg(feature = "deepseek")]
193        pub reasoning_content: Option<Vec<LogprobeContent>>,
194        /// A list of message refusal tokens with log probability information.
195        pub refusal: Option<Vec<LogprobeContent>>,
196    }
197
198    /// A list of message content tokens with log probability information.
199    #[derive(Debug, Deserialize, Clone)]
200    pub struct LogprobeContent {
201        pub token: String,
202        pub logprob: f32,
203        pub bytes: Option<Vec<u8>>,
204        pub top_logprobs: Vec<TopLogprob>,
205    }
206
207    /// List of the most likely tokens and their log probability, at this
208    /// token position. In rare cases, there may be fewer than the number of requested top_logprobs returned.
209    #[derive(Debug, Deserialize, Clone)]
210    pub struct TopLogprob {
211        pub token: String,
212        pub logprob: f32,
213        pub bytes: Option<Vec<u8>>,
214    }
215
216    #[derive(Debug, Deserialize, Clone)]
217    pub struct CompletionUsage {
218        /// Number of tokens in the generated completion.
219        pub completion_tokens: usize,
220        /// Number of tokens in the prompt.
221        pub prompt_tokens: usize,
222
223        /// DeepSeek: number of tokens in the prompt that hits the context cache.
224        #[cfg(feature = "deepseek")]
225        pub prompt_cache_hit_tokens: Option<usize>,
226        /// DeepSeek: number of tokens in the prompt that misses the context cache.
227        #[cfg(feature = "deepseek")]
228        pub prompt_cache_miss_tokens: Option<usize>,
229
230        /// Total number of tokens used in the request (prompt + completion).
231        pub total_tokens: usize,
232        /// Breakdown of tokens used in a completion.
233        pub completion_tokens_details: Option<CompletionTokensDetails>,
234        /// Breakdown of tokens used in the prompt.
235        pub prompt_tokens_details: Option<PromptTokensDetails>,
236    }
237
238    #[derive(Debug, Deserialize, Clone)]
239    pub struct CompletionTokensDetails {
240        /// When using Predicted Outputs, the number of tokens in the prediction that
241        /// appeared in the completion.
242        pub accepted_prediction_tokens: Option<usize>,
243        /// Audio input tokens generated by the model.
244        pub audio_tokens: Option<usize>,
245        /// Tokens generated by the model for reasoning.
246        pub reasoning_tokens: Option<usize>,
247        /// When using Predicted Outputs, the number of tokens in the prediction that did
248        /// not appear in the completion. However, like reasoning tokens, these tokens are
249        /// still counted in the total completion tokens for purposes of billing, output,
250        /// and context window limits.
251        pub rejected_prediction_tokens: Option<usize>,
252    }
253
254    #[derive(Debug, Deserialize, Clone)]
255    pub struct PromptTokensDetails {
256        /// Audio input tokens present in the prompt.
257        pub audio_tokens: Option<usize>,
258        /// Cached tokens present in the prompt.
259        pub cached_tokens: Option<usize>,
260    }
261
262    crate::impl_from_str!(ChatCompletionChunk);
263
264    #[cfg(test)]
265    mod test {
266        use std::str::FromStr;
267
268        use super::*;
269
270        #[test]
271        fn streaming_example_deepseek() {
272            let streams = vec![
273                r#"{"id": "1f633d8bfc032625086f14113c411638", "choices": [{"index": 0, "delta": {"content": "", "role": "assistant"}, "finish_reason": null, "logprobs": null}], "created": 1718345013, "model": "deepseek-chat", "system_fingerprint": "fp_a49d71b8a1", "object": "chat.completion.chunk", "usage": null}"#,
274                r#"{"choices": [{"delta": {"content": "Hello", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
275                r#"{"choices": [{"delta": {"content": "!", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
276                r#"{"choices": [{"delta": {"content": " How", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
277                r#"{"choices": [{"delta": {"content": " can", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
278                r#"{"choices": [{"delta": {"content": " I", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
279                r#"{"choices": [{"delta": {"content": " assist", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
280                r#"{"choices": [{"delta": {"content": " you", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
281                r#"{"choices": [{"delta": {"content": " today", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
282                r#"{"choices": [{"delta": {"content": "?", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
283                r#"{"choices": [{"delta": {"content": "", "role": null}, "finish_reason": "stop", "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1", "usage": {"completion_tokens": 9, "prompt_tokens": 17, "total_tokens": 26}}"#,
284            ];
285
286            for stream in streams {
287                let parsed = ChatCompletionChunk::from_str(stream);
288                match parsed {
289                    Ok(completion) => {
290                        println!("Deserialized: {:#?}", completion);
291                    }
292                    Err(e) => {
293                        panic!("Failed to deserialize {}: {}", stream, e);
294                    }
295                }
296            }
297        }
298
299        #[test]
300        fn streaming_example_qwen() {
301            let streams = vec![
302                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":"assistant","tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
303                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"我是","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
304                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"来自","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
305                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"阿里","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
306                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"云的超大规模","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
307                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"语言","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
308                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"模型","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
309                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
310                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"问。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
311                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":"stop","index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
312                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":{"completion_tokens":17,"prompt_tokens":22,"total_tokens":39,"completion_tokens_details":null,"prompt_tokens_details":{"audio_tokens":null,"cached_tokens":0}}}"#,
313            ];
314
315            for stream in streams {
316                let parsed = ChatCompletionChunk::from_str(stream);
317                match parsed {
318                    Ok(completion) => {
319                        println!("Deserialized: {:#?}", completion);
320                    }
321                    Err(e) => {
322                        panic!("Failed to deserialize {}: {}", stream, e);
323                    }
324                }
325            }
326        }
327
328        /// Azure OpenAI streams `annotations` (URL citations from "on your
329        /// data" deployments) and `audio` inside `delta`, outside the
330        /// official chunk schema; both are gated behind the `azure` feature
331        /// and must deserialize instead of being dropped.
332        #[cfg(feature = "azure")]
333        #[test]
334        fn streaming_example_azure_annotations_and_audio() {
335            let chunk = ChatCompletionChunk::from_str(
336                r#"{"id":"chatcmpl-abc","choices":[{"delta":{"content":"According to the doc","annotations":[{"type":"url_citation","url_citation":{"start_index":0,"end_index":20,"title":"Azure Docs","url":"https://learn.microsoft.com/azure"}}],"audio":{"id":"audio_abc","data":"SGVsbG8=","expires_at":1735113344,"transcript":"Hello"}},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"gpt-4o","object":"chat.completion.chunk","system_fingerprint":"fp_abc"}"#,
337            )
338            .expect("chunk with annotations and audio must deserialize");
339
340            let delta = &chunk.choices[0].delta;
341            let annotations = delta.annotations.as_ref().expect("annotations");
342            assert_eq!(annotations.len(), 1);
343            assert_eq!(annotations[0].url_citation.title, "Azure Docs");
344            assert_eq!(
345                annotations[0].url_citation.url,
346                "https://learn.microsoft.com/azure"
347            );
348
349            let audio = delta.audio.as_ref().expect("audio");
350            assert_eq!(audio.id, "audio_abc");
351            assert_eq!(audio.transcript, "Hello");
352        }
353    }
354}
355
356pub mod no_streaming {
357    //! Non-streaming chat completion response.
358
359    /// Alias for `crate::chat::ChatCompletion`, which is shared
360    /// by many other modules. This alias is for compatibility.
361    pub type ChatCompletion = crate::chat::ChatCompletion;
362}