Skip to main content

openai_interface/chat/create/
response.rs

1//! This module provides structures for streaming and non-streaming
2//! chat completion responses.
3
4pub mod streaming {
5    //! Streaming chat completion response.
6
7    use serde::Deserialize;
8
9    use crate::chat::ServiceTier;
10
11    #[derive(Debug, Deserialize, Clone)]
12    pub struct ChatCompletionChunk {
13        /// A unique identifier for the chat completion.
14        pub id: String,
15        /// A list of chat completion choices. Can be more than one
16        /// if `n` is greater than 1. Can also be empty for the last chunk if you set
17        /// `stream_options: {"include_usage": true}`.
18        pub choices: Vec<CompletionChunkChoice>,
19        /// The Unix timestamp (in seconds) of when the chat completion was created.
20        /// Each chunk has the same timestamp.
21        pub created: u64,
22        /// The model used for the chat completion.
23        pub model: String,
24        /// The object type, which is always `chat.completion.chunk`
25        pub object: ChatCompletionChunkObject,
26        /// Specifies the processing type used for serving the request.
27        ///
28        /// When the `service_tier` parameter is set, the response body will include the
29        /// `service_tier` value based on the processing mode actually used to serve the
30        /// request. This response value may be different from the value set in the
31        /// request parameter.
32        pub service_tier: Option<ServiceTier>,
33        /// This fingerprint represents the backend configuration that the model runs with.
34        /// Can be used in conjunction with the `seed` request parameter to understand when
35        /// backend changes have been made that might impact determinism.
36        pub system_fingerprint: Option<String>,
37        /// An optional field that will only be present when you set
38        /// `stream_options: {"include_usage": true}` in your request. When present, it
39        /// contains a null value **except for the last chunk** which contains the token
40        /// usage statistics for the entire request.
41        ///
42        /// **NOTE:** If the stream is interrupted or cancelled, you may not receive the
43        /// final usage chunk which contains the total token usage for the request.
44        pub usage: Option<CompletionUsage>,
45    }
46
47    #[derive(Debug, Deserialize, Clone)]
48    pub enum ChatCompletionChunkObject {
49        #[serde(rename = "chat.completion.chunk")]
50        ChatCompletionChunk,
51    }
52
53    #[derive(Debug, Deserialize, Clone)]
54    pub struct CompletionChunkChoice {
55        /// A chat completion delta generated by streamed model responses.
56        pub delta: ChoiceDelta,
57        /// The index of the choice in the list of choices.
58        pub index: u32,
59        /// Log probability information for the choice.
60        pub logprobs: Option<ChoiceLogprobs>,
61        /// The reason the model stopped generating tokens.
62        ///
63        /// This will be `stop` if the model hit a natural stop point or a provided stop
64        /// sequence, `length` if the maximum number of tokens specified in the request was
65        /// reached, `content_filter` if content was omitted due to a flag from our content
66        /// filters, `tool_calls` if the model called a tool, or `function_call`
67        /// (deprecated) if the model called a function.
68        pub finish_reason: Option<FinishReason>,
69    }
70
71    #[derive(Debug, Deserialize, Clone)]
72    #[serde(rename_all = "snake_case")]
73    pub enum FinishReason {
74        /// The maximum number of tokens specified in the request was reached.
75        Length,
76        /// The model hit a natural stop point or a provided stop sequence.
77        Stop,
78        /// Content was omitted due to a flag from our content filters.
79        ContentFilter,
80        /// The model called a function (deprecated).
81        FunctionCall,
82        /// The model called a tool.
83        ToolCalls,
84        /// DeepSeek: the request is interrupted due to insufficient resource
85        /// of the inference system.
86        #[cfg(feature = "deepseek")]
87        InsufficientSystemResource,
88    }
89
90    #[derive(Debug, Deserialize, Clone)]
91    pub struct ChoiceDelta {
92        /// The contents of the chunk message.
93        pub content: Option<String>,
94        /// DeepSeek: the reasoning contents of the chunk message. Only
95        /// present for thinking models.
96        #[cfg(feature = "deepseek")]
97        pub reasoning_content: Option<String>,
98        /// Deprecated and replaced by `tool_calls`.
99        ///
100        /// The name and arguments of a function that should be called, as generated by the
101        /// model.
102        pub function_call: Option<ChoiceDeltaFunctionCall>,
103        /// The refusal message generated by the model.
104        pub refusal: Option<String>,
105        /// The role of the author of this message.
106        pub role: Option<CompletionRole>,
107        /// A list of tool calls generated by the model, such as function calls.
108        pub tool_calls: Option<Vec<ChoiceDeltaToolCall>>,
109    }
110
111    #[derive(Debug, Deserialize, Clone)]
112    pub struct ChoiceDeltaToolCallFunction {
113        /// The arguments to call the function with, as generated by the model in JSON
114        /// format. Note that the model does not always generate valid JSON, and may
115        /// hallucinate parameters not defined by your function schema. Validate the
116        /// arguments in your code before calling your function.
117        pub arguments: Option<String>,
118        /// The name of the function to call.
119        pub name: Option<String>,
120    }
121
122    #[derive(Debug, Deserialize, Clone)]
123    pub struct ChoiceDeltaFunctionCall {
124        /// The arguments to call the function with, as generated by the model in JSON
125        /// format. Note that the model does not always generate valid JSON, and may
126        /// hallucinate parameters not defined by your function schema. Validate the
127        /// arguments in your code before calling your function.
128        pub arguments: Option<String>,
129        /// The name of the function to call.
130        pub name: Option<String>,
131    }
132
133    #[derive(Debug, Deserialize, Clone)]
134    pub struct ChoiceDeltaToolCall {
135        /// The index of the tool call in the list of tool calls.
136        pub index: usize,
137        /// The ID of the tool call.
138        pub id: Option<String>,
139        /// The function that the model called.
140        pub function: Option<ChoiceDeltaToolCallFunction>,
141        /// The type of the tool. Currently, only `function` is supported.
142        #[serde(rename = "type")]
143        pub type_: Option<ChoiceDeltaToolCallType>,
144    }
145
146    #[derive(Debug, Deserialize, Clone)]
147    #[serde(rename_all = "snake_case")]
148    pub enum ChoiceDeltaToolCallType {
149        Function,
150    }
151
152    #[derive(Debug, Deserialize, Clone)]
153    #[serde(rename_all = "snake_case")]
154    pub enum CompletionRole {
155        Assistant,
156        Developer,
157        System,
158        Tool,
159        User,
160    }
161
162    /// Log probability information for a choice.
163    #[derive(Debug, Deserialize, Clone)]
164    pub struct ChoiceLogprobs {
165        /// A list of message content tokens with log probability information.
166        pub content: Option<Vec<LogprobeContent>>,
167        /// DeepSeek: a list of reasoning content tokens with log probability
168        /// information. Only present for thinking models.
169        #[cfg(feature = "deepseek")]
170        pub reasoning_content: Option<Vec<LogprobeContent>>,
171        /// A list of message refusal tokens with log probability information.
172        pub refusal: Option<Vec<LogprobeContent>>,
173    }
174
175    /// A list of message content tokens with log probability information.
176    #[derive(Debug, Deserialize, Clone)]
177    pub struct LogprobeContent {
178        pub token: String,
179        pub logprob: f32,
180        pub bytes: Option<Vec<u8>>,
181        pub top_logprobs: Vec<TopLogprob>,
182    }
183
184    /// List of the most likely tokens and their log probability, at this
185    /// token position. In rare cases, there may be fewer than the number of requested top_logprobs returned.
186    #[derive(Debug, Deserialize, Clone)]
187    pub struct TopLogprob {
188        pub token: String,
189        pub logprob: f32,
190        pub bytes: Option<Vec<u8>>,
191    }
192
193    #[derive(Debug, Deserialize, Clone)]
194    pub struct CompletionUsage {
195        /// Number of tokens in the generated completion.
196        pub completion_tokens: usize,
197        /// Number of tokens in the prompt.
198        pub prompt_tokens: usize,
199
200        /// DeepSeek: number of tokens in the prompt that hits the context cache.
201        #[cfg(feature = "deepseek")]
202        pub prompt_cache_hit_tokens: Option<usize>,
203        /// DeepSeek: number of tokens in the prompt that misses the context cache.
204        #[cfg(feature = "deepseek")]
205        pub prompt_cache_miss_tokens: Option<usize>,
206
207        /// Total number of tokens used in the request (prompt + completion).
208        pub total_tokens: usize,
209        /// Breakdown of tokens used in a completion.
210        pub completion_tokens_details: Option<CompletionTokensDetails>,
211        /// Breakdown of tokens used in the prompt.
212        pub prompt_tokens_details: Option<PromptTokensDetails>,
213    }
214
215    #[derive(Debug, Deserialize, Clone)]
216    pub struct CompletionTokensDetails {
217        /// When using Predicted Outputs, the number of tokens in the prediction that
218        /// appeared in the completion.
219        pub accepted_prediction_tokens: Option<usize>,
220        /// Audio input tokens generated by the model.
221        pub audio_tokens: Option<usize>,
222        /// Tokens generated by the model for reasoning.
223        pub reasoning_tokens: Option<usize>,
224        /// When using Predicted Outputs, the number of tokens in the prediction that did
225        /// not appear in the completion. However, like reasoning tokens, these tokens are
226        /// still counted in the total completion tokens for purposes of billing, output,
227        /// and context window limits.
228        pub rejected_prediction_tokens: Option<usize>,
229    }
230
231    #[derive(Debug, Deserialize, Clone)]
232    pub struct PromptTokensDetails {
233        /// Audio input tokens present in the prompt.
234        pub audio_tokens: Option<usize>,
235        /// Cached tokens present in the prompt.
236        pub cached_tokens: Option<usize>,
237    }
238
239    crate::impl_from_str!(ChatCompletionChunk);
240
241    #[cfg(test)]
242    mod test {
243        use std::str::FromStr;
244
245        use super::*;
246
247        #[test]
248        fn streaming_example_deepseek() {
249            let streams = vec![
250                r#"{"id": "1f633d8bfc032625086f14113c411638", "choices": [{"index": 0, "delta": {"content": "", "role": "assistant"}, "finish_reason": null, "logprobs": null}], "created": 1718345013, "model": "deepseek-chat", "system_fingerprint": "fp_a49d71b8a1", "object": "chat.completion.chunk", "usage": null}"#,
251                r#"{"choices": [{"delta": {"content": "Hello", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
252                r#"{"choices": [{"delta": {"content": "!", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
253                r#"{"choices": [{"delta": {"content": " How", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
254                r#"{"choices": [{"delta": {"content": " can", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
255                r#"{"choices": [{"delta": {"content": " I", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
256                r#"{"choices": [{"delta": {"content": " assist", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
257                r#"{"choices": [{"delta": {"content": " you", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
258                r#"{"choices": [{"delta": {"content": " today", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
259                r#"{"choices": [{"delta": {"content": "?", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
260                r#"{"choices": [{"delta": {"content": "", "role": null}, "finish_reason": "stop", "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1", "usage": {"completion_tokens": 9, "prompt_tokens": 17, "total_tokens": 26}}"#,
261            ];
262
263            for stream in streams {
264                let parsed = ChatCompletionChunk::from_str(stream);
265                match parsed {
266                    Ok(completion) => {
267                        println!("Deserialized: {:#?}", completion);
268                    }
269                    Err(e) => {
270                        panic!("Failed to deserialize {}: {}", stream, e);
271                    }
272                }
273            }
274        }
275
276        #[test]
277        fn streaming_example_qwen() {
278            let streams = vec![
279                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":"assistant","tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
280                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"我是","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
281                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"来自","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
282                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"阿里","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
283                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"云的超大规模","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
284                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"语言","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
285                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"模型","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
286                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
287                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"问。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
288                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":"stop","index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
289                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":{"completion_tokens":17,"prompt_tokens":22,"total_tokens":39,"completion_tokens_details":null,"prompt_tokens_details":{"audio_tokens":null,"cached_tokens":0}}}"#,
290            ];
291
292            for stream in streams {
293                let parsed = ChatCompletionChunk::from_str(stream);
294                match parsed {
295                    Ok(completion) => {
296                        println!("Deserialized: {:#?}", completion);
297                    }
298                    Err(e) => {
299                        panic!("Failed to deserialize {}: {}", stream, e);
300                    }
301                }
302            }
303        }
304    }
305}
306
307pub mod no_streaming {
308    //! Non-streaming chat completion response.
309
310    /// Alias for `crate::chat::ChatCompletion`, which is shared
311    /// by many other modules. This alias is for compatibility.
312    pub type ChatCompletion = crate::chat::ChatCompletion;
313}