Skip to main content

openai_interface/chat/create/
response.rs

1//! This module provides structures for streaming and non-streaming
2//! chat completion responses.
3
4pub mod streaming {
5    //! Streaming chat completion response.
6
7    use serde::Deserialize;
8
9    use crate::chat::ServiceTier;
10
11    #[derive(Debug, Deserialize, Clone)]
12    pub struct ChatCompletionChunk {
13        /// A unique identifier for the chat completion.
14        pub id: String,
15        /// A list of chat completion choices. Can be more than one
16        /// if `n` is greater than 1. Can also be empty for the last chunk if you set
17        /// `stream_options: {"include_usage": true}`.
18        pub choices: Vec<CompletionChunkChoice>,
19        /// The Unix timestamp (in seconds) of when the chat completion was created.
20        /// Each chunk has the same timestamp.
21        pub created: u64,
22        /// The model used for the chat completion.
23        pub model: String,
24        /// The object type, which is always `chat.completion.chunk`
25        pub object: ChatCompletionChunkObject,
26        /// Specifies the processing type used for serving the request.
27        ///
28        /// When the `service_tier` parameter is set, the response body will include the
29        /// `service_tier` value based on the processing mode actually used to serve the
30        /// request. This response value may be different from the value set in the
31        /// request parameter.
32        pub service_tier: Option<ServiceTier>,
33        /// This fingerprint represents the backend configuration that the model runs with.
34        /// Can be used in conjunction with the `seed` request parameter to understand when
35        /// backend changes have been made that might impact determinism.
36        pub system_fingerprint: Option<String>,
37        /// An optional field that will only be present when you set
38        /// `stream_options: {"include_usage": true}` in your request. When present, it
39        /// contains a null value **except for the last chunk** which contains the token
40        /// usage statistics for the entire request.
41        ///
42        /// **NOTE:** If the stream is interrupted or cancelled, you may not receive the
43        /// final usage chunk which contains the total token usage for the request.
44        pub usage: Option<CompletionUsage>,
45        /// Moderation results for the request input and generated output.
46        ///
47        /// Present on the moderation chunk when moderated completions are
48        /// requested via the `moderation` request parameter.
49        pub moderation: Option<crate::chat::ChatModeration>,
50    }
51
52    #[derive(Debug, Deserialize, Clone)]
53    pub enum ChatCompletionChunkObject {
54        #[serde(rename = "chat.completion.chunk")]
55        ChatCompletionChunk,
56    }
57
58    #[derive(Debug, Deserialize, Clone)]
59    pub struct CompletionChunkChoice {
60        /// A chat completion delta generated by streamed model responses.
61        pub delta: ChoiceDelta,
62        /// The index of the choice in the list of choices.
63        pub index: u32,
64        /// Log probability information for the choice.
65        pub logprobs: Option<ChoiceLogprobs>,
66        /// The reason the model stopped generating tokens.
67        ///
68        /// This will be `stop` if the model hit a natural stop point or a provided stop
69        /// sequence, `length` if the maximum number of tokens specified in the request was
70        /// reached, `content_filter` if content was omitted due to a flag from our content
71        /// filters, `tool_calls` if the model called a tool, or `function_call`
72        /// (deprecated) if the model called a function.
73        pub finish_reason: Option<FinishReason>,
74    }
75
76    #[derive(Debug, Deserialize, Clone)]
77    #[serde(rename_all = "snake_case")]
78    pub enum FinishReason {
79        /// The maximum number of tokens specified in the request was reached.
80        Length,
81        /// The model hit a natural stop point or a provided stop sequence.
82        Stop,
83        /// Content was omitted due to a flag from our content filters.
84        ContentFilter,
85        /// The model called a function (deprecated).
86        FunctionCall,
87        /// The model called a tool.
88        ToolCalls,
89        /// DeepSeek: the request is interrupted due to insufficient resource
90        /// of the inference system.
91        #[cfg(feature = "deepseek")]
92        InsufficientSystemResource,
93    }
94
95    #[derive(Debug, Deserialize, Clone)]
96    pub struct ChoiceDelta {
97        /// The contents of the chunk message.
98        pub content: Option<String>,
99        /// DeepSeek: the reasoning contents of the chunk message. Only
100        /// present for thinking models.
101        #[cfg(feature = "deepseek")]
102        pub reasoning_content: Option<String>,
103        /// Deprecated and replaced by `tool_calls`.
104        ///
105        /// The name and arguments of a function that should be called, as generated by the
106        /// model.
107        pub function_call: Option<ChoiceDeltaFunctionCall>,
108        /// The refusal message generated by the model.
109        pub refusal: Option<String>,
110        /// The role of the author of this message.
111        pub role: Option<CompletionRole>,
112        /// A list of tool calls generated by the model, such as function calls.
113        pub tool_calls: Option<Vec<ChoiceDeltaToolCall>>,
114    }
115
116    #[derive(Debug, Deserialize, Clone)]
117    pub struct ChoiceDeltaToolCallFunction {
118        /// The arguments to call the function with, as generated by the model in JSON
119        /// format. Note that the model does not always generate valid JSON, and may
120        /// hallucinate parameters not defined by your function schema. Validate the
121        /// arguments in your code before calling your function.
122        pub arguments: Option<String>,
123        /// The name of the function to call.
124        pub name: Option<String>,
125    }
126
127    #[derive(Debug, Deserialize, Clone)]
128    pub struct ChoiceDeltaFunctionCall {
129        /// The arguments to call the function with, as generated by the model in JSON
130        /// format. Note that the model does not always generate valid JSON, and may
131        /// hallucinate parameters not defined by your function schema. Validate the
132        /// arguments in your code before calling your function.
133        pub arguments: Option<String>,
134        /// The name of the function to call.
135        pub name: Option<String>,
136    }
137
138    #[derive(Debug, Deserialize, Clone)]
139    pub struct ChoiceDeltaToolCall {
140        /// The index of the tool call in the list of tool calls.
141        pub index: usize,
142        /// The ID of the tool call.
143        pub id: Option<String>,
144        /// The function that the model called.
145        pub function: Option<ChoiceDeltaToolCallFunction>,
146        /// The type of the tool. Currently, only `function` is supported.
147        #[serde(rename = "type")]
148        pub type_: Option<ChoiceDeltaToolCallType>,
149    }
150
151    #[derive(Debug, Deserialize, Clone)]
152    #[serde(rename_all = "snake_case")]
153    pub enum ChoiceDeltaToolCallType {
154        Function,
155    }
156
157    #[derive(Debug, Deserialize, Clone)]
158    #[serde(rename_all = "snake_case")]
159    pub enum CompletionRole {
160        Assistant,
161        Developer,
162        System,
163        Tool,
164        User,
165    }
166
167    /// Log probability information for a choice.
168    #[derive(Debug, Deserialize, Clone)]
169    pub struct ChoiceLogprobs {
170        /// A list of message content tokens with log probability information.
171        pub content: Option<Vec<LogprobeContent>>,
172        /// DeepSeek: a list of reasoning content tokens with log probability
173        /// information. Only present for thinking models.
174        #[cfg(feature = "deepseek")]
175        pub reasoning_content: Option<Vec<LogprobeContent>>,
176        /// A list of message refusal tokens with log probability information.
177        pub refusal: Option<Vec<LogprobeContent>>,
178    }
179
180    /// A list of message content tokens with log probability information.
181    #[derive(Debug, Deserialize, Clone)]
182    pub struct LogprobeContent {
183        pub token: String,
184        pub logprob: f32,
185        pub bytes: Option<Vec<u8>>,
186        pub top_logprobs: Vec<TopLogprob>,
187    }
188
189    /// List of the most likely tokens and their log probability, at this
190    /// token position. In rare cases, there may be fewer than the number of requested top_logprobs returned.
191    #[derive(Debug, Deserialize, Clone)]
192    pub struct TopLogprob {
193        pub token: String,
194        pub logprob: f32,
195        pub bytes: Option<Vec<u8>>,
196    }
197
198    #[derive(Debug, Deserialize, Clone)]
199    pub struct CompletionUsage {
200        /// Number of tokens in the generated completion.
201        pub completion_tokens: usize,
202        /// Number of tokens in the prompt.
203        pub prompt_tokens: usize,
204
205        /// DeepSeek: number of tokens in the prompt that hits the context cache.
206        #[cfg(feature = "deepseek")]
207        pub prompt_cache_hit_tokens: Option<usize>,
208        /// DeepSeek: number of tokens in the prompt that misses the context cache.
209        #[cfg(feature = "deepseek")]
210        pub prompt_cache_miss_tokens: Option<usize>,
211
212        /// Total number of tokens used in the request (prompt + completion).
213        pub total_tokens: usize,
214        /// Breakdown of tokens used in a completion.
215        pub completion_tokens_details: Option<CompletionTokensDetails>,
216        /// Breakdown of tokens used in the prompt.
217        pub prompt_tokens_details: Option<PromptTokensDetails>,
218    }
219
220    #[derive(Debug, Deserialize, Clone)]
221    pub struct CompletionTokensDetails {
222        /// When using Predicted Outputs, the number of tokens in the prediction that
223        /// appeared in the completion.
224        pub accepted_prediction_tokens: Option<usize>,
225        /// Audio input tokens generated by the model.
226        pub audio_tokens: Option<usize>,
227        /// Tokens generated by the model for reasoning.
228        pub reasoning_tokens: Option<usize>,
229        /// When using Predicted Outputs, the number of tokens in the prediction that did
230        /// not appear in the completion. However, like reasoning tokens, these tokens are
231        /// still counted in the total completion tokens for purposes of billing, output,
232        /// and context window limits.
233        pub rejected_prediction_tokens: Option<usize>,
234    }
235
236    #[derive(Debug, Deserialize, Clone)]
237    pub struct PromptTokensDetails {
238        /// Audio input tokens present in the prompt.
239        pub audio_tokens: Option<usize>,
240        /// Cached tokens present in the prompt.
241        pub cached_tokens: Option<usize>,
242    }
243
244    crate::impl_from_str!(ChatCompletionChunk);
245
246    #[cfg(test)]
247    mod test {
248        use std::str::FromStr;
249
250        use super::*;
251
252        #[test]
253        fn streaming_example_deepseek() {
254            let streams = vec![
255                r#"{"id": "1f633d8bfc032625086f14113c411638", "choices": [{"index": 0, "delta": {"content": "", "role": "assistant"}, "finish_reason": null, "logprobs": null}], "created": 1718345013, "model": "deepseek-chat", "system_fingerprint": "fp_a49d71b8a1", "object": "chat.completion.chunk", "usage": null}"#,
256                r#"{"choices": [{"delta": {"content": "Hello", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
257                r#"{"choices": [{"delta": {"content": "!", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
258                r#"{"choices": [{"delta": {"content": " How", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
259                r#"{"choices": [{"delta": {"content": " can", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
260                r#"{"choices": [{"delta": {"content": " I", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
261                r#"{"choices": [{"delta": {"content": " assist", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
262                r#"{"choices": [{"delta": {"content": " you", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
263                r#"{"choices": [{"delta": {"content": " today", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
264                r#"{"choices": [{"delta": {"content": "?", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
265                r#"{"choices": [{"delta": {"content": "", "role": null}, "finish_reason": "stop", "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1", "usage": {"completion_tokens": 9, "prompt_tokens": 17, "total_tokens": 26}}"#,
266            ];
267
268            for stream in streams {
269                let parsed = ChatCompletionChunk::from_str(stream);
270                match parsed {
271                    Ok(completion) => {
272                        println!("Deserialized: {:#?}", completion);
273                    }
274                    Err(e) => {
275                        panic!("Failed to deserialize {}: {}", stream, e);
276                    }
277                }
278            }
279        }
280
281        #[test]
282        fn streaming_example_qwen() {
283            let streams = vec![
284                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":"assistant","tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
285                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"我是","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
286                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"来自","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
287                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"阿里","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
288                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"云的超大规模","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
289                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"语言","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
290                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"模型","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
291                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
292                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"问。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
293                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":"stop","index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
294                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":{"completion_tokens":17,"prompt_tokens":22,"total_tokens":39,"completion_tokens_details":null,"prompt_tokens_details":{"audio_tokens":null,"cached_tokens":0}}}"#,
295            ];
296
297            for stream in streams {
298                let parsed = ChatCompletionChunk::from_str(stream);
299                match parsed {
300                    Ok(completion) => {
301                        println!("Deserialized: {:#?}", completion);
302                    }
303                    Err(e) => {
304                        panic!("Failed to deserialize {}: {}", stream, e);
305                    }
306                }
307            }
308        }
309    }
310}
311
312pub mod no_streaming {
313    //! Non-streaming chat completion response.
314
315    /// Alias for `crate::chat::ChatCompletion`, which is shared
316    /// by many other modules. This alias is for compatibility.
317    pub type ChatCompletion = crate::chat::ChatCompletion;
318}