openai-interface 0.13.0

A low-level Rust interface for the OpenAI API
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
//! This module provides structures for streaming and non-streaming
//! chat completion responses.

pub mod streaming {
    //! Streaming chat completion response.

    use serde::{Deserialize, Serialize};

    use crate::chat::ServiceTier;

    /// Deserializes a possibly-null, possibly-missing list as an empty
    /// `Vec`. Some OpenAI-compatible backends (e.g. vLLM) terminate the
    /// stream with a usage-only chunk whose `choices` is `null`; without
    /// this, the chunk — and with it the only copy of the usage statistics —
    /// would fail to deserialize.
    fn null_to_empty_vec<'de, D, T>(deserializer: D) -> Result<Vec<T>, D::Error>
    where
        D: serde::Deserializer<'de>,
        T: Deserialize<'de>,
    {
        Ok(Option::<Vec<T>>::deserialize(deserializer)?.unwrap_or_default())
    }

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChatCompletionChunk {
        /// A unique identifier for the chat completion.
        pub id: String,
        /// A list of chat completion choices. Can be more than one
        /// if `n` is greater than 1. Empty for the final usage-only chunk
        /// (see `stream_options: {"include_usage": true}`); some backends
        /// send that chunk with `"choices": null` or omit the key entirely,
        /// which deserializes as an empty list too.
        #[serde(default, deserialize_with = "null_to_empty_vec")]
        pub choices: Vec<CompletionChunkChoice>,
        /// The Unix timestamp (in seconds) of when the chat completion was created.
        /// Each chunk has the same timestamp.
        pub created: u64,
        /// The model used for the chat completion.
        pub model: String,
        /// The object type, which is always `chat.completion.chunk`.
        ///
        /// `Some` only when the backend sends a recognized value; some
        /// non-OpenAI gateways omit or repurpose the field.
        pub object: Option<ChatCompletionChunkObject>,
        /// Specifies the processing type used for serving the request.
        ///
        /// When the `service_tier` parameter is set, the response body will include the
        /// `service_tier` value based on the processing mode actually used to serve the
        /// request. This response value may be different from the value set in the
        /// request parameter.
        pub service_tier: Option<ServiceTier>,
        /// This fingerprint represents the backend configuration that the model runs with.
        /// Can be used in conjunction with the `seed` request parameter to understand when
        /// backend changes have been made that might impact determinism.
        pub system_fingerprint: Option<String>,
        /// An optional field that will only be present when you set
        /// `stream_options: {"include_usage": true}` in your request. When present, it
        /// contains a null value **except for the last chunk** which contains the token
        /// usage statistics for the entire request.
        ///
        /// **NOTE:** If the stream is interrupted or cancelled, you may not receive the
        /// final usage chunk which contains the total token usage for the request.
        pub usage: Option<CompletionUsage>,
        /// Moderation results for the request input and generated output.
        ///
        /// Present on the moderation chunk when moderated completions are
        /// requested via the `moderation` request parameter.
        pub moderation: Option<crate::chat::ChatModeration>,

        /// vLLM: the prompt's token IDs after chat-template rendering. Sent
        /// on the first chunk only.
        #[cfg(feature = "vllm")]
        pub prompt_token_ids: Option<Vec<u32>>,
        /// vLLM: the fully rendered prompt text. Only sent on the first
        /// chunk, and only when the request set `vllm_chat.return_prompt_text`.
        #[cfg(feature = "vllm")]
        pub prompt_text: Option<String>,
    }

    crate::wire_string_enum! {
        /// The object type, which is always `chat.completion.chunk`.
        pub enum ChatCompletionChunkObject {
            ChatCompletionChunk => "chat.completion.chunk",
        }
    }

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct CompletionChunkChoice {
        /// A chat completion delta generated by streamed model responses.
        pub delta: ChoiceDelta,
        /// The index of the choice in the list of choices.
        pub index: u32,
        /// Log probability information for the choice.
        pub logprobs: Option<ChoiceLogprobs>,
        /// The reason the model stopped generating tokens.
        ///
        /// This will be `stop` if the model hit a natural stop point or a provided stop
        /// sequence, `length` if the maximum number of tokens specified in the request was
        /// reached, `content_filter` if content was omitted due to a flag from our content
        /// filters, `tool_calls` if the model called a tool, or `function_call`
        /// (deprecated) if the model called a function.
        pub finish_reason: Option<FinishReason>,

        /// vLLM: which terminator ended generation — the matched stop
        /// string, or the matched token ID. Not part of the OpenAI chunk
        /// schema.
        #[cfg(feature = "vllm")]
        pub stop_reason: Option<crate::vllm::StopReason>,
        /// vLLM: the generated token IDs for this chunk. Only set when the
        /// request set `vllm_chat.return_token_ids`.
        #[cfg(feature = "vllm")]
        pub token_ids: Option<Vec<u32>>,
    }

    pub use crate::chat::FinishReason;

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChoiceDelta {
        /// The contents of the chunk message.
        pub content: Option<String>,
        /// The reasoning contents of the chunk message. Only present for
        /// thinking models (DeepSeek, Qwen3, and other reasoning models
        /// served by OpenAI-compatible backends).
        #[cfg(feature = "reasoning")]
        pub reasoning_content: Option<String>,
        /// vLLM: the reasoning contents of the chunk message, under the key
        /// vLLM actually streams.
        ///
        /// vLLM serializes the chain of thought as `reasoning`, not
        /// `reasoning_content`, so for a vLLM backend the field above stays
        /// `None`. [`ChatCompletionAccumulator`](crate::chat::create::accumulator::ChatCompletionAccumulator)
        /// accumulates both and exposes this one through its `reasoning`
        /// accessor.
        #[cfg(feature = "vllm")]
        pub reasoning: Option<String>,
        /// Deprecated and replaced by `tool_calls`.
        ///
        /// The name and arguments of a function that should be called, as generated by the
        /// model.
        pub function_call: Option<ChoiceDeltaFunctionCall>,
        /// The refusal message generated by the model.
        pub refusal: Option<String>,
        /// The role of the author of this message.
        pub role: Option<CompletionRole>,
        /// A list of tool calls generated by the model, such as function calls.
        pub tool_calls: Option<Vec<ChoiceDeltaToolCall>>,
        /// Annotations for the chunk message, such as URL citations emitted
        /// by search deployments.
        ///
        /// Not part of the official OpenAI chunk schema, but Azure OpenAI
        /// ("on your data" / search deployments) streams `annotations`
        /// inside `delta`, and dropping them silently loses the citations.
        pub annotations: Option<Vec<crate::chat::Annotation>>,
        /// Data about a streamed audio response from the model.
        ///
        /// Not part of the official OpenAI chunk schema, but audio-capable
        /// Azure OpenAI deployments stream `audio` inside `delta`.
        pub audio: Option<crate::chat::ChatCompletionAudio>,
    }

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChoiceDeltaToolCallFunction {
        /// The arguments to call the function with, as generated by the model in JSON
        /// format. Note that the model does not always generate valid JSON, and may
        /// hallucinate parameters not defined by your function schema. Validate the
        /// arguments in your code before calling your function.
        pub arguments: Option<String>,
        /// The name of the function to call.
        pub name: Option<String>,
    }

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChoiceDeltaFunctionCall {
        /// The arguments to call the function with, as generated by the model in JSON
        /// format. Note that the model does not always generate valid JSON, and may
        /// hallucinate parameters not defined by your function schema. Validate the
        /// arguments in your code before calling your function.
        pub arguments: Option<String>,
        /// The name of the function to call.
        pub name: Option<String>,
    }

    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChoiceDeltaToolCall {
        /// The index of the tool call in the list of tool calls.
        pub index: u32,
        /// The ID of the tool call.
        pub id: Option<String>,
        /// The function that the model called.
        pub function: Option<ChoiceDeltaToolCallFunction>,
        /// The type of the tool. Currently, only `function` is supported.
        #[serde(rename = "type")]
        pub type_: Option<ChoiceDeltaToolCallType>,
    }

    crate::wire_string_enum! {
        /// The type of a streamed tool call.
        pub enum ChoiceDeltaToolCallType {
            /// The tool call invokes a function.
            Function => "function",
            /// The tool call invokes a custom tool.
            Custom => "custom",
        }
    }

    pub use crate::chat::Role as CompletionRole;

    /// Log probability information for a choice.
    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct ChoiceLogprobs {
        /// A list of message content tokens with log probability information.
        pub content: Option<Vec<LogprobeContent>>,
        /// A list of reasoning content tokens with log probability
        /// information. Only present for thinking models.
        #[cfg(feature = "reasoning")]
        pub reasoning_content: Option<Vec<LogprobeContent>>,
        /// A list of message refusal tokens with log probability information.
        pub refusal: Option<Vec<LogprobeContent>>,
    }

    /// A list of message content tokens with log probability information.
    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct LogprobeContent {
        pub token: String,
        pub logprob: f32,
        pub bytes: Option<Vec<u8>>,
        pub top_logprobs: Vec<TopLogprob>,
    }

    /// List of the most likely tokens and their log probability, at this
    /// token position. In rare cases, there may be fewer than the number of requested top_logprobs returned.
    #[derive(Debug, Deserialize, Serialize, Clone)]
    pub struct TopLogprob {
        pub token: String,
        pub logprob: f32,
        pub bytes: Option<Vec<u8>>,
    }

    pub use crate::chat::{CompletionTokensDetails, CompletionUsage, PromptTokensDetails};

    crate::impl_from_str!(ChatCompletionChunk);

    #[cfg(test)]
    mod test {
        use std::str::FromStr;

        use super::*;

        #[test]
        fn streaming_example_deepseek() {
            let streams = vec![
                r#"{"id": "1f633d8bfc032625086f14113c411638", "choices": [{"index": 0, "delta": {"content": "", "role": "assistant"}, "finish_reason": null, "logprobs": null}], "created": 1718345013, "model": "deepseek-chat", "system_fingerprint": "fp_a49d71b8a1", "object": "chat.completion.chunk", "usage": null}"#,
                r#"{"choices": [{"delta": {"content": "Hello", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": "!", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " How", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " can", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " I", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " assist", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " you", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": " today", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": "?", "role": "assistant"}, "finish_reason": null, "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1"}"#,
                r#"{"choices": [{"delta": {"content": "", "role": null}, "finish_reason": "stop", "index": 0, "logprobs": null}], "created": 1718345013, "id": "1f633d8bfc032625086f14113c411638", "model": "deepseek-chat", "object": "chat.completion.chunk", "system_fingerprint": "fp_a49d71b8a1", "usage": {"completion_tokens": 9, "prompt_tokens": 17, "total_tokens": 26}}"#,
            ];

            for stream in streams {
                let parsed = ChatCompletionChunk::from_str(stream);
                match parsed {
                    Ok(completion) => {
                        println!("Deserialized: {:#?}", completion);
                    }
                    Err(e) => {
                        panic!("Failed to deserialize {}: {}", stream, e);
                    }
                }
            }
        }

        #[test]
        fn streaming_example_qwen() {
            let streams = vec![
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":"assistant","tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"我是","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"来自","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"阿里","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"云的超大规模","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"语言","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"模型","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"问。","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[{"delta":{"content":"","function_call":null,"refusal":null,"role":null,"tool_calls":null},"finish_reason":"stop","index":0,"logprobs":null}],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":null}"#,
                r#"{"id":"chatcmpl-e30f5ae7-3063-93c4-90fe-beb5f900bd57","choices":[],"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","service_tier":null,"system_fingerprint":null,"usage":{"completion_tokens":17,"prompt_tokens":22,"total_tokens":39,"completion_tokens_details":null,"prompt_tokens_details":{"audio_tokens":null,"cached_tokens":0}}}"#,
            ];

            for stream in streams {
                let parsed = ChatCompletionChunk::from_str(stream);
                match parsed {
                    Ok(completion) => {
                        println!("Deserialized: {:#?}", completion);
                    }
                    Err(e) => {
                        panic!("Failed to deserialize {}: {}", stream, e);
                    }
                }
            }
        }

        /// Azure OpenAI streams `annotations` (URL citations from "on your
        /// data" deployments) and `audio` inside `delta`, outside the
        /// official chunk schema; they must deserialize instead of being
        /// dropped.
        #[test]
        fn streaming_example_azure_annotations_and_audio() {
            let chunk = ChatCompletionChunk::from_str(
                r#"{"id":"chatcmpl-abc","choices":[{"delta":{"content":"According to the doc","annotations":[{"type":"url_citation","url_citation":{"start_index":0,"end_index":20,"title":"Azure Docs","url":"https://learn.microsoft.com/azure"}}],"audio":{"id":"audio_abc","data":"SGVsbG8=","expires_at":1735113344,"transcript":"Hello"}},"finish_reason":null,"index":0,"logprobs":null}],"created":1735113344,"model":"gpt-4o","object":"chat.completion.chunk","system_fingerprint":"fp_abc"}"#,
            )
            .expect("chunk with annotations and audio must deserialize");

            let delta = &chunk.choices[0].delta;
            let annotations = delta.annotations.as_ref().expect("annotations");
            assert_eq!(annotations.len(), 1);
            assert_eq!(annotations[0].url_citation.title, "Azure Docs");
            assert_eq!(
                annotations[0].url_citation.url,
                "https://learn.microsoft.com/azure"
            );

            let audio = delta.audio.as_ref().expect("audio");
            assert_eq!(audio.id, "audio_abc");
            assert_eq!(audio.transcript, "Hello");
        }

        /// vLLM-style backends terminate the stream with a usage-only chunk
        /// whose `choices` is `null`. The chunk must parse, yield no
        /// choices, and keep the usage statistics.
        #[test]
        fn usage_chunk_with_null_choices_parses() {
            let parsed = ChatCompletionChunk::from_str(
                r#"{"id":"chatcmpl-1","choices":null,"created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","usage":{"completion_tokens":17,"prompt_tokens":22,"total_tokens":39}}"#,
            )
            .expect("usage chunk with null choices must deserialize");
            assert!(parsed.choices.is_empty());
            assert_eq!(parsed.usage.expect("usage").total_tokens, 39);
        }

        #[test]
        fn usage_chunk_with_missing_choices_parses() {
            let parsed = ChatCompletionChunk::from_str(
                r#"{"id":"chatcmpl-1","created":1735113344,"model":"qwen-plus","object":"chat.completion.chunk","usage":{"completion_tokens":1,"prompt_tokens":2,"total_tokens":3}}"#,
            )
            .expect("usage chunk without a choices key must deserialize");
            assert!(parsed.choices.is_empty());
        }

        /// Gateways invent finish reasons (`eos`, provider-specific codes).
        /// They must not kill the chunk, and must round-trip unchanged.
        #[test]
        fn unknown_finish_reason_is_preserved() {
            let parsed = ChatCompletionChunk::from_str(
                r#"{"id":"1","choices":[{"index":0,"delta":{},"finish_reason":"eos"}],"created":1,"model":"m","object":"chat.completion.chunk"}"#,
            )
            .expect("chunk with unknown finish_reason must deserialize");

            let finish_reason = parsed.choices[0]
                .finish_reason
                .as_ref()
                .expect("finish_reason");
            assert_eq!(finish_reason.as_str(), "eos");
            assert_eq!(finish_reason.to_string(), "eos");
            assert_eq!(
                serde_json::to_value(finish_reason).unwrap(),
                serde_json::json!("eos")
            );
        }

        #[test]
        fn unknown_role_is_preserved() {
            let parsed = ChatCompletionChunk::from_str(
                r#"{"id":"1","choices":[{"index":0,"delta":{"role":"model","content":"hi"},"finish_reason":null}],"created":1,"model":"m","object":"chat.completion.chunk"}"#,
            )
            .expect("chunk with unknown role must deserialize");
            let role = parsed.choices[0].delta.role.as_ref().expect("role");
            assert_eq!(role.as_str(), "model");
        }

        #[test]
        fn missing_or_unknown_object_field_parses() {
            let missing =
                ChatCompletionChunk::from_str(r#"{"id":"1","choices":[],"created":1,"model":"m"}"#)
                    .expect("chunk without object must deserialize");
            assert!(missing.object.is_none());

            let weird = ChatCompletionChunk::from_str(
                r#"{"id":"1","choices":[],"created":1,"model":"m","object":"vendor.custom.chunk"}"#,
            )
            .expect("chunk with unknown object must deserialize");
            assert_eq!(
                weird.object.expect("object").as_str(),
                "vendor.custom.chunk"
            );
        }

        /// Responses must serialize back out for proxying/logging.
        #[test]
        fn chunk_round_trips_through_json() {
            let parsed = ChatCompletionChunk::from_str(
                r#"{"id":"1","choices":[{"index":0,"delta":{"role":"assistant","content":"Hi"},"finish_reason":null}],"created":1,"model":"m","object":"chat.completion.chunk"}"#,
            )
            .expect("chunk must deserialize");

            let json = serde_json::to_value(&parsed).unwrap();
            assert_eq!(json["object"], "chat.completion.chunk");
            assert_eq!(json["choices"][0]["delta"]["content"], "Hi");

            let reparsed = serde_json::from_value::<ChatCompletionChunk>(json).unwrap();
            assert_eq!(reparsed.id, parsed.id);
            assert_eq!(
                reparsed.choices[0].delta.content,
                parsed.choices[0].delta.content
            );
        }
    }
}

pub mod no_streaming {
    //! Non-streaming chat completion response.

    /// Alias for `crate::chat::ChatCompletion`, which is shared
    /// by many other modules. This alias is for compatibility.
    pub type ChatCompletion = crate::chat::ChatCompletion;
}