Skip to main content

openai_interface/chat/
mod.rs

1//! # Chat Completions API Module
2//!
3//! This module provides components shared by many submodules.
4
5use std::str::FromStr;
6
7use serde::{Deserialize, Serialize};
8
9use crate::errors::OapiError;
10
11pub mod create;
12pub mod delete;
13pub mod retrieve;
14pub mod update;
15
16/// The service tier used for processing the request.
17///
18/// This enum represents the different service tiers that can be specified when
19/// making a request to the API. Each tier corresponds to different performance
20/// characteristics and pricing models.
21#[derive(Debug, Serialize, Deserialize, Clone)]
22#[serde(rename_all = "lowercase")]
23pub enum ServiceTier {
24    /// Automatically select the service tier based on project settings.
25    Auto,
26    /// Use the default service tier with standard pricing and performance.
27    Default,
28    /// Use the flex service tier for flexible processing requirements.
29    Flex,
30    /// Use the scale service tier for scalable processing needs.
31    Scale,
32    /// Use the priority service tier for high-priority requests.
33    Priority,
34    /// Fast mode. Request-level opt-in for
35    /// [Fast mode](https://platform.openai.com/docs/guides/fast-mode); the
36    /// response reports the actual tier as `priority`.
37    Fast,
38}
39
40#[derive(Debug, Deserialize)]
41pub struct ChatCompletion {
42    /// A unique identifier for the chat completion.
43    pub id: String,
44    /// A list of chat completion choices. Can be more than one
45    /// if `n` is greater than 1.
46    pub choices: Vec<Choice>,
47    /// The Unix timestamp (in seconds) of when the chat completion was created.
48    pub created: u64,
49    /// The model used for the chat completion.
50    pub model: String,
51    /// Specifies the processing type used for serving the request.
52    ///
53    /// - If set to 'auto', then the request will be processed with the service tier
54    ///   configured in the Project settings. Unless otherwise configured, the Project
55    ///   will use 'default'.
56    /// - If set to 'default', then the request will be processed with the standard
57    ///   pricing and performance for the selected model.
58    /// - If set to '[flex](https://platform.openai.com/docs/guides/flex-processing)' or
59    ///   '[priority](https://openai.com/api-priority-processing/)', then the request
60    ///   will be processed with the corresponding service tier.
61    /// - When not set, the default behavior is 'auto'.
62    ///
63    /// When the `service_tier` parameter is set, the response body will include the
64    /// `service_tier` value based on the processing mode actually used to serve the
65    /// request. This response value may be different from the value set in the
66    /// parameter.
67    pub service_tier: Option<ServiceTier>,
68    /// The system fingerprint used for the chat completion.
69    /// Can be used in conjunction with the `seed` request parameter to understand when
70    /// backend changes have been made that might impact determinism.
71    pub system_fingerprint: Option<String>,
72    /// The object type, which is always `chat.completion`.
73    pub object: ChatCompletionObject,
74    /// Usage statistics for the completion request.
75    pub usage: Option<CompletionUsage>,
76}
77
78/// The object type, which is always `chat.completion`.
79#[derive(Debug, Deserialize)]
80pub enum ChatCompletionObject {
81    /// The object type is always `chat.completion`.
82    #[serde(rename = "chat.completion")]
83    ChatCompletion,
84}
85
86#[derive(Debug, Deserialize)]
87pub struct Choice {
88    /// The reason the model stopped generating tokens.
89    ///
90    /// This will be `stop` if the model hit a natural stop point or a provided stop
91    /// sequence, `length` if the maximum number of tokens specified in the request was
92    /// reached, `content_filter` if content was omitted due to a flag from our content
93    /// filters, `tool_calls` if the model called a tool, or `function_call`
94    /// (deprecated) if the model called a function.
95    pub finish_reason: FinishReason,
96    /// The index of the choice in the list of choices.
97    pub index: usize,
98    /// Log probability information for the choice.
99    pub logprobs: Option<ChoiceLogprobs>,
100    /// A chat completion message generated by the model.
101    pub message: ChatCompletionMessage,
102}
103
104#[derive(Debug, Deserialize, PartialEq)]
105#[serde(rename_all = "snake_case")]
106pub enum FinishReason {
107    Length,
108    Stop,
109    ToolCalls,
110    FunctionCall,
111    ContentFilter,
112    /// DeepSeek: the request is interrupted due to insufficient resource
113    /// of the inference system.
114    #[cfg(feature = "deepseek")]
115    InsufficientSystemResource,
116}
117
118#[derive(Debug, Deserialize)]
119pub struct ChatCompletionMessage {
120    /// The role of the author of this message. This shall always
121    /// be ResponseRole::Assistant
122    pub role: ResponseRole,
123    /// If the audio output modality is requested, this object contains data
124    /// about the audio response from the model.
125    /// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
126    pub audio: Option<ChatCompletionAudio>,
127    /// The contents of the message.
128    pub content: Option<String>,
129    /// DeepSeek: for thinking mode only. The reasoning contents of the
130    /// assistant message, before the final answer.
131    #[cfg(feature = "deepseek")]
132    pub reasoning_content: Option<String>,
133    /// The tool calls generated by the model, such as function calls.
134    /// Tool calls deserialization is not supported yet.
135    pub tool_calls: Option<Vec<ChatCompletionMessageToolCall>>,
136    /// The refusal message generated by the model.
137    pub refusal: Option<String>,
138    /// Annotations for the message, when applicable, such as URL citations
139    /// when the model uses a web search tool.
140    pub annotations: Option<Vec<Annotation>>,
141}
142
143/// If the audio output modality is requested, this object contains data about
144/// the audio response from the model.
145/// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
146#[derive(Debug, Deserialize, Clone)]
147pub struct ChatCompletionAudio {
148    /// Unique identifier for this audio response.
149    pub id: String,
150    /// Base64 encoded audio bytes generated by the model, in the format
151    /// specified in the request.
152    pub data: String,
153    /// The Unix timestamp (in seconds) for when this audio response will no
154    /// longer be accessible on the server for use in multi-turn conversations.
155    pub expires_at: u64,
156    /// Transcript of the audio generated by the model.
157    pub transcript: String,
158}
159
160/// An annotation for a chat completion message.
161#[derive(Debug, Deserialize, Clone)]
162pub struct Annotation {
163    /// The type of the annotation. Always `url_citation`.
164    #[serde(rename = "type")]
165    pub type_: AnnotationType,
166    /// The URL citation.
167    pub url_citation: UrlCitation,
168}
169
170#[derive(Debug, Deserialize, Clone)]
171#[serde(rename_all = "snake_case")]
172pub enum AnnotationType {
173    /// A URL citation when using web search.
174    UrlCitation,
175}
176
177/// A URL citation when the model uses a web search tool.
178#[derive(Debug, Deserialize, Clone)]
179pub struct UrlCitation {
180    /// The index of the first character of the URL citation in the message.
181    pub start_index: usize,
182    /// The index of the last character of the URL citation in the message.
183    pub end_index: usize,
184    /// The title of the web resource.
185    pub title: String,
186    /// The URL of the web resource.
187    pub url: String,
188}
189
190#[derive(Debug, Deserialize)]
191#[serde(tag = "type", rename_all = "snake_case")]
192pub enum ChatCompletionMessageToolCall {
193    /// The type of the tool. Currently, only `function` is supported.
194    /// The field { type = "function" } is added automatically.
195    Function {
196        /// The ID of the tool call.
197        id: String,
198        /// The function that the model called.
199        function: MessageToolCallFunction,
200    },
201    /// The type of the tool. Always `custom`.
202    /// The field { type = "custom" } is added automatically.
203    Custom {
204        /// The id of the tool call.
205        id: String,
206        /// The custom tool that the model called.
207        custom: MessageToolCallCustom,
208    },
209}
210
211#[derive(Debug, Deserialize)]
212pub struct MessageToolCallCustom {
213    /// The input for the custom tool call generated by the model.
214    pub input: String,
215    /// The name of the custom tool to call.
216    pub name: String,
217}
218
219#[derive(Debug, Deserialize)]
220pub struct MessageToolCallFunction {
221    /// The arguments to call the function with, as generated by the model in JSON
222    /// format. Note that the model does not always generate valid JSON, and may
223    /// hallucinate parameters not defined by your function schema. Validate the
224    /// arguments in your code before calling your function.
225    pub arguments: String,
226    /// The name of the function to call.
227    pub name: String,
228}
229
230#[derive(Debug, Deserialize)]
231#[serde(rename_all = "snake_case")]
232pub enum ResponseRole {
233    /// The role of the response message is always assistant.
234    Assistant,
235}
236
237#[derive(Debug, Deserialize)]
238pub struct ChoiceLogprobs {
239    /// A list of message content tokens with log probability information.
240    pub content: Option<Vec<TokenLogProb>>,
241    /// DeepSeek: a list of reasoning content tokens with log probability
242    /// information. Only present for thinking models.
243    #[cfg(feature = "deepseek")]
244    pub reasoning_content: Option<Vec<TokenLogProb>>,
245    /// A list of message refusal tokens with log probability information.
246    pub refusal: Option<Vec<TokenLogProb>>,
247}
248
249#[derive(Debug, Deserialize)]
250pub struct TokenLogProb {
251    /// The token.
252    pub token: String,
253    /// The log probability of this token, if it is within the top 20 most likely
254    /// tokens. Otherwise, the value `-9999.0` is used to signify that the token is very
255    /// unlikely.
256    pub logprob: f32,
257    /// A list of integers representing the UTF-8 bytes representation of the token.
258    ///
259    /// Useful in instances where characters are represented by multiple tokens and
260    /// their byte representations must be combined to generate the correct text
261    /// representation. Can be `null` if there is no bytes representation for the token.
262    pub bytes: Option<Vec<u8>>,
263    /// List of the most likely tokens and their log probability, at this token
264    /// position. In rare cases, there may be fewer than the number of requested
265    /// `top_logprobs` returned.
266    pub top_logprobs: Vec<TopLogprob>,
267}
268
269#[derive(Debug, Deserialize)]
270pub struct TopLogprob {
271    /// The token.
272    pub token: String,
273    /// A list of integers representing the UTF-8 bytes representation of the token.
274    ///
275    /// Useful in instances where characters are represented by multiple tokens and
276    /// their byte representations must be combined to generate the correct text
277    /// representation. Can be `null` if there is no bytes representation for the token.
278    pub logprob: f32,
279    /// List of the most likely tokens and their log probability, at this token
280    /// position. In rare cases, there may be fewer than the number of requested
281    /// `top_logprobs` returned.
282    pub bytes: Option<Vec<u8>>,
283}
284
285#[derive(Debug, Deserialize)]
286pub struct CompletionUsage {
287    /// Number of tokens in the generated completion.
288    pub completion_tokens: usize,
289    /// Number of tokens in the prompt.
290    pub prompt_tokens: usize,
291
292    /// DeepSeek: number of tokens in the prompt that hits the context cache.
293    #[cfg(feature = "deepseek")]
294    pub prompt_cache_hit_tokens: Option<usize>,
295    /// DeepSeek: number of tokens in the prompt that misses the context cache.
296    #[cfg(feature = "deepseek")]
297    pub prompt_cache_miss_tokens: Option<usize>,
298
299    /// Total number of tokens used in the request (prompt + completion).
300    pub total_tokens: usize,
301    /// Breakdown of tokens used in a completion.
302    pub completion_tokens_details: Option<CompletionTokensDetails>,
303    /// Breakdown of tokens used in the prompt.
304    pub prompt_tokens_details: Option<PromptTokensDetails>,
305}
306
307#[derive(Debug, Deserialize)]
308pub struct CompletionTokensDetails {
309    /// When using Predicted Outputs, the number of tokens in the prediction that
310    /// appeared in the completion.
311    pub accepted_prediction_tokens: Option<usize>,
312    /// Audio input tokens generated by the model.
313    pub audio_tokens: Option<usize>,
314    /// Tokens generated by the model for reasoning.
315    pub reasoning_tokens: Option<usize>,
316    /// When using Predicted Outputs, the number of tokens in the prediction that did
317    /// not appear in the completion. However, like reasoning tokens, these tokens are
318    /// still counted in the total completion tokens for purposes of billing, output,
319    /// and context window limits.
320    pub rejected_prediction_tokens: Option<usize>,
321}
322
323#[derive(Debug, Deserialize)]
324pub struct PromptTokensDetails {
325    /// Audio input tokens present in the prompt.
326    pub audio_tokens: Option<usize>,
327    /// Cached tokens present in the prompt.
328    pub cached_tokens: Option<usize>,
329}
330
331impl FromStr for ChatCompletion {
332    type Err = crate::errors::OapiError;
333
334    fn from_str(content: &str) -> Result<Self, Self::Err> {
335        let parse_result: Result<ChatCompletion, _> = serde_json::from_str(content)
336            .map_err(|e| OapiError::DeserializationError(e.to_string()));
337        parse_result
338    }
339}
340
341#[cfg(test)]
342mod test {
343    use super::*;
344
345    #[test]
346    fn service_tier_parses_official_values() {
347        for (raw, is_fast) in [
348            (r#""auto""#, false),
349            (r#""default""#, false),
350            (r#""flex""#, false),
351            (r#""scale""#, false),
352            (r#""priority""#, false),
353            (r#""fast""#, true),
354        ] {
355            let tier: ServiceTier =
356                serde_json::from_str(raw).unwrap_or_else(|e| panic!("failed to parse {raw}: {e}"));
357            assert_eq!(matches!(tier, ServiceTier::Fast), is_fast, "raw: {raw}");
358        }
359    }
360
361    #[test]
362    fn no_streaming_example_deepseek() {
363        let json = r#"{
364          "id": "30f6413a-a827-4cf3-9898-f13a8634b798",
365          "object": "chat.completion",
366          "created": 1757944111,
367          "model": "deepseek-chat",
368          "choices": [
369            {
370              "index": 0,
371              "message": {
372                "role": "assistant",
373                "content": "Hello! How can I help you today? 😊"
374              },
375              "logprobs": null,
376              "finish_reason": "stop"
377            }
378          ],
379          "usage": {
380            "prompt_tokens": 10,
381            "completion_tokens": 11,
382            "total_tokens": 21,
383            "prompt_tokens_details": {
384              "cached_tokens": 0
385            },
386            "prompt_cache_hit_tokens": 0,
387            "prompt_cache_miss_tokens": 10
388          },
389          "system_fingerprint": "fp_08f168e49b_prod0820_fp8_kvcache"
390        }"#;
391
392        let parsed = ChatCompletion::from_str(json);
393        match parsed {
394            Ok(_) => {}
395            Err(e) => {
396                panic!("Failed to deserialize: {}", e);
397            }
398        }
399    }
400
401    #[test]
402    fn no_streaming_example_qwen() {
403        let json = r#"{
404            "choices": [
405                {
406                    "message": {
407                        "role": "assistant",
408                        "content": "我是阿里云开发的一款超大规模语言模型,我叫通义千问。"
409                    },
410                    "finish_reason": "stop",
411                    "index": 0,
412                    "logprobs": null
413                }
414            ],
415            "object": "chat.completion",
416            "usage": {
417                "prompt_tokens": 3019,
418                "completion_tokens": 104,
419                "total_tokens": 3123,
420                "prompt_tokens_details": {
421                    "cached_tokens": 2048
422                }
423            },
424            "created": 1735120033,
425            "system_fingerprint": null,
426            "model": "qwen-plus",
427            "id": "chatcmpl-6ada9ed2-7f33-9de2-8bb0-78bd4035025a"
428        }"#;
429
430        let parsed = ChatCompletion::from_str(json);
431        match parsed {
432            Ok(_) => {}
433            Err(e) => {
434                panic!("Failed to deserialize: {}", e);
435            }
436        }
437    }
438}