openai_interface/chat/mod.rs
1//! # Chat Completions API Module
2//!
3//! This module provides components shared by many submodules.
4
5use std::str::FromStr;
6
7use serde::{Deserialize, Serialize};
8
9use crate::errors::OapiError;
10
11pub mod create;
12pub mod delete;
13pub mod retrieve;
14pub mod update;
15
16/// The service tier used for processing the request.
17///
18/// This enum represents the different service tiers that can be specified when
19/// making a request to the API. Each tier corresponds to different performance
20/// characteristics and pricing models.
21#[derive(Debug, Serialize, Deserialize, Clone)]
22#[serde(rename_all = "lowercase")]
23pub enum ServiceTier {
24 /// Automatically select the service tier based on project settings.
25 Auto,
26 /// Use the default service tier with standard pricing and performance.
27 Default,
28 /// Use the flex service tier for flexible processing requirements.
29 Flex,
30 /// Use the scale service tier for scalable processing needs.
31 Scale,
32 /// Use the priority service tier for high-priority requests.
33 Priority,
34}
35
36#[derive(Debug, Deserialize)]
37pub struct ChatCompletion {
38 /// A unique identifier for the chat completion.
39 pub id: String,
40 /// A list of chat completion choices. Can be more than one
41 /// if `n` is greater than 1.
42 pub choices: Vec<Choice>,
43 /// The Unix timestamp (in seconds) of when the chat completion was created.
44 pub created: u64,
45 /// The model used for the chat completion.
46 pub model: String,
47 /// Specifies the processing type used for serving the request.
48 ///
49 /// - If set to 'auto', then the request will be processed with the service tier
50 /// configured in the Project settings. Unless otherwise configured, the Project
51 /// will use 'default'.
52 /// - If set to 'default', then the request will be processed with the standard
53 /// pricing and performance for the selected model.
54 /// - If set to '[flex](https://platform.openai.com/docs/guides/flex-processing)' or
55 /// '[priority](https://openai.com/api-priority-processing/)', then the request
56 /// will be processed with the corresponding service tier.
57 /// - When not set, the default behavior is 'auto'.
58 ///
59 /// When the `service_tier` parameter is set, the response body will include the
60 /// `service_tier` value based on the processing mode actually used to serve the
61 /// request. This response value may be different from the value set in the
62 /// parameter.
63 pub service_tier: Option<ServiceTier>,
64 /// The system fingerprint used for the chat completion.
65 /// Can be used in conjunction with the `seed` request parameter to understand when
66 /// backend changes have been made that might impact determinism.
67 pub system_fingerprint: Option<String>,
68 /// The object type, which is always `chat.completion`.
69 pub object: ChatCompletionObject,
70 /// Usage statistics for the completion request.
71 pub usage: Option<CompletionUsage>,
72}
73
74/// The object type, which is always `chat.completion`.
75#[derive(Debug, Deserialize)]
76pub enum ChatCompletionObject {
77 /// The object type is always `chat.completion`.
78 #[serde(rename = "chat.completion")]
79 ChatCompletion,
80}
81
82#[derive(Debug, Deserialize)]
83pub struct Choice {
84 /// The reason the model stopped generating tokens.
85 ///
86 /// This will be `stop` if the model hit a natural stop point or a provided stop
87 /// sequence, `length` if the maximum number of tokens specified in the request was
88 /// reached, `content_filter` if content was omitted due to a flag from our content
89 /// filters, `tool_calls` if the model called a tool, or `function_call`
90 /// (deprecated) if the model called a function.
91 pub finish_reason: FinishReason,
92 /// The index of the choice in the list of choices.
93 pub index: usize,
94 /// Log probability information for the choice.
95 pub logprobs: Option<ChoiceLogprobs>,
96 /// A chat completion message generated by the model.
97 pub message: ChatCompletionMessage,
98}
99
100#[derive(Debug, Deserialize, PartialEq)]
101#[serde(rename_all = "snake_case")]
102pub enum FinishReason {
103 Length,
104 Stop,
105 ToolCalls,
106 FunctionCall,
107 ContentFilter,
108 /// DeepSeek: the request is interrupted due to insufficient resource
109 /// of the inference system.
110 #[cfg(feature = "deepseek")]
111 InsufficientSystemResource,
112}
113
114#[derive(Debug, Deserialize)]
115pub struct ChatCompletionMessage {
116 /// The role of the author of this message. This shall always
117 /// be ResponseRole::Assistant
118 pub role: ResponseRole,
119 /// If the audio output modality is requested, this object contains data
120 /// about the audio response from the model.
121 /// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
122 pub audio: Option<ChatCompletionAudio>,
123 /// The contents of the message.
124 pub content: Option<String>,
125 /// DeepSeek: for thinking mode only. The reasoning contents of the
126 /// assistant message, before the final answer.
127 #[cfg(feature = "deepseek")]
128 pub reasoning_content: Option<String>,
129 /// The tool calls generated by the model, such as function calls.
130 /// Tool calls deserialization is not supported yet.
131 pub tool_calls: Option<Vec<ChatCompletionMessageToolCall>>,
132 /// The refusal message generated by the model.
133 pub refusal: Option<String>,
134 /// Annotations for the message, when applicable, such as URL citations
135 /// when the model uses a web search tool.
136 pub annotations: Option<Vec<Annotation>>,
137}
138
139/// If the audio output modality is requested, this object contains data about
140/// the audio response from the model.
141/// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
142#[derive(Debug, Deserialize, Clone)]
143pub struct ChatCompletionAudio {
144 /// Unique identifier for this audio response.
145 pub id: String,
146 /// Base64 encoded audio bytes generated by the model, in the format
147 /// specified in the request.
148 pub data: String,
149 /// The Unix timestamp (in seconds) for when this audio response will no
150 /// longer be accessible on the server for use in multi-turn conversations.
151 pub expires_at: u64,
152 /// Transcript of the audio generated by the model.
153 pub transcript: String,
154}
155
156/// An annotation for a chat completion message.
157#[derive(Debug, Deserialize, Clone)]
158pub struct Annotation {
159 /// The type of the annotation. Always `url_citation`.
160 #[serde(rename = "type")]
161 pub type_: AnnotationType,
162 /// The URL citation.
163 pub url_citation: UrlCitation,
164}
165
166#[derive(Debug, Deserialize, Clone)]
167#[serde(rename_all = "snake_case")]
168pub enum AnnotationType {
169 /// A URL citation when using web search.
170 UrlCitation,
171}
172
173/// A URL citation when the model uses a web search tool.
174#[derive(Debug, Deserialize, Clone)]
175pub struct UrlCitation {
176 /// The index of the first character of the URL citation in the message.
177 pub start_index: usize,
178 /// The index of the last character of the URL citation in the message.
179 pub end_index: usize,
180 /// The title of the web resource.
181 pub title: String,
182 /// The URL of the web resource.
183 pub url: String,
184}
185
186#[derive(Debug, Deserialize)]
187#[serde(tag = "type", rename_all = "snake_case")]
188pub enum ChatCompletionMessageToolCall {
189 /// The type of the tool. Currently, only `function` is supported.
190 /// The field { type = "function" } is added automatically.
191 Function {
192 /// The ID of the tool call.
193 id: String,
194 /// The function that the model called.
195 function: MessageToolCallFunction,
196 },
197 /// The type of the tool. Always `custom`.
198 /// The field { type = "custom" } is added automatically.
199 Custom {
200 /// The id of the tool call.
201 id: String,
202 /// The custom tool that the model called.
203 custom: MessageToolCallCustom,
204 },
205}
206
207#[derive(Debug, Deserialize)]
208pub struct MessageToolCallCustom {
209 /// The input for the custom tool call generated by the model.
210 pub input: String,
211 /// The name of the custom tool to call.
212 pub name: String,
213}
214
215#[derive(Debug, Deserialize)]
216pub struct MessageToolCallFunction {
217 /// The arguments to call the function with, as generated by the model in JSON
218 /// format. Note that the model does not always generate valid JSON, and may
219 /// hallucinate parameters not defined by your function schema. Validate the
220 /// arguments in your code before calling your function.
221 pub arguments: String,
222 /// The name of the function to call.
223 pub name: String,
224}
225
226#[derive(Debug, Deserialize)]
227#[serde(rename_all = "snake_case")]
228pub enum ResponseRole {
229 /// The role of the response message is always assistant.
230 Assistant,
231}
232
233#[derive(Debug, Deserialize)]
234pub struct ChoiceLogprobs {
235 /// A list of message content tokens with log probability information.
236 pub content: Option<Vec<TokenLogProb>>,
237 /// DeepSeek: a list of reasoning content tokens with log probability
238 /// information. Only present for thinking models.
239 #[cfg(feature = "deepseek")]
240 pub reasoning_content: Option<Vec<TokenLogProb>>,
241 /// A list of message refusal tokens with log probability information.
242 pub refusal: Option<Vec<TokenLogProb>>,
243}
244
245#[derive(Debug, Deserialize)]
246pub struct TokenLogProb {
247 /// The token.
248 pub token: String,
249 /// The log probability of this token, if it is within the top 20 most likely
250 /// tokens. Otherwise, the value `-9999.0` is used to signify that the token is very
251 /// unlikely.
252 pub logprob: f32,
253 /// A list of integers representing the UTF-8 bytes representation of the token.
254 ///
255 /// Useful in instances where characters are represented by multiple tokens and
256 /// their byte representations must be combined to generate the correct text
257 /// representation. Can be `null` if there is no bytes representation for the token.
258 pub bytes: Option<Vec<u8>>,
259 /// List of the most likely tokens and their log probability, at this token
260 /// position. In rare cases, there may be fewer than the number of requested
261 /// `top_logprobs` returned.
262 pub top_logprobs: Vec<TopLogprob>,
263}
264
265#[derive(Debug, Deserialize)]
266pub struct TopLogprob {
267 /// The token.
268 pub token: String,
269 /// A list of integers representing the UTF-8 bytes representation of the token.
270 ///
271 /// Useful in instances where characters are represented by multiple tokens and
272 /// their byte representations must be combined to generate the correct text
273 /// representation. Can be `null` if there is no bytes representation for the token.
274 pub logprob: f32,
275 /// List of the most likely tokens and their log probability, at this token
276 /// position. In rare cases, there may be fewer than the number of requested
277 /// `top_logprobs` returned.
278 pub bytes: Option<Vec<u8>>,
279}
280
281#[derive(Debug, Deserialize)]
282pub struct CompletionUsage {
283 /// Number of tokens in the generated completion.
284 pub completion_tokens: usize,
285 /// Number of tokens in the prompt.
286 pub prompt_tokens: usize,
287
288 /// DeepSeek: number of tokens in the prompt that hits the context cache.
289 #[cfg(feature = "deepseek")]
290 pub prompt_cache_hit_tokens: Option<usize>,
291 /// DeepSeek: number of tokens in the prompt that misses the context cache.
292 #[cfg(feature = "deepseek")]
293 pub prompt_cache_miss_tokens: Option<usize>,
294
295 /// Total number of tokens used in the request (prompt + completion).
296 pub total_tokens: usize,
297 /// Breakdown of tokens used in a completion.
298 pub completion_tokens_details: Option<CompletionTokensDetails>,
299 /// Breakdown of tokens used in the prompt.
300 pub prompt_tokens_details: Option<PromptTokensDetails>,
301}
302
303#[derive(Debug, Deserialize)]
304pub struct CompletionTokensDetails {
305 /// When using Predicted Outputs, the number of tokens in the prediction that
306 /// appeared in the completion.
307 pub accepted_prediction_tokens: Option<usize>,
308 /// Audio input tokens generated by the model.
309 pub audio_tokens: Option<usize>,
310 /// Tokens generated by the model for reasoning.
311 pub reasoning_tokens: Option<usize>,
312 /// When using Predicted Outputs, the number of tokens in the prediction that did
313 /// not appear in the completion. However, like reasoning tokens, these tokens are
314 /// still counted in the total completion tokens for purposes of billing, output,
315 /// and context window limits.
316 pub rejected_prediction_tokens: Option<usize>,
317}
318
319#[derive(Debug, Deserialize)]
320pub struct PromptTokensDetails {
321 /// Audio input tokens present in the prompt.
322 pub audio_tokens: Option<usize>,
323 /// Cached tokens present in the prompt.
324 pub cached_tokens: Option<usize>,
325}
326
327impl FromStr for ChatCompletion {
328 type Err = crate::errors::OapiError;
329
330 fn from_str(content: &str) -> Result<Self, Self::Err> {
331 let parse_result: Result<ChatCompletion, _> = serde_json::from_str(content)
332 .map_err(|e| OapiError::DeserializationError(e.to_string()));
333 parse_result
334 }
335}
336
337#[cfg(test)]
338mod test {
339 use super::*;
340
341 #[test]
342 fn no_streaming_example_deepseek() {
343 let json = r#"{
344 "id": "30f6413a-a827-4cf3-9898-f13a8634b798",
345 "object": "chat.completion",
346 "created": 1757944111,
347 "model": "deepseek-chat",
348 "choices": [
349 {
350 "index": 0,
351 "message": {
352 "role": "assistant",
353 "content": "Hello! How can I help you today? 😊"
354 },
355 "logprobs": null,
356 "finish_reason": "stop"
357 }
358 ],
359 "usage": {
360 "prompt_tokens": 10,
361 "completion_tokens": 11,
362 "total_tokens": 21,
363 "prompt_tokens_details": {
364 "cached_tokens": 0
365 },
366 "prompt_cache_hit_tokens": 0,
367 "prompt_cache_miss_tokens": 10
368 },
369 "system_fingerprint": "fp_08f168e49b_prod0820_fp8_kvcache"
370 }"#;
371
372 let parsed = ChatCompletion::from_str(json);
373 match parsed {
374 Ok(_) => {}
375 Err(e) => {
376 panic!("Failed to deserialize: {}", e);
377 }
378 }
379 }
380
381 #[test]
382 fn no_streaming_example_qwen() {
383 let json = r#"{
384 "choices": [
385 {
386 "message": {
387 "role": "assistant",
388 "content": "我是阿里云开发的一款超大规模语言模型,我叫通义千问。"
389 },
390 "finish_reason": "stop",
391 "index": 0,
392 "logprobs": null
393 }
394 ],
395 "object": "chat.completion",
396 "usage": {
397 "prompt_tokens": 3019,
398 "completion_tokens": 104,
399 "total_tokens": 3123,
400 "prompt_tokens_details": {
401 "cached_tokens": 2048
402 }
403 },
404 "created": 1735120033,
405 "system_fingerprint": null,
406 "model": "qwen-plus",
407 "id": "chatcmpl-6ada9ed2-7f33-9de2-8bb0-78bd4035025a"
408 }"#;
409
410 let parsed = ChatCompletion::from_str(json);
411 match parsed {
412 Ok(_) => {}
413 Err(e) => {
414 panic!("Failed to deserialize: {}", e);
415 }
416 }
417 }
418}