openai_interface/chat/mod.rs
1//! # Chat Completions API Module
2//!
3//! This module provides components shared by many submodules.
4
5use std::str::FromStr;
6
7use serde::{Deserialize, Serialize};
8
9use crate::errors::OapiError;
10
11pub mod create;
12pub mod delete;
13pub mod retrieve;
14pub mod update;
15
16crate::wire_string_enum! {
17 /// The service tier used for processing the request.
18 ///
19 /// This enum represents the different service tiers that can be specified when
20 /// making a request to the API. Each tier corresponds to different performance
21 /// characteristics and pricing models.
22 pub enum ServiceTier {
23 /// Automatically select the service tier based on project settings.
24 Auto => "auto",
25 /// Use the default service tier with standard pricing and performance.
26 Default => "default",
27 /// Use the flex service tier for flexible processing requirements.
28 Flex => "flex",
29 /// Use the scale service tier for scalable processing needs.
30 Scale => "scale",
31 /// Use the priority service tier for high-priority requests.
32 Priority => "priority",
33 /// Fast mode. Request-level opt-in for
34 /// [Fast mode](https://platform.openai.com/docs/guides/fast-mode); the
35 /// response reports the actual tier as `priority`.
36 Fast => "fast",
37 }
38}
39
40/// Moderation results for the request input and generated output.
41///
42/// Present on the response when moderated completions are requested via the
43/// `moderation` request parameter.
44#[derive(Debug, Deserialize, Serialize, Clone)]
45pub struct ChatModeration {
46 /// Moderation for the request input.
47 pub input: ModerationSide,
48 /// Moderation for the generated output.
49 pub output: ModerationSide,
50}
51
52/// The moderation outcome for one side (input or output) of the completion:
53/// either successful results or an error produced while attempting
54/// moderation.
55#[derive(Debug, Deserialize, Serialize, Clone)]
56#[serde(tag = "type", rename_all = "snake_case")]
57pub enum ModerationSide {
58 /// Successful moderation results. Always `moderation_results`.
59 ModerationResults {
60 /// The moderation model used to generate the results.
61 model: String,
62 /// A list of moderation results.
63 results: Vec<ModerationSideResult>,
64 },
65 /// An error produced while attempting moderation. Always `error`.
66 Error {
67 /// The error code.
68 code: String,
69 /// The error message.
70 message: String,
71 },
72}
73
74/// A moderation result produced for the response input or generated output.
75#[derive(Debug, Deserialize, Serialize, Clone)]
76pub struct ModerationSideResult {
77 /// A dictionary of moderation categories to booleans, `true` if the
78 /// input is flagged under this category.
79 pub categories: std::collections::HashMap<String, bool>,
80 /// Which modalities of input are reflected by the score for each
81 /// category, e.g. `["text"]` or `["text", "image"]`.
82 pub category_applied_input_types: std::collections::HashMap<String, Vec<String>>,
83 /// A dictionary of moderation categories to scores.
84 pub category_scores: std::collections::HashMap<String, f64>,
85 /// A boolean indicating whether the content was flagged by any category.
86 pub flagged: bool,
87 /// The moderation model that produced this result.
88 pub model: String,
89 /// The object type, which is always `moderation_result`.
90 #[serde(rename = "type")]
91 pub type_: ModerationResultType,
92}
93
94crate::wire_string_enum! {
95 /// The object type of a moderation result. Always `moderation_result`.
96 pub enum ModerationResultType {
97 ModerationResult => "moderation_result",
98 }
99}
100
101#[derive(Debug, Deserialize, Serialize, Clone)]
102pub struct ChatCompletion {
103 /// A unique identifier for the chat completion.
104 pub id: String,
105 /// A list of chat completion choices. Can be more than one
106 /// if `n` is greater than 1.
107 pub choices: Vec<Choice>,
108 /// The Unix timestamp (in seconds) of when the chat completion was created.
109 pub created: u64,
110 /// The model used for the chat completion.
111 pub model: String,
112 /// Specifies the processing type used for serving the request.
113 ///
114 /// - If set to 'auto', then the request will be processed with the service tier
115 /// configured in the Project settings. Unless otherwise configured, the Project
116 /// will use 'default'.
117 /// - If set to 'default', then the request will be processed with the standard
118 /// pricing and performance for the selected model.
119 /// - If set to '[flex](https://platform.openai.com/docs/guides/flex-processing)' or
120 /// '[priority](https://openai.com/api-priority-processing/)', then the request
121 /// will be processed with the corresponding service tier.
122 /// - When not set, the default behavior is 'auto'.
123 ///
124 /// When the `service_tier` parameter is set, the response body will include the
125 /// `service_tier` value based on the processing mode actually used to serve the
126 /// request. This response value may be different from the value set in the
127 /// parameter.
128 pub service_tier: Option<ServiceTier>,
129 /// The system fingerprint used for the chat completion.
130 /// Can be used in conjunction with the `seed` request parameter to understand when
131 /// backend changes have been made that might impact determinism.
132 pub system_fingerprint: Option<String>,
133 /// The object type, which is always `chat.completion`.
134 ///
135 /// `Some` only when the backend sends a recognized value; some
136 /// non-OpenAI gateways omit or repurpose the field.
137 pub object: Option<ChatCompletionObject>,
138 /// Usage statistics for the completion request.
139 pub usage: Option<CompletionUsage>,
140 /// Moderation results for the request input and generated output.
141 ///
142 /// Present when moderated completions are requested via the `moderation`
143 /// request parameter.
144 pub moderation: Option<ChatModeration>,
145
146 /// vLLM: log probabilities of the prompt tokens, one entry per prompt
147 /// position, or `null` for positions the server did not report. Each
148 /// entry maps a token ID to its [`crate::vllm::Logprob`].
149 ///
150 /// Requested with `vllm_sampling.prompt_logprobs`; OpenAI has no
151 /// equivalent.
152 #[cfg(feature = "vllm")]
153 pub prompt_logprobs: Option<Vec<Option<std::collections::HashMap<u32, crate::vllm::Logprob>>>>,
154 /// vLLM: the prompt's token IDs after chat-template rendering.
155 #[cfg(feature = "vllm")]
156 pub prompt_token_ids: Option<Vec<u32>>,
157 /// vLLM: the fully rendered prompt text.
158 ///
159 /// Only set when the request set `vllm_chat.return_prompt_text`.
160 #[cfg(feature = "vllm")]
161 pub prompt_text: Option<String>,
162 /// vLLM: KV-transfer parameters for disaggregated prefill, echoing and
163 /// extending the request's `kv_transfer_params`.
164 #[cfg(feature = "vllm")]
165 pub kv_transfer_params: Option<serde_json::Map<String, serde_json::Value>>,
166 /// vLLM: encoder-cache transfer parameters, echoing and extending the
167 /// request's `ec_transfer_params`.
168 #[cfg(feature = "vllm")]
169 pub ec_transfer_params: Option<serde_json::Map<String, serde_json::Value>>,
170}
171
172crate::wire_string_enum! {
173 /// The object type, which is always `chat.completion`.
174 pub enum ChatCompletionObject {
175 /// The object type is always `chat.completion`.
176 ChatCompletion => "chat.completion",
177 }
178}
179
180#[derive(Debug, Deserialize, Serialize, Clone)]
181pub struct Choice {
182 /// The reason the model stopped generating tokens.
183 ///
184 /// This will be `stop` if the model hit a natural stop point or a provided stop
185 /// sequence, `length` if the maximum number of tokens specified in the request was
186 /// reached, `content_filter` if content was omitted due to a flag from our content
187 /// filters, `tool_calls` if the model called a tool, or `function_call`
188 /// (deprecated) if the model called a function.
189 pub finish_reason: FinishReason,
190 /// The index of the choice in the list of choices.
191 pub index: u32,
192 /// Log probability information for the choice.
193 pub logprobs: Option<ChoiceLogprobs>,
194 /// A chat completion message generated by the model.
195 pub message: ChatCompletionMessage,
196
197 /// vLLM: which terminator ended generation — the matched stop string, or
198 /// the matched token ID. Not part of the OpenAI schema; `finish_reason`
199 /// alone only reports `stop` for both cases.
200 #[cfg(feature = "vllm")]
201 pub stop_reason: Option<crate::vllm::StopReason>,
202 /// vLLM: the generated token IDs, for tracing tokens in agent scenarios.
203 /// Only set when the request set `vllm_chat.return_token_ids`.
204 #[cfg(feature = "vllm")]
205 pub token_ids: Option<Vec<u32>>,
206 /// vLLM: per-token expert routing decisions for mixture-of-experts
207 /// models, as base64-encoded NumPy `.npy` bytes of shape
208 /// `(num_tokens - 1, num_layers, num_experts_per_tok)`.
209 ///
210 /// Only set when the server runs with `--enable-return-routed-experts`.
211 #[cfg(feature = "vllm")]
212 pub routed_experts: Option<String>,
213}
214
215crate::wire_string_enum! {
216 /// The reason the model stopped generating tokens.
217 ///
218 /// Values that are not part of the official API (some gateways emit
219 /// their own, e.g. `eos`) are preserved as
220 /// [`FinishReason::Unknown`] instead of failing deserialization.
221 pub enum FinishReason {
222 /// The maximum number of tokens specified in the request was reached.
223 Length => "length",
224 /// The model hit a natural stop point or a provided stop sequence.
225 Stop => "stop",
226 /// The model called a tool.
227 ToolCalls => "tool_calls",
228 /// The model called a function (deprecated).
229 FunctionCall => "function_call",
230 /// Content was omitted due to a flag from our content filters.
231 ContentFilter => "content_filter",
232 /// DeepSeek: the request is interrupted due to insufficient resource
233 /// of the inference system.
234 InsufficientSystemResource => "insufficient_system_resource",
235 }
236}
237
238#[derive(Debug, Deserialize, Serialize, Clone)]
239pub struct ChatCompletionMessage {
240 /// The role of the author of this message. This shall always
241 /// be [`Role::Assistant`]
242 pub role: Role,
243 /// If the audio output modality is requested, this object contains data
244 /// about the audio response from the model.
245 /// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
246 pub audio: Option<ChatCompletionAudio>,
247 /// The contents of the message.
248 pub content: Option<String>,
249 /// For thinking models: the reasoning contents of the assistant
250 /// message, before the final answer.
251 #[cfg(feature = "reasoning")]
252 pub reasoning_content: Option<String>,
253 /// vLLM: the reasoning contents of the assistant message, under the key
254 /// vLLM actually emits.
255 ///
256 /// vLLM accepts `reasoning_content` on the way in but serializes the
257 /// chain of thought as `reasoning` on the way out — in its `ChatMessage`
258 /// the two names are aliases of one field, and `reasoning` is the
259 /// serialization name. Current vLLM does **not** emit
260 /// `reasoning_content` in responses, so for a vLLM backend the field
261 /// above stays `None` and this one carries the chain of thought. Map it
262 /// back onto `reasoning_content` when feeding it into a follow-up
263 /// request.
264 #[cfg(feature = "vllm")]
265 pub reasoning: Option<String>,
266 /// The tool calls generated by the model, such as function calls.
267 /// Tool calls deserialization is not supported yet.
268 pub tool_calls: Option<Vec<ChatCompletionMessageToolCall>>,
269 /// The refusal message generated by the model.
270 pub refusal: Option<String>,
271 /// Annotations for the message, when applicable, such as URL citations
272 /// when the model uses a web search tool.
273 pub annotations: Option<Vec<Annotation>>,
274}
275
276/// If the audio output modality is requested, this object contains data about
277/// the audio response from the model.
278/// [Learn more from OpenAI](https://platform.openai.com/docs/guides/audio).
279#[derive(Debug, Deserialize, Serialize, Clone)]
280pub struct ChatCompletionAudio {
281 /// Unique identifier for this audio response.
282 pub id: String,
283 /// Base64 encoded audio bytes generated by the model, in the format
284 /// specified in the request.
285 pub data: String,
286 /// The Unix timestamp (in seconds) for when this audio response will no
287 /// longer be accessible on the server for use in multi-turn conversations.
288 pub expires_at: u64,
289 /// Transcript of the audio generated by the model.
290 pub transcript: String,
291}
292
293/// An annotation for a chat completion message.
294#[derive(Debug, Deserialize, Serialize, Clone)]
295pub struct Annotation {
296 /// The type of the annotation. Always `url_citation`.
297 #[serde(rename = "type")]
298 pub type_: AnnotationType,
299 /// The URL citation.
300 pub url_citation: UrlCitation,
301}
302
303crate::wire_string_enum! {
304 /// The type of an annotation.
305 pub enum AnnotationType {
306 /// A URL citation when using web search.
307 UrlCitation => "url_citation",
308 }
309}
310
311/// A URL citation when the model uses a web search tool.
312#[derive(Debug, Deserialize, Serialize, Clone)]
313pub struct UrlCitation {
314 /// The index of the first character of the URL citation in the message.
315 pub start_index: usize,
316 /// The index of the last character of the URL citation in the message.
317 pub end_index: usize,
318 /// The title of the web resource.
319 pub title: String,
320 /// The URL of the web resource.
321 pub url: String,
322}
323
324#[derive(Debug, Deserialize, Serialize, Clone)]
325#[serde(tag = "type", rename_all = "snake_case")]
326pub enum ChatCompletionMessageToolCall {
327 /// The type of the tool. Currently, only `function` is supported.
328 /// The field { type = "function" } is added automatically.
329 Function {
330 /// The ID of the tool call.
331 id: String,
332 /// The function that the model called.
333 function: MessageToolCallFunction,
334 },
335 /// The type of the tool. Always `custom`.
336 /// The field { type = "custom" } is added automatically.
337 Custom {
338 /// The id of the tool call.
339 id: String,
340 /// The custom tool that the model called.
341 custom: MessageToolCallCustom,
342 },
343}
344
345#[derive(Debug, Deserialize, Serialize, Clone)]
346pub struct MessageToolCallCustom {
347 /// The input for the custom tool call generated by the model.
348 pub input: String,
349 /// The name of the custom tool to call.
350 pub name: String,
351}
352
353#[derive(Debug, Deserialize, Serialize, Clone)]
354pub struct MessageToolCallFunction {
355 /// The arguments to call the function with, as generated by the model in JSON
356 /// format. Note that the model does not always generate valid JSON, and may
357 /// hallucinate parameters not defined by your function schema. Validate the
358 /// arguments in your code before calling your function.
359 pub arguments: String,
360 /// The name of the function to call.
361 pub name: String,
362}
363
364crate::wire_string_enum! {
365 /// The role of the author of a message, as reported in responses and
366 /// streamed deltas.
367 ///
368 /// Values that are not part of the official API are preserved as
369 /// [`Role::Unknown`] instead of failing deserialization.
370 pub enum Role {
371 /// The author is the model.
372 Assistant => "assistant",
373 /// The author is a developer-defined persona.
374 Developer => "developer",
375 /// The author is the system prompt.
376 System => "system",
377 /// The author is a tool response.
378 Tool => "tool",
379 /// The author is the end user.
380 User => "user",
381 }
382}
383
384/// Legacy alias of [`Role`].
385pub type ResponseRole = Role;
386
387#[derive(Debug, Deserialize, Serialize, Clone)]
388pub struct ChoiceLogprobs {
389 /// A list of message content tokens with log probability information.
390 pub content: Option<Vec<TokenLogProb>>,
391 /// A list of reasoning content tokens with log probability information.
392 /// Only present for thinking models.
393 #[cfg(feature = "reasoning")]
394 pub reasoning_content: Option<Vec<TokenLogProb>>,
395 /// A list of message refusal tokens with log probability information.
396 pub refusal: Option<Vec<TokenLogProb>>,
397}
398
399#[derive(Debug, Deserialize, Serialize, Clone)]
400pub struct TokenLogProb {
401 /// The token.
402 pub token: String,
403 /// The log probability of this token, if it is within the top 20 most likely
404 /// tokens. Otherwise, the value `-9999.0` is used to signify that the token is very
405 /// unlikely.
406 pub logprob: f32,
407 /// A list of integers representing the UTF-8 bytes representation of the token.
408 ///
409 /// Useful in instances where characters are represented by multiple tokens and
410 /// their byte representations must be combined to generate the correct text
411 /// representation. Can be `null` if there is no bytes representation for the token.
412 pub bytes: Option<Vec<u8>>,
413 /// List of the most likely tokens and their log probability, at this token
414 /// position. In rare cases, there may be fewer than the number of requested
415 /// `top_logprobs` returned.
416 pub top_logprobs: Vec<TopLogprob>,
417}
418
419#[derive(Debug, Deserialize, Serialize, Clone)]
420pub struct TopLogprob {
421 /// The token.
422 pub token: String,
423 /// A list of integers representing the UTF-8 bytes representation of the token.
424 ///
425 /// Useful in instances where characters are represented by multiple tokens and
426 /// their byte representations must be combined to generate the correct text
427 /// representation. Can be `null` if there is no bytes representation for the token.
428 pub logprob: f32,
429 /// List of the most likely tokens and their log probability, at this token
430 /// position. In rare cases, there may be fewer than the number of requested
431 /// `top_logprobs` returned.
432 pub bytes: Option<Vec<u8>>,
433}
434
435#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
436pub struct CompletionUsage {
437 /// Number of tokens in the generated completion.
438 pub completion_tokens: u64,
439 /// Number of tokens in the prompt.
440 pub prompt_tokens: u64,
441
442 /// DeepSeek: number of tokens in the prompt that hits the context cache.
443 #[cfg(feature = "deepseek")]
444 pub prompt_cache_hit_tokens: Option<u64>,
445 /// DeepSeek: number of tokens in the prompt that misses the context cache.
446 #[cfg(feature = "deepseek")]
447 pub prompt_cache_miss_tokens: Option<u64>,
448
449 /// Total number of tokens used in the request (prompt + completion).
450 pub total_tokens: u64,
451 /// Breakdown of tokens used in a completion.
452 pub completion_tokens_details: Option<CompletionTokensDetails>,
453 /// Breakdown of tokens used in the prompt.
454 pub prompt_tokens_details: Option<PromptTokensDetails>,
455}
456
457#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
458pub struct CompletionTokensDetails {
459 /// When using Predicted Outputs, the number of tokens in the prediction that
460 /// appeared in the completion.
461 pub accepted_prediction_tokens: Option<u64>,
462 /// Audio input tokens generated by the model.
463 pub audio_tokens: Option<u64>,
464 /// Tokens generated by the model for reasoning.
465 pub reasoning_tokens: Option<u64>,
466 /// When using Predicted Outputs, the number of tokens in the prediction that did
467 /// not appear in the completion. However, like reasoning tokens, these tokens are
468 /// still counted in the total completion tokens for purposes of billing, output,
469 /// and context window limits.
470 pub rejected_prediction_tokens: Option<u64>,
471 /// vLLM: how many of the completion tokens were produced by speculative
472 /// decoding. Not part of the OpenAI schema.
473 #[cfg(feature = "vllm")]
474 pub num_speculative_tokens: Option<u64>,
475}
476
477#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
478pub struct PromptTokensDetails {
479 /// Audio input tokens present in the prompt.
480 pub audio_tokens: Option<u64>,
481 /// Cached tokens present in the prompt.
482 pub cached_tokens: Option<u64>,
483}
484
485impl FromStr for ChatCompletion {
486 type Err = crate::errors::OapiError;
487
488 fn from_str(content: &str) -> Result<Self, Self::Err> {
489 let parse_result: Result<ChatCompletion, _> = serde_json::from_str(content)
490 .map_err(|e| OapiError::DeserializationError(e.to_string()));
491 parse_result
492 }
493}
494
495#[cfg(test)]
496mod test {
497 use super::*;
498
499 #[test]
500 fn service_tier_parses_official_values() {
501 for (raw, is_fast) in [
502 (r#""auto""#, false),
503 (r#""default""#, false),
504 (r#""flex""#, false),
505 (r#""scale""#, false),
506 (r#""priority""#, false),
507 (r#""fast""#, true),
508 ] {
509 let tier: ServiceTier =
510 serde_json::from_str(raw).unwrap_or_else(|e| panic!("failed to parse {raw}: {e}"));
511 assert_eq!(matches!(tier, ServiceTier::Fast), is_fast, "raw: {raw}");
512 }
513 }
514
515 #[test]
516 fn no_streaming_example_deepseek() {
517 let json = r#"{
518 "id": "30f6413a-a827-4cf3-9898-f13a8634b798",
519 "object": "chat.completion",
520 "created": 1757944111,
521 "model": "deepseek-chat",
522 "choices": [
523 {
524 "index": 0,
525 "message": {
526 "role": "assistant",
527 "content": "Hello! How can I help you today? 😊"
528 },
529 "logprobs": null,
530 "finish_reason": "stop"
531 }
532 ],
533 "usage": {
534 "prompt_tokens": 10,
535 "completion_tokens": 11,
536 "total_tokens": 21,
537 "prompt_tokens_details": {
538 "cached_tokens": 0
539 },
540 "prompt_cache_hit_tokens": 0,
541 "prompt_cache_miss_tokens": 10
542 },
543 "system_fingerprint": "fp_08f168e49b_prod0820_fp8_kvcache"
544 }"#;
545
546 let parsed = ChatCompletion::from_str(json);
547 match parsed {
548 Ok(_) => {}
549 Err(e) => {
550 panic!("Failed to deserialize: {}", e);
551 }
552 }
553 }
554
555 #[test]
556 fn no_streaming_example_qwen() {
557 let json = r#"{
558 "choices": [
559 {
560 "message": {
561 "role": "assistant",
562 "content": "我是阿里云开发的一款超大规模语言模型,我叫通义千问。"
563 },
564 "finish_reason": "stop",
565 "index": 0,
566 "logprobs": null
567 }
568 ],
569 "object": "chat.completion",
570 "usage": {
571 "prompt_tokens": 3019,
572 "completion_tokens": 104,
573 "total_tokens": 3123,
574 "prompt_tokens_details": {
575 "cached_tokens": 2048
576 }
577 },
578 "created": 1735120033,
579 "system_fingerprint": null,
580 "model": "qwen-plus",
581 "id": "chatcmpl-6ada9ed2-7f33-9de2-8bb0-78bd4035025a"
582 }"#;
583
584 let parsed = ChatCompletion::from_str(json);
585 match parsed {
586 Ok(_) => {}
587 Err(e) => {
588 panic!("Failed to deserialize: {}", e);
589 }
590 }
591 }
592}