Skip to main content

openai_interface/chat/create/
request.rs

1//! This module contains the request body and POST method for the chat completion API.
2
3use std::collections::HashMap;
4
5use serde::{Deserialize, Serialize};
6use url::Url;
7
8use crate::{
9    chat::ServiceTier,
10    errors::OapiError,
11    rest::post::{Post, PostNoStream, PostStream},
12};
13
14/// Creates a model response for the given chat conversation.
15///
16/// # Example
17///
18/// ```rust,no_run
19/// use futures_util::StreamExt;
20/// use openai_interface::chat::create::request::{Message, RequestBody};
21/// use openai_interface::rest::{default_client, post::PostStream, RequestOptions};
22///
23/// const DEEPSEEK_CHAT_URL: &'static str = "https://api.deepseek.com";
24/// const DEEPSEEK_MODEL: &'static str = "deepseek-v4-flash";
25///
26/// #[tokio::main]
27/// async fn main() -> Result<(), Box<dyn std::error::Error>> {
28///     // Needs the `ferritls` cargo feature; drop this line if you install
29///     // your own rustls crypto provider (see `openai_interface::rest`).
30///     # #[cfg(feature = "ferritls")]
31///     openai_interface::rest::install_crypto_provider().ok();
32///
33///     let request = RequestBody {
34///         messages: vec![
35///             Message::system("This is a request of test purpose. Reply briefly"),
36///             Message::user("What's your name?"),
37///         ],
38///         model: DEEPSEEK_MODEL.to_string(),
39///         stream: Some(true),
40///         ..Default::default()
41///     };
42///
43///     let mut response = request
44///         .get_stream_response_string(&default_client(), DEEPSEEK_CHAT_URL, &RequestOptions::bearer("YOUR_API_KEY"))
45///         .await?;
46///
47///     while let Some(chunk) = response.next().await {
48///         println!("{}", chunk?);
49///     }
50///     Ok(())
51/// }
52/// ```
53#[derive(Serialize, Deserialize, Debug, Default, Clone)]
54pub struct RequestBody {
55    /// Parameters for audio output. Required when audio output is requested
56    /// with `modalities: ["audio"]`.
57    /// [Learn more](https://platform.openai.com/docs/guides/audio).
58    #[serde(skip_serializing_if = "Option::is_none")]
59    pub audio: Option<ChatCompletionAudioParam>,
60
61    /// Number between -2.0 and 2.0. Positive values penalize new tokens based on their
62    /// existing frequency in the text so far, decreasing the model's likelihood to
63    /// repeat the same line verbatim.
64    #[serde(skip_serializing_if = "Option::is_none")]
65    pub frequency_penalty: Option<f32>,
66
67    /// Whether to return log probabilities of the output tokens or not. If true,
68    /// returns the log probabilities of each output token returned in the `content` of
69    /// `message`.
70    #[serde(skip_serializing_if = "Option::is_none")]
71    pub logprobs: Option<bool>,
72
73    /// An upper bound for the number of tokens that can be generated for a completion,
74    /// including visible output tokens and reasoning tokens.
75    #[serde(skip_serializing_if = "Option::is_none")]
76    pub max_completion_tokens: Option<u32>,
77
78    /// The maximum number of tokens that can be generated in the chat completion.
79    /// Deprecated according to OpenAI's Python SDK in favour of
80    /// `max_completion_tokens`.
81    #[serde(skip_serializing_if = "Option::is_none")]
82    pub max_tokens: Option<u32>,
83
84    /// A list of messages comprising the conversation so far.
85    pub messages: Vec<Message>,
86
87    /// Modify the likelihood of specified tokens appearing in the completion.
88    ///
89    /// Accepts a JSON object that maps tokens (specified by their token ID in
90    /// the tokenizer) to an associated bias value from -100 to 100.
91    #[serde(skip_serializing_if = "Option::is_none")]
92    pub logit_bias: Option<HashMap<u32, i32>>,
93
94    /// Configuration for running moderation on the request input and
95    /// generated output.
96    #[serde(skip_serializing_if = "Option::is_none")]
97    pub moderation: Option<ChatModerationParam>,
98
99    /// Set of 16 key-value pairs that can be attached to an object. This can be useful
100    /// for storing additional information about the object in a structured format, and
101    /// querying for objects via API or the dashboard.
102    ///
103    /// Keys are strings with a maximum length of 64 characters. Values are strings with
104    /// a maximum length of 512 characters.
105    #[serde(skip_serializing_if = "Option::is_none")]
106    pub metadata: Option<HashMap<String, String>>,
107
108    /// Output types that you would like the model to generate. Most models are capable
109    /// of generating text, which is the default:
110    ///
111    /// `["text"]`
112    ///
113    /// The `gpt-4o-audio-preview` model can also be used to
114    /// [generate audio](https://platform.openai.com/docs/guides/audio). To request that
115    /// this model generate both text and audio responses, you can use:
116    ///
117    /// `["text", "audio"]`
118    #[serde(skip_serializing_if = "Option::is_none")]
119    pub modalities: Option<Vec<Modality>>,
120
121    /// Name of the model to use to generate the response.
122    pub model: String, // The type of this attribute needs improvements.
123
124    /// How many chat completion choices to generate for each input message. Note that
125    /// you will be charged based on the number of generated tokens across all of the
126    /// choices. Keep `n` as `1` to minimize costs.
127    #[serde(skip_serializing_if = "Option::is_none")]
128    pub n: Option<u32>,
129
130    /// Whether to enable
131    /// [parallel function calling](https://platform.openai.com/docs/guides/function-calling#configuring-parallel-function-calling)
132    /// during tool use.
133    #[serde(skip_serializing_if = "Option::is_none")]
134    pub parallel_tool_calls: Option<bool>,
135
136    /// Static predicted output content, such as the content of a text file that is
137    /// being regenerated.
138    #[serde(skip_serializing_if = "Option::is_none")]
139    pub prediction: Option<ChatCompletionPredictionContentParam>,
140
141    /// Number between -2.0 and 2.0. Positive values penalize new tokens based on
142    /// whether they appear in the text so far, increasing the model's likelihood to
143    /// talk about new topics.
144    #[serde(skip_serializing_if = "Option::is_none")]
145    pub presence_penalty: Option<f32>,
146
147    /// Used by OpenAI to cache responses for similar requests to optimize your cache
148    /// hit rates. Replaces the `user` field.
149    /// [Learn more](https://platform.openai.com/docs/guides/prompt-caching).
150    #[serde(skip_serializing_if = "Option::is_none")]
151    pub prompt_cache_key: Option<String>,
152
153    /// Options for prompt caching. Supported for `gpt-5.6` and later models.
154    /// By default, OpenAI automatically chooses one implicit cache breakpoint;
155    /// set `mode` to `explicit` to disable the implicit breakpoint.
156    /// [Learn more](https://platform.openai.com/docs/guides/prompt-caching).
157    #[serde(skip_serializing_if = "Option::is_none")]
158    pub prompt_cache_options: Option<PromptCacheOptions>,
159
160    /// Constrains effort on reasoning for
161    /// [reasoning models](https://platform.openai.com/docs/guides/reasoning).
162    /// Currently supported values are `none`, `minimal`, `low`, `medium`,
163    /// `high`, `xhigh`, and `max` (model-dependent). Reducing reasoning
164    /// effort can result in faster responses and fewer tokens used on
165    /// reasoning in a response. Defaults are provider- and model-dependent:
166    /// e.g. `medium` for GPT-5.5. Providers map unsupported values to the
167    /// nearest effort level.
168    #[serde(skip_serializing_if = "Option::is_none")]
169    pub reasoning_effort: Option<ReasoningEffort>,
170
171    /// specifying the format that the model must output.
172    ///
173    /// Setting to `{ "type": "json_schema", "json_schema": {...} }` enables Structured
174    /// Outputs which ensures the model will match your supplied JSON schema. Learn more
175    /// in the
176    /// [Structured Outputs guide](https://platform.openai.com/docs/guides/structured-outputs).
177    /// Setting to `{ "type": "json_object" }` enables the older JSON mode, which
178    /// ensures the message the model generates is valid JSON. Using `json_schema` is
179    /// preferred for models that support it.
180    #[serde(skip_serializing_if = "Option::is_none")]
181    pub response_format: Option<ResponseFormat>,
182
183    /// A stable identifier used to help detect users of your application that may be
184    /// violating OpenAI's usage policies. The IDs should be a string that uniquely
185    /// identifies each user. It is recommended to hash their username or email address, in
186    /// order to avoid sending any identifying information.
187    #[serde(skip_serializing_if = "Option::is_none")]
188    pub safety_identifier: Option<String>,
189
190    /// If specified, the system will make a best effort to sample deterministically. Determinism
191    /// is not guaranteed, and you should refer to the `system_fingerprint` response parameter to
192    /// monitor changes in the backend.
193    #[serde(skip_serializing_if = "Option::is_none")]
194    pub seed: Option<i64>,
195
196    /// Specifies the processing type used for serving the request.
197    ///
198    /// - If set to 'auto', then the request will be processed with the service tier
199    ///   configured in the Project settings. Unless otherwise configured, the Project
200    ///   will use 'default'.
201    /// - If set to 'default', then the request will be processed with the standard
202    ///   pricing and performance for the selected model.
203    /// - If set to '[flex](https://platform.openai.com/docs/guides/flex-processing)' or
204    ///   '[priority](https://openai.com/api-priority-processing/)', then the request
205    ///   will be processed with the corresponding service tier.
206    /// - When not set, the default behavior is 'auto'.
207    ///
208    /// When the `service_tier` parameter is set, the response body will include the
209    /// `service_tier` value based on the processing mode actually used to serve the
210    /// request. This response value may be different from the value set in the
211    /// parameter.
212    #[serde(skip_serializing_if = "Option::is_none")]
213    pub service_tier: Option<ServiceTier>,
214
215    /// Up to 4 sequences where the API will stop generating further tokens. The
216    /// returned text will not contain the stop sequence.
217    #[serde(skip_serializing_if = "Option::is_none")]
218    pub stop: Option<StopKeywords>,
219
220    /// Whether or not to store the output of this chat completion request for use in
221    /// our [model distillation](https://platform.openai.com/docs/guides/distillation)
222    /// or [evals](https://platform.openai.com/docs/guides/evals) products.
223    ///
224    /// Supports text and image inputs. Note: image inputs over 8MB will be dropped.
225    #[serde(skip_serializing_if = "Option::is_none")]
226    pub store: Option<bool>,
227
228    /// Whether to stream back partial progress. If set to `true` (or left as
229    /// `Some(true)`), tokens will be sent as data-only
230    /// [server-sent events](https://developer.mozilla.org/en-US/docs/Web/API/Server-sent_events/Using_server-sent_events#Event_stream_format)
231    /// as they become available, with the stream terminated by a `data: [DONE]`
232    /// message.
233    ///
234    /// Although it is optional, you should explicitly designate it
235    /// for an expected response.
236    #[serde(skip_serializing_if = "Option::is_none")]
237    pub stream: Option<bool>,
238
239    /// Options for streaming response. Only set this when you set `stream: true`
240    #[serde(skip_serializing_if = "Option::is_none")]
241    pub stream_options: Option<StreamOptions>,
242
243    /// What sampling temperature to use, between 0 and 2. Higher values like 0.8 will
244    /// make the output more random, while lower values like 0.2 will make it more
245    /// focused and deterministic. It is generally recommended to alter this or `top_p` but
246    /// not both.
247    #[serde(skip_serializing_if = "Option::is_none")]
248    pub temperature: Option<f32>,
249
250    /// An alternative to sampling with temperature, called nucleus sampling, where the
251    /// model considers the results of the tokens with top_p probability mass. So 0.1
252    /// means only the tokens comprising the top 10% probability mass are considered.
253    ///
254    /// It is generally recommended to alter this or `temperature` but not both.
255    #[serde(skip_serializing_if = "Option::is_none")]
256    pub top_p: Option<f32>,
257
258    /// Controls which (if any) tool is called by the model. `none` means the model will
259    /// not call any tool and instead generates a message. `auto` means the model can
260    /// pick between generating a message or calling one or more tools. `required` means
261    /// the model must call one or more tools. Specifying a particular tool via
262    /// `{"type": "function", "function": {"name": "my_function"}}` forces the model to
263    /// call that tool.
264    #[serde(skip_serializing_if = "Option::is_none")]
265    pub tool_choice: Option<ToolChoice>,
266
267    /// A list of tools the model may call.
268    #[serde(skip_serializing_if = "Option::is_none")]
269    pub tools: Option<Vec<RequestTool>>,
270
271    /// An integer between 0 and 20 specifying the number of most likely tokens to
272    /// return at each token position, each with an associated log probability.
273    /// `logprobs` must be set to `true` if this parameter is used.
274    #[serde(skip_serializing_if = "Option::is_none")]
275    pub top_logprobs: Option<u32>,
276
277    /// DeepSeek / Z.ai GLM: controls the switch between thinking and
278    /// non-thinking mode. Defaults to `enabled`. GLM also uses the nested
279    /// `clear_thinking` flag (see [`Thinking`]). See
280    /// [the DeepSeek API reference](https://api-docs.deepseek.com/api/create-chat-completion)
281    /// and [the GLM reference](https://docs.bigmodel.cn/api-reference/模型-api/对话补全).
282    #[cfg(any(feature = "deepseek", feature = "zai"))]
283    #[serde(skip_serializing_if = "Option::is_none")]
284    pub thinking: Option<Thinking>,
285
286    /// DeepSeek / Z.ai GLM: a custom end-user ID. Do not include user privacy
287    /// information. DeepSeek allows `[a-zA-Z0-9\-_]` up to 512 characters and
288    /// uses it to distinguish user identities for content-safety review,
289    /// isolate KVCache and schedule users; GLM allows 6–128 characters.
290    #[cfg(any(feature = "deepseek", feature = "zai"))]
291    #[serde(skip_serializing_if = "Option::is_none")]
292    pub user_id: Option<String>,
293
294    /// Qwen: whether to enable thinking mode for hybrid-thinking models such
295    /// as Qwen3. When set to `true`, the thinking content is returned in the
296    /// `reasoning_content` field.
297    #[cfg(feature = "qwen")]
298    #[serde(skip_serializing_if = "Option::is_none")]
299    pub enable_thinking: Option<bool>,
300    /// Qwen: the maximum number of tokens available for the model's thinking
301    /// (chain-of-thought) process.
302    #[cfg(feature = "qwen")]
303    #[serde(skip_serializing_if = "Option::is_none")]
304    pub thinking_budget: Option<u32>,
305    /// Qwen / vLLM: the size of the candidate set for sampling during
306    /// generation. Set to `null` or a value greater than 100 to disable
307    /// `top_k` sampling.
308    ///
309    /// Both providers spell this key the same way, so it lives here rather
310    /// than in the `vllm::SamplingParams` struct; defining it in both places
311    /// would emit the key twice.
312    #[cfg(any(feature = "qwen", feature = "vllm"))]
313    #[serde(skip_serializing_if = "Option::is_none")]
314    pub top_k: Option<u32>,
315
316    /// vLLM / Z.ai GLM: a caller-chosen request identifier. vLLM uses it to
317    /// replace the generated UUID (it must be unique or the server rejects the
318    /// request); GLM records it for tracing and generates one when omitted
319    /// (6–64 characters).
320    ///
321    /// Both providers spell this key the same way, so it lives here rather
322    /// than in `vllm::ChatParams`; defining it in both places would emit the
323    /// key twice.
324    #[cfg(any(feature = "vllm", feature = "zai"))]
325    #[serde(skip_serializing_if = "Option::is_none")]
326    pub request_id: Option<String>,
327
328    /// Z.ai / GLM: whether to sample (`true`, the default) using `temperature`
329    /// and `top_p`, or decode greedily (`false`), in which case both are
330    /// ignored. OpenAI has no equivalent key; greedy output is requested there
331    /// with `temperature: 0`.
332    #[cfg(feature = "zai")]
333    #[serde(skip_serializing_if = "Option::is_none")]
334    pub do_sample: Option<bool>,
335
336    /// Z.ai / GLM: whether tool-call output is streamed incrementally
337    /// (`true`) or sent whole (`false`, the default).
338    #[cfg(feature = "zai")]
339    #[serde(skip_serializing_if = "Option::is_none")]
340    pub tool_stream: Option<bool>,
341
342    /// This field is being replaced by `safety_identifier` and `prompt_cache_key`. Use
343    /// `prompt_cache_key` instead to maintain caching optimizations. A stable
344    /// identifier for your end-users. Used to boost cache hit rates by better bucketing
345    /// similar requests and to help OpenAI detect and prevent abuse.
346    /// [Learn more](https://platform.openai.com/docs/guides/safety-best-practices#safety-identifiers).
347    #[serde(skip_serializing_if = "Option::is_none")]
348    pub user: Option<String>,
349
350    /// Constrains the verbosity of the model's response. Lower values will result in
351    /// more concise responses, while higher values will result in more verbose
352    /// responses. Currently supported values are `low`, `medium`, and `high`.
353    #[serde(skip_serializing_if = "Option::is_none")]
354    pub verbosity: Option<LowMediumHighEnum>,
355
356    /// This tool searches the web for relevant results to use in a response. Learn more
357    /// about the
358    /// [web search tool](https://platform.openai.com/docs/guides/tools-web-search?api-mode=chat).
359    #[serde(rename = "web_search_options", skip_serializing_if = "Option::is_none")]
360    pub web_search_options: Option<WebSearchOptions>,
361
362    /// vLLM: extra sampling parameters (`min_p`, `repetition_penalty`,
363    /// `stop_token_ids`, `prompt_logprobs`, ...) that OpenAI's API does not
364    /// define. Flattened into the top level of the request body.
365    #[cfg(feature = "vllm")]
366    #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
367    pub vllm_sampling: Option<crate::vllm::SamplingParams>,
368
369    /// vLLM: extra chat parameters (`chat_template_kwargs`,
370    /// `structured_outputs`, `kv_transfer_params`, `priority`, ...) that
371    /// OpenAI's API does not define. Flattened into the top level of the
372    /// request body.
373    #[cfg(feature = "vllm")]
374    #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
375    pub vllm_chat: Option<crate::vllm::ChatParams>,
376
377    /// Z.ai / GLM: platform-ecosystem parameters (`watermark_enabled`) that
378    /// are tied to Zhipu's platform rather than to text generation. Flattened
379    /// into the top level of the request body. The generic GLM controls
380    /// (`do_sample`, `tool_stream`) are plain fields above instead.
381    #[cfg(feature = "zai")]
382    #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
383    pub zai_platform: Option<crate::zai::PlatformParams>,
384
385    /// Other request bodies that are not in standard OpenAI API and
386    /// not covered by the fields above.
387    #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
388    pub extra_body_map: Option<serde_json::Map<String, serde_json::Value>>,
389}
390
391/// A message in the conversation, tagged by `role`.
392///
393/// Construct messages through the convenience constructors
394/// ([`Message::system`], [`Message::user`], [`Message::assistant`],
395/// [`Message::tool`], [`Message::function`], [`Message::developer`]) or by
396/// building the payload structs directly (`Message::User(UserMessage {
397/// ..Default::default() })`), which stays source-compatible when new
398/// optional fields are added.
399///
400/// Deserialization of an unknown `role` is an error: a message with an
401/// unrecognized role cannot be forwarded, so it is treated as invalid input
402/// rather than silently mapped onto a catch-all.
403#[derive(Serialize, Deserialize, Debug, Clone)]
404#[serde(tag = "role", rename_all = "lowercase")]
405pub enum Message {
406    /// The role of the message author is `system`.
407    /// The field `{ role = "system" }` is added automatically.
408    System(SystemMessage),
409    /// The role of the message author is `user`.
410    /// The field `{ role = "user" }` is added automatically.
411    User(UserMessage),
412    /// The role of the message author is `assistant`.
413    /// The field `{ role = "assistant" }` is added automatically.
414    Assistant(AssistantMessage),
415    /// The role of the message author is `tool`.
416    /// The field `{ role = "tool" }` is added automatically.
417    Tool(ToolMessage),
418    /// The role of the message author is `function`.
419    /// The field `{ role = "function" }` is added automatically.
420    Function(FunctionMessage),
421    /// The role of the message author is `developer`.
422    /// The field `{ role = "developer" }` is added automatically.
423    Developer(DeveloperMessage),
424}
425
426impl Message {
427    /// A system message with the given content: plain text, or an array of
428    /// text content parts.
429    #[must_use]
430    pub fn system(content: impl Into<MessageContent>) -> Self {
431        Self::System(SystemMessage {
432            content: content.into(),
433            name: None,
434        })
435    }
436
437    /// A user message with the given content: plain text, or an array of
438    /// multimodal content parts (`text`, `image_url`, `input_audio`,
439    /// `file`).
440    #[must_use]
441    pub fn user(content: impl Into<MessageContent>) -> Self {
442        Self::User(UserMessage {
443            content: content.into(),
444            name: None,
445        })
446    }
447
448    /// An assistant message with the given text content. Build
449    /// [`AssistantMessage`] directly for tool calls, audio, or reasoning
450    /// content.
451    #[must_use]
452    pub fn assistant(content: impl Into<String>) -> Self {
453        Self::Assistant(AssistantMessage {
454            content: Some(content.into()),
455            ..Default::default()
456        })
457    }
458
459    /// A tool message responding to the tool call with the given ID.
460    #[must_use]
461    pub fn tool(content: impl Into<MessageContent>, tool_call_id: impl Into<String>) -> Self {
462        Self::Tool(ToolMessage {
463            content: content.into(),
464            tool_call_id: tool_call_id.into(),
465        })
466    }
467
468    /// A deprecated `function` message responding to the named function
469    /// call.
470    #[must_use]
471    pub fn function(name: impl Into<String>, content: impl Into<String>) -> Self {
472        Self::Function(FunctionMessage {
473            content: content.into(),
474            name: name.into(),
475        })
476    }
477
478    /// A developer message with the given content: plain text, or an array
479    /// of text content parts.
480    #[must_use]
481    pub fn developer(content: impl Into<MessageContent>) -> Self {
482        Self::Developer(DeveloperMessage {
483            content: content.into(),
484            name: None,
485        })
486    }
487}
488
489/// A `system` message payload.
490#[derive(Serialize, Deserialize, Debug, Clone, Default)]
491pub struct SystemMessage {
492    /// The contents of the system message: plain text, or an array of
493    /// text content parts.
494    pub content: MessageContent,
495    /// An optional name for the participant.
496    ///
497    /// Provides the model information to differentiate between
498    /// participants of the same role.
499    #[serde(skip_serializing_if = "Option::is_none")]
500    pub name: Option<String>,
501}
502
503/// A `user` message payload.
504#[derive(Serialize, Deserialize, Debug, Clone, Default)]
505pub struct UserMessage {
506    /// The contents of the user message: plain text, or an array of
507    /// multimodal content parts (`text`, `image_url`, `input_audio`,
508    /// `file`).
509    pub content: MessageContent,
510    /// An optional name for the participant.
511    ///
512    /// Provides the model information to differentiate between
513    /// participants of the same role.
514    #[serde(skip_serializing_if = "Option::is_none")]
515    pub name: Option<String>,
516}
517
518/// An `assistant` message payload.
519#[derive(Serialize, Deserialize, Debug, Clone, Default)]
520pub struct AssistantMessage {
521    /// The contents of the assistant message. Required unless `tool_calls`
522    /// or `function_call` is specified. (Note that `function_call` is deprecated
523    /// in favour of `tool_calls`.)
524    pub content: Option<String>,
525    /// Data about a previous audio response from the model. Required for
526    /// multi-turn audio conversations.
527    #[serde(skip_serializing_if = "Option::is_none")]
528    pub audio: Option<AssistantAudio>,
529    /// The refusal message by the assistant.
530    #[serde(skip_serializing_if = "Option::is_none")]
531    pub refusal: Option<String>,
532    #[serde(skip_serializing_if = "Option::is_none")]
533    pub name: Option<String>,
534    /// DeepSeek (Beta): set this to `true` to force the model to start its
535    /// answer by the content of the supplied prefix in this assistant
536    /// message. Requires `base_url = "https://api.deepseek.com/beta"`.
537    #[cfg(feature = "deepseek")]
538    #[serde(default, skip_serializing_if = "is_false")]
539    pub prefix: bool,
540    /// The reasoning contents of the assistant message produced by thinking
541    /// models (DeepSeek, Qwen3, and other reasoning models served by
542    /// OpenAI-compatible backends), before the final answer. Feed it back
543    /// in multi-turn thinking conversations; DeepSeek's Beta
544    /// [Chat Prefix Completion](https://api-docs.deepseek.com/guides/chat_prefix_completion)
545    /// also uses it as the CoT input of the last assistant message
546    /// (with `prefix` set to `true`).
547    #[cfg(feature = "reasoning")]
548    #[serde(skip_serializing_if = "Option::is_none")]
549    pub reasoning_content: Option<String>,
550
551    /// The tool calls generated by the model, such as function calls.
552    #[serde(skip_serializing_if = "Option::is_none")]
553    pub tool_calls: Option<Vec<AssistantToolCall>>,
554}
555
556/// A `tool` message payload.
557#[derive(Serialize, Deserialize, Debug, Clone, Default)]
558pub struct ToolMessage {
559    /// The contents of the tool message: plain text, or an array of
560    /// text content parts.
561    pub content: MessageContent,
562    /// Tool call that this message is responding to.
563    pub tool_call_id: String,
564}
565
566/// A deprecated `function` message payload.
567#[derive(Serialize, Deserialize, Debug, Clone, Default)]
568pub struct FunctionMessage {
569    /// The contents of the function message.
570    pub content: String,
571    /// The name of the function to call.
572    pub name: String,
573}
574
575/// A `developer` message payload.
576#[derive(Serialize, Deserialize, Debug, Clone, Default)]
577pub struct DeveloperMessage {
578    /// The contents of the developer message: plain text, or an array of
579    /// text content parts.
580    pub content: MessageContent,
581    /// An optional name for the participant.
582    ///
583    /// Provides the model information to differentiate between
584    /// participants of the same role.
585    #[serde(skip_serializing_if = "Option::is_none")]
586    pub name: Option<String>,
587}
588
589/// The contents of a user message: either plain text, or an array of
590/// multimodal content parts.
591#[derive(Debug, Serialize, Deserialize, Clone)]
592#[serde(untagged)]
593pub enum MessageContent {
594    /// A plain-text message content.
595    Text(String),
596    /// An array of multimodal content parts (`text`, `image_url`,
597    /// `input_audio`, `file`).
598    Parts(Vec<ContentPart>),
599}
600
601impl From<&str> for MessageContent {
602    fn from(value: &str) -> Self {
603        Self::Text(value.to_string())
604    }
605}
606
607impl From<String> for MessageContent {
608    fn from(value: String) -> Self {
609        Self::Text(value)
610    }
611}
612
613impl From<Vec<ContentPart>> for MessageContent {
614    fn from(value: Vec<ContentPart>) -> Self {
615        Self::Parts(value)
616    }
617}
618
619impl Default for MessageContent {
620    fn default() -> Self {
621        Self::Text(String::new())
622    }
623}
624
625/// A content part of a multimodal user message.
626#[derive(Debug, Serialize, Deserialize, Clone)]
627#[serde(tag = "type", rename_all = "snake_case")]
628pub enum ContentPart {
629    /// Learn about [text inputs](https://platform.openai.com/docs/guides/text).
630    Text {
631        /// The text content.
632        text: String,
633        /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
634        /// from the request's `prompt_cache_options.ttl`.
635        #[serde(skip_serializing_if = "Option::is_none")]
636        prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
637    },
638    /// Learn about [image inputs](https://platform.openai.com/docs/guides/vision).
639    ImageUrl {
640        /// Contains either an image URL or a data URL for a base64 encoded image.
641        image_url: ContentPartImageUrl,
642        /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
643        /// from the request's `prompt_cache_options.ttl`.
644        #[serde(skip_serializing_if = "Option::is_none")]
645        prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
646    },
647    /// Learn about [audio inputs](https://platform.openai.com/docs/guides/audio).
648    InputAudio {
649        /// The audio input data and its format.
650        input_audio: ContentPartInputAudio,
651        /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
652        /// from the request's `prompt_cache_options.ttl`.
653        #[serde(skip_serializing_if = "Option::is_none")]
654        prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
655    },
656    /// Learn about [file inputs](https://platform.openai.com/docs/guides/text).
657    File {
658        /// The file input: base64 data, an uploaded file ID, or both with a
659        /// filename.
660        file: ContentPartFile,
661        /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
662        /// from the request's `prompt_cache_options.ttl`.
663        #[serde(skip_serializing_if = "Option::is_none")]
664        prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
665    },
666}
667
668/// Marks the exact end of a reusable prompt prefix.
669#[derive(Debug, Serialize, Deserialize, Clone)]
670pub struct PromptCacheBreakpoint {
671    /// The breakpoint mode. Always `explicit`.
672    pub mode: PromptCacheBreakpointMode,
673}
674
675/// The breakpoint mode. Always `explicit`.
676#[derive(Debug, Serialize, Deserialize, Clone)]
677#[serde(rename_all = "lowercase")]
678pub enum PromptCacheBreakpointMode {
679    Explicit,
680}
681
682/// Contains either an image URL or a data URL for a base64 encoded image.
683#[derive(Debug, Serialize, Deserialize, Clone)]
684pub struct ContentPartImageUrl {
685    /// Either a URL of the image or the base64 encoded image data.
686    pub url: String,
687    /// Specifies the detail level of the image.
688    /// [Learn more](https://platform.openai.com/docs/guides/vision#low-or-high-fidelity-image-understanding).
689    ///
690    /// vLLM does not support this field and rejects requests that set it.
691    #[serde(skip_serializing_if = "Option::is_none")]
692    pub detail: Option<ImageDetail>,
693}
694
695/// The detail level of an image input.
696#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
697#[serde(rename_all = "lowercase")]
698pub enum ImageDetail {
699    Auto,
700    Low,
701    High,
702}
703
704/// Base64 encoded audio input data.
705#[derive(Debug, Serialize, Deserialize, Clone)]
706pub struct ContentPartInputAudio {
707    /// Base64 encoded audio data.
708    pub data: String,
709    /// The format of the encoded audio data. Currently supports `wav` and
710    /// `mp3`.
711    pub format: InputAudioFormat,
712}
713
714/// The format of the encoded audio data.
715#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
716#[serde(rename_all = "lowercase")]
717pub enum InputAudioFormat {
718    Wav,
719    Mp3,
720}
721
722/// A file input for a content part. At least one of `file_data` and
723/// `file_id` should be provided.
724#[derive(Debug, Serialize, Deserialize, Clone, Default)]
725pub struct ContentPartFile {
726    /// The base64 encoded file data, used when passing the file to the model
727    /// as a string.
728    #[serde(skip_serializing_if = "Option::is_none")]
729    pub file_data: Option<String>,
730    /// The ID of an uploaded file to use as input.
731    #[serde(skip_serializing_if = "Option::is_none")]
732    pub file_id: Option<String>,
733    /// The name of the file, used when passing the file to the model as a
734    /// string.
735    #[serde(skip_serializing_if = "Option::is_none")]
736    pub filename: Option<String>,
737}
738
739/// Configuration for running moderation on the request input and generated
740/// output.
741#[derive(Debug, Serialize, Deserialize, Clone)]
742pub struct ChatModerationParam {
743    /// The moderation model to use for moderated completions, e.g.
744    /// `omni-moderation-latest`.
745    pub model: String,
746    /// The policy to apply to moderated response input and output.
747    #[serde(skip_serializing_if = "Option::is_none")]
748    pub policy: Option<ModerationPolicyParam>,
749}
750
751/// The policy to apply to moderated response input and output.
752#[derive(Debug, Serialize, Deserialize, Clone, Default)]
753pub struct ModerationPolicyParam {
754    /// The moderation policy for the response input.
755    #[serde(skip_serializing_if = "Option::is_none")]
756    pub input: Option<ModerationPolicySideParam>,
757    /// The moderation policy for the response output.
758    #[serde(skip_serializing_if = "Option::is_none")]
759    pub output: Option<ModerationPolicySideParam>,
760}
761
762/// The moderation policy for one side (input or output) of the response.
763#[derive(Debug, Serialize, Deserialize, Clone)]
764pub struct ModerationPolicySideParam {
765    /// `score` returns moderation results; `block` additionally blocks
766    /// flagged content.
767    pub mode: ModerationPolicyMode,
768}
769
770/// The moderation policy mode.
771#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
772#[serde(rename_all = "lowercase")]
773pub enum ModerationPolicyMode {
774    Score,
775    Block,
776}
777
778/// Options for prompt caching.
779#[derive(Debug, Serialize, Deserialize, Clone, Default)]
780pub struct PromptCacheOptions {
781    /// Controls whether OpenAI automatically creates an implicit cache
782    /// breakpoint. Defaults to `implicit`.
783    #[serde(skip_serializing_if = "Option::is_none")]
784    pub mode: Option<PromptCacheMode>,
785    /// The minimum lifetime applied to every implicit and explicit cache
786    /// breakpoint written by the request. Defaults to `30m`, currently the
787    /// only supported value.
788    #[serde(skip_serializing_if = "Option::is_none")]
789    pub ttl: Option<PromptCacheTtl>,
790}
791
792/// The prompt cache breakpoint mode.
793#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
794#[serde(rename_all = "lowercase")]
795pub enum PromptCacheMode {
796    Implicit,
797    Explicit,
798}
799
800/// The prompt cache TTL. Currently only `30m` is supported.
801#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
802pub enum PromptCacheTtl {
803    #[serde(rename = "30m")]
804    ThirtyMinutes,
805}
806
807#[derive(Debug, Serialize, Deserialize, Clone)]
808#[serde(tag = "type", rename_all = "lowercase")]
809pub enum AssistantToolCall {
810    Function {
811        /// The ID of the tool call.
812        id: String,
813        /// The function that the model called.
814        function: ToolCallFunction,
815    },
816    Custom {
817        /// The ID of the tool call.
818        id: String,
819        /// The custom tool that the model called.
820        custom: ToolCallCustom,
821    },
822}
823
824#[derive(Debug, Serialize, Deserialize, Clone)]
825pub struct ToolCallFunction {
826    /// The arguments to call the function with, as generated by the model in JSON
827    /// format. Note that the model does not always generate valid JSON, and may
828    /// hallucinate parameters not defined by your function schema. Validate the
829    /// arguments in your code before calling your function.
830    pub arguments: String,
831    /// The name of the function to call.
832    pub name: String,
833}
834
835#[derive(Debug, Serialize, Deserialize, Clone)]
836pub struct ToolCallCustom {
837    /// The input for the custom tool call generated by the model.
838    pub input: String,
839    /// The name of the custom tool to call.
840    pub name: String,
841}
842
843/// Data about a previous audio response from the model, referenced in an
844/// assistant message for multi-turn audio conversations.
845#[derive(Debug, Serialize, Deserialize, Clone)]
846pub struct AssistantAudio {
847    /// Unique identifier for a previous audio response in a multi-turn
848    /// conversation.
849    pub id: String,
850    /// The audio data (base64 encoded) to insert as context. Optional.
851    #[serde(skip_serializing_if = "Option::is_none")]
852    pub data: Option<String>,
853}
854
855#[derive(Debug, Serialize, Deserialize, Clone)]
856#[serde(tag = "type", rename_all = "snake_case")]
857pub enum ResponseFormat {
858    /// The type of response format being defined. Always `json_schema`.
859    JsonSchema {
860        /// Structured Outputs configuration options, including a JSON Schema.
861        json_schema: JSONSchema,
862    },
863    /// The type of response format being defined. Always `json_object`.
864    JsonObject,
865    /// The type of response format being defined. Always `text`.
866    Text,
867}
868
869#[derive(Debug, Serialize, Deserialize, Clone)]
870pub struct JSONSchema {
871    /// The name of the response format. Must be a-z, A-Z, 0-9, or contain
872    /// underscores and dashes, with a maximum length of 64.
873    pub name: String,
874    /// A description of what the response format is for, used by the model to determine
875    /// how to respond in the format.
876    #[serde(skip_serializing_if = "Option::is_none")]
877    pub description: Option<String>,
878    /// The schema for the response format, described as a JSON Schema object. Learn how
879    /// to build JSON schemas [here](https://json-schema.org/).
880    #[serde(skip_serializing_if = "Option::is_none")]
881    pub schema: Option<serde_json::Map<String, serde_json::Value>>,
882    /// Whether to enable strict schema adherence when generating the output. If set to
883    /// true, the model will always follow the exact schema defined in the `schema`
884    /// field. Only a subset of JSON Schema is supported when `strict` is `true`. To
885    /// learn more, read the
886    /// [Structured Outputs guide](https://platform.openai.com/docs/guides/structured-outputs).
887    #[serde(skip_serializing_if = "Option::is_none")]
888    pub strict: Option<bool>,
889}
890
891#[derive(Serialize, Deserialize, Debug, Clone)]
892#[serde(rename_all = "snake_case")]
893pub enum Modality {
894    Text,
895    Audio,
896}
897
898/// Parameters for audio output of a chat completion.
899#[derive(Serialize, Deserialize, Debug, Clone)]
900pub struct ChatCompletionAudioParam {
901    /// Specifies the output audio format. Must be one of `wav`, `aac`, `mp3`,
902    /// `flac`, `opus`, or `pcm16`.
903    pub format: AudioFormat,
904    /// The voice the model uses to respond.
905    pub voice: Voice,
906}
907
908/// The output audio format of a chat completion.
909#[derive(Serialize, Deserialize, Debug, Clone)]
910#[serde(rename_all = "snake_case")]
911pub enum AudioFormat {
912    Wav,
913    Aac,
914    Mp3,
915    Flac,
916    Opus,
917    Pcm16,
918}
919
920/// The voice the model uses to respond with audio output.
921#[derive(Serialize, Deserialize, Debug, Clone)]
922#[serde(untagged)]
923pub enum Voice {
924    /// A built-in voice name, e.g. `alloy`, `ash`, `ballad`, `coral`, `echo`,
925    /// `sage`, `shimmer`, or `verse`.
926    BuiltIn(String),
927    /// A custom voice reference, e.g. `{ "id": "voice_1234" }`.
928    Custom {
929        /// The custom voice ID, e.g. `voice_1234`.
930        id: String,
931    },
932}
933
934#[derive(Serialize, Deserialize, Debug, Clone)]
935pub struct ChatCompletionPredictionContentParam {
936    /// The content that should be matched when generating a model response. If
937    /// generated tokens would match this content, the entire model response can be
938    /// returned much more quickly.
939    pub content: ChatCompletionPredictionContentParamContent,
940
941    /// The type of the predicted content you want to provide.
942    /// This type is currently always `content`.
943    #[serde(rename = "type")]
944    pub type_: ChatCompletionPredictionContentParamType,
945}
946
947#[derive(Serialize, Deserialize, Debug, Clone)]
948#[serde(untagged)]
949pub enum ChatCompletionPredictionContentParamContent {
950    Text(String),
951    ChatCompletionContentPartTextParam {
952        /// The text content.
953        text: String,
954        /// The type of the content part.
955        #[serde(rename = "type")]
956        type_: ChatCompletionContentPartTextParamType,
957    },
958}
959
960#[derive(Serialize, Deserialize, Debug, Clone)]
961#[serde(rename_all = "snake_case")]
962pub enum ChatCompletionContentPartTextParamType {
963    Text,
964}
965
966#[derive(Serialize, Deserialize, Debug, Clone)]
967#[serde(rename_all = "snake_case")]
968pub enum ChatCompletionPredictionContentParamType {
969    Content,
970}
971
972/// DeepSeek: skip-serialization helper for the Beta `prefix` message field.
973#[cfg(feature = "deepseek")]
974#[inline]
975fn is_false(value: &bool) -> bool {
976    !value
977}
978
979#[derive(Serialize, Deserialize, Debug, Clone)]
980#[serde(untagged)]
981pub enum StopKeywords {
982    Word(String),
983    Words(Vec<String>),
984}
985
986#[derive(Serialize, Deserialize, Debug, Clone)]
987#[serde(rename_all = "snake_case")]
988pub enum LowMediumHighEnum {
989    Low,
990    Medium,
991    High,
992}
993
994#[derive(Serialize, Deserialize, Debug, Clone, Default)]
995pub struct WebSearchOptions {
996    /// High level guidance for the amount of context window space to use for the
997    /// search. One of `low`, `medium`, or `high`. `medium` is the default.
998    #[serde(skip_serializing_if = "Option::is_none")]
999    pub search_context_size: Option<LowMediumHighEnum>,
1000
1001    #[serde(skip_serializing_if = "Option::is_none")]
1002    pub user_location: Option<WebSearchOptionsUserLocation>,
1003}
1004
1005#[derive(Serialize, Deserialize, Debug, Clone)]
1006#[serde(tag = "type", rename_all = "snake_case")]
1007pub enum WebSearchOptionsUserLocation {
1008    /// The type of location approximation. Always `approximate`.
1009    Approximate {
1010        /// Approximate location parameters for the search.
1011        approximate: WebSearchOptionsUserLocationApproximate,
1012    },
1013}
1014
1015#[derive(Serialize, Deserialize, Debug, Clone, Default)]
1016pub struct WebSearchOptionsUserLocationApproximate {
1017    /// Free text input for the city of the user, e.g. `San Francisco`.
1018    #[serde(skip_serializing_if = "Option::is_none")]
1019    pub city: Option<String>,
1020
1021    /// The two-letter [ISO country code](https://en.wikipedia.org/wiki/ISO_3166-1) of
1022    /// the user, e.g. `US`.
1023    #[serde(skip_serializing_if = "Option::is_none")]
1024    pub country: Option<String>,
1025
1026    /// Free text input for the region of the user, e.g. `California`.
1027    #[serde(skip_serializing_if = "Option::is_none")]
1028    pub region: Option<String>,
1029
1030    /// The [IANA timezone](https://timeapi.io/documentation/iana-timezones) of the
1031    /// user, e.g. `America/Los_Angeles`.
1032    #[serde(skip_serializing_if = "Option::is_none")]
1033    pub timezone: Option<String>,
1034}
1035
1036#[derive(Serialize, Deserialize, Debug, Clone)]
1037pub struct StreamOptions {
1038    /// If set, an additional chunk will be streamed before the `data: [DONE]` message.
1039    ///
1040    /// The `usage` field on this chunk shows the token usage statistics for the entire
1041    /// request, and the `choices` field will always be an empty array.
1042    ///
1043    /// All other chunks will also include a `usage` field, but with a null value.
1044    /// **NOTE:** If the stream is interrupted, you may not receive the final usage
1045    /// chunk which contains the total token usage for the request.
1046    pub include_usage: bool,
1047}
1048
1049#[derive(Serialize, Deserialize, Debug, Clone)]
1050#[serde(tag = "type", rename_all = "snake_case")]
1051pub enum RequestTool {
1052    /// The type of the tool. Currently, only `function` is supported.
1053    Function { function: ToolFunction },
1054    /// The type of the custom tool. Always `custom`.
1055    Custom {
1056        /// Properties of the custom tool.
1057        custom: ToolCustom,
1058    },
1059    /// Z.ai / GLM: the `retrieval` tool, grounding the answer in one of
1060    /// Zhipu's knowledge bases. Always `retrieval`.
1061    #[cfg(feature = "zai")]
1062    Retrieval {
1063        /// Properties of the retrieval tool.
1064        retrieval: crate::zai::RetrievalTool,
1065    },
1066    /// Z.ai / GLM: the `web_search` tool, letting the model call Zhipu's web
1067    /// search. Always `web_search`.
1068    #[cfg(feature = "zai")]
1069    WebSearch {
1070        /// Properties of the web-search tool.
1071        web_search: crate::zai::WebSearchTool,
1072    },
1073}
1074
1075#[derive(Serialize, Deserialize, Debug, Clone)]
1076pub struct ToolFunction {
1077    /// The name of the function to be called. Must be a-z, A-Z, 0-9, or
1078    /// contain underscores and dashes, with a maximum length
1079    /// of 64.
1080    pub name: String,
1081    /// A description of what the function does, used by the model to choose when and
1082    /// how to call the function.
1083    #[serde(skip_serializing_if = "Option::is_none")]
1084    pub description: Option<String>,
1085    /// The parameters the functions accepts, described as a JSON Schema object.
1086    ///
1087    /// See the
1088    /// [openai function calling guide](https://platform.openai.com/docs/guides/function-calling)
1089    /// for examples, and the
1090    /// [JSON Schema reference](https://json-schema.org/understanding-json-schema/) for
1091    /// documentation about the format.
1092    ///
1093    /// Omitting `parameters` defines a function with an empty parameter list.
1094    #[serde(skip_serializing_if = "Option::is_none")]
1095    pub parameters: Option<serde_json::Map<String, serde_json::Value>>,
1096    /// Whether to enable strict schema adherence when generating the function call.
1097    ///
1098    /// If set to true, the model will follow the exact schema defined in the
1099    /// `parameters` field. Only a subset of JSON Schema is supported when `strict` is
1100    /// `true`. Learn more about Structured Outputs in the
1101    /// [openai function calling guide](https://platform.openai.com/docs/guides/function-calling).
1102    #[serde(skip_serializing_if = "Option::is_none")]
1103    pub strict: Option<bool>,
1104}
1105
1106#[derive(Serialize, Deserialize, Debug, Clone)]
1107pub struct ToolCustom {
1108    /// The name of the custom tool, used to identify it in tool calls.
1109    pub name: String,
1110    /// Optional description of the custom tool, used to provide more context.
1111    #[serde(skip_serializing_if = "Option::is_none")]
1112    pub description: Option<String>,
1113    /// The input format for the custom tool. Default is unconstrained text.
1114    #[serde(skip_serializing_if = "Option::is_none")]
1115    pub format: Option<ToolCustomFormat>,
1116}
1117
1118#[derive(Serialize, Deserialize, Debug, Clone)]
1119#[serde(rename_all = "snake_case", tag = "type")]
1120pub enum ToolCustomFormat {
1121    /// Unconstrained text format. Always `text`.
1122    Text,
1123    /// Grammar format. Always `grammar`.
1124    Grammar {
1125        /// Your chosen grammar.
1126        grammar: ToolCustomFormatGrammarGrammar,
1127    },
1128}
1129
1130#[derive(Debug, Serialize, Deserialize, Clone)]
1131pub struct ToolCustomFormatGrammarGrammar {
1132    /// The grammar definition.
1133    pub definition: String,
1134    /// The syntax of the grammar definition. One of `lark` or `regex`.
1135    pub syntax: ToolCustomFormatGrammarGrammarSyntax,
1136}
1137
1138#[derive(Debug, Serialize, Deserialize, Clone)]
1139#[serde(rename_all = "snake_case")]
1140pub enum ToolCustomFormatGrammarGrammarSyntax {
1141    Lark,
1142    Regex,
1143}
1144
1145#[derive(Debug, Serialize, Deserialize, Clone)]
1146#[serde(rename_all = "snake_case")]
1147pub enum ToolChoice {
1148    None,
1149    Auto,
1150    Required,
1151    #[serde(untagged)]
1152    Specific(ToolChoiceSpecific),
1153}
1154
1155#[derive(Debug, Serialize, Deserialize, Clone)]
1156#[serde(rename_all = "snake_case", tag = "type")]
1157pub enum ToolChoiceSpecific {
1158    /// Allowed tool configuration type. Always `allowed_tools`.
1159    AllowedTools {
1160        /// Constrains the tools available to the model to a pre-defined set.
1161        allowed_tools: ToolChoiceAllowedTools,
1162    },
1163    /// For function calling, the type is always `function`.
1164    Function { function: ToolChoiceFunction },
1165    /// For custom tool calling, the type is always `custom`.
1166    Custom { custom: ToolChoiceCustom },
1167}
1168
1169#[derive(Debug, Serialize, Deserialize, Clone)]
1170pub struct ToolChoiceAllowedTools {
1171    /// Constrains the tools available to the model to a pre-defined set.
1172    ///
1173    /// - `auto` allows the model to pick from among the allowed tools and generate a
1174    ///   message.
1175    /// - `required` requires the model to call one or more of the allowed tools.
1176    pub mode: ToolChoiceAllowedToolsMode,
1177    /// A list of tool definitions that the model should be allowed to call.
1178    ///
1179    /// For the Chat Completions API, the list of tool definitions might look like:
1180    ///
1181    /// ```json
1182    /// [
1183    ///   { "type": "function", "function": { "name": "get_weather" } },
1184    ///   { "type": "function", "function": { "name": "get_time" } }
1185    /// ]
1186    /// ```
1187    pub tools: Vec<serde_json::Map<String, serde_json::Value>>,
1188}
1189
1190/// The mode for allowed tools in tool choice.
1191///
1192/// Controls how the model should handle the set of allowed tools:
1193///
1194/// - `auto` allows the model to pick from among the allowed tools and generate a
1195///   message.
1196/// - `required` requires the model to call one or more of the allowed tools.
1197#[derive(Debug, Serialize, Deserialize, Clone)]
1198#[serde(rename_all = "lowercase")]
1199pub enum ToolChoiceAllowedToolsMode {
1200    /// The model can choose whether to use the allowed tools or not.
1201    Auto,
1202    /// The model must use at least one of the allowed tools.
1203    Required,
1204}
1205
1206#[derive(Debug, Serialize, Deserialize, Clone)]
1207pub struct ToolChoiceFunction {
1208    /// The name of the function to call.
1209    pub name: String,
1210}
1211
1212#[derive(Debug, Serialize, Deserialize, Clone)]
1213pub struct ToolChoiceCustom {
1214    /// The name of the custom tool to call.
1215    pub name: String,
1216}
1217
1218/// Controls the switch between thinking and non-thinking mode.
1219///
1220/// Shared by DeepSeek and Z.ai / 智谱 GLM, which use the same `thinking`
1221/// key and the same `type` values; GLM additionally supports the
1222/// `clear_thinking` flag. Gated on `any(deepseek, zai)` so enabling either
1223/// provider feature makes the field available, and enabling both never
1224/// emits the `thinking` key twice.
1225#[cfg(any(feature = "deepseek", feature = "zai"))]
1226#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq, Default)]
1227pub struct Thinking {
1228    /// Whether to use thinking mode (`enabled`) or non-thinking mode
1229    /// (`disabled`). Defaults to `enabled`.
1230    #[serde(rename = "type")]
1231    pub type_: ThinkingType,
1232    /// Z.ai / GLM: whether to clear the previous turn's `reasoning_content`
1233    /// from the context (`true`, GLM's default) or preserve it (`false`).
1234    #[cfg(feature = "zai")]
1235    #[serde(skip_serializing_if = "Option::is_none")]
1236    pub clear_thinking: Option<bool>,
1237}
1238
1239/// Whether thinking mode is enabled or disabled.
1240#[cfg(any(feature = "deepseek", feature = "zai"))]
1241#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq, Default)]
1242#[serde(rename_all = "lowercase")]
1243pub enum ThinkingType {
1244    /// Thinking mode is on. The default for both DeepSeek and GLM.
1245    #[default]
1246    Enabled,
1247    /// Thinking mode is off.
1248    Disabled,
1249}
1250
1251/// Constrains the effort on reasoning for reasoning models. This is an
1252/// official OpenAI parameter; reasoning providers such as DeepSeek and Qwen
1253/// accept a subset of these values and map the rest to their nearest effort
1254/// level.
1255#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq)]
1256#[serde(rename_all = "lowercase")]
1257pub enum ReasoningEffort {
1258    None,
1259    Minimal,
1260    Low,
1261    Medium,
1262    High,
1263    Xhigh,
1264    Max,
1265}
1266
1267impl RequestBody {
1268    /// Whether this request asks for a streamed response. Defaults to
1269    /// `false` when [`RequestBody::stream`] is `None`.
1270    pub fn is_streaming(&self) -> bool {
1271        self.stream.unwrap_or(false)
1272    }
1273}
1274
1275impl Post for RequestBody {
1276    fn is_streaming(&self) -> bool {
1277        RequestBody::is_streaming(self)
1278    }
1279
1280    /// Builds the URL for the request.
1281    ///
1282    /// `base_url` should be like <https://api.openai.com/v1>
1283    fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
1284        let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
1285        url.path_segments_mut()
1286            .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
1287            .push("chat")
1288            .push("completions");
1289
1290        Ok(url.to_string())
1291    }
1292}
1293
1294impl PostNoStream for RequestBody {
1295    type Response = super::response::no_streaming::ChatCompletion;
1296}
1297
1298impl PostStream for RequestBody {
1299    type Response = super::response::streaming::ChatCompletionChunk;
1300}
1301
1302#[cfg(test)]
1303mod request_test {
1304    use futures_util::StreamExt;
1305
1306    use super::*;
1307
1308    const DEEPSEEK_CHAT_URL: &str = "https://api.deepseek.com";
1309    const DEEPSEEK_MODEL: &str = "deepseek-v4-flash";
1310
1311    fn deepseek_api_key() -> Option<String> {
1312        std::env::var("DEEPSEEK_API_KEY")
1313            .ok()
1314            .map(|key| key.trim().to_string())
1315            .filter(|key| !key.is_empty())
1316    }
1317
1318    #[tokio::test]
1319    async fn test_deepseek_no_stream() {
1320        let Some(api_key) = deepseek_api_key() else {
1321            println!("Skipping: set DEEPSEEK_API_KEY to run this test");
1322            return;
1323        };
1324
1325        let request = RequestBody {
1326            messages: vec![
1327                Message::system("This is a request of test purpose. Reply briefly"),
1328                Message::user("What's your name?"),
1329            ],
1330            model: DEEPSEEK_MODEL.to_string(),
1331            stream: Some(false),
1332            ..Default::default()
1333        };
1334
1335        let response = request
1336            .get_response_string(
1337                &crate::rest::default_client(),
1338                DEEPSEEK_CHAT_URL,
1339                &crate::rest::RequestOptions::bearer(&api_key),
1340            )
1341            .await
1342            .unwrap();
1343
1344        println!("{}", response);
1345
1346        assert!(response.to_ascii_lowercase().contains("deepseek"));
1347    }
1348
1349    #[tokio::test]
1350    async fn test_deepseek_stream() {
1351        let Some(api_key) = deepseek_api_key() else {
1352            println!("Skipping: set DEEPSEEK_API_KEY to run this test");
1353            return;
1354        };
1355
1356        let request = RequestBody {
1357            messages: vec![
1358                Message::system("This is a request of test purpose. Reply briefly"),
1359                Message::user("Who are you?"),
1360            ],
1361            model: DEEPSEEK_MODEL.to_string(),
1362            stream: Some(true),
1363            ..Default::default()
1364        };
1365
1366        let mut response = request
1367            .get_stream_response_string(
1368                &crate::rest::default_client(),
1369                DEEPSEEK_CHAT_URL,
1370                &crate::rest::RequestOptions::bearer(&api_key),
1371            )
1372            .await
1373            .unwrap();
1374
1375        while let Some(chunk) = response.next().await {
1376            println!("{}", chunk.unwrap());
1377        }
1378    }
1379
1380    /// Assistant tool calls serialize with the official `type` tag
1381    /// (`{"type":"function",...}` / `{"type":"custom",...}`), not `role`.
1382    #[test]
1383    fn assistant_tool_call_serialization() {
1384        let function_call = AssistantToolCall::Function {
1385            id: "call_abc".to_string(),
1386            function: ToolCallFunction {
1387                arguments: "{\"city\":\"paris\"}".to_string(),
1388                name: "get_weather".to_string(),
1389            },
1390        };
1391        let json = serde_json::to_string(&function_call).unwrap();
1392        assert!(json.contains(r#""type":"function""#), "json: {json}");
1393        assert!(!json.contains(r#""role""#), "json: {json}");
1394
1395        let custom_call = AssistantToolCall::Custom {
1396            id: "call_def".to_string(),
1397            custom: ToolCallCustom {
1398                input: "2+2".to_string(),
1399                name: "calculator".to_string(),
1400            },
1401        };
1402        let json = serde_json::to_string(&custom_call).unwrap();
1403        assert!(json.contains(r#""type":"custom""#), "json: {json}");
1404        assert!(!json.contains(r#""role""#), "json: {json}");
1405    }
1406
1407    /// The `prediction` parameter sends its discriminator as `type`, not
1408    /// as the Rust field name `type_`.
1409    #[test]
1410    fn prediction_type_serialization() {
1411        let prediction = ChatCompletionPredictionContentParam {
1412            content: ChatCompletionPredictionContentParamContent::Text(
1413                "The capital of France is Paris.".to_string(),
1414            ),
1415            type_: ChatCompletionPredictionContentParamType::Content,
1416        };
1417        let json = serde_json::to_string(&prediction).unwrap();
1418        assert!(json.contains(r#""type":"content""#), "json: {json}");
1419        assert!(!json.contains("type_"), "json: {json}");
1420    }
1421
1422    /// `tool_choice: allowed_tools` sends `tools` as a JSON array of tool
1423    /// definitions, matching the official `Iterable[Dict[str, object]]`.
1424    #[test]
1425    fn allowed_tools_choice_serialization() {
1426        let mut weather = serde_json::Map::new();
1427        weather.insert("type".to_string(), serde_json::json!("function"));
1428        weather.insert(
1429            "function".to_string(),
1430            serde_json::json!({ "name": "get_weather" }),
1431        );
1432
1433        let choice = ToolChoiceSpecific::AllowedTools {
1434            allowed_tools: ToolChoiceAllowedTools {
1435                mode: ToolChoiceAllowedToolsMode::Required,
1436                tools: vec![weather],
1437            },
1438        };
1439        let json = serde_json::to_string(&choice).unwrap();
1440        assert!(json.contains(r#""type":"allowed_tools""#), "json: {json}");
1441        assert!(json.contains(r#""mode":"required""#), "json: {json}");
1442        // `tools` must serialize as an array, not an object.
1443        assert!(json.contains(r#""tools":[{"#), "json: {json}");
1444    }
1445
1446    /// `web_search_options` sends `search_context_size` as optional and the
1447    /// user location nested under an `approximate` key.
1448    #[test]
1449    fn web_search_options_serialization() {
1450        let options = WebSearchOptions {
1451            search_context_size: None,
1452            user_location: Some(WebSearchOptionsUserLocation::Approximate {
1453                approximate: WebSearchOptionsUserLocationApproximate {
1454                    city: Some("San Francisco".to_string()),
1455                    country: None,
1456                    region: None,
1457                    timezone: None,
1458                },
1459            }),
1460        };
1461        let json = serde_json::to_string(&options).unwrap();
1462        assert!(!json.contains("search_context_size"), "json: {json}");
1463        assert!(json.contains(r#""type":"approximate""#), "json: {json}");
1464        assert!(
1465            json.contains(r#""approximate":{"city":"San Francisco"}"#),
1466            "json: {json}"
1467        );
1468    }
1469
1470    /// `JSONSchema`/`ToolFunction` optional fields are omitted when unset.
1471    #[test]
1472    fn json_schema_optional_fields_serialization() {
1473        let schema = JSONSchema {
1474            name: "Answer".to_string(),
1475            description: None,
1476            schema: None,
1477            strict: None,
1478        };
1479        let json = serde_json::to_string(&schema).unwrap();
1480        assert_eq!(json, r#"{"name":"Answer"}"#);
1481
1482        let function = ToolFunction {
1483            name: "get_weather".to_string(),
1484            description: None,
1485            parameters: None,
1486            strict: None,
1487        };
1488        let json = serde_json::to_string(&function).unwrap();
1489        assert_eq!(json, r#"{"name":"get_weather"}"#);
1490    }
1491
1492    /// Plain-text user messages keep the official wire format: `content`
1493    /// is a JSON string, not a parts array.
1494    #[test]
1495    fn user_text_content_serialization() {
1496        let request = RequestBody {
1497            messages: vec![Message::user("Hi")],
1498            model: "gpt-4o".to_string(),
1499            ..Default::default()
1500        };
1501
1502        let json = serde_json::to_string(&request).unwrap();
1503        assert!(json.contains(r#""content":"Hi""#), "json: {json}");
1504    }
1505
1506    /// Messages deserialize from client JSON through the payload structs
1507    /// (request-side parity for proxies and servers).
1508    #[test]
1509    fn message_deserialization() {
1510        let system: Message =
1511            serde_json::from_str(r#"{"role":"system","content":"Be terse"}"#).unwrap();
1512        assert!(matches!(
1513            system,
1514            Message::System(SystemMessage {
1515                content: MessageContent::Text(_),
1516                name: None
1517            })
1518        ));
1519
1520        let user: Message =
1521            serde_json::from_str(r#"{"role":"user","content":"Hi","name":"jimmy"}"#).unwrap();
1522        let Message::User(user) = user else {
1523            panic!("must be a user message");
1524        };
1525        assert_eq!(user.name.as_deref(), Some("jimmy"));
1526
1527        let assistant: Message = serde_json::from_str(
1528            r#"{"role":"assistant","content":null,"tool_calls":[{"type":"function","id":"call_1","function":{"name":"f","arguments":"{}"}}]}"#,
1529        )
1530        .unwrap();
1531        let Message::Assistant(assistant) = assistant else {
1532            panic!("must be an assistant message");
1533        };
1534        assert_eq!(assistant.content, None);
1535        assert_eq!(assistant.tool_calls.expect("tool calls").len(), 1);
1536
1537        let tool: Message =
1538            serde_json::from_str(r#"{"role":"tool","content":"42","tool_call_id":"call_1"}"#)
1539                .unwrap();
1540        let Message::Tool(tool) = tool else {
1541            panic!("must be a tool message");
1542        };
1543        assert_eq!(tool.tool_call_id, "call_1");
1544
1545        let developer: Message =
1546            serde_json::from_str(r#"{"role":"developer","content":"New rules"}"#).unwrap();
1547        assert!(matches!(developer, Message::Developer(_)));
1548
1549        let function: Message =
1550            serde_json::from_str(r#"{"role":"function","name":"f","content":"ok"}"#).unwrap();
1551        assert!(matches!(function, Message::Function(_)));
1552    }
1553
1554    /// An unknown role is a hard error: such a message cannot be forwarded
1555    /// to any backend, so it must not be silently mapped onto a catch-all.
1556    #[test]
1557    fn unknown_role_fails_deserialization() {
1558        let result = serde_json::from_str::<Message>(r#"{"role":"weird","content":"x"}"#);
1559        assert!(result.is_err(), "unknown roles must be rejected");
1560    }
1561
1562    /// A full request body deserializes back from client JSON; unknown
1563    /// top-level fields are captured into `extra_body_map` and survive
1564    /// re-serialization, so proxying is lossless.
1565    #[test]
1566    fn request_body_deserializes_with_extra_fields() {
1567        let json = r#"{
1568            "model": "qwen-plus",
1569            "messages": [{"role": "user", "content": "Hi"}],
1570            "stream": true,
1571            "vendor_extension": {"depth": 3}
1572        }"#;
1573        let request = serde_json::from_str::<RequestBody>(json).unwrap();
1574        assert_eq!(request.model, "qwen-plus");
1575        assert_eq!(request.stream, Some(true));
1576        assert_eq!(request.messages.len(), 1);
1577
1578        let extra = request
1579            .extra_body_map
1580            .as_ref()
1581            .expect("extra fields captured");
1582        assert_eq!(
1583            extra.get("vendor_extension"),
1584            Some(&serde_json::json!({"depth": 3}))
1585        );
1586
1587        let serialized = serde_json::to_value(&request).unwrap();
1588        assert_eq!(serialized["vendor_extension"]["depth"], 3);
1589    }
1590
1591    /// The convenience constructors produce the official wire shapes.
1592    #[test]
1593    fn message_constructors() {
1594        let request = RequestBody {
1595            messages: vec![
1596                Message::system("Be terse"),
1597                Message::user("Hi"),
1598                Message::assistant("Hello!"),
1599                Message::tool(r#"{"temp":21}"#, "call_1"),
1600            ],
1601            model: "gpt-4o".to_string(),
1602            ..Default::default()
1603        };
1604
1605        let json = serde_json::to_string(&request).unwrap();
1606        assert!(json.contains(r#""role":"system","content":"Be terse""#),);
1607        assert!(json.contains(r#""role":"user","content":"Hi""#));
1608        assert!(json.contains(r#""role":"assistant","content":"Hello!""#));
1609        assert!(
1610            json.contains(r#""role":"tool","content":"{\"temp\":21}","tool_call_id":"call_1""#)
1611        );
1612    }
1613
1614    /// System, developer and tool messages serialize `content` as a plain
1615    /// string by default and as a text-part array when parts are supplied
1616    /// (the official "string or array of content parts" shapes).
1617    #[test]
1618    fn system_developer_tool_content_serialization() {
1619        let request = RequestBody {
1620            messages: vec![
1621                Message::system("Be terse"),
1622                Message::developer(MessageContent::Parts(vec![ContentPart::Text {
1623                    text: "Prefer Rust".to_string(),
1624                    prompt_cache_breakpoint: None,
1625                }])),
1626                Message::tool(
1627                    MessageContent::Parts(vec![ContentPart::Text {
1628                        text: r#"{"temp": 21}"#.to_string(),
1629                        prompt_cache_breakpoint: None,
1630                    }]),
1631                    "call_1",
1632                ),
1633            ],
1634            model: "gpt-4o".to_string(),
1635            ..Default::default()
1636        };
1637
1638        let json = serde_json::to_string(&request).unwrap();
1639        assert!(
1640            json.contains(r#""role":"system","content":"Be terse""#),
1641            "json: {json}"
1642        );
1643        assert!(
1644            json.contains(r#""role":"developer","content":[{"type":"text","text":"Prefer Rust"}]"#),
1645            "json: {json}"
1646        );
1647        assert!(
1648            json.contains(
1649                r#""role":"tool","content":[{"type":"text","text":"{\"temp\": 21}"}],"tool_call_id":"call_1""#
1650            ),
1651            "json: {json}"
1652        );
1653    }
1654
1655    /// Multimodal user messages serialize as content-part arrays with the
1656    /// official shapes, including `prompt_cache_breakpoint`.
1657    #[test]
1658    fn multimodal_content_serialization() {
1659        let request = RequestBody {
1660            messages: vec![Message::user(MessageContent::Parts(vec![
1661                ContentPart::ImageUrl {
1662                    image_url: ContentPartImageUrl {
1663                        url: "https://example.com/cat.png".to_string(),
1664                        detail: Some(ImageDetail::High),
1665                    },
1666                    prompt_cache_breakpoint: None,
1667                },
1668                ContentPart::Text {
1669                    text: "What's in this image?".to_string(),
1670                    prompt_cache_breakpoint: Some(PromptCacheBreakpoint {
1671                        mode: PromptCacheBreakpointMode::Explicit,
1672                    }),
1673                },
1674            ]))],
1675            model: "gpt-4o".to_string(),
1676            ..Default::default()
1677        };
1678
1679        let json = serde_json::to_string(&request).unwrap();
1680        assert!(json.contains(r#""type":"image_url""#), "json: {json}");
1681        assert!(
1682            json.contains(r#""url":"https://example.com/cat.png""#),
1683            "json: {json}"
1684        );
1685        assert!(json.contains(r#""detail":"high""#), "json: {json}");
1686        assert!(json.contains(r#""type":"text""#), "json: {json}");
1687        assert!(
1688            json.contains(r#""prompt_cache_breakpoint":{"mode":"explicit"}"#),
1689            "json: {json}"
1690        );
1691    }
1692
1693    /// `input_audio` and `file` content parts serialize with the official
1694    /// shapes.
1695    #[test]
1696    fn audio_and_file_content_serialization() {
1697        let content = MessageContent::Parts(vec![
1698            ContentPart::InputAudio {
1699                input_audio: ContentPartInputAudio {
1700                    data: "aGVsbG8=".to_string(),
1701                    format: InputAudioFormat::Wav,
1702                },
1703                prompt_cache_breakpoint: None,
1704            },
1705            ContentPart::File {
1706                file: ContentPartFile {
1707                    file_id: Some("file-abc".to_string()),
1708                    ..Default::default()
1709                },
1710                prompt_cache_breakpoint: None,
1711            },
1712        ]);
1713
1714        let json = serde_json::to_string(&content).unwrap();
1715        assert!(json.contains(r#""type":"input_audio""#), "json: {json}");
1716        assert!(json.contains(r#""data":"aGVsbG8=""#), "json: {json}");
1717        assert!(json.contains(r#""format":"wav""#), "json: {json}");
1718        assert!(json.contains(r#""type":"file""#), "json: {json}");
1719        assert!(
1720            json.contains(r#""file":{"file_id":"file-abc"}"#),
1721            "json: {json}"
1722        );
1723        // Optional file fields are omitted when unset.
1724        assert!(!json.contains("file_data"), "json: {json}");
1725    }
1726
1727    /// `logit_bias`, `moderation` and `prompt_cache_options` serialize as
1728    /// the official request parameters (token-id keys as JSON strings).
1729    #[test]
1730    fn new_params_serialization() {
1731        let mut logit_bias = HashMap::new();
1732        logit_bias.insert(40u32, -100i32);
1733
1734        let request = RequestBody {
1735            messages: vec![Message::user("Hi")],
1736            model: "gpt-5".to_string(),
1737            logit_bias: Some(logit_bias),
1738            moderation: Some(ChatModerationParam {
1739                model: "omni-moderation-latest".to_string(),
1740                policy: Some(ModerationPolicyParam {
1741                    input: Some(ModerationPolicySideParam {
1742                        mode: ModerationPolicyMode::Block,
1743                    }),
1744                    output: None,
1745                }),
1746            }),
1747            prompt_cache_options: Some(PromptCacheOptions {
1748                mode: Some(PromptCacheMode::Explicit),
1749                ttl: Some(PromptCacheTtl::ThirtyMinutes),
1750            }),
1751            ..Default::default()
1752        };
1753
1754        let json = serde_json::to_string(&request).unwrap();
1755        assert!(json.contains(r#""logit_bias":{"40":-100}"#), "json: {json}");
1756        assert!(
1757            json.contains(
1758                r#""moderation":{"model":"omni-moderation-latest","policy":{"input":{"mode":"block"}}}"#
1759            ),
1760            "json: {json}"
1761        );
1762        assert!(
1763            json.contains(r#""prompt_cache_options":{"mode":"explicit","ttl":"30m"}"#),
1764            "json: {json}"
1765        );
1766    }
1767
1768    /// Serializes the OpenAI `reasoning_effort` parameter.
1769    #[test]
1770    fn reasoning_effort_serialization() {
1771        let request = RequestBody {
1772            messages: vec![Message::user("What's your name?")],
1773            model: "gpt-5".to_string(),
1774            reasoning_effort: Some(ReasoningEffort::Xhigh),
1775            ..Default::default()
1776        };
1777
1778        let json = serde_json::to_string(&request).unwrap();
1779        assert!(
1780            json.contains(r#""reasoning_effort":"xhigh""#),
1781            "json: {json}"
1782        );
1783    }
1784
1785    /// Serializes the DeepSeek Beta chat prefix completion fields.
1786    #[cfg(feature = "deepseek")]
1787    #[test]
1788    fn deepseek_assistant_prefix_serialization() {
1789        let request = RequestBody {
1790            messages: vec![
1791                Message::user("Please write quick sort code"),
1792                Message::Assistant(AssistantMessage {
1793                    content: Some("```python\n".to_string()),
1794                    prefix: true,
1795                    ..Default::default()
1796                }),
1797            ],
1798            model: DEEPSEEK_MODEL.to_string(),
1799            ..Default::default()
1800        };
1801
1802        let json = serde_json::to_string(&request).unwrap();
1803        assert!(json.contains(r#""prefix":true"#), "json: {json}");
1804    }
1805
1806    /// Serializes the DeepSeek `thinking`, `reasoning_effort` and `user_id`
1807    /// request parameters.
1808    #[cfg(feature = "deepseek")]
1809    #[test]
1810    fn deepseek_thinking_params_serialization() {
1811        let request = RequestBody {
1812            messages: vec![Message::user("What's your name?")],
1813            model: DEEPSEEK_MODEL.to_string(),
1814            thinking: Some(Thinking {
1815                type_: ThinkingType::Disabled,
1816                ..Default::default()
1817            }),
1818            user_id: Some("user-123".to_string()),
1819            ..Default::default()
1820        };
1821
1822        let json = serde_json::to_string(&request).unwrap();
1823        assert!(
1824            json.contains(r#""thinking":{"type":"disabled"}"#),
1825            "json: {json}"
1826        );
1827        assert!(json.contains(r#""user_id":"user-123""#), "json: {json}");
1828    }
1829
1830    /// Serializes the Qwen `enable_thinking`, `thinking_budget` and `top_k`
1831    /// request parameters.
1832    #[cfg(feature = "qwen")]
1833    #[test]
1834    fn qwen_params_serialization() {
1835        let request = RequestBody {
1836            messages: vec![Message::user("What's your name?")],
1837            model: "qwen-plus".to_string(),
1838            enable_thinking: Some(false),
1839            thinking_budget: Some(1024),
1840            top_k: Some(20),
1841            ..Default::default()
1842        };
1843
1844        let json = serde_json::to_string(&request).unwrap();
1845        assert!(json.contains(r#""enable_thinking":false"#), "json: {json}");
1846        assert!(json.contains(r#""thinking_budget":1024"#), "json: {json}");
1847        assert!(json.contains(r#""top_k":20"#), "json: {json}");
1848    }
1849
1850    const QWEN_CHAT_URL: &str = "https://dashscope.aliyuncs.com/compatible-mode/v1";
1851    /// Qwen's multimodal flash model: accepts text, image and audio inputs
1852    /// through its OpenAI-compatible endpoint.
1853    const QWEN_MULTIMODAL_MODEL: &str = "qwen3.8-flash";
1854
1855    fn qwen_api_key() -> Option<String> {
1856        std::env::var("QWEN_API_KEY")
1857            .ok()
1858            .map(|key| key.trim().to_string())
1859            .filter(|key| !key.is_empty())
1860    }
1861
1862    /// Real request: a user message with an `image_url` content part. The
1863    /// image is the football sample used in Alibaba Cloud Model Studio's own
1864    /// documentation. Requires `QWEN_API_KEY`; skipped otherwise.
1865    #[tokio::test]
1866    async fn test_qwen_image_input() -> Result<(), anyhow::Error> {
1867        let Some(api_key) = qwen_api_key() else {
1868            println!("Skipping: set QWEN_API_KEY to run this test");
1869            return Ok(());
1870        };
1871
1872        let request = RequestBody {
1873            messages: vec![
1874                Message::system("This is a request of test purpose. Reply briefly"),
1875                Message::user(MessageContent::Parts(vec![
1876                    ContentPart::ImageUrl {
1877                        image_url: ContentPartImageUrl {
1878                            url: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20241108/xzsgiz/football1.jpg"
1879                                .to_string(),
1880                            detail: None,
1881                        },
1882                        prompt_cache_breakpoint: None,
1883                    },
1884                    ContentPart::Text {
1885                        text: "What is shown in this image? Answer with one short sentence."
1886                            .to_string(),
1887                        prompt_cache_breakpoint: None,
1888                    },
1889                ])),
1890            ],
1891            model: QWEN_MULTIMODAL_MODEL.to_string(),
1892            ..Default::default()
1893        };
1894
1895        let response = request
1896            .get_response(
1897                &crate::rest::default_client(),
1898                QWEN_CHAT_URL,
1899                &crate::rest::RequestOptions::bearer(&api_key),
1900            )
1901            .await?;
1902
1903        let content = response.choices[0]
1904            .message
1905            .content
1906            .clone()
1907            .unwrap_or_default();
1908        println!("image response: {content}");
1909        assert!(
1910            !content.trim().is_empty(),
1911            "empty content for a valid image request"
1912        );
1913        Ok(())
1914    }
1915
1916    /// Real request: a user message with an `input_audio` content part
1917    /// carrying a public audio URL (the cherry sample from the Model Studio
1918    /// docs), answered by the streaming response. Requires `QWEN_API_KEY`;
1919    /// skipped otherwise.
1920    ///
1921    /// Uses `qwen-omni-turbo`: Qwen's Omni models are the multimodal class
1922    /// that accepts audio input on the OpenAI-compatible endpoint, and they
1923    /// require `stream: true`. (`qwen3.8-flash` rejects `input_audio` with a
1924    /// provider-side `400 incorrect modal 'audio'` error, verified with
1925    /// plain curl.)
1926    #[tokio::test]
1927    async fn test_qwen_audio_input() -> Result<(), anyhow::Error> {
1928        let Some(api_key) = qwen_api_key() else {
1929            println!("Skipping: set QWEN_API_KEY to run this test");
1930            return Ok(());
1931        };
1932
1933        let request = RequestBody {
1934            messages: vec![Message::user(MessageContent::Parts(vec![
1935                ContentPart::InputAudio {
1936                    input_audio: ContentPartInputAudio {
1937                        data: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20250211/tixcef/cherry.wav"
1938                            .to_string(),
1939                        format: InputAudioFormat::Wav,
1940                    },
1941                    prompt_cache_breakpoint: None,
1942                },
1943                ContentPart::Text {
1944                    text: "What does the speaker say in this audio? Reply briefly."
1945                        .to_string(),
1946                    prompt_cache_breakpoint: None,
1947                },
1948            ]))],
1949            model: "qwen-omni-turbo".to_string(),
1950            stream: Some(true),
1951            modalities: Some(vec![Modality::Text]),
1952            ..Default::default()
1953        };
1954
1955        let mut stream = request
1956            .get_stream_response(
1957                &crate::rest::default_client(),
1958                QWEN_CHAT_URL,
1959                &crate::rest::RequestOptions::bearer(&api_key),
1960            )
1961            .await?;
1962
1963        let mut message = String::new();
1964        while let Some(chunk) = stream.next().await {
1965            let chunk = chunk?;
1966            if let Some(choice) = chunk.choices.first()
1967                && let Some(content) = choice.delta.content.as_deref()
1968            {
1969                message.push_str(content);
1970            }
1971        }
1972
1973        println!("audio response: {message}");
1974        assert!(
1975            !message.trim().is_empty(),
1976            "empty content for a valid audio request"
1977        );
1978        Ok(())
1979    }
1980
1981    /// Real request: a plain-text user message (the wire format of
1982    /// [`MessageContent::Text`]). Requires `QWEN_API_KEY`; skipped otherwise.
1983    #[tokio::test]
1984    async fn test_qwen_text_input() -> Result<(), anyhow::Error> {
1985        let Some(api_key) = qwen_api_key() else {
1986            println!("Skipping: set QWEN_API_KEY to run this test");
1987            return Ok(());
1988        };
1989
1990        let request = RequestBody {
1991            messages: vec![Message::user("Reply with exactly one word.")],
1992            model: QWEN_MULTIMODAL_MODEL.to_string(),
1993            ..Default::default()
1994        };
1995
1996        let response = request
1997            .get_response(
1998                &crate::rest::default_client(),
1999                QWEN_CHAT_URL,
2000                &crate::rest::RequestOptions::bearer(&api_key),
2001            )
2002            .await?;
2003
2004        let content = response.choices[0]
2005            .message
2006            .content
2007            .clone()
2008            .unwrap_or_default();
2009        println!("text response: {content}");
2010        assert!(!content.trim().is_empty(), "empty content for text input");
2011        Ok(())
2012    }
2013
2014    /// Real request: streaming a multimodal (image + text) user message.
2015    /// Requires `QWEN_API_KEY`; skipped otherwise.
2016    #[tokio::test]
2017    async fn test_qwen_multimodal_stream() -> Result<(), anyhow::Error> {
2018        let Some(api_key) = qwen_api_key() else {
2019            println!("Skipping: set QWEN_API_KEY to run this test");
2020            return Ok(());
2021        };
2022
2023        let request = RequestBody {
2024            messages: vec![Message::user(MessageContent::Parts(vec![
2025                ContentPart::ImageUrl {
2026                    image_url: ContentPartImageUrl {
2027                        url: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20241108/xzsgiz/football1.jpg"
2028                            .to_string(),
2029                        detail: None,
2030                    },
2031                    prompt_cache_breakpoint: None,
2032                },
2033                ContentPart::Text {
2034                    text: "What is shown in this image? Answer with one short sentence."
2035                        .to_string(),
2036                    prompt_cache_breakpoint: None,
2037                },
2038            ]))],
2039            model: QWEN_MULTIMODAL_MODEL.to_string(),
2040            stream: Some(true),
2041            ..Default::default()
2042        };
2043
2044        let mut stream = request
2045            .get_stream_response(
2046                &crate::rest::default_client(),
2047                QWEN_CHAT_URL,
2048                &crate::rest::RequestOptions::bearer(&api_key),
2049            )
2050            .await?;
2051
2052        let mut message = String::new();
2053        while let Some(chunk) = stream.next().await {
2054            let chunk = chunk?;
2055            if let Some(choice) = chunk.choices.first()
2056                && let Some(content) = choice.delta.content.as_deref()
2057            {
2058                message.push_str(content);
2059            }
2060        }
2061
2062        println!("streamed message: {message}");
2063        assert!(
2064            !message.trim().is_empty(),
2065            "empty streamed content for a valid image request"
2066        );
2067        Ok(())
2068    }
2069}