openai_interface/chat/create/request.rs
1//! This module contains the request body and POST method for the chat completion API.
2
3use std::collections::HashMap;
4
5use serde::{Deserialize, Serialize};
6use url::Url;
7
8use crate::{
9 chat::ServiceTier,
10 errors::OapiError,
11 rest::post::{Post, PostNoStream, PostStream},
12};
13
14/// Creates a model response for the given chat conversation.
15///
16/// # Example
17///
18/// ```rust,no_run
19/// use futures_util::StreamExt;
20/// use openai_interface::chat::create::request::{Message, RequestBody};
21/// use openai_interface::rest::{default_client, post::PostStream, RequestOptions};
22///
23/// const DEEPSEEK_CHAT_URL: &'static str = "https://api.deepseek.com";
24/// const DEEPSEEK_MODEL: &'static str = "deepseek-v4-flash";
25///
26/// #[tokio::main]
27/// async fn main() -> Result<(), Box<dyn std::error::Error>> {
28/// // Needs the `ferritls` cargo feature; drop this line if you install
29/// // your own rustls crypto provider (see `openai_interface::rest`).
30/// # #[cfg(feature = "ferritls")]
31/// openai_interface::rest::install_crypto_provider().ok();
32///
33/// let request = RequestBody {
34/// messages: vec![
35/// Message::system("This is a request of test purpose. Reply briefly"),
36/// Message::user("What's your name?"),
37/// ],
38/// model: DEEPSEEK_MODEL.to_string(),
39/// stream: Some(true),
40/// ..Default::default()
41/// };
42///
43/// let mut response = request
44/// .get_stream_response_string(&default_client(), DEEPSEEK_CHAT_URL, &RequestOptions::bearer("YOUR_API_KEY"))
45/// .await?;
46///
47/// while let Some(chunk) = response.next().await {
48/// println!("{}", chunk?);
49/// }
50/// Ok(())
51/// }
52/// ```
53#[derive(Serialize, Deserialize, Debug, Default, Clone)]
54pub struct RequestBody {
55 /// Parameters for audio output. Required when audio output is requested
56 /// with `modalities: ["audio"]`.
57 /// [Learn more](https://platform.openai.com/docs/guides/audio).
58 #[serde(skip_serializing_if = "Option::is_none")]
59 pub audio: Option<ChatCompletionAudioParam>,
60
61 /// Number between -2.0 and 2.0. Positive values penalize new tokens based on their
62 /// existing frequency in the text so far, decreasing the model's likelihood to
63 /// repeat the same line verbatim.
64 #[serde(skip_serializing_if = "Option::is_none")]
65 pub frequency_penalty: Option<f32>,
66
67 /// Whether to return log probabilities of the output tokens or not. If true,
68 /// returns the log probabilities of each output token returned in the `content` of
69 /// `message`.
70 #[serde(skip_serializing_if = "Option::is_none")]
71 pub logprobs: Option<bool>,
72
73 /// An upper bound for the number of tokens that can be generated for a completion,
74 /// including visible output tokens and reasoning tokens.
75 #[serde(skip_serializing_if = "Option::is_none")]
76 pub max_completion_tokens: Option<u32>,
77
78 /// The maximum number of tokens that can be generated in the chat completion.
79 /// Deprecated according to OpenAI's Python SDK in favour of
80 /// `max_completion_tokens`.
81 #[serde(skip_serializing_if = "Option::is_none")]
82 pub max_tokens: Option<u32>,
83
84 /// A list of messages comprising the conversation so far.
85 pub messages: Vec<Message>,
86
87 /// Modify the likelihood of specified tokens appearing in the completion.
88 ///
89 /// Accepts a JSON object that maps tokens (specified by their token ID in
90 /// the tokenizer) to an associated bias value from -100 to 100.
91 #[serde(skip_serializing_if = "Option::is_none")]
92 pub logit_bias: Option<HashMap<u32, i32>>,
93
94 /// Configuration for running moderation on the request input and
95 /// generated output.
96 #[serde(skip_serializing_if = "Option::is_none")]
97 pub moderation: Option<ChatModerationParam>,
98
99 /// Set of 16 key-value pairs that can be attached to an object. This can be useful
100 /// for storing additional information about the object in a structured format, and
101 /// querying for objects via API or the dashboard.
102 ///
103 /// Keys are strings with a maximum length of 64 characters. Values are strings with
104 /// a maximum length of 512 characters.
105 #[serde(skip_serializing_if = "Option::is_none")]
106 pub metadata: Option<HashMap<String, String>>,
107
108 /// Output types that you would like the model to generate. Most models are capable
109 /// of generating text, which is the default:
110 ///
111 /// `["text"]`
112 ///
113 /// The `gpt-4o-audio-preview` model can also be used to
114 /// [generate audio](https://platform.openai.com/docs/guides/audio). To request that
115 /// this model generate both text and audio responses, you can use:
116 ///
117 /// `["text", "audio"]`
118 #[serde(skip_serializing_if = "Option::is_none")]
119 pub modalities: Option<Vec<Modality>>,
120
121 /// Name of the model to use to generate the response.
122 pub model: String, // The type of this attribute needs improvements.
123
124 /// How many chat completion choices to generate for each input message. Note that
125 /// you will be charged based on the number of generated tokens across all of the
126 /// choices. Keep `n` as `1` to minimize costs.
127 #[serde(skip_serializing_if = "Option::is_none")]
128 pub n: Option<u32>,
129
130 /// Whether to enable
131 /// [parallel function calling](https://platform.openai.com/docs/guides/function-calling#configuring-parallel-function-calling)
132 /// during tool use.
133 #[serde(skip_serializing_if = "Option::is_none")]
134 pub parallel_tool_calls: Option<bool>,
135
136 /// Static predicted output content, such as the content of a text file that is
137 /// being regenerated.
138 #[serde(skip_serializing_if = "Option::is_none")]
139 pub prediction: Option<ChatCompletionPredictionContentParam>,
140
141 /// Number between -2.0 and 2.0. Positive values penalize new tokens based on
142 /// whether they appear in the text so far, increasing the model's likelihood to
143 /// talk about new topics.
144 #[serde(skip_serializing_if = "Option::is_none")]
145 pub presence_penalty: Option<f32>,
146
147 /// Used by OpenAI to cache responses for similar requests to optimize your cache
148 /// hit rates. Replaces the `user` field.
149 /// [Learn more](https://platform.openai.com/docs/guides/prompt-caching).
150 #[serde(skip_serializing_if = "Option::is_none")]
151 pub prompt_cache_key: Option<String>,
152
153 /// Options for prompt caching. Supported for `gpt-5.6` and later models.
154 /// By default, OpenAI automatically chooses one implicit cache breakpoint;
155 /// set `mode` to `explicit` to disable the implicit breakpoint.
156 /// [Learn more](https://platform.openai.com/docs/guides/prompt-caching).
157 #[serde(skip_serializing_if = "Option::is_none")]
158 pub prompt_cache_options: Option<PromptCacheOptions>,
159
160 /// Constrains effort on reasoning for
161 /// [reasoning models](https://platform.openai.com/docs/guides/reasoning).
162 /// Currently supported values are `none`, `minimal`, `low`, `medium`,
163 /// `high`, `xhigh`, and `max` (model-dependent). Reducing reasoning
164 /// effort can result in faster responses and fewer tokens used on
165 /// reasoning in a response. Defaults are provider- and model-dependent:
166 /// e.g. `medium` for GPT-5.5. Providers map unsupported values to the
167 /// nearest effort level.
168 #[serde(skip_serializing_if = "Option::is_none")]
169 pub reasoning_effort: Option<ReasoningEffort>,
170
171 /// specifying the format that the model must output.
172 ///
173 /// Setting to `{ "type": "json_schema", "json_schema": {...} }` enables Structured
174 /// Outputs which ensures the model will match your supplied JSON schema. Learn more
175 /// in the
176 /// [Structured Outputs guide](https://platform.openai.com/docs/guides/structured-outputs).
177 /// Setting to `{ "type": "json_object" }` enables the older JSON mode, which
178 /// ensures the message the model generates is valid JSON. Using `json_schema` is
179 /// preferred for models that support it.
180 #[serde(skip_serializing_if = "Option::is_none")]
181 pub response_format: Option<ResponseFormat>,
182
183 /// A stable identifier used to help detect users of your application that may be
184 /// violating OpenAI's usage policies. The IDs should be a string that uniquely
185 /// identifies each user. It is recommended to hash their username or email address, in
186 /// order to avoid sending any identifying information.
187 #[serde(skip_serializing_if = "Option::is_none")]
188 pub safety_identifier: Option<String>,
189
190 /// If specified, the system will make a best effort to sample deterministically. Determinism
191 /// is not guaranteed, and you should refer to the `system_fingerprint` response parameter to
192 /// monitor changes in the backend.
193 #[serde(skip_serializing_if = "Option::is_none")]
194 pub seed: Option<i64>,
195
196 /// Specifies the processing type used for serving the request.
197 ///
198 /// - If set to 'auto', then the request will be processed with the service tier
199 /// configured in the Project settings. Unless otherwise configured, the Project
200 /// will use 'default'.
201 /// - If set to 'default', then the request will be processed with the standard
202 /// pricing and performance for the selected model.
203 /// - If set to '[flex](https://platform.openai.com/docs/guides/flex-processing)' or
204 /// '[priority](https://openai.com/api-priority-processing/)', then the request
205 /// will be processed with the corresponding service tier.
206 /// - When not set, the default behavior is 'auto'.
207 ///
208 /// When the `service_tier` parameter is set, the response body will include the
209 /// `service_tier` value based on the processing mode actually used to serve the
210 /// request. This response value may be different from the value set in the
211 /// parameter.
212 #[serde(skip_serializing_if = "Option::is_none")]
213 pub service_tier: Option<ServiceTier>,
214
215 /// Up to 4 sequences where the API will stop generating further tokens. The
216 /// returned text will not contain the stop sequence.
217 #[serde(skip_serializing_if = "Option::is_none")]
218 pub stop: Option<StopKeywords>,
219
220 /// Whether or not to store the output of this chat completion request for use in
221 /// our [model distillation](https://platform.openai.com/docs/guides/distillation)
222 /// or [evals](https://platform.openai.com/docs/guides/evals) products.
223 ///
224 /// Supports text and image inputs. Note: image inputs over 8MB will be dropped.
225 #[serde(skip_serializing_if = "Option::is_none")]
226 pub store: Option<bool>,
227
228 /// Whether to stream back partial progress. If set to `true` (or left as
229 /// `Some(true)`), tokens will be sent as data-only
230 /// [server-sent events](https://developer.mozilla.org/en-US/docs/Web/API/Server-sent_events/Using_server-sent_events#Event_stream_format)
231 /// as they become available, with the stream terminated by a `data: [DONE]`
232 /// message.
233 ///
234 /// Although it is optional, you should explicitly designate it
235 /// for an expected response.
236 #[serde(skip_serializing_if = "Option::is_none")]
237 pub stream: Option<bool>,
238
239 /// Options for streaming response. Only set this when you set `stream: true`
240 #[serde(skip_serializing_if = "Option::is_none")]
241 pub stream_options: Option<StreamOptions>,
242
243 /// What sampling temperature to use, between 0 and 2. Higher values like 0.8 will
244 /// make the output more random, while lower values like 0.2 will make it more
245 /// focused and deterministic. It is generally recommended to alter this or `top_p` but
246 /// not both.
247 #[serde(skip_serializing_if = "Option::is_none")]
248 pub temperature: Option<f32>,
249
250 /// An alternative to sampling with temperature, called nucleus sampling, where the
251 /// model considers the results of the tokens with top_p probability mass. So 0.1
252 /// means only the tokens comprising the top 10% probability mass are considered.
253 ///
254 /// It is generally recommended to alter this or `temperature` but not both.
255 #[serde(skip_serializing_if = "Option::is_none")]
256 pub top_p: Option<f32>,
257
258 /// Controls which (if any) tool is called by the model. `none` means the model will
259 /// not call any tool and instead generates a message. `auto` means the model can
260 /// pick between generating a message or calling one or more tools. `required` means
261 /// the model must call one or more tools. Specifying a particular tool via
262 /// `{"type": "function", "function": {"name": "my_function"}}` forces the model to
263 /// call that tool.
264 #[serde(skip_serializing_if = "Option::is_none")]
265 pub tool_choice: Option<ToolChoice>,
266
267 /// A list of tools the model may call.
268 #[serde(skip_serializing_if = "Option::is_none")]
269 pub tools: Option<Vec<RequestTool>>,
270
271 /// An integer between 0 and 20 specifying the number of most likely tokens to
272 /// return at each token position, each with an associated log probability.
273 /// `logprobs` must be set to `true` if this parameter is used.
274 #[serde(skip_serializing_if = "Option::is_none")]
275 pub top_logprobs: Option<u32>,
276
277 /// DeepSeek / Z.ai GLM: controls the switch between thinking and
278 /// non-thinking mode. Defaults to `enabled`. GLM also uses the nested
279 /// `clear_thinking` flag (see [`Thinking`]). See
280 /// [the DeepSeek API reference](https://api-docs.deepseek.com/api/create-chat-completion)
281 /// and [the GLM reference](https://docs.bigmodel.cn/api-reference/模型-api/对话补全).
282 #[cfg(any(feature = "deepseek", feature = "zai"))]
283 #[serde(skip_serializing_if = "Option::is_none")]
284 pub thinking: Option<Thinking>,
285
286 /// DeepSeek / Z.ai GLM: a custom end-user ID. Do not include user privacy
287 /// information. DeepSeek allows `[a-zA-Z0-9\-_]` up to 512 characters and
288 /// uses it to distinguish user identities for content-safety review,
289 /// isolate KVCache and schedule users; GLM allows 6–128 characters.
290 #[cfg(any(feature = "deepseek", feature = "zai"))]
291 #[serde(skip_serializing_if = "Option::is_none")]
292 pub user_id: Option<String>,
293
294 /// Qwen: whether to enable thinking mode for hybrid-thinking models such
295 /// as Qwen3. When set to `true`, the thinking content is returned in the
296 /// `reasoning_content` field.
297 #[cfg(feature = "qwen")]
298 #[serde(skip_serializing_if = "Option::is_none")]
299 pub enable_thinking: Option<bool>,
300 /// Qwen: the maximum number of tokens available for the model's thinking
301 /// (chain-of-thought) process.
302 #[cfg(feature = "qwen")]
303 #[serde(skip_serializing_if = "Option::is_none")]
304 pub thinking_budget: Option<u32>,
305 /// Qwen / vLLM: the size of the candidate set for sampling during
306 /// generation. Set to `null` or a value greater than 100 to disable
307 /// `top_k` sampling.
308 ///
309 /// Both providers spell this key the same way, so it lives here rather
310 /// than in the `vllm::SamplingParams` struct; defining it in both places
311 /// would emit the key twice.
312 #[cfg(any(feature = "qwen", feature = "vllm"))]
313 #[serde(skip_serializing_if = "Option::is_none")]
314 pub top_k: Option<u32>,
315
316 /// vLLM / Z.ai GLM: a caller-chosen request identifier. vLLM uses it to
317 /// replace the generated UUID (it must be unique or the server rejects the
318 /// request); GLM records it for tracing and generates one when omitted
319 /// (6–64 characters).
320 ///
321 /// Both providers spell this key the same way, so it lives here rather
322 /// than in `vllm::ChatParams`; defining it in both places would emit the
323 /// key twice.
324 #[cfg(any(feature = "vllm", feature = "zai"))]
325 #[serde(skip_serializing_if = "Option::is_none")]
326 pub request_id: Option<String>,
327
328 /// Z.ai / GLM: whether to sample (`true`, the default) using `temperature`
329 /// and `top_p`, or decode greedily (`false`), in which case both are
330 /// ignored. OpenAI has no equivalent key; greedy output is requested there
331 /// with `temperature: 0`.
332 #[cfg(feature = "zai")]
333 #[serde(skip_serializing_if = "Option::is_none")]
334 pub do_sample: Option<bool>,
335
336 /// Z.ai / GLM: whether tool-call output is streamed incrementally
337 /// (`true`) or sent whole (`false`, the default).
338 #[cfg(feature = "zai")]
339 #[serde(skip_serializing_if = "Option::is_none")]
340 pub tool_stream: Option<bool>,
341
342 /// This field is being replaced by `safety_identifier` and `prompt_cache_key`. Use
343 /// `prompt_cache_key` instead to maintain caching optimizations. A stable
344 /// identifier for your end-users. Used to boost cache hit rates by better bucketing
345 /// similar requests and to help OpenAI detect and prevent abuse.
346 /// [Learn more](https://platform.openai.com/docs/guides/safety-best-practices#safety-identifiers).
347 #[serde(skip_serializing_if = "Option::is_none")]
348 pub user: Option<String>,
349
350 /// Constrains the verbosity of the model's response. Lower values will result in
351 /// more concise responses, while higher values will result in more verbose
352 /// responses. Currently supported values are `low`, `medium`, and `high`.
353 #[serde(skip_serializing_if = "Option::is_none")]
354 pub verbosity: Option<LowMediumHighEnum>,
355
356 /// This tool searches the web for relevant results to use in a response. Learn more
357 /// about the
358 /// [web search tool](https://platform.openai.com/docs/guides/tools-web-search?api-mode=chat).
359 #[serde(rename = "web_search_options", skip_serializing_if = "Option::is_none")]
360 pub web_search_options: Option<WebSearchOptions>,
361
362 /// vLLM: extra sampling parameters (`min_p`, `repetition_penalty`,
363 /// `stop_token_ids`, `prompt_logprobs`, ...) that OpenAI's API does not
364 /// define. Flattened into the top level of the request body.
365 #[cfg(feature = "vllm")]
366 #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
367 pub vllm_sampling: Option<crate::vllm::SamplingParams>,
368
369 /// vLLM: extra chat parameters (`chat_template_kwargs`,
370 /// `structured_outputs`, `kv_transfer_params`, `priority`, ...) that
371 /// OpenAI's API does not define. Flattened into the top level of the
372 /// request body.
373 #[cfg(feature = "vllm")]
374 #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
375 pub vllm_chat: Option<crate::vllm::ChatParams>,
376
377 /// Z.ai / GLM: platform-ecosystem parameters (`watermark_enabled`) that
378 /// are tied to Zhipu's platform rather than to text generation. Flattened
379 /// into the top level of the request body. The generic GLM controls
380 /// (`do_sample`, `tool_stream`) are plain fields above instead.
381 #[cfg(feature = "zai")]
382 #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
383 pub zai_platform: Option<crate::zai::PlatformParams>,
384
385 /// Other request bodies that are not in standard OpenAI API and
386 /// not covered by the fields above.
387 #[serde(flatten, default, skip_serializing_if = "Option::is_none")]
388 pub extra_body_map: Option<serde_json::Map<String, serde_json::Value>>,
389}
390
391/// A message in the conversation, tagged by `role`.
392///
393/// Construct messages through the convenience constructors
394/// ([`Message::system`], [`Message::user`], [`Message::assistant`],
395/// [`Message::tool`], [`Message::function`], [`Message::developer`]) or by
396/// building the payload structs directly (`Message::User(UserMessage {
397/// ..Default::default() })`), which stays source-compatible when new
398/// optional fields are added.
399///
400/// Deserialization of an unknown `role` is an error: a message with an
401/// unrecognized role cannot be forwarded, so it is treated as invalid input
402/// rather than silently mapped onto a catch-all.
403#[derive(Serialize, Deserialize, Debug, Clone)]
404#[serde(tag = "role", rename_all = "lowercase")]
405pub enum Message {
406 /// The role of the message author is `system`.
407 /// The field `{ role = "system" }` is added automatically.
408 System(SystemMessage),
409 /// The role of the message author is `user`.
410 /// The field `{ role = "user" }` is added automatically.
411 User(UserMessage),
412 /// The role of the message author is `assistant`.
413 /// The field `{ role = "assistant" }` is added automatically.
414 Assistant(AssistantMessage),
415 /// The role of the message author is `tool`.
416 /// The field `{ role = "tool" }` is added automatically.
417 Tool(ToolMessage),
418 /// The role of the message author is `function`.
419 /// The field `{ role = "function" }` is added automatically.
420 Function(FunctionMessage),
421 /// The role of the message author is `developer`.
422 /// The field `{ role = "developer" }` is added automatically.
423 Developer(DeveloperMessage),
424}
425
426impl Message {
427 /// A system message with the given content: plain text, or an array of
428 /// text content parts.
429 #[must_use]
430 pub fn system(content: impl Into<MessageContent>) -> Self {
431 Self::System(SystemMessage {
432 content: content.into(),
433 name: None,
434 })
435 }
436
437 /// A user message with the given content: plain text, or an array of
438 /// multimodal content parts (`text`, `image_url`, `input_audio`,
439 /// `file`).
440 #[must_use]
441 pub fn user(content: impl Into<MessageContent>) -> Self {
442 Self::User(UserMessage {
443 content: content.into(),
444 name: None,
445 })
446 }
447
448 /// An assistant message with the given text content. Build
449 /// [`AssistantMessage`] directly for tool calls, audio, or reasoning
450 /// content.
451 #[must_use]
452 pub fn assistant(content: impl Into<String>) -> Self {
453 Self::Assistant(AssistantMessage {
454 content: Some(content.into()),
455 ..Default::default()
456 })
457 }
458
459 /// A tool message responding to the tool call with the given ID.
460 #[must_use]
461 pub fn tool(content: impl Into<MessageContent>, tool_call_id: impl Into<String>) -> Self {
462 Self::Tool(ToolMessage {
463 content: content.into(),
464 tool_call_id: tool_call_id.into(),
465 })
466 }
467
468 /// A deprecated `function` message responding to the named function
469 /// call.
470 #[must_use]
471 pub fn function(name: impl Into<String>, content: impl Into<String>) -> Self {
472 Self::Function(FunctionMessage {
473 content: content.into(),
474 name: name.into(),
475 })
476 }
477
478 /// A developer message with the given content: plain text, or an array
479 /// of text content parts.
480 #[must_use]
481 pub fn developer(content: impl Into<MessageContent>) -> Self {
482 Self::Developer(DeveloperMessage {
483 content: content.into(),
484 name: None,
485 })
486 }
487}
488
489/// A `system` message payload.
490#[derive(Serialize, Deserialize, Debug, Clone, Default)]
491pub struct SystemMessage {
492 /// The contents of the system message: plain text, or an array of
493 /// text content parts.
494 pub content: MessageContent,
495 /// An optional name for the participant.
496 ///
497 /// Provides the model information to differentiate between
498 /// participants of the same role.
499 #[serde(skip_serializing_if = "Option::is_none")]
500 pub name: Option<String>,
501}
502
503/// A `user` message payload.
504#[derive(Serialize, Deserialize, Debug, Clone, Default)]
505pub struct UserMessage {
506 /// The contents of the user message: plain text, or an array of
507 /// multimodal content parts (`text`, `image_url`, `input_audio`,
508 /// `file`).
509 pub content: MessageContent,
510 /// An optional name for the participant.
511 ///
512 /// Provides the model information to differentiate between
513 /// participants of the same role.
514 #[serde(skip_serializing_if = "Option::is_none")]
515 pub name: Option<String>,
516}
517
518/// An `assistant` message payload.
519#[derive(Serialize, Deserialize, Debug, Clone, Default)]
520pub struct AssistantMessage {
521 /// The contents of the assistant message. Required unless `tool_calls`
522 /// or `function_call` is specified. (Note that `function_call` is deprecated
523 /// in favour of `tool_calls`.)
524 pub content: Option<String>,
525 /// Data about a previous audio response from the model. Required for
526 /// multi-turn audio conversations.
527 #[serde(skip_serializing_if = "Option::is_none")]
528 pub audio: Option<AssistantAudio>,
529 /// The refusal message by the assistant.
530 #[serde(skip_serializing_if = "Option::is_none")]
531 pub refusal: Option<String>,
532 #[serde(skip_serializing_if = "Option::is_none")]
533 pub name: Option<String>,
534 /// DeepSeek (Beta): set this to `true` to force the model to start its
535 /// answer by the content of the supplied prefix in this assistant
536 /// message. Requires `base_url = "https://api.deepseek.com/beta"`.
537 #[cfg(feature = "deepseek")]
538 #[serde(default, skip_serializing_if = "is_false")]
539 pub prefix: bool,
540 /// The reasoning contents of the assistant message produced by thinking
541 /// models (DeepSeek, Qwen3, and other reasoning models served by
542 /// OpenAI-compatible backends), before the final answer. Feed it back
543 /// in multi-turn thinking conversations; DeepSeek's Beta
544 /// [Chat Prefix Completion](https://api-docs.deepseek.com/guides/chat_prefix_completion)
545 /// also uses it as the CoT input of the last assistant message
546 /// (with `prefix` set to `true`).
547 #[cfg(feature = "reasoning")]
548 #[serde(skip_serializing_if = "Option::is_none")]
549 pub reasoning_content: Option<String>,
550
551 /// The tool calls generated by the model, such as function calls.
552 #[serde(skip_serializing_if = "Option::is_none")]
553 pub tool_calls: Option<Vec<AssistantToolCall>>,
554}
555
556/// A `tool` message payload.
557#[derive(Serialize, Deserialize, Debug, Clone, Default)]
558pub struct ToolMessage {
559 /// The contents of the tool message: plain text, or an array of
560 /// text content parts.
561 pub content: MessageContent,
562 /// Tool call that this message is responding to.
563 pub tool_call_id: String,
564}
565
566/// A deprecated `function` message payload.
567#[derive(Serialize, Deserialize, Debug, Clone, Default)]
568pub struct FunctionMessage {
569 /// The contents of the function message.
570 pub content: String,
571 /// The name of the function to call.
572 pub name: String,
573}
574
575/// A `developer` message payload.
576#[derive(Serialize, Deserialize, Debug, Clone, Default)]
577pub struct DeveloperMessage {
578 /// The contents of the developer message: plain text, or an array of
579 /// text content parts.
580 pub content: MessageContent,
581 /// An optional name for the participant.
582 ///
583 /// Provides the model information to differentiate between
584 /// participants of the same role.
585 #[serde(skip_serializing_if = "Option::is_none")]
586 pub name: Option<String>,
587}
588
589/// The contents of a user message: either plain text, or an array of
590/// multimodal content parts.
591#[derive(Debug, Serialize, Deserialize, Clone)]
592#[serde(untagged)]
593pub enum MessageContent {
594 /// A plain-text message content.
595 Text(String),
596 /// An array of multimodal content parts (`text`, `image_url`,
597 /// `input_audio`, `file`).
598 Parts(Vec<ContentPart>),
599}
600
601impl From<&str> for MessageContent {
602 fn from(value: &str) -> Self {
603 Self::Text(value.to_string())
604 }
605}
606
607impl From<String> for MessageContent {
608 fn from(value: String) -> Self {
609 Self::Text(value)
610 }
611}
612
613impl From<Vec<ContentPart>> for MessageContent {
614 fn from(value: Vec<ContentPart>) -> Self {
615 Self::Parts(value)
616 }
617}
618
619impl Default for MessageContent {
620 fn default() -> Self {
621 Self::Text(String::new())
622 }
623}
624
625/// A content part of a multimodal user message.
626#[derive(Debug, Serialize, Deserialize, Clone)]
627#[serde(tag = "type", rename_all = "snake_case")]
628pub enum ContentPart {
629 /// Learn about [text inputs](https://platform.openai.com/docs/guides/text).
630 Text {
631 /// The text content.
632 text: String,
633 /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
634 /// from the request's `prompt_cache_options.ttl`.
635 #[serde(skip_serializing_if = "Option::is_none")]
636 prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
637 },
638 /// Learn about [image inputs](https://platform.openai.com/docs/guides/vision).
639 ImageUrl {
640 /// Contains either an image URL or a data URL for a base64 encoded image.
641 image_url: ContentPartImageUrl,
642 /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
643 /// from the request's `prompt_cache_options.ttl`.
644 #[serde(skip_serializing_if = "Option::is_none")]
645 prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
646 },
647 /// Learn about [audio inputs](https://platform.openai.com/docs/guides/audio).
648 InputAudio {
649 /// The audio input data and its format.
650 input_audio: ContentPartInputAudio,
651 /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
652 /// from the request's `prompt_cache_options.ttl`.
653 #[serde(skip_serializing_if = "Option::is_none")]
654 prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
655 },
656 /// Learn about [file inputs](https://platform.openai.com/docs/guides/text).
657 File {
658 /// The file input: base64 data, an uploaded file ID, or both with a
659 /// filename.
660 file: ContentPartFile,
661 /// Marks the exact end of a reusable prompt prefix. Inherits its TTL
662 /// from the request's `prompt_cache_options.ttl`.
663 #[serde(skip_serializing_if = "Option::is_none")]
664 prompt_cache_breakpoint: Option<PromptCacheBreakpoint>,
665 },
666}
667
668/// Marks the exact end of a reusable prompt prefix.
669#[derive(Debug, Serialize, Deserialize, Clone)]
670pub struct PromptCacheBreakpoint {
671 /// The breakpoint mode. Always `explicit`.
672 pub mode: PromptCacheBreakpointMode,
673}
674
675/// The breakpoint mode. Always `explicit`.
676#[derive(Debug, Serialize, Deserialize, Clone)]
677#[serde(rename_all = "lowercase")]
678pub enum PromptCacheBreakpointMode {
679 Explicit,
680}
681
682/// Contains either an image URL or a data URL for a base64 encoded image.
683#[derive(Debug, Serialize, Deserialize, Clone)]
684pub struct ContentPartImageUrl {
685 /// Either a URL of the image or the base64 encoded image data.
686 pub url: String,
687 /// Specifies the detail level of the image.
688 /// [Learn more](https://platform.openai.com/docs/guides/vision#low-or-high-fidelity-image-understanding).
689 ///
690 /// vLLM does not support this field and rejects requests that set it.
691 #[serde(skip_serializing_if = "Option::is_none")]
692 pub detail: Option<ImageDetail>,
693}
694
695/// The detail level of an image input.
696#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
697#[serde(rename_all = "lowercase")]
698pub enum ImageDetail {
699 Auto,
700 Low,
701 High,
702}
703
704/// Base64 encoded audio input data.
705#[derive(Debug, Serialize, Deserialize, Clone)]
706pub struct ContentPartInputAudio {
707 /// Base64 encoded audio data.
708 pub data: String,
709 /// The format of the encoded audio data. Currently supports `wav` and
710 /// `mp3`.
711 pub format: InputAudioFormat,
712}
713
714/// The format of the encoded audio data.
715#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
716#[serde(rename_all = "lowercase")]
717pub enum InputAudioFormat {
718 Wav,
719 Mp3,
720}
721
722/// A file input for a content part. At least one of `file_data` and
723/// `file_id` should be provided.
724#[derive(Debug, Serialize, Deserialize, Clone, Default)]
725pub struct ContentPartFile {
726 /// The base64 encoded file data, used when passing the file to the model
727 /// as a string.
728 #[serde(skip_serializing_if = "Option::is_none")]
729 pub file_data: Option<String>,
730 /// The ID of an uploaded file to use as input.
731 #[serde(skip_serializing_if = "Option::is_none")]
732 pub file_id: Option<String>,
733 /// The name of the file, used when passing the file to the model as a
734 /// string.
735 #[serde(skip_serializing_if = "Option::is_none")]
736 pub filename: Option<String>,
737}
738
739/// Configuration for running moderation on the request input and generated
740/// output.
741#[derive(Debug, Serialize, Deserialize, Clone)]
742pub struct ChatModerationParam {
743 /// The moderation model to use for moderated completions, e.g.
744 /// `omni-moderation-latest`.
745 pub model: String,
746 /// The policy to apply to moderated response input and output.
747 #[serde(skip_serializing_if = "Option::is_none")]
748 pub policy: Option<ModerationPolicyParam>,
749}
750
751/// The policy to apply to moderated response input and output.
752#[derive(Debug, Serialize, Deserialize, Clone, Default)]
753pub struct ModerationPolicyParam {
754 /// The moderation policy for the response input.
755 #[serde(skip_serializing_if = "Option::is_none")]
756 pub input: Option<ModerationPolicySideParam>,
757 /// The moderation policy for the response output.
758 #[serde(skip_serializing_if = "Option::is_none")]
759 pub output: Option<ModerationPolicySideParam>,
760}
761
762/// The moderation policy for one side (input or output) of the response.
763#[derive(Debug, Serialize, Deserialize, Clone)]
764pub struct ModerationPolicySideParam {
765 /// `score` returns moderation results; `block` additionally blocks
766 /// flagged content.
767 pub mode: ModerationPolicyMode,
768}
769
770/// The moderation policy mode.
771#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
772#[serde(rename_all = "lowercase")]
773pub enum ModerationPolicyMode {
774 Score,
775 Block,
776}
777
778/// Options for prompt caching.
779#[derive(Debug, Serialize, Deserialize, Clone, Default)]
780pub struct PromptCacheOptions {
781 /// Controls whether OpenAI automatically creates an implicit cache
782 /// breakpoint. Defaults to `implicit`.
783 #[serde(skip_serializing_if = "Option::is_none")]
784 pub mode: Option<PromptCacheMode>,
785 /// The minimum lifetime applied to every implicit and explicit cache
786 /// breakpoint written by the request. Defaults to `30m`, currently the
787 /// only supported value.
788 #[serde(skip_serializing_if = "Option::is_none")]
789 pub ttl: Option<PromptCacheTtl>,
790}
791
792/// The prompt cache breakpoint mode.
793#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
794#[serde(rename_all = "lowercase")]
795pub enum PromptCacheMode {
796 Implicit,
797 Explicit,
798}
799
800/// The prompt cache TTL. Currently only `30m` is supported.
801#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
802pub enum PromptCacheTtl {
803 #[serde(rename = "30m")]
804 ThirtyMinutes,
805}
806
807#[derive(Debug, Serialize, Deserialize, Clone)]
808#[serde(tag = "type", rename_all = "lowercase")]
809pub enum AssistantToolCall {
810 Function {
811 /// The ID of the tool call.
812 id: String,
813 /// The function that the model called.
814 function: ToolCallFunction,
815 },
816 Custom {
817 /// The ID of the tool call.
818 id: String,
819 /// The custom tool that the model called.
820 custom: ToolCallCustom,
821 },
822}
823
824#[derive(Debug, Serialize, Deserialize, Clone)]
825pub struct ToolCallFunction {
826 /// The arguments to call the function with, as generated by the model in JSON
827 /// format. Note that the model does not always generate valid JSON, and may
828 /// hallucinate parameters not defined by your function schema. Validate the
829 /// arguments in your code before calling your function.
830 pub arguments: String,
831 /// The name of the function to call.
832 pub name: String,
833}
834
835#[derive(Debug, Serialize, Deserialize, Clone)]
836pub struct ToolCallCustom {
837 /// The input for the custom tool call generated by the model.
838 pub input: String,
839 /// The name of the custom tool to call.
840 pub name: String,
841}
842
843/// Data about a previous audio response from the model, referenced in an
844/// assistant message for multi-turn audio conversations.
845#[derive(Debug, Serialize, Deserialize, Clone)]
846pub struct AssistantAudio {
847 /// Unique identifier for a previous audio response in a multi-turn
848 /// conversation.
849 pub id: String,
850 /// The audio data (base64 encoded) to insert as context. Optional.
851 #[serde(skip_serializing_if = "Option::is_none")]
852 pub data: Option<String>,
853}
854
855#[derive(Debug, Serialize, Deserialize, Clone)]
856#[serde(tag = "type", rename_all = "snake_case")]
857pub enum ResponseFormat {
858 /// The type of response format being defined. Always `json_schema`.
859 JsonSchema {
860 /// Structured Outputs configuration options, including a JSON Schema.
861 json_schema: JSONSchema,
862 },
863 /// The type of response format being defined. Always `json_object`.
864 JsonObject,
865 /// The type of response format being defined. Always `text`.
866 Text,
867}
868
869#[derive(Debug, Serialize, Deserialize, Clone)]
870pub struct JSONSchema {
871 /// The name of the response format. Must be a-z, A-Z, 0-9, or contain
872 /// underscores and dashes, with a maximum length of 64.
873 pub name: String,
874 /// A description of what the response format is for, used by the model to determine
875 /// how to respond in the format.
876 #[serde(skip_serializing_if = "Option::is_none")]
877 pub description: Option<String>,
878 /// The schema for the response format, described as a JSON Schema object. Learn how
879 /// to build JSON schemas [here](https://json-schema.org/).
880 #[serde(skip_serializing_if = "Option::is_none")]
881 pub schema: Option<serde_json::Map<String, serde_json::Value>>,
882 /// Whether to enable strict schema adherence when generating the output. If set to
883 /// true, the model will always follow the exact schema defined in the `schema`
884 /// field. Only a subset of JSON Schema is supported when `strict` is `true`. To
885 /// learn more, read the
886 /// [Structured Outputs guide](https://platform.openai.com/docs/guides/structured-outputs).
887 #[serde(skip_serializing_if = "Option::is_none")]
888 pub strict: Option<bool>,
889}
890
891#[derive(Serialize, Deserialize, Debug, Clone)]
892#[serde(rename_all = "snake_case")]
893pub enum Modality {
894 Text,
895 Audio,
896}
897
898/// Parameters for audio output of a chat completion.
899#[derive(Serialize, Deserialize, Debug, Clone)]
900pub struct ChatCompletionAudioParam {
901 /// Specifies the output audio format. Must be one of `wav`, `aac`, `mp3`,
902 /// `flac`, `opus`, or `pcm16`.
903 pub format: AudioFormat,
904 /// The voice the model uses to respond.
905 pub voice: Voice,
906}
907
908/// The output audio format of a chat completion.
909#[derive(Serialize, Deserialize, Debug, Clone)]
910#[serde(rename_all = "snake_case")]
911pub enum AudioFormat {
912 Wav,
913 Aac,
914 Mp3,
915 Flac,
916 Opus,
917 Pcm16,
918}
919
920/// The voice the model uses to respond with audio output.
921#[derive(Serialize, Deserialize, Debug, Clone)]
922#[serde(untagged)]
923pub enum Voice {
924 /// A built-in voice name, e.g. `alloy`, `ash`, `ballad`, `coral`, `echo`,
925 /// `sage`, `shimmer`, or `verse`.
926 BuiltIn(String),
927 /// A custom voice reference, e.g. `{ "id": "voice_1234" }`.
928 Custom {
929 /// The custom voice ID, e.g. `voice_1234`.
930 id: String,
931 },
932}
933
934#[derive(Serialize, Deserialize, Debug, Clone)]
935pub struct ChatCompletionPredictionContentParam {
936 /// The content that should be matched when generating a model response. If
937 /// generated tokens would match this content, the entire model response can be
938 /// returned much more quickly.
939 pub content: ChatCompletionPredictionContentParamContent,
940
941 /// The type of the predicted content you want to provide.
942 /// This type is currently always `content`.
943 #[serde(rename = "type")]
944 pub type_: ChatCompletionPredictionContentParamType,
945}
946
947#[derive(Serialize, Deserialize, Debug, Clone)]
948#[serde(untagged)]
949pub enum ChatCompletionPredictionContentParamContent {
950 Text(String),
951 ChatCompletionContentPartTextParam {
952 /// The text content.
953 text: String,
954 /// The type of the content part.
955 #[serde(rename = "type")]
956 type_: ChatCompletionContentPartTextParamType,
957 },
958}
959
960#[derive(Serialize, Deserialize, Debug, Clone)]
961#[serde(rename_all = "snake_case")]
962pub enum ChatCompletionContentPartTextParamType {
963 Text,
964}
965
966#[derive(Serialize, Deserialize, Debug, Clone)]
967#[serde(rename_all = "snake_case")]
968pub enum ChatCompletionPredictionContentParamType {
969 Content,
970}
971
972/// DeepSeek: skip-serialization helper for the Beta `prefix` message field.
973#[cfg(feature = "deepseek")]
974#[inline]
975fn is_false(value: &bool) -> bool {
976 !value
977}
978
979#[derive(Serialize, Deserialize, Debug, Clone)]
980#[serde(untagged)]
981pub enum StopKeywords {
982 Word(String),
983 Words(Vec<String>),
984}
985
986#[derive(Serialize, Deserialize, Debug, Clone)]
987#[serde(rename_all = "snake_case")]
988pub enum LowMediumHighEnum {
989 Low,
990 Medium,
991 High,
992}
993
994#[derive(Serialize, Deserialize, Debug, Clone, Default)]
995pub struct WebSearchOptions {
996 /// High level guidance for the amount of context window space to use for the
997 /// search. One of `low`, `medium`, or `high`. `medium` is the default.
998 #[serde(skip_serializing_if = "Option::is_none")]
999 pub search_context_size: Option<LowMediumHighEnum>,
1000
1001 #[serde(skip_serializing_if = "Option::is_none")]
1002 pub user_location: Option<WebSearchOptionsUserLocation>,
1003}
1004
1005#[derive(Serialize, Deserialize, Debug, Clone)]
1006#[serde(tag = "type", rename_all = "snake_case")]
1007pub enum WebSearchOptionsUserLocation {
1008 /// The type of location approximation. Always `approximate`.
1009 Approximate {
1010 /// Approximate location parameters for the search.
1011 approximate: WebSearchOptionsUserLocationApproximate,
1012 },
1013}
1014
1015#[derive(Serialize, Deserialize, Debug, Clone, Default)]
1016pub struct WebSearchOptionsUserLocationApproximate {
1017 /// Free text input for the city of the user, e.g. `San Francisco`.
1018 #[serde(skip_serializing_if = "Option::is_none")]
1019 pub city: Option<String>,
1020
1021 /// The two-letter [ISO country code](https://en.wikipedia.org/wiki/ISO_3166-1) of
1022 /// the user, e.g. `US`.
1023 #[serde(skip_serializing_if = "Option::is_none")]
1024 pub country: Option<String>,
1025
1026 /// Free text input for the region of the user, e.g. `California`.
1027 #[serde(skip_serializing_if = "Option::is_none")]
1028 pub region: Option<String>,
1029
1030 /// The [IANA timezone](https://timeapi.io/documentation/iana-timezones) of the
1031 /// user, e.g. `America/Los_Angeles`.
1032 #[serde(skip_serializing_if = "Option::is_none")]
1033 pub timezone: Option<String>,
1034}
1035
1036#[derive(Serialize, Deserialize, Debug, Clone)]
1037pub struct StreamOptions {
1038 /// If set, an additional chunk will be streamed before the `data: [DONE]` message.
1039 ///
1040 /// The `usage` field on this chunk shows the token usage statistics for the entire
1041 /// request, and the `choices` field will always be an empty array.
1042 ///
1043 /// All other chunks will also include a `usage` field, but with a null value.
1044 /// **NOTE:** If the stream is interrupted, you may not receive the final usage
1045 /// chunk which contains the total token usage for the request.
1046 pub include_usage: bool,
1047}
1048
1049#[derive(Serialize, Deserialize, Debug, Clone)]
1050#[serde(tag = "type", rename_all = "snake_case")]
1051pub enum RequestTool {
1052 /// The type of the tool. Currently, only `function` is supported.
1053 Function { function: ToolFunction },
1054 /// The type of the custom tool. Always `custom`.
1055 Custom {
1056 /// Properties of the custom tool.
1057 custom: ToolCustom,
1058 },
1059 /// Z.ai / GLM: the `retrieval` tool, grounding the answer in one of
1060 /// Zhipu's knowledge bases. Always `retrieval`.
1061 #[cfg(feature = "zai")]
1062 Retrieval {
1063 /// Properties of the retrieval tool.
1064 retrieval: crate::zai::RetrievalTool,
1065 },
1066 /// Z.ai / GLM: the `web_search` tool, letting the model call Zhipu's web
1067 /// search. Always `web_search`.
1068 #[cfg(feature = "zai")]
1069 WebSearch {
1070 /// Properties of the web-search tool.
1071 web_search: crate::zai::WebSearchTool,
1072 },
1073}
1074
1075#[derive(Serialize, Deserialize, Debug, Clone)]
1076pub struct ToolFunction {
1077 /// The name of the function to be called. Must be a-z, A-Z, 0-9, or
1078 /// contain underscores and dashes, with a maximum length
1079 /// of 64.
1080 pub name: String,
1081 /// A description of what the function does, used by the model to choose when and
1082 /// how to call the function.
1083 #[serde(skip_serializing_if = "Option::is_none")]
1084 pub description: Option<String>,
1085 /// The parameters the functions accepts, described as a JSON Schema object.
1086 ///
1087 /// See the
1088 /// [openai function calling guide](https://platform.openai.com/docs/guides/function-calling)
1089 /// for examples, and the
1090 /// [JSON Schema reference](https://json-schema.org/understanding-json-schema/) for
1091 /// documentation about the format.
1092 ///
1093 /// Omitting `parameters` defines a function with an empty parameter list.
1094 #[serde(skip_serializing_if = "Option::is_none")]
1095 pub parameters: Option<serde_json::Map<String, serde_json::Value>>,
1096 /// Whether to enable strict schema adherence when generating the function call.
1097 ///
1098 /// If set to true, the model will follow the exact schema defined in the
1099 /// `parameters` field. Only a subset of JSON Schema is supported when `strict` is
1100 /// `true`. Learn more about Structured Outputs in the
1101 /// [openai function calling guide](https://platform.openai.com/docs/guides/function-calling).
1102 #[serde(skip_serializing_if = "Option::is_none")]
1103 pub strict: Option<bool>,
1104}
1105
1106#[derive(Serialize, Deserialize, Debug, Clone)]
1107pub struct ToolCustom {
1108 /// The name of the custom tool, used to identify it in tool calls.
1109 pub name: String,
1110 /// Optional description of the custom tool, used to provide more context.
1111 #[serde(skip_serializing_if = "Option::is_none")]
1112 pub description: Option<String>,
1113 /// The input format for the custom tool. Default is unconstrained text.
1114 #[serde(skip_serializing_if = "Option::is_none")]
1115 pub format: Option<ToolCustomFormat>,
1116}
1117
1118#[derive(Serialize, Deserialize, Debug, Clone)]
1119#[serde(rename_all = "snake_case", tag = "type")]
1120pub enum ToolCustomFormat {
1121 /// Unconstrained text format. Always `text`.
1122 Text,
1123 /// Grammar format. Always `grammar`.
1124 Grammar {
1125 /// Your chosen grammar.
1126 grammar: ToolCustomFormatGrammarGrammar,
1127 },
1128}
1129
1130#[derive(Debug, Serialize, Deserialize, Clone)]
1131pub struct ToolCustomFormatGrammarGrammar {
1132 /// The grammar definition.
1133 pub definition: String,
1134 /// The syntax of the grammar definition. One of `lark` or `regex`.
1135 pub syntax: ToolCustomFormatGrammarGrammarSyntax,
1136}
1137
1138#[derive(Debug, Serialize, Deserialize, Clone)]
1139#[serde(rename_all = "snake_case")]
1140pub enum ToolCustomFormatGrammarGrammarSyntax {
1141 Lark,
1142 Regex,
1143}
1144
1145#[derive(Debug, Serialize, Deserialize, Clone)]
1146#[serde(rename_all = "snake_case")]
1147pub enum ToolChoice {
1148 None,
1149 Auto,
1150 Required,
1151 #[serde(untagged)]
1152 Specific(ToolChoiceSpecific),
1153}
1154
1155#[derive(Debug, Serialize, Deserialize, Clone)]
1156#[serde(rename_all = "snake_case", tag = "type")]
1157pub enum ToolChoiceSpecific {
1158 /// Allowed tool configuration type. Always `allowed_tools`.
1159 AllowedTools {
1160 /// Constrains the tools available to the model to a pre-defined set.
1161 allowed_tools: ToolChoiceAllowedTools,
1162 },
1163 /// For function calling, the type is always `function`.
1164 Function { function: ToolChoiceFunction },
1165 /// For custom tool calling, the type is always `custom`.
1166 Custom { custom: ToolChoiceCustom },
1167}
1168
1169#[derive(Debug, Serialize, Deserialize, Clone)]
1170pub struct ToolChoiceAllowedTools {
1171 /// Constrains the tools available to the model to a pre-defined set.
1172 ///
1173 /// - `auto` allows the model to pick from among the allowed tools and generate a
1174 /// message.
1175 /// - `required` requires the model to call one or more of the allowed tools.
1176 pub mode: ToolChoiceAllowedToolsMode,
1177 /// A list of tool definitions that the model should be allowed to call.
1178 ///
1179 /// For the Chat Completions API, the list of tool definitions might look like:
1180 ///
1181 /// ```json
1182 /// [
1183 /// { "type": "function", "function": { "name": "get_weather" } },
1184 /// { "type": "function", "function": { "name": "get_time" } }
1185 /// ]
1186 /// ```
1187 pub tools: Vec<serde_json::Map<String, serde_json::Value>>,
1188}
1189
1190/// The mode for allowed tools in tool choice.
1191///
1192/// Controls how the model should handle the set of allowed tools:
1193///
1194/// - `auto` allows the model to pick from among the allowed tools and generate a
1195/// message.
1196/// - `required` requires the model to call one or more of the allowed tools.
1197#[derive(Debug, Serialize, Deserialize, Clone)]
1198#[serde(rename_all = "lowercase")]
1199pub enum ToolChoiceAllowedToolsMode {
1200 /// The model can choose whether to use the allowed tools or not.
1201 Auto,
1202 /// The model must use at least one of the allowed tools.
1203 Required,
1204}
1205
1206#[derive(Debug, Serialize, Deserialize, Clone)]
1207pub struct ToolChoiceFunction {
1208 /// The name of the function to call.
1209 pub name: String,
1210}
1211
1212#[derive(Debug, Serialize, Deserialize, Clone)]
1213pub struct ToolChoiceCustom {
1214 /// The name of the custom tool to call.
1215 pub name: String,
1216}
1217
1218/// Controls the switch between thinking and non-thinking mode.
1219///
1220/// Shared by DeepSeek and Z.ai / 智谱 GLM, which use the same `thinking`
1221/// key and the same `type` values; GLM additionally supports the
1222/// `clear_thinking` flag. Gated on `any(deepseek, zai)` so enabling either
1223/// provider feature makes the field available, and enabling both never
1224/// emits the `thinking` key twice.
1225#[cfg(any(feature = "deepseek", feature = "zai"))]
1226#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq, Default)]
1227pub struct Thinking {
1228 /// Whether to use thinking mode (`enabled`) or non-thinking mode
1229 /// (`disabled`). Defaults to `enabled`.
1230 #[serde(rename = "type")]
1231 pub type_: ThinkingType,
1232 /// Z.ai / GLM: whether to clear the previous turn's `reasoning_content`
1233 /// from the context (`true`, GLM's default) or preserve it (`false`).
1234 #[cfg(feature = "zai")]
1235 #[serde(skip_serializing_if = "Option::is_none")]
1236 pub clear_thinking: Option<bool>,
1237}
1238
1239/// Whether thinking mode is enabled or disabled.
1240#[cfg(any(feature = "deepseek", feature = "zai"))]
1241#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq, Default)]
1242#[serde(rename_all = "lowercase")]
1243pub enum ThinkingType {
1244 /// Thinking mode is on. The default for both DeepSeek and GLM.
1245 #[default]
1246 Enabled,
1247 /// Thinking mode is off.
1248 Disabled,
1249}
1250
1251/// Constrains the effort on reasoning for reasoning models. This is an
1252/// official OpenAI parameter; reasoning providers such as DeepSeek and Qwen
1253/// accept a subset of these values and map the rest to their nearest effort
1254/// level.
1255#[derive(Debug, Serialize, Deserialize, Clone, PartialEq, Eq)]
1256#[serde(rename_all = "lowercase")]
1257pub enum ReasoningEffort {
1258 None,
1259 Minimal,
1260 Low,
1261 Medium,
1262 High,
1263 Xhigh,
1264 Max,
1265}
1266
1267impl RequestBody {
1268 /// Whether this request asks for a streamed response. Defaults to
1269 /// `false` when [`RequestBody::stream`] is `None`.
1270 pub fn is_streaming(&self) -> bool {
1271 self.stream.unwrap_or(false)
1272 }
1273}
1274
1275impl Post for RequestBody {
1276 fn is_streaming(&self) -> bool {
1277 RequestBody::is_streaming(self)
1278 }
1279
1280 /// Builds the URL for the request.
1281 ///
1282 /// `base_url` should be like <https://api.openai.com/v1>
1283 fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
1284 let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
1285 url.path_segments_mut()
1286 .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
1287 .push("chat")
1288 .push("completions");
1289
1290 Ok(url.to_string())
1291 }
1292}
1293
1294impl PostNoStream for RequestBody {
1295 type Response = super::response::no_streaming::ChatCompletion;
1296}
1297
1298impl PostStream for RequestBody {
1299 type Response = super::response::streaming::ChatCompletionChunk;
1300}
1301
1302#[cfg(test)]
1303mod request_test {
1304 use futures_util::StreamExt;
1305
1306 use super::*;
1307
1308 const DEEPSEEK_CHAT_URL: &str = "https://api.deepseek.com";
1309 const DEEPSEEK_MODEL: &str = "deepseek-v4-flash";
1310
1311 fn deepseek_api_key() -> Option<String> {
1312 std::env::var("DEEPSEEK_API_KEY")
1313 .ok()
1314 .map(|key| key.trim().to_string())
1315 .filter(|key| !key.is_empty())
1316 }
1317
1318 #[tokio::test]
1319 async fn test_deepseek_no_stream() {
1320 let Some(api_key) = deepseek_api_key() else {
1321 println!("Skipping: set DEEPSEEK_API_KEY to run this test");
1322 return;
1323 };
1324
1325 let request = RequestBody {
1326 messages: vec![
1327 Message::system("This is a request of test purpose. Reply briefly"),
1328 Message::user("What's your name?"),
1329 ],
1330 model: DEEPSEEK_MODEL.to_string(),
1331 stream: Some(false),
1332 ..Default::default()
1333 };
1334
1335 let response = request
1336 .get_response_string(
1337 &crate::rest::default_client(),
1338 DEEPSEEK_CHAT_URL,
1339 &crate::rest::RequestOptions::bearer(&api_key),
1340 )
1341 .await
1342 .unwrap();
1343
1344 println!("{}", response);
1345
1346 assert!(response.to_ascii_lowercase().contains("deepseek"));
1347 }
1348
1349 #[tokio::test]
1350 async fn test_deepseek_stream() {
1351 let Some(api_key) = deepseek_api_key() else {
1352 println!("Skipping: set DEEPSEEK_API_KEY to run this test");
1353 return;
1354 };
1355
1356 let request = RequestBody {
1357 messages: vec![
1358 Message::system("This is a request of test purpose. Reply briefly"),
1359 Message::user("Who are you?"),
1360 ],
1361 model: DEEPSEEK_MODEL.to_string(),
1362 stream: Some(true),
1363 ..Default::default()
1364 };
1365
1366 let mut response = request
1367 .get_stream_response_string(
1368 &crate::rest::default_client(),
1369 DEEPSEEK_CHAT_URL,
1370 &crate::rest::RequestOptions::bearer(&api_key),
1371 )
1372 .await
1373 .unwrap();
1374
1375 while let Some(chunk) = response.next().await {
1376 println!("{}", chunk.unwrap());
1377 }
1378 }
1379
1380 /// Assistant tool calls serialize with the official `type` tag
1381 /// (`{"type":"function",...}` / `{"type":"custom",...}`), not `role`.
1382 #[test]
1383 fn assistant_tool_call_serialization() {
1384 let function_call = AssistantToolCall::Function {
1385 id: "call_abc".to_string(),
1386 function: ToolCallFunction {
1387 arguments: "{\"city\":\"paris\"}".to_string(),
1388 name: "get_weather".to_string(),
1389 },
1390 };
1391 let json = serde_json::to_string(&function_call).unwrap();
1392 assert!(json.contains(r#""type":"function""#), "json: {json}");
1393 assert!(!json.contains(r#""role""#), "json: {json}");
1394
1395 let custom_call = AssistantToolCall::Custom {
1396 id: "call_def".to_string(),
1397 custom: ToolCallCustom {
1398 input: "2+2".to_string(),
1399 name: "calculator".to_string(),
1400 },
1401 };
1402 let json = serde_json::to_string(&custom_call).unwrap();
1403 assert!(json.contains(r#""type":"custom""#), "json: {json}");
1404 assert!(!json.contains(r#""role""#), "json: {json}");
1405 }
1406
1407 /// The `prediction` parameter sends its discriminator as `type`, not
1408 /// as the Rust field name `type_`.
1409 #[test]
1410 fn prediction_type_serialization() {
1411 let prediction = ChatCompletionPredictionContentParam {
1412 content: ChatCompletionPredictionContentParamContent::Text(
1413 "The capital of France is Paris.".to_string(),
1414 ),
1415 type_: ChatCompletionPredictionContentParamType::Content,
1416 };
1417 let json = serde_json::to_string(&prediction).unwrap();
1418 assert!(json.contains(r#""type":"content""#), "json: {json}");
1419 assert!(!json.contains("type_"), "json: {json}");
1420 }
1421
1422 /// `tool_choice: allowed_tools` sends `tools` as a JSON array of tool
1423 /// definitions, matching the official `Iterable[Dict[str, object]]`.
1424 #[test]
1425 fn allowed_tools_choice_serialization() {
1426 let mut weather = serde_json::Map::new();
1427 weather.insert("type".to_string(), serde_json::json!("function"));
1428 weather.insert(
1429 "function".to_string(),
1430 serde_json::json!({ "name": "get_weather" }),
1431 );
1432
1433 let choice = ToolChoiceSpecific::AllowedTools {
1434 allowed_tools: ToolChoiceAllowedTools {
1435 mode: ToolChoiceAllowedToolsMode::Required,
1436 tools: vec![weather],
1437 },
1438 };
1439 let json = serde_json::to_string(&choice).unwrap();
1440 assert!(json.contains(r#""type":"allowed_tools""#), "json: {json}");
1441 assert!(json.contains(r#""mode":"required""#), "json: {json}");
1442 // `tools` must serialize as an array, not an object.
1443 assert!(json.contains(r#""tools":[{"#), "json: {json}");
1444 }
1445
1446 /// `web_search_options` sends `search_context_size` as optional and the
1447 /// user location nested under an `approximate` key.
1448 #[test]
1449 fn web_search_options_serialization() {
1450 let options = WebSearchOptions {
1451 search_context_size: None,
1452 user_location: Some(WebSearchOptionsUserLocation::Approximate {
1453 approximate: WebSearchOptionsUserLocationApproximate {
1454 city: Some("San Francisco".to_string()),
1455 country: None,
1456 region: None,
1457 timezone: None,
1458 },
1459 }),
1460 };
1461 let json = serde_json::to_string(&options).unwrap();
1462 assert!(!json.contains("search_context_size"), "json: {json}");
1463 assert!(json.contains(r#""type":"approximate""#), "json: {json}");
1464 assert!(
1465 json.contains(r#""approximate":{"city":"San Francisco"}"#),
1466 "json: {json}"
1467 );
1468 }
1469
1470 /// `JSONSchema`/`ToolFunction` optional fields are omitted when unset.
1471 #[test]
1472 fn json_schema_optional_fields_serialization() {
1473 let schema = JSONSchema {
1474 name: "Answer".to_string(),
1475 description: None,
1476 schema: None,
1477 strict: None,
1478 };
1479 let json = serde_json::to_string(&schema).unwrap();
1480 assert_eq!(json, r#"{"name":"Answer"}"#);
1481
1482 let function = ToolFunction {
1483 name: "get_weather".to_string(),
1484 description: None,
1485 parameters: None,
1486 strict: None,
1487 };
1488 let json = serde_json::to_string(&function).unwrap();
1489 assert_eq!(json, r#"{"name":"get_weather"}"#);
1490 }
1491
1492 /// Plain-text user messages keep the official wire format: `content`
1493 /// is a JSON string, not a parts array.
1494 #[test]
1495 fn user_text_content_serialization() {
1496 let request = RequestBody {
1497 messages: vec![Message::user("Hi")],
1498 model: "gpt-4o".to_string(),
1499 ..Default::default()
1500 };
1501
1502 let json = serde_json::to_string(&request).unwrap();
1503 assert!(json.contains(r#""content":"Hi""#), "json: {json}");
1504 }
1505
1506 /// Messages deserialize from client JSON through the payload structs
1507 /// (request-side parity for proxies and servers).
1508 #[test]
1509 fn message_deserialization() {
1510 let system: Message =
1511 serde_json::from_str(r#"{"role":"system","content":"Be terse"}"#).unwrap();
1512 assert!(matches!(
1513 system,
1514 Message::System(SystemMessage {
1515 content: MessageContent::Text(_),
1516 name: None
1517 })
1518 ));
1519
1520 let user: Message =
1521 serde_json::from_str(r#"{"role":"user","content":"Hi","name":"jimmy"}"#).unwrap();
1522 let Message::User(user) = user else {
1523 panic!("must be a user message");
1524 };
1525 assert_eq!(user.name.as_deref(), Some("jimmy"));
1526
1527 let assistant: Message = serde_json::from_str(
1528 r#"{"role":"assistant","content":null,"tool_calls":[{"type":"function","id":"call_1","function":{"name":"f","arguments":"{}"}}]}"#,
1529 )
1530 .unwrap();
1531 let Message::Assistant(assistant) = assistant else {
1532 panic!("must be an assistant message");
1533 };
1534 assert_eq!(assistant.content, None);
1535 assert_eq!(assistant.tool_calls.expect("tool calls").len(), 1);
1536
1537 let tool: Message =
1538 serde_json::from_str(r#"{"role":"tool","content":"42","tool_call_id":"call_1"}"#)
1539 .unwrap();
1540 let Message::Tool(tool) = tool else {
1541 panic!("must be a tool message");
1542 };
1543 assert_eq!(tool.tool_call_id, "call_1");
1544
1545 let developer: Message =
1546 serde_json::from_str(r#"{"role":"developer","content":"New rules"}"#).unwrap();
1547 assert!(matches!(developer, Message::Developer(_)));
1548
1549 let function: Message =
1550 serde_json::from_str(r#"{"role":"function","name":"f","content":"ok"}"#).unwrap();
1551 assert!(matches!(function, Message::Function(_)));
1552 }
1553
1554 /// An unknown role is a hard error: such a message cannot be forwarded
1555 /// to any backend, so it must not be silently mapped onto a catch-all.
1556 #[test]
1557 fn unknown_role_fails_deserialization() {
1558 let result = serde_json::from_str::<Message>(r#"{"role":"weird","content":"x"}"#);
1559 assert!(result.is_err(), "unknown roles must be rejected");
1560 }
1561
1562 /// A full request body deserializes back from client JSON; unknown
1563 /// top-level fields are captured into `extra_body_map` and survive
1564 /// re-serialization, so proxying is lossless.
1565 #[test]
1566 fn request_body_deserializes_with_extra_fields() {
1567 let json = r#"{
1568 "model": "qwen-plus",
1569 "messages": [{"role": "user", "content": "Hi"}],
1570 "stream": true,
1571 "vendor_extension": {"depth": 3}
1572 }"#;
1573 let request = serde_json::from_str::<RequestBody>(json).unwrap();
1574 assert_eq!(request.model, "qwen-plus");
1575 assert_eq!(request.stream, Some(true));
1576 assert_eq!(request.messages.len(), 1);
1577
1578 let extra = request
1579 .extra_body_map
1580 .as_ref()
1581 .expect("extra fields captured");
1582 assert_eq!(
1583 extra.get("vendor_extension"),
1584 Some(&serde_json::json!({"depth": 3}))
1585 );
1586
1587 let serialized = serde_json::to_value(&request).unwrap();
1588 assert_eq!(serialized["vendor_extension"]["depth"], 3);
1589 }
1590
1591 /// The convenience constructors produce the official wire shapes.
1592 #[test]
1593 fn message_constructors() {
1594 let request = RequestBody {
1595 messages: vec![
1596 Message::system("Be terse"),
1597 Message::user("Hi"),
1598 Message::assistant("Hello!"),
1599 Message::tool(r#"{"temp":21}"#, "call_1"),
1600 ],
1601 model: "gpt-4o".to_string(),
1602 ..Default::default()
1603 };
1604
1605 let json = serde_json::to_string(&request).unwrap();
1606 assert!(json.contains(r#""role":"system","content":"Be terse""#),);
1607 assert!(json.contains(r#""role":"user","content":"Hi""#));
1608 assert!(json.contains(r#""role":"assistant","content":"Hello!""#));
1609 assert!(
1610 json.contains(r#""role":"tool","content":"{\"temp\":21}","tool_call_id":"call_1""#)
1611 );
1612 }
1613
1614 /// System, developer and tool messages serialize `content` as a plain
1615 /// string by default and as a text-part array when parts are supplied
1616 /// (the official "string or array of content parts" shapes).
1617 #[test]
1618 fn system_developer_tool_content_serialization() {
1619 let request = RequestBody {
1620 messages: vec![
1621 Message::system("Be terse"),
1622 Message::developer(MessageContent::Parts(vec![ContentPart::Text {
1623 text: "Prefer Rust".to_string(),
1624 prompt_cache_breakpoint: None,
1625 }])),
1626 Message::tool(
1627 MessageContent::Parts(vec![ContentPart::Text {
1628 text: r#"{"temp": 21}"#.to_string(),
1629 prompt_cache_breakpoint: None,
1630 }]),
1631 "call_1",
1632 ),
1633 ],
1634 model: "gpt-4o".to_string(),
1635 ..Default::default()
1636 };
1637
1638 let json = serde_json::to_string(&request).unwrap();
1639 assert!(
1640 json.contains(r#""role":"system","content":"Be terse""#),
1641 "json: {json}"
1642 );
1643 assert!(
1644 json.contains(r#""role":"developer","content":[{"type":"text","text":"Prefer Rust"}]"#),
1645 "json: {json}"
1646 );
1647 assert!(
1648 json.contains(
1649 r#""role":"tool","content":[{"type":"text","text":"{\"temp\": 21}"}],"tool_call_id":"call_1""#
1650 ),
1651 "json: {json}"
1652 );
1653 }
1654
1655 /// Multimodal user messages serialize as content-part arrays with the
1656 /// official shapes, including `prompt_cache_breakpoint`.
1657 #[test]
1658 fn multimodal_content_serialization() {
1659 let request = RequestBody {
1660 messages: vec![Message::user(MessageContent::Parts(vec![
1661 ContentPart::ImageUrl {
1662 image_url: ContentPartImageUrl {
1663 url: "https://example.com/cat.png".to_string(),
1664 detail: Some(ImageDetail::High),
1665 },
1666 prompt_cache_breakpoint: None,
1667 },
1668 ContentPart::Text {
1669 text: "What's in this image?".to_string(),
1670 prompt_cache_breakpoint: Some(PromptCacheBreakpoint {
1671 mode: PromptCacheBreakpointMode::Explicit,
1672 }),
1673 },
1674 ]))],
1675 model: "gpt-4o".to_string(),
1676 ..Default::default()
1677 };
1678
1679 let json = serde_json::to_string(&request).unwrap();
1680 assert!(json.contains(r#""type":"image_url""#), "json: {json}");
1681 assert!(
1682 json.contains(r#""url":"https://example.com/cat.png""#),
1683 "json: {json}"
1684 );
1685 assert!(json.contains(r#""detail":"high""#), "json: {json}");
1686 assert!(json.contains(r#""type":"text""#), "json: {json}");
1687 assert!(
1688 json.contains(r#""prompt_cache_breakpoint":{"mode":"explicit"}"#),
1689 "json: {json}"
1690 );
1691 }
1692
1693 /// `input_audio` and `file` content parts serialize with the official
1694 /// shapes.
1695 #[test]
1696 fn audio_and_file_content_serialization() {
1697 let content = MessageContent::Parts(vec![
1698 ContentPart::InputAudio {
1699 input_audio: ContentPartInputAudio {
1700 data: "aGVsbG8=".to_string(),
1701 format: InputAudioFormat::Wav,
1702 },
1703 prompt_cache_breakpoint: None,
1704 },
1705 ContentPart::File {
1706 file: ContentPartFile {
1707 file_id: Some("file-abc".to_string()),
1708 ..Default::default()
1709 },
1710 prompt_cache_breakpoint: None,
1711 },
1712 ]);
1713
1714 let json = serde_json::to_string(&content).unwrap();
1715 assert!(json.contains(r#""type":"input_audio""#), "json: {json}");
1716 assert!(json.contains(r#""data":"aGVsbG8=""#), "json: {json}");
1717 assert!(json.contains(r#""format":"wav""#), "json: {json}");
1718 assert!(json.contains(r#""type":"file""#), "json: {json}");
1719 assert!(
1720 json.contains(r#""file":{"file_id":"file-abc"}"#),
1721 "json: {json}"
1722 );
1723 // Optional file fields are omitted when unset.
1724 assert!(!json.contains("file_data"), "json: {json}");
1725 }
1726
1727 /// `logit_bias`, `moderation` and `prompt_cache_options` serialize as
1728 /// the official request parameters (token-id keys as JSON strings).
1729 #[test]
1730 fn new_params_serialization() {
1731 let mut logit_bias = HashMap::new();
1732 logit_bias.insert(40u32, -100i32);
1733
1734 let request = RequestBody {
1735 messages: vec![Message::user("Hi")],
1736 model: "gpt-5".to_string(),
1737 logit_bias: Some(logit_bias),
1738 moderation: Some(ChatModerationParam {
1739 model: "omni-moderation-latest".to_string(),
1740 policy: Some(ModerationPolicyParam {
1741 input: Some(ModerationPolicySideParam {
1742 mode: ModerationPolicyMode::Block,
1743 }),
1744 output: None,
1745 }),
1746 }),
1747 prompt_cache_options: Some(PromptCacheOptions {
1748 mode: Some(PromptCacheMode::Explicit),
1749 ttl: Some(PromptCacheTtl::ThirtyMinutes),
1750 }),
1751 ..Default::default()
1752 };
1753
1754 let json = serde_json::to_string(&request).unwrap();
1755 assert!(json.contains(r#""logit_bias":{"40":-100}"#), "json: {json}");
1756 assert!(
1757 json.contains(
1758 r#""moderation":{"model":"omni-moderation-latest","policy":{"input":{"mode":"block"}}}"#
1759 ),
1760 "json: {json}"
1761 );
1762 assert!(
1763 json.contains(r#""prompt_cache_options":{"mode":"explicit","ttl":"30m"}"#),
1764 "json: {json}"
1765 );
1766 }
1767
1768 /// Serializes the OpenAI `reasoning_effort` parameter.
1769 #[test]
1770 fn reasoning_effort_serialization() {
1771 let request = RequestBody {
1772 messages: vec![Message::user("What's your name?")],
1773 model: "gpt-5".to_string(),
1774 reasoning_effort: Some(ReasoningEffort::Xhigh),
1775 ..Default::default()
1776 };
1777
1778 let json = serde_json::to_string(&request).unwrap();
1779 assert!(
1780 json.contains(r#""reasoning_effort":"xhigh""#),
1781 "json: {json}"
1782 );
1783 }
1784
1785 /// Serializes the DeepSeek Beta chat prefix completion fields.
1786 #[cfg(feature = "deepseek")]
1787 #[test]
1788 fn deepseek_assistant_prefix_serialization() {
1789 let request = RequestBody {
1790 messages: vec![
1791 Message::user("Please write quick sort code"),
1792 Message::Assistant(AssistantMessage {
1793 content: Some("```python\n".to_string()),
1794 prefix: true,
1795 ..Default::default()
1796 }),
1797 ],
1798 model: DEEPSEEK_MODEL.to_string(),
1799 ..Default::default()
1800 };
1801
1802 let json = serde_json::to_string(&request).unwrap();
1803 assert!(json.contains(r#""prefix":true"#), "json: {json}");
1804 }
1805
1806 /// Serializes the DeepSeek `thinking`, `reasoning_effort` and `user_id`
1807 /// request parameters.
1808 #[cfg(feature = "deepseek")]
1809 #[test]
1810 fn deepseek_thinking_params_serialization() {
1811 let request = RequestBody {
1812 messages: vec![Message::user("What's your name?")],
1813 model: DEEPSEEK_MODEL.to_string(),
1814 thinking: Some(Thinking {
1815 type_: ThinkingType::Disabled,
1816 ..Default::default()
1817 }),
1818 user_id: Some("user-123".to_string()),
1819 ..Default::default()
1820 };
1821
1822 let json = serde_json::to_string(&request).unwrap();
1823 assert!(
1824 json.contains(r#""thinking":{"type":"disabled"}"#),
1825 "json: {json}"
1826 );
1827 assert!(json.contains(r#""user_id":"user-123""#), "json: {json}");
1828 }
1829
1830 /// Serializes the Qwen `enable_thinking`, `thinking_budget` and `top_k`
1831 /// request parameters.
1832 #[cfg(feature = "qwen")]
1833 #[test]
1834 fn qwen_params_serialization() {
1835 let request = RequestBody {
1836 messages: vec![Message::user("What's your name?")],
1837 model: "qwen-plus".to_string(),
1838 enable_thinking: Some(false),
1839 thinking_budget: Some(1024),
1840 top_k: Some(20),
1841 ..Default::default()
1842 };
1843
1844 let json = serde_json::to_string(&request).unwrap();
1845 assert!(json.contains(r#""enable_thinking":false"#), "json: {json}");
1846 assert!(json.contains(r#""thinking_budget":1024"#), "json: {json}");
1847 assert!(json.contains(r#""top_k":20"#), "json: {json}");
1848 }
1849
1850 const QWEN_CHAT_URL: &str = "https://dashscope.aliyuncs.com/compatible-mode/v1";
1851 /// Qwen's multimodal flash model: accepts text, image and audio inputs
1852 /// through its OpenAI-compatible endpoint.
1853 const QWEN_MULTIMODAL_MODEL: &str = "qwen3.8-flash";
1854
1855 fn qwen_api_key() -> Option<String> {
1856 std::env::var("QWEN_API_KEY")
1857 .ok()
1858 .map(|key| key.trim().to_string())
1859 .filter(|key| !key.is_empty())
1860 }
1861
1862 /// Real request: a user message with an `image_url` content part. The
1863 /// image is the football sample used in Alibaba Cloud Model Studio's own
1864 /// documentation. Requires `QWEN_API_KEY`; skipped otherwise.
1865 #[tokio::test]
1866 async fn test_qwen_image_input() -> Result<(), anyhow::Error> {
1867 let Some(api_key) = qwen_api_key() else {
1868 println!("Skipping: set QWEN_API_KEY to run this test");
1869 return Ok(());
1870 };
1871
1872 let request = RequestBody {
1873 messages: vec![
1874 Message::system("This is a request of test purpose. Reply briefly"),
1875 Message::user(MessageContent::Parts(vec![
1876 ContentPart::ImageUrl {
1877 image_url: ContentPartImageUrl {
1878 url: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20241108/xzsgiz/football1.jpg"
1879 .to_string(),
1880 detail: None,
1881 },
1882 prompt_cache_breakpoint: None,
1883 },
1884 ContentPart::Text {
1885 text: "What is shown in this image? Answer with one short sentence."
1886 .to_string(),
1887 prompt_cache_breakpoint: None,
1888 },
1889 ])),
1890 ],
1891 model: QWEN_MULTIMODAL_MODEL.to_string(),
1892 ..Default::default()
1893 };
1894
1895 let response = request
1896 .get_response(
1897 &crate::rest::default_client(),
1898 QWEN_CHAT_URL,
1899 &crate::rest::RequestOptions::bearer(&api_key),
1900 )
1901 .await?;
1902
1903 let content = response.choices[0]
1904 .message
1905 .content
1906 .clone()
1907 .unwrap_or_default();
1908 println!("image response: {content}");
1909 assert!(
1910 !content.trim().is_empty(),
1911 "empty content for a valid image request"
1912 );
1913 Ok(())
1914 }
1915
1916 /// Real request: a user message with an `input_audio` content part
1917 /// carrying a public audio URL (the cherry sample from the Model Studio
1918 /// docs), answered by the streaming response. Requires `QWEN_API_KEY`;
1919 /// skipped otherwise.
1920 ///
1921 /// Uses `qwen-omni-turbo`: Qwen's Omni models are the multimodal class
1922 /// that accepts audio input on the OpenAI-compatible endpoint, and they
1923 /// require `stream: true`. (`qwen3.8-flash` rejects `input_audio` with a
1924 /// provider-side `400 incorrect modal 'audio'` error, verified with
1925 /// plain curl.)
1926 #[tokio::test]
1927 async fn test_qwen_audio_input() -> Result<(), anyhow::Error> {
1928 let Some(api_key) = qwen_api_key() else {
1929 println!("Skipping: set QWEN_API_KEY to run this test");
1930 return Ok(());
1931 };
1932
1933 let request = RequestBody {
1934 messages: vec![Message::user(MessageContent::Parts(vec![
1935 ContentPart::InputAudio {
1936 input_audio: ContentPartInputAudio {
1937 data: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20250211/tixcef/cherry.wav"
1938 .to_string(),
1939 format: InputAudioFormat::Wav,
1940 },
1941 prompt_cache_breakpoint: None,
1942 },
1943 ContentPart::Text {
1944 text: "What does the speaker say in this audio? Reply briefly."
1945 .to_string(),
1946 prompt_cache_breakpoint: None,
1947 },
1948 ]))],
1949 model: "qwen-omni-turbo".to_string(),
1950 stream: Some(true),
1951 modalities: Some(vec![Modality::Text]),
1952 ..Default::default()
1953 };
1954
1955 let mut stream = request
1956 .get_stream_response(
1957 &crate::rest::default_client(),
1958 QWEN_CHAT_URL,
1959 &crate::rest::RequestOptions::bearer(&api_key),
1960 )
1961 .await?;
1962
1963 let mut message = String::new();
1964 while let Some(chunk) = stream.next().await {
1965 let chunk = chunk?;
1966 if let Some(choice) = chunk.choices.first()
1967 && let Some(content) = choice.delta.content.as_deref()
1968 {
1969 message.push_str(content);
1970 }
1971 }
1972
1973 println!("audio response: {message}");
1974 assert!(
1975 !message.trim().is_empty(),
1976 "empty content for a valid audio request"
1977 );
1978 Ok(())
1979 }
1980
1981 /// Real request: a plain-text user message (the wire format of
1982 /// [`MessageContent::Text`]). Requires `QWEN_API_KEY`; skipped otherwise.
1983 #[tokio::test]
1984 async fn test_qwen_text_input() -> Result<(), anyhow::Error> {
1985 let Some(api_key) = qwen_api_key() else {
1986 println!("Skipping: set QWEN_API_KEY to run this test");
1987 return Ok(());
1988 };
1989
1990 let request = RequestBody {
1991 messages: vec![Message::user("Reply with exactly one word.")],
1992 model: QWEN_MULTIMODAL_MODEL.to_string(),
1993 ..Default::default()
1994 };
1995
1996 let response = request
1997 .get_response(
1998 &crate::rest::default_client(),
1999 QWEN_CHAT_URL,
2000 &crate::rest::RequestOptions::bearer(&api_key),
2001 )
2002 .await?;
2003
2004 let content = response.choices[0]
2005 .message
2006 .content
2007 .clone()
2008 .unwrap_or_default();
2009 println!("text response: {content}");
2010 assert!(!content.trim().is_empty(), "empty content for text input");
2011 Ok(())
2012 }
2013
2014 /// Real request: streaming a multimodal (image + text) user message.
2015 /// Requires `QWEN_API_KEY`; skipped otherwise.
2016 #[tokio::test]
2017 async fn test_qwen_multimodal_stream() -> Result<(), anyhow::Error> {
2018 let Some(api_key) = qwen_api_key() else {
2019 println!("Skipping: set QWEN_API_KEY to run this test");
2020 return Ok(());
2021 };
2022
2023 let request = RequestBody {
2024 messages: vec![Message::user(MessageContent::Parts(vec![
2025 ContentPart::ImageUrl {
2026 image_url: ContentPartImageUrl {
2027 url: "https://help-static-aliyun-doc.aliyuncs.com/file-manage-files/zh-CN/20241108/xzsgiz/football1.jpg"
2028 .to_string(),
2029 detail: None,
2030 },
2031 prompt_cache_breakpoint: None,
2032 },
2033 ContentPart::Text {
2034 text: "What is shown in this image? Answer with one short sentence."
2035 .to_string(),
2036 prompt_cache_breakpoint: None,
2037 },
2038 ]))],
2039 model: QWEN_MULTIMODAL_MODEL.to_string(),
2040 stream: Some(true),
2041 ..Default::default()
2042 };
2043
2044 let mut stream = request
2045 .get_stream_response(
2046 &crate::rest::default_client(),
2047 QWEN_CHAT_URL,
2048 &crate::rest::RequestOptions::bearer(&api_key),
2049 )
2050 .await?;
2051
2052 let mut message = String::new();
2053 while let Some(chunk) = stream.next().await {
2054 let chunk = chunk?;
2055 if let Some(choice) = chunk.choices.first()
2056 && let Some(content) = choice.delta.content.as_deref()
2057 {
2058 message.push_str(content);
2059 }
2060 }
2061
2062 println!("streamed message: {message}");
2063 assert!(
2064 !message.trim().is_empty(),
2065 "empty streamed content for a valid image request"
2066 );
2067 Ok(())
2068 }
2069}