openai_interface/zai.rs
1//! Z.ai / 智谱 GLM (BigModel) proprietary extensions to the OpenAI-compatible
2//! API.
3//!
4//! Everything in this module is gated on the `zai` cargo feature.
5//!
6//! GLM is served behind an OpenAI-compatible endpoint at
7//! `https://open.bigmodel.cn/api/paas/v4` (the international brand is Z.ai,
8//! `https://api.z.ai/api/paas/v4`), authenticated with the usual
9//! `Authorization: Bearer <key>` header. Most of the request body is standard
10//! OpenAI; this module collects the parts that are not.
11//!
12//! GLM's divergences fall into two groups, handled differently:
13//!
14//! - **Generic controls** that GLM merely spells its own way live directly on
15//! [`RequestBody`](crate::chat::create::request::RequestBody) as `zai`-gated
16//! fields — `do_sample` and `tool_stream`. The keys GLM shares with another
17//! provider are *unified* rather than duplicated: `thinking` and `user_id`
18//! are gated on `any(deepseek, zai)`, `request_id` on `any(vllm, zai)`, so
19//! enabling several provider features at once never emits a key twice.
20//! - **Platform-ecosystem extensions** — features tied to Zhipu's own platform
21//! rather than to text generation — are grouped here:
22//! - [`PlatformParams`], `#[serde(flatten)]`ed into the request body through
23//! `RequestBody::zai_platform` (currently the `watermark_enabled`
24//! compliance flag).
25//! - the [`retrieval`](RetrievalTool) and [`web_search`](WebSearchTool) tool
26//! types, added as `zai`-gated variants of
27//! [`RequestTool`](crate::chat::create::request::RequestTool);
28//! - the [`web_search`](WebSearchResult) results GLM returns at the top level
29//! of a chat completion.
30//!
31//! ```rust
32//! # #[cfg(feature = "zai")] {
33//! use openai_interface::chat::create::request::{Message, RequestBody};
34//! use openai_interface::zai::{PlatformParams, SearchEngine, WebSearchTool};
35//!
36//! let request = RequestBody {
37//! messages: vec![Message::user("最近有什么关于 Rust 的新闻?")],
38//! model: "glm-4.6".to_string(),
39//! do_sample: Some(false),
40//! zai_platform: Some(PlatformParams {
41//! watermark_enabled: Some(false),
42//! }),
43//! tools: Some(vec![
44//! openai_interface::chat::create::request::RequestTool::WebSearch {
45//! web_search: WebSearchTool {
46//! search_engine: Some(SearchEngine::SearchProJina),
47//! enable: Some(true),
48//! ..Default::default()
49//! },
50//! },
51//! ]),
52//! ..Default::default()
53//! };
54//!
55//! let json = serde_json::to_value(&request).unwrap();
56//! assert_eq!(json["do_sample"], serde_json::json!(false));
57//! assert_eq!(json["watermark_enabled"], serde_json::json!(false));
58//! assert_eq!(json["tools"][0]["type"], serde_json::json!("web_search"));
59//! # }
60//! ```
61//!
62//! # Fields that need no gate
63//!
64//! - `reasoning_effort` is already an ungated
65//! [`RequestBody`](crate::chat::create::request::RequestBody) field, and its
66//! [`ReasoningEffort`](crate::chat::create::request::ReasoningEffort) enum
67//! already covers every GLM value (`none`, `minimal`, `low`, `medium`,
68//! `high`, `xhigh`, `max`). GLM only honours it when `thinking` is enabled,
69//! and the value-to-depth mapping is model-specific.
70//! - `temperature`, `top_p` and `max_tokens` are standard keys; GLM simply
71//! accepts narrower ranges (`temperature` and `top_p` in `[0, 1]`, two
72//! decimals; `max_tokens` up to `131072`). No extra field models a range.
73//! - `reasoning_content` on the response message and streamed delta is covered
74//! by the always-available `reasoning` feature.
75//! - GLM's extra `finish_reason` values (`sensitive`,
76//! `model_context_window_exceeded`, `network_error`) are preserved by
77//! [`FinishReason::Unknown`](crate::chat::FinishReason) rather than modelled
78//! as variants, so an unexpected terminator never fails deserialization.
79//!
80//! # Not implemented
81//!
82//! GLM also exposes an `mcp` tool type, image/video generation, embeddings and
83//! a separate GLM Coding Plan endpoint. None are documented precisely enough
84//! to model here; this feature covers the fields GLM adds to the OpenAI chat
85//! completions endpoint.
86//!
87//! See [the GLM chat-completions
88//! reference](https://docs.bigmodel.cn/api-reference/模型-api/对话补全) and its
89//! [Z.ai equivalent](https://docs.z.ai/api-reference/llm/chat-completion).
90#![cfg(feature = "zai")]
91
92use serde::{Deserialize, Serialize};
93
94/// GLM platform-ecosystem request parameters, flattened into the request body.
95///
96/// These are Zhipu-platform features rather than text-generation controls, so
97/// they are grouped here instead of sitting loose on
98/// [`RequestBody`](crate::chat::create::request::RequestBody). Flattened
99/// through `RequestBody::zai_platform`; see the [module docs](crate::zai).
100#[derive(Serialize, Deserialize, Debug, Clone, Default)]
101pub struct PlatformParams {
102 /// Whether AI-generated output carries Zhipu's explicit and implicit
103 /// watermark. GLM's compliant default is `true`; `false` is only accepted
104 /// for accounts that have signed the corresponding waiver.
105 #[serde(skip_serializing_if = "Option::is_none")]
106 pub watermark_enabled: Option<bool>,
107}
108
109/// GLM's `retrieval` tool: grounds the answer in one of Zhipu's knowledge
110/// bases.
111///
112/// A `zai`-gated variant of
113/// [`RequestTool`](crate::chat::create::request::RequestTool); the wire shape
114/// is `{"type": "retrieval", "retrieval": { ... }}`.
115#[derive(Serialize, Deserialize, Debug, Clone)]
116pub struct RetrievalTool {
117 /// The identifier of the knowledge base to retrieve from. Required by GLM.
118 pub knowledge_id: String,
119 /// An optional template controlling how retrieved chunks are prompted.
120 #[serde(skip_serializing_if = "Option::is_none")]
121 pub prompt_template: Option<String>,
122}
123
124/// GLM's `web_search` tool: lets the model call Zhipu's web search.
125///
126/// A `zai`-gated variant of
127/// [`RequestTool`](crate::chat::create::request::RequestTool); the wire shape
128/// is `{"type": "web_search", "web_search": { ... }}`. Every field except
129/// `search_engine` is optional and omitted when unset, so GLM applies its own
130/// default.
131#[derive(Serialize, Deserialize, Debug, Clone, Default)]
132pub struct WebSearchTool {
133 /// The search backend. Required by GLM; currently only
134 /// [`SearchEngine::SearchProJina`] is accepted.
135 #[serde(skip_serializing_if = "Option::is_none")]
136 pub search_engine: Option<SearchEngine>,
137 /// Whether the tool is enabled. GLM's default is `false`.
138 #[serde(skip_serializing_if = "Option::is_none")]
139 pub enable: Option<bool>,
140 /// An explicit query to search for, instead of letting the model phrase it.
141 #[serde(skip_serializing_if = "Option::is_none")]
142 pub search_query: Option<String>,
143 /// Number of results to return, `1..=50`. GLM's default is `10`.
144 #[serde(skip_serializing_if = "Option::is_none")]
145 pub count: Option<u32>,
146 /// Restrict results to this domain.
147 #[serde(skip_serializing_if = "Option::is_none")]
148 pub search_domain_filter: Option<String>,
149 /// Only return results published within this window. GLM's default is
150 /// [`SearchRecencyFilter::NoLimit`].
151 #[serde(skip_serializing_if = "Option::is_none")]
152 pub search_recency_filter: Option<SearchRecencyFilter>,
153 /// How much of each page to keep. GLM's default is
154 /// [`ContentSize::Medium`].
155 #[serde(skip_serializing_if = "Option::is_none")]
156 pub content_size: Option<ContentSize>,
157 /// Whether search results are placed before or after the model's answer.
158 /// GLM's default is [`ResultSequence::After`].
159 #[serde(skip_serializing_if = "Option::is_none")]
160 pub result_sequence: Option<ResultSequence>,
161 /// Return the raw search results alongside the answer. GLM's default is
162 /// `false`.
163 #[serde(skip_serializing_if = "Option::is_none")]
164 pub search_result: Option<bool>,
165 /// Force a search even when the model thinks none is needed. GLM's default
166 /// is `false`.
167 #[serde(skip_serializing_if = "Option::is_none")]
168 pub require_search: Option<bool>,
169 /// Extra instructions steering what the model searches for.
170 #[serde(skip_serializing_if = "Option::is_none")]
171 pub search_prompt: Option<String>,
172}
173
174crate::wire_string_enum! {
175 /// The web-search backend GLM should use.
176 pub enum SearchEngine {
177 /// Zhipu's Jina-powered pro search. Currently the only accepted value.
178 SearchProJina => "search_pro_jina",
179 }
180}
181
182crate::wire_string_enum! {
183 /// How recent a web-search result may be.
184 pub enum SearchRecencyFilter {
185 /// Published within the last day.
186 OneDay => "oneDay",
187 /// Published within the last week.
188 OneWeek => "oneWeek",
189 /// Published within the last month.
190 OneMonth => "oneMonth",
191 /// Published within the last year.
192 OneYear => "oneYear",
193 /// No recency restriction (GLM's default).
194 NoLimit => "noLimit",
195 }
196}
197
198crate::wire_string_enum! {
199 /// How much of each searched page to keep.
200 pub enum ContentSize {
201 /// A medium-length extract (GLM's default).
202 Medium => "medium",
203 /// A longer extract.
204 High => "high",
205 }
206}
207
208crate::wire_string_enum! {
209 /// Where the search results are placed relative to the answer.
210 pub enum ResultSequence {
211 /// Results before the answer.
212 Before => "before",
213 /// Results after the answer (GLM's default).
214 After => "after",
215 }
216}
217
218/// One web-search result GLM returns in the top-level `web_search` array of a
219/// chat completion.
220///
221/// Not part of the OpenAI schema. Every field is optional: GLM populates what
222/// the search backend returned, and a given result may omit several of them.
223#[derive(Serialize, Deserialize, Debug, Clone, Default, PartialEq)]
224pub struct WebSearchResult {
225 /// The result's title.
226 #[serde(skip_serializing_if = "Option::is_none")]
227 pub title: Option<String>,
228 /// The snippet or extracted content of the result.
229 #[serde(skip_serializing_if = "Option::is_none")]
230 pub content: Option<String>,
231 /// The result's URL.
232 #[serde(skip_serializing_if = "Option::is_none")]
233 pub link: Option<String>,
234 /// A media URL associated with the result, when any.
235 #[serde(skip_serializing_if = "Option::is_none")]
236 pub media: Option<String>,
237 /// The source site's icon URL, when any.
238 #[serde(skip_serializing_if = "Option::is_none")]
239 pub icon: Option<String>,
240 /// The citation marker GLM associates with this result.
241 #[serde(skip_serializing_if = "Option::is_none")]
242 pub refer: Option<String>,
243 /// The publication date of the result, as a string.
244 #[serde(skip_serializing_if = "Option::is_none")]
245 pub publish_date: Option<String>,
246}
247
248#[cfg(test)]
249mod tests {
250 use super::*;
251 use crate::chat::create::request::{Message, RequestBody, RequestTool};
252
253 /// The platform parameters must land at the top level of the body, and
254 /// survive a round trip without leaking into the `extra_body_map`
255 /// catch-all (the crate is used to build proxies).
256 #[test]
257 fn platform_params_flatten_to_top_level() {
258 let request = RequestBody {
259 messages: vec![Message::user("Hello")],
260 model: "glm-4.6".to_string(),
261 do_sample: Some(false),
262 tool_stream: Some(true),
263 zai_platform: Some(PlatformParams {
264 watermark_enabled: Some(false),
265 }),
266 ..Default::default()
267 };
268
269 let json = serde_json::to_value(&request).unwrap();
270 assert_eq!(json["do_sample"], serde_json::json!(false));
271 assert_eq!(json["tool_stream"], serde_json::json!(true));
272 assert_eq!(json["watermark_enabled"], serde_json::json!(false));
273
274 let parsed: RequestBody = serde_json::from_str(
275 r#"{
276 "model": "glm-4.6",
277 "messages": [{"role": "user", "content": "Hello"}],
278 "do_sample": false,
279 "watermark_enabled": false,
280 "some_future_zai_field": 42
281 }"#,
282 )
283 .unwrap();
284 assert_eq!(parsed.do_sample, Some(false));
285 assert_eq!(
286 parsed
287 .zai_platform
288 .as_ref()
289 .expect("zai_platform")
290 .watermark_enabled,
291 Some(false)
292 );
293 // Only the genuinely unknown key may fall through to the catch-all.
294 let extra = parsed.extra_body_map.as_ref().expect("extra_body_map");
295 assert_eq!(extra.len(), 1, "extra_body_map: {extra:?}");
296 assert_eq!(extra["some_future_zai_field"], 42);
297 }
298
299 /// The `retrieval` and `web_search` tools serialize under their `type`
300 /// tag with the payload nested under the matching key.
301 #[test]
302 fn tools_serialize_with_type_tag() {
303 let request = RequestBody {
304 messages: vec![Message::user("Hello")],
305 model: "glm-4.6".to_string(),
306 tools: Some(vec![
307 RequestTool::Retrieval {
308 retrieval: RetrievalTool {
309 knowledge_id: "kb-123".to_string(),
310 prompt_template: Some("使用以下资料回答".to_string()),
311 },
312 },
313 RequestTool::WebSearch {
314 web_search: WebSearchTool {
315 search_engine: Some(SearchEngine::SearchProJina),
316 enable: Some(true),
317 count: Some(5),
318 search_recency_filter: Some(SearchRecencyFilter::OneWeek),
319 content_size: Some(ContentSize::High),
320 result_sequence: Some(ResultSequence::Before),
321 ..Default::default()
322 },
323 },
324 ]),
325 ..Default::default()
326 };
327
328 let json = serde_json::to_value(&request).unwrap();
329 assert_eq!(json["tools"][0]["type"], serde_json::json!("retrieval"));
330 assert_eq!(
331 json["tools"][0]["retrieval"]["knowledge_id"],
332 serde_json::json!("kb-123")
333 );
334 assert_eq!(json["tools"][1]["type"], serde_json::json!("web_search"));
335 assert_eq!(
336 json["tools"][1]["web_search"]["search_engine"],
337 serde_json::json!("search_pro_jina")
338 );
339 assert_eq!(
340 json["tools"][1]["web_search"]["search_recency_filter"],
341 serde_json::json!("oneWeek")
342 );
343 // Unset web_search fields stay off the wire.
344 assert!(json["tools"][1]["web_search"].get("search_query").is_none());
345 }
346
347 /// The response `web_search` array parses into `WebSearchResult`s, and an
348 /// unknown search engine is preserved rather than rejected.
349 #[test]
350 fn web_search_result_parses_and_engine_tolerates_unknown() {
351 let result: WebSearchResult = serde_json::from_value(serde_json::json!({
352 "title": "Rust 1.88 released",
353 "link": "https://example.com/rust",
354 "refer": "1",
355 "publish_date": "2026-09-01"
356 }))
357 .unwrap();
358 assert_eq!(result.title.as_deref(), Some("Rust 1.88 released"));
359 assert_eq!(result.refer.as_deref(), Some("1"));
360 assert!(result.content.is_none());
361
362 let engine: SearchEngine = serde_json::from_str(r#""search_pro_jina""#).unwrap();
363 assert_eq!(engine, SearchEngine::SearchProJina);
364 let unknown: SearchEngine = serde_json::from_str(r#""some_future_engine""#).unwrap();
365 assert_eq!(
366 unknown,
367 SearchEngine::Unknown("some_future_engine".to_string())
368 );
369 assert_eq!(unknown.as_str(), "some_future_engine");
370 }
371}