openai-interface 0.14.0

A low-level Rust interface for the OpenAI API
Documentation
//! Z.ai / 智谱 GLM (BigModel) proprietary extensions to the OpenAI-compatible
//! API.
//!
//! Everything in this module is gated on the `zai` cargo feature.
//!
//! GLM is served behind an OpenAI-compatible endpoint at
//! `https://open.bigmodel.cn/api/paas/v4` (the international brand is Z.ai,
//! `https://api.z.ai/api/paas/v4`), authenticated with the usual
//! `Authorization: Bearer <key>` header. Most of the request body is standard
//! OpenAI; this module collects the parts that are not.
//!
//! GLM's divergences fall into two groups, handled differently:
//!
//! - **Generic controls** that GLM merely spells its own way live directly on
//!   [`RequestBody`](crate::chat::create::request::RequestBody) as `zai`-gated
//!   fields — `do_sample` and `tool_stream`. The keys GLM shares with another
//!   provider are *unified* rather than duplicated: `thinking` and `user_id`
//!   are gated on `any(deepseek, zai)`, `request_id` on `any(vllm, zai)`, so
//!   enabling several provider features at once never emits a key twice.
//! - **Platform-ecosystem extensions** — features tied to Zhipu's own platform
//!   rather than to text generation — are grouped here:
//!   - [`PlatformParams`], `#[serde(flatten)]`ed into the request body through
//!     `RequestBody::zai_platform` (currently the `watermark_enabled`
//!     compliance flag).
//!   - the [`retrieval`](RetrievalTool) and [`web_search`](WebSearchTool) tool
//!     types, added as `zai`-gated variants of
//!     [`RequestTool`](crate::chat::create::request::RequestTool);
//!   - the [`web_search`](WebSearchResult) results GLM returns at the top level
//!     of a chat completion.
//!
//! ```rust
//! # #[cfg(feature = "zai")] {
//! use openai_interface::chat::create::request::{Message, RequestBody};
//! use openai_interface::zai::{PlatformParams, SearchEngine, WebSearchTool};
//!
//! let request = RequestBody {
//!     messages: vec![Message::user("最近有什么关于 Rust 的新闻?")],
//!     model: "glm-4.6".to_string(),
//!     do_sample: Some(false),
//!     zai_platform: Some(PlatformParams {
//!         watermark_enabled: Some(false),
//!     }),
//!     tools: Some(vec![
//!         openai_interface::chat::create::request::RequestTool::WebSearch {
//!             web_search: WebSearchTool {
//!                 search_engine: Some(SearchEngine::SearchProJina),
//!                 enable: Some(true),
//!                 ..Default::default()
//!             },
//!         },
//!     ]),
//!     ..Default::default()
//! };
//!
//! let json = serde_json::to_value(&request).unwrap();
//! assert_eq!(json["do_sample"], serde_json::json!(false));
//! assert_eq!(json["watermark_enabled"], serde_json::json!(false));
//! assert_eq!(json["tools"][0]["type"], serde_json::json!("web_search"));
//! # }
//! ```
//!
//! # Fields that need no gate
//!
//! - `reasoning_effort` is already an ungated
//!   [`RequestBody`](crate::chat::create::request::RequestBody) field, and its
//!   [`ReasoningEffort`](crate::chat::create::request::ReasoningEffort) enum
//!   already covers every GLM value (`none`, `minimal`, `low`, `medium`,
//!   `high`, `xhigh`, `max`). GLM only honours it when `thinking` is enabled,
//!   and the value-to-depth mapping is model-specific.
//! - `temperature`, `top_p` and `max_tokens` are standard keys; GLM simply
//!   accepts narrower ranges (`temperature` and `top_p` in `[0, 1]`, two
//!   decimals; `max_tokens` up to `131072`). No extra field models a range.
//! - `reasoning_content` on the response message and streamed delta is covered
//!   by the always-available `reasoning` feature.
//! - GLM's extra `finish_reason` values (`sensitive`,
//!   `model_context_window_exceeded`, `network_error`) are preserved by
//!   [`FinishReason::Unknown`](crate::chat::FinishReason) rather than modelled
//!   as variants, so an unexpected terminator never fails deserialization.
//!
//! # Not implemented
//!
//! GLM also exposes an `mcp` tool type, image/video generation, embeddings and
//! a separate GLM Coding Plan endpoint. None are documented precisely enough
//! to model here; this feature covers the fields GLM adds to the OpenAI chat
//! completions endpoint.
//!
//! See [the GLM chat-completions
//! reference](https://docs.bigmodel.cn/api-reference/模型-api/对话补全) and its
//! [Z.ai equivalent](https://docs.z.ai/api-reference/llm/chat-completion).
#![cfg(feature = "zai")]

use serde::{Deserialize, Serialize};

/// GLM platform-ecosystem request parameters, flattened into the request body.
///
/// These are Zhipu-platform features rather than text-generation controls, so
/// they are grouped here instead of sitting loose on
/// [`RequestBody`](crate::chat::create::request::RequestBody). Flattened
/// through `RequestBody::zai_platform`; see the [module docs](crate::zai).
#[derive(Serialize, Deserialize, Debug, Clone, Default)]
pub struct PlatformParams {
    /// Whether AI-generated output carries Zhipu's explicit and implicit
    /// watermark. GLM's compliant default is `true`; `false` is only accepted
    /// for accounts that have signed the corresponding waiver.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub watermark_enabled: Option<bool>,
}

/// GLM's `retrieval` tool: grounds the answer in one of Zhipu's knowledge
/// bases.
///
/// A `zai`-gated variant of
/// [`RequestTool`](crate::chat::create::request::RequestTool); the wire shape
/// is `{"type": "retrieval", "retrieval": { ... }}`.
#[derive(Serialize, Deserialize, Debug, Clone)]
pub struct RetrievalTool {
    /// The identifier of the knowledge base to retrieve from. Required by GLM.
    pub knowledge_id: String,
    /// An optional template controlling how retrieved chunks are prompted.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub prompt_template: Option<String>,
}

/// GLM's `web_search` tool: lets the model call Zhipu's web search.
///
/// A `zai`-gated variant of
/// [`RequestTool`](crate::chat::create::request::RequestTool); the wire shape
/// is `{"type": "web_search", "web_search": { ... }}`. Every field except
/// `search_engine` is optional and omitted when unset, so GLM applies its own
/// default.
#[derive(Serialize, Deserialize, Debug, Clone, Default)]
pub struct WebSearchTool {
    /// The search backend. Required by GLM; currently only
    /// [`SearchEngine::SearchProJina`] is accepted.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_engine: Option<SearchEngine>,
    /// Whether the tool is enabled. GLM's default is `false`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub enable: Option<bool>,
    /// An explicit query to search for, instead of letting the model phrase it.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_query: Option<String>,
    /// Number of results to return, `1..=50`. GLM's default is `10`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub count: Option<u32>,
    /// Restrict results to this domain.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_domain_filter: Option<String>,
    /// Only return results published within this window. GLM's default is
    /// [`SearchRecencyFilter::NoLimit`].
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_recency_filter: Option<SearchRecencyFilter>,
    /// How much of each page to keep. GLM's default is
    /// [`ContentSize::Medium`].
    #[serde(skip_serializing_if = "Option::is_none")]
    pub content_size: Option<ContentSize>,
    /// Whether search results are placed before or after the model's answer.
    /// GLM's default is [`ResultSequence::After`].
    #[serde(skip_serializing_if = "Option::is_none")]
    pub result_sequence: Option<ResultSequence>,
    /// Return the raw search results alongside the answer. GLM's default is
    /// `false`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_result: Option<bool>,
    /// Force a search even when the model thinks none is needed. GLM's default
    /// is `false`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub require_search: Option<bool>,
    /// Extra instructions steering what the model searches for.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub search_prompt: Option<String>,
}

crate::wire_string_enum! {
    /// The web-search backend GLM should use.
    pub enum SearchEngine {
        /// Zhipu's Jina-powered pro search. Currently the only accepted value.
        SearchProJina => "search_pro_jina",
    }
}

crate::wire_string_enum! {
    /// How recent a web-search result may be.
    pub enum SearchRecencyFilter {
        /// Published within the last day.
        OneDay => "oneDay",
        /// Published within the last week.
        OneWeek => "oneWeek",
        /// Published within the last month.
        OneMonth => "oneMonth",
        /// Published within the last year.
        OneYear => "oneYear",
        /// No recency restriction (GLM's default).
        NoLimit => "noLimit",
    }
}

crate::wire_string_enum! {
    /// How much of each searched page to keep.
    pub enum ContentSize {
        /// A medium-length extract (GLM's default).
        Medium => "medium",
        /// A longer extract.
        High => "high",
    }
}

crate::wire_string_enum! {
    /// Where the search results are placed relative to the answer.
    pub enum ResultSequence {
        /// Results before the answer.
        Before => "before",
        /// Results after the answer (GLM's default).
        After => "after",
    }
}

/// One web-search result GLM returns in the top-level `web_search` array of a
/// chat completion.
///
/// Not part of the OpenAI schema. Every field is optional: GLM populates what
/// the search backend returned, and a given result may omit several of them.
#[derive(Serialize, Deserialize, Debug, Clone, Default, PartialEq)]
pub struct WebSearchResult {
    /// The result's title.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub title: Option<String>,
    /// The snippet or extracted content of the result.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub content: Option<String>,
    /// The result's URL.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub link: Option<String>,
    /// A media URL associated with the result, when any.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub media: Option<String>,
    /// The source site's icon URL, when any.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub icon: Option<String>,
    /// The citation marker GLM associates with this result.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub refer: Option<String>,
    /// The publication date of the result, as a string.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub publish_date: Option<String>,
}

#[cfg(test)]
mod tests {
    use super::*;
    use crate::chat::create::request::{Message, RequestBody, RequestTool};

    /// The platform parameters must land at the top level of the body, and
    /// survive a round trip without leaking into the `extra_body_map`
    /// catch-all (the crate is used to build proxies).
    #[test]
    fn platform_params_flatten_to_top_level() {
        let request = RequestBody {
            messages: vec![Message::user("Hello")],
            model: "glm-4.6".to_string(),
            do_sample: Some(false),
            tool_stream: Some(true),
            zai_platform: Some(PlatformParams {
                watermark_enabled: Some(false),
            }),
            ..Default::default()
        };

        let json = serde_json::to_value(&request).unwrap();
        assert_eq!(json["do_sample"], serde_json::json!(false));
        assert_eq!(json["tool_stream"], serde_json::json!(true));
        assert_eq!(json["watermark_enabled"], serde_json::json!(false));

        let parsed: RequestBody = serde_json::from_str(
            r#"{
                "model": "glm-4.6",
                "messages": [{"role": "user", "content": "Hello"}],
                "do_sample": false,
                "watermark_enabled": false,
                "some_future_zai_field": 42
            }"#,
        )
        .unwrap();
        assert_eq!(parsed.do_sample, Some(false));
        assert_eq!(
            parsed
                .zai_platform
                .as_ref()
                .expect("zai_platform")
                .watermark_enabled,
            Some(false)
        );
        // Only the genuinely unknown key may fall through to the catch-all.
        let extra = parsed.extra_body_map.as_ref().expect("extra_body_map");
        assert_eq!(extra.len(), 1, "extra_body_map: {extra:?}");
        assert_eq!(extra["some_future_zai_field"], 42);
    }

    /// The `retrieval` and `web_search` tools serialize under their `type`
    /// tag with the payload nested under the matching key.
    #[test]
    fn tools_serialize_with_type_tag() {
        let request = RequestBody {
            messages: vec![Message::user("Hello")],
            model: "glm-4.6".to_string(),
            tools: Some(vec![
                RequestTool::Retrieval {
                    retrieval: RetrievalTool {
                        knowledge_id: "kb-123".to_string(),
                        prompt_template: Some("使用以下资料回答".to_string()),
                    },
                },
                RequestTool::WebSearch {
                    web_search: WebSearchTool {
                        search_engine: Some(SearchEngine::SearchProJina),
                        enable: Some(true),
                        count: Some(5),
                        search_recency_filter: Some(SearchRecencyFilter::OneWeek),
                        content_size: Some(ContentSize::High),
                        result_sequence: Some(ResultSequence::Before),
                        ..Default::default()
                    },
                },
            ]),
            ..Default::default()
        };

        let json = serde_json::to_value(&request).unwrap();
        assert_eq!(json["tools"][0]["type"], serde_json::json!("retrieval"));
        assert_eq!(
            json["tools"][0]["retrieval"]["knowledge_id"],
            serde_json::json!("kb-123")
        );
        assert_eq!(json["tools"][1]["type"], serde_json::json!("web_search"));
        assert_eq!(
            json["tools"][1]["web_search"]["search_engine"],
            serde_json::json!("search_pro_jina")
        );
        assert_eq!(
            json["tools"][1]["web_search"]["search_recency_filter"],
            serde_json::json!("oneWeek")
        );
        // Unset web_search fields stay off the wire.
        assert!(json["tools"][1]["web_search"].get("search_query").is_none());
    }

    /// The response `web_search` array parses into `WebSearchResult`s, and an
    /// unknown search engine is preserved rather than rejected.
    #[test]
    fn web_search_result_parses_and_engine_tolerates_unknown() {
        let result: WebSearchResult = serde_json::from_value(serde_json::json!({
            "title": "Rust 1.88 released",
            "link": "https://example.com/rust",
            "refer": "1",
            "publish_date": "2026-09-01"
        }))
        .unwrap();
        assert_eq!(result.title.as_deref(), Some("Rust 1.88 released"));
        assert_eq!(result.refer.as_deref(), Some("1"));
        assert!(result.content.is_none());

        let engine: SearchEngine = serde_json::from_str(r#""search_pro_jina""#).unwrap();
        assert_eq!(engine, SearchEngine::SearchProJina);
        let unknown: SearchEngine = serde_json::from_str(r#""some_future_engine""#).unwrap();
        assert_eq!(
            unknown,
            SearchEngine::Unknown("some_future_engine".to_string())
        );
        assert_eq!(unknown.as_str(), "some_future_engine");
    }
}