openai-interface 0.13.0

A low-level Rust interface for the OpenAI API
Documentation
//! Given a prompt and/or an input image, the model will generate a new
//! image.
//!
//! Submodules: `generate`, `edit`, and `variation` for specific image
//! operations. Streaming image generation (`stream` / `partial_images`) is
//! not supported yet.
//!
//! > ![warn] The endpoints in this module are untested!
//! > No OpenAI-compatible provider accessible to this project (DeepSeek,
//! > Alibaba Cloud Model Studio) implements the `/images/*` endpoints, and
//! > no OpenAI API key was available for testing. If you encounter any
//! > issues, please report them on the repository.

pub mod edit;
pub mod generate;
pub mod variation;

use serde::{Deserialize, Serialize};

/// Represents the content or the URL of an image generated by the OpenAI
/// API.
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct Image {
    /// The base64-encoded JSON of the generated image.
    ///
    /// Returned by default for the GPT image models, and only present if
    /// `response_format` is set to `b64_json` for `dall-e-2` and `dall-e-3`.
    pub b64_json: Option<String>,
    /// For `dall-e-3` only, the revised prompt that was used to generate the
    /// image.
    pub revised_prompt: Option<String>,
    /// When using `dall-e-2` or `dall-e-3`, the URL of the generated image
    /// if `response_format` is set to `url` (default value). Unsupported for
    /// the GPT image models.
    pub url: Option<String>,
}

/// The response from the image generation endpoint.
#[derive(Debug, Deserialize, Serialize, Clone)]
pub struct ImagesResponse {
    /// The Unix timestamp (in seconds) of when the image was created.
    pub created: u64,
    /// The list of generated images.
    pub data: Option<Vec<Image>>,
    /// The background parameter used for the image generation. Either
    /// `transparent` or `opaque`. Echoed back for the GPT image models.
    pub background: Option<String>,
    /// The output format of the image generation. Either `png`, `webp`, or
    /// `jpeg`. Echoed back for the GPT image models.
    pub output_format: Option<String>,
    /// The quality of the image generated. Either `low`, `medium`, or
    /// `high`. Echoed back for the GPT image models.
    pub quality: Option<String>,
    /// The size of the image generated. Echoed back for the GPT image
    /// models.
    pub size: Option<String>,
    /// For `gpt-image-1` only, the token usage information for the image
    /// generation.
    pub usage: Option<Usage>,
}

/// For `gpt-image-1` only, the token usage information for the image
/// generation.
#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
pub struct Usage {
    /// The number of tokens (images and text) in the input prompt.
    pub input_tokens: u64,
    /// The input tokens detailed information for the image generation.
    pub input_tokens_details: UsageInputTokensDetails,
    /// The number of output tokens generated by the model.
    pub output_tokens: u64,
    /// The total number of tokens (images and text) used for the image
    /// generation.
    pub total_tokens: u64,
    /// The output token details for the image generation.
    pub output_tokens_details: Option<UsageOutputTokensDetails>,
}

/// The input tokens detailed information for the image generation.
#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
pub struct UsageInputTokensDetails {
    /// The number of image tokens in the input prompt.
    pub image_tokens: u64,
    /// The number of text tokens in the input prompt.
    pub text_tokens: u64,
}

/// The output token details for the image generation.
#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
pub struct UsageOutputTokensDetails {
    /// The number of image output tokens generated by the model.
    pub image_tokens: u64,
    /// The number of text output tokens generated by the model.
    pub text_tokens: u64,
}

crate::impl_from_str!(ImagesResponse);

/// The background of the generated image(s).
#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
#[serde(rename_all = "snake_case")]
pub enum Background {
    Transparent,
    Opaque,
    Auto,
}

/// The format in which the generated images are returned. This parameter is
/// only supported for the GPT image models.
#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
#[serde(rename_all = "snake_case")]
pub enum OutputFormat {
    Png,
    Jpeg,
    Webp,
}

/// The format in which generated images with `dall-e-2` and `dall-e-3` are
/// returned. Must be one of `url` or `b64_json`.
#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
#[serde(rename_all = "snake_case")]
pub enum ImageResponseFormat {
    Url,
    B64Json,
}

/// Serializes a unit-variant enum value to its bare string literal, e.g.
/// `Url` -> `"url"`. Used for multipart text fields.
pub(super) fn enum_to_literal<T: Serialize>(value: &T) -> Result<String, crate::errors::OapiError> {
    serde_json::to_string(value)
        .map(|json| json.trim_matches('"').to_string())
        .map_err(|e| {
            crate::errors::OapiError::ResponseError(format!("Failed to serialize field: {e}"))
        })
}

#[cfg(test)]
mod tests {
    /// Deserializes a `dall-e-3` style image response with `url` data.
    ///
    /// No accessible provider implements this endpoint, so this fixture is
    /// NOT captured from a live response. The structure follows the schema
    /// of openai-python `types/image.py` + `types/images_response.py`; the
    /// values are constructed for the test.
    #[test]
    fn parse_url_response() {
        let content = r#"{
            "created": 1706745938,
            "data": [
                {
                    "url": "https://example.com/image.png",
                    "revised_prompt": "a painted nebula with stars"
                }
            ]
        }"#;

        let response: super::ImagesResponse = content.parse().unwrap();
        assert_eq!(response.created, 1706745938);
        let data = response.data.unwrap();
        assert_eq!(data.len(), 1);
        assert_eq!(
            data[0].url.as_deref(),
            Some("https://example.com/image.png")
        );
        assert_eq!(
            data[0].revised_prompt.as_deref(),
            Some("a painted nebula with stars")
        );
        assert_eq!(data[0].b64_json, None);
        assert_eq!(response.usage, None);
    }

    /// Deserializes a `gpt-image-1` style image response with `b64_json`
    /// data and usage.
    ///
    /// No accessible provider implements this endpoint, so this fixture is
    /// NOT captured from a live response. The structure follows the schema
    /// of openai-python `types/images_response.py` (including the top-level
    /// echoed `background` / `output_format` / `quality` / `size` fields and
    /// the `usage` object); the values are constructed for the test.
    #[test]
    fn parse_b64_response() {
        let content = r#"{
            "created": 1706745938,
            "data": [
                {
                    "b64_json": "aGVsbG8gd29ybGQ="
                }
            ],
            "background": "transparent",
            "output_format": "png",
            "quality": "high",
            "size": "1024x1024",
            "usage": {
                "input_tokens": 10,
                "input_tokens_details": {
                    "image_tokens": 5,
                    "text_tokens": 5
                },
                "output_tokens": 4096,
                "total_tokens": 4106,
                "output_tokens_details": {
                    "image_tokens": 4096,
                    "text_tokens": 0
                }
            }
        }"#;

        let response: super::ImagesResponse = content.parse().unwrap();
        assert_eq!(response.background.as_deref(), Some("transparent"));
        assert_eq!(response.output_format.as_deref(), Some("png"));
        assert_eq!(response.quality.as_deref(), Some("high"));
        assert_eq!(response.size.as_deref(), Some("1024x1024"));
        let usage = response.usage.unwrap();
        assert_eq!(usage.input_tokens, 10);
        assert_eq!(usage.input_tokens_details.image_tokens, 5);
        assert_eq!(usage.input_tokens_details.text_tokens, 5);
        assert_eq!(usage.output_tokens, 4096);
        assert_eq!(usage.total_tokens, 4106);
        assert_eq!(
            usage.output_tokens_details.as_ref().unwrap().image_tokens,
            4096
        );
        let data = response.data.unwrap();
        assert_eq!(data[0].b64_json.as_deref(), Some("aGVsbG8gd29ybGQ="));
        assert_eq!(data[0].url, None);
    }
}