Skip to main content

openai_interface/images/
mod.rs

1//! Given a prompt and/or an input image, the model will generate a new
2//! image.
3//!
4//! Submodules: `generate`, `edit`, and `variation` for specific image
5//! operations. Streaming image generation (`stream` / `partial_images`) is
6//! not supported yet.
7//!
8//! > ![warn] The endpoints in this module are untested!
9//! > No OpenAI-compatible provider accessible to this project (DeepSeek,
10//! > Alibaba Cloud Model Studio) implements the `/images/*` endpoints, and
11//! > no OpenAI API key was available for testing. If you encounter any
12//! > issues, please report them on the repository.
13
14pub mod edit;
15pub mod generate;
16pub mod variation;
17
18use serde::{Deserialize, Serialize};
19
20/// Represents the content or the URL of an image generated by the OpenAI
21/// API.
22#[derive(Debug, Deserialize, Serialize, Clone)]
23pub struct Image {
24    /// The base64-encoded JSON of the generated image.
25    ///
26    /// Returned by default for the GPT image models, and only present if
27    /// `response_format` is set to `b64_json` for `dall-e-2` and `dall-e-3`.
28    pub b64_json: Option<String>,
29    /// For `dall-e-3` only, the revised prompt that was used to generate the
30    /// image.
31    pub revised_prompt: Option<String>,
32    /// When using `dall-e-2` or `dall-e-3`, the URL of the generated image
33    /// if `response_format` is set to `url` (default value). Unsupported for
34    /// the GPT image models.
35    pub url: Option<String>,
36}
37
38/// The response from the image generation endpoint.
39#[derive(Debug, Deserialize, Serialize, Clone)]
40pub struct ImagesResponse {
41    /// The Unix timestamp (in seconds) of when the image was created.
42    pub created: u64,
43    /// The list of generated images.
44    pub data: Option<Vec<Image>>,
45    /// The background parameter used for the image generation. Either
46    /// `transparent` or `opaque`. Echoed back for the GPT image models.
47    pub background: Option<String>,
48    /// The output format of the image generation. Either `png`, `webp`, or
49    /// `jpeg`. Echoed back for the GPT image models.
50    pub output_format: Option<String>,
51    /// The quality of the image generated. Either `low`, `medium`, or
52    /// `high`. Echoed back for the GPT image models.
53    pub quality: Option<String>,
54    /// The size of the image generated. Echoed back for the GPT image
55    /// models.
56    pub size: Option<String>,
57    /// For `gpt-image-1` only, the token usage information for the image
58    /// generation.
59    pub usage: Option<Usage>,
60}
61
62/// For `gpt-image-1` only, the token usage information for the image
63/// generation.
64#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
65pub struct Usage {
66    /// The number of tokens (images and text) in the input prompt.
67    pub input_tokens: u64,
68    /// The input tokens detailed information for the image generation.
69    pub input_tokens_details: UsageInputTokensDetails,
70    /// The number of output tokens generated by the model.
71    pub output_tokens: u64,
72    /// The total number of tokens (images and text) used for the image
73    /// generation.
74    pub total_tokens: u64,
75    /// The output token details for the image generation.
76    pub output_tokens_details: Option<UsageOutputTokensDetails>,
77}
78
79/// The input tokens detailed information for the image generation.
80#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
81pub struct UsageInputTokensDetails {
82    /// The number of image tokens in the input prompt.
83    pub image_tokens: u64,
84    /// The number of text tokens in the input prompt.
85    pub text_tokens: u64,
86}
87
88/// The output token details for the image generation.
89#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
90pub struct UsageOutputTokensDetails {
91    /// The number of image output tokens generated by the model.
92    pub image_tokens: u64,
93    /// The number of text output tokens generated by the model.
94    pub text_tokens: u64,
95}
96
97crate::impl_from_str!(ImagesResponse);
98
99/// The background of the generated image(s).
100#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Background {
103    Transparent,
104    Opaque,
105    Auto,
106}
107
108/// The format in which the generated images are returned. This parameter is
109/// only supported for the GPT image models.
110#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
111#[serde(rename_all = "snake_case")]
112pub enum OutputFormat {
113    Png,
114    Jpeg,
115    Webp,
116}
117
118/// The format in which generated images with `dall-e-2` and `dall-e-3` are
119/// returned. Must be one of `url` or `b64_json`.
120#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
121#[serde(rename_all = "snake_case")]
122pub enum ImageResponseFormat {
123    Url,
124    B64Json,
125}
126
127/// Serializes a unit-variant enum value to its bare string literal, e.g.
128/// `Url` -> `"url"`. Used for multipart text fields.
129pub(super) fn enum_to_literal<T: Serialize>(value: &T) -> Result<String, crate::errors::OapiError> {
130    serde_json::to_string(value)
131        .map(|json| json.trim_matches('"').to_string())
132        .map_err(|e| {
133            crate::errors::OapiError::ResponseError(format!("Failed to serialize field: {e}"))
134        })
135}
136
137#[cfg(test)]
138mod tests {
139    /// Deserializes a `dall-e-3` style image response with `url` data.
140    ///
141    /// No accessible provider implements this endpoint, so this fixture is
142    /// NOT captured from a live response. The structure follows the schema
143    /// of openai-python `types/image.py` + `types/images_response.py`; the
144    /// values are constructed for the test.
145    #[test]
146    fn parse_url_response() {
147        let content = r#"{
148            "created": 1706745938,
149            "data": [
150                {
151                    "url": "https://example.com/image.png",
152                    "revised_prompt": "a painted nebula with stars"
153                }
154            ]
155        }"#;
156
157        let response: super::ImagesResponse = content.parse().unwrap();
158        assert_eq!(response.created, 1706745938);
159        let data = response.data.unwrap();
160        assert_eq!(data.len(), 1);
161        assert_eq!(
162            data[0].url.as_deref(),
163            Some("https://example.com/image.png")
164        );
165        assert_eq!(
166            data[0].revised_prompt.as_deref(),
167            Some("a painted nebula with stars")
168        );
169        assert_eq!(data[0].b64_json, None);
170        assert_eq!(response.usage, None);
171    }
172
173    /// Deserializes a `gpt-image-1` style image response with `b64_json`
174    /// data and usage.
175    ///
176    /// No accessible provider implements this endpoint, so this fixture is
177    /// NOT captured from a live response. The structure follows the schema
178    /// of openai-python `types/images_response.py` (including the top-level
179    /// echoed `background` / `output_format` / `quality` / `size` fields and
180    /// the `usage` object); the values are constructed for the test.
181    #[test]
182    fn parse_b64_response() {
183        let content = r#"{
184            "created": 1706745938,
185            "data": [
186                {
187                    "b64_json": "aGVsbG8gd29ybGQ="
188                }
189            ],
190            "background": "transparent",
191            "output_format": "png",
192            "quality": "high",
193            "size": "1024x1024",
194            "usage": {
195                "input_tokens": 10,
196                "input_tokens_details": {
197                    "image_tokens": 5,
198                    "text_tokens": 5
199                },
200                "output_tokens": 4096,
201                "total_tokens": 4106,
202                "output_tokens_details": {
203                    "image_tokens": 4096,
204                    "text_tokens": 0
205                }
206            }
207        }"#;
208
209        let response: super::ImagesResponse = content.parse().unwrap();
210        assert_eq!(response.background.as_deref(), Some("transparent"));
211        assert_eq!(response.output_format.as_deref(), Some("png"));
212        assert_eq!(response.quality.as_deref(), Some("high"));
213        assert_eq!(response.size.as_deref(), Some("1024x1024"));
214        let usage = response.usage.unwrap();
215        assert_eq!(usage.input_tokens, 10);
216        assert_eq!(usage.input_tokens_details.image_tokens, 5);
217        assert_eq!(usage.input_tokens_details.text_tokens, 5);
218        assert_eq!(usage.output_tokens, 4096);
219        assert_eq!(usage.total_tokens, 4106);
220        assert_eq!(
221            usage.output_tokens_details.as_ref().unwrap().image_tokens,
222            4096
223        );
224        let data = response.data.unwrap();
225        assert_eq!(data[0].b64_json.as_deref(), Some("aGVsbG8gd29ybGQ="));
226        assert_eq!(data[0].url, None);
227    }
228}