Skip to main content

openai_interface/images/
mod.rs

1//! Given a prompt and/or an input image, the model will generate a new
2//! image.
3//!
4//! Submodules: `generate`, `edit`, and `variation` for specific image
5//! operations. Streaming image generation (`stream` / `partial_images`) is
6//! not supported yet.
7//!
8//! > ![warn] The endpoints in this module are untested!
9//! > No OpenAI-compatible provider accessible to this project (DeepSeek,
10//! > Alibaba Cloud Model Studio) implements the `/images/*` endpoints, and
11//! > no OpenAI API key was available for testing. If you encounter any
12//! > issues, please report them on the repository.
13
14pub mod edit;
15pub mod generate;
16pub mod variation;
17
18use serde::Deserialize;
19
20/// Represents the content or the URL of an image generated by the OpenAI
21/// API.
22#[derive(Debug, Deserialize, Clone)]
23pub struct Image {
24    /// The base64-encoded JSON of the generated image.
25    ///
26    /// Returned by default for the GPT image models, and only present if
27    /// `response_format` is set to `b64_json` for `dall-e-2` and `dall-e-3`.
28    pub b64_json: Option<String>,
29    /// For `dall-e-3` only, the revised prompt that was used to generate the
30    /// image.
31    pub revised_prompt: Option<String>,
32    /// When using `dall-e-2` or `dall-e-3`, the URL of the generated image
33    /// if `response_format` is set to `url` (default value). Unsupported for
34    /// the GPT image models.
35    pub url: Option<String>,
36}
37
38/// The response from the image generation endpoint.
39#[derive(Debug, Deserialize, Clone)]
40pub struct ImagesResponse {
41    /// The Unix timestamp (in seconds) of when the image was created.
42    pub created: u64,
43    /// The list of generated images.
44    pub data: Option<Vec<Image>>,
45    /// The background parameter used for the image generation. Either
46    /// `transparent` or `opaque`. Echoed back for the GPT image models.
47    pub background: Option<String>,
48    /// The output format of the image generation. Either `png`, `webp`, or
49    /// `jpeg`. Echoed back for the GPT image models.
50    pub output_format: Option<String>,
51    /// The quality of the image generated. Either `low`, `medium`, or
52    /// `high`. Echoed back for the GPT image models.
53    pub quality: Option<String>,
54    /// The size of the image generated. Echoed back for the GPT image
55    /// models.
56    pub size: Option<String>,
57    /// For `gpt-image-1` only, the token usage information for the image
58    /// generation.
59    pub usage: Option<Usage>,
60}
61
62/// For `gpt-image-1` only, the token usage information for the image
63/// generation.
64#[derive(Debug, Deserialize, Clone, PartialEq)]
65pub struct Usage {
66    /// The number of tokens (images and text) in the input prompt.
67    pub input_tokens: u64,
68    /// The input tokens detailed information for the image generation.
69    pub input_tokens_details: UsageInputTokensDetails,
70    /// The number of output tokens generated by the model.
71    pub output_tokens: u64,
72    /// The total number of tokens (images and text) used for the image
73    /// generation.
74    pub total_tokens: u64,
75    /// The output token details for the image generation.
76    pub output_tokens_details: Option<UsageOutputTokensDetails>,
77}
78
79/// The input tokens detailed information for the image generation.
80#[derive(Debug, Deserialize, Clone, PartialEq)]
81pub struct UsageInputTokensDetails {
82    /// The number of image tokens in the input prompt.
83    pub image_tokens: u64,
84    /// The number of text tokens in the input prompt.
85    pub text_tokens: u64,
86}
87
88/// The output token details for the image generation.
89#[derive(Debug, Deserialize, Clone, PartialEq)]
90pub struct UsageOutputTokensDetails {
91    /// The number of image output tokens generated by the model.
92    pub image_tokens: u64,
93    /// The number of text output tokens generated by the model.
94    pub text_tokens: u64,
95}
96
97crate::impl_from_str!(ImagesResponse);
98
99/// The background of the generated image(s).
100#[derive(Debug, Serialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Background {
103    Transparent,
104    Opaque,
105    Auto,
106}
107
108/// The format in which the generated images are returned. This parameter is
109/// only supported for the GPT image models.
110#[derive(Debug, Serialize, Clone, Copy)]
111#[serde(rename_all = "snake_case")]
112pub enum OutputFormat {
113    Png,
114    Jpeg,
115    Webp,
116}
117
118/// The format in which generated images with `dall-e-2` and `dall-e-3` are
119/// returned. Must be one of `url` or `b64_json`.
120#[derive(Debug, Serialize, Clone, Copy)]
121#[serde(rename_all = "snake_case")]
122pub enum ImageResponseFormat {
123    Url,
124    B64Json,
125}
126
127use serde::Serialize;
128
129/// Serializes a unit-variant enum value to its bare string literal, e.g.
130/// `Url` -> `"url"`. Used for multipart text fields.
131pub(super) fn enum_to_literal<T: Serialize>(value: &T) -> Result<String, crate::errors::OapiError> {
132    serde_json::to_string(value)
133        .map(|json| json.trim_matches('"').to_string())
134        .map_err(|e| {
135            crate::errors::OapiError::ResponseError(format!("Failed to serialize field: {e}"))
136        })
137}
138
139#[cfg(test)]
140mod tests {
141    /// Deserializes a `dall-e-3` style image response with `url` data.
142    ///
143    /// No accessible provider implements this endpoint, so this fixture is
144    /// NOT captured from a live response. The structure follows the schema
145    /// of openai-python `types/image.py` + `types/images_response.py`; the
146    /// values are constructed for the test.
147    #[test]
148    fn parse_url_response() {
149        let content = r#"{
150            "created": 1706745938,
151            "data": [
152                {
153                    "url": "https://example.com/image.png",
154                    "revised_prompt": "a painted nebula with stars"
155                }
156            ]
157        }"#;
158
159        let response: super::ImagesResponse = content.parse().unwrap();
160        assert_eq!(response.created, 1706745938);
161        let data = response.data.unwrap();
162        assert_eq!(data.len(), 1);
163        assert_eq!(
164            data[0].url.as_deref(),
165            Some("https://example.com/image.png")
166        );
167        assert_eq!(
168            data[0].revised_prompt.as_deref(),
169            Some("a painted nebula with stars")
170        );
171        assert_eq!(data[0].b64_json, None);
172        assert_eq!(response.usage, None);
173    }
174
175    /// Deserializes a `gpt-image-1` style image response with `b64_json`
176    /// data and usage.
177    ///
178    /// No accessible provider implements this endpoint, so this fixture is
179    /// NOT captured from a live response. The structure follows the schema
180    /// of openai-python `types/images_response.py` (including the top-level
181    /// echoed `background` / `output_format` / `quality` / `size` fields and
182    /// the `usage` object); the values are constructed for the test.
183    #[test]
184    fn parse_b64_response() {
185        let content = r#"{
186            "created": 1706745938,
187            "data": [
188                {
189                    "b64_json": "aGVsbG8gd29ybGQ="
190                }
191            ],
192            "background": "transparent",
193            "output_format": "png",
194            "quality": "high",
195            "size": "1024x1024",
196            "usage": {
197                "input_tokens": 10,
198                "input_tokens_details": {
199                    "image_tokens": 5,
200                    "text_tokens": 5
201                },
202                "output_tokens": 4096,
203                "total_tokens": 4106,
204                "output_tokens_details": {
205                    "image_tokens": 4096,
206                    "text_tokens": 0
207                }
208            }
209        }"#;
210
211        let response: super::ImagesResponse = content.parse().unwrap();
212        assert_eq!(response.background.as_deref(), Some("transparent"));
213        assert_eq!(response.output_format.as_deref(), Some("png"));
214        assert_eq!(response.quality.as_deref(), Some("high"));
215        assert_eq!(response.size.as_deref(), Some("1024x1024"));
216        let usage = response.usage.unwrap();
217        assert_eq!(usage.input_tokens, 10);
218        assert_eq!(usage.input_tokens_details.image_tokens, 5);
219        assert_eq!(usage.input_tokens_details.text_tokens, 5);
220        assert_eq!(usage.output_tokens, 4096);
221        assert_eq!(usage.total_tokens, 4106);
222        assert_eq!(
223            usage.output_tokens_details.as_ref().unwrap().image_tokens,
224            4096
225        );
226        let data = response.data.unwrap();
227        assert_eq!(data[0].b64_json.as_deref(), Some("aGVsbG8gd29ybGQ="));
228        assert_eq!(data[0].url, None);
229    }
230}