openai_interface/images/mod.rs
1//! Given a prompt and/or an input image, the model will generate a new
2//! image.
3//!
4//! Submodules: `generate`, `edit`, and `variation` for specific image
5//! operations. Streaming image generation (`stream` / `partial_images`) is
6//! not supported yet.
7//!
8//! > ![warn] The endpoints in this module are untested!
9//! > No OpenAI-compatible provider accessible to this project (DeepSeek,
10//! > Alibaba Cloud Model Studio) implements the `/images/*` endpoints, and
11//! > no OpenAI API key was available for testing. If you encounter any
12//! > issues, please report them on the repository.
13
14pub mod edit;
15pub mod generate;
16pub mod variation;
17
18use serde::{Deserialize, Serialize};
19
20/// Represents the content or the URL of an image generated by the OpenAI
21/// API.
22#[derive(Debug, Deserialize, Serialize, Clone)]
23pub struct Image {
24 /// The base64-encoded JSON of the generated image.
25 ///
26 /// Returned by default for the GPT image models, and only present if
27 /// `response_format` is set to `b64_json` for `dall-e-2` and `dall-e-3`.
28 pub b64_json: Option<String>,
29 /// For `dall-e-3` only, the revised prompt that was used to generate the
30 /// image.
31 pub revised_prompt: Option<String>,
32 /// When using `dall-e-2` or `dall-e-3`, the URL of the generated image
33 /// if `response_format` is set to `url` (default value). Unsupported for
34 /// the GPT image models.
35 pub url: Option<String>,
36}
37
38/// The response from the image generation endpoint.
39#[derive(Debug, Deserialize, Serialize, Clone)]
40pub struct ImagesResponse {
41 /// The Unix timestamp (in seconds) of when the image was created.
42 pub created: u64,
43 /// The list of generated images.
44 pub data: Option<Vec<Image>>,
45 /// The background parameter used for the image generation. Either
46 /// `transparent` or `opaque`. Echoed back for the GPT image models.
47 pub background: Option<String>,
48 /// The output format of the image generation. Either `png`, `webp`, or
49 /// `jpeg`. Echoed back for the GPT image models.
50 pub output_format: Option<String>,
51 /// The quality of the image generated. Either `low`, `medium`, or
52 /// `high`. Echoed back for the GPT image models.
53 pub quality: Option<String>,
54 /// The size of the image generated. Echoed back for the GPT image
55 /// models.
56 pub size: Option<String>,
57 /// For `gpt-image-1` only, the token usage information for the image
58 /// generation.
59 pub usage: Option<Usage>,
60}
61
62/// For `gpt-image-1` only, the token usage information for the image
63/// generation.
64#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
65pub struct Usage {
66 /// The number of tokens (images and text) in the input prompt.
67 pub input_tokens: u64,
68 /// The input tokens detailed information for the image generation.
69 pub input_tokens_details: UsageInputTokensDetails,
70 /// The number of output tokens generated by the model.
71 pub output_tokens: u64,
72 /// The total number of tokens (images and text) used for the image
73 /// generation.
74 pub total_tokens: u64,
75 /// The output token details for the image generation.
76 pub output_tokens_details: Option<UsageOutputTokensDetails>,
77}
78
79/// The input tokens detailed information for the image generation.
80#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
81pub struct UsageInputTokensDetails {
82 /// The number of image tokens in the input prompt.
83 pub image_tokens: u64,
84 /// The number of text tokens in the input prompt.
85 pub text_tokens: u64,
86}
87
88/// The output token details for the image generation.
89#[derive(Debug, Deserialize, Serialize, Clone, PartialEq)]
90pub struct UsageOutputTokensDetails {
91 /// The number of image output tokens generated by the model.
92 pub image_tokens: u64,
93 /// The number of text output tokens generated by the model.
94 pub text_tokens: u64,
95}
96
97crate::impl_from_str!(ImagesResponse);
98
99/// The background of the generated image(s).
100#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Background {
103 Transparent,
104 Opaque,
105 Auto,
106}
107
108/// The format in which the generated images are returned. This parameter is
109/// only supported for the GPT image models.
110#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
111#[serde(rename_all = "snake_case")]
112pub enum OutputFormat {
113 Png,
114 Jpeg,
115 Webp,
116}
117
118/// The format in which generated images with `dall-e-2` and `dall-e-3` are
119/// returned. Must be one of `url` or `b64_json`.
120#[derive(Debug, Serialize, Deserialize, Clone, Copy)]
121#[serde(rename_all = "snake_case")]
122pub enum ImageResponseFormat {
123 Url,
124 B64Json,
125}
126
127/// Serializes a unit-variant enum value to its bare string literal, e.g.
128/// `Url` -> `"url"`. Used for multipart text fields.
129pub(super) fn enum_to_literal<T: Serialize>(value: &T) -> Result<String, crate::errors::OapiError> {
130 serde_json::to_string(value)
131 .map(|json| json.trim_matches('"').to_string())
132 .map_err(|e| {
133 crate::errors::OapiError::ResponseError(format!("Failed to serialize field: {e}"))
134 })
135}
136
137#[cfg(test)]
138mod tests {
139 /// Deserializes a `dall-e-3` style image response with `url` data.
140 ///
141 /// No accessible provider implements this endpoint, so this fixture is
142 /// NOT captured from a live response. The structure follows the schema
143 /// of openai-python `types/image.py` + `types/images_response.py`; the
144 /// values are constructed for the test.
145 #[test]
146 fn parse_url_response() {
147 let content = r#"{
148 "created": 1706745938,
149 "data": [
150 {
151 "url": "https://example.com/image.png",
152 "revised_prompt": "a painted nebula with stars"
153 }
154 ]
155 }"#;
156
157 let response: super::ImagesResponse = content.parse().unwrap();
158 assert_eq!(response.created, 1706745938);
159 let data = response.data.unwrap();
160 assert_eq!(data.len(), 1);
161 assert_eq!(
162 data[0].url.as_deref(),
163 Some("https://example.com/image.png")
164 );
165 assert_eq!(
166 data[0].revised_prompt.as_deref(),
167 Some("a painted nebula with stars")
168 );
169 assert_eq!(data[0].b64_json, None);
170 assert_eq!(response.usage, None);
171 }
172
173 /// Deserializes a `gpt-image-1` style image response with `b64_json`
174 /// data and usage.
175 ///
176 /// No accessible provider implements this endpoint, so this fixture is
177 /// NOT captured from a live response. The structure follows the schema
178 /// of openai-python `types/images_response.py` (including the top-level
179 /// echoed `background` / `output_format` / `quality` / `size` fields and
180 /// the `usage` object); the values are constructed for the test.
181 #[test]
182 fn parse_b64_response() {
183 let content = r#"{
184 "created": 1706745938,
185 "data": [
186 {
187 "b64_json": "aGVsbG8gd29ybGQ="
188 }
189 ],
190 "background": "transparent",
191 "output_format": "png",
192 "quality": "high",
193 "size": "1024x1024",
194 "usage": {
195 "input_tokens": 10,
196 "input_tokens_details": {
197 "image_tokens": 5,
198 "text_tokens": 5
199 },
200 "output_tokens": 4096,
201 "total_tokens": 4106,
202 "output_tokens_details": {
203 "image_tokens": 4096,
204 "text_tokens": 0
205 }
206 }
207 }"#;
208
209 let response: super::ImagesResponse = content.parse().unwrap();
210 assert_eq!(response.background.as_deref(), Some("transparent"));
211 assert_eq!(response.output_format.as_deref(), Some("png"));
212 assert_eq!(response.quality.as_deref(), Some("high"));
213 assert_eq!(response.size.as_deref(), Some("1024x1024"));
214 let usage = response.usage.unwrap();
215 assert_eq!(usage.input_tokens, 10);
216 assert_eq!(usage.input_tokens_details.image_tokens, 5);
217 assert_eq!(usage.input_tokens_details.text_tokens, 5);
218 assert_eq!(usage.output_tokens, 4096);
219 assert_eq!(usage.total_tokens, 4106);
220 assert_eq!(
221 usage.output_tokens_details.as_ref().unwrap().image_tokens,
222 4096
223 );
224 let data = response.data.unwrap();
225 assert_eq!(data[0].b64_json.as_deref(), Some("aGVsbG8gd29ybGQ="));
226 assert_eq!(data[0].url, None);
227 }
228}