openai_interface/images/mod.rs
1//! Given a prompt and/or an input image, the model will generate a new
2//! image.
3//!
4//! Submodules: `generate`, `edit`, and `variation` for specific image
5//! operations. Streaming image generation (`stream` / `partial_images`) is
6//! not supported yet.
7//!
8//! > ![warn] The endpoints in this module are untested!
9//! > No OpenAI-compatible provider accessible to this project (DeepSeek,
10//! > Alibaba Cloud Model Studio) implements the `/images/*` endpoints, and
11//! > no OpenAI API key was available for testing. If you encounter any
12//! > issues, please report them on the repository.
13
14pub mod edit;
15pub mod generate;
16pub mod variation;
17
18use serde::Deserialize;
19
20/// Represents the content or the URL of an image generated by the OpenAI
21/// API.
22#[derive(Debug, Deserialize, Clone)]
23pub struct Image {
24 /// The base64-encoded JSON of the generated image.
25 ///
26 /// Returned by default for the GPT image models, and only present if
27 /// `response_format` is set to `b64_json` for `dall-e-2` and `dall-e-3`.
28 pub b64_json: Option<String>,
29 /// For `dall-e-3` only, the revised prompt that was used to generate the
30 /// image.
31 pub revised_prompt: Option<String>,
32 /// When using `dall-e-2` or `dall-e-3`, the URL of the generated image
33 /// if `response_format` is set to `url` (default value). Unsupported for
34 /// the GPT image models.
35 pub url: Option<String>,
36}
37
38/// The response from the image generation endpoint.
39#[derive(Debug, Deserialize, Clone)]
40pub struct ImagesResponse {
41 /// The Unix timestamp (in seconds) of when the image was created.
42 pub created: u64,
43 /// The list of generated images.
44 pub data: Option<Vec<Image>>,
45 /// The background parameter used for the image generation. Either
46 /// `transparent` or `opaque`. Echoed back for the GPT image models.
47 pub background: Option<String>,
48 /// The output format of the image generation. Either `png`, `webp`, or
49 /// `jpeg`. Echoed back for the GPT image models.
50 pub output_format: Option<String>,
51 /// The quality of the image generated. Either `low`, `medium`, or
52 /// `high`. Echoed back for the GPT image models.
53 pub quality: Option<String>,
54 /// The size of the image generated. Echoed back for the GPT image
55 /// models.
56 pub size: Option<String>,
57 /// For `gpt-image-1` only, the token usage information for the image
58 /// generation.
59 pub usage: Option<Usage>,
60}
61
62/// For `gpt-image-1` only, the token usage information for the image
63/// generation.
64#[derive(Debug, Deserialize, Clone, PartialEq)]
65pub struct Usage {
66 /// The number of tokens (images and text) in the input prompt.
67 pub input_tokens: u64,
68 /// The input tokens detailed information for the image generation.
69 pub input_tokens_details: UsageInputTokensDetails,
70 /// The number of output tokens generated by the model.
71 pub output_tokens: u64,
72 /// The total number of tokens (images and text) used for the image
73 /// generation.
74 pub total_tokens: u64,
75 /// The output token details for the image generation.
76 pub output_tokens_details: Option<UsageOutputTokensDetails>,
77}
78
79/// The input tokens detailed information for the image generation.
80#[derive(Debug, Deserialize, Clone, PartialEq)]
81pub struct UsageInputTokensDetails {
82 /// The number of image tokens in the input prompt.
83 pub image_tokens: u64,
84 /// The number of text tokens in the input prompt.
85 pub text_tokens: u64,
86}
87
88/// The output token details for the image generation.
89#[derive(Debug, Deserialize, Clone, PartialEq)]
90pub struct UsageOutputTokensDetails {
91 /// The number of image output tokens generated by the model.
92 pub image_tokens: u64,
93 /// The number of text output tokens generated by the model.
94 pub text_tokens: u64,
95}
96
97crate::impl_from_str!(ImagesResponse);
98
99/// The background of the generated image(s).
100#[derive(Debug, Serialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Background {
103 Transparent,
104 Opaque,
105 Auto,
106}
107
108/// The format in which the generated images are returned. This parameter is
109/// only supported for the GPT image models.
110#[derive(Debug, Serialize, Clone, Copy)]
111#[serde(rename_all = "snake_case")]
112pub enum OutputFormat {
113 Png,
114 Jpeg,
115 Webp,
116}
117
118/// The format in which generated images with `dall-e-2` and `dall-e-3` are
119/// returned. Must be one of `url` or `b64_json`.
120#[derive(Debug, Serialize, Clone, Copy)]
121#[serde(rename_all = "snake_case")]
122pub enum ImageResponseFormat {
123 Url,
124 B64Json,
125}
126
127use serde::Serialize;
128
129/// Serializes a unit-variant enum value to its bare string literal, e.g.
130/// `Url` -> `"url"`. Used for multipart text fields.
131pub(super) fn enum_to_literal<T: Serialize>(value: &T) -> Result<String, crate::errors::OapiError> {
132 serde_json::to_string(value)
133 .map(|json| json.trim_matches('"').to_string())
134 .map_err(|e| {
135 crate::errors::OapiError::ResponseError(format!("Failed to serialize field: {e}"))
136 })
137}
138
139#[cfg(test)]
140mod tests {
141 /// Deserializes a `dall-e-3` style image response with `url` data.
142 ///
143 /// No accessible provider implements this endpoint, so this fixture is
144 /// NOT captured from a live response. The structure follows the schema
145 /// of openai-python `types/image.py` + `types/images_response.py`; the
146 /// values are constructed for the test.
147 #[test]
148 fn parse_url_response() {
149 let content = r#"{
150 "created": 1706745938,
151 "data": [
152 {
153 "url": "https://example.com/image.png",
154 "revised_prompt": "a painted nebula with stars"
155 }
156 ]
157 }"#;
158
159 let response: super::ImagesResponse = content.parse().unwrap();
160 assert_eq!(response.created, 1706745938);
161 let data = response.data.unwrap();
162 assert_eq!(data.len(), 1);
163 assert_eq!(
164 data[0].url.as_deref(),
165 Some("https://example.com/image.png")
166 );
167 assert_eq!(
168 data[0].revised_prompt.as_deref(),
169 Some("a painted nebula with stars")
170 );
171 assert_eq!(data[0].b64_json, None);
172 assert_eq!(response.usage, None);
173 }
174
175 /// Deserializes a `gpt-image-1` style image response with `b64_json`
176 /// data and usage.
177 ///
178 /// No accessible provider implements this endpoint, so this fixture is
179 /// NOT captured from a live response. The structure follows the schema
180 /// of openai-python `types/images_response.py` (including the top-level
181 /// echoed `background` / `output_format` / `quality` / `size` fields and
182 /// the `usage` object); the values are constructed for the test.
183 #[test]
184 fn parse_b64_response() {
185 let content = r#"{
186 "created": 1706745938,
187 "data": [
188 {
189 "b64_json": "aGVsbG8gd29ybGQ="
190 }
191 ],
192 "background": "transparent",
193 "output_format": "png",
194 "quality": "high",
195 "size": "1024x1024",
196 "usage": {
197 "input_tokens": 10,
198 "input_tokens_details": {
199 "image_tokens": 5,
200 "text_tokens": 5
201 },
202 "output_tokens": 4096,
203 "total_tokens": 4106,
204 "output_tokens_details": {
205 "image_tokens": 4096,
206 "text_tokens": 0
207 }
208 }
209 }"#;
210
211 let response: super::ImagesResponse = content.parse().unwrap();
212 assert_eq!(response.background.as_deref(), Some("transparent"));
213 assert_eq!(response.output_format.as_deref(), Some("png"));
214 assert_eq!(response.quality.as_deref(), Some("high"));
215 assert_eq!(response.size.as_deref(), Some("1024x1024"));
216 let usage = response.usage.unwrap();
217 assert_eq!(usage.input_tokens, 10);
218 assert_eq!(usage.input_tokens_details.image_tokens, 5);
219 assert_eq!(usage.input_tokens_details.text_tokens, 5);
220 assert_eq!(usage.output_tokens, 4096);
221 assert_eq!(usage.total_tokens, 4106);
222 assert_eq!(
223 usage.output_tokens_details.as_ref().unwrap().image_tokens,
224 4096
225 );
226 let data = response.data.unwrap();
227 assert_eq!(data[0].b64_json.as_deref(), Some("aGVsbG8gd29ybGQ="));
228 assert_eq!(data[0].url, None);
229 }
230}