Skip to main content

openai_interface/audio/
speech.rs

1//! Generates audio from the input text.
2//!
3//! Endpoint: `POST /audio/speech` (JSON request / binary response). The
4//! response body is the raw audio bytes in the requested format.
5//!
6//! Streaming audio is supported through the optional `stream_format`
7//! parameter: when set, send the request via
8//! [`PostBinary::get_stream_response_bytes`](crate::rest::post::PostBinary::get_stream_response_bytes)
9//! to receive the response as a stream of raw byte chunks (for
10//! `stream_format = "sse"` the server-sent-event frames are carried inside
11//! those bytes; parsing them is left to the caller).
12//!
13//! > ![warn] This module is untested!
14//! > No OpenAI-compatible provider accessible to this project implements
15//! > this endpoint, and no OpenAI API key was available for testing. If you
16//! > encounter any issues, please report them on the repository.
17
18use serde::Serialize;
19use url::Url;
20
21use crate::{
22    errors::OapiError,
23    rest::post::{Post, PostBinary},
24};
25
26/// Generates audio from the input text.
27#[derive(Debug, Serialize, Default, Clone)]
28pub struct SpeechRequest {
29    /// The text to generate audio for. The maximum length is 4096
30    /// characters.
31    pub input: String,
32    /// One of the available TTS models: `tts-1`, `tts-1-hd`,
33    /// `gpt-4o-mini-tts`, or `gpt-4o-mini-tts-2025-12-15`.
34    pub model: String,
35    /// The voice to use when generating the audio.
36    ///
37    /// Supported built-in voices are `alloy`, `ash`, `ballad`, `coral`,
38    /// `echo`, `fable`, `onyx`, `nova`, `sage`, `shimmer`, `verse`, `marin`,
39    /// and `cedar`. A custom voice object with an `id` may also be
40    /// provided, for example `{ "id": "voice_1234" }`.
41    pub voice: SpeechVoice,
42    /// Control the voice of your generated audio with additional
43    /// instructions. Does not work with `tts-1` or `tts-1-hd`.
44    #[serde(skip_serializing_if = "Option::is_none")]
45    pub instructions: Option<String>,
46    /// The format to audio in. Supported formats are `mp3`, `opus`, `aac`,
47    /// `flac`, `wav`, and `pcm`.
48    #[serde(skip_serializing_if = "Option::is_none")]
49    pub response_format: Option<SpeechFormat>,
50    /// The speed of the generated audio. Select a value from `0.25` to
51    /// `4.0`. `1.0` is the default.
52    #[serde(skip_serializing_if = "Option::is_none")]
53    pub speed: Option<f32>,
54    /// The format to stream the audio in. Supported formats are `sse` and
55    /// `audio`. `sse` is not supported for `tts-1` or `tts-1-hd`.
56    ///
57    /// When set, send the request through
58    /// [`PostBinary::get_stream_response_bytes`] to receive the response as
59    /// a stream of byte chunks instead of a single buffered body.
60    #[serde(skip_serializing_if = "Option::is_none")]
61    pub stream_format: Option<SpeechStreamFormat>,
62}
63
64/// The format to stream generated audio in.
65#[derive(Debug, Serialize, Clone, Copy)]
66#[serde(rename_all = "lowercase")]
67pub enum SpeechStreamFormat {
68    /// Server-sent events carrying audio chunks (not supported for `tts-1`
69    /// or `tts-1-hd`).
70    Sse,
71    /// Raw streamed audio bytes.
72    Audio,
73}
74
75/// The voice to use when generating audio.
76#[derive(Debug, Serialize, Clone)]
77#[serde(untagged)]
78pub enum SpeechVoice {
79    /// A built-in voice name, e.g. `alloy`, `ash`, `coral`, or `shimmer`.
80    BuiltIn(String),
81    /// A custom voice reference, e.g. `{ "id": "voice_1234" }`.
82    Custom {
83        /// The custom voice ID, e.g. `voice_1234`.
84        id: String,
85    },
86}
87
88impl Default for SpeechVoice {
89    fn default() -> Self {
90        Self::BuiltIn(String::new())
91    }
92}
93
94/// The format of the generated audio.
95#[derive(Debug, Serialize, Clone, Copy)]
96#[serde(rename_all = "snake_case")]
97pub enum SpeechFormat {
98    Mp3,
99    Opus,
100    Aac,
101    Flac,
102    Wav,
103    Pcm,
104}
105
106impl Post for SpeechRequest {
107    #[inline]
108    fn is_streaming(&self) -> bool {
109        false
110    }
111
112    /// Builds the URL for the request.
113    ///
114    /// `base_url` should be like <https://api.openai.com/v1>
115    fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
116        let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
117        url.path_segments_mut()
118            .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
119            .push("audio")
120            .push("speech");
121
122        Ok(url.to_string())
123    }
124}
125
126impl PostBinary for SpeechRequest {}
127
128#[cfg(test)]
129mod tests {
130    use super::*;
131
132    /// Serializes a speech request.
133    #[test]
134    fn request_serialization() {
135        let request = SpeechRequest {
136            input: "The quick brown fox jumped over the lazy dog.".to_string(),
137            model: "gpt-4o-mini-tts".to_string(),
138            voice: SpeechVoice::BuiltIn("alloy".to_string()),
139            instructions: Some("Voice: cheerful.".to_string()),
140            response_format: Some(SpeechFormat::Wav),
141            speed: Some(1.5),
142            stream_format: None,
143        };
144
145        let json = serde_json::to_string(&request).unwrap();
146        assert!(
147            json.contains(r#""input":"The quick brown fox jumped over the lazy dog.""#),
148            "json: {json}"
149        );
150        assert!(
151            json.contains(r#""model":"gpt-4o-mini-tts""#),
152            "json: {json}"
153        );
154        assert!(json.contains(r#""voice":"alloy""#), "json: {json}");
155        assert!(
156            json.contains(r#""instructions":"Voice: cheerful.""#),
157            "json: {json}"
158        );
159        assert!(json.contains(r#""response_format":"wav""#), "json: {json}");
160        assert!(json.contains(r#""speed":1.5"#), "json: {json}");
161    }
162
163    /// A custom voice reference serializes as an object with an `id`.
164    #[test]
165    fn custom_voice_serialization() {
166        let request = SpeechRequest {
167            input: "Hello".to_string(),
168            model: "gpt-4o-mini-tts".to_string(),
169            voice: SpeechVoice::Custom {
170                id: "voice_1234".to_string(),
171            },
172            ..Default::default()
173        };
174
175        let json = serde_json::to_string(&request).unwrap();
176        assert!(
177            json.contains(r#""voice":{"id":"voice_1234"}"#),
178            "json: {json}"
179        );
180    }
181
182    /// `stream_format` serializes with the official literal values and is
183    /// omitted when unset.
184    #[test]
185    fn stream_format_serialization() {
186        let request = SpeechRequest {
187            input: "Hello".to_string(),
188            model: "gpt-4o-mini-tts".to_string(),
189            voice: SpeechVoice::BuiltIn("alloy".to_string()),
190            stream_format: Some(SpeechStreamFormat::Sse),
191            ..Default::default()
192        };
193        let json = serde_json::to_string(&request).unwrap();
194        assert!(json.contains(r#""stream_format":"sse""#), "json: {json}");
195
196        let request = SpeechRequest {
197            stream_format: Some(SpeechStreamFormat::Audio),
198            ..request
199        };
200        let json = serde_json::to_string(&request).unwrap();
201        assert!(json.contains(r#""stream_format":"audio""#), "json: {json}");
202
203        let request = SpeechRequest {
204            stream_format: None,
205            ..request
206        };
207        let json = serde_json::to_string(&request).unwrap();
208        assert!(!json.contains("stream_format"), "json: {json}");
209    }
210
211    #[test]
212    fn test_build_url() {
213        let request = SpeechRequest::default();
214        let url = request.build_url("https://api.openai.com/v1/").unwrap();
215        assert_eq!(url, "https://api.openai.com/v1/audio/speech");
216    }
217}