Skip to main content

openai_interface/audio/
speech.rs

1//! Generates audio from the input text.
2//!
3//! Endpoint: `POST /audio/speech` (JSON request / binary response). The
4//! response body is the raw audio bytes in the requested format.
5//!
6//! Streaming audio (`stream_format`) is not supported yet.
7//!
8//! > ![warn] This module is untested!
9//! > No OpenAI-compatible provider accessible to this project implements
10//! > this endpoint, and no OpenAI API key was available for testing. If you
11//! > encounter any issues, please report them on the repository.
12
13use serde::Serialize;
14use url::Url;
15
16use crate::{
17    errors::OapiError,
18    rest::post::{Post, PostBinary},
19};
20
21/// Generates audio from the input text.
22#[derive(Debug, Serialize, Default, Clone)]
23pub struct SpeechRequest {
24    /// The text to generate audio for. The maximum length is 4096
25    /// characters.
26    pub input: String,
27    /// One of the available TTS models: `tts-1`, `tts-1-hd`,
28    /// `gpt-4o-mini-tts`, or `gpt-4o-mini-tts-2025-12-15`.
29    pub model: String,
30    /// The voice to use when generating the audio.
31    ///
32    /// Supported built-in voices are `alloy`, `ash`, `ballad`, `coral`,
33    /// `echo`, `fable`, `onyx`, `nova`, `sage`, `shimmer`, `verse`, `marin`,
34    /// and `cedar`. A custom voice object with an `id` may also be
35    /// provided, for example `{ "id": "voice_1234" }`.
36    pub voice: SpeechVoice,
37    /// Control the voice of your generated audio with additional
38    /// instructions. Does not work with `tts-1` or `tts-1-hd`.
39    #[serde(skip_serializing_if = "Option::is_none")]
40    pub instructions: Option<String>,
41    /// The format to audio in. Supported formats are `mp3`, `opus`, `aac`,
42    /// `flac`, `wav`, and `pcm`.
43    #[serde(skip_serializing_if = "Option::is_none")]
44    pub response_format: Option<SpeechFormat>,
45    /// The speed of the generated audio. Select a value from `0.25` to
46    /// `4.0`. `1.0` is the default.
47    #[serde(skip_serializing_if = "Option::is_none")]
48    pub speed: Option<f32>,
49}
50
51/// The voice to use when generating audio.
52#[derive(Debug, Serialize, Clone)]
53#[serde(untagged)]
54pub enum SpeechVoice {
55    /// A built-in voice name, e.g. `alloy`, `ash`, `coral`, or `shimmer`.
56    BuiltIn(String),
57    /// A custom voice reference, e.g. `{ "id": "voice_1234" }`.
58    Custom {
59        /// The custom voice ID, e.g. `voice_1234`.
60        id: String,
61    },
62}
63
64impl Default for SpeechVoice {
65    fn default() -> Self {
66        Self::BuiltIn(String::new())
67    }
68}
69
70/// The format of the generated audio.
71#[derive(Debug, Serialize, Clone, Copy)]
72#[serde(rename_all = "snake_case")]
73pub enum SpeechFormat {
74    Mp3,
75    Opus,
76    Aac,
77    Flac,
78    Wav,
79    Pcm,
80}
81
82impl Post for SpeechRequest {
83    #[inline]
84    fn is_streaming(&self) -> bool {
85        false
86    }
87
88    /// Builds the URL for the request.
89    ///
90    /// `base_url` should be like <https://api.openai.com/v1>
91    fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
92        let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
93        url.path_segments_mut()
94            .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
95            .push("audio")
96            .push("speech");
97
98        Ok(url.to_string())
99    }
100}
101
102impl PostBinary for SpeechRequest {}
103
104#[cfg(test)]
105mod tests {
106    use super::*;
107
108    /// Serializes a speech request.
109    #[test]
110    fn request_serialization() {
111        let request = SpeechRequest {
112            input: "The quick brown fox jumped over the lazy dog.".to_string(),
113            model: "gpt-4o-mini-tts".to_string(),
114            voice: SpeechVoice::BuiltIn("alloy".to_string()),
115            instructions: Some("Voice: cheerful.".to_string()),
116            response_format: Some(SpeechFormat::Wav),
117            speed: Some(1.5),
118        };
119
120        let json = serde_json::to_string(&request).unwrap();
121        assert!(
122            json.contains(r#""input":"The quick brown fox jumped over the lazy dog.""#),
123            "json: {json}"
124        );
125        assert!(
126            json.contains(r#""model":"gpt-4o-mini-tts""#),
127            "json: {json}"
128        );
129        assert!(json.contains(r#""voice":"alloy""#), "json: {json}");
130        assert!(
131            json.contains(r#""instructions":"Voice: cheerful.""#),
132            "json: {json}"
133        );
134        assert!(json.contains(r#""response_format":"wav""#), "json: {json}");
135        assert!(json.contains(r#""speed":1.5"#), "json: {json}");
136    }
137
138    /// A custom voice reference serializes as an object with an `id`.
139    #[test]
140    fn custom_voice_serialization() {
141        let request = SpeechRequest {
142            input: "Hello".to_string(),
143            model: "gpt-4o-mini-tts".to_string(),
144            voice: SpeechVoice::Custom {
145                id: "voice_1234".to_string(),
146            },
147            ..Default::default()
148        };
149
150        let json = serde_json::to_string(&request).unwrap();
151        assert!(
152            json.contains(r#""voice":{"id":"voice_1234"}"#),
153            "json: {json}"
154        );
155    }
156
157    #[test]
158    fn test_build_url() {
159        let request = SpeechRequest::default();
160        let url = request.build_url("https://api.openai.com/v1/").unwrap();
161        assert_eq!(url, "https://api.openai.com/v1/audio/speech");
162    }
163}