openai-interface 0.6.0

A low-level Rust interface for the OpenAI API
Documentation
//! Generates audio from the input text.
//!
//! Endpoint: `POST /audio/speech` (JSON request / binary response). The
//! response body is the raw audio bytes in the requested format.
//!
//! Streaming audio (`stream_format`) is not supported yet.
//!
//! > ![warn] This module is untested!
//! > No OpenAI-compatible provider accessible to this project implements
//! > this endpoint, and no OpenAI API key was available for testing. If you
//! > encounter any issues, please report them on the repository.

use serde::Serialize;
use url::Url;

use crate::{
    errors::OapiError,
    rest::post::{Post, PostBinary},
};

/// Generates audio from the input text.
#[derive(Debug, Serialize, Default, Clone)]
pub struct SpeechRequest {
    /// The text to generate audio for. The maximum length is 4096
    /// characters.
    pub input: String,
    /// One of the available TTS models: `tts-1`, `tts-1-hd`,
    /// `gpt-4o-mini-tts`, or `gpt-4o-mini-tts-2025-12-15`.
    pub model: String,
    /// The voice to use when generating the audio.
    ///
    /// Supported built-in voices are `alloy`, `ash`, `ballad`, `coral`,
    /// `echo`, `fable`, `onyx`, `nova`, `sage`, `shimmer`, `verse`, `marin`,
    /// and `cedar`. A custom voice object with an `id` may also be
    /// provided, for example `{ "id": "voice_1234" }`.
    pub voice: SpeechVoice,
    /// Control the voice of your generated audio with additional
    /// instructions. Does not work with `tts-1` or `tts-1-hd`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub instructions: Option<String>,
    /// The format to audio in. Supported formats are `mp3`, `opus`, `aac`,
    /// `flac`, `wav`, and `pcm`.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub response_format: Option<SpeechFormat>,
    /// The speed of the generated audio. Select a value from `0.25` to
    /// `4.0`. `1.0` is the default.
    #[serde(skip_serializing_if = "Option::is_none")]
    pub speed: Option<f32>,
}

/// The voice to use when generating audio.
#[derive(Debug, Serialize, Clone)]
#[serde(untagged)]
pub enum SpeechVoice {
    /// A built-in voice name, e.g. `alloy`, `ash`, `coral`, or `shimmer`.
    BuiltIn(String),
    /// A custom voice reference, e.g. `{ "id": "voice_1234" }`.
    Custom {
        /// The custom voice ID, e.g. `voice_1234`.
        id: String,
    },
}

impl Default for SpeechVoice {
    fn default() -> Self {
        Self::BuiltIn(String::new())
    }
}

/// The format of the generated audio.
#[derive(Debug, Serialize, Clone, Copy)]
#[serde(rename_all = "snake_case")]
pub enum SpeechFormat {
    Mp3,
    Opus,
    Aac,
    Flac,
    Wav,
    Pcm,
}

impl Post for SpeechRequest {
    #[inline]
    fn is_streaming(&self) -> bool {
        false
    }

    /// Builds the URL for the request.
    ///
    /// `base_url` should be like <https://api.openai.com/v1>
    fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
        let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
        url.path_segments_mut()
            .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
            .push("audio")
            .push("speech");

        Ok(url.to_string())
    }
}

impl PostBinary for SpeechRequest {}

#[cfg(test)]
mod tests {
    use super::*;

    /// Serializes a speech request.
    #[test]
    fn request_serialization() {
        let request = SpeechRequest {
            input: "The quick brown fox jumped over the lazy dog.".to_string(),
            model: "gpt-4o-mini-tts".to_string(),
            voice: SpeechVoice::BuiltIn("alloy".to_string()),
            instructions: Some("Voice: cheerful.".to_string()),
            response_format: Some(SpeechFormat::Wav),
            speed: Some(1.5),
        };

        let json = serde_json::to_string(&request).unwrap();
        assert!(
            json.contains(r#""input":"The quick brown fox jumped over the lazy dog.""#),
            "json: {json}"
        );
        assert!(
            json.contains(r#""model":"gpt-4o-mini-tts""#),
            "json: {json}"
        );
        assert!(json.contains(r#""voice":"alloy""#), "json: {json}");
        assert!(
            json.contains(r#""instructions":"Voice: cheerful.""#),
            "json: {json}"
        );
        assert!(json.contains(r#""response_format":"wav""#), "json: {json}");
        assert!(json.contains(r#""speed":1.5"#), "json: {json}");
    }

    /// A custom voice reference serializes as an object with an `id`.
    #[test]
    fn custom_voice_serialization() {
        let request = SpeechRequest {
            input: "Hello".to_string(),
            model: "gpt-4o-mini-tts".to_string(),
            voice: SpeechVoice::Custom {
                id: "voice_1234".to_string(),
            },
            ..Default::default()
        };

        let json = serde_json::to_string(&request).unwrap();
        assert!(
            json.contains(r#""voice":{"id":"voice_1234"}"#),
            "json: {json}"
        );
    }

    #[test]
    fn test_build_url() {
        let request = SpeechRequest::default();
        let url = request.build_url("https://api.openai.com/v1/").unwrap();
        assert_eq!(url, "https://api.openai.com/v1/audio/speech");
    }
}