Skip to main content

openai_interface/audio/
transcriptions.rs

1//! Transcribes audio into the input language.
2//!
3//! Endpoint: `POST /audio/transcriptions` (multipart/form-data request).
4//!
5//! Response shapes depend on `response_format`:
6//!
7//! - `json` and `verbose_json` deserialize into the typed
8//!   [`TranscriptionResponse`] via `get_response`.
9//! - `text`, `srt`, and `vtt` return plain text; use
10//!   `get_response_string` for those.
11//!
12//! Streaming transcriptions (`stream`) is not supported yet.
13//!
14//! > ![warn] This module is untested!
15//! > No OpenAI-compatible provider accessible to this project implements
16//! > this endpoint, and no OpenAI API key was available for testing. If you
17//! > encounter any issues, please report them on the repository.
18
19use std::path::PathBuf;
20
21use serde::{Deserialize, Serialize};
22use url::Url;
23
24use crate::{
25    audio::AudioResponseFormat,
26    errors::OapiError,
27    rest::post::{Post, PostNoStream},
28};
29
30/// Transcribes audio into the input language.
31///
32/// The `keywords`, `languages`, `known_speaker_names`, and
33/// `known_speaker_references` parameters are not covered by this type yet.
34#[derive(Debug, Serialize, Default, Clone)]
35pub struct TranscriptionRequest {
36    /// The audio file (as a path) to transcribe, in one of these formats:
37    /// flac, mp3, mp4, mpeg, mpga, m4a, ogg, wav, or webm. The request must
38    /// include enough format metadata for the file to be identified; an
39    /// extension-bearing filename satisfies this.
40    #[serde(skip_serializing)]
41    pub file: PathBuf,
42    /// ID of the model to use. The options are `gpt-transcribe`,
43    /// `gpt-4o-transcribe`, `gpt-4o-mini-transcribe`, `whisper-1` (which is
44    /// powered by the open source Whisper V2 model), and
45    /// `gpt-4o-transcribe-diarize`.
46    pub model: String,
47    /// The language of the input audio. Supplying the input language in
48    /// [ISO-639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)
49    /// (e.g. `en`) format will improve accuracy and latency.
50    #[serde(skip_serializing_if = "Option::is_none")]
51    pub language: Option<String>,
52    /// An optional text to guide the model's style or continue a previous
53    /// audio segment. The prompt should match the audio language.
54    #[serde(skip_serializing_if = "Option::is_none")]
55    pub prompt: Option<String>,
56    /// The format of the output, in one of these options: `json`, `text`,
57    /// `srt`, `verbose_json`, or `vtt`. For `gpt-4o-transcribe` and
58    /// `gpt-4o-mini-transcribe`, the only supported format is `json`.
59    ///
60    /// With `json` / `verbose_json`, use `get_response`; with `text` /
61    /// `srt` / `vtt`, use `get_response_string`.
62    #[serde(skip_serializing_if = "Option::is_none")]
63    pub response_format: Option<AudioResponseFormat>,
64    /// The sampling temperature, between 0 and 1. Higher values like 0.8
65    /// will make the output more random, while lower values like 0.2 will
66    /// make it more focused and deterministic.
67    #[serde(skip_serializing_if = "Option::is_none")]
68    pub temperature: Option<f32>,
69    /// The timestamp granularities to populate for this transcription.
70    /// `response_format` must be set to `verbose_json` to use timestamp
71    /// granularities. Either or both of these options are supported: `word`,
72    /// or `segment`.
73    #[serde(skip_serializing_if = "Option::is_none")]
74    pub timestamp_granularities: Option<Vec<TimestampGranularity>>,
75    /// Controls how the audio is cut into chunks.
76    ///
77    /// When set to `auto`, the server first normalizes loudness and then
78    /// uses voice activity detection (VAD) to choose boundaries. A
79    /// `server_vad` object can be provided to tweak VAD detection
80    /// parameters manually. If unset, the audio is transcribed as a single
81    /// block.
82    #[serde(skip_serializing)]
83    pub chunking_strategy: Option<ChunkingStrategy>,
84    /// Additional information to include in the transcription response.
85    /// `logprobs` will return the log probabilities of the tokens in the
86    /// response.
87    #[serde(skip_serializing_if = "Option::is_none")]
88    pub include: Option<Vec<Include>>,
89}
90
91/// The timestamp granularities to populate for a transcription.
92#[derive(Debug, Serialize, Clone, Copy)]
93#[serde(rename_all = "snake_case")]
94pub enum TimestampGranularity {
95    Word,
96    Segment,
97}
98
99/// Additional information to include in the transcription response.
100#[derive(Debug, Serialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Include {
103    Logprobs,
104}
105
106/// Controls how the audio is cut into chunks.
107#[derive(Debug, Clone)]
108pub enum ChunkingStrategy {
109    /// The server normalizes loudness and then uses voice activity detection
110    /// (VAD) to choose boundaries.
111    Auto,
112    /// Tweak VAD detection parameters manually.
113    ServerVad(ServerVadConfig),
114}
115
116/// Manual VAD detection parameters, sent as the `server_vad` chunking
117/// strategy.
118#[derive(Debug, Clone, Default)]
119pub struct ServerVadConfig {
120    /// Amount of audio to include before the VAD detected speech (in
121    /// milliseconds).
122    pub prefix_padding_ms: Option<u32>,
123    /// Duration of silence to detect speech stop (in milliseconds).
124    pub silence_duration_ms: Option<u32>,
125    /// Sensitivity threshold (0.0 to 1.0) for voice activity detection.
126    pub threshold: Option<f32>,
127}
128
129/// The typed transcription response: `json` yields a plain
130/// [`Transcription`], `verbose_json` yields a
131/// [`TranscriptionVerbose`].
132#[derive(Debug, Deserialize, Clone)]
133#[serde(untagged)]
134pub enum TranscriptionResponse {
135    /// The `verbose_json` response shape (requires `duration` and
136    /// `language`).
137    Verbose(TranscriptionVerbose),
138    /// The `json` response shape.
139    Plain(Transcription),
140}
141
142/// Represents a transcription response returned by the model.
143#[derive(Debug, Deserialize, Clone)]
144pub struct Transcription {
145    /// The transcribed text.
146    pub text: String,
147    /// The languages detected in the audio.
148    ///
149    /// Returned by `gpt-transcribe`. An empty array indicates that no
150    /// language could be reliably detected.
151    pub languages: Option<Vec<TranscriptionLanguage>>,
152    /// The log probabilities of the tokens in the transcription.
153    ///
154    /// Only returned with the models `gpt-4o-transcribe` and
155    /// `gpt-4o-mini-transcribe` if `logprobs` is added to the `include`
156    /// array.
157    pub logprobs: Option<Vec<TranscriptionLogprob>>,
158    /// Usage statistics for the request.
159    pub usage: Option<TranscriptionUsage>,
160}
161
162/// A language detected in transcribed audio.
163#[derive(Debug, Deserialize, Clone, PartialEq)]
164pub struct TranscriptionLanguage {
165    /// The code of a language detected in the audio.
166    pub code: String,
167}
168
169/// The log probability of a token in the transcription.
170#[derive(Debug, Deserialize, Clone, PartialEq)]
171pub struct TranscriptionLogprob {
172    /// The token in the transcription.
173    pub token: Option<String>,
174    /// The bytes of the token.
175    pub bytes: Option<Vec<f32>>,
176    /// The log probability of the token.
177    pub logprob: Option<f32>,
178}
179
180/// Usage statistics for a transcription request. Billed either by token
181/// usage or by audio input duration, discriminated by `type`.
182#[derive(Debug, Deserialize, Clone, PartialEq)]
183#[serde(tag = "type", rename_all = "snake_case")]
184pub enum TranscriptionUsage {
185    /// Usage statistics for models billed by token usage.
186    Tokens {
187        /// Number of input tokens billed for this request.
188        input_tokens: u64,
189        /// Number of output tokens generated.
190        output_tokens: u64,
191        /// Total number of tokens used (input + output).
192        total_tokens: u64,
193        /// Details about the input tokens billed for this request.
194        input_token_details: Option<UsageTokensInputTokenDetails>,
195    },
196    /// Usage statistics for models billed by audio input duration.
197    Duration {
198        /// Duration of the input audio in seconds.
199        seconds: f64,
200    },
201}
202
203/// Details about the input tokens billed for a request.
204#[derive(Debug, Deserialize, Clone, PartialEq)]
205pub struct UsageTokensInputTokenDetails {
206    /// Number of audio tokens billed for this request.
207    pub audio_tokens: Option<u64>,
208    /// Number of text tokens billed for this request.
209    pub text_tokens: Option<u64>,
210}
211
212/// Represents a verbose json transcription response.
213#[derive(Debug, Deserialize, Clone)]
214pub struct TranscriptionVerbose {
215    /// The duration of the input audio.
216    pub duration: f64,
217    /// The language of the input audio.
218    pub language: String,
219    /// The transcribed text.
220    pub text: String,
221    /// Segments of the transcribed text and their corresponding details.
222    pub segments: Option<Vec<TranscriptionSegment>>,
223    /// Usage statistics for models billed by audio input duration.
224    pub usage: Option<TranscriptionVerboseUsage>,
225    /// Extracted words and their corresponding timestamps.
226    pub words: Option<Vec<TranscriptionWord>>,
227}
228
229/// Usage statistics for models billed by audio input duration.
230#[derive(Debug, Deserialize, Clone)]
231pub struct TranscriptionVerboseUsage {
232    /// Duration of the input audio in seconds.
233    pub seconds: f64,
234}
235
236/// A segment of the transcribed text and its corresponding details.
237#[derive(Debug, Deserialize, Clone)]
238pub struct TranscriptionSegment {
239    /// Unique identifier of the segment.
240    pub id: u64,
241    /// Average logprob of the segment.
242    ///
243    /// If the value is lower than -1, consider the logprobs failed.
244    pub avg_logprob: f64,
245    /// Compression ratio of the segment.
246    ///
247    /// If the value is greater than 2.4, consider the compression failed.
248    pub compression_ratio: f64,
249    /// End time of the segment in seconds.
250    pub end: f64,
251    /// Probability of no speech in the segment.
252    pub no_speech_prob: f64,
253    /// Seek offset of the segment.
254    pub seek: u64,
255    /// Start time of the segment in seconds.
256    pub start: f64,
257    /// Temperature parameter used for generating the segment.
258    pub temperature: f64,
259    /// Text content of the segment.
260    pub text: String,
261    /// Array of token IDs for the text content.
262    pub tokens: Vec<u64>,
263}
264
265/// An extracted word and its corresponding timestamp.
266#[derive(Debug, Deserialize, Clone)]
267pub struct TranscriptionWord {
268    /// End time of the word in seconds.
269    pub end: f64,
270    /// Start time of the word in seconds.
271    pub start: f64,
272    /// The text content of the word.
273    pub word: String,
274}
275
276crate::impl_from_str!(TranscriptionResponse);
277
278impl Post for TranscriptionRequest {
279    #[inline]
280    fn is_streaming(&self) -> bool {
281        false
282    }
283
284    /// Builds the URL for the request.
285    ///
286    /// `base_url` should be like <https://api.openai.com/v1>
287    fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
288        let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
289        url.path_segments_mut()
290            .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
291            .push("audio")
292            .push("transcriptions");
293
294        Ok(url.to_string())
295    }
296}
297
298impl PostNoStream for TranscriptionRequest {
299    type Response = TranscriptionResponse;
300
301    /// Sends a transcription POST request using multipart/form-data format,
302    /// following the field layout of the official SDK.
303    async fn get_response_string(
304        &self,
305        client: &reqwest::Client,
306        url: &str,
307        key: &str,
308    ) -> Result<String, OapiError> {
309        if !self.file.exists() {
310            return Err(OapiError::FileNotFoundError(self.file.clone()));
311        }
312
313        let content = tokio::fs::read(&self.file).await?;
314        let file_name = self
315            .file
316            .file_name()
317            .and_then(|name| name.to_str())
318            .ok_or_else(|| OapiError::ResponseError("Invalid file name".to_string()))?
319            .to_string();
320
321        let file_part = reqwest::multipart::Part::bytes(content).file_name(file_name);
322        let mut form = reqwest::multipart::Form::new().part("file", file_part);
323
324        form = form.text("model", self.model.clone());
325
326        if let Some(language) = &self.language {
327            form = form.text("language", language.clone());
328        }
329        if let Some(prompt) = &self.prompt {
330            form = form.text("prompt", prompt.clone());
331        }
332        if let Some(response_format) = self.response_format {
333            let literal = crate::audio::enum_to_literal(&response_format)?;
334            form = form.text("response_format", literal);
335        }
336        if let Some(temperature) = self.temperature {
337            form = form.text("temperature", temperature.to_string());
338        }
339        // List parameters are sent as repeated `name[]` parts, matching the
340        // official SDK.
341        if let Some(granularities) = &self.timestamp_granularities {
342            for granularity in granularities {
343                let literal = crate::audio::enum_to_literal(granularity)?;
344                form = form.text("timestamp_granularities[]", literal);
345            }
346        }
347        if let Some(include) = &self.include {
348            for item in include {
349                let literal = crate::audio::enum_to_literal(item)?;
350                form = form.text("include[]", literal);
351            }
352        }
353        if let Some(chunking_strategy) = &self.chunking_strategy {
354            let value = match chunking_strategy {
355                ChunkingStrategy::Auto => "auto".to_string(),
356                ChunkingStrategy::ServerVad(config) => {
357                    let mut map = serde_json::Map::new();
358                    map.insert("type".to_string(), "server_vad".into());
359                    if let Some(v) = config.prefix_padding_ms {
360                        map.insert("prefix_padding_ms".to_string(), v.into());
361                    }
362                    if let Some(v) = config.silence_duration_ms {
363                        map.insert("silence_duration_ms".to_string(), v.into());
364                    }
365                    if let Some(v) = config.threshold {
366                        map.insert("threshold".to_string(), v.into());
367                    }
368                    serde_json::to_string(&map).map_err(|e| {
369                        OapiError::ResponseError(format!(
370                            "Failed to serialize chunking_strategy: {e}"
371                        ))
372                    })?
373                }
374            };
375            form = form.text("chunking_strategy", value);
376        }
377
378        let response = client
379            .post(url)
380            .header("Accept", "application/json")
381            .bearer_auth(key)
382            .multipart(form)
383            .send()
384            .await?;
385
386        crate::rest::response_text_checked(response).await
387    }
388}
389
390#[cfg(test)]
391mod tests {
392    use super::*;
393
394    #[test]
395    fn test_build_url() {
396        let request = TranscriptionRequest::default();
397        let url = request.build_url("https://api.openai.com/v1/").unwrap();
398        assert_eq!(url, "https://api.openai.com/v1/audio/transcriptions");
399    }
400
401    /// Enum literals serialize to their official wire values.
402    #[test]
403    fn enum_literals() {
404        assert_eq!(
405            crate::audio::enum_to_literal(&AudioResponseFormat::VerboseJson).unwrap(),
406            "verbose_json"
407        );
408        assert_eq!(
409            crate::audio::enum_to_literal(&TimestampGranularity::Word).unwrap(),
410            "word"
411        );
412        assert_eq!(
413            crate::audio::enum_to_literal(&Include::Logprobs).unwrap(),
414            "logprobs"
415        );
416    }
417
418    /// Deserializes a `json` response (plain transcription shape).
419    ///
420    /// No accessible provider implements this endpoint, so this fixture is
421    /// NOT captured from a live response. The structure follows the schema
422    /// of openai-python `types/audio/transcription.py`; the values are
423    /// constructed for the test.
424    #[test]
425    fn parse_plain_response() {
426        let content = r#"{
427            "text": "The quick brown fox jumped over the lazy dog."
428        }"#;
429
430        let response: TranscriptionResponse = content.parse().unwrap();
431        let TranscriptionResponse::Plain(transcription) = response else {
432            panic!("expected plain transcription");
433        };
434        assert_eq!(
435            transcription.text,
436            "The quick brown fox jumped over the lazy dog."
437        );
438        assert_eq!(transcription.languages, None);
439        assert_eq!(transcription.logprobs, None);
440        assert_eq!(transcription.usage, None);
441    }
442
443    /// Deserializes a `verbose_json` response with segments, words, and
444    /// duration-billed usage.
445    ///
446    /// No accessible provider implements this endpoint, so this fixture is
447    /// NOT captured from a live response. The structure follows the schema
448    /// of openai-python `types/audio/transcription_verbose.py` +
449    /// `transcription_segment.py` + `transcription_word.py`; the values are
450    /// constructed for the test.
451    #[test]
452    fn parse_verbose_response() {
453        let content = r#"{
454            "duration": 8.47,
455            "language": "english",
456            "text": "The quick brown fox jumped over the lazy dog.",
457            "segments": [
458                {
459                    "id": 0,
460                    "avg_logprob": -0.2365,
461                    "compression_ratio": 1.7174,
462                    "end": 3.48,
463                    "no_speech_prob": 0.01485,
464                    "seek": 0,
465                    "start": 0.0,
466                    "temperature": 0.0,
467                    "text": " The quick brown fox jumped over the lazy dog.",
468                    "tokens": [464, 2069, 7586, 21831, 18045, 625, 262, 16931, 3290, 13]
469                }
470            ],
471            "words": [
472                {
473                    "end": 0.36,
474                    "start": 0.06,
475                    "word": "The"
476                }
477            ],
478            "usage": {
479                "type": "duration",
480                "seconds": 8.47
481            }
482        }"#;
483
484        let response: TranscriptionResponse = content.parse().unwrap();
485        let TranscriptionResponse::Verbose(verbose) = response else {
486            panic!("expected verbose transcription");
487        };
488        assert_eq!(verbose.duration, 8.47);
489        assert_eq!(verbose.language, "english");
490        assert_eq!(
491            verbose.text,
492            "The quick brown fox jumped over the lazy dog."
493        );
494        let segments = verbose.segments.unwrap();
495        assert_eq!(segments.len(), 1);
496        assert_eq!(segments[0].id, 0);
497        assert_eq!(segments[0].start, 0.0);
498        assert_eq!(segments[0].end, 3.48);
499        assert_eq!(segments[0].tokens.len(), 10);
500        let words = verbose.words.unwrap();
501        assert_eq!(words[0].word, "The");
502        assert_eq!(words[0].start, 0.06);
503        let usage = verbose.usage.unwrap();
504        assert_eq!(usage.seconds, 8.47);
505    }
506
507    /// Deserializes a token-billed usage object (discriminated by `type`).
508    ///
509    /// No accessible provider implements this endpoint, so this fixture is
510    /// NOT captured from a live response. The structure follows the schema
511    /// of openai-python `types/audio/transcription.py` (`UsageTokens`);
512    /// the values are constructed for the test.
513    #[test]
514    fn parse_tokens_usage() {
515        let content = r#"{
516            "text": "Hello.",
517            "usage": {
518                "type": "tokens",
519                "input_tokens": 76,
520                "output_tokens": 13,
521                "total_tokens": 89,
522                "input_token_details": {
523                    "audio_tokens": 76,
524                    "text_tokens": 0
525                }
526            }
527        }"#;
528
529        let response: TranscriptionResponse = content.parse().unwrap();
530        let TranscriptionResponse::Plain(transcription) = response else {
531            panic!("expected plain transcription");
532        };
533        let TranscriptionUsage::Tokens {
534            input_tokens,
535            output_tokens,
536            total_tokens,
537            input_token_details,
538        } = transcription.usage.unwrap()
539        else {
540            panic!("expected tokens usage");
541        };
542        assert_eq!(input_tokens, 76);
543        assert_eq!(output_tokens, 13);
544        assert_eq!(total_tokens, 89);
545        let details = input_token_details.unwrap();
546        assert_eq!(details.audio_tokens, Some(76));
547        assert_eq!(details.text_tokens, Some(0));
548    }
549}