openai_interface/audio/transcriptions.rs
1//! Transcribes audio into the input language.
2//!
3//! Endpoint: `POST /audio/transcriptions` (multipart/form-data request).
4//!
5//! Response shapes depend on `response_format`:
6//!
7//! - `json` and `verbose_json` deserialize into the typed
8//! [`TranscriptionResponse`] via `get_response`.
9//! - `text`, `srt`, and `vtt` return plain text; use
10//! `get_response_string` for those.
11//!
12//! Streaming transcriptions (`stream`) is not supported yet.
13//!
14//! > ![warn] This module is untested!
15//! > No OpenAI-compatible provider accessible to this project implements
16//! > this endpoint, and no OpenAI API key was available for testing. If you
17//! > encounter any issues, please report them on the repository.
18
19use std::path::PathBuf;
20
21use serde::{Deserialize, Serialize};
22use url::Url;
23
24use crate::{
25 audio::AudioResponseFormat,
26 errors::OapiError,
27 rest::post::{Post, PostNoStream},
28};
29
30/// Transcribes audio into the input language.
31///
32/// The `keywords`, `languages`, `known_speaker_names`, and
33/// `known_speaker_references` parameters are not covered by this type yet.
34#[derive(Debug, Serialize, Default, Clone)]
35pub struct TranscriptionRequest {
36 /// The audio file (as a path) to transcribe, in one of these formats:
37 /// flac, mp3, mp4, mpeg, mpga, m4a, ogg, wav, or webm. The request must
38 /// include enough format metadata for the file to be identified; an
39 /// extension-bearing filename satisfies this.
40 #[serde(skip_serializing)]
41 pub file: PathBuf,
42 /// ID of the model to use. The options are `gpt-transcribe`,
43 /// `gpt-4o-transcribe`, `gpt-4o-mini-transcribe`, `whisper-1` (which is
44 /// powered by the open source Whisper V2 model), and
45 /// `gpt-4o-transcribe-diarize`.
46 pub model: String,
47 /// The language of the input audio. Supplying the input language in
48 /// [ISO-639-1](https://en.wikipedia.org/wiki/List_of_ISO_639-1_codes)
49 /// (e.g. `en`) format will improve accuracy and latency.
50 #[serde(skip_serializing_if = "Option::is_none")]
51 pub language: Option<String>,
52 /// An optional text to guide the model's style or continue a previous
53 /// audio segment. The prompt should match the audio language.
54 #[serde(skip_serializing_if = "Option::is_none")]
55 pub prompt: Option<String>,
56 /// The format of the output, in one of these options: `json`, `text`,
57 /// `srt`, `verbose_json`, or `vtt`. For `gpt-4o-transcribe` and
58 /// `gpt-4o-mini-transcribe`, the only supported format is `json`.
59 ///
60 /// With `json` / `verbose_json`, use `get_response`; with `text` /
61 /// `srt` / `vtt`, use `get_response_string`.
62 #[serde(skip_serializing_if = "Option::is_none")]
63 pub response_format: Option<AudioResponseFormat>,
64 /// The sampling temperature, between 0 and 1. Higher values like 0.8
65 /// will make the output more random, while lower values like 0.2 will
66 /// make it more focused and deterministic.
67 #[serde(skip_serializing_if = "Option::is_none")]
68 pub temperature: Option<f32>,
69 /// The timestamp granularities to populate for this transcription.
70 /// `response_format` must be set to `verbose_json` to use timestamp
71 /// granularities. Either or both of these options are supported: `word`,
72 /// or `segment`.
73 #[serde(skip_serializing_if = "Option::is_none")]
74 pub timestamp_granularities: Option<Vec<TimestampGranularity>>,
75 /// Controls how the audio is cut into chunks.
76 ///
77 /// When set to `auto`, the server first normalizes loudness and then
78 /// uses voice activity detection (VAD) to choose boundaries. A
79 /// `server_vad` object can be provided to tweak VAD detection
80 /// parameters manually. If unset, the audio is transcribed as a single
81 /// block.
82 #[serde(skip_serializing)]
83 pub chunking_strategy: Option<ChunkingStrategy>,
84 /// Additional information to include in the transcription response.
85 /// `logprobs` will return the log probabilities of the tokens in the
86 /// response.
87 #[serde(skip_serializing_if = "Option::is_none")]
88 pub include: Option<Vec<Include>>,
89}
90
91/// The timestamp granularities to populate for a transcription.
92#[derive(Debug, Serialize, Clone, Copy)]
93#[serde(rename_all = "snake_case")]
94pub enum TimestampGranularity {
95 Word,
96 Segment,
97}
98
99/// Additional information to include in the transcription response.
100#[derive(Debug, Serialize, Clone, Copy)]
101#[serde(rename_all = "snake_case")]
102pub enum Include {
103 Logprobs,
104}
105
106/// Controls how the audio is cut into chunks.
107#[derive(Debug, Clone)]
108pub enum ChunkingStrategy {
109 /// The server normalizes loudness and then uses voice activity detection
110 /// (VAD) to choose boundaries.
111 Auto,
112 /// Tweak VAD detection parameters manually.
113 ServerVad(ServerVadConfig),
114}
115
116/// Manual VAD detection parameters, sent as the `server_vad` chunking
117/// strategy.
118#[derive(Debug, Clone, Default)]
119pub struct ServerVadConfig {
120 /// Amount of audio to include before the VAD detected speech (in
121 /// milliseconds).
122 pub prefix_padding_ms: Option<u32>,
123 /// Duration of silence to detect speech stop (in milliseconds).
124 pub silence_duration_ms: Option<u32>,
125 /// Sensitivity threshold (0.0 to 1.0) for voice activity detection.
126 pub threshold: Option<f32>,
127}
128
129/// The typed transcription response: `json` yields a plain
130/// [`Transcription`], `verbose_json` yields a
131/// [`TranscriptionVerbose`].
132#[derive(Debug, Deserialize, Clone)]
133#[serde(untagged)]
134pub enum TranscriptionResponse {
135 /// The `verbose_json` response shape (requires `duration` and
136 /// `language`).
137 Verbose(TranscriptionVerbose),
138 /// The `json` response shape.
139 Plain(Transcription),
140}
141
142/// Represents a transcription response returned by the model.
143#[derive(Debug, Deserialize, Clone)]
144pub struct Transcription {
145 /// The transcribed text.
146 pub text: String,
147 /// The languages detected in the audio.
148 ///
149 /// Returned by `gpt-transcribe`. An empty array indicates that no
150 /// language could be reliably detected.
151 pub languages: Option<Vec<TranscriptionLanguage>>,
152 /// The log probabilities of the tokens in the transcription.
153 ///
154 /// Only returned with the models `gpt-4o-transcribe` and
155 /// `gpt-4o-mini-transcribe` if `logprobs` is added to the `include`
156 /// array.
157 pub logprobs: Option<Vec<TranscriptionLogprob>>,
158 /// Usage statistics for the request.
159 pub usage: Option<TranscriptionUsage>,
160}
161
162/// A language detected in transcribed audio.
163#[derive(Debug, Deserialize, Clone, PartialEq)]
164pub struct TranscriptionLanguage {
165 /// The code of a language detected in the audio.
166 pub code: String,
167}
168
169/// The log probability of a token in the transcription.
170#[derive(Debug, Deserialize, Clone, PartialEq)]
171pub struct TranscriptionLogprob {
172 /// The token in the transcription.
173 pub token: Option<String>,
174 /// The bytes of the token.
175 pub bytes: Option<Vec<f32>>,
176 /// The log probability of the token.
177 pub logprob: Option<f32>,
178}
179
180/// Usage statistics for a transcription request. Billed either by token
181/// usage or by audio input duration, discriminated by `type`.
182#[derive(Debug, Deserialize, Clone, PartialEq)]
183#[serde(tag = "type", rename_all = "snake_case")]
184pub enum TranscriptionUsage {
185 /// Usage statistics for models billed by token usage.
186 Tokens {
187 /// Number of input tokens billed for this request.
188 input_tokens: u64,
189 /// Number of output tokens generated.
190 output_tokens: u64,
191 /// Total number of tokens used (input + output).
192 total_tokens: u64,
193 /// Details about the input tokens billed for this request.
194 input_token_details: Option<UsageTokensInputTokenDetails>,
195 },
196 /// Usage statistics for models billed by audio input duration.
197 Duration {
198 /// Duration of the input audio in seconds.
199 seconds: f64,
200 },
201}
202
203/// Details about the input tokens billed for a request.
204#[derive(Debug, Deserialize, Clone, PartialEq)]
205pub struct UsageTokensInputTokenDetails {
206 /// Number of audio tokens billed for this request.
207 pub audio_tokens: Option<u64>,
208 /// Number of text tokens billed for this request.
209 pub text_tokens: Option<u64>,
210}
211
212/// Represents a verbose json transcription response.
213#[derive(Debug, Deserialize, Clone)]
214pub struct TranscriptionVerbose {
215 /// The duration of the input audio.
216 pub duration: f64,
217 /// The language of the input audio.
218 pub language: String,
219 /// The transcribed text.
220 pub text: String,
221 /// Segments of the transcribed text and their corresponding details.
222 pub segments: Option<Vec<TranscriptionSegment>>,
223 /// Usage statistics for models billed by audio input duration.
224 pub usage: Option<TranscriptionVerboseUsage>,
225 /// Extracted words and their corresponding timestamps.
226 pub words: Option<Vec<TranscriptionWord>>,
227}
228
229/// Usage statistics for models billed by audio input duration.
230#[derive(Debug, Deserialize, Clone)]
231pub struct TranscriptionVerboseUsage {
232 /// Duration of the input audio in seconds.
233 pub seconds: f64,
234}
235
236/// A segment of the transcribed text and its corresponding details.
237#[derive(Debug, Deserialize, Clone)]
238pub struct TranscriptionSegment {
239 /// Unique identifier of the segment.
240 pub id: u64,
241 /// Average logprob of the segment.
242 ///
243 /// If the value is lower than -1, consider the logprobs failed.
244 pub avg_logprob: f64,
245 /// Compression ratio of the segment.
246 ///
247 /// If the value is greater than 2.4, consider the compression failed.
248 pub compression_ratio: f64,
249 /// End time of the segment in seconds.
250 pub end: f64,
251 /// Probability of no speech in the segment.
252 pub no_speech_prob: f64,
253 /// Seek offset of the segment.
254 pub seek: u64,
255 /// Start time of the segment in seconds.
256 pub start: f64,
257 /// Temperature parameter used for generating the segment.
258 pub temperature: f64,
259 /// Text content of the segment.
260 pub text: String,
261 /// Array of token IDs for the text content.
262 pub tokens: Vec<u64>,
263}
264
265/// An extracted word and its corresponding timestamp.
266#[derive(Debug, Deserialize, Clone)]
267pub struct TranscriptionWord {
268 /// End time of the word in seconds.
269 pub end: f64,
270 /// Start time of the word in seconds.
271 pub start: f64,
272 /// The text content of the word.
273 pub word: String,
274}
275
276crate::impl_from_str!(TranscriptionResponse);
277
278impl Post for TranscriptionRequest {
279 #[inline]
280 fn is_streaming(&self) -> bool {
281 false
282 }
283
284 /// Builds the URL for the request.
285 ///
286 /// `base_url` should be like <https://api.openai.com/v1>
287 fn build_url(&self, base_url: &str) -> Result<String, OapiError> {
288 let mut url = Url::parse(base_url.trim_end_matches('/')).map_err(OapiError::UrlError)?;
289 url.path_segments_mut()
290 .map_err(|_| OapiError::UrlCannotBeBase(base_url.to_string()))?
291 .push("audio")
292 .push("transcriptions");
293
294 Ok(url.to_string())
295 }
296}
297
298impl PostNoStream for TranscriptionRequest {
299 type Response = TranscriptionResponse;
300
301 /// Sends a transcription POST request using multipart/form-data format,
302 /// following the field layout of the official SDK.
303 async fn get_response_string(
304 &self,
305 client: &reqwest::Client,
306 url: &str,
307 key: &str,
308 ) -> Result<String, OapiError> {
309 if !self.file.exists() {
310 return Err(OapiError::FileNotFoundError(self.file.clone()));
311 }
312
313 let content = tokio::fs::read(&self.file).await?;
314 let file_name = self
315 .file
316 .file_name()
317 .and_then(|name| name.to_str())
318 .ok_or_else(|| OapiError::ResponseError("Invalid file name".to_string()))?
319 .to_string();
320
321 let file_part = reqwest::multipart::Part::bytes(content).file_name(file_name);
322 let mut form = reqwest::multipart::Form::new().part("file", file_part);
323
324 form = form.text("model", self.model.clone());
325
326 if let Some(language) = &self.language {
327 form = form.text("language", language.clone());
328 }
329 if let Some(prompt) = &self.prompt {
330 form = form.text("prompt", prompt.clone());
331 }
332 if let Some(response_format) = self.response_format {
333 let literal = crate::audio::enum_to_literal(&response_format)?;
334 form = form.text("response_format", literal);
335 }
336 if let Some(temperature) = self.temperature {
337 form = form.text("temperature", temperature.to_string());
338 }
339 // List parameters are sent as repeated `name[]` parts, matching the
340 // official SDK.
341 if let Some(granularities) = &self.timestamp_granularities {
342 for granularity in granularities {
343 let literal = crate::audio::enum_to_literal(granularity)?;
344 form = form.text("timestamp_granularities[]", literal);
345 }
346 }
347 if let Some(include) = &self.include {
348 for item in include {
349 let literal = crate::audio::enum_to_literal(item)?;
350 form = form.text("include[]", literal);
351 }
352 }
353 if let Some(chunking_strategy) = &self.chunking_strategy {
354 let value = match chunking_strategy {
355 ChunkingStrategy::Auto => "auto".to_string(),
356 ChunkingStrategy::ServerVad(config) => {
357 let mut map = serde_json::Map::new();
358 map.insert("type".to_string(), "server_vad".into());
359 if let Some(v) = config.prefix_padding_ms {
360 map.insert("prefix_padding_ms".to_string(), v.into());
361 }
362 if let Some(v) = config.silence_duration_ms {
363 map.insert("silence_duration_ms".to_string(), v.into());
364 }
365 if let Some(v) = config.threshold {
366 map.insert("threshold".to_string(), v.into());
367 }
368 serde_json::to_string(&map).map_err(|e| {
369 OapiError::ResponseError(format!(
370 "Failed to serialize chunking_strategy: {e}"
371 ))
372 })?
373 }
374 };
375 form = form.text("chunking_strategy", value);
376 }
377
378 let response = client
379 .post(url)
380 .header("Accept", "application/json")
381 .bearer_auth(key)
382 .multipart(form)
383 .send()
384 .await?;
385
386 crate::rest::response_text_checked(response).await
387 }
388}
389
390#[cfg(test)]
391mod tests {
392 use super::*;
393
394 #[test]
395 fn test_build_url() {
396 let request = TranscriptionRequest::default();
397 let url = request.build_url("https://api.openai.com/v1/").unwrap();
398 assert_eq!(url, "https://api.openai.com/v1/audio/transcriptions");
399 }
400
401 /// Enum literals serialize to their official wire values.
402 #[test]
403 fn enum_literals() {
404 assert_eq!(
405 crate::audio::enum_to_literal(&AudioResponseFormat::VerboseJson).unwrap(),
406 "verbose_json"
407 );
408 assert_eq!(
409 crate::audio::enum_to_literal(&TimestampGranularity::Word).unwrap(),
410 "word"
411 );
412 assert_eq!(
413 crate::audio::enum_to_literal(&Include::Logprobs).unwrap(),
414 "logprobs"
415 );
416 }
417
418 /// Deserializes a `json` response (plain transcription shape).
419 ///
420 /// No accessible provider implements this endpoint, so this fixture is
421 /// NOT captured from a live response. The structure follows the schema
422 /// of openai-python `types/audio/transcription.py`; the values are
423 /// constructed for the test.
424 #[test]
425 fn parse_plain_response() {
426 let content = r#"{
427 "text": "The quick brown fox jumped over the lazy dog."
428 }"#;
429
430 let response: TranscriptionResponse = content.parse().unwrap();
431 let TranscriptionResponse::Plain(transcription) = response else {
432 panic!("expected plain transcription");
433 };
434 assert_eq!(
435 transcription.text,
436 "The quick brown fox jumped over the lazy dog."
437 );
438 assert_eq!(transcription.languages, None);
439 assert_eq!(transcription.logprobs, None);
440 assert_eq!(transcription.usage, None);
441 }
442
443 /// Deserializes a `verbose_json` response with segments, words, and
444 /// duration-billed usage.
445 ///
446 /// No accessible provider implements this endpoint, so this fixture is
447 /// NOT captured from a live response. The structure follows the schema
448 /// of openai-python `types/audio/transcription_verbose.py` +
449 /// `transcription_segment.py` + `transcription_word.py`; the values are
450 /// constructed for the test.
451 #[test]
452 fn parse_verbose_response() {
453 let content = r#"{
454 "duration": 8.47,
455 "language": "english",
456 "text": "The quick brown fox jumped over the lazy dog.",
457 "segments": [
458 {
459 "id": 0,
460 "avg_logprob": -0.2365,
461 "compression_ratio": 1.7174,
462 "end": 3.48,
463 "no_speech_prob": 0.01485,
464 "seek": 0,
465 "start": 0.0,
466 "temperature": 0.0,
467 "text": " The quick brown fox jumped over the lazy dog.",
468 "tokens": [464, 2069, 7586, 21831, 18045, 625, 262, 16931, 3290, 13]
469 }
470 ],
471 "words": [
472 {
473 "end": 0.36,
474 "start": 0.06,
475 "word": "The"
476 }
477 ],
478 "usage": {
479 "type": "duration",
480 "seconds": 8.47
481 }
482 }"#;
483
484 let response: TranscriptionResponse = content.parse().unwrap();
485 let TranscriptionResponse::Verbose(verbose) = response else {
486 panic!("expected verbose transcription");
487 };
488 assert_eq!(verbose.duration, 8.47);
489 assert_eq!(verbose.language, "english");
490 assert_eq!(
491 verbose.text,
492 "The quick brown fox jumped over the lazy dog."
493 );
494 let segments = verbose.segments.unwrap();
495 assert_eq!(segments.len(), 1);
496 assert_eq!(segments[0].id, 0);
497 assert_eq!(segments[0].start, 0.0);
498 assert_eq!(segments[0].end, 3.48);
499 assert_eq!(segments[0].tokens.len(), 10);
500 let words = verbose.words.unwrap();
501 assert_eq!(words[0].word, "The");
502 assert_eq!(words[0].start, 0.06);
503 let usage = verbose.usage.unwrap();
504 assert_eq!(usage.seconds, 8.47);
505 }
506
507 /// Deserializes a token-billed usage object (discriminated by `type`).
508 ///
509 /// No accessible provider implements this endpoint, so this fixture is
510 /// NOT captured from a live response. The structure follows the schema
511 /// of openai-python `types/audio/transcription.py` (`UsageTokens`);
512 /// the values are constructed for the test.
513 #[test]
514 fn parse_tokens_usage() {
515 let content = r#"{
516 "text": "Hello.",
517 "usage": {
518 "type": "tokens",
519 "input_tokens": 76,
520 "output_tokens": 13,
521 "total_tokens": 89,
522 "input_token_details": {
523 "audio_tokens": 76,
524 "text_tokens": 0
525 }
526 }
527 }"#;
528
529 let response: TranscriptionResponse = content.parse().unwrap();
530 let TranscriptionResponse::Plain(transcription) = response else {
531 panic!("expected plain transcription");
532 };
533 let TranscriptionUsage::Tokens {
534 input_tokens,
535 output_tokens,
536 total_tokens,
537 input_token_details,
538 } = transcription.usage.unwrap()
539 else {
540 panic!("expected tokens usage");
541 };
542 assert_eq!(input_tokens, 76);
543 assert_eq!(output_tokens, 13);
544 assert_eq!(total_tokens, 89);
545 let details = input_token_details.unwrap();
546 assert_eq!(details.audio_tokens, Some(76));
547 assert_eq!(details.text_tokens, Some(0));
548 }
549}