pub mod audio;
pub mod backend;
pub mod benchmark;
pub mod bessel;
pub mod channel_layout;
pub mod decode;
pub mod denoiser;
pub mod encode;
pub mod fft;
pub mod gain;
#[cfg(feature = "live")]
pub mod live;
pub mod loudness;
pub mod metadata;
pub mod models;
pub mod noise;
pub mod perceptual;
pub mod postfilter;
pub mod resample;
pub mod service;
pub mod stft;
pub mod stream;
pub mod vad;
pub mod window;
pub use audio::{
ensure_memory_limit, estimate_audio_memory_bytes, estimate_audio_working_set_bytes,
estimate_file_memory_bytes, estimate_stream_memory_bytes, read_audio, read_wav, read_wav_bytes,
sanitize_sample, write_audio, write_wav, write_wav_bytes, write_wav_channel_mask, Audio,
WavStreamReader, WavStreamWriter,
};
pub use backend::{
decode_mid_side, encode_mid_side, Backend, BackendOptions, ChannelMode, OnnxModelConfig,
SgmseProfile,
};
pub use channel_layout::{ChannelLayout, ChannelMask, ChannelPosition, PanInfo};
pub use decode::{decode_file, AudioFormat, DecodedPcm};
pub use denoiser::{Denoiser, DenoiserConfig, Preset, ProcessingMode, StreamingDenoiser};
pub use encode::{AacEncoder, DownmixMode, EncodeOptions, OutputFormat};
pub use gain::{Algorithm, SpecSubLaw};
pub use window::{WindowParams, WindowType};
pub fn denoise_file<P1, P2>(input: P1, output: P2, config: DenoiserConfig) -> Result<Audio, String>
where
P1: AsRef<std::path::Path>,
P2: AsRef<std::path::Path>,
{
denoise_file_with_backend(input, output, config, Backend::Classical)
}
pub fn denoise_file_with_backend<P1, P2>(
input: P1,
output: P2,
config: DenoiserConfig,
backend: Backend,
) -> Result<Audio, String>
where
P1: AsRef<std::path::Path>,
P2: AsRef<std::path::Path>,
{
denoise_file_with_backend_opts(input, output, config, backend, EncodeOptions::default())
}
pub fn denoise_file_with_backend_opts<P1, P2>(
input: P1,
output: P2,
config: DenoiserConfig,
backend: Backend,
encode_opts: EncodeOptions,
) -> Result<Audio, String>
where
P1: AsRef<std::path::Path>,
P2: AsRef<std::path::Path>,
{
denoise_file_with_backend_config(
input,
output,
config,
backend,
encode_opts,
BackendOptions::default(),
)
}
pub fn denoise_file_with_backend_config<P1, P2>(
input: P1,
output: P2,
config: DenoiserConfig,
backend: Backend,
encode_opts: EncodeOptions,
backend_options: BackendOptions,
) -> Result<Audio, String>
where
P1: AsRef<std::path::Path>,
P2: AsRef<std::path::Path>,
{
let input = input.as_ref();
let output = output.as_ref();
let metadata = metadata::read_extended(input)?;
let mut audio = read_audio(input)?;
denoise_audio_with_backend_config(&mut audio, config, backend, &backend_options)?;
write_audio(output, &audio, encode_opts)?;
if let Some(metadata) = metadata {
metadata::write_extended(metadata, output)?;
}
Ok(audio)
}
pub fn denoise_audio_with_backend_config(
audio: &mut Audio,
mut config: DenoiserConfig,
backend: Backend,
backend_options: &BackendOptions,
) -> Result<std::time::Duration, String> {
config.sample_rate = audio.sample_rate;
audio.sanitize_samples();
let t0 = std::time::Instant::now();
audio.channels = if config.vad {
process_with_vad(
backend,
&audio.channels,
audio.sample_rate,
&config,
backend_options,
)?
} else {
backend::process_channels(
backend,
&audio.channels,
audio.sample_rate,
&config,
backend_options,
)?
};
audio.sanitize_samples();
let elapsed = t0.elapsed();
eprintln!(
"denoize: {:?} | {}ch x {} frames ({:.2}s) in {:.2?} ({:.1}x realtime)",
backend,
audio.channels(),
audio.frames(),
audio.frames() as f64 / audio.sample_rate as f64,
elapsed,
(audio.frames() as f64 / audio.sample_rate as f64) / elapsed.as_secs_f64().max(1e-9),
);
Ok(elapsed)
}
fn process_with_vad(
backend: Backend,
channels: &[Vec<f64>],
sample_rate: u32,
config: &DenoiserConfig,
backend_options: &BackendOptions,
) -> Result<Vec<Vec<f64>>, String> {
let regions = vad::speech_regions(channels, sample_rate);
let fade_frames = (sample_rate as usize / 50).max(1); let silence_gain = config.vad_silence_gain;
let speech_mix = config.vad_speech_mix;
let mut output: Vec<Vec<f64>> = channels
.iter()
.map(|channel| channel.iter().map(|sample| sample * silence_gain).collect())
.collect();
for region in regions {
let input: Vec<Vec<f64>> = channels
.iter()
.map(|channel| {
channel[region.start.min(channel.len())..region.end.min(channel.len())].to_vec()
})
.collect();
let enhanced =
backend::process_channels(backend, &input, sample_rate, config, backend_options)?;
for (channel_index, enhanced_channel) in enhanced.iter().enumerate() {
let Some(destination) = output.get_mut(channel_index) else {
continue;
};
let original = &channels[channel_index];
for (offset, sample) in enhanced_channel.iter().enumerate() {
let index = region.start + offset;
if index >= destination.len() || index >= original.len() || index >= region.end {
break;
}
let target = sample * speech_mix + original[index] * (1.0 - speech_mix);
let weight = vad_mix_weight(offset, region.end - region.start, fade_frames);
destination[index] = destination[index] * (1.0 - weight) + target * weight;
}
}
}
Ok(output)
}
fn vad_mix_weight(offset: usize, length: usize, fade_frames: usize) -> f64 {
let from_start = offset.min(fade_frames) as f64 / fade_frames.max(1) as f64;
let from_end =
length.saturating_sub(offset + 1).min(fade_frames) as f64 / fade_frames.max(1) as f64;
from_start.min(from_end).clamp(0.0, 1.0)
}
#[cfg(test)]
mod vad_mix_tests {
use super::vad_mix_weight;
#[test]
fn fades_vad_region_edges_without_exceeding_unity() {
assert_eq!(vad_mix_weight(0, 100, 10), 0.0);
assert_eq!(vad_mix_weight(99, 100, 10), 0.0);
assert_eq!(vad_mix_weight(50, 100, 10), 1.0);
assert!((vad_mix_weight(5, 100, 10) - 0.5).abs() < f64::EPSILON);
}
}
#[cfg(test)]
mod input_safety_tests {
use super::*;
#[test]
fn high_level_processing_sanitizes_nonfinite_samples_and_keeps_empty_audio_safe() {
let mut audio = Audio {
sample_rate: 16_000,
channels: vec![vec![f64::NAN, f64::INFINITY, -f64::INFINITY, 2.0, -2.0]],
bits_per_sample: 32,
sample_format: hound::SampleFormat::Float,
channel_mask: None,
};
denoise_audio_with_backend_config(
&mut audio,
DenoiserConfig::default(16_000),
Backend::Classical,
&BackendOptions::default(),
)
.unwrap();
assert!(audio.channels[0].iter().all(|sample| sample.is_finite()));
assert!(audio.channels[0].iter().all(|sample| sample.abs() <= 1.0));
let mut empty = Audio {
sample_rate: 16_000,
channels: vec![Vec::new()],
bits_per_sample: 32,
sample_format: hound::SampleFormat::Float,
channel_mask: None,
};
denoise_audio_with_backend_config(
&mut empty,
DenoiserConfig::default(16_000),
Backend::Classical,
&BackendOptions::default(),
)
.unwrap();
assert_eq!(empty.frames(), 0);
}
}