From b10927eee0729d1209bf6d3ec171c6280640a6e8 Mon Sep 17 00:00:00 2001 From: Roman Ernst Date: Wed, 30 Sep 2026 20:22:11 +0000 Subject: [PATCH] fix: avoid heap allocations in the steady-state echo canceller path Echo cancellation runs in real-time audio callbacks, where allocating memory can block. With AEC3 enabled, processing one 10 ms frame allocated about 40 times: - EchoRemover::process_capture created eleven per-channel working vectors per 4 ms block. They now live in a CaptureScratch struct that is allocated once and reset to the same zero/default state per block. - EchoCanceller3 built a Vec> view per sub-frame for the frame blocker (capture and render). FrameBlocker::insert_sub_frame_and_extract_block now accepts owned buffers as well, so no view is built. - AudioBuffer::copy_to_buffer resampled into a temporary vector; it now resamples straight into the destination channel. The output is bit-identical to 0.2.0 (same SHA-256 of 60 s of double-talk capture output). A new test, tests/no_allocation.rs, counts allocations in steady-state processing (40,500 before, 0 after). Co-Authored-By: Claude --- crates/sonora-aec3/src/echo_remover.rs | 115 ++++++++++++++++++++---- crates/sonora-aec3/src/frame_blocker.rs | 15 +++- crates/sonora/src/audio_buffer.rs | 6 +- crates/sonora/src/echo_canceller3.rs | 25 ++---- crates/sonora/tests/no_allocation.rs | 114 +++++++++++++++++++++++ 5 files changed, 236 insertions(+), 39 deletions(-) create mode 100644 crates/sonora/tests/no_allocation.rs diff --git a/crates/sonora-aec3/src/echo_remover.rs b/crates/sonora-aec3/src/echo_remover.rs index ec7df85..d5faee9 100644 --- a/crates/sonora-aec3/src/echo_remover.rs +++ b/crates/sonora-aec3/src/echo_remover.rs @@ -3,7 +3,7 @@ //! //! Ported from `modules/audio_processing/aec3/echo_remover.h/cc`. -use std::ptr; +use std::{fmt, mem, ptr}; use crate::aec_state::{AecState, AecStateUpdate}; use crate::aec3_fft::{Aec3Fft, Window}; @@ -151,6 +151,69 @@ pub(crate) struct EchoRemover { block_counter: usize, gain_change_hangover: i32, refined_filter_output_last_selected: bool, + scratch: CaptureScratch, +} + +/// Per-channel working buffers of [`EchoRemover::process_capture`], +/// allocated once so that processing a block does not allocate. +#[derive(Default)] +struct CaptureScratch { + e: Vec<[f32; FFT_LENGTH_BY_2]>, + y2: Vec<[f32; FFT_LENGTH_BY_2_PLUS_1]>, + e2: Vec<[f32; FFT_LENGTH_BY_2_PLUS_1]>, + r2: Vec<[f32; FFT_LENGTH_BY_2_PLUS_1]>, + r2_unbounded: Vec<[f32; FFT_LENGTH_BY_2_PLUS_1]>, + s2_linear: Vec<[f32; FFT_LENGTH_BY_2_PLUS_1]>, + y_fft: Vec, + e_fft: Vec, + comfort_noise: Vec, + high_band_comfort_noise: Vec, + subtractor_output: Vec, +} + +impl CaptureScratch { + fn new(num_capture_channels: usize) -> Self { + Self { + e: vec![[0.0; FFT_LENGTH_BY_2]; num_capture_channels], + y2: vec![[0.0; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels], + e2: vec![[0.0; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels], + r2: vec![[0.0; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels], + r2_unbounded: vec![[0.0; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels], + s2_linear: vec![[0.0; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels], + y_fft: vec![FftData::default(); num_capture_channels], + e_fft: vec![FftData::default(); num_capture_channels], + comfort_noise: vec![FftData::default(); num_capture_channels], + high_band_comfort_noise: vec![FftData::default(); num_capture_channels], + subtractor_output: (0..num_capture_channels) + .map(|_| SubtractorOutput::default()) + .collect(), + } + } + + /// Resets every buffer to the state `new` creates, without allocating. + fn reset(&mut self) { + self.e.fill([0.0; FFT_LENGTH_BY_2]); + self.y2.fill([0.0; FFT_LENGTH_BY_2_PLUS_1]); + self.e2.fill([0.0; FFT_LENGTH_BY_2_PLUS_1]); + self.r2.fill([0.0; FFT_LENGTH_BY_2_PLUS_1]); + self.r2_unbounded.fill([0.0; FFT_LENGTH_BY_2_PLUS_1]); + self.s2_linear.fill([0.0; FFT_LENGTH_BY_2_PLUS_1]); + self.y_fft.fill(FftData::default()); + self.e_fft.fill(FftData::default()); + self.comfort_noise.fill(FftData::default()); + self.high_band_comfort_noise.fill(FftData::default()); + for output in &mut self.subtractor_output { + *output = SubtractorOutput::default(); + } + } +} + +impl fmt::Debug for CaptureScratch { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.debug_struct("CaptureScratch") + .field("num_channels", &self.e.len()) + .finish_non_exhaustive() + } } impl EchoRemover { @@ -184,6 +247,7 @@ impl EchoRemover { block_counter: 0, gain_change_hangover: 0, refined_filter_output_last_selected: true, + scratch: CaptureScratch::new(num_capture_channels), } } @@ -221,20 +285,27 @@ impl EchoRemover { ); debug_assert_eq!(capture.num_channels(), num_capture_channels); - // Per-channel working storage. - let mut e = vec![[0.0f32; FFT_LENGTH_BY_2]; num_capture_channels]; - let mut y2 = vec![[0.0f32; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels]; - let mut e2 = vec![[0.0f32; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels]; - let mut r2 = vec![[0.0f32; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels]; - let mut r2_unbounded = vec![[0.0f32; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels]; - let mut s2_linear = vec![[0.0f32; FFT_LENGTH_BY_2_PLUS_1]; num_capture_channels]; - let mut y_fft = vec![FftData::default(); num_capture_channels]; - let mut e_fft = vec![FftData::default(); num_capture_channels]; - let mut comfort_noise = vec![FftData::default(); num_capture_channels]; - let mut high_band_comfort_noise = vec![FftData::default(); num_capture_channels]; - let mut subtractor_output: Vec = (0..num_capture_channels) - .map(|_| SubtractorOutput::default()) - .collect(); + // Per-channel working storage. It lives in `self.scratch` so that no + // memory is allocated per block (this runs in real-time audio + // callbacks). The buffers are moved out of `self` for the duration of + // the call (moving a `Vec` does not allocate), so they can be borrowed + // independently of the other fields, and reset to the same + // zero/default state a fresh allocation would have. + let mut scratch = mem::take(&mut self.scratch); + scratch.reset(); + let CaptureScratch { + mut e, + mut y2, + mut e2, + mut r2, + mut r2_unbounded, + mut s2_linear, + mut y_fft, + mut e_fft, + mut comfort_noise, + mut high_band_comfort_noise, + mut subtractor_output, + } = scratch; self.aec_state .update_capture_saturation(capture_signal_saturation); @@ -439,6 +510,20 @@ impl EchoRemover { // Update the metrics. self.metrics .update(&self.aec_state, &self.cng.noise_spectrum()[0], &g); + + self.scratch = CaptureScratch { + e, + y2, + e2, + r2, + r2_unbounded, + s2_linear, + y_fft, + e_fft, + comfort_noise, + high_band_comfort_noise, + subtractor_output, + }; } /// Updates the status on whether echo leakage is detected. diff --git a/crates/sonora-aec3/src/frame_blocker.rs b/crates/sonora-aec3/src/frame_blocker.rs index 11ed676..dc36937 100644 --- a/crates/sonora-aec3/src/frame_blocker.rs +++ b/crates/sonora-aec3/src/frame_blocker.rs @@ -32,19 +32,26 @@ impl FrameBlocker { /// Inserts one 80-sample sub-frame and extracts one 64-sample block. /// /// `sub_frame` is indexed as `sub_frame[band][channel]`, where each inner - /// slice has `SUB_FRAME_LENGTH` (80) samples. - pub fn insert_sub_frame_and_extract_block( + /// slice has `SUB_FRAME_LENGTH` (80) samples. It accepts both borrowed + /// views (`Vec<&[f32]>` per band) and owned buffers (`Vec>` per + /// band), so callers do not need to build a temporary view per call. + pub fn insert_sub_frame_and_extract_block( &mut self, - sub_frame: &[Vec<&[f32]>], + sub_frame: &[Band], block: &mut Block, - ) { + ) where + Band: AsRef<[Channel]>, + Channel: AsRef<[f32]>, + { debug_assert_eq!(self.num_bands, block.num_bands()); debug_assert_eq!(self.num_bands, sub_frame.len()); for (band, (buf_band, sf_band)) in self.buffer.iter_mut().zip(sub_frame.iter()).enumerate() { debug_assert_eq!(self.num_channels, block.num_channels()); + let sf_band = sf_band.as_ref(); debug_assert_eq!(self.num_channels, sf_band.len()); for (channel, (buf_ch, sf_ch)) in buf_band.iter_mut().zip(sf_band.iter()).enumerate() { + let sf_ch = sf_ch.as_ref(); debug_assert!(buf_ch.len() <= BLOCK_SIZE - 16); debug_assert_eq!(SUB_FRAME_LENGTH, sf_ch.len()); diff --git a/crates/sonora/src/audio_buffer.rs b/crates/sonora/src/audio_buffer.rs index 6cfa1e5..2839f0e 100644 --- a/crates/sonora/src/audio_buffer.rs +++ b/crates/sonora/src/audio_buffer.rs @@ -318,10 +318,10 @@ impl AudioBuffer { let buf_frames = buffer.num_frames(); if resampling_needed { for i in 0..self.num_channels { + // Resample straight into the destination channel instead of a + // temporary vector, so this real-time path does not allocate. let src = self.data.bands(i); - let mut temp = vec![0.0f32; buf_frames]; - self.output_resamplers[i].resample(src, &mut temp); - buffer.channel_mut(i)[..buf_frames].copy_from_slice(&temp); + self.output_resamplers[i].resample(src, &mut buffer.channel_mut(i)[..buf_frames]); } } else { for i in 0..self.num_channels { diff --git a/crates/sonora/src/echo_canceller3.rs b/crates/sonora/src/echo_canceller3.rs index ffd2a73..7d86350 100644 --- a/crates/sonora/src/echo_canceller3.rs +++ b/crates/sonora/src/echo_canceller3.rs @@ -560,15 +560,10 @@ impl EchoCanceller3 { &mut self.capture_sub_frame_view, ); - // Convert sub-frame slices to the format FrameBlocker expects. - let sub_frame_refs: Vec> = self - .capture_sub_frame_view - .iter() - .map(|band| band.iter().map(|ch| ch.as_slice()).collect()) - .collect(); - - self.capture_blocker - .insert_sub_frame_and_extract_block(&sub_frame_refs, &mut self.capture_block); + self.capture_blocker.insert_sub_frame_and_extract_block( + &self.capture_sub_frame_view, + &mut self.capture_block, + ); // Process through block processor. let echo_path_gain_change = level_change || aec_reference_is_downmixed_stereo; @@ -679,14 +674,10 @@ impl EchoCanceller3 { &mut self.render_sub_frame_view, ); - let sub_frame_refs: Vec> = self - .render_sub_frame_view - .iter() - .map(|band| band.iter().map(|ch| ch.as_slice()).collect()) - .collect(); - - self.render_blocker - .insert_sub_frame_and_extract_block(&sub_frame_refs, &mut self.render_block); + self.render_blocker.insert_sub_frame_and_extract_block( + &self.render_sub_frame_view, + &mut self.render_block, + ); self.block_processor.buffer_render(&self.render_block); } diff --git a/crates/sonora/tests/no_allocation.rs b/crates/sonora/tests/no_allocation.rs new file mode 100644 index 0000000..a15b56c --- /dev/null +++ b/crates/sonora/tests/no_allocation.rs @@ -0,0 +1,114 @@ +//! Echo cancellation runs in real-time audio callbacks, where allocating +//! memory can block. This test checks that steady-state processing with the +//! echo canceller enabled does not allocate. It counts allocations of the +//! current thread only, so tests running in parallel do not interfere. + +use std::alloc::{GlobalAlloc, Layout, System}; +use std::cell::Cell; + +use sonora::config::EchoCanceller; +use sonora::{AudioProcessing, Config, StreamConfig}; + +thread_local! { + static COUNTING: Cell = const { Cell::new(false) }; + static ALLOCATIONS: Cell = const { Cell::new(0) }; +} + +fn count() { + if COUNTING.with(Cell::get) { + ALLOCATIONS.with(|a| a.set(a.get() + 1)); + } +} + +struct CountingAllocator; + +// SAFETY: forwards every call unchanged to the system allocator and only +// counts calls. +unsafe impl GlobalAlloc for CountingAllocator { + unsafe fn alloc(&self, layout: Layout) -> *mut u8 { + count(); + // SAFETY: same contract as the caller's. + unsafe { System.alloc(layout) } + } + + unsafe fn dealloc(&self, ptr: *mut u8, layout: Layout) { + // SAFETY: same contract as the caller's. + unsafe { System.dealloc(ptr, layout) } + } + + unsafe fn realloc(&self, ptr: *mut u8, layout: Layout, new_size: usize) -> *mut u8 { + count(); + // SAFETY: same contract as the caller's. + unsafe { System.realloc(ptr, layout, new_size) } + } +} + +#[global_allocator] +static ALLOCATOR: CountingAllocator = CountingAllocator; + +/// Deterministic noise in [-1, 1] (xorshift64). +struct Noise(u64); + +impl Noise { + fn next(&mut self) -> f32 { + self.0 ^= self.0 << 13; + self.0 ^= self.0 >> 7; + self.0 ^= self.0 << 17; + ((self.0 >> 40) as f32 / (1u64 << 23) as f32) - 1.0 + } +} + +#[test] +fn echo_cancellation_does_not_allocate_in_steady_state() { + const RATE: u32 = 48_000; + const FRAME: usize = 480; + let stream = StreamConfig::new(RATE, 1); + let mut apm = AudioProcessing::builder() + .config(Config { + echo_canceller: Some(EchoCanceller::default()), + ..Config::default() + }) + .capture_config(stream) + .render_config(stream) + .build(); + + // Far end with pauses, an echo delayed by 60 ms and near-end talk every + // fourth second (double talk), so that all echo canceller paths run. + let frames = 1500; + let mut noise = Noise(0x1234_5678_9abc_def1); + let far: Vec = (0..frames * FRAME) + .map(|n| { + let active = ((n as f32 / RATE as f32) * 9.4).sin() > -0.3; + if active { 0.3 * noise.next() } else { 0.0 } + }) + .collect(); + let mic: Vec = (0..far.len()) + .map(|n| { + let echo = if n >= 2880 { 0.3 * far[n - 2880] } else { 0.0 }; + let near = if (n / RATE as usize) % 4 == 3 { + 0.2 * noise.next() + } else { + 0.0 + }; + echo + near + 0.001 * noise.next() + }) + .collect(); + + let mut render_out = [0.0f32; FRAME]; + let mut capture_out = [0.0f32; FRAME]; + for f in 0..frames { + // Allow lazy initialisation during the first 5 seconds. + if f == 500 { + COUNTING.with(|c| c.set(true)); + } + let range = f * FRAME..(f + 1) * FRAME; + apm.process_render_f32(&[&far[range.clone()]], &mut [&mut render_out]) + .unwrap(); + apm.process_capture_f32(&[&mic[range]], &mut [&mut capture_out]) + .unwrap(); + let _ = apm.statistics(); + } + COUNTING.with(|c| c.set(false)); + + assert_eq!(ALLOCATIONS.with(Cell::get), 0); +}