Skip to main content

rustynes_apu/
blip.rs

1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the band-limited (BLEP) synthesis is derived from blip_buf by Shay Green (Blargg), LGPL-2.1-or-later (GPLv3-compatible). See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! Band-limited synthesis for the APU's audio output.
6//!
7//! # What this is
8//!
9//! A streaming **BLEP (Band-Limited Step) decimator** that takes the
10//! per-CPU-cycle mixer output and produces band-limited samples at the
11//! host audio rate (default 44.1 kHz).
12//!
13//! The technique is band-limited step (BLEP) synthesis — the same general
14//! approach popularized by Shay Green's `blip_buf` and used by many emulators.
15//! Provenance: the band-limited-step technique is **derived from Shay Green's
16//! `blip_buf`** (LGPL-2.1-or-later, which is GPLv3-compatible); our polyphase
17//! kernel in [`crate::blip_kernel`] uses a finer 32-phase resolution than
18//! `blip_buf`. See NOTICE and docs/originality-and-provenance.md (Section 1):
19//!
20//! - Pre-compute a polyphase windowed-sinc kernel ([`crate::blip_kernel`])
21//!   keyed by `PHASES = 32` sub-output-sample fractional offsets, with
22//!   `TAPS = 32` coefficients per row giving the FIR a ±16-output-sample
23//!   reach.
24//! - For each CPU-rate input, compute the amplitude **delta** from the
25//!   previous value.
26//! - Position the delta at its fractional output-sample location and add
27//!   `delta * kernel[phase][m]` to `TAPS` positions of a **host-rate
28//!   delta buffer**.
29//! - Integrate the delta buffer (running cumulative sum) to recover the
30//!   reconstructed signal, then push through the existing analog
31//!   [`crate::mixer::FilterChain`] (90 Hz HPF + 440 Hz HPF + 14 kHz LPF).
32//!
33//! This is the textbook BLEP structure: each abrupt step in the input is
34//! replaced by the band-limited equivalent (`sinc * window`-shaped step
35//! response), eliminating the alias products that a naive sample-and-
36//! hold decimator leaves above Nyquist.
37//!
38//! # Why this replaces the previous ratio-counter decimator
39//!
40//! The pre-v0.9.x decimator was a ratio-counter with sample-and-hold
41//! reconstruction — functionally a rectangular-window decimator that
42//! left aliased energy above Nyquist (~22.05 kHz at 44.1 kHz host rate).
43//! The existing 14 kHz LPF in the [`crate::mixer::FilterChain`] suppressed
44//! the audible portion, but the band beyond 14 kHz contained alias-
45//! pumping products that the analog filter could not fully kill. With
46//! the v0.9.x mapper-audio extensions (VRC6 / Sunsoft 5B / Namco 163 /
47//! MMC5) adding wavetable + envelope-modulated FM-like complexity, the
48//! alias floor started clearing the LPF's stopband attenuation in spots.
49//!
50//! The polyphase BLEP decimator pushes the alias rejection well below
51//! -60 dB across the audible band even for input frequencies above the
52//! host Nyquist (verified by the spectral FFT regression test in
53//! `tests/spectral.rs`).
54//!
55//! # API contract
56//!
57//! Unchanged from the previous decimator:
58//!
59//! - [`BlipBuf::new(sample_rate, cpu_rate)`] — same signature.
60//! - [`BlipBuf::add_sample(value)`] — called once per CPU cycle.
61//! - [`BlipBuf::drain`], [`BlipBuf::drain_all`], [`BlipBuf::len`],
62//!   [`BlipBuf::is_empty`], [`BlipBuf::reset`] — same semantics.
63//!
64//! Save-state compat: the snapshot module reads/writes `sample_rate`,
65//! `cpu_rate`, `phase`, `filter`, and `held_value`. All five are
66//! preserved on this rewrite. The internal delta ring is NOT serialized
67//! — same intentional behavior as before (the ring's contents are sub-
68//! audio-sample-window detail; a fresh restored state begins emitting
69//! samples as soon as `tick()` runs forward).
70//!
71//! # Determinism
72//!
73//! All math is `f32` with a fixed operation order. The kernel is pre-
74//! computed bit-identically across builds (see
75//! [`crate::blip_kernel::Kernel`]). No allocations on the hot path.
76
77#[cfg(test)]
78use crate::blip_kernel::PHASES;
79use crate::blip_kernel::{Kernel, TAPS};
80use crate::mixer::FilterChain;
81use alloc::vec::Vec;
82
83/// The most capacity [`BlipBuf::drain_all`] keeps for the next frame, in
84/// samples (v2.7.5). A frame is about 800 samples at 48 kHz, so four frames'
85/// worth keeps the per-frame benefit at any common output rate while a caller
86/// that once let audio pile up (the probe engine's undrained trials, ~24,000
87/// samples) does not leave a buffer that size resident for good.
88const DRAIN_ALL_KEEP_CAPACITY: usize = 4096;
89
90/// CPU cycles per second, NTSC.
91pub const CPU_HZ_NTSC: f64 = 1_789_773.0;
92/// CPU cycles per second, PAL (slightly slower).
93pub const CPU_HZ_PAL: f64 = 1_662_607.0;
94
95/// Size of the host-rate delta ring buffer. Must be a power of two for
96/// the modulo via bitmask. Held large enough to comfortably absorb a
97/// burst of pending output samples plus the kernel's TAPS-sample reach.
98/// 4096 samples ≈ 93 ms of audio at 44.1 kHz — far more than any single
99/// frame's worth of output (~735 samples).
100const RING_SIZE: usize = 4096;
101const RING_MASK: usize = RING_SIZE - 1;
102
103/// Streaming BLEP decimator + filter chain that feeds host-rate samples
104/// to the frontend's audio thread.
105#[derive(Debug, Clone)]
106pub struct BlipBuf {
107    /// Host sample rate (Hz).
108    pub(crate) sample_rate: u32,
109    /// CPU rate (Hz, fractional).
110    pub(crate) cpu_rate: f64,
111    /// Fractional output-sample position. Advances by `step = sample_rate
112    /// / cpu_rate` per input sample. Whenever the integer part advances,
113    /// one host-rate output sample is "ready" (its delta contributions
114    /// have been written for at least `TAPS/2` future samples ahead, so
115    /// it's safe to read out).
116    pub(crate) phase: f64,
117    /// Step per input sample (`sample_rate / cpu_rate`).
118    step: f64,
119    /// Output filter chain (90 Hz HPF + 440 Hz HPF + 14 kHz LPF).
120    pub(crate) filter: FilterChain,
121    /// Output ring of finalized post-filter samples awaiting drain.
122    pub(crate) samples: Vec<f32>,
123    /// Most recent input value handed to [`Self::add_sample`]. The next
124    /// call's delta is `value - held_value`.
125    pub(crate) held_value: f32,
126    /// Pre-computed polyphase windowed-sinc kernel.
127    kernel: Kernel,
128    /// Host-rate delta ring buffer. Each input delta scatters its
129    /// `kernel * delta` contributions across `TAPS` positions of this
130    /// buffer; the buffer is then integrated (cumulative sum) to
131    /// recover the actual sample stream.
132    delta_ring: [f32; RING_SIZE],
133    /// Integer output-sample index of the **current** input's scatter
134    /// center. Each scatter writes to ring positions
135    /// `[head - TAPS/2, head + TAPS/2)`. The next emit reads at
136    /// `head - TAPS/2 - 1` (one past the leftmost scatter slot).
137    /// Wraps within `RING_SIZE` via `& RING_MASK`.
138    head: usize,
139    /// True once `head` has advanced far enough that the leftmost
140    /// scatter slot `head - TAPS/2` has settled (no future input can
141    /// write to it). Output emission gated on this flag — for the very
142    /// first `TAPS/2 + 1` inputs, the integrator is just warming up
143    /// and no samples are emitted yet. This costs one frame of startup
144    /// latency (~16 samples = 0.36 ms @ 44.1 kHz, well below human-
145    /// perceptible).
146    primed: bool,
147    /// Running integrator state (cumulative sum of consumed delta-ring
148    /// entries since reset). Converts the scattered delta stream back
149    /// into an absolute-amplitude sample stream.
150    integrator: f32,
151}
152
153impl BlipBuf {
154    /// Create a new band-limited buffer.
155    ///
156    /// `sample_rate` is the host audio rate in Hz (typically 44 100).
157    /// `cpu_rate` is the NES CPU rate in Hz ([`CPU_HZ_NTSC`] or
158    /// [`CPU_HZ_PAL`]).
159    #[must_use]
160    pub fn new(sample_rate: u32, cpu_rate: f64) -> Self {
161        let step = f64::from(sample_rate) / cpu_rate;
162        let mut b = Self {
163            sample_rate,
164            cpu_rate,
165            phase: 0.0,
166            step,
167            filter: FilterChain::new(sample_rate),
168            samples: Vec::with_capacity(8192),
169            held_value: 0.0,
170            kernel: Kernel::new(),
171            delta_ring: [0.0; RING_SIZE],
172            // Start `head` at TAPS so the initial scatter (writing to
173            // `head - TAPS/2 .. head + TAPS/2`) lands at indices
174            // `[TAPS/2, 3*TAPS/2)` and never goes negative.
175            head: TAPS,
176            primed: false,
177            integrator: 0.0,
178        };
179        b.reset();
180        b
181    }
182
183    /// v2.1.3 — swap the analog output-filter model (see
184    /// [`crate::mixer::FilterModel`]), rebuilding the filter chain at the
185    /// current sample rate. This resets the filter's IIR state, so switching
186    /// while audio is playing (the Settings selector applies it live) produces a
187    /// brief transient — as with any live filter swap. The frontend also applies
188    /// it at ROM load / power-cycle, where there is no audible discontinuity.
189    pub fn set_filter_model(&mut self, model: crate::mixer::FilterModel) {
190        self.filter = crate::mixer::FilterChain::for_model(self.sample_rate, model);
191    }
192
193    /// Reset to silence. Empties the input ring, the pending output
194    /// queue, and the filter chain state.
195    pub fn reset(&mut self) {
196        self.phase = 0.0;
197        self.filter.reset();
198        self.samples.clear();
199        self.held_value = 0.0;
200        self.delta_ring = [0.0; RING_SIZE];
201        self.head = TAPS;
202        self.primed = false;
203        self.integrator = 0.0;
204    }
205
206    /// The resampler state a save state must carry for a restore to resume
207    /// the exact sample stream (v2.9.9, NL-12): the head's ring position,
208    /// whether warm-up is over, the integrator, and the `TAPS` delta-ring
209    /// slots a scatter can still reach. Every other slot is zero: an emitted
210    /// slot is zeroed as it is consumed, and no scatter writes outside
211    /// `[head - TAPS/2, head + TAPS/2)`.
212    pub(crate) fn live_state(&self) -> (u16, bool, f32, [f32; TAPS]) {
213        let mut window = [0.0; TAPS];
214        let start = self.head.wrapping_sub(TAPS / 2);
215        for (i, v) in window.iter_mut().enumerate() {
216            *v = self.delta_ring[start.wrapping_add(i) & RING_MASK];
217        }
218        // RING_SIZE is 4096, so the masked head fits a u16.
219        #[allow(clippy::cast_possible_truncation)]
220        let head = (self.head & RING_MASK) as u16;
221        (head, self.primed, self.integrator, window)
222    }
223
224    /// Install a state [`Self::live_state`] produced. The caller validates
225    /// the values; `head` is masked into the ring here.
226    pub(crate) fn set_live_state(
227        &mut self,
228        head: u16,
229        primed: bool,
230        integrator: f32,
231        window: &[f32; TAPS],
232    ) {
233        self.delta_ring = [0.0; RING_SIZE];
234        self.head = usize::from(head) & RING_MASK;
235        self.primed = primed;
236        self.integrator = integrator;
237        let start = self.head.wrapping_sub(TAPS / 2);
238        for (i, &v) in window.iter().enumerate() {
239            self.delta_ring[start.wrapping_add(i) & RING_MASK] = v;
240        }
241    }
242
243    /// Add one mixed sample at CPU resolution. The buffer accumulates
244    /// host-rate samples internally; drain via [`Self::drain`] or
245    /// [`Self::drain_all`].
246    ///
247    /// The mixer's output is approximately in `[-0.5, 0.5]` for the 2A03
248    /// alone, with on-cart audio expansions pushing the absolute range
249    /// up somewhat. We clamp `value` defensively so a stray NaN/Inf or
250    /// runaway mapper can't propagate non-finite values into the FIR
251    /// state. The clamp range is far outside any sane mixer output, so
252    /// the gate is purely a saturation backstop, not a normal-operation
253    /// limiter.
254    #[inline]
255    pub fn add_sample(&mut self, value: f32) {
256        // Defensive saturation. Real mixer output never exceeds ~1.5; the
257        // clamp catches NaN/Inf from a runaway mapper-audio path so the
258        // FIR can't propagate non-finite values through the ring.
259        let value = if value.is_finite() {
260            value.clamp(-4.0, 4.0)
261        } else {
262            0.0
263        };
264        let delta = value - self.held_value;
265        self.held_value = value;
266
267        // Scatter the delta into the host-rate ring via the kernel row
268        // matching the input's fractional output-sample position.
269        //
270        // `self.phase` ∈ [0, 1) is the fractional output-sample position
271        // of THIS input within the current output-grid interval. The
272        // kernel row matching this fractional offset spreads the delta
273        // across `TAPS` output samples centered at `head`. The center
274        // sits at `head + TAPS/2 - 1` (the right-half tap that's closest
275        // to the input's position); we walk the kernel from `head` for
276        // `TAPS` slots.
277        if delta != 0.0 {
278            #[allow(clippy::cast_possible_truncation)]
279            let row = self.kernel.row(self.phase as f32);
280            // Center the scatter at `head` (the current integer output
281            // index). Tap 0 lands at `head - TAPS/2`; tap TAPS-1 lands
282            // at `head + TAPS/2 - 1`. The kernel itself encodes the
283            // sub-sample fractional shift via the row index.
284            //
285            // v2.8.0 Phase 4b — the 32-tap window wraps the ring at most
286            // once, so split it into (at most) two CONTIGUOUS runs instead
287            // of masking every index: each run is a plain SAXPY LLVM
288            // auto-vectorizes (SSE2/NEON/wasm-simd), which the per-tap
289            // `& RING_MASK` form structurally prevented. Per-slot math is
290            // unchanged (`slot += delta * coeff`, one touch per slot, mul
291            // then add — no FMA contraction), so output is bit-identical
292            // to the scalar form.
293            let start = self.head.wrapping_sub(TAPS / 2) & RING_MASK;
294            let first = (RING_SIZE - start).min(TAPS);
295            for (slot, &coeff) in self.delta_ring[start..start + first]
296                .iter_mut()
297                .zip(&row[..first])
298            {
299                *slot += delta * coeff;
300            }
301            for (slot, &coeff) in self.delta_ring[..TAPS - first]
302                .iter_mut()
303                .zip(&row[first..])
304            {
305                *slot += delta * coeff;
306            }
307        }
308
309        // Advance phase. When it crosses 1.0, advance `head` and
310        // emit the output sample that just fell out of the scatter
311        // window (one slot to the left of `head - TAPS/2`).
312        self.phase += self.step;
313        while self.phase >= 1.0 {
314            self.phase -= 1.0;
315            self.head = self.head.wrapping_add(1);
316
317            // The slot at `head - TAPS/2 - 1` is now permanently
318            // finalized: any future scatter writes to `[new_head -
319            // TAPS/2, new_head + TAPS/2)`, which doesn't reach back
320            // this far. Emit it.
321            //
322            // During warm-up (the first TAPS/2 + 1 head advances), the
323            // initial slots haven't been written by any scatter yet
324            // (they're still zero from construction); we burn through
325            // them silently to flush the startup phase, then start
326            // emitting once `primed` is set.
327            let emit_idx = self.head.wrapping_sub(TAPS / 2 + 1) & RING_MASK;
328            self.integrator += self.delta_ring[emit_idx];
329            self.delta_ring[emit_idx] = 0.0;
330
331            if !self.primed {
332                // Have we advanced past the initial dead zone? The
333                // first scatter wrote at `head_initial = TAPS` (i.e.,
334                // covering `[TAPS/2, 3*TAPS/2)`). We start emitting
335                // once `head >= TAPS + TAPS/2 + 1`, i.e., the emit
336                // slot has caught up with the first scatter's right
337                // edge.
338                if self.head > TAPS + TAPS / 2 {
339                    self.primed = true;
340                }
341                continue;
342            }
343
344            let filtered = self.filter.process(self.integrator);
345            self.samples.push(filtered);
346        }
347    }
348
349    /// Drain finalized samples into `out`. Returns the number written.
350    /// Excess pending samples are kept; if `out.len() < self.samples.len()`
351    /// the remainder waits for the next drain.
352    pub fn drain(&mut self, out: &mut [f32]) -> usize {
353        let n = self.samples.len().min(out.len());
354        out[..n].copy_from_slice(&self.samples[..n]);
355        self.samples.drain(..n);
356        n
357    }
358
359    /// Drain all finalized samples into a new `Vec`.
360    ///
361    /// The filled buffer goes to the caller, and a fresh one with the same
362    /// capacity takes its place (v2.7.5, core audit IMP-07). A buffer returned
363    /// by value must be replaced every call whatever happens; the choice is
364    /// only whether its replacement starts empty. With `mem::take` it started
365    /// at capacity zero and regrew by doubling through the next frame's pushes
366    /// (about nine reallocations for a frame's ~800 samples); now it is one
367    /// allocation of the size the last frame needed, up to
368    /// `DRAIN_ALL_KEEP_CAPACITY` samples. Measured -0.89% frame
369    /// time on a steady-state frame-then-drain workload, reproduced on two
370    /// runs (`docs/performance.md` §v2.7.5). Callers that drain into their own
371    /// buffer with [`Self::drain`] allocate nothing and are unaffected. The
372    /// samples themselves are untouched, so the output is byte-identical.
373    #[must_use]
374    pub fn drain_all(&mut self) -> Vec<f32> {
375        // The `core` path keeps this module portable to `#![no_std]` builds.
376        // See `docs/architecture.md` §no_std boundary.
377        // Nothing queued: hand back an unallocated `Vec` and keep the buffer,
378        // rather than allocate a replacement for an empty one.
379        if self.samples.is_empty() {
380            return Vec::new();
381        }
382        // Clamped so a one-off backlog is not kept as a high-water mark
383        // (see `DRAIN_ALL_KEEP_CAPACITY`); a per-frame drain is far below it.
384        let capacity = self.samples.capacity().min(DRAIN_ALL_KEEP_CAPACITY);
385        core::mem::replace(&mut self.samples, Vec::with_capacity(capacity))
386    }
387
388    /// Number of samples currently buffered, awaiting drain.
389    #[must_use]
390    pub fn len(&self) -> usize {
391        self.samples.len()
392    }
393
394    /// Whether the pending-output queue is empty.
395    #[must_use]
396    pub fn is_empty(&self) -> bool {
397        self.samples.is_empty()
398    }
399}
400
401#[cfg(test)]
402mod tests {
403    use super::*;
404
405    /// Total CPU cycles in 1 second of NTSC playback. Used as a "long-
406    /// run" fixture for the determinism / DC tests.
407    const ONE_SECOND_NTSC: usize = 1_789_773;
408
409    /// 10 frames at 60 Hz (used by spectral tests + decay tests).
410    const TEN_FRAMES_NTSC: usize = ONE_SECOND_NTSC / 6;
411
412    /// v2.7.5 (core audit IMP-07): `drain_all` hands the filled buffer to the
413    /// caller and keeps a same-capacity one for the next frame. With
414    /// `mem::take` the buffer restarted at capacity zero and regrew by doubling
415    /// through every frame's pushes (about nine reallocations for a frame's
416    /// ~800 samples). Measured on a steady-state frame-then-drain workload:
417    /// -0.89% frame time on two runs (docs/performance.md §v2.7.5).
418    #[test]
419    fn drain_all_keeps_room_for_the_next_frame() {
420        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
421        for _ in 0..TEN_FRAMES_NTSC / 10 {
422            b.add_sample(0.25);
423        }
424        let first = b.drain_all();
425        assert!(!first.is_empty(), "a frame produces samples");
426        assert!(
427            b.samples.capacity() >= first.len(),
428            "the next frame must not start from capacity zero: {} < {}",
429            b.samples.capacity(),
430            first.len()
431        );
432        assert!(b.is_empty(), "the drained samples are gone");
433    }
434
435    /// An empty drain hands back an empty, unallocated `Vec` and keeps the
436    /// buffer as it is (agy on #553): polling with nothing queued used to
437    /// allocate a replacement and return the old, empty buffer to be freed.
438    #[test]
439    fn an_empty_drain_all_allocates_nothing() {
440        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
441        for _ in 0..TEN_FRAMES_NTSC / 10 {
442            b.add_sample(0.25);
443        }
444        let _ = b.drain_all();
445        let kept = b.samples.capacity();
446        let empty = b.drain_all();
447        assert_eq!(empty, [] as [f32; 0]);
448        assert_eq!(empty.capacity(), 0, "no buffer handed out for nothing");
449        assert_eq!(b.samples.capacity(), kept, "the kept buffer stays");
450    }
451
452    /// The kept capacity is clamped (agy on #553): a caller that lets audio
453    /// pile up once -- the probe engine's undrained trials, ~24,000 samples --
454    /// must not leave a buffer that size resident for good. The per-frame case
455    /// (~800 samples) is far below the clamp, so it keeps its full benefit.
456    #[test]
457    fn drain_all_does_not_keep_a_high_water_mark() {
458        let mut b = BlipBuf::new(48_000, CPU_HZ_NTSC);
459        for _ in 0..ONE_SECOND_NTSC {
460            b.add_sample(0.25);
461        }
462        let backlog = b.drain_all();
463        assert!(
464            backlog.len() > DRAIN_ALL_KEEP_CAPACITY,
465            "a long undrained run"
466        );
467        assert!(
468            b.samples.capacity() <= DRAIN_ALL_KEEP_CAPACITY,
469            "kept {} after a {}-sample backlog",
470            b.samples.capacity(),
471            backlog.len()
472        );
473    }
474
475    #[test]
476    fn empty_buffer_reads_zero_samples() {
477        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
478        let mut out = [0.0_f32; 16];
479        assert_eq!(b.drain(&mut out), 0);
480        assert!(b.is_empty());
481        assert_eq!(b.len(), 0);
482    }
483
484    #[test]
485    fn emits_expected_sample_count() {
486        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
487        for _ in 0..ONE_SECOND_NTSC {
488            b.add_sample(0.0);
489        }
490        let drained = b.drain_all();
491        let n = drained.len();
492        // Expected: 44_100 samples ± a small warm-up loss. The BLEP
493        // structure delays output by `TAPS/2 + 1` host samples after
494        // construction so each emitted sample has received its full
495        // kernel scatter — that's ~17 samples of startup delay at our
496        // TAPS=32 configuration. After 1 s, output count is
497        // `44_100 - 17 ± rounding`.
498        let max_loss = TAPS / 2 + 4;
499        assert!(
500            n <= 44_100 && n + max_loss >= 44_100,
501            "expected 44100 - {max_loss}..=44100 samples in 1 s, got {n}"
502        );
503    }
504
505    #[test]
506    fn dc_passes_through_with_settled_gain() {
507        // A constant 0.5 input. The FIR scatters NO delta (the input
508        // never changes after the first sample), so the integrator
509        // converges to 0.5. The HPFs then attenuate the DC; after
510        // ~200 000 samples the output is near zero.
511        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
512        for _ in 0..200_000 {
513            b.add_sample(0.5);
514        }
515        let drained = b.drain_all();
516        let last = *drained.last().unwrap();
517        assert!(last.abs() < 0.05, "DC not removed; last sample = {last}");
518    }
519
520    #[test]
521    fn drain_empties_buffer() {
522        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
523        for _ in 0..1000 {
524            b.add_sample(0.0);
525        }
526        let _ = b.drain_all();
527        assert!(b.is_empty());
528    }
529
530    #[test]
531    fn single_delta_produces_band_limited_step() {
532        // Feed silence, then a unit step. The output should be a
533        // band-limited ramp (sinc-ringing) reaching the new amplitude
534        // over ~TAPS host-rate samples.
535        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
536        for _ in 0..10_000 {
537            b.add_sample(0.0);
538        }
539        for _ in 0..10_000 {
540            b.add_sample(1.0);
541        }
542        // Push enough samples to fully settle the FIR + HPF response.
543        for _ in 0..50_000 {
544            b.add_sample(1.0);
545        }
546        let drained = b.drain_all();
547        // After 50_000+ samples of constant 1.0, the HPFs drive output
548        // back to ~0 (DC blocked).
549        let last = *drained.last().unwrap();
550        assert!(
551            last.abs() < 0.05,
552            "constant input not DC-blocked: last = {last}"
553        );
554        // And the step transient produced finite peak < the saturation
555        // clip (1.5 is our cap before clamp).
556        let max_abs = drained.iter().map(|s| s.abs()).fold(0.0_f32, f32::max);
557        assert!(
558            max_abs > 0.0 && max_abs < 1.5,
559            "step transient max = {max_abs}, expected (0, 1.5)"
560        );
561    }
562
563    #[test]
564    fn opposing_deltas_cancel_to_dc() {
565        // Equal amounts of +1 and -1 in long runs, end with HPF-blocked
566        // output. The integrator must NOT drift.
567        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
568        for _ in 0..100_000 {
569            b.add_sample(1.0);
570        }
571        for _ in 0..100_000 {
572            b.add_sample(-1.0);
573        }
574        let drained = b.drain_all();
575        let last = *drained.last().unwrap();
576        assert!(last.abs() < 0.05, "didn't cancel; last = {last}");
577    }
578
579    #[test]
580    fn deterministic_across_runs() {
581        // Same input sequence produces bit-identical output.
582        let drive = |b: &mut BlipBuf| {
583            for i in 0..TEN_FRAMES_NTSC {
584                #[allow(clippy::cast_precision_loss)]
585                let v = ((i as f32) * 0.0001).sin() * 0.5;
586                b.add_sample(v);
587            }
588        };
589        let mut a = BlipBuf::new(44_100, CPU_HZ_NTSC);
590        drive(&mut a);
591        let av = a.drain_all();
592        let mut c = BlipBuf::new(44_100, CPU_HZ_NTSC);
593        drive(&mut c);
594        let cv = c.drain_all();
595        assert_eq!(av.len(), cv.len(), "same input, different output length");
596        for (i, (x, y)) in av.iter().zip(cv.iter()).enumerate() {
597            assert_eq!(
598                x.to_bits(),
599                y.to_bits(),
600                "non-determinism at index {i}: {x} vs {y}"
601            );
602        }
603    }
604
605    #[test]
606    fn saturation_clips_extreme_values() {
607        // A pathological NaN / Inf input must not poison the output ring.
608        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
609        b.add_sample(f32::NAN);
610        b.add_sample(f32::INFINITY);
611        b.add_sample(f32::NEG_INFINITY);
612        for _ in 0..200 {
613            b.add_sample(1.0e20);
614        }
615        // No panics, drained samples are all finite.
616        let drained = b.drain_all();
617        for v in &drained {
618            assert!(v.is_finite(), "output poisoned by extreme input: {v}");
619        }
620        assert!(b.held_value.is_finite());
621    }
622
623    #[test]
624    fn reset_clears_all_state() {
625        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
626        for _ in 0..1000 {
627            b.add_sample(0.5);
628        }
629        b.reset();
630        assert_eq!(b.phase, 0.0);
631        assert_eq!(b.held_value, 0.0);
632        assert!(b.is_empty());
633        for v in &b.delta_ring {
634            assert_eq!(*v, 0.0);
635        }
636        assert_eq!(b.integrator, 0.0);
637    }
638
639    #[test]
640    fn drain_slice_handles_partial_read() {
641        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
642        for _ in 0..10_000 {
643            b.add_sample(0.0);
644        }
645        let queued = b.len();
646        assert!(queued > 0);
647        let mut out = [0.0_f32; 16];
648        let n = b.drain(&mut out);
649        assert_eq!(n, 16);
650        assert_eq!(b.len(), queued - 16);
651    }
652
653    #[test]
654    fn long_run_produces_finite_output() {
655        // Sweep amplitudes so the FIR convolution + integrator path is
656        // exercised. All drained samples must be finite (no NaN/Inf
657        // leakage from the kernel-row lookup or integrator drift).
658        let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
659        for i in 0..(ONE_SECOND_NTSC / 100) {
660            #[allow(clippy::cast_precision_loss)]
661            let v = ((i % 100) as f32) / 100.0 - 0.5;
662            b.add_sample(v);
663        }
664        let drained = b.drain_all();
665        for (i, v) in drained.iter().enumerate() {
666            assert!(v.is_finite(), "non-finite at index {i}: {v}");
667        }
668    }
669
670    #[test]
671    fn kernel_phases_accessible() {
672        // Sanity: PHASES + TAPS are the kernel's dimensional parameters.
673        // Powers of two keep the delta_ring's RING_MASK modulo working
674        // and allow the kernel-row lookup to stay branch-free under
675        // the hood. PHASES ≥ 32 ensures sub-sample-phase quantization
676        // noise stays below the spectral acceptance gate (verified by
677        // `tests/spectral.rs`).
678        assert!(PHASES.is_power_of_two() && PHASES >= 32);
679        assert!(TAPS.is_power_of_two() && TAPS >= 16);
680    }
681}