rustynes_apu/blip.rs
1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the band-limited (BLEP) synthesis is derived from blip_buf by Shay Green (Blargg), LGPL-2.1-or-later (GPLv3-compatible). See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! Band-limited synthesis for the APU's audio output.
6//!
7//! # What this is
8//!
9//! A streaming **BLEP (Band-Limited Step) decimator** that takes the
10//! per-CPU-cycle mixer output and produces band-limited samples at the
11//! host audio rate (default 44.1 kHz).
12//!
13//! The technique is band-limited step (BLEP) synthesis — the same general
14//! approach popularized by Shay Green's `blip_buf` and used by many emulators.
15//! Provenance: the band-limited-step technique is **derived from Shay Green's
16//! `blip_buf`** (LGPL-2.1-or-later, which is GPLv3-compatible); our polyphase
17//! kernel in [`crate::blip_kernel`] uses a finer 32-phase resolution than
18//! `blip_buf`. See NOTICE and docs/originality-and-provenance.md (Section 1):
19//!
20//! - Pre-compute a polyphase windowed-sinc kernel ([`crate::blip_kernel`])
21//! keyed by `PHASES = 32` sub-output-sample fractional offsets, with
22//! `TAPS = 32` coefficients per row giving the FIR a ±16-output-sample
23//! reach.
24//! - For each CPU-rate input, compute the amplitude **delta** from the
25//! previous value.
26//! - Position the delta at its fractional output-sample location and add
27//! `delta * kernel[phase][m]` to `TAPS` positions of a **host-rate
28//! delta buffer**.
29//! - Integrate the delta buffer (running cumulative sum) to recover the
30//! reconstructed signal, then push through the existing analog
31//! [`crate::mixer::FilterChain`] (90 Hz HPF + 440 Hz HPF + 14 kHz LPF).
32//!
33//! This is the textbook BLEP structure: each abrupt step in the input is
34//! replaced by the band-limited equivalent (`sinc * window`-shaped step
35//! response), eliminating the alias products that a naive sample-and-
36//! hold decimator leaves above Nyquist.
37//!
38//! # Why this replaces the previous ratio-counter decimator
39//!
40//! The pre-v0.9.x decimator was a ratio-counter with sample-and-hold
41//! reconstruction — functionally a rectangular-window decimator that
42//! left aliased energy above Nyquist (~22.05 kHz at 44.1 kHz host rate).
43//! The existing 14 kHz LPF in the [`crate::mixer::FilterChain`] suppressed
44//! the audible portion, but the band beyond 14 kHz contained alias-
45//! pumping products that the analog filter could not fully kill. With
46//! the v0.9.x mapper-audio extensions (VRC6 / Sunsoft 5B / Namco 163 /
47//! MMC5) adding wavetable + envelope-modulated FM-like complexity, the
48//! alias floor started clearing the LPF's stopband attenuation in spots.
49//!
50//! The polyphase BLEP decimator pushes the alias rejection well below
51//! -60 dB across the audible band even for input frequencies above the
52//! host Nyquist (verified by the spectral FFT regression test in
53//! `tests/spectral.rs`).
54//!
55//! # API contract
56//!
57//! Unchanged from the previous decimator:
58//!
59//! - [`BlipBuf::new(sample_rate, cpu_rate)`] — same signature.
60//! - [`BlipBuf::add_sample(value)`] — called once per CPU cycle.
61//! - [`BlipBuf::drain`], [`BlipBuf::drain_all`], [`BlipBuf::len`],
62//! [`BlipBuf::is_empty`], [`BlipBuf::reset`] — same semantics.
63//!
64//! Save-state compat: the snapshot module reads/writes `sample_rate`,
65//! `cpu_rate`, `phase`, `filter`, and `held_value`. All five are
66//! preserved on this rewrite. The internal delta ring is NOT serialized
67//! — same intentional behavior as before (the ring's contents are sub-
68//! audio-sample-window detail; a fresh restored state begins emitting
69//! samples as soon as `tick()` runs forward).
70//!
71//! # Determinism
72//!
73//! All math is `f32` with a fixed operation order. The kernel is pre-
74//! computed bit-identically across builds (see
75//! [`crate::blip_kernel::Kernel`]). No allocations on the hot path.
76
77#[cfg(test)]
78use crate::blip_kernel::PHASES;
79use crate::blip_kernel::{Kernel, TAPS};
80use crate::mixer::FilterChain;
81use alloc::vec::Vec;
82
83/// The most capacity [`BlipBuf::drain_all`] keeps for the next frame, in
84/// samples (v2.7.5). A frame is about 800 samples at 48 kHz, so four frames'
85/// worth keeps the per-frame benefit at any common output rate while a caller
86/// that once let audio pile up (the probe engine's undrained trials, ~24,000
87/// samples) does not leave a buffer that size resident for good.
88const DRAIN_ALL_KEEP_CAPACITY: usize = 4096;
89
90/// CPU cycles per second, NTSC.
91pub const CPU_HZ_NTSC: f64 = 1_789_773.0;
92/// CPU cycles per second, PAL (slightly slower).
93pub const CPU_HZ_PAL: f64 = 1_662_607.0;
94
95/// Size of the host-rate delta ring buffer. Must be a power of two for
96/// the modulo via bitmask. Held large enough to comfortably absorb a
97/// burst of pending output samples plus the kernel's TAPS-sample reach.
98/// 4096 samples ≈ 93 ms of audio at 44.1 kHz — far more than any single
99/// frame's worth of output (~735 samples).
100const RING_SIZE: usize = 4096;
101const RING_MASK: usize = RING_SIZE - 1;
102
103/// Streaming BLEP decimator + filter chain that feeds host-rate samples
104/// to the frontend's audio thread.
105#[derive(Debug, Clone)]
106pub struct BlipBuf {
107 /// Host sample rate (Hz).
108 pub(crate) sample_rate: u32,
109 /// CPU rate (Hz, fractional).
110 pub(crate) cpu_rate: f64,
111 /// Fractional output-sample position. Advances by `step = sample_rate
112 /// / cpu_rate` per input sample. Whenever the integer part advances,
113 /// one host-rate output sample is "ready" (its delta contributions
114 /// have been written for at least `TAPS/2` future samples ahead, so
115 /// it's safe to read out).
116 pub(crate) phase: f64,
117 /// Step per input sample (`sample_rate / cpu_rate`).
118 step: f64,
119 /// Output filter chain (90 Hz HPF + 440 Hz HPF + 14 kHz LPF).
120 pub(crate) filter: FilterChain,
121 /// Output ring of finalized post-filter samples awaiting drain.
122 pub(crate) samples: Vec<f32>,
123 /// Most recent input value handed to [`Self::add_sample`]. The next
124 /// call's delta is `value - held_value`.
125 pub(crate) held_value: f32,
126 /// Pre-computed polyphase windowed-sinc kernel.
127 kernel: Kernel,
128 /// Host-rate delta ring buffer. Each input delta scatters its
129 /// `kernel * delta` contributions across `TAPS` positions of this
130 /// buffer; the buffer is then integrated (cumulative sum) to
131 /// recover the actual sample stream.
132 delta_ring: [f32; RING_SIZE],
133 /// Integer output-sample index of the **current** input's scatter
134 /// center. Each scatter writes to ring positions
135 /// `[head - TAPS/2, head + TAPS/2)`. The next emit reads at
136 /// `head - TAPS/2 - 1` (one past the leftmost scatter slot).
137 /// Wraps within `RING_SIZE` via `& RING_MASK`.
138 head: usize,
139 /// True once `head` has advanced far enough that the leftmost
140 /// scatter slot `head - TAPS/2` has settled (no future input can
141 /// write to it). Output emission gated on this flag — for the very
142 /// first `TAPS/2 + 1` inputs, the integrator is just warming up
143 /// and no samples are emitted yet. This costs one frame of startup
144 /// latency (~16 samples = 0.36 ms @ 44.1 kHz, well below human-
145 /// perceptible).
146 primed: bool,
147 /// Running integrator state (cumulative sum of consumed delta-ring
148 /// entries since reset). Converts the scattered delta stream back
149 /// into an absolute-amplitude sample stream.
150 integrator: f32,
151}
152
153impl BlipBuf {
154 /// Create a new band-limited buffer.
155 ///
156 /// `sample_rate` is the host audio rate in Hz (typically 44 100).
157 /// `cpu_rate` is the NES CPU rate in Hz ([`CPU_HZ_NTSC`] or
158 /// [`CPU_HZ_PAL`]).
159 #[must_use]
160 pub fn new(sample_rate: u32, cpu_rate: f64) -> Self {
161 let step = f64::from(sample_rate) / cpu_rate;
162 let mut b = Self {
163 sample_rate,
164 cpu_rate,
165 phase: 0.0,
166 step,
167 filter: FilterChain::new(sample_rate),
168 samples: Vec::with_capacity(8192),
169 held_value: 0.0,
170 kernel: Kernel::new(),
171 delta_ring: [0.0; RING_SIZE],
172 // Start `head` at TAPS so the initial scatter (writing to
173 // `head - TAPS/2 .. head + TAPS/2`) lands at indices
174 // `[TAPS/2, 3*TAPS/2)` and never goes negative.
175 head: TAPS,
176 primed: false,
177 integrator: 0.0,
178 };
179 b.reset();
180 b
181 }
182
183 /// v2.1.3 — swap the analog output-filter model (see
184 /// [`crate::mixer::FilterModel`]), rebuilding the filter chain at the
185 /// current sample rate. This resets the filter's IIR state, so switching
186 /// while audio is playing (the Settings selector applies it live) produces a
187 /// brief transient — as with any live filter swap. The frontend also applies
188 /// it at ROM load / power-cycle, where there is no audible discontinuity.
189 pub fn set_filter_model(&mut self, model: crate::mixer::FilterModel) {
190 self.filter = crate::mixer::FilterChain::for_model(self.sample_rate, model);
191 }
192
193 /// Reset to silence. Empties the input ring, the pending output
194 /// queue, and the filter chain state.
195 pub fn reset(&mut self) {
196 self.phase = 0.0;
197 self.filter.reset();
198 self.samples.clear();
199 self.held_value = 0.0;
200 self.delta_ring = [0.0; RING_SIZE];
201 self.head = TAPS;
202 self.primed = false;
203 self.integrator = 0.0;
204 }
205
206 /// The resampler state a save state must carry for a restore to resume
207 /// the exact sample stream (v2.9.9, NL-12): the head's ring position,
208 /// whether warm-up is over, the integrator, and the `TAPS` delta-ring
209 /// slots a scatter can still reach. Every other slot is zero: an emitted
210 /// slot is zeroed as it is consumed, and no scatter writes outside
211 /// `[head - TAPS/2, head + TAPS/2)`.
212 pub(crate) fn live_state(&self) -> (u16, bool, f32, [f32; TAPS]) {
213 let mut window = [0.0; TAPS];
214 let start = self.head.wrapping_sub(TAPS / 2);
215 for (i, v) in window.iter_mut().enumerate() {
216 *v = self.delta_ring[start.wrapping_add(i) & RING_MASK];
217 }
218 // RING_SIZE is 4096, so the masked head fits a u16.
219 #[allow(clippy::cast_possible_truncation)]
220 let head = (self.head & RING_MASK) as u16;
221 (head, self.primed, self.integrator, window)
222 }
223
224 /// Install a state [`Self::live_state`] produced. The caller validates
225 /// the values; `head` is masked into the ring here.
226 pub(crate) fn set_live_state(
227 &mut self,
228 head: u16,
229 primed: bool,
230 integrator: f32,
231 window: &[f32; TAPS],
232 ) {
233 self.delta_ring = [0.0; RING_SIZE];
234 self.head = usize::from(head) & RING_MASK;
235 self.primed = primed;
236 self.integrator = integrator;
237 let start = self.head.wrapping_sub(TAPS / 2);
238 for (i, &v) in window.iter().enumerate() {
239 self.delta_ring[start.wrapping_add(i) & RING_MASK] = v;
240 }
241 }
242
243 /// Add one mixed sample at CPU resolution. The buffer accumulates
244 /// host-rate samples internally; drain via [`Self::drain`] or
245 /// [`Self::drain_all`].
246 ///
247 /// The mixer's output is approximately in `[-0.5, 0.5]` for the 2A03
248 /// alone, with on-cart audio expansions pushing the absolute range
249 /// up somewhat. We clamp `value` defensively so a stray NaN/Inf or
250 /// runaway mapper can't propagate non-finite values into the FIR
251 /// state. The clamp range is far outside any sane mixer output, so
252 /// the gate is purely a saturation backstop, not a normal-operation
253 /// limiter.
254 #[inline]
255 pub fn add_sample(&mut self, value: f32) {
256 // Defensive saturation. Real mixer output never exceeds ~1.5; the
257 // clamp catches NaN/Inf from a runaway mapper-audio path so the
258 // FIR can't propagate non-finite values through the ring.
259 let value = if value.is_finite() {
260 value.clamp(-4.0, 4.0)
261 } else {
262 0.0
263 };
264 let delta = value - self.held_value;
265 self.held_value = value;
266
267 // Scatter the delta into the host-rate ring via the kernel row
268 // matching the input's fractional output-sample position.
269 //
270 // `self.phase` ∈ [0, 1) is the fractional output-sample position
271 // of THIS input within the current output-grid interval. The
272 // kernel row matching this fractional offset spreads the delta
273 // across `TAPS` output samples centered at `head`. The center
274 // sits at `head + TAPS/2 - 1` (the right-half tap that's closest
275 // to the input's position); we walk the kernel from `head` for
276 // `TAPS` slots.
277 if delta != 0.0 {
278 #[allow(clippy::cast_possible_truncation)]
279 let row = self.kernel.row(self.phase as f32);
280 // Center the scatter at `head` (the current integer output
281 // index). Tap 0 lands at `head - TAPS/2`; tap TAPS-1 lands
282 // at `head + TAPS/2 - 1`. The kernel itself encodes the
283 // sub-sample fractional shift via the row index.
284 //
285 // v2.8.0 Phase 4b — the 32-tap window wraps the ring at most
286 // once, so split it into (at most) two CONTIGUOUS runs instead
287 // of masking every index: each run is a plain SAXPY LLVM
288 // auto-vectorizes (SSE2/NEON/wasm-simd), which the per-tap
289 // `& RING_MASK` form structurally prevented. Per-slot math is
290 // unchanged (`slot += delta * coeff`, one touch per slot, mul
291 // then add — no FMA contraction), so output is bit-identical
292 // to the scalar form.
293 let start = self.head.wrapping_sub(TAPS / 2) & RING_MASK;
294 let first = (RING_SIZE - start).min(TAPS);
295 for (slot, &coeff) in self.delta_ring[start..start + first]
296 .iter_mut()
297 .zip(&row[..first])
298 {
299 *slot += delta * coeff;
300 }
301 for (slot, &coeff) in self.delta_ring[..TAPS - first]
302 .iter_mut()
303 .zip(&row[first..])
304 {
305 *slot += delta * coeff;
306 }
307 }
308
309 // Advance phase. When it crosses 1.0, advance `head` and
310 // emit the output sample that just fell out of the scatter
311 // window (one slot to the left of `head - TAPS/2`).
312 self.phase += self.step;
313 while self.phase >= 1.0 {
314 self.phase -= 1.0;
315 self.head = self.head.wrapping_add(1);
316
317 // The slot at `head - TAPS/2 - 1` is now permanently
318 // finalized: any future scatter writes to `[new_head -
319 // TAPS/2, new_head + TAPS/2)`, which doesn't reach back
320 // this far. Emit it.
321 //
322 // During warm-up (the first TAPS/2 + 1 head advances), the
323 // initial slots haven't been written by any scatter yet
324 // (they're still zero from construction); we burn through
325 // them silently to flush the startup phase, then start
326 // emitting once `primed` is set.
327 let emit_idx = self.head.wrapping_sub(TAPS / 2 + 1) & RING_MASK;
328 self.integrator += self.delta_ring[emit_idx];
329 self.delta_ring[emit_idx] = 0.0;
330
331 if !self.primed {
332 // Have we advanced past the initial dead zone? The
333 // first scatter wrote at `head_initial = TAPS` (i.e.,
334 // covering `[TAPS/2, 3*TAPS/2)`). We start emitting
335 // once `head >= TAPS + TAPS/2 + 1`, i.e., the emit
336 // slot has caught up with the first scatter's right
337 // edge.
338 if self.head > TAPS + TAPS / 2 {
339 self.primed = true;
340 }
341 continue;
342 }
343
344 let filtered = self.filter.process(self.integrator);
345 self.samples.push(filtered);
346 }
347 }
348
349 /// Drain finalized samples into `out`. Returns the number written.
350 /// Excess pending samples are kept; if `out.len() < self.samples.len()`
351 /// the remainder waits for the next drain.
352 pub fn drain(&mut self, out: &mut [f32]) -> usize {
353 let n = self.samples.len().min(out.len());
354 out[..n].copy_from_slice(&self.samples[..n]);
355 self.samples.drain(..n);
356 n
357 }
358
359 /// Drain all finalized samples into a new `Vec`.
360 ///
361 /// The filled buffer goes to the caller, and a fresh one with the same
362 /// capacity takes its place (v2.7.5, core audit IMP-07). A buffer returned
363 /// by value must be replaced every call whatever happens; the choice is
364 /// only whether its replacement starts empty. With `mem::take` it started
365 /// at capacity zero and regrew by doubling through the next frame's pushes
366 /// (about nine reallocations for a frame's ~800 samples); now it is one
367 /// allocation of the size the last frame needed, up to
368 /// `DRAIN_ALL_KEEP_CAPACITY` samples. Measured -0.89% frame
369 /// time on a steady-state frame-then-drain workload, reproduced on two
370 /// runs (`docs/performance.md` §v2.7.5). Callers that drain into their own
371 /// buffer with [`Self::drain`] allocate nothing and are unaffected. The
372 /// samples themselves are untouched, so the output is byte-identical.
373 #[must_use]
374 pub fn drain_all(&mut self) -> Vec<f32> {
375 // The `core` path keeps this module portable to `#![no_std]` builds.
376 // See `docs/architecture.md` §no_std boundary.
377 // Nothing queued: hand back an unallocated `Vec` and keep the buffer,
378 // rather than allocate a replacement for an empty one.
379 if self.samples.is_empty() {
380 return Vec::new();
381 }
382 // Clamped so a one-off backlog is not kept as a high-water mark
383 // (see `DRAIN_ALL_KEEP_CAPACITY`); a per-frame drain is far below it.
384 let capacity = self.samples.capacity().min(DRAIN_ALL_KEEP_CAPACITY);
385 core::mem::replace(&mut self.samples, Vec::with_capacity(capacity))
386 }
387
388 /// Number of samples currently buffered, awaiting drain.
389 #[must_use]
390 pub fn len(&self) -> usize {
391 self.samples.len()
392 }
393
394 /// Whether the pending-output queue is empty.
395 #[must_use]
396 pub fn is_empty(&self) -> bool {
397 self.samples.is_empty()
398 }
399}
400
401#[cfg(test)]
402mod tests {
403 use super::*;
404
405 /// Total CPU cycles in 1 second of NTSC playback. Used as a "long-
406 /// run" fixture for the determinism / DC tests.
407 const ONE_SECOND_NTSC: usize = 1_789_773;
408
409 /// 10 frames at 60 Hz (used by spectral tests + decay tests).
410 const TEN_FRAMES_NTSC: usize = ONE_SECOND_NTSC / 6;
411
412 /// v2.7.5 (core audit IMP-07): `drain_all` hands the filled buffer to the
413 /// caller and keeps a same-capacity one for the next frame. With
414 /// `mem::take` the buffer restarted at capacity zero and regrew by doubling
415 /// through every frame's pushes (about nine reallocations for a frame's
416 /// ~800 samples). Measured on a steady-state frame-then-drain workload:
417 /// -0.89% frame time on two runs (docs/performance.md §v2.7.5).
418 #[test]
419 fn drain_all_keeps_room_for_the_next_frame() {
420 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
421 for _ in 0..TEN_FRAMES_NTSC / 10 {
422 b.add_sample(0.25);
423 }
424 let first = b.drain_all();
425 assert!(!first.is_empty(), "a frame produces samples");
426 assert!(
427 b.samples.capacity() >= first.len(),
428 "the next frame must not start from capacity zero: {} < {}",
429 b.samples.capacity(),
430 first.len()
431 );
432 assert!(b.is_empty(), "the drained samples are gone");
433 }
434
435 /// An empty drain hands back an empty, unallocated `Vec` and keeps the
436 /// buffer as it is (agy on #553): polling with nothing queued used to
437 /// allocate a replacement and return the old, empty buffer to be freed.
438 #[test]
439 fn an_empty_drain_all_allocates_nothing() {
440 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
441 for _ in 0..TEN_FRAMES_NTSC / 10 {
442 b.add_sample(0.25);
443 }
444 let _ = b.drain_all();
445 let kept = b.samples.capacity();
446 let empty = b.drain_all();
447 assert_eq!(empty, [] as [f32; 0]);
448 assert_eq!(empty.capacity(), 0, "no buffer handed out for nothing");
449 assert_eq!(b.samples.capacity(), kept, "the kept buffer stays");
450 }
451
452 /// The kept capacity is clamped (agy on #553): a caller that lets audio
453 /// pile up once -- the probe engine's undrained trials, ~24,000 samples --
454 /// must not leave a buffer that size resident for good. The per-frame case
455 /// (~800 samples) is far below the clamp, so it keeps its full benefit.
456 #[test]
457 fn drain_all_does_not_keep_a_high_water_mark() {
458 let mut b = BlipBuf::new(48_000, CPU_HZ_NTSC);
459 for _ in 0..ONE_SECOND_NTSC {
460 b.add_sample(0.25);
461 }
462 let backlog = b.drain_all();
463 assert!(
464 backlog.len() > DRAIN_ALL_KEEP_CAPACITY,
465 "a long undrained run"
466 );
467 assert!(
468 b.samples.capacity() <= DRAIN_ALL_KEEP_CAPACITY,
469 "kept {} after a {}-sample backlog",
470 b.samples.capacity(),
471 backlog.len()
472 );
473 }
474
475 #[test]
476 fn empty_buffer_reads_zero_samples() {
477 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
478 let mut out = [0.0_f32; 16];
479 assert_eq!(b.drain(&mut out), 0);
480 assert!(b.is_empty());
481 assert_eq!(b.len(), 0);
482 }
483
484 #[test]
485 fn emits_expected_sample_count() {
486 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
487 for _ in 0..ONE_SECOND_NTSC {
488 b.add_sample(0.0);
489 }
490 let drained = b.drain_all();
491 let n = drained.len();
492 // Expected: 44_100 samples ± a small warm-up loss. The BLEP
493 // structure delays output by `TAPS/2 + 1` host samples after
494 // construction so each emitted sample has received its full
495 // kernel scatter — that's ~17 samples of startup delay at our
496 // TAPS=32 configuration. After 1 s, output count is
497 // `44_100 - 17 ± rounding`.
498 let max_loss = TAPS / 2 + 4;
499 assert!(
500 n <= 44_100 && n + max_loss >= 44_100,
501 "expected 44100 - {max_loss}..=44100 samples in 1 s, got {n}"
502 );
503 }
504
505 #[test]
506 fn dc_passes_through_with_settled_gain() {
507 // A constant 0.5 input. The FIR scatters NO delta (the input
508 // never changes after the first sample), so the integrator
509 // converges to 0.5. The HPFs then attenuate the DC; after
510 // ~200 000 samples the output is near zero.
511 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
512 for _ in 0..200_000 {
513 b.add_sample(0.5);
514 }
515 let drained = b.drain_all();
516 let last = *drained.last().unwrap();
517 assert!(last.abs() < 0.05, "DC not removed; last sample = {last}");
518 }
519
520 #[test]
521 fn drain_empties_buffer() {
522 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
523 for _ in 0..1000 {
524 b.add_sample(0.0);
525 }
526 let _ = b.drain_all();
527 assert!(b.is_empty());
528 }
529
530 #[test]
531 fn single_delta_produces_band_limited_step() {
532 // Feed silence, then a unit step. The output should be a
533 // band-limited ramp (sinc-ringing) reaching the new amplitude
534 // over ~TAPS host-rate samples.
535 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
536 for _ in 0..10_000 {
537 b.add_sample(0.0);
538 }
539 for _ in 0..10_000 {
540 b.add_sample(1.0);
541 }
542 // Push enough samples to fully settle the FIR + HPF response.
543 for _ in 0..50_000 {
544 b.add_sample(1.0);
545 }
546 let drained = b.drain_all();
547 // After 50_000+ samples of constant 1.0, the HPFs drive output
548 // back to ~0 (DC blocked).
549 let last = *drained.last().unwrap();
550 assert!(
551 last.abs() < 0.05,
552 "constant input not DC-blocked: last = {last}"
553 );
554 // And the step transient produced finite peak < the saturation
555 // clip (1.5 is our cap before clamp).
556 let max_abs = drained.iter().map(|s| s.abs()).fold(0.0_f32, f32::max);
557 assert!(
558 max_abs > 0.0 && max_abs < 1.5,
559 "step transient max = {max_abs}, expected (0, 1.5)"
560 );
561 }
562
563 #[test]
564 fn opposing_deltas_cancel_to_dc() {
565 // Equal amounts of +1 and -1 in long runs, end with HPF-blocked
566 // output. The integrator must NOT drift.
567 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
568 for _ in 0..100_000 {
569 b.add_sample(1.0);
570 }
571 for _ in 0..100_000 {
572 b.add_sample(-1.0);
573 }
574 let drained = b.drain_all();
575 let last = *drained.last().unwrap();
576 assert!(last.abs() < 0.05, "didn't cancel; last = {last}");
577 }
578
579 #[test]
580 fn deterministic_across_runs() {
581 // Same input sequence produces bit-identical output.
582 let drive = |b: &mut BlipBuf| {
583 for i in 0..TEN_FRAMES_NTSC {
584 #[allow(clippy::cast_precision_loss)]
585 let v = ((i as f32) * 0.0001).sin() * 0.5;
586 b.add_sample(v);
587 }
588 };
589 let mut a = BlipBuf::new(44_100, CPU_HZ_NTSC);
590 drive(&mut a);
591 let av = a.drain_all();
592 let mut c = BlipBuf::new(44_100, CPU_HZ_NTSC);
593 drive(&mut c);
594 let cv = c.drain_all();
595 assert_eq!(av.len(), cv.len(), "same input, different output length");
596 for (i, (x, y)) in av.iter().zip(cv.iter()).enumerate() {
597 assert_eq!(
598 x.to_bits(),
599 y.to_bits(),
600 "non-determinism at index {i}: {x} vs {y}"
601 );
602 }
603 }
604
605 #[test]
606 fn saturation_clips_extreme_values() {
607 // A pathological NaN / Inf input must not poison the output ring.
608 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
609 b.add_sample(f32::NAN);
610 b.add_sample(f32::INFINITY);
611 b.add_sample(f32::NEG_INFINITY);
612 for _ in 0..200 {
613 b.add_sample(1.0e20);
614 }
615 // No panics, drained samples are all finite.
616 let drained = b.drain_all();
617 for v in &drained {
618 assert!(v.is_finite(), "output poisoned by extreme input: {v}");
619 }
620 assert!(b.held_value.is_finite());
621 }
622
623 #[test]
624 fn reset_clears_all_state() {
625 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
626 for _ in 0..1000 {
627 b.add_sample(0.5);
628 }
629 b.reset();
630 assert_eq!(b.phase, 0.0);
631 assert_eq!(b.held_value, 0.0);
632 assert!(b.is_empty());
633 for v in &b.delta_ring {
634 assert_eq!(*v, 0.0);
635 }
636 assert_eq!(b.integrator, 0.0);
637 }
638
639 #[test]
640 fn drain_slice_handles_partial_read() {
641 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
642 for _ in 0..10_000 {
643 b.add_sample(0.0);
644 }
645 let queued = b.len();
646 assert!(queued > 0);
647 let mut out = [0.0_f32; 16];
648 let n = b.drain(&mut out);
649 assert_eq!(n, 16);
650 assert_eq!(b.len(), queued - 16);
651 }
652
653 #[test]
654 fn long_run_produces_finite_output() {
655 // Sweep amplitudes so the FIR convolution + integrator path is
656 // exercised. All drained samples must be finite (no NaN/Inf
657 // leakage from the kernel-row lookup or integrator drift).
658 let mut b = BlipBuf::new(44_100, CPU_HZ_NTSC);
659 for i in 0..(ONE_SECOND_NTSC / 100) {
660 #[allow(clippy::cast_precision_loss)]
661 let v = ((i % 100) as f32) / 100.0 - 0.5;
662 b.add_sample(v);
663 }
664 let drained = b.drain_all();
665 for (i, v) in drained.iter().enumerate() {
666 assert!(v.is_finite(), "non-finite at index {i}: {v}");
667 }
668 }
669
670 #[test]
671 fn kernel_phases_accessible() {
672 // Sanity: PHASES + TAPS are the kernel's dimensional parameters.
673 // Powers of two keep the delta_ring's RING_MASK modulo working
674 // and allow the kernel-row lookup to stay branch-free under
675 // the hood. PHASES ≥ 32 ensures sub-sample-phase quantization
676 // noise stays below the spectral acceptance gate (verified by
677 // `tests/spectral.rs`).
678 assert!(PHASES.is_power_of_two() && PHASES >= 32);
679 assert!(TAPS.is_power_of_two() && TAPS >= 16);
680 }
681}