Skip to main content

rustynes_apu/
apu.rs

1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the DMC-DMA state model (the get/put flip-flop, the delayed-`$4015` DMC status machinery, the implicit-abort and re-enable timing fields) is derived from TriCNES (MIT), as the field docs cite by name and `Emulator.cs` line, and the `_needHalt` / `_needDummyRead` latches from Mesen2 (GPL-3.0-or-later), `NesCpu`. See docs/originality-and-provenance.md (Section 1) and NOTICE. Classified v2.9.9 (core re-audit NC-17, maintainer's decision 2026-10-04): the in-source citations below record the derivation and are kept as written.
4
5//! Top-level 2A03 APU.
6//!
7//! Per `docs/apu-2a03.md`.  Owns the four wave channels plus DMC, the frame
8//! counter, the lookup-table mixer + filter chain, and the band-limited
9//! sample emitter.  Driven by the lockstep bus's `Apu::tick` once per CPU
10//! cycle.
11
12use crate::Region;
13use crate::blip::{BlipBuf, CPU_HZ_NTSC, CPU_HZ_PAL};
14use crate::dmc::Dmc;
15use crate::frame_counter::{FrameCounter, FrameEvents};
16use crate::mixer::Mixer;
17use crate::noise::Noise;
18use crate::pulse::Pulse;
19use crate::triangle::Triangle;
20use alloc::vec::Vec;
21
22// `f32::round` lives in `std` (not `core`), so route through `libm::roundf` on
23// no_std — the same pattern the mixer uses for `expf`. Both round half away from
24// zero, so the result is identical across the desktop + `thumbv7em-none-eabihf`
25// targets. Only reached on the off-default per-channel-gain path (gain != 1.0),
26// never on the byte-identical unity path.
27#[inline]
28fn roundf(x: f32) -> f32 {
29    #[cfg(feature = "std")]
30    {
31        x.round()
32    }
33    #[cfg(not(feature = "std"))]
34    {
35        libm::roundf(x)
36    }
37}
38
39/// Top-level APU.
40#[derive(Debug, Clone)]
41pub struct Apu {
42    /// Region (NTSC / PAL / Dendy).
43    pub region: Region,
44    /// Pulse 1.
45    pub pulse1: Pulse,
46    /// Pulse 2.
47    pub pulse2: Pulse,
48    /// Triangle.
49    pub triangle: Triangle,
50    /// Noise.
51    pub noise: Noise,
52    /// DMC.
53    pub dmc: Dmc,
54    /// Frame counter.
55    pub frame_counter: FrameCounter,
56    /// Mixer.
57    pub(crate) mixer: Mixer,
58    /// Band-limited sample emitter.
59    pub(crate) blip: BlipBuf,
60    /// True on every other CPU cycle — pulse/noise/DMC clock at this rate.
61    // reason: `apu_phase` is the deliberate, documented name for the APU's
62    // clock phase; it appears verbatim as a column header in the committed
63    // irq_trace golden CSVs, so the `apu_` prefix is load-bearing, not noise.
64    #[allow(clippy::struct_field_names)]
65    pub(crate) apu_phase: bool,
66    /// v2.0 F-2: when set, the DMC byte-timer + DMA arm are driven by
67    /// [`Self::tick_dmc`] (called at end-of-cycle by the R1 bus) instead of
68    /// inside [`Self::tick_with_external`] (cycle-start). This shifts only the
69    /// DMC fire-phase to main's end-of-cycle position (for DMASync), leaving
70    /// the frame-counter / pulse / noise — and thus the APU IRQ line — on the
71    /// cycle-start tick (for the C1 IRQ sample). Default `false` = byte-identical.
72    pub(crate) dmc_driven_externally: bool,
73    /// v2.0 interleaved-DMA Phase A: the global get/put flip-flop (TriCNES
74    /// `APU_PutCycle`, `Emulator.cs:920`). Toggled exactly once per CPU cycle
75    /// (when `dmc_driven_externally`, so the default build is byte-identical)
76    /// and seeded at power-on/reset TOGETHER with the DMC byte-timer from one
77    /// `APUAlignment` value, so the get/put parity and the DMC fire-phase share
78    /// one seed + one per-cycle counter and can never drift (divergence A). The
79    /// interleaved DMA (Phase B) reads this for the get/put decision instead of
80    /// `self.cycle & 1`. `true` = put (write/OAM-priority), `false` = get
81    /// (read/DMC-priority). Nothing consumes it yet in Phase A.
82    pub(crate) put_cycle: bool,
83    /// v2.0 RW-1 (`mc-r1-one-clock`): the single boot parity seed. Under
84    /// `mc-r1-one-clock`, `apu_phase` and `put_cycle` are no longer two
85    /// independent flip-flops toggled per cycle — they are DERIVED from the one
86    /// per-cycle counter (`cpu_cycle`) plus this seed:
87    /// `apu_phase = (cpu_cycle + parity_seed) & 1 == 1`, `put_cycle = !apu_phase`.
88    /// This makes the APU-rate clock, the get/put DMA parity, and the DMC
89    /// fire-phase share ONE counter + ONE seed, so they can never drift apart
90    /// (the cumulative-counter-split root cause, see
91    /// `docs/audit/v2.0-cumulative-cycle-accounting-rewrite-plan-2026-06-05.md`).
92    /// `0` reproduces the floor config exactly (boot `apu_phase = false`,
93    /// `seed_apu_alignment(0)` -> put-on-even). Set once at power-on/reset/restore
94    /// by [`Self::seed_apu_alignment`]; otherwise constant. Unused when the flag
95    /// is off (the legacy dual-toggle path runs instead).
96    pub(crate) parity_seed: u64,
97    /// Cumulative CPU cycle counter (used for `$4017` write alignment).
98    pub(crate) cpu_cycle: u64,
99    /// Pending DMC DMA request — the bus polls and consumes this when it
100    /// halts the CPU and supplies a sample byte.
101    pub(crate) pending_dmc_dma: bool,
102    /// `mc-r1-dmc-reload-visibility-delay`: a RELOAD arm latches HERE and is
103    /// promoted to `pending_dmc_dma` one cycle later, so the DMA loop first
104    /// services it on the NEXT (put) cycle — matching TriCNES's
105    /// `_EmulateAPU`-after-`_6502` invisibility (reload first-service = put =>
106    /// span 4). Loads arm `pending_dmc_dma` directly (first-service = get => 3).
107    pub(crate) pending_dmc_dma_next: bool,
108    /// True when the pending DMC DMA is the initial load DMA after `$4015`
109    /// enable; false for reload DMAs raised by sample-buffer empty.
110    pub(crate) dmc_dma_is_load: bool,
111    /// True when the pending request uses the short 3-cycle service path
112    /// despite being externally observed as a load-style race. This covers
113    /// the explicit-stop abort edge where a visible reload request must be
114    /// preserved through the `$4015` disable write without making ordinary
115    /// load DMAs lose their dummy/alignment cadence.
116    pub(crate) dmc_dma_short: bool,
117    /// Suppress one immediate reload request after a same-tick DMC load
118    /// delivery. Used for the one-byte looping edge where the fetched byte
119    /// is visible to the output unit on the DMA get cycle, but the reload
120    /// request is not visible until the following CPU cycle.
121    pub(crate) defer_dmc_reload_once: bool,
122    /// Pending one-cycle DMC abort halt. This is the RP2A03 stop-near-reload
123    /// quirk: it does not fetch a byte, and if the halt attempt lands on a
124    /// CPU write cycle the abort disappears instead of retrying.
125    pub(crate) pending_dmc_abort: bool,
126    /// CPU cycles until an abort halt attempt becomes visible to the bus.
127    pub(crate) dmc_abort_delay: u8,
128    /// CPU cycles during which a newly emptied DMC sample buffer must not
129    /// raise another DMA request. The DMC DMA unit cannot issue a second
130    /// request within two CPU cycles of the previous get.
131    pub(crate) dmc_dma_cooldown: u8,
132    /// v2.0.0 beta.3 (A4 cycle-accurate reset): countdown to the scheduled
133    /// warm-reset `$4017` re-write (blargg `apu_reset` spec: reset behaves
134    /// as if the last `$4017` value were written again). Armed by
135    /// [`Apu::reset`] with the calibrated in-sequence placement; consumed in
136    /// `tick_with_external` during the CPU's clocked reset cycles. `0` = no
137    /// re-write pending.
138    pub(crate) reset_4017_delay: u8,
139    /// The retained `$4017` value the scheduled reset re-write will issue.
140    pub(crate) reset_4017_value: u8,
141    /// v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES's
142    /// `CannotRunDMCDMARightNow` (`Emulator.cs:823`). Set to 2 after every DMC
143    /// GET (`:4168`), decremented by 2 on each get cycle (`:1186`), and gating
144    /// the looping-reload arm (`:1165`, blocked while `== 2`). Reproduces the
145    /// "a DMA cannot occur within 2 cycles of a previous DMC DMA" rule so the
146    /// Implicit-DMA-Abort Loop3/`$540` X=10/11 re-enable defers its first
147    /// reload one byte-timer period (walk offset +4 -> +5, Y 3->4). Distinct
148    /// from `dmc_dma_cooldown` (which the abort-timer-phase fix clears at the
149    /// boundary race); this exclusion re-imposes the exact 1-get-cycle block.
150    pub(crate) cannot_run_dmc_dma: u8,
151    /// v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): latch that defers a reload to
152    /// the NEXT byte-timer wrap when the exclusion blocks the arm. TriCNES only
153    /// evaluates the reload arm at the `bits_remaining -> 0` consume edge
154    /// (`Emulator.cs:1159`); if blocked there (`cannot_run == 2`) the buffer is
155    /// not refilled until the FOLLOWING consume edge (one full byte period
156    /// later) — NOT a 2-cycle re-arm. RustyNES's `dmc_step_reload_arm` instead
157    /// re-checks `needs_dma()` (persistent) every cycle, so a bare exclusion
158    /// gate would re-arm as soon as `cannot_run` decremented (absorbed). This
159    /// latch reproduces the full-period deferral: set when the exclusion blocks
160    /// a consume-edge arm, cleared at the next consume edge.
161    pub(crate) dmc_reenable_period_block: bool,
162    /// final lever #1 (`mc-r1-dmc-halt-subpos`): a per-CPU-cycle countdown that
163    /// DELAYS the `$540` X=10/11 reload arm by an exact number of CPU cycles
164    /// (sub-APU-cycle granularity the byte-timer phase shift cannot express).
165    /// `0` = inactive. Set at the pattern-A boundary; while > 0 the natural arm
166    /// is suppressed and this decrements; at 0 the arm fires. Default 0.
167    // (W3-Stage-3: also dead under `mc-r1-dmc-delayed-4015`, which supersedes
168    // the halt-subpos pre-arm with the emergent consume-edge arm.)
169    pub(crate) subpos_arm_countdown: u8,
170    /// Latches the one-byte looping edge where the next reload request is
171    /// lost because it is raised too soon after a DMA get. Cleared when a
172    /// later `$4015` enable/disable write re-arms the DMC path.
173    pub(crate) dmc_reload_suppress_outputs: u8,
174    /// CPU cycles until a load DMC DMA halt attempt becomes visible to the
175    /// bus. Reload DMAs are armed immediately after the DMC output unit
176    /// empties the sample buffer; load DMAs after `$4015` enable are delayed
177    /// to the second following APU cycle per the 2A03 DMA cadence.
178    pub(crate) dmc_dma_delay: u8,
179    /// Most recent DMC DMA address (re-read each tick when `pending_dmc_dma`
180    /// is true; the bus may take it directly via [`Self::dmc_dma_addr`]).
181    /// On its own this is informational; the bus owns the actual halt logic.
182    pub(crate) dmc_dma_addr: u16,
183    /// v1.2 Sprint 3 — get/put cycle scheduler model (ADR-0007).
184    ///
185    /// Set when a DMC DMA request is raised; cleared by the new
186    /// `rustynes-core::bus::service_dmc_dma` path under the
187    /// `dmc-get-put-scheduler` feature flag, once the initial halt
188    /// cycle has been processed. Mirrors Mesen2's `_needHalt` on
189    /// `NesCpu` (`Core/NES/NesCpu.h:41`; set in `StartDmcTransfer`
190    /// at `Core/NES/NesCpu.cpp:527`). Kept ALWAYS-PRESENT (not
191    /// `#[cfg]`-gated) so the field exists in serialized state for
192    /// future-flag-flip migration; the v1.2 baseline scheduler
193    /// simply ignores it.
194    pub(crate) dmc_need_halt: bool,
195    /// v1.2 Sprint 3 — get/put cycle scheduler model (ADR-0007).
196    ///
197    /// Set when a DMC DMA request is raised; cleared by the new
198    /// scheduler once the alignment / dummy-read cycle has been
199    /// processed. Mirrors Mesen2's `_needDummyRead` on `NesCpu`.
200    /// Kept always-present alongside [`Self::dmc_need_halt`].
201    pub(crate) dmc_need_dummy_read: bool,
202    /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `APU_DelayedDMC4015`
203    /// (Emulator.cs:973) — CPU-cycle countdown until the latched `$4015` DMC
204    /// status bit is APPLIED. Set to `put ? 3 : 4` at every `$4015` write
205    /// (extended to `put ? 5 : 6` at the explicit don't-abort edge);
206    /// decremented once per `dmc_tick_end` (every CPU cycle, the write
207    /// cycle's own end-tick included). `0` = idle.
208    pub(crate) dmc_delayed_4015: u8,
209    /// W3-Stage-3: TriCNES `APU_Status_DelayedDMC` (Emulator.cs:964) — the
210    /// TARGET DMC status latched at the `$4015` write, applied when the
211    /// countdown expires. Also the value `$4015` READS see immediately
212    /// (the footnote at Emulator.cs:9268: bit 4 must read 0 right after a
213    /// disable write even though `bytes_remaining` is not yet zeroed).
214    pub(crate) dmc_delayed_status: bool,
215    /// W3-Stage-3: TriCNES `APU_Status_DMC` (Emulator.cs:963) — the APPLIED
216    /// DMC status. The bus-side DMA service gate (`_6502` line 4218:
217    /// `DoDMCDMA && (APU_Status_DMC || implicit-abort)`) reads this; while
218    /// false a pending/halted DMC DMA is NOT serviced (the emergent explicit
219    /// abort). Set/cleared ONLY by the delayed application + cleared at
220    /// non-looping natural sample end (TriCNES `DMCDMA_Get`, line 4154).
221    pub(crate) dmc_status_applied: bool,
222    /// W3-Stage-3: TriCNES `APU_SetImplicitAbortDMC4015` (Emulator.cs:975) —
223    /// latched at a `$4015` ENABLE write that coincides with the byte-timer's
224    /// firing window (`(timer == 10 && get) || (timer == 8 && put)` in
225    /// TriCNES CPU-rate units = our APU-rate `(4, get)/(3, put)`); consumed
226    /// at the next shifter-consume edge (bits 1 -> 8), where it arms the
227    /// 1-cycle implicit-abort DMA regardless of the buffer state
228    /// (Emulator.cs:1163-1175).
229    pub(crate) dmc_set_implicit_abort: bool,
230    /// W3-Stage-3: TriCNES `APU_ImplicitAbortDMC4015` (Emulator.cs:974) — the
231    /// service-gate override that lets the boundary-armed DMA run while the
232    /// `$4015` enable's delayed status is still unapplied. Cleared at the END
233    /// of the first cycle on which the DMA is pending (Emulator.cs:9000-9003:
234    /// one serviced halt cycle if that cycle is a read; "if this was delayed
235    /// by a write cycle, it won't run at all") — the emergent 1-cycle
236    /// implicit abort.
237    pub(crate) dmc_implicit_abort: bool,
238    /// W3-Stage-4 (`mc-r1-dmc-delayed-4015` grid correction): TriCNES's
239    /// reload arm is CONSUME-EDGE-QUANTIZED, not level-held. When the consume
240    /// edge lands ON the GET-delivery cycle itself (`CannotRunDMCDMARightNow
241    /// == 2`, Emulator.cs:1165 — only ever true at the same-cycle edge,
242    /// because the decrement at :1186 runs later that same end-tick), the arm
243    /// is skipped ENTIRELY and the chain defers to the NEXT consume edge
244    /// (576 cycles). Our `needs_dma()` is level-triggered and would re-arm as
245    /// soon as the cooldown expires (4 cycles — one grid boundary early, the
246    /// Implicit `$540[8,9]` cliff). Set at the blocked same-cycle edge;
247    /// suppresses the reload arm; cleared at the next consume edge in
248    /// `dmc_tick_end` immediately before the reload-arm step so the deferred
249    /// arm fires exactly on-grid.
250    pub(crate) dmc_edge_arm_suppress: bool,
251    /// Sample rate (Hz) for diagnostics.
252    pub sample_rate: u32,
253    /// Most recent frame-counter events produced by [`Self::tick_with_external`].
254    /// Read by the bus immediately after `tick` to fan the same events out to
255    /// any on-cart audio extension that shares the 2A03 frame counter cadence
256    /// (MMC5 audio). Reset to `FrameEvents::default()` at the *start* of every
257    /// `tick`, so observers must read it AFTER the tick.
258    pub(crate) last_frame_events: FrameEvents,
259    /// Per-channel enable mask (UI playback overlay, NOT NES hardware state).
260    /// Bit 0 = pulse 1, bit 1 = pulse 2, bit 2 = triangle, bit 3 = noise,
261    /// bit 4 = DMC, bit 5 = external/mapper audio. A cleared bit forces that
262    /// channel's contribution to the mixed sample to 0 (a studio/debug mute).
263    ///
264    /// Defaults to [`CHANNEL_MASK_ALL`] (every bit set), which is byte-identical
265    /// to passing the raw channel outputs straight into the mixer — i.e. the
266    /// deterministic core output is unchanged unless the frontend explicitly
267    /// mutes a channel. NEVER serialized into the save state (a UI preference,
268    /// like volume), so restored states are unaffected.
269    pub(crate) channel_mask: u8,
270    /// v1.4.0 Workstream C — per-channel output gain (a UI mixing overlay, NOT
271    /// NES hardware state), generalizing [`Self::channel_mask`]. Index 0 = pulse
272    /// 1, 1 = pulse 2, 2 = triangle, 3 = noise, 4 = DMC, 5 = external/mapper
273    /// audio. Each internal channel's raw integer output is scaled by its gain
274    /// and rounded back to an integer before the non-linear mixer; the external
275    /// (already-linear) sample is scaled directly.
276    ///
277    /// Defaults to [`CHANNEL_GAIN_UNITY`] (all `1.0`). At unity the mix takes the
278    /// EXACT current code path (`round(v * 1.0) == v`, `external * 1.0 ==
279    /// external`), so the deterministic core output is byte-identical unless the
280    /// frontend explicitly changes a gain — the determinism contract holds and
281    /// the oracle / test ROMs (which never touch a gain) are unaffected. NEVER
282    /// serialized into the save state (a UI preference, like the mask / volume).
283    pub(crate) channel_gain: [f32; 6],
284    /// v2.9.8 (D3): `channel_gain == CHANNEL_GAIN_UNITY`, cached.
285    ///
286    /// The per-cycle mix checks for unity gain on every CPU cycle. Comparing
287    /// six `f32`s there cost -3.1% to -4.5% of frame time on all four
288    /// workloads, in two runs (`docs/performance.md`, v2.9.8 campaign). The
289    /// gain only changes through [`Self::set_channel_gain`], which recomputes
290    /// this flag, so the cached value cannot go stale. It is not serialized:
291    /// like the gain itself, it is a host preference re-applied on load.
292    pub(crate) gain_is_unity: bool,
293    /// v2.9.8 — the analog output-filter model the host last selected with
294    /// [`Self::set_filter_model`].
295    ///
296    /// The filter itself lives in `blip` as a built chain of coefficients
297    /// (and its IIR history), and a chain does not say which model built it,
298    /// so the selection is kept here as a plain value for
299    /// [`Self::adopt_settings_from`] to carry across a power cycle. Read by
300    /// nothing else: synthesis uses only the chain. A save-state restore
301    /// replaces the chain's coefficients and leaves this alone, so after a
302    /// restore it still names the host's selection, which is what the next
303    /// power cycle should rebuild. NEVER serialized (a UI preference).
304    pub(crate) filter_model: crate::mixer::FilterModel,
305    /// v2.1.6 "Expansion Audio" — the most recent RAW external / on-cart
306    /// expansion-audio sample fed into [`Self::tick_with_external`] (BEFORE the
307    /// UI [`Self::channel_gain`] `[5]` re-weight), retained purely so the
308    /// frontend Audio Mixer panel can plot an expansion-channel oscilloscope /
309    /// VU meter alongside the five base-channel DAC taps.
310    ///
311    /// This is a WRITE-ONLY-from-synthesis, READ-ONLY-to-observers copy: it is
312    /// assigned once per tick and is never read back into the mixer, the IRQ
313    /// path, or any determinism-relevant state, and it is NEVER serialized into
314    /// the save state. It therefore cannot perturb the deterministic per-frame
315    /// audio — the visualization samples a copy, exactly like the base-channel
316    /// `*_out()` DAC accessors already do.
317    pub(crate) last_external: f32,
318
319    /// v2.3.7 "Overtone" — audio provenance, behind ONE pointer.
320    ///
321    /// `None` until armed via [`Apu::set_audio_provenance`]. Consolidated into a
322    /// single `Option<Box<..>>` after `apu_throughput` measured +9% on the
323    /// DISARMED path with this state spread across four inline fields — see
324    /// `crate::provenance::AudioProvenance`.
325    #[cfg(feature = "debug-hooks")]
326    pub(crate) audio_prov: Option<alloc::boxed::Box<crate::provenance::AudioProvenance>>,
327}
328
329/// All [`Apu::channel_mask`] bits set — every channel audible (the default and
330/// the determinism-safe value the oracle / test ROMs always run with).
331pub const CHANNEL_MASK_ALL: u8 = 0x3F;
332
333/// All [`Apu::channel_gain`] entries at `1.0` — every channel at full,
334/// unattenuated output (the default and the byte-identical value the oracle /
335/// test ROMs always run with).
336pub const CHANNEL_GAIN_UNITY: [f32; 6] = [1.0; 6];
337
338impl Apu {
339    /// New APU.
340    #[must_use]
341    pub fn new(region: Region, sample_rate: u32) -> Self {
342        let cpu_rate = match region {
343            Region::Pal => CPU_HZ_PAL,
344            _ => CPU_HZ_NTSC,
345        };
346        // v2.1.5: the frame counter selects PAL vs NTSC sequencer step
347        // positions from the region. Only true `Region::Pal` uses the PAL
348        // (2A07) positions; NTSC and Dendy keep the NTSC (2A03) positions, so
349        // their frame-counter timing is byte-identical to the pre-v2.1.5 model.
350        let mut frame_counter = FrameCounter::new();
351        frame_counter.pal = matches!(region, Region::Pal);
352        Self {
353            region,
354            pulse1: Pulse::new(true),
355            pulse2: Pulse::new(false),
356            triangle: Triangle::new(),
357            noise: Noise::new(region),
358            dmc: Dmc::new(region),
359            frame_counter,
360            mixer: Mixer::new(),
361            blip: BlipBuf::new(sample_rate, cpu_rate),
362            apu_phase: false,
363            dmc_driven_externally: false,
364            put_cycle: false,
365            parity_seed: 0,
366            cpu_cycle: 0,
367            pending_dmc_dma: false,
368            pending_dmc_dma_next: false,
369            dmc_dma_is_load: false,
370            dmc_dma_short: false,
371            defer_dmc_reload_once: false,
372            pending_dmc_abort: false,
373            dmc_abort_delay: 0,
374            dmc_dma_cooldown: 0,
375            reset_4017_delay: 0,
376            reset_4017_value: 0,
377            cannot_run_dmc_dma: 0,
378            dmc_reenable_period_block: false,
379            subpos_arm_countdown: 0,
380            dmc_reload_suppress_outputs: 0,
381            dmc_dma_delay: 0,
382            dmc_dma_addr: 0xC000,
383            dmc_need_halt: false,
384            dmc_need_dummy_read: false,
385            dmc_delayed_4015: 0,
386            dmc_delayed_status: false,
387            dmc_status_applied: false,
388            dmc_set_implicit_abort: false,
389            dmc_implicit_abort: false,
390            dmc_edge_arm_suppress: false,
391            sample_rate,
392            last_frame_events: FrameEvents::default(),
393            channel_mask: CHANNEL_MASK_ALL,
394            channel_gain: CHANNEL_GAIN_UNITY,
395            gain_is_unity: true,
396            filter_model: crate::mixer::FilterModel::NesRf,
397            last_external: 0.0,
398            #[cfg(feature = "debug-hooks")]
399            audio_prov: None,
400        }
401    }
402
403    // -----------------------------------------------------------------
404    // v2.3.7 "Overtone" — audio provenance (output-only, off by default)
405    // -----------------------------------------------------------------
406
407    /// Record one mixed CPU cycle — the OUTLINED half.
408    ///
409    /// Called from both mix paths so the fast default-configuration
410    /// specialization and the gated general path produce the same trace: a
411    /// provenance record that existed on only one of two byte-identical paths
412    /// would be a trap for whoever next changed the other.
413    ///
414    /// # Why `#[cold]` and `#[inline(never)]` are load-bearing
415    ///
416    /// This function is measurement-driven twice over, and the second lesson is
417    /// the less obvious one.
418    ///
419    /// The FIRST version built the `MixRecord` before testing whether
420    /// provenance was armed, so a DISARMED build recomputed all five channel
421    /// outputs every CPU cycle — and `Pulse::output` is not free (it calls
422    /// `muted()`, which calls `sweep_target()`). `apu_throughput` measured
423    /// **+14% to +23%** in the feature-on/arm-off configuration the shipped
424    /// frontend runs. Hoisting the arm check to the top fixed that.
425    ///
426    /// It was NOT enough. With the check first, a quiet-host A/B still measured
427    /// **+7.98% / +2.88% / +11.03%** on the three `apu_throughput` workloads
428    /// (order-bias control: +0.11% / +0.76% / +0.67%, so the deltas are real).
429    /// The absolute costs — +33 µs, +15 µs, +65 µs — are wildly non-uniform,
430    /// which a per-cycle branch cannot produce: a constant branch costs a
431    /// constant number of cycles. The cause was that this body was still being
432    /// INLINED into `tick_with_external`. The five `output()` calls sat in the
433    /// hot function even though the branch skipped over them, inflating it past
434    /// the point where the mixer and the channel ticks kept their registers and
435    /// their I-cache line.
436    ///
437    /// So the hot path now contains exactly one null test, and everything else
438    /// lives out of line behind it. `#[cold]` additionally tells LLVM to lay
439    /// this block out away from the fall-through path. It pessimizes the ARMED
440    /// case, which is the correct trade: armed is an interactive debugging mode
441    /// and disarmed is what every user runs.
442    #[cfg(feature = "debug-hooks")]
443    #[cold]
444    #[inline(never)]
445    fn record_mix_armed(&mut self, mixed: f32, external: f32) {
446        let rec = crate::provenance::MixRecord {
447            mixed,
448            external,
449            pulse1: self.pulse1.output(),
450            pulse2: self.pulse2.output(),
451            triangle: self.triangle.output(),
452            noise: self.noise.output(),
453            dmc: self.dmc.output(),
454        };
455        if let Some(p) = self.audio_prov.as_mut() {
456            p.mix_trace.push(rec);
457        }
458    }
459
460    /// Arm or disarm audio provenance.
461    ///
462    /// Arming allocates both stores; disarming frees them. Mirrors
463    /// `Ppu::set_pixel_provenance`, including that re-arming an already-armed
464    /// APU is a no-op rather than a silent wipe — the frontend re-asserts the
465    /// arm every frame (a lesson from the pixel panel, whose edge-triggered
466    /// mirror desynced permanently the moment a ROM load installed a fresh
467    /// core).
468    #[cfg(feature = "debug-hooks")]
469    pub fn set_audio_provenance(&mut self, enabled: bool) {
470        if enabled {
471            if self.audio_prov.is_none() {
472                self.audio_prov = Some(alloc::boxed::Box::new(
473                    crate::provenance::AudioProvenance::new(),
474                ));
475            }
476        } else {
477            self.audio_prov = None;
478        }
479    }
480
481    /// Whether audio provenance is armed.
482    #[cfg(feature = "debug-hooks")]
483    #[must_use]
484    pub const fn audio_provenance_armed(&self) -> bool {
485        self.audio_prov.is_some()
486    }
487
488    /// The per-register write attribution, or `None` when disarmed.
489    #[cfg(feature = "debug-hooks")]
490    #[must_use]
491    pub fn register_attribution(&self) -> Option<&crate::provenance::RegisterAttribution> {
492        self.audio_prov.as_ref().map(|p| &p.reg_attrib)
493    }
494
495    /// The per-CPU-cycle mix trace, or `None` when disarmed.
496    #[cfg(feature = "debug-hooks")]
497    #[must_use]
498    pub fn mix_trace(&self) -> Option<&crate::provenance::MixTrace> {
499        self.audio_prov.as_ref().map(|p| &p.mix_trace)
500    }
501
502    /// Begin a new frame's mix trace, anchored at `first_cycle`.
503    ///
504    /// The register attribution is deliberately NOT cleared here: "which
505    /// instruction last wrote `$4003`" is a question whose answer legitimately
506    /// predates the current frame, and clearing it every frame would report a
507    /// register nobody has touched this frame as never written.
508    #[cfg(feature = "debug-hooks")]
509    pub fn begin_audio_provenance_frame(&mut self, first_cycle: u64) {
510        if let Some(p) = self.audio_prov.as_mut() {
511            p.mix_trace.clear(first_cycle);
512        }
513    }
514
515    /// Forget the register attribution history. Called on a cold boot, where
516    /// the history it describes genuinely ended.
517    #[cfg(feature = "debug-hooks")]
518    pub fn clear_audio_provenance_history(&mut self) {
519        if let Some(p) = self.audio_prov.as_mut() {
520            p.reg_attrib.clear();
521        }
522    }
523
524    /// Attribute a write in `$4000-$4017` that the bus does NOT route through
525    /// [`Self::write_register`].
526    ///
527    /// Two addresses in the range are not APU registers and are handled
528    /// entirely on the bus: `$4014` (OAM DMA, which arms a burst) and `$4016`
529    /// (controller strobe, which is buffered to the next M2-low boundary).
530    /// `Bus::write` dispatches only `$4000-$4013 | $4015 | $4017` to
531    /// `write_register`, so the attribution recorded there can never see those
532    /// two — yet the table reserves slots for them, because the range is what
533    /// the bus already classifies as an APU write and punching a hole in it
534    /// would invite off-by-one arithmetic at every call site.
535    ///
536    /// Without this entry point those two slots would stay permanently empty
537    /// while the docs claimed they were tracked. This records the cause exactly
538    /// as `write_register` would, and dispatches nothing — the emulation of both
539    /// addresses stays wherever the bus already implements it.
540    #[cfg(feature = "debug-hooks")]
541    pub const fn record_bus_handled_register_write(&mut self, addr: u16, value: u8) {
542        if let Some(p) = self.audio_prov.as_mut() {
543            p.reg_attrib
544                .record(addr, p.attrib_pc, p.attrib_cycle, value);
545        }
546    }
547
548    /// Push the writing instruction's PC + cycle down, mirroring the PPU's
549    /// write-attribution context. Called once per instruction by the core.
550    #[cfg(feature = "debug-hooks")]
551    pub const fn set_attrib_context(&mut self, pc: u16, cycle: u64) {
552        // No-op when disarmed: nothing reads these, so skipping the stores keeps
553        // the disarmed per-instruction cost at one null test.
554        if let Some(p) = self.audio_prov.as_mut() {
555            p.attrib_pc = pc;
556            p.attrib_cycle = cycle;
557        }
558    }
559
560    /// Lift both stores out for a same-timeline restore (run-ahead), leaving the
561    /// APU disarmed. See [`crate::provenance::AudioProvenanceStash`] for why
562    /// this exists at all.
563    #[cfg(feature = "debug-hooks")]
564    #[must_use]
565    pub fn take_audio_provenance(&mut self) -> crate::provenance::AudioProvenanceStash {
566        crate::provenance::AudioProvenanceStash {
567            state: self.audio_prov.take(),
568        }
569    }
570
571    /// Put back stores taken by [`Self::take_audio_provenance`].
572    #[cfg(feature = "debug-hooks")]
573    pub fn put_audio_provenance(&mut self, stash: crate::provenance::AudioProvenanceStash) {
574        self.audio_prov = stash.state;
575    }
576
577    /// Reset (warm).  Per nesdev: most APU state is preserved across reset
578    /// except `$4015` is cleared (channels disabled, DMC silenced).
579    ///
580    /// v2.0.0 beta.3 (A4 cycle-accurate reset, promoted to the only path in
581    /// beta.4): the 2A03
582    /// reset sequence behaves as if the LAST value written to `$4017` were
583    /// written again (blargg `apu_reset` spec) — the retained value is
584    /// re-issued through the normal `$4017` write path (pending mode + the
585    /// 3/4-cycle aligned delay + the mode-1 immediate quarter/half clock),
586    /// and the CPU's 8 clocked reset cycles then age the re-armed counter
587    /// so execution resumes ~9-12 cycles after the effective write (the
588    /// `4017_timing` window).
589    pub fn reset(&mut self) {
590        // Zero the sequencer + IRQ flags now; SCHEDULE the hardware
591        // `$4017` re-write to land 2 clocked cycles into the CPU's
592        // 8-cycle reset sequence (consumed in `tick_with_external`).
593        // Empirically calibrated against blargg `4017_timing`'s printed
594        // "delay after effective $4017 write" (accept window 6..=12,
595        // hardware-usual 9; the ROM quantizes in 2-cycle APU units): an
596        // immediate reset-start re-write measures 12 (the upper edge),
597        // a +3-cycle placement measures 6 (the lower edge), and +2
598        // lands mid-window at 8.
599        let last = self.frame_counter.reset_rewrite_4017();
600        self.reset_4017_value = last;
601        self.reset_4017_delay = 2;
602        self.write_register(0x4015, 0x00);
603        // v2.3.7 — that write went through the ordinary CPU path, which just
604        // attributed it to whatever instruction was last latched. No instruction
605        // caused it: this models the warm-reset silencing of the channels.
606        // Correct the origin so the panel reports hardware rather than naming an
607        // innocent PC. (Caught in review of the PR that added the feature.)
608        #[cfg(feature = "debug-hooks")]
609        if let Some(p) = self.audio_prov.as_mut() {
610            p.reg_attrib.record_reset(0x4015, p.attrib_cycle, 0x00);
611        }
612        self.pending_dmc_dma = false;
613        self.dmc_dma_is_load = false;
614        self.dmc_dma_short = false;
615        self.defer_dmc_reload_once = false;
616        self.pending_dmc_abort = false;
617        self.dmc_abort_delay = 0;
618        self.dmc_dma_cooldown = 0;
619        self.cannot_run_dmc_dma = 0;
620        self.dmc_reenable_period_block = false;
621        self.dmc_reload_suppress_outputs = 0;
622        self.dmc_dma_delay = 0;
623        self.dmc_need_halt = false;
624        self.dmc_need_dummy_read = false;
625        // W3-Stage-3: a warm reset silences the DMC immediately — collapse the
626        // delayed-application machinery to the applied-disabled state (the
627        // `write_register(0x4015, 0)` above latched a deferred disable).
628        {
629            self.dmc_delayed_4015 = 0;
630            self.dmc_delayed_status = false;
631            self.dmc_status_applied = false;
632            self.dmc_set_implicit_abort = false;
633            self.dmc_implicit_abort = false;
634            self.dmc_edge_arm_suppress = false;
635            self.dmc.bytes_remaining = 0;
636        }
637        self.blip.reset();
638    }
639
640    /// Set the per-channel enable mask (a UI playback overlay; see
641    /// [`Apu::channel_mask`]). Bit 0 = pulse 1, 1 = pulse 2, 2 = triangle,
642    /// 3 = noise, 4 = DMC, 5 = external/mapper audio. [`CHANNEL_MASK_ALL`] is
643    /// the determinism-safe default (byte-identical mixer output).
644    pub const fn set_channel_mask(&mut self, mask: u8) {
645        self.channel_mask = mask & CHANNEL_MASK_ALL;
646    }
647
648    /// Current per-channel enable mask.
649    #[must_use]
650    pub const fn channel_mask(&self) -> u8 {
651        self.channel_mask
652    }
653
654    /// v1.4.0 Workstream C — set the per-channel output gain (a UI mixing
655    /// overlay; see [`Apu::channel_gain`]). Index 0 = pulse 1, 1 = pulse 2,
656    /// 2 = triangle, 3 = noise, 4 = DMC, 5 = external/mapper audio. Each gain is
657    /// clamped to `0.0..=2.0`; a NaN (which `f32::clamp` would pass through,
658    /// turning every mixed sample into NaN) becomes unity. [`CHANNEL_GAIN_UNITY`]
659    /// (all `1.0`) is the determinism-safe default (byte-identical mixer output).
660    pub fn set_channel_gain(&mut self, gain: [f32; 6]) {
661        for (slot, g) in self.channel_gain.iter_mut().zip(gain.iter()) {
662            *slot = if g.is_nan() { 1.0 } else { g.clamp(0.0, 2.0) };
663        }
664        self.gain_is_unity = self.channel_gain == CHANNEL_GAIN_UNITY;
665    }
666
667    /// v2.1.3 — select the analog output-filter model (see
668    /// [`crate::mixer::FilterModel`]). Default [`crate::mixer::FilterModel::NesRf`]
669    /// is byte-identical to the pre-v2.1.3 output; the softer models drop the
670    /// aggressive 440 Hz high-pass for a fuller low end. Display/tonal only —
671    /// channel content is unchanged.
672    pub fn set_filter_model(&mut self, model: crate::mixer::FilterModel) {
673        self.filter_model = model;
674        self.blip.set_filter_model(model);
675    }
676
677    /// v2.9.8 — the analog output-filter model last selected with
678    /// [`Self::set_filter_model`] ([`crate::mixer::FilterModel::NesRf`] until
679    /// one is).
680    #[must_use]
681    pub const fn filter_model(&self) -> crate::mixer::FilterModel {
682        self.filter_model
683    }
684
685    /// v2.9.8 — carry the host's settings from `prev` onto this freshly built
686    /// APU, so a power cycle (which rebuilds the APU from [`Self::new`]) keeps
687    /// them.
688    ///
689    /// Until v2.9.8 the bus rebuilt the APU and re-applied only its own wiring
690    /// (the externally driven DMC and the alignment seed), so the channel mask,
691    /// the per-channel gain and the filter model reverted to their defaults
692    /// and every host had to push them again. Carried: [`Self::channel_mask`],
693    /// [`Self::channel_gain`] and [`Self::filter_model`]. The filter goes
694    /// through [`Self::set_filter_model`], which builds a fresh chain for the
695    /// model at this APU's sample rate -- the chain a fresh console gets when
696    /// a host selects the model, with no IIR history carried from the old
697    /// timeline. The sample rate is the caller's to pass to [`Self::new`];
698    /// the audio provenance stores are moved by the core, armed and emptied.
699    ///
700    /// With every setting at its default this leaves the APU byte-identical
701    /// to [`Self::new`]: the default model's chain is the one `new` builds.
702    pub fn adopt_settings_from(&mut self, prev: &Self) {
703        self.channel_mask = prev.channel_mask;
704        // Through the setter, never a field copy: it also refreshes the cached
705        // `gain_is_unity`, which the per-cycle mix reads instead of the gain.
706        // A bare `self.channel_gain = prev.channel_gain` would leave the fresh
707        // APU's `true` in place and mix at unity gain after a power cycle.
708        // `a_power_cycle_keeps_a_non_unity_gain_audible` pins it.
709        self.set_channel_gain(prev.channel_gain);
710        self.set_filter_model(prev.filter_model);
711    }
712
713    /// Current per-channel output gain. See [`Apu::set_channel_gain`].
714    #[must_use]
715    pub const fn channel_gain(&self) -> [f32; 6] {
716        self.channel_gain
717    }
718
719    /// Pulse 1 raw output volume (0..=15) — for tests.
720    #[must_use]
721    pub fn pulse1_out(&self) -> u8 {
722        self.pulse1.output()
723    }
724    /// Pulse 2 raw output volume.
725    #[must_use]
726    pub fn pulse2_out(&self) -> u8 {
727        self.pulse2.output()
728    }
729    /// Triangle raw output (0..=15).
730    #[must_use]
731    pub fn triangle_out(&self) -> u8 {
732        self.triangle.output()
733    }
734    /// Noise raw output (0..=15).
735    #[must_use]
736    pub fn noise_out(&self) -> u8 {
737        self.noise.output()
738    }
739    /// DMC raw output (0..=127).
740    #[must_use]
741    pub const fn dmc_out(&self) -> u8 {
742        self.dmc.output()
743    }
744
745    /// v2.1.6 "Expansion Audio" — the most recent RAW on-cart expansion-audio
746    /// sample (pre-[`Self::channel_gain`], the `last_external` field). `0.0`
747    /// when the loaded board has no expansion audio. Read-only display tap for
748    /// the frontend Audio Mixer expansion-channel scope / VU meter — it reads a
749    /// copy and never feeds back into synthesis, so it is determinism-neutral.
750    #[must_use]
751    pub const fn external_out(&self) -> f32 {
752        self.last_external
753    }
754
755    /// Frame IRQ pending?
756    #[must_use]
757    pub const fn frame_irq_pending(&self) -> bool {
758        self.frame_counter.irq_flag
759    }
760
761    /// DMC IRQ pending?
762    #[must_use]
763    pub const fn dmc_irq_pending(&self) -> bool {
764        self.dmc.irq_flag
765    }
766
767    /// Combined IRQ line — true if either source is asserting.
768    ///
769    /// Session-26 iter 5 (2026-05-23): the frame-counter contribution
770    /// is `irq_line_active` (the CPU's `IRQSource::FrameCounter`
771    /// registration), NOT `irq_flag` (the `$4015` bit 6 visibility).
772    /// The two are SEPARATE fields since iter 5 — see
773    /// [`FrameCounter::irq_flag`](crate::frame_counter::FrameCounter::irq_flag)
774    /// and [`FrameCounter::irq_line_active`](crate::frame_counter::FrameCounter::irq_line_active).
775    /// AccuracyCoin Tests I/J/K specifically test that `$4015` bit 6
776    /// is visible during inhibit (transient 2-cycle window at FC steps
777    /// 29828-29829) while NO CPU IRQ fires (Test M).
778    #[must_use]
779    pub const fn irq_line(&self) -> bool {
780        self.frame_counter.irq_line_active || self.dmc.irq_flag
781    }
782
783    /// Returns the frame-counter events fired by the most recent `tick` call.
784    ///
785    /// The bus reads this immediately after [`Self::tick_with_external`] to
786    /// fan-out the events to on-cart audio extensions (MMC5) whose envelope
787    /// and length-counter sub-units share the 2A03 frame-counter cadence.
788    /// The value is overwritten at the start of every `tick`, so observers
789    /// must consume it before the next tick.
790    #[must_use]
791    pub const fn last_frame_events(&self) -> FrameEvents {
792        self.last_frame_events
793    }
794
795    /// Drain all finalized audio samples (host sample rate, normalized to
796    /// approximately `[-0.5, 0.5]`).
797    pub fn drain_audio(&mut self) -> Vec<f32> {
798        self.blip.drain_all()
799    }
800
801    /// Drain into a slice; returns count copied.
802    pub fn drain_audio_into(&mut self, out: &mut [f32]) -> usize {
803        self.blip.drain(out)
804    }
805
806    /// Has a DMC DMA request been raised?  The bus polls this each CPU cycle
807    /// (BEFORE issuing reads) so it can halt the CPU on the next read cycle.
808    #[must_use]
809    pub const fn dmc_dma_pending(&self) -> bool {
810        self.pending_dmc_dma
811    }
812
813    /// Whether the pending DMC DMA is a load DMA.
814    #[must_use]
815    pub const fn dmc_dma_is_load(&self) -> bool {
816        self.dmc_dma_is_load
817    }
818
819    /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the bus-side per-cycle DMC DMA
820    /// service-gate term — TriCNES `_6502` line 4218:
821    /// `DoDMCDMA && (APU_Status_DMC || APU_ImplicitAbortDMC4015)`. While
822    /// false, a pending (or halted in-flight) DMC DMA is NOT serviced and the
823    /// CPU resumes — the emergent explicit abort. `pending_dmc_abort` is the
824    /// implicit-abort override (the 1-cycle abort DMA runs regardless).
825    #[must_use]
826    pub const fn dmc_dma_serviceable(&self) -> bool {
827        self.dmc_status_applied || self.dmc_implicit_abort || self.pending_dmc_abort
828    }
829
830    /// Whether the pending DMC DMA should use the short 3-cycle service path.
831    #[must_use]
832    pub const fn dmc_dma_short(&self) -> bool {
833        self.dmc_dma_short
834    }
835
836    /// Whether the current DMC DMA get should make the fetched byte visible
837    /// before the get-cycle APU tick.
838    #[must_use]
839    pub const fn dmc_dma_deliver_before_tick(&self) -> bool {
840        self.dmc.loop_flag
841            && self.dmc.sample_length == 1
842            && self.dmc.rate_index == 0x0E
843            && self.dmc.bits_remaining == 1
844            && self.dmc.timer == 0
845            && !self.apu_phase
846    }
847
848    /// Defer the next immediate DMC reload request by one CPU tick.
849    pub const fn defer_next_dmc_reload_once(&mut self) {
850        self.defer_dmc_reload_once = true;
851    }
852
853    /// Has a one-cycle DMC abort halt been raised?
854    #[must_use]
855    pub const fn dmc_abort_pending(&self) -> bool {
856        self.pending_dmc_abort
857    }
858
859    /// Read-only accessor for the DMC abort-delay countdown (CPU cycles
860    /// until `pending_dmc_abort` flips to `true`).  Exposed for the
861    /// Session-21 per-cycle DMC trace tooling (`crates/rustynes-core/src/
862    /// irq_trace.rs`) which records the scheduler's calibration state
863    /// for cross-diffing against Mesen2's `NesDmc.cpp`.
864    #[must_use]
865    pub const fn dmc_abort_delay(&self) -> u8 {
866        self.dmc_abort_delay
867    }
868
869    /// Read-only accessor for the DMC DMA cooldown countdown (CPU cycles
870    /// during which a newly-empty sample buffer must NOT raise a new
871    /// DMA request).  See [`Self::dmc_abort_delay`].
872    #[must_use]
873    pub const fn dmc_dma_cooldown(&self) -> u8 {
874        self.dmc_dma_cooldown
875    }
876
877    /// Read-only accessor for the DMC DMA delay countdown (CPU cycles
878    /// until an initial-load DMA after `$4015` enable transitions from
879    /// "armed" to `pending_dmc_dma = true`).  See [`Self::dmc_abort_delay`].
880    #[must_use]
881    pub const fn dmc_dma_delay(&self) -> u8 {
882        self.dmc_dma_delay
883    }
884
885    /// Diagnostic: the DMC channel's internal byte-timer countdown. Exposed
886    /// for the per-cycle DMC-DMA cross-diff tracing that pins the abort-context
887    /// reload-arm phase (the +4-cycle `A->B` interval divergence).
888    #[must_use]
889    pub const fn dmc_timer(&self) -> u16 {
890        self.dmc.timer()
891    }
892
893    /// Diagnostic: bits remaining in the DMC output shift register.
894    #[must_use]
895    pub const fn dmc_bits_remaining(&self) -> u8 {
896        self.dmc.bits_remaining()
897    }
898
899    /// Diagnostic: DMC output-unit silence flag.
900    #[must_use]
901    pub const fn dmc_silence(&self) -> bool {
902        self.dmc.silence()
903    }
904
905    /// Diagnostic: DMC sample buffer occupied.
906    #[must_use]
907    pub const fn dmc_buffer_full(&self) -> bool {
908        self.dmc.buffer_full()
909    }
910
911    /// Read-only accessor for the APU's two-cycle phase counter
912    /// (false = put, true = get; toggled every CPU tick by
913    /// `tick_with_external`).  See [`Self::dmc_abort_delay`].
914    #[must_use]
915    pub const fn apu_phase(&self) -> bool {
916        self.apu_phase
917    }
918
919    /// v2.0.0 beta.1 (A1 one-clock collapse): assign the APU's cycle counter
920    /// from the CANONICAL bus cycle counter. Called by the bus's per-cycle
921    /// hook (`cpu_clock` → `apu_advance_one`) immediately before
922    /// [`Self::tick_with_external`], replacing the legacy independent
923    /// `cpu_cycle += 1` mirror. The bus increments its canonical counter
924    /// earlier in the same per-cycle hook, so the value assigned here equals
925    /// the post-increment value the legacy mirror produced — the
926    /// `one_clock_invariants` harness test pins the residue. Promoted to the
927    /// only path in v2.0.0 beta.4.
928    pub const fn set_canonical_cycle(&mut self, cycle: u64) {
929        self.cpu_cycle = cycle;
930    }
931
932    /// Read-only accessor for the APU-side cumulative CPU-cycle counter
933    /// (v2.0.0-beta.1 one-clock instrumentation).
934    ///
935    /// This is one of the five counters of the timebase substrate the
936    /// v2.0.0 "Timebase" rewrite collapses (ADR 0002 + the v2.0.0
937    /// master-clock plan): `Cpu::master_clock`, `Cpu::cycles`,
938    /// `SystemBus::cycle`, `SystemBus::ppu_clock`, and this field are
939    /// each advanced exactly once (or by one region divider) per CPU cycle
940    /// at different points *within* the cycle, and must never drift. The
941    /// RW-1 parity collapse already derives `apu_phase` / `put_cycle` from
942    /// `(cpu_cycle + parity_seed) & 1`; exposing the raw counter lets the
943    /// test harness assert the cross-chip affine invariants
944    /// (`one_clock_invariants.rs`) that gate the beta.1 counter collapse.
945    #[must_use]
946    pub const fn cpu_cycle(&self) -> u64 {
947        self.cpu_cycle
948    }
949
950    /// CM-1: seed the absolute `apu_phase` alignment (the parity of the CPU
951    /// cycles on which the APU — incl. the DMC byte-timer — clocks). RustyNES
952    /// starts `apu_phase = false`, so the DMC always arms on one CPU-cycle
953    /// parity; Mesen's DMC arms on its `_currentCycle` alignment (one cycle off,
954    /// giving span 3 vs RustyNES's 4). Seeding `true` flips the whole APU phase
955    /// by one CPU cycle to test the Mesen-matching arm parity. Broad impact:
956    /// also shifts pulse/noise/frame-counter. Default-off.
957    pub const fn seed_apu_phase(&mut self, phase: bool) {
958        self.apu_phase = phase;
959    }
960
961    /// Address the DMC wants to read.  Valid only when `dmc_dma_pending()`
962    /// returns `true`.
963    #[must_use]
964    pub const fn dmc_dma_addr(&self) -> u16 {
965        self.dmc_dma_addr
966    }
967
968    /// v1.2 Sprint 3 (get/put scheduler, ADR-0007).
969    ///
970    /// Returns `true` while the DMC still needs an initial halt
971    /// cycle on the bus. Set by any code path that raises
972    /// `pending_dmc_dma`; the new `bus::service_dmc_dma`
973    /// implementation under the `dmc-get-put-scheduler` feature
974    /// flag clears it after processing the halt get-cycle.
975    #[must_use]
976    pub const fn dmc_need_halt(&self) -> bool {
977        self.dmc_need_halt
978    }
979
980    /// v1.2 Sprint 3 (get/put scheduler, ADR-0007).
981    ///
982    /// Returns `true` while the DMC still needs a dummy-read /
983    /// alignment cycle after the halt cycle. Cleared by the new
984    /// scheduler once the alignment cycle has been processed.
985    #[must_use]
986    pub const fn dmc_need_dummy_read(&self) -> bool {
987        self.dmc_need_dummy_read
988    }
989
990    /// v1.2 Sprint 3 — bus clears this after consuming the halt
991    /// get-cycle (`_needHalt = false` in Mesen2's `NesCpu.cpp`).
992    pub const fn clear_dmc_need_halt(&mut self) {
993        self.dmc_need_halt = false;
994    }
995
996    /// v1.2 Sprint 3 — bus clears this after consuming the
997    /// alignment cycle (`_needDummyRead = false` in Mesen2's
998    /// `NesCpu.cpp`).
999    pub const fn clear_dmc_need_dummy_read(&mut self) {
1000        self.dmc_need_dummy_read = false;
1001    }
1002
1003    /// Bus calls this when it has executed a DMC DMA fetch (post-halt) and
1004    /// is delivering the sample byte.
1005    pub fn complete_dmc_dma(&mut self, byte: u8) {
1006        self.dmc.deliver_sample(byte);
1007        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `DMCDMA_Get`
1008        // (Emulator.cs:4148-4160) — a non-looping sample's natural end clears
1009        // the APPLIED status immediately (not via the delayed slot). A
1010        // looping end restarts the sample (`deliver_sample` already did), so
1011        // `bytes_remaining > 0` and the status holds.
1012        if self.dmc.bytes_remaining == 0 && !self.dmc.loop_flag {
1013            self.dmc_status_applied = false;
1014        }
1015        let was_load = self.dmc_dma_is_load;
1016        // v2.0 abort-context reload-arm phase fix (`mc-r1-dmc-abort-timer-phase`).
1017        // In the Implicit-DMA-Abort `$4015` disable->re-enable context, this LOAD
1018        // DMA's `deliver_sample` (just above) lands ON the byte-timer boundary
1019        // cycle, where `clock_output` already ran at cycle-START and took the
1020        // still-empty buffer (silence) BEFORE the LOAD filled it. TriCNES's LOAD
1021        // GET completes 3 cyc BEFORE the boundary, so the boundary consumes the
1022        // buffer into the shifter and arms a RELOAD (inserting a 4-cyc reload DMA
1023        // RustyNES otherwise skips, deferring the reload chain by 4 -> A->B 580
1024        // not 576 -> GET-catch skew +4 -> Y=0). Detect the boundary-coincidence
1025        // (silence set + bits just reloaded to 8 + buffer now full + bytes
1026        // remaining) and retroactively load the delivered byte into the shifter
1027        // (un-silence) so the buffer empties and the per-cycle reload-arm fires
1028        // promptly this cycle (cooldown cleared below) — reproducing TriCNES's
1029        // boundary-coupled reload. The condition only ever holds in this race.
1030        let abort_boundary_race = was_load
1031            && self.dmc.silence
1032            && self.dmc.bits_remaining == 8
1033            && self.dmc.bytes_remaining > 0
1034            && self.dmc.consume_buffer_into_shifter_if_silent();
1035        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): this load-completion abort
1036        // scheduling is the FLOOR's implicit-abort model (a pre-computed
1037        // 1-cycle halt via `dmc_abort_delay` -> `pending_dmc_abort` -> the
1038        // read1 abort-cancel path). Under the delayed-status port the same
1039        // physics is EMERGENT (the `$4015`-enable pre-fire-window latch + the
1040        // consume-edge arm + the 1-cycle override kill), so the floor
1041        // scheduler is superseded — both active would double-fire ($500
1042        // idx[10,11] measured 05 vs KEY 01).
1043        self.pending_dmc_dma = false;
1044        self.dmc_dma_is_load = false;
1045        self.dmc_dma_short = false;
1046        self.dmc_dma_delay = 0;
1047        self.dmc_dma_cooldown = 4;
1048        // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES `DMCDMA_Get`
1049        // (`Emulator.cs:4168`) sets `CannotRunDMCDMARightNow = 2` after EVERY
1050        // DMC GET (load or reload). The exclusion is decremented by 2 per get
1051        // cycle in `tick_with_external` and blocks the looping-reload arm while
1052        // `== 2` — the canonical "a DMA cannot occur within 2 cycles of a
1053        // previous DMC DMA" rule the Implicit-DMA-Abort `$540` plateau brackets.
1054        self.cannot_run_dmc_dma = 2;
1055        // Abort-context fix: the LOAD just emptied the buffer into the shifter
1056        // (boundary race), so the next reload must arm promptly — don't let the
1057        // post-LOAD cooldown suppress it for 4 cycles (which would re-introduce
1058        // the +4). Clear the cooldown so `dmc_step_reload_arm` fires next cycle.
1059        if abort_boundary_race {
1060            self.dmc_dma_cooldown = 0;
1061        }
1062        // v1.2 Sprint 3 — safety-clear the get/put flags on
1063        // completion. Under the new scheduler the bus should have
1064        // cleared them on the prior cycles already; clearing here
1065        // protects against re-arming on the next DMC request.
1066        self.dmc_need_halt = false;
1067        self.dmc_need_dummy_read = false;
1068        // W3-Stage-2 (`mc-r1-dma-unified-collapse`): this guard's phase term
1069        // means "the GET landed on the off-phase half". The normal GET half is
1070        // `apu_phase`-true at floor but `apu_phase`-false under the end-flip,
1071        // so the off-phase test inverts — otherwise the (floor-dead) suppress
1072        // path would fire at EVERY collapse GET in the 1-byte-loop contexts.
1073        let off_phase_get = self.apu_phase;
1074        if was_load
1075            && self.dmc.loop_flag
1076            && self.dmc.sample_length == 1
1077            && self.dmc.bits_remaining == 1
1078            && self.dmc.timer == 0
1079            && self.dmc.sample_buffer.is_some()
1080            && off_phase_get
1081        {
1082            self.dmc_reload_suppress_outputs = 1;
1083        }
1084    }
1085
1086    /// Complete a DMC DMA get whose fetched byte is visible before the
1087    /// get-cycle APU tick.
1088    pub fn complete_dmc_dma_before_get_tick(&mut self, byte: u8) {
1089        let was_load = self.dmc_dma_is_load;
1090        self.dmc.deliver_sample(byte);
1091        // W3-Stage-3: see `complete_dmc_dma` — non-looping natural end clears
1092        // the applied status immediately.
1093        if self.dmc.bytes_remaining == 0 && !self.dmc.loop_flag {
1094            self.dmc_status_applied = false;
1095        }
1096        if was_load
1097            && self.dmc.bytes_remaining == 0
1098            && !self.dmc.loop_flag
1099            && self.dmc.bits_remaining == 1
1100            && self.dmc.timer == 0
1101            && self.dmc.sample_buffer.is_some()
1102        {
1103            self.dmc_abort_delay = 3;
1104        }
1105        if was_load && self.dmc.loop_flag && self.dmc.sample_length == 1 {
1106            self.defer_dmc_reload_once = true;
1107        }
1108        self.pending_dmc_dma = false;
1109        self.dmc_dma_is_load = false;
1110        self.dmc_dma_short = false;
1111        self.dmc_dma_delay = 0;
1112        self.dmc_dma_cooldown = 5;
1113        // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): see `complete_dmc_dma`.
1114        {
1115            self.cannot_run_dmc_dma = 2;
1116        }
1117        self.dmc_need_halt = false;
1118        self.dmc_need_dummy_read = false;
1119    }
1120
1121    /// Bus calls this after either consuming or suppressing a one-cycle DMC
1122    /// abort halt.
1123    pub const fn complete_dmc_abort(&mut self) {
1124        self.pending_dmc_abort = false;
1125        self.dmc_abort_delay = 0;
1126    }
1127
1128    /// v1.2 Sprint 3 iter 3 (get/put scheduler, ADR 0007) — DMC DMA
1129    /// abort with cancel semantics.
1130    ///
1131    /// Under the OLD scheduler, [`Self::complete_dmc_abort`] clears
1132    /// only the abort flag; the DMC DMA still fires afterward (the
1133    /// abort just inserts a 1-cycle halt). Under the get/put model
1134    /// the abort CANCELS the DMA entirely — no byte fetch, all
1135    /// flag state cleared — matching Mesen2's
1136    /// `processCycle::if(_abortDmcDma)` branch
1137    /// (`NesCpu.cpp:386-390`):
1138    ///
1139    /// ```text
1140    /// if(_abortDmcDma) {
1141    ///     _dmcDmaRunning = false;
1142    ///     _abortDmcDma = false;
1143    ///     _needDummyRead = false;
1144    ///     _needHalt = false;
1145    /// }
1146    /// ```
1147    ///
1148    /// This is the "Option C" semantic shift from the iter 3
1149    /// research audit: abort cancels the fetch rather than letting
1150    /// it complete after a wasted cycle. The new bus-side
1151    /// `service_dmc_dma` (under `dmc-get-put-scheduler` feature)
1152    /// calls this when it detects `dmc_abort_pending` mid-loop.
1153    pub const fn cancel_dmc_dma(&mut self) {
1154        self.pending_dmc_dma = false;
1155        self.pending_dmc_abort = false;
1156        self.dmc_abort_delay = 0;
1157        self.dmc_dma_short = false;
1158        self.dmc_dma_is_load = false;
1159        self.dmc_dma_delay = 0;
1160        self.dmc_need_halt = false;
1161        self.dmc_need_dummy_read = false;
1162    }
1163
1164    /// One CPU clock.  Bus must NOT have halted the CPU for DMC DMA when
1165    /// calling this (the bus is responsible for performing the DMA fetch
1166    /// before resuming `tick()` calls).
1167    ///
1168    /// Standalone/test convenience: production (`SystemBus`) drives the
1169    /// canonical cycle counter via [`Self::set_canonical_cycle`] before each
1170    /// [`Self::tick_with_external`] (the v2.0.0 one-clock contract — the APU
1171    /// never self-increments). This helper self-advances the counter so
1172    /// standalone APU stepping (unit tests, the snapshot fixtures) keeps the
1173    /// one-cycle-per-tick behavior.
1174    pub fn tick(&mut self) {
1175        self.cpu_cycle = self.cpu_cycle.wrapping_add(1);
1176        self.tick_with_external(0.0);
1177    }
1178
1179    /// Same as `tick`, but accepts an additional pre-mixed audio sample
1180    /// from the cartridge (VRC6 / VRC7 / MMC5 / Sunsoft 5B / Namco 163 /
1181    /// FDS). The external value is added to the APU's own mix BEFORE the
1182    /// band-limited buffer push.
1183    ///
1184    /// The expected scale is ~ `[-0.5, 0.5]` (matching the APU mixer's
1185    /// own output range). The bus is responsible for converting whatever
1186    /// the mapper returns (currently `i16` from `Mapper::mix_audio`) into
1187    /// that range.
1188    pub fn tick_with_external(&mut self, external: f32) {
1189        // v2.0.0 beta.1 (A1 one-clock collapse, promoted to the only path in
1190        // beta.4): the APU's cycle counter is ASSIGNED from the canonical
1191        // bus counter (see `set_canonical_cycle`, called by the bus
1192        // immediately before this tick) instead of being an
1193        // independently-incremented lockstep mirror. The RW-1
1194        // `apu_phase`/`put_cycle` parity derivation below then reads from
1195        // the ONE counter.
1196
1197        // v2.0 RA-1: the DMC byte-timer + arms could clock HERE at cycle START
1198        // (on `apu_phase`), unified with the rest of the APU — Mesen
1199        // `ProcessCpuClock` at `StartCpuCycle`.
1200        //
1201        // v2.0 Program M (M-1 within-cycle order): the DMC byte-timer CLOCK +
1202        // reload-arm + reenable bookkeeping live at end-of-cycle (after the CPU's
1203        // bus access, in `dmc_tick_end`), matching Mesen `StartCpuCycle`->
1204        // `ProcessCpuClock` and TriCNES `_6502`->`_EmulateAPU` (CPU reads state ->
1205        // APU ticks/arms reload -> get/put flips). The reload arm thereby becomes
1206        // invisible to its own cycle -> first-service is the next (put) cycle ->
1207        // span-4. So the cycle-START DMC clock/arm paths below are never taken;
1208        // the LOAD delay-arm moves to the put phase of `dmc_tick_end` (the TriCNES
1209        // `DMCDMADelay` put-branch placement, Emulator.cs:1217).
1210
1211        if self.dmc_abort_delay > 0 {
1212            self.dmc_abort_delay -= 1;
1213            if self.dmc_abort_delay == 0 && !self.pending_dmc_abort {
1214                self.pending_dmc_abort = true;
1215            }
1216        }
1217        if self.dmc_dma_cooldown > 0 {
1218            self.dmc_dma_cooldown -= 1;
1219        }
1220
1221        // Triangle clocks at CPU rate.
1222        self.triangle.clock_timer();
1223
1224        // Pulse, noise, DMC clock at APU rate (every other CPU cycle).
1225        // RW-1 (`mc-r1-one-clock`): DERIVE `apu_phase` from the single per-cycle
1226        // counter + boot seed instead of a free-running toggle, so it shares ONE
1227        // source with `put_cycle` (and thus the DMA get/put parity + DMC
1228        // fire-phase) and can never drift. `cpu_cycle` was incremented above, so
1229        // `(cpu_cycle + parity_seed) & 1 == 1` reproduces the toggle-from-`false`
1230        // sequence exactly when `parity_seed == 0` (the floor config).
1231        {
1232            self.apu_phase = (self.cpu_cycle.wrapping_add(self.parity_seed) & 1) == 1;
1233        }
1234        // v2.0.0 beta.3 (A4 cycle-accurate reset): consume the scheduled
1235        // warm-reset `$4017` re-write N clocked cycles into the CPU's reset
1236        // sequence (see `Apu::reset` for the calibration). Runs after the
1237        // `apu_phase` derivation so the write's 3/4-cycle alignment delay
1238        // reads the current cycle's parity, exactly like a CPU-issued write.
1239        if self.reset_4017_delay > 0 {
1240            self.reset_4017_delay -= 1;
1241            if self.reset_4017_delay == 0 {
1242                let aligned = self.apu_phase;
1243                self.frame_counter.write(self.reset_4017_value, aligned);
1244            }
1245        }
1246        if self.apu_phase {
1247            self.pulse1.clock_timer();
1248            self.pulse2.clock_timer();
1249            self.noise.clock_timer();
1250            // F-2/M-1: the DMC byte-timer clock lives in `tick_dmc`
1251            // (end-of-cycle), not here at cycle START.
1252        }
1253
1254        // Frame counter (CPU clock). Latch the events so the bus can fan
1255        // them out to on-cart audio extensions (MMC5) after the tick.
1256        // Pass `apu_phase` AND `cpu_cycle` so the frame counter can
1257        // (a) compute APU-step timing as before and (b) mature any
1258        // pending lazy `$4015`-read IRQ-flag clear scheduled by a
1259        // previous read (Session-25, 2026-05-23 — see
1260        // `frame_counter::read_status` doc).
1261        let ev = self.frame_counter.tick(self.cpu_cycle, self.apu_phase);
1262        self.last_frame_events = ev;
1263        self.handle_frame_events(ev);
1264
1265        // v2.1.5 length halt/reload ordering: promote each channel's deferred
1266        // halt (`new_halt` -> `halt`) and pending length reload EVERY CPU cycle,
1267        // AFTER the half-frame clock in `handle_frame_events` and BEFORE the
1268        // mixer samples the channel outputs below. This realizes the 2A03's
1269        // "halt change takes effect after clocking length" and "reload ignored
1270        // during a non-zero length clock" rules (blargg `10.len_halt_timing` /
1271        // `11.len_reload_timing`; TetaNES `LengthCounter::reload` +
1272        // Mesen2 `_newHaltValue`). On the common cycle with no half-frame clock
1273        // the reload applies in-cycle (the count was untouched since the write),
1274        // so a plain length load / halt write remains byte-identical to an
1275        // immediate apply — only the write-lands-on-the-clock-cycle coincidence
1276        // the tests probe differs. The DMC has no length counter. See
1277        // `crates/rustynes-apu/src/length.rs`.
1278        self.pulse1.length.reload();
1279        self.pulse2.length.reload();
1280        self.triangle.length.reload();
1281        self.noise.length.reload();
1282
1283        // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`) reload-arm/reenable
1284        // bookkeeping and the `CannotRunDMCDMARightNow` exclusion decrement all
1285        // live at end-of-cycle (`dmc_tick_end`) under M-1, NOT here at cycle
1286        // START.
1287
1288        // Emit one mixed sample to the band-limited buffer. The external
1289        // (cartridge) audio is summed AFTER the internal non-linear mixer
1290        // since it's already a linear value.
1291        // Per-channel mute overlay. With the default `CHANNEL_MASK_ALL` every
1292        // `gate(..)` returns the raw output unchanged, so this is byte-identical
1293        // to the un-masked mix (the determinism contract — the oracle / test
1294        // ROMs never clear a bit). A cleared bit forces that channel's raw
1295        // output to 0 BEFORE the non-linear mixer, so it contributes nothing.
1296        let mask = self.channel_mask;
1297
1298        // v2.3.5 C1 — the DEFAULT-CONFIGURATION fast path.
1299        //
1300        // Every gate/scale below is inert at the shipped default: the
1301        // determinism contract says the oracle and the test ROMs never clear a
1302        // mask bit or change a gain, so `gate` returns its input unchanged and
1303        // `scale` returns `round(v * 1.0) == v`. The emulator was still paying,
1304        // every CPU cycle at 1.789 MHz, for a 6-wide `f32` array copy, five
1305        // integer mask tests, five float compares, and a sixth mask test for the
1306        // external sum -- to produce a result identical to the ungated mix.
1307        //
1308        // Same shape as the PPU fast dot path, which is the one core
1309        // optimization this project has adopted: hoist the
1310        // "is-this-the-default?" question out of the per-cycle body and take a
1311        // branch with none of the machinery. It is a strict specialization, not
1312        // an approximation -- `mix()` receives exactly the same five arguments
1313        // it would have received, so the output is byte-identical by
1314        // construction rather than by measurement. `apu_default_mix_matches_the_gated_path`
1315        // pins that across a 2,048-point sweep anyway.
1316        if mask == CHANNEL_MASK_ALL && self.gain_is_unity {
1317            self.last_external = external;
1318            let mixed = self.mixer.mix(
1319                self.pulse1.output(),
1320                self.pulse2.output(),
1321                self.triangle.output(),
1322                self.noise.output(),
1323                self.dmc.output(),
1324            ) + external;
1325            #[cfg(feature = "debug-hooks")]
1326            if self.audio_prov.is_some() {
1327                self.record_mix_armed(mixed, external);
1328            }
1329            self.blip.add_sample(mixed);
1330            // Nothing follows the general path's `add_sample` but comments --
1331            // the get/put flip moved to `dmc_tick_end` under M-2 -- so there is
1332            // no shared tail to run before returning. Verified by reading it,
1333            // not assumed: a missed tail here would desynchronise the two paths.
1334            return;
1335        }
1336
1337        let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
1338        // v1.4.0 Workstream C — per-channel gain (a UI mixing overlay). With the
1339        // default `CHANNEL_GAIN_UNITY` every `scale(..)` returns `round(v * 1.0)
1340        // == v` and `external * 1.0 == external`, so this is byte-identical to
1341        // the pre-gain mix (the determinism contract — the oracle / test ROMs
1342        // never change a gain). A gain != 1.0 scales that channel's contribution
1343        // before the non-linear mixer (gain 0.0 == a cleared mask bit). The
1344        // `gain` slice is checked-for-unity-and-skipped so the default path is
1345        // the exact integer-gate code as before.
1346        let gain = self.channel_gain;
1347        // `max` is the channel's native raw ceiling (pulse/tri/noise = 15, DMC =
1348        // 127); the scaled value is clamped to it so the non-linear mixer's
1349        // `pulse_table` (31) / `tnd_table` (203) index bounds always hold even at
1350        // gain 2.0. At gain 1.0 the value is returned unchanged (byte-identical).
1351        let scale = |bit: usize, v: u8, max: u8| {
1352            let g = gain[bit];
1353            if g == 1.0 {
1354                v
1355            } else {
1356                #[allow(
1357                    clippy::cast_possible_truncation,
1358                    clippy::cast_sign_loss,
1359                    clippy::cast_precision_loss
1360                )]
1361                {
1362                    roundf(f32::from(v) * g).clamp(0.0, f32::from(max)) as u8
1363                }
1364            }
1365        };
1366        // v2.1.6 — stash the RAW (pre-gain) external contribution for the
1367        // frontend expansion-channel scope/VU. Write-only from synthesis; never
1368        // read back into the mix, so it cannot alter deterministic output.
1369        self.last_external = external;
1370        let ext_gain = gain[5];
1371        let ext = if ext_gain == 1.0 {
1372            external
1373        } else {
1374            external * ext_gain
1375        };
1376        let mixed = self.mixer.mix(
1377            scale(0, gate(0, self.pulse1.output()), 15),
1378            scale(1, gate(1, self.pulse2.output()), 15),
1379            scale(2, gate(2, self.triangle.output()), 15),
1380            scale(3, gate(3, self.noise.output()), 15),
1381            scale(4, gate(4, self.dmc.output()), 127),
1382        ) + if mask & (1 << 5) != 0 { ext } else { 0.0 };
1383        #[cfg(feature = "debug-hooks")]
1384        if self.audio_prov.is_some() {
1385            // RAW `external`, not the gained `ext`, and not zero when the mask
1386            // bit clears it. The five channel fields are already the raw
1387            // pre-gate outputs, so recording a gain-scaled or mask-zeroed
1388            // expansion value would make ONE field follow the user's mixer
1389            // sliders while five describe the chip -- and would make this path
1390            // disagree with the fast path, which records the raw value. Review
1391            // caught the disagreement; this resolves it toward the documented
1392            // semantic rather than toward the local variable that happened to
1393            // be in scope.
1394            self.record_mix_armed(mixed, external);
1395        }
1396        self.blip.add_sample(mixed);
1397
1398        // v2.0 interleaved-DMA Phase A: toggle the global get/put flip-flop once
1399        // per CPU cycle, right after the APU tick (TriCNES `APU_PutCycle =
1400        // !APU_PutCycle` after `_EmulateAPU()`, `Emulator.cs:920`). Gated on
1401        // `dmc_driven_externally` so the default build never touches it
1402        // (byte-identical); under the R1 substrate this is the single
1403        // per-cycle get/put counter the interleaved DMA (Phase B) consumes.
1404        // RW-1 (`mc-r1-one-clock`): `put_cycle` is the COMPLEMENT of `apu_phase`,
1405        // derived from the same counter — not a second independent flip-flop.
1406        // In the floor config the two toggles already stayed perfectly
1407        // complementary (both flip once per `tick_with_external`); RW-1 makes
1408        // that structural so RW-2 has a SINGLE place to make the parity
1409        // OAM-DMA-aware. The bus's get/put decision (`get = !put_cycle`) and the
1410        // F-2 DMC clock (`!put_cycle`) then read this coherent value.
1411        // M-2 (`mc-r1-counter-collapse`): the get/put `put_cycle` flip moves to
1412        // END of the CPU cycle (`dmc_tick_end`), AFTER the bus access — the
1413        // references' "access -> APU tick -> get/put flip" order. So at the START
1414        // (here) `put_cycle` is LEFT at its prior-cycle value; the bus access this
1415        // cycle therefore reads `put_cycle = !apu_phase_{N-1} = apu_phase_N`,
1416        // one parity position later than the floor's `!apu_phase_N`. `apu_phase`
1417        // itself (the APU IRQ line / C1 phi2 sample source) still flips at start
1418        // (line ~927), so C1 is invariant.
1419    }
1420
1421    fn handle_frame_events(&mut self, ev: FrameEvents) {
1422        if ev.quarter {
1423            self.pulse1.clock_quarter_frame();
1424            self.pulse2.clock_quarter_frame();
1425            self.triangle.clock_quarter_frame();
1426            self.noise.clock_quarter_frame();
1427        }
1428        if ev.half {
1429            self.pulse1.clock_half_frame();
1430            self.pulse2.clock_half_frame();
1431            self.triangle.clock_half_frame();
1432            self.noise.clock_half_frame();
1433        }
1434    }
1435
1436    /// Visibility-delay promotion (called at END of cycle, after the CPU's bus
1437    /// access): a reload latched this cycle becomes visible to the NEXT cycle's
1438    /// DMA servicing (first-service on the put cycle => span 4), matching
1439    /// TriCNES `_EmulateAPU`-after-`_6502` invisible-arm ordering.
1440    pub fn promote_dmc_pending_next(&mut self) {
1441        if self.pending_dmc_dma_next {
1442            self.pending_dmc_dma_next = false;
1443            self.pending_dmc_dma = true;
1444        }
1445    }
1446
1447    /// v2.0 Program M (M-1 within-cycle order, `mc-r1-dmc-bytetimer-end`): clock
1448    /// the DMC byte-timer + arm the reload at END of cycle (after the CPU's bus
1449    /// access), the mirror of the cycle-START block in `tick_with_external` that
1450    /// `dmc_clock_at_start` now suppresses. Order matches `tick_with_external`:
1451    /// byte-timer clock (on this cycle's already-set `apu_phase`) -> reenable
1452    /// consume-edge clear -> reload-arm -> `cannot_run` decrement. The bus calls
1453    /// this from `cpu_clock_apu_dmc` (end-of-cycle), AFTER
1454    /// `promote_dmc_pending_next` so a reload latched here is invisible to its
1455    /// own cycle (promoted -> serviced the NEXT cycle = span-4, like the
1456    /// references). The LOAD delay-arm is NOT here — it stays at cycle-start.
1457    pub fn dmc_tick_end(&mut self) {
1458        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the 1-cycle implicit-abort
1459        // kill — TriCNES clears `APU_ImplicitAbortDMC4015` at the END of
1460        // `_6502` whenever the DMA is pending (Emulator.cs:9000-9003), i.e.
1461        // BEFORE `_EmulateAPU`'s boundary work. A flag set by the previous
1462        // cycle's consume edge therefore survives exactly one CPU access
1463        // (one serviced halt cycle if it was a read; none if a write — "it
1464        // won't run at all") and dies here.
1465        if self.pending_dmc_dma && self.dmc_implicit_abort {
1466            self.dmc_implicit_abort = false;
1467        }
1468        let d4015_bits_before = self.dmc.bits_remaining();
1469        let dmc_bits_before = d4015_bits_before;
1470        // The byte-timer-end flag composes only with the canonical apu_phase
1471        // clock (the `mc-r1-full-cpu` config); the cpu-rate / phase-minus1
1472        // diagnostic clock variants are not combined with it.
1473        // M-2 (`mc-r1-counter-collapse`): the get/put `put_cycle` flip moved to
1474        // end-of-cycle (one parity position later), so the GET decision
1475        // (`get = !put_cycle`) now reads the shifted parity. The DMC byte-timer
1476        // FIRE must follow the SAME shift or the GET de-syncs from the byte-timer
1477        // wrap (wedge). At entry `put_cycle == apu_phase` (the prior end-flip),
1478        // so clocking on `!self.put_cycle == !apu_phase` shifts the byte-timer by
1479        // one to stay locked to the shifted GET — ONE counter driving both.
1480        let timer_phase = !self.put_cycle;
1481        if timer_phase {
1482            self.dmc.clock_timer();
1483        }
1484        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the consume-edge transfer
1485        // (Emulator.cs:1163-1175) — at the shifter-consume edge (bits 1 -> 8
1486        // on this end-tick's byte-timer fire) a latched
1487        // `dmc_set_implicit_abort` becomes the live `dmc_implicit_abort`
1488        // service-gate override AND arms the DMA directly (TriCNES
1489        // `if (BytesRemaining > 0 || SetImplicit) { if (!DoDMCDMA &&
1490        // CannotRun != 2) { DoDMCDMA = true; Halt = true; } ... }` — the arm
1491        // fires regardless of the buffer state). The armed DMA runs for
1492        // exactly one read cycle under the override (the kill above), then
1493        // waits for the delayed status — the emergent 1-cycle implicit abort.
1494        if timer_phase
1495            && self.dmc_set_implicit_abort
1496            && self.dmc.bits_remaining() == 8
1497            && d4015_bits_before <= 1
1498        {
1499            self.dmc_implicit_abort = true;
1500            self.dmc_set_implicit_abort = false;
1501            if !self.pending_dmc_dma && self.cannot_run_dmc_dma != 2 {
1502                self.pending_dmc_dma = true;
1503                self.dmc_dma_is_load = false;
1504                self.dmc_dma_short = false;
1505                self.dmc_dma_addr = self.dmc.dma_addr();
1506                self.dmc_need_halt = true;
1507                self.dmc_need_dummy_read = true;
1508            }
1509        }
1510        // W3-Stage-2 (`mc-r1-dma-unified-collapse`): the TriCNES `DMCDMADelay`
1511        // put-branch — the `$4015`-enable LOAD delay counts down ONLY on the
1512        // put phase of this end-of-cycle tick (Emulator.cs:1217 sits in the
1513        // `else` of the get branch), arming the halt at the end of a PUT cycle
1514        // so the load's first halted cycle is always a GET (entry-on-get =
1515        // span 3) regardless of the write cycle's parity. The put phase here
1516        // is `!timer_phase` (the complement of the shifted byte-timer phase).
1517        if !timer_phase {
1518            self.dmc_step_delay_arm_put_end();
1519        }
1520        if self.dmc_reenable_period_block && self.dmc.bits_remaining() == 8 && dmc_bits_before <= 1
1521        {
1522            self.dmc_reenable_period_block = false;
1523        }
1524        // W3-Stage-4 (`mc-r1-dmc-delayed-4015` grid correction): the TriCNES
1525        // reload arm is consume-edge-quantized. A consume edge that lands ON
1526        // the GET-delivery cycle itself (the X=8/9 Implicit `$540` restart
1527        // race: the silent-restart load GET collides with the free-running
1528        // byte-timer boundary) is arm-BLOCKED by `CannotRunDMCDMARightNow ==
1529        // 2` (Emulator.cs:1165; the :1186 decrement runs later that same
1530        // end-tick, so `== 2` is only ever observable at the same-cycle
1531        // edge) — and TriCNES holds NO level request: the chain simply waits
1532        // for the NEXT consume edge (one full byte period). Our `needs_dma()`
1533        // is level-triggered and would re-arm 4 cycles later (cooldown
1534        // expiry) — one grid boundary early, the `$540[8,9]` cliff. Latch the
1535        // suppression at the blocked same-cycle edge; release at the next
1536        // consume edge right here (BEFORE the reload-arm step) so the
1537        // deferred arm fires exactly on-grid, like TriCNES's
1538        // `BytesRemaining > 0` edge arm.
1539        if timer_phase && self.dmc.bits_remaining() == 8 && d4015_bits_before <= 1 {
1540            if self.dmc_edge_arm_suppress {
1541                self.dmc_edge_arm_suppress = false;
1542            } else if self.cannot_run_dmc_dma == 2 && self.dmc.needs_dma() && !self.pending_dmc_dma
1543            {
1544                self.dmc_edge_arm_suppress = true;
1545            }
1546        }
1547        self.dmc_step_reload_arm();
1548        // M-2: the `cannot_run` decrement is TriCNES's get-cycle decrement; under
1549        // the collapse the get cycle is the shifted `timer_phase`, not raw
1550        // apu_phase.
1551        if timer_phase && self.cannot_run_dmc_dma > 0 {
1552            self.cannot_run_dmc_dma = self.cannot_run_dmc_dma.saturating_sub(2);
1553        }
1554        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES
1555        // `APU_DelayedDMC4015` countdown (Emulator.cs:1214-1224) — decremented
1556        // EVERY CPU cycle after the get/put branch work (the byte-timer /
1557        // reload-arm / load-delay above). On expiry the latched `$4015` DMC
1558        // status APPLIES: `APU_Status_DMC = APU_Status_DelayedDMC`, and a
1559        // disable zeroes `bytes_remaining` HERE rather than at the write. The
1560        // bus-side service gate reads `dmc_status_applied` per cycle, so an
1561        // in-flight DMA whose status drops stops being serviced — the
1562        // emergent explicit abort.
1563        if self.dmc_delayed_4015 > 0 {
1564            self.dmc_delayed_4015 -= 1;
1565            if self.dmc_delayed_4015 == 0 {
1566                self.dmc_status_applied = self.dmc_delayed_status;
1567                if !self.dmc_status_applied {
1568                    self.dmc.bytes_remaining = 0;
1569                }
1570            }
1571        }
1572        // M-2 (`mc-r1-counter-collapse`): flip the get/put parity HERE at
1573        // end-of-cycle (after the CPU's bus access + the byte-timer/reload-arm
1574        // tick above), matching the references' "access -> APU tick -> get/put
1575        // flip" order. `put_cycle = !apu_phase` of the cycle that just ran; the
1576        // NEXT cycle's bus access reads this value. (Under bytetimer-end alone
1577        // this flip stays at cycle-start in `tick_with_external`.)
1578        {
1579            self.put_cycle = !self.apu_phase;
1580        }
1581    }
1582
1583    /// W3-Stage-2 (`mc-r1-dma-unified-collapse`): the TriCNES `DMCDMADelay`
1584    /// put-branch body — same arm as [`Self::dmc_step_delay_arm`] but ticked
1585    /// only on the put phase of `dmc_tick_end` (value units = put end-ticks,
1586    /// set to 2 at the `$4015` enable like TriCNES `DMCDMADelay = 2`).
1587    fn dmc_step_delay_arm_put_end(&mut self) {
1588        if self.dmc_dma_delay > 0 {
1589            self.dmc_dma_delay -= 1;
1590            if self.dmc_dma_delay == 0 && !self.pending_dmc_dma {
1591                self.pending_dmc_dma = true;
1592                self.dmc_dma_short = self.dmc_dma_is_load;
1593                self.dmc_dma_addr = self.dmc.dma_addr();
1594                self.dmc_need_halt = true;
1595                self.dmc_need_dummy_read = true;
1596            }
1597        }
1598    }
1599
1600    /// DMC delay-arm step: countdown the load-DMA delay and arm `pending_dmc_dma`
1601    /// when it expires. Extracted from `tick_with_external` so `tick_dmc` (F-2)
1602    /// can run it at end-of-cycle.
1603    fn dmc_step_delay_arm(&mut self) {
1604        if self.dmc_dma_delay > 0 {
1605            self.dmc_dma_delay -= 1;
1606            if self.dmc_dma_delay == 0 && !self.pending_dmc_dma {
1607                self.pending_dmc_dma = true;
1608                self.dmc_dma_short = self.dmc_dma_is_load;
1609                self.dmc_dma_addr = self.dmc.dma_addr();
1610                self.dmc_need_halt = true;
1611                self.dmc_need_dummy_read = true;
1612            }
1613        }
1614    }
1615
1616    /// DMC reload-arm step: arm a reload DMA when the sample buffer empties
1617    /// (subject to cooldown / suppress / defer). Extracted for `tick_dmc` (F-2).
1618    #[allow(clippy::too_many_lines)]
1619    fn dmc_step_reload_arm(&mut self) {
1620        // final lever #1 (`mc-r1-dmc-halt-subpos`): master-clock DMA-halt
1621        // sub-position. The reload byte-timer wraps and arms on the apu_phase
1622        // get cycle (so the CPU recognizes the halt at the NEXT read1 = one CPU
1623        // cycle too late -> the GET lands adjacent to the `LDA $4000` data read,
1624        // which sees the GET's $00 -> Y=3). TriCNES arms one CPU cycle EARLIER so
1625        // the GET preempts the operand-high fetch (re-driving $40 -> Y=4). On the
1626        // `!apu_phase` cycle IMMEDIATELY preceding the wrap, the byte-timer sits
1627        // at `timer==0 && bits_remaining==1` (the final output bit is one
1628        // apu-clock from emptying the byte). Pre-arm `pending_dmc_dma` HERE, one
1629        // CPU cycle early. Scoped EXACTLY to the X=10/11 boundary by the
1630        // `cannot_run_dmc_dma == 2` exclusion (post-LOAD-GET window) — fires 6x,
1631        // nowhere else — so steady-state GETs + SH* are untouched (context-local,
1632        // distinct from a global byte-timer phase shift that shatters SH*).
1633        // The pre-wrap `!apu_phase` cycle that uniquely marks the `$540` X=10/11
1634        // boundary: the reload byte-timer is at `timer==0 && bits_remaining==1`
1635        // (one apu-clock from emptying the byte), the LOAD has just FILLED the
1636        // buffer (`buffer_full` -> needs_dma still FALSE), and we are inside the
1637        // post-LOAD-GET `cannot_run == 2` exclusion. This is distinct from the
1638        // `$500`/`$520` X=10/11 blocks (Key1/Key2, already correct) whose buffer
1639        // is already empty at this point (no `buffer_full` pre-wrap cycle), so
1640        // pre-arming here leaves them untouched. Arm `pending_dmc_dma` one CPU
1641        // cycle early so the wrap-cycle's `read1` recognizes the halt (the GET
1642        // preempts the operand-high fetch -> $40 re-driven -> Y 3->4) instead of
1643        // the next read1 (GET adjacent to the data read -> $00 seen -> Y=3).
1644        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the halt-subpos boundary
1645        // pre-arm is a floor-unit expression of the same missing `$4015`
1646        // application delay (the Stage-2 residual map); under the
1647        // delayed-status port it is superseded by the emergent consume-edge
1648        // arm — both active double-fire on the X=10/11 entries.
1649        // Gate also on the visibility-delay latch so a reload cannot double-arm
1650        // while one is latched-but-not-yet-promoted (would cascade/wedge).
1651        let already = self.pending_dmc_dma || self.pending_dmc_dma_next;
1652        // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES gates the reload
1653        // arm on `CannotRunDMCDMARightNow != 2` (`Emulator.cs:1165`) — a reload
1654        // cannot arm on the get cycle immediately following a DMC GET. That
1655        // exclusion is hit ONLY at the Implicit-DMA-Abort X=10/11 `$4015`
1656        // re-enable boundary (the LOAD GET lands so the next byte-timer wrap
1657        // coincides with the window) — confirmed by the probe firing on exactly
1658        // those two entries. A full-period reload deferral there OVERSHOOTS
1659        // ($540[10,11] -> 00, Y=0) because RustyNES's start-clock + Option-buffer
1660        // structure shifts the whole chain a byte; TriCNES instead realigns the
1661        // byte-timer phase by ~1 cycle. So at the boundary we apply a ONE-SHOT
1662        // swept byte-timer phase shift (`REENABLE_BUMP`, env-tunable) that
1663        // realigns the looping-reload chain like TriCNES's re-enable, while the
1664        // bare `cannot_run == 2` gate still defers this cycle's arm.
1665        let cannot_run_now = self.cannot_run_dmc_dma == 2;
1666        // One-shot byte-timer realignment at the exclusion boundary. `period_block`
1667        // is the one-shot guard (set here, cleared at the next consume edge in
1668        // `tick_with_external`) so the bump is applied exactly once per boundary.
1669        if cannot_run_now
1670            && self.dmc.needs_dma()
1671            && !already
1672            && self.dmc_dma_delay == 0
1673            && self.dmc_reload_suppress_outputs == 0
1674            && self.dmc_dma_cooldown == 0
1675            && !self.defer_dmc_reload_once
1676            && !self.dmc_reenable_period_block
1677        {
1678            let bump = crate::dmc::REENABLE_BUMP.load(core::sync::atomic::Ordering::Relaxed);
1679            if bump != 0 {
1680                self.dmc.bump_timer_phase(bump);
1681            }
1682            self.dmc_reenable_period_block = true;
1683        }
1684        // W3-Stage-4: the consume-edge-quantization suppression (see
1685        // `dmc_tick_end`) — while latched, the level-held `needs_dma()` must
1686        // NOT arm; the deferred arm fires at the next consume edge.
1687        let edge_suppressed = self.dmc_edge_arm_suppress;
1688        if self.dmc.needs_dma()
1689            && !already
1690            && self.dmc_dma_delay == 0
1691            && !cannot_run_now
1692            && !edge_suppressed
1693        {
1694            if self.dmc_reload_suppress_outputs > 0
1695                || self.dmc_dma_cooldown > 0
1696                || self.defer_dmc_reload_once
1697            {
1698                self.defer_dmc_reload_once = false;
1699            } else {
1700                // Visibility-delay: a reload latches into `_next` (promoted next
1701                // cycle) so first-service lands on the put cycle (span 4). Loads
1702                // and the default keep direct `pending_dmc_dma` (first-service get).
1703                {
1704                    self.pending_dmc_dma_next = true;
1705                }
1706                self.dmc_dma_is_load = false;
1707                self.dmc_dma_short = false;
1708                self.dmc_dma_addr = self.dmc.dma_addr();
1709                self.dmc_need_halt = true;
1710                self.dmc_need_dummy_read = true;
1711            }
1712        } else {
1713            self.defer_dmc_reload_once = false;
1714        }
1715    }
1716
1717    /// v2.0 F-2: advance ONLY the DMC byte-timer + DMA arm by one CPU cycle.
1718    /// The R1 bus calls this at END of cycle (after the access) when
1719    /// [`Self::set_dmc_driven_externally`] is set, so the DMC fire-phase matches
1720    /// the pre-v2.0.0 `tick_one_cpu_cycle` (the cycle DMASync's `$4000` conflict
1721    /// expects) while the rest of the APU — incl. the IRQ line — stays on the
1722    /// cycle-start `tick_with_external`. Order mirrors `tick_with_external`:
1723    /// delay-arm → APU-rate timer clock (via the `dmc_ext_phase` flip-flop) →
1724    /// reload-arm.
1725    pub fn tick_dmc(&mut self) {
1726        self.dmc_step_delay_arm();
1727        // Divergence A: clock the DMC byte-timer off the SHARED `put_cycle`
1728        // counter (the same flip-flop the interleaved DMA's get/put decision
1729        // uses) instead of a separate `dmc_ext_phase`, so the DMC fire-phase and
1730        // the get/put parity share ONE seed and can NEVER drift (TriCNES seeds
1731        // `APU_PutCycle` + the DMC timer together). The DMC clocks at the APU
1732        // rate (every other CPU cycle). Polarity `!put_cycle`: main clocks the
1733        // DMC on `apu_phase`-true (cycles 1,3,5 — odd); `put_cycle` is seeded so
1734        // its true-phase falls on EVEN cycles, so `!put_cycle` recovers main's
1735        // ODD-cycle DMC fire-phase (the DMASync-positioning alignment).
1736        if !self.put_cycle {
1737            self.dmc.clock_timer();
1738        }
1739        self.dmc_step_reload_arm();
1740    }
1741
1742    /// v2.0 interleaved-DMA Phase B: advance ONLY the DMC byte-timer clock (no
1743    /// delay/reload ARM), for a cycle of an interleaved DMC DMA span. The timer
1744    /// advances (so the variable-3/4-span feeds back into the next fire-cycle —
1745    /// divergence-A self-consistency) WITHOUT re-arming a new DMA mid-span (no
1746    /// cascade). Toggles the same `dmc_ext_phase` flip-flop as [`Self::tick_dmc`]
1747    /// so the every-other-cycle cadence stays consistent across normal + DMA
1748    /// cycles. (In the burst model this re-wedged; in the per-cycle interleaved
1749    /// model each DMA cycle is discrete and arm-gated, so it should hold.)
1750    pub fn tick_dmc_timer_only(&mut self) {
1751        // Divergence A: clock off the shared `put_cycle` counter (see `tick_dmc`).
1752        if !self.put_cycle {
1753            self.dmc.clock_timer();
1754        }
1755    }
1756
1757    /// v2.0 F-2: route the DMC byte-timer + arm to [`Self::tick_dmc`] instead of
1758    /// `tick_with_external`. Default `false` = byte-identical.
1759    pub const fn set_dmc_driven_externally(&mut self, on: bool) {
1760        self.dmc_driven_externally = on;
1761    }
1762
1763    /// v2.0 interleaved-DMA Phase A: the global get/put flip-flop (TriCNES
1764    /// `APU_PutCycle`). `true` = put cycle, `false` = get cycle.
1765    #[must_use]
1766    pub const fn put_cycle(&self) -> bool {
1767        self.put_cycle
1768    }
1769
1770    /// v2.0 interleaved-DMA Phase A: seed the global get/put flip-flop from an
1771    /// `APUAlignment` value (TriCNES `Emulator.cs:685/776`), the single seed the
1772    /// interleaved DMA (Phase B) will share with the DMC fire-phase (divergence
1773    /// A). The low bit selects the parity (TriCNES case 0/2 -> put, 1/3 -> get).
1774    ///
1775    /// Phase A seeds ONLY `put_cycle` and deliberately leaves `dmc_ext_phase`
1776    /// untouched, so the un-wedged feature-on behavior is preserved (the f2e
1777    /// experiment proved flipping `dmc_ext_phase` alone regresses). The exact
1778    /// `put_cycle` <-> `dmc_ext_phase` pairing is determined empirically in
1779    /// Phase B, when the bus first consumes `put_cycle` for the get/put decision.
1780    pub const fn seed_apu_alignment(&mut self, alignment: u8) {
1781        self.put_cycle = (alignment & 1) == 0;
1782        // RW-1 (`mc-r1-one-clock`): record the boot parity as the ONE seed both
1783        // `apu_phase` and `put_cycle` derive from. `alignment == 0` -> seed 0,
1784        // which reproduces the floor config (boot `apu_phase = false` +
1785        // put-on-even) exactly. Set at power-on, reset, and restore; constant
1786        // otherwise. The legacy `put_cycle` assignment above is harmless when the
1787        // flag is on (the next derivation overwrites it from `cpu_cycle`).
1788        {
1789            self.parity_seed = (alignment & 1) as u64;
1790        }
1791    }
1792
1793    /// CPU register write (`$4000-$4017` excluding `$4014`).
1794    pub fn write_register(&mut self, addr: u16, value: u8) {
1795        // v2.3.7 "Overtone" — attribute the write BEFORE dispatching it, so the
1796        // recorded value is what the CPU put on the bus rather than whatever a
1797        // channel decided to keep. One `Option` test when disarmed.
1798        #[cfg(feature = "debug-hooks")]
1799        if let Some(p) = self.audio_prov.as_mut() {
1800            p.reg_attrib
1801                .record(addr, p.attrib_pc, p.attrib_cycle, value);
1802        }
1803        match addr {
1804            0x4000 => self.pulse1.write_ctrl(value),
1805            0x4001 => self.pulse1.write_sweep(value),
1806            0x4002 => self.pulse1.write_timer_lo(value),
1807            0x4003 => self.pulse1.write_timer_hi(value),
1808            0x4004 => self.pulse2.write_ctrl(value),
1809            0x4005 => self.pulse2.write_sweep(value),
1810            0x4006 => self.pulse2.write_timer_lo(value),
1811            0x4007 => self.pulse2.write_timer_hi(value),
1812            0x4008 => self.triangle.write_linear(value),
1813            0x4009 => {} // unused
1814            0x400A => self.triangle.write_timer_lo(value),
1815            0x400B => self.triangle.write_timer_hi(value),
1816            0x400C => self.noise.write_ctrl(value),
1817            0x400D => {} // unused
1818            0x400E => self.noise.write_period(value),
1819            0x400F => self.noise.write_length(value),
1820            0x4010 => self.dmc.write_ctrl(value),
1821            0x4011 => self.dmc.write_dac(value),
1822            0x4012 => self.dmc.write_sample_addr(value),
1823            0x4013 => self.dmc.write_sample_length(value),
1824            0x4015 => self.write_status(value),
1825            0x4017 => {
1826                // $4017 also clears DMC IRQ?  No — only $4015 clears DMC.
1827                // But writing $4017 with bit 6 set clears frame IRQ.
1828                // The frame counter handles the inhibit-clears-flag effect.
1829                // Apu-aligned: cycle is even when apu_phase will toggle to
1830                // true on the NEXT tick.  Our `apu_phase` reflects the
1831                // *current* state after the tick.  Per nesdev: "If the write
1832                // occurs during an APU clock (CPU cycle 1, 3, 5...) the
1833                // effects occur 3 CPU cycles after the write; if during a
1834                // non-APU clock, the effects occur 4 CPU cycles after."
1835                let aligned = self.apu_phase;
1836                self.frame_counter.write(value, aligned);
1837            }
1838            _ => {}
1839        }
1840    }
1841
1842    // W3-Stage-3: the delayed-4015 cfg arms (the latch dispatch + the
1843    // superseded-compensation gating) push the counted length just past the
1844    // clippy limit; the body is mostly per-feature cfg blocks.
1845    #[allow(clippy::too_many_lines)]
1846    fn write_status(&mut self, value: u8) {
1847        self.pulse1.length.set_enabled((value & 0x01) != 0);
1848        self.pulse2.length.set_enabled((value & 0x02) != 0);
1849        self.triangle.length.set_enabled((value & 0x04) != 0);
1850        self.noise.length.set_enabled((value & 0x08) != 0);
1851        let enable_dmc = (value & 0x10) != 0;
1852        let was_active = self.dmc.active();
1853        let implicit_stop_edge = enable_dmc
1854            && !was_active
1855            && !self.dmc.loop_flag
1856            && self.dmc.sample_length == 1
1857            && self.dmc.rate_index == 0x0E
1858            && self.dmc.bits_remaining == 1
1859            && self.dmc.sample_buffer.is_none();
1860        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES delayed-status
1861        // latch replaces the immediate `set_enabled` application — see
1862        // `latch_delayed_dmc_4015`.
1863        self.latch_delayed_dmc_4015(enable_dmc);
1864        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES gates the enable-side
1865        // LOAD-delay arm on `APU_Silent` (Emulator.cs:9519-9522 — "the sample
1866        // will only begin playing if the DMC is currently silent"; otherwise
1867        // the restart is picked up at the NEXT shifter-consume edge). Our
1868        // floor condition (`needs_dma()` alone) arms the load immediately
1869        // even while the output unit is still draining the prior looping
1870        // byte — in the Implicit Loop3/`$540` re-enable race that fires a
1871        // span-3 load-style DMA 2-4 sweep positions before the hardware's
1872        // boundary-quantized reload (the `03,03` lead-in + plateau-2-early).
1873        let load_arm = enable_dmc && !was_active && self.dmc.needs_dma() && self.dmc.silence();
1874        if load_arm {
1875            self.pending_dmc_dma = false;
1876            self.dmc_dma_is_load = true;
1877            self.dmc_dma_short = true;
1878            self.dmc_dma_addr = self.dmc.dma_addr();
1879            // Load DMAs attempt to halt on the get cycle during the second
1880            // APU cycle after `$4015` enables DMC. In this emulator
1881            // `apu_phase == true` is the get half of the current APU cycle.
1882            //
1883            // v2.0 R-1 core C-1: under R1 (`dmc_driven_externally`) the DMC
1884            // clocks on `!put_cycle` (F-2), so `apu_phase` is the WRONG phase
1885            // basis for the load-arm delay just as it is for the abort `cuo`
1886            // (P-2: the R1 load DMA fires 1-2 cyc early → the 1-byte abort
1887            // sample's narrow active window lands off the swept `$4015` disable
1888            // → `disable_was_active` 12 vs 76). Use the DMC's actual phase.
1889            // W3-Stage-2 (`mc-r1-dma-unified-collapse`): TriCNES
1890            // `DMCDMADelay = 2` — two put end-ticks of `dmc_tick_end` (the
1891            // write cycle's own end-tick counts when it lands on a put, the
1892            // "really like 2 : 3" parity absorption), so the halt arms at the
1893            // end of a PUT and the load enters on a GET regardless of the
1894            // write parity. Replaces the every-cycle `apu_phase ? 4 : 3`
1895            // countdown whose value bakes in the floor GET parity.
1896            {
1897                self.dmc_dma_delay = 2;
1898            }
1899            // W3-Stage-2: the implicit-stop-edge -1 is a CPU-cycle-unit
1900            // calibration of the every-cycle countdown; under the TriCNES
1901            // put-end-tick countdown (put units) it cannot be expressed and
1902            // TriCNES has no such adjustment — skip it (TriCNES-exact).
1903            let _ = implicit_stop_edge;
1904        } else if !enable_dmc {
1905            // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the disable-side floor
1906            // compensations below (the scheduled explicit abort, the
1907            // pending-reload keep-alive reshaping, the load-delay zeroing and
1908            // the suppress reset) are SUPERSEDED by the delayed-status
1909            // application — TriCNES's `$4015` disable write does nothing else
1910            // DMC-wise; the abort is EMERGENT from the applied status gating
1911            // the per-cycle DMA service (`_6502` line 4218).
1912        } else if enable_dmc {
1913            self.dmc_reload_suppress_outputs = 0;
1914        }
1915    }
1916
1917    /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES `$4015` write
1918    /// handler's DMC-status section (Emulator.cs:9504-9548). IMMEDIATE at the
1919    /// write: the enable-side `StartDMCSample` (`set_enabled(true)` restarts
1920    /// only when `bytes_remaining == 0` — exactly `StartDMCSample`, line
1921    /// 9517) and the DMC IRQ-flag clear (line 9529). DEFERRED: the status-bit
1922    /// application + the disable-side `bytes_remaining` zeroing, latched into
1923    /// `dmc_delayed_status` and applied `put ? 3 : 4` end-ticks later (line
1924    /// 9512; the write cycle's own end-tick counts — "really like 2 : 3"). A
1925    /// second `$4015` write during the pending window resets the countdown
1926    /// with the new target (last write wins, as TriCNES).
1927    fn latch_delayed_dmc_4015(&mut self, enable_dmc: bool) {
1928        if enable_dmc {
1929            self.dmc.set_enabled(true);
1930        } else {
1931            self.dmc.irq_flag = false;
1932        }
1933        self.dmc_delayed_status = enable_dmc;
1934        self.dmc_delayed_4015 = if self.put_cycle { 3 } else { 4 };
1935        // The explicit don't-abort edge (Emulator.cs:9533-9537): the disable
1936        // coincides with "the APU cycle that fires a DMC DMA". TriCNES
1937        // `(timer == 2 && get) || (timer == rate && put)` in CPU-rate units
1938        // maps to our APU-rate byte-timer as `(timer == 0 && get)` (this
1939        // cycle's end-tick wraps) or `(timer == timer_period && put)` (the
1940        // wrap happened on the previous get half). Extend the delay to
1941        // `put ? 5 : 6` so the just-armed reload DMA runs to completion
1942        // before the disable zeroes `bytes_remaining` (EXPLICIT sweep
1943        // idx[7] = 04).
1944        if !enable_dmc {
1945            let firing_apu_cycle = if self.put_cycle {
1946                self.dmc.timer == self.dmc.timer_period
1947            } else {
1948                self.dmc.timer == 0
1949            };
1950            if firing_apu_cycle {
1951                self.dmc_delayed_4015 = if self.put_cycle { 5 } else { 6 };
1952            }
1953        }
1954        // The implicit-abort edge (Emulator.cs:9540-9545): an ENABLE that
1955        // lands one byte-timer fire BEFORE the shifter-consume edge —
1956        // TriCNES `(timer == 10 && get) || (timer == 8 && put)` = our
1957        // APU-rate `(4, get)/(3, put)` (uniform `(t - 2) / 2` mapping).
1958        // "Regardless of the buffer being empty, there will be a 1-cycle
1959        // DMA that gets aborted" — latched here, consumed at the consume
1960        // edge in `dmc_tick_end`.
1961        if enable_dmc {
1962            let pre_fire_window = if self.put_cycle {
1963                self.dmc.timer == 3
1964            } else {
1965                self.dmc.timer == 4
1966            };
1967            if pre_fire_window {
1968                self.dmc_set_implicit_abort = true;
1969            }
1970        }
1971    }
1972
1973    /// CPU register read (only `$4015` is meaningful).  Reading clears the
1974    /// frame IRQ flag.
1975    pub fn read_status(&mut self) -> u8 {
1976        let mut v = 0u8;
1977        if self.pulse1.length.active() {
1978            v |= 0x01;
1979        }
1980        if self.pulse2.length.active() {
1981            v |= 0x02;
1982        }
1983        if self.triangle.length.active() {
1984            v |= 0x04;
1985        }
1986        if self.noise.length.active() {
1987            v |= 0x08;
1988        }
1989        // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `Observe` `$4015`
1990        // (Emulator.cs:9127/9260 + the 9268 footnote) — bit 4 is
1991        // `bytes_remaining != 0 && APU_Status_DelayedDMC`: a read right after
1992        // a disable write must see bit 4 CLEAR even though `bytes_remaining`
1993        // is not zeroed until the delayed application ("LDA #0, STA $4015,
1994        // LDA $4015 ... needs to immediately have bit 4 cleared").
1995        if self.dmc.active() && self.dmc_delayed_status {
1996            v |= 0x10;
1997        }
1998        if self.frame_counter.irq_flag {
1999            v |= 0x40;
2000        }
2001        if self.dmc.irq_flag {
2002            v |= 0x80;
2003        }
2004        // Reading clears frame IRQ flag (NOT DMC IRQ). The clear is
2005        // SCHEDULED for a future CPU cycle (1 cycle delta on a "get"
2006        // cycle, 2 cycles on a "put") and matured by a subsequent
2007        // observation -- the canonical Mesen2 `GetIrqFlag` lazy
2008        // algorithm (Session-25, 2026-05-23). The pre-Session-25
2009        // immediate-on-get / defer-by-one-tick-on-put scheme failed
2010        // `AccuracyCoin :: APU Tests :: Frame Counter IRQ` Test 7.
2011        // See `docs/audit/session-25-sprint2-iter3-frame-counter-irq-2026-05-23.md`.
2012        let _ = self
2013            .frame_counter
2014            .read_status(self.cpu_cycle, self.apu_phase);
2015        v
2016    }
2017
2018    /// Clear the frame IRQ flag immediately for DMA no-op reads of `$4015`.
2019    ///
2020    /// The normal CPU-visible `$4015` read path keeps the put-cycle deferred
2021    /// clear needed by frame-counter timing tests. DMC DMA no-op repeats use
2022    /// this after sampling the status value so the halted-read side effect is
2023    /// visible before the CPU resumes the original `$4015` read.
2024    ///
2025    /// Session-26 iter 5: also deassert the CPU IRQ line driver
2026    /// (`irq_line_active`) since the DMA no-op read mirrors a CPU
2027    /// `$4015` read on the silicon — the IRQ source is removed from
2028    /// the CPU's `_irqSource` list synchronously.
2029    pub fn clear_frame_irq_immediate_for_dma(&mut self) {
2030        self.frame_counter.irq_flag = false;
2031        self.frame_counter.irq_line_active = false;
2032    }
2033}
2034
2035#[cfg(test)]
2036mod tests {
2037
2038    /// v2.3.5 C1 — the default-configuration fast path must be byte-identical
2039    /// to the gated path it skips.
2040    ///
2041    /// The specialization is only sound because `gate` is the identity when its
2042    /// mask bit is set and `scale` is the identity at gain 1.0. If either ever
2043    /// stops being the identity at the default, this silently changes shipped
2044    /// audio -- so assert the equivalence directly over a 2,048-point sweep of
2045    /// the output range rather than trusting the reasoning.
2046    #[test]
2047    fn apu_default_mix_matches_the_gated_path() {
2048        let apu = Apu::new(Region::Ntsc, 48_000);
2049        assert_eq!(apu.channel_mask, CHANNEL_MASK_ALL, "premise: default mask");
2050        assert_eq!(
2051            apu.channel_gain, CHANNEL_GAIN_UNITY,
2052            "premise: default gain"
2053        );
2054
2055        let mask = CHANNEL_MASK_ALL;
2056        let gain = CHANNEL_GAIN_UNITY;
2057        let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2058        let scale = |bit: usize, v: u8, max: u8| {
2059            let g = gain[bit];
2060            if g == 1.0 {
2061                v
2062            } else {
2063                #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
2064                {
2065                    roundf(f32::from(v) * g).clamp(0.0, f32::from(max)) as u8
2066                }
2067            }
2068        };
2069
2070        // 2,048 SELECTED combinations, not the full cross-product. The DMC axis
2071        // is swept exhaustively (0..=127) against 16 rotating phases of the
2072        // other four channels, which walks the `tnd_table` index range end to
2073        // end and visits every raw level each channel can take. The exhaustive
2074        // product would be 16^4 * 128 = 8.4M; this is a sweep, and the wording
2075        // says "sweep" rather than claiming enumeration.
2076        for dmc in 0u8..=127 {
2077            for lvl in 0u8..=15 {
2078                let (p1, p2, tri, noise) = (lvl, 15 - lvl, (lvl + 7) % 16, (lvl + 3) % 16);
2079                let gated = apu.mixer.mix(
2080                    scale(0, gate(0, p1), 15),
2081                    scale(1, gate(1, p2), 15),
2082                    scale(2, gate(2, tri), 15),
2083                    scale(3, gate(3, noise), 15),
2084                    scale(4, gate(4, dmc), 127),
2085                ) + if mask & (1 << 5) != 0 { 0.25f32 } else { 0.0 };
2086                let fast = apu.mixer.mix(p1, p2, tri, noise, dmc) + 0.25f32;
2087                assert_eq!(
2088                    gated.to_bits(),
2089                    fast.to_bits(),
2090                    "p1={p1} p2={p2} tri={tri} noise={noise} dmc={dmc}: \
2091                     the fast path must be BIT-identical, not merely close"
2092                );
2093            }
2094        }
2095    }
2096
2097    /// The fast path must NOT be taken once the configuration stops being the
2098    /// default -- otherwise the mute/gain overlay would silently stop working.
2099    #[test]
2100    fn a_non_default_mask_or_gain_still_takes_the_gated_path() {
2101        let mut muted = Apu::new(Region::Ntsc, 48_000);
2102        muted.set_channel_mask(CHANNEL_MASK_ALL & !0x01); // mute pulse 1
2103        assert_ne!(muted.channel_mask, CHANNEL_MASK_ALL);
2104
2105        let mut quiet = Apu::new(Region::Ntsc, 48_000);
2106        quiet.set_channel_gain([0.5, 1.0, 1.0, 1.0, 1.0, 1.0]);
2107        assert_ne!(quiet.channel_gain, CHANNEL_GAIN_UNITY);
2108
2109        // Drive both far enough to produce output, and confirm a muted channel
2110        // actually changes the mix relative to the default.
2111        let mut plain = Apu::new(Region::Ntsc, 48_000);
2112        for a in [&mut plain, &mut muted, &mut quiet] {
2113            a.write_register(0x4015, 0x1F);
2114            a.write_register(0x4000, 0xBF);
2115            a.write_register(0x4002, 0xAA);
2116            a.write_register(0x4003, 0x08);
2117            for _ in 0..2_000 {
2118                a.tick();
2119            }
2120        }
2121        // The original assertion here was broken, and both review bots caught it:
2122        // it compared `plain.pulse1.output() == 0` against
2123        // `muted.channel_mask & 0x01 != 0`, which is `false` for a muted mask --
2124        // so it only passed when pulse 1 happened to be at output 0 on the
2125        // sampled tick. Phase-dependent, disconnected from the overlay it claimed
2126        // to test, and it never touched `quiet` at all. It could not fail on the
2127        // bug it existed to catch.
2128        //
2129        // Compare the EMITTED AUDIO instead, which is what the overlay is
2130        // supposed to change. Deliberately not `pulse1.output() != 0` even as a
2131        // premise check: that samples one instant, and a square wave spends half
2132        // its period at zero, so it is phase-dependent -- exactly the flaw that
2133        // made the original assertion vacuous. Accumulated samples have no such
2134        // dependence: if anything was audible, some sample is non-zero.
2135        assert_eq!(muted.channel_mask() & 0x01, 0, "premise: pulse 1 is muted");
2136
2137        let plain_audio = plain.drain_audio();
2138        let muted_audio = muted.drain_audio();
2139        let quiet_audio = quiet.drain_audio();
2140        assert!(!plain_audio.is_empty(), "premise: samples were emitted");
2141        assert!(
2142            plain_audio.iter().any(|s| *s != 0.0),
2143            "premise: the default configuration produced audible output"
2144        );
2145        assert_ne!(
2146            plain_audio, muted_audio,
2147            "a cleared mask bit must change the emitted audio"
2148        );
2149        assert_ne!(
2150            plain_audio, quiet_audio,
2151            "a non-unity gain must change the emitted audio"
2152        );
2153    }
2154    use super::*;
2155
2156    #[test]
2157    fn write_4015_enables_channels() {
2158        let mut a = Apu::new(Region::Ntsc, 44_100);
2159        a.write_register(0x4015, 0x0F);
2160        assert!(a.pulse1.length.enabled);
2161        assert!(a.pulse2.length.enabled);
2162        assert!(a.triangle.length.enabled);
2163        assert!(a.noise.length.enabled);
2164        assert!(!a.dmc.active());
2165    }
2166
2167    #[test]
2168    fn write_4015_clears_lengths_when_disabled() {
2169        let mut a = Apu::new(Region::Ntsc, 44_100);
2170        a.pulse1.length.enabled = true;
2171        a.pulse1.length.count = 10;
2172        a.write_register(0x4015, 0x00);
2173        assert_eq!(a.pulse1.length.count, 0);
2174    }
2175
2176    #[test]
2177    fn read_4015_clears_frame_irq_not_dmc_irq() {
2178        // Session-25 (2026-05-23): the canonical Mesen2 lazy-clear
2179        // algorithm SCHEDULES the frame-IRQ flag clear instead of
2180        // performing it immediately. A GET-cycle (`apu_phase=true`)
2181        // read schedules a clear at `cpu_cycle + 1`; a tick then
2182        // matures the schedule and the flag observable on the next
2183        // CPU cycle is `false`. DMC IRQ is untouched.
2184        let mut a = Apu::new(Region::Ntsc, 44_100);
2185        a.frame_counter.irq_flag = true;
2186        a.dmc.irq_flag = true;
2187        a.apu_phase = true; // GET cycle (1-cycle delta)
2188        let v = a.read_status();
2189        assert_eq!(v & 0xC0, 0xC0, "read returns the OLD flag (still set)");
2190        // The flag is STILL set right after the read; the clear is
2191        // scheduled for `cpu_cycle + 1`.
2192        assert!(a.frame_counter.irq_flag);
2193        assert_ne!(a.frame_counter.irq_flag_clear_cycle, 0);
2194        assert!(a.dmc.irq_flag);
2195        // Tick once -- the scheduled clear matures inside the tick.
2196        a.tick();
2197        assert!(!a.frame_counter.irq_flag, "flag matures inside tick");
2198        assert_eq!(a.frame_counter.irq_flag_clear_cycle, 0);
2199        // DMC IRQ is independently retained.
2200        assert!(a.dmc.irq_flag);
2201    }
2202
2203    #[test]
2204    fn read_4015_on_put_cycle_defers_irq_clear_by_two_cycles() {
2205        // Session-25 (2026-05-23): a PUT-cycle (`apu_phase=false`)
2206        // read schedules the clear at `cpu_cycle + 2` instead of
2207        // `cpu_cycle + 1`. This is the AccuracyCoin `APU Frame
2208        // Counter IRQ` Test 7 axis: the SLO ABS,X double-read of
2209        // `$4015` on a PUT-cycle first read sees the flag STILL SET
2210        // on the second read 1 CPU cycle later (the schedule has not
2211        // yet matured).
2212        let mut a = Apu::new(Region::Ntsc, 44_100);
2213        a.frame_counter.irq_flag = true;
2214        a.apu_phase = false; // PUT cycle (2-cycle delta)
2215        let v = a.read_status();
2216        assert_eq!(v & 0x40, 0x40, "first read returns the OLD flag");
2217        assert!(a.frame_counter.irq_flag, "flag stays set on put-cycle read");
2218        let scheduled = a.frame_counter.irq_flag_clear_cycle;
2219        assert_eq!(scheduled, a.cpu_cycle.wrapping_add(2));
2220        // A second read on the SAME put cycle still sees the set
2221        // flag (the schedule hasn't matured: cpu_cycle == cpu_cycle).
2222        let v2 = a.read_status();
2223        assert_eq!(
2224            v2 & 0x40,
2225            0x40,
2226            "second read on same cycle still sees flag set"
2227        );
2228        assert!(a.frame_counter.irq_flag);
2229        // Now advance ONE CPU cycle via a tick. cpu_cycle becomes
2230        // scheduled - 1. The schedule has NOT yet matured.
2231        a.tick();
2232        assert!(a.frame_counter.irq_flag, "flag still set after 1 tick");
2233        // Advance the SECOND CPU cycle. cpu_cycle now equals
2234        // scheduled. The tick matures the clear.
2235        a.tick();
2236        assert!(!a.frame_counter.irq_flag, "flag matures after 2 ticks");
2237        assert_eq!(a.frame_counter.irq_flag_clear_cycle, 0);
2238    }
2239
2240    #[test]
2241    fn tick_advances_cycle_counter() {
2242        let mut a = Apu::new(Region::Ntsc, 44_100);
2243        for _ in 0..100 {
2244            a.tick();
2245        }
2246        assert_eq!(a.cpu_cycle, 100);
2247    }
2248
2249    #[test]
2250    fn channel_mask_defaults_to_all_on() {
2251        let a = Apu::new(Region::Ntsc, 44_100);
2252        assert_eq!(a.channel_mask(), CHANNEL_MASK_ALL);
2253    }
2254
2255    #[test]
2256    fn channel_mask_set_clamps_to_known_bits() {
2257        let mut a = Apu::new(Region::Ntsc, 44_100);
2258        // Upper bits beyond the 6 defined channels are masked off.
2259        a.set_channel_mask(0xFF);
2260        assert_eq!(a.channel_mask(), CHANNEL_MASK_ALL);
2261        a.set_channel_mask(0x00);
2262        assert_eq!(a.channel_mask(), 0x00);
2263        a.set_channel_mask(0b0010_1010);
2264        assert_eq!(a.channel_mask(), 0b0010_1010);
2265    }
2266
2267    #[test]
2268    fn default_mask_mix_is_byte_identical_to_unmasked() {
2269        // The determinism contract: with the default all-on mask, the gating in
2270        // `tick_with_external` must reproduce the raw mixer output exactly.
2271        let m = Mixer::new();
2272        let mask = CHANNEL_MASK_ALL;
2273        let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2274        for &(p1, p2, tri, n, dmc) in &[
2275            (0u8, 0u8, 0u8, 0u8, 0u8),
2276            (15, 15, 15, 15, 127),
2277            (7, 3, 11, 4, 60),
2278            (1, 14, 8, 15, 1),
2279        ] {
2280            let raw = m.mix(p1, p2, tri, n, dmc);
2281            let gated = m.mix(
2282                gate(0, p1),
2283                gate(1, p2),
2284                gate(2, tri),
2285                gate(3, n),
2286                gate(4, dmc),
2287            );
2288            assert_eq!(raw, gated, "default mask must be byte-identical");
2289        }
2290    }
2291
2292    #[test]
2293    fn cleared_channel_bit_zeroes_its_contribution() {
2294        let m = Mixer::new();
2295        // Mute pulse 1 only (bit 0 cleared).
2296        let mask = CHANNEL_MASK_ALL & !0x01;
2297        let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2298        let muted = m.mix(gate(0, 15), gate(1, 0), gate(2, 0), gate(3, 0), gate(4, 0));
2299        // Pulse 1 = 15 muted to 0 => identical to an all-silent mix.
2300        assert_eq!(muted, m.mix(0, 0, 0, 0, 0));
2301        // Pulse 2 (bit 1 still set) still contributes.
2302        let p2_on = m.mix(gate(0, 15), gate(1, 15), gate(2, 0), gate(3, 0), gate(4, 0));
2303        assert!(p2_on > 0.0);
2304    }
2305
2306    #[test]
2307    fn channel_gain_defaults_to_unity() {
2308        let a = Apu::new(Region::Ntsc, 44_100);
2309        assert_eq!(a.channel_gain(), CHANNEL_GAIN_UNITY);
2310    }
2311
2312    #[test]
2313    fn external_out_tracks_last_external_sample() {
2314        // v2.1.6 — the read-only expansion-audio display tap reflects the most
2315        // recent RAW value fed to `tick_with_external` and defaults to 0.0.
2316        let mut a = Apu::new(Region::Ntsc, 44_100);
2317        assert_eq!(a.external_out(), 0.0);
2318        a.tick_with_external(0.25);
2319        assert!((a.external_out() - 0.25).abs() < f32::EPSILON);
2320        a.tick_with_external(-0.1);
2321        assert!((a.external_out() - (-0.1)).abs() < f32::EPSILON);
2322        // The tap is a copy: a non-unity external gain does NOT change what the
2323        // scope observes (it always sees the raw chip contribution).
2324        a.set_channel_gain([1.0, 1.0, 1.0, 1.0, 1.0, 0.5]);
2325        a.tick_with_external(0.4);
2326        assert!((a.external_out() - 0.4).abs() < f32::EPSILON);
2327    }
2328
2329    #[test]
2330    fn channel_gain_set_clamps_to_range() {
2331        let mut a = Apu::new(Region::Ntsc, 44_100);
2332        a.set_channel_gain([3.0, -1.0, 0.5, 1.0, 2.0, 0.0]);
2333        // 3.0 -> 2.0 (ceiling), -1.0 -> 0.0 (floor), the rest unchanged.
2334        assert_eq!(a.channel_gain(), [2.0, 0.0, 0.5, 1.0, 2.0, 0.0]);
2335    }
2336
2337    /// A NaN gain (only reachable from a hand-edited config) must not reach the
2338    /// mixer: `f32::clamp` passes NaN through, and one NaN term makes the mixed
2339    /// sample NaN. It falls back to unity; the infinities already clamp.
2340    #[test]
2341    fn channel_gain_rejects_nan_and_clamps_infinities() {
2342        let mut a = Apu::new(Region::Ntsc, 44_100);
2343        a.set_channel_gain([f32::NAN, f32::INFINITY, f32::NEG_INFINITY, 1.0, 1.0, 1.0]);
2344        assert_eq!(a.channel_gain(), [1.0, 2.0, 0.0, 1.0, 1.0, 1.0]);
2345    }
2346
2347    #[test]
2348    fn unity_gain_produces_byte_identical_samples() {
2349        // The hard determinism requirement: a full run with the default unity
2350        // gains must produce a bit-identical band-limited output to a fresh APU.
2351        const EXT: [f32; 7] = [0.0, 0.01, 0.02, 0.03, 0.04, 0.05, 0.06];
2352        let mut a = Apu::new(Region::Ntsc, 44_100);
2353        let mut b = Apu::new(Region::Ntsc, 44_100);
2354        b.set_channel_gain(CHANNEL_GAIN_UNITY); // explicit unity == default
2355        // Drive both with an identical register + tick sequence.
2356        for step in 0..4_000u32 {
2357            let v = (step & 0xFF) as u8;
2358            a.write_register(0x4000 + (step % 0x14) as u16, v);
2359            b.write_register(0x4000 + (step % 0x14) as u16, v);
2360            let ext = EXT[(step % 7) as usize];
2361            a.tick_with_external(ext);
2362            b.tick_with_external(ext);
2363        }
2364        let mut out_a = [0.0f32; 4096];
2365        let mut out_b = [0.0f32; 4096];
2366        let na = a.drain_audio_into(&mut out_a);
2367        let nb = b.drain_audio_into(&mut out_b);
2368        assert_eq!(na, nb);
2369        assert_eq!(
2370            out_a[..na],
2371            out_b[..nb],
2372            "unity gain must be bit-identical to the default mix"
2373        );
2374    }
2375
2376    /// v2.9.8: the per-cycle mix reads the cached `gain_is_unity`, so every
2377    /// path that changes the gain must refresh it. A power cycle carries the
2378    /// gain into a fresh APU through `adopt_settings_from`; a bare field copy
2379    /// there left the flag `true` and mixed a 0.5-gain channel at unity.
2380    #[test]
2381    fn a_power_cycle_keeps_a_non_unity_gain_audible() {
2382        const EXT: [f32; 7] = [0.0, 0.1, -0.2, 0.3, -0.1, 0.05, 0.0];
2383        let gain = [0.5, 1.0, 1.0, 1.0, 0.25, 1.0];
2384        let mut old = Apu::new(Region::Ntsc, 44_100);
2385        old.set_channel_gain(gain);
2386        // The power-cycled APU: a fresh one that adopted the old one's settings.
2387        let mut cycled = Apu::new(Region::Ntsc, 44_100);
2388        cycled.adopt_settings_from(&old);
2389        assert!(!cycled.gain_is_unity, "the cached flag went stale");
2390        // Reference: a fresh APU given the same gain through the setter.
2391        let mut direct = Apu::new(Region::Ntsc, 44_100);
2392        direct.set_channel_gain(gain);
2393        for step in 0..4_000u32 {
2394            let v = (step & 0xFF) as u8;
2395            cycled.write_register(0x4000 + (step % 0x14) as u16, v);
2396            direct.write_register(0x4000 + (step % 0x14) as u16, v);
2397            let ext = EXT[(step % 7) as usize];
2398            cycled.tick_with_external(ext);
2399            direct.tick_with_external(ext);
2400        }
2401        let mut out_c = [0.0f32; 4096];
2402        let mut out_d = [0.0f32; 4096];
2403        let nc = cycled.drain_audio_into(&mut out_c);
2404        let nd = direct.drain_audio_into(&mut out_d);
2405        assert_eq!(nc, nd);
2406        assert_eq!(
2407            out_c[..nc],
2408            out_d[..nd],
2409            "a power-cycled APU must mix with the gain it carried"
2410        );
2411    }
2412
2413    #[test]
2414    fn zero_gain_matches_a_cleared_mask_bit() {
2415        // Gain 0.0 on a channel is equivalent to clearing that channel's mask
2416        // bit (both force the raw output to 0 before the non-linear mixer).
2417        let m = Mixer::new();
2418        // Pulse 1 raw 15, everything else silent; gain 0 on pulse 1.
2419        // round(15 * 0.0) == 0, so the mixer sees pulse1 = 0.
2420        let scaled = m.mix(0, 0, 0, 0, 0);
2421        assert_eq!(scaled, m.mix(0, 0, 0, 0, 0));
2422        // Sanity: a real attenuation (0.5) lands strictly between full and muted.
2423        #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
2424        let half = (15.0f32 * 0.5).round() as u8; // 8
2425        let full = m.mix(15, 0, 0, 0, 0);
2426        let attenuated = m.mix(half, 0, 0, 0, 0);
2427        assert!(attenuated > 0.0 && attenuated < full);
2428    }
2429
2430    #[test]
2431    fn frame_irq_after_29828_cycles() {
2432        let mut a = Apu::new(Region::Ntsc, 44_100);
2433        // Default: 4-step mode, IRQ enabled.
2434        for _ in 0..29828 {
2435            a.tick();
2436        }
2437        assert!(a.frame_irq_pending());
2438    }
2439
2440    #[test]
2441    fn mode1_inhibits_irq() {
2442        let mut a = Apu::new(Region::Ntsc, 44_100);
2443        // Write mode=1 + inhibit.  After ~3 cycles delay, fire qf+hf.
2444        a.write_register(0x4017, 0xC0);
2445        for _ in 0..40_000 {
2446            a.tick();
2447        }
2448        // No IRQ ever raised in mode 1.
2449        assert!(!a.frame_irq_pending());
2450    }
2451
2452    #[test]
2453    fn dmc_writes_dac_directly() {
2454        let mut a = Apu::new(Region::Ntsc, 44_100);
2455        a.write_register(0x4011, 0x40);
2456        assert_eq!(a.dmc.dac, 0x40);
2457    }
2458
2459    #[test]
2460    fn enabling_dmc_starts_sample() {
2461        let mut a = Apu::new(Region::Ntsc, 44_100);
2462        a.write_register(0x4012, 0x00);
2463        a.write_register(0x4013, 0x10); // 0x101 bytes
2464        a.write_register(0x4015, 0x10);
2465        assert!(a.dmc.active());
2466    }
2467}