rustynes_apu/apu.rs
1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the DMC-DMA state model (the get/put flip-flop, the delayed-`$4015` DMC status machinery, the implicit-abort and re-enable timing fields) is derived from TriCNES (MIT), as the field docs cite by name and `Emulator.cs` line, and the `_needHalt` / `_needDummyRead` latches from Mesen2 (GPL-3.0-or-later), `NesCpu`. See docs/originality-and-provenance.md (Section 1) and NOTICE. Classified v2.9.9 (core re-audit NC-17, maintainer's decision 2026-10-04): the in-source citations below record the derivation and are kept as written.
4
5//! Top-level 2A03 APU.
6//!
7//! Per `docs/apu-2a03.md`. Owns the four wave channels plus DMC, the frame
8//! counter, the lookup-table mixer + filter chain, and the band-limited
9//! sample emitter. Driven by the lockstep bus's `Apu::tick` once per CPU
10//! cycle.
11
12use crate::Region;
13use crate::blip::{BlipBuf, CPU_HZ_NTSC, CPU_HZ_PAL};
14use crate::dmc::Dmc;
15use crate::frame_counter::{FrameCounter, FrameEvents};
16use crate::mixer::Mixer;
17use crate::noise::Noise;
18use crate::pulse::Pulse;
19use crate::triangle::Triangle;
20use alloc::vec::Vec;
21
22// `f32::round` lives in `std` (not `core`), so route through `libm::roundf` on
23// no_std — the same pattern the mixer uses for `expf`. Both round half away from
24// zero, so the result is identical across the desktop + `thumbv7em-none-eabihf`
25// targets. Only reached on the off-default per-channel-gain path (gain != 1.0),
26// never on the byte-identical unity path.
27#[inline]
28fn roundf(x: f32) -> f32 {
29 #[cfg(feature = "std")]
30 {
31 x.round()
32 }
33 #[cfg(not(feature = "std"))]
34 {
35 libm::roundf(x)
36 }
37}
38
39/// Top-level APU.
40#[derive(Debug, Clone)]
41pub struct Apu {
42 /// Region (NTSC / PAL / Dendy).
43 pub region: Region,
44 /// Pulse 1.
45 pub pulse1: Pulse,
46 /// Pulse 2.
47 pub pulse2: Pulse,
48 /// Triangle.
49 pub triangle: Triangle,
50 /// Noise.
51 pub noise: Noise,
52 /// DMC.
53 pub dmc: Dmc,
54 /// Frame counter.
55 pub frame_counter: FrameCounter,
56 /// Mixer.
57 pub(crate) mixer: Mixer,
58 /// Band-limited sample emitter.
59 pub(crate) blip: BlipBuf,
60 /// True on every other CPU cycle — pulse/noise/DMC clock at this rate.
61 // reason: `apu_phase` is the deliberate, documented name for the APU's
62 // clock phase; it appears verbatim as a column header in the committed
63 // irq_trace golden CSVs, so the `apu_` prefix is load-bearing, not noise.
64 #[allow(clippy::struct_field_names)]
65 pub(crate) apu_phase: bool,
66 /// v2.0 F-2: when set, the DMC byte-timer + DMA arm are driven by
67 /// [`Self::tick_dmc`] (called at end-of-cycle by the R1 bus) instead of
68 /// inside [`Self::tick_with_external`] (cycle-start). This shifts only the
69 /// DMC fire-phase to main's end-of-cycle position (for DMASync), leaving
70 /// the frame-counter / pulse / noise — and thus the APU IRQ line — on the
71 /// cycle-start tick (for the C1 IRQ sample). Default `false` = byte-identical.
72 pub(crate) dmc_driven_externally: bool,
73 /// v2.0 interleaved-DMA Phase A: the global get/put flip-flop (TriCNES
74 /// `APU_PutCycle`, `Emulator.cs:920`). Toggled exactly once per CPU cycle
75 /// (when `dmc_driven_externally`, so the default build is byte-identical)
76 /// and seeded at power-on/reset TOGETHER with the DMC byte-timer from one
77 /// `APUAlignment` value, so the get/put parity and the DMC fire-phase share
78 /// one seed + one per-cycle counter and can never drift (divergence A). The
79 /// interleaved DMA (Phase B) reads this for the get/put decision instead of
80 /// `self.cycle & 1`. `true` = put (write/OAM-priority), `false` = get
81 /// (read/DMC-priority). Nothing consumes it yet in Phase A.
82 pub(crate) put_cycle: bool,
83 /// v2.0 RW-1 (`mc-r1-one-clock`): the single boot parity seed. Under
84 /// `mc-r1-one-clock`, `apu_phase` and `put_cycle` are no longer two
85 /// independent flip-flops toggled per cycle — they are DERIVED from the one
86 /// per-cycle counter (`cpu_cycle`) plus this seed:
87 /// `apu_phase = (cpu_cycle + parity_seed) & 1 == 1`, `put_cycle = !apu_phase`.
88 /// This makes the APU-rate clock, the get/put DMA parity, and the DMC
89 /// fire-phase share ONE counter + ONE seed, so they can never drift apart
90 /// (the cumulative-counter-split root cause, see
91 /// `docs/audit/v2.0-cumulative-cycle-accounting-rewrite-plan-2026-06-05.md`).
92 /// `0` reproduces the floor config exactly (boot `apu_phase = false`,
93 /// `seed_apu_alignment(0)` -> put-on-even). Set once at power-on/reset/restore
94 /// by [`Self::seed_apu_alignment`]; otherwise constant. Unused when the flag
95 /// is off (the legacy dual-toggle path runs instead).
96 pub(crate) parity_seed: u64,
97 /// Cumulative CPU cycle counter (used for `$4017` write alignment).
98 pub(crate) cpu_cycle: u64,
99 /// Pending DMC DMA request — the bus polls and consumes this when it
100 /// halts the CPU and supplies a sample byte.
101 pub(crate) pending_dmc_dma: bool,
102 /// `mc-r1-dmc-reload-visibility-delay`: a RELOAD arm latches HERE and is
103 /// promoted to `pending_dmc_dma` one cycle later, so the DMA loop first
104 /// services it on the NEXT (put) cycle — matching TriCNES's
105 /// `_EmulateAPU`-after-`_6502` invisibility (reload first-service = put =>
106 /// span 4). Loads arm `pending_dmc_dma` directly (first-service = get => 3).
107 pub(crate) pending_dmc_dma_next: bool,
108 /// True when the pending DMC DMA is the initial load DMA after `$4015`
109 /// enable; false for reload DMAs raised by sample-buffer empty.
110 pub(crate) dmc_dma_is_load: bool,
111 /// True when the pending request uses the short 3-cycle service path
112 /// despite being externally observed as a load-style race. This covers
113 /// the explicit-stop abort edge where a visible reload request must be
114 /// preserved through the `$4015` disable write without making ordinary
115 /// load DMAs lose their dummy/alignment cadence.
116 pub(crate) dmc_dma_short: bool,
117 /// Suppress one immediate reload request after a same-tick DMC load
118 /// delivery. Used for the one-byte looping edge where the fetched byte
119 /// is visible to the output unit on the DMA get cycle, but the reload
120 /// request is not visible until the following CPU cycle.
121 pub(crate) defer_dmc_reload_once: bool,
122 /// Pending one-cycle DMC abort halt. This is the RP2A03 stop-near-reload
123 /// quirk: it does not fetch a byte, and if the halt attempt lands on a
124 /// CPU write cycle the abort disappears instead of retrying.
125 pub(crate) pending_dmc_abort: bool,
126 /// CPU cycles until an abort halt attempt becomes visible to the bus.
127 pub(crate) dmc_abort_delay: u8,
128 /// CPU cycles during which a newly emptied DMC sample buffer must not
129 /// raise another DMA request. The DMC DMA unit cannot issue a second
130 /// request within two CPU cycles of the previous get.
131 pub(crate) dmc_dma_cooldown: u8,
132 /// v2.0.0 beta.3 (A4 cycle-accurate reset): countdown to the scheduled
133 /// warm-reset `$4017` re-write (blargg `apu_reset` spec: reset behaves
134 /// as if the last `$4017` value were written again). Armed by
135 /// [`Apu::reset`] with the calibrated in-sequence placement; consumed in
136 /// `tick_with_external` during the CPU's clocked reset cycles. `0` = no
137 /// re-write pending.
138 pub(crate) reset_4017_delay: u8,
139 /// The retained `$4017` value the scheduled reset re-write will issue.
140 pub(crate) reset_4017_value: u8,
141 /// v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES's
142 /// `CannotRunDMCDMARightNow` (`Emulator.cs:823`). Set to 2 after every DMC
143 /// GET (`:4168`), decremented by 2 on each get cycle (`:1186`), and gating
144 /// the looping-reload arm (`:1165`, blocked while `== 2`). Reproduces the
145 /// "a DMA cannot occur within 2 cycles of a previous DMC DMA" rule so the
146 /// Implicit-DMA-Abort Loop3/`$540` X=10/11 re-enable defers its first
147 /// reload one byte-timer period (walk offset +4 -> +5, Y 3->4). Distinct
148 /// from `dmc_dma_cooldown` (which the abort-timer-phase fix clears at the
149 /// boundary race); this exclusion re-imposes the exact 1-get-cycle block.
150 pub(crate) cannot_run_dmc_dma: u8,
151 /// v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): latch that defers a reload to
152 /// the NEXT byte-timer wrap when the exclusion blocks the arm. TriCNES only
153 /// evaluates the reload arm at the `bits_remaining -> 0` consume edge
154 /// (`Emulator.cs:1159`); if blocked there (`cannot_run == 2`) the buffer is
155 /// not refilled until the FOLLOWING consume edge (one full byte period
156 /// later) — NOT a 2-cycle re-arm. RustyNES's `dmc_step_reload_arm` instead
157 /// re-checks `needs_dma()` (persistent) every cycle, so a bare exclusion
158 /// gate would re-arm as soon as `cannot_run` decremented (absorbed). This
159 /// latch reproduces the full-period deferral: set when the exclusion blocks
160 /// a consume-edge arm, cleared at the next consume edge.
161 pub(crate) dmc_reenable_period_block: bool,
162 /// final lever #1 (`mc-r1-dmc-halt-subpos`): a per-CPU-cycle countdown that
163 /// DELAYS the `$540` X=10/11 reload arm by an exact number of CPU cycles
164 /// (sub-APU-cycle granularity the byte-timer phase shift cannot express).
165 /// `0` = inactive. Set at the pattern-A boundary; while > 0 the natural arm
166 /// is suppressed and this decrements; at 0 the arm fires. Default 0.
167 // (W3-Stage-3: also dead under `mc-r1-dmc-delayed-4015`, which supersedes
168 // the halt-subpos pre-arm with the emergent consume-edge arm.)
169 pub(crate) subpos_arm_countdown: u8,
170 /// Latches the one-byte looping edge where the next reload request is
171 /// lost because it is raised too soon after a DMA get. Cleared when a
172 /// later `$4015` enable/disable write re-arms the DMC path.
173 pub(crate) dmc_reload_suppress_outputs: u8,
174 /// CPU cycles until a load DMC DMA halt attempt becomes visible to the
175 /// bus. Reload DMAs are armed immediately after the DMC output unit
176 /// empties the sample buffer; load DMAs after `$4015` enable are delayed
177 /// to the second following APU cycle per the 2A03 DMA cadence.
178 pub(crate) dmc_dma_delay: u8,
179 /// Most recent DMC DMA address (re-read each tick when `pending_dmc_dma`
180 /// is true; the bus may take it directly via [`Self::dmc_dma_addr`]).
181 /// On its own this is informational; the bus owns the actual halt logic.
182 pub(crate) dmc_dma_addr: u16,
183 /// v1.2 Sprint 3 — get/put cycle scheduler model (ADR-0007).
184 ///
185 /// Set when a DMC DMA request is raised; cleared by the new
186 /// `rustynes-core::bus::service_dmc_dma` path under the
187 /// `dmc-get-put-scheduler` feature flag, once the initial halt
188 /// cycle has been processed. Mirrors Mesen2's `_needHalt` on
189 /// `NesCpu` (`Core/NES/NesCpu.h:41`; set in `StartDmcTransfer`
190 /// at `Core/NES/NesCpu.cpp:527`). Kept ALWAYS-PRESENT (not
191 /// `#[cfg]`-gated) so the field exists in serialized state for
192 /// future-flag-flip migration; the v1.2 baseline scheduler
193 /// simply ignores it.
194 pub(crate) dmc_need_halt: bool,
195 /// v1.2 Sprint 3 — get/put cycle scheduler model (ADR-0007).
196 ///
197 /// Set when a DMC DMA request is raised; cleared by the new
198 /// scheduler once the alignment / dummy-read cycle has been
199 /// processed. Mirrors Mesen2's `_needDummyRead` on `NesCpu`.
200 /// Kept always-present alongside [`Self::dmc_need_halt`].
201 pub(crate) dmc_need_dummy_read: bool,
202 /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `APU_DelayedDMC4015`
203 /// (Emulator.cs:973) — CPU-cycle countdown until the latched `$4015` DMC
204 /// status bit is APPLIED. Set to `put ? 3 : 4` at every `$4015` write
205 /// (extended to `put ? 5 : 6` at the explicit don't-abort edge);
206 /// decremented once per `dmc_tick_end` (every CPU cycle, the write
207 /// cycle's own end-tick included). `0` = idle.
208 pub(crate) dmc_delayed_4015: u8,
209 /// W3-Stage-3: TriCNES `APU_Status_DelayedDMC` (Emulator.cs:964) — the
210 /// TARGET DMC status latched at the `$4015` write, applied when the
211 /// countdown expires. Also the value `$4015` READS see immediately
212 /// (the footnote at Emulator.cs:9268: bit 4 must read 0 right after a
213 /// disable write even though `bytes_remaining` is not yet zeroed).
214 pub(crate) dmc_delayed_status: bool,
215 /// W3-Stage-3: TriCNES `APU_Status_DMC` (Emulator.cs:963) — the APPLIED
216 /// DMC status. The bus-side DMA service gate (`_6502` line 4218:
217 /// `DoDMCDMA && (APU_Status_DMC || implicit-abort)`) reads this; while
218 /// false a pending/halted DMC DMA is NOT serviced (the emergent explicit
219 /// abort). Set/cleared ONLY by the delayed application + cleared at
220 /// non-looping natural sample end (TriCNES `DMCDMA_Get`, line 4154).
221 pub(crate) dmc_status_applied: bool,
222 /// W3-Stage-3: TriCNES `APU_SetImplicitAbortDMC4015` (Emulator.cs:975) —
223 /// latched at a `$4015` ENABLE write that coincides with the byte-timer's
224 /// firing window (`(timer == 10 && get) || (timer == 8 && put)` in
225 /// TriCNES CPU-rate units = our APU-rate `(4, get)/(3, put)`); consumed
226 /// at the next shifter-consume edge (bits 1 -> 8), where it arms the
227 /// 1-cycle implicit-abort DMA regardless of the buffer state
228 /// (Emulator.cs:1163-1175).
229 pub(crate) dmc_set_implicit_abort: bool,
230 /// W3-Stage-3: TriCNES `APU_ImplicitAbortDMC4015` (Emulator.cs:974) — the
231 /// service-gate override that lets the boundary-armed DMA run while the
232 /// `$4015` enable's delayed status is still unapplied. Cleared at the END
233 /// of the first cycle on which the DMA is pending (Emulator.cs:9000-9003:
234 /// one serviced halt cycle if that cycle is a read; "if this was delayed
235 /// by a write cycle, it won't run at all") — the emergent 1-cycle
236 /// implicit abort.
237 pub(crate) dmc_implicit_abort: bool,
238 /// W3-Stage-4 (`mc-r1-dmc-delayed-4015` grid correction): TriCNES's
239 /// reload arm is CONSUME-EDGE-QUANTIZED, not level-held. When the consume
240 /// edge lands ON the GET-delivery cycle itself (`CannotRunDMCDMARightNow
241 /// == 2`, Emulator.cs:1165 — only ever true at the same-cycle edge,
242 /// because the decrement at :1186 runs later that same end-tick), the arm
243 /// is skipped ENTIRELY and the chain defers to the NEXT consume edge
244 /// (576 cycles). Our `needs_dma()` is level-triggered and would re-arm as
245 /// soon as the cooldown expires (4 cycles — one grid boundary early, the
246 /// Implicit `$540[8,9]` cliff). Set at the blocked same-cycle edge;
247 /// suppresses the reload arm; cleared at the next consume edge in
248 /// `dmc_tick_end` immediately before the reload-arm step so the deferred
249 /// arm fires exactly on-grid.
250 pub(crate) dmc_edge_arm_suppress: bool,
251 /// Sample rate (Hz) for diagnostics.
252 pub sample_rate: u32,
253 /// Most recent frame-counter events produced by [`Self::tick_with_external`].
254 /// Read by the bus immediately after `tick` to fan the same events out to
255 /// any on-cart audio extension that shares the 2A03 frame counter cadence
256 /// (MMC5 audio). Reset to `FrameEvents::default()` at the *start* of every
257 /// `tick`, so observers must read it AFTER the tick.
258 pub(crate) last_frame_events: FrameEvents,
259 /// Per-channel enable mask (UI playback overlay, NOT NES hardware state).
260 /// Bit 0 = pulse 1, bit 1 = pulse 2, bit 2 = triangle, bit 3 = noise,
261 /// bit 4 = DMC, bit 5 = external/mapper audio. A cleared bit forces that
262 /// channel's contribution to the mixed sample to 0 (a studio/debug mute).
263 ///
264 /// Defaults to [`CHANNEL_MASK_ALL`] (every bit set), which is byte-identical
265 /// to passing the raw channel outputs straight into the mixer — i.e. the
266 /// deterministic core output is unchanged unless the frontend explicitly
267 /// mutes a channel. NEVER serialized into the save state (a UI preference,
268 /// like volume), so restored states are unaffected.
269 pub(crate) channel_mask: u8,
270 /// v1.4.0 Workstream C — per-channel output gain (a UI mixing overlay, NOT
271 /// NES hardware state), generalizing [`Self::channel_mask`]. Index 0 = pulse
272 /// 1, 1 = pulse 2, 2 = triangle, 3 = noise, 4 = DMC, 5 = external/mapper
273 /// audio. Each internal channel's raw integer output is scaled by its gain
274 /// and rounded back to an integer before the non-linear mixer; the external
275 /// (already-linear) sample is scaled directly.
276 ///
277 /// Defaults to [`CHANNEL_GAIN_UNITY`] (all `1.0`). At unity the mix takes the
278 /// EXACT current code path (`round(v * 1.0) == v`, `external * 1.0 ==
279 /// external`), so the deterministic core output is byte-identical unless the
280 /// frontend explicitly changes a gain — the determinism contract holds and
281 /// the oracle / test ROMs (which never touch a gain) are unaffected. NEVER
282 /// serialized into the save state (a UI preference, like the mask / volume).
283 pub(crate) channel_gain: [f32; 6],
284 /// v2.9.8 (D3): `channel_gain == CHANNEL_GAIN_UNITY`, cached.
285 ///
286 /// The per-cycle mix checks for unity gain on every CPU cycle. Comparing
287 /// six `f32`s there cost -3.1% to -4.5% of frame time on all four
288 /// workloads, in two runs (`docs/performance.md`, v2.9.8 campaign). The
289 /// gain only changes through [`Self::set_channel_gain`], which recomputes
290 /// this flag, so the cached value cannot go stale. It is not serialized:
291 /// like the gain itself, it is a host preference re-applied on load.
292 pub(crate) gain_is_unity: bool,
293 /// v2.9.8 — the analog output-filter model the host last selected with
294 /// [`Self::set_filter_model`].
295 ///
296 /// The filter itself lives in `blip` as a built chain of coefficients
297 /// (and its IIR history), and a chain does not say which model built it,
298 /// so the selection is kept here as a plain value for
299 /// [`Self::adopt_settings_from`] to carry across a power cycle. Read by
300 /// nothing else: synthesis uses only the chain. A save-state restore
301 /// replaces the chain's coefficients and leaves this alone, so after a
302 /// restore it still names the host's selection, which is what the next
303 /// power cycle should rebuild. NEVER serialized (a UI preference).
304 pub(crate) filter_model: crate::mixer::FilterModel,
305 /// v2.1.6 "Expansion Audio" — the most recent RAW external / on-cart
306 /// expansion-audio sample fed into [`Self::tick_with_external`] (BEFORE the
307 /// UI [`Self::channel_gain`] `[5]` re-weight), retained purely so the
308 /// frontend Audio Mixer panel can plot an expansion-channel oscilloscope /
309 /// VU meter alongside the five base-channel DAC taps.
310 ///
311 /// This is a WRITE-ONLY-from-synthesis, READ-ONLY-to-observers copy: it is
312 /// assigned once per tick and is never read back into the mixer, the IRQ
313 /// path, or any determinism-relevant state, and it is NEVER serialized into
314 /// the save state. It therefore cannot perturb the deterministic per-frame
315 /// audio — the visualization samples a copy, exactly like the base-channel
316 /// `*_out()` DAC accessors already do.
317 pub(crate) last_external: f32,
318
319 /// v2.3.7 "Overtone" — audio provenance, behind ONE pointer.
320 ///
321 /// `None` until armed via [`Apu::set_audio_provenance`]. Consolidated into a
322 /// single `Option<Box<..>>` after `apu_throughput` measured +9% on the
323 /// DISARMED path with this state spread across four inline fields — see
324 /// `crate::provenance::AudioProvenance`.
325 #[cfg(feature = "debug-hooks")]
326 pub(crate) audio_prov: Option<alloc::boxed::Box<crate::provenance::AudioProvenance>>,
327}
328
329/// All [`Apu::channel_mask`] bits set — every channel audible (the default and
330/// the determinism-safe value the oracle / test ROMs always run with).
331pub const CHANNEL_MASK_ALL: u8 = 0x3F;
332
333/// All [`Apu::channel_gain`] entries at `1.0` — every channel at full,
334/// unattenuated output (the default and the byte-identical value the oracle /
335/// test ROMs always run with).
336pub const CHANNEL_GAIN_UNITY: [f32; 6] = [1.0; 6];
337
338impl Apu {
339 /// New APU.
340 #[must_use]
341 pub fn new(region: Region, sample_rate: u32) -> Self {
342 let cpu_rate = match region {
343 Region::Pal => CPU_HZ_PAL,
344 _ => CPU_HZ_NTSC,
345 };
346 // v2.1.5: the frame counter selects PAL vs NTSC sequencer step
347 // positions from the region. Only true `Region::Pal` uses the PAL
348 // (2A07) positions; NTSC and Dendy keep the NTSC (2A03) positions, so
349 // their frame-counter timing is byte-identical to the pre-v2.1.5 model.
350 let mut frame_counter = FrameCounter::new();
351 frame_counter.pal = matches!(region, Region::Pal);
352 Self {
353 region,
354 pulse1: Pulse::new(true),
355 pulse2: Pulse::new(false),
356 triangle: Triangle::new(),
357 noise: Noise::new(region),
358 dmc: Dmc::new(region),
359 frame_counter,
360 mixer: Mixer::new(),
361 blip: BlipBuf::new(sample_rate, cpu_rate),
362 apu_phase: false,
363 dmc_driven_externally: false,
364 put_cycle: false,
365 parity_seed: 0,
366 cpu_cycle: 0,
367 pending_dmc_dma: false,
368 pending_dmc_dma_next: false,
369 dmc_dma_is_load: false,
370 dmc_dma_short: false,
371 defer_dmc_reload_once: false,
372 pending_dmc_abort: false,
373 dmc_abort_delay: 0,
374 dmc_dma_cooldown: 0,
375 reset_4017_delay: 0,
376 reset_4017_value: 0,
377 cannot_run_dmc_dma: 0,
378 dmc_reenable_period_block: false,
379 subpos_arm_countdown: 0,
380 dmc_reload_suppress_outputs: 0,
381 dmc_dma_delay: 0,
382 dmc_dma_addr: 0xC000,
383 dmc_need_halt: false,
384 dmc_need_dummy_read: false,
385 dmc_delayed_4015: 0,
386 dmc_delayed_status: false,
387 dmc_status_applied: false,
388 dmc_set_implicit_abort: false,
389 dmc_implicit_abort: false,
390 dmc_edge_arm_suppress: false,
391 sample_rate,
392 last_frame_events: FrameEvents::default(),
393 channel_mask: CHANNEL_MASK_ALL,
394 channel_gain: CHANNEL_GAIN_UNITY,
395 gain_is_unity: true,
396 filter_model: crate::mixer::FilterModel::NesRf,
397 last_external: 0.0,
398 #[cfg(feature = "debug-hooks")]
399 audio_prov: None,
400 }
401 }
402
403 // -----------------------------------------------------------------
404 // v2.3.7 "Overtone" — audio provenance (output-only, off by default)
405 // -----------------------------------------------------------------
406
407 /// Record one mixed CPU cycle — the OUTLINED half.
408 ///
409 /// Called from both mix paths so the fast default-configuration
410 /// specialization and the gated general path produce the same trace: a
411 /// provenance record that existed on only one of two byte-identical paths
412 /// would be a trap for whoever next changed the other.
413 ///
414 /// # Why `#[cold]` and `#[inline(never)]` are load-bearing
415 ///
416 /// This function is measurement-driven twice over, and the second lesson is
417 /// the less obvious one.
418 ///
419 /// The FIRST version built the `MixRecord` before testing whether
420 /// provenance was armed, so a DISARMED build recomputed all five channel
421 /// outputs every CPU cycle — and `Pulse::output` is not free (it calls
422 /// `muted()`, which calls `sweep_target()`). `apu_throughput` measured
423 /// **+14% to +23%** in the feature-on/arm-off configuration the shipped
424 /// frontend runs. Hoisting the arm check to the top fixed that.
425 ///
426 /// It was NOT enough. With the check first, a quiet-host A/B still measured
427 /// **+7.98% / +2.88% / +11.03%** on the three `apu_throughput` workloads
428 /// (order-bias control: +0.11% / +0.76% / +0.67%, so the deltas are real).
429 /// The absolute costs — +33 µs, +15 µs, +65 µs — are wildly non-uniform,
430 /// which a per-cycle branch cannot produce: a constant branch costs a
431 /// constant number of cycles. The cause was that this body was still being
432 /// INLINED into `tick_with_external`. The five `output()` calls sat in the
433 /// hot function even though the branch skipped over them, inflating it past
434 /// the point where the mixer and the channel ticks kept their registers and
435 /// their I-cache line.
436 ///
437 /// So the hot path now contains exactly one null test, and everything else
438 /// lives out of line behind it. `#[cold]` additionally tells LLVM to lay
439 /// this block out away from the fall-through path. It pessimizes the ARMED
440 /// case, which is the correct trade: armed is an interactive debugging mode
441 /// and disarmed is what every user runs.
442 #[cfg(feature = "debug-hooks")]
443 #[cold]
444 #[inline(never)]
445 fn record_mix_armed(&mut self, mixed: f32, external: f32) {
446 let rec = crate::provenance::MixRecord {
447 mixed,
448 external,
449 pulse1: self.pulse1.output(),
450 pulse2: self.pulse2.output(),
451 triangle: self.triangle.output(),
452 noise: self.noise.output(),
453 dmc: self.dmc.output(),
454 };
455 if let Some(p) = self.audio_prov.as_mut() {
456 p.mix_trace.push(rec);
457 }
458 }
459
460 /// Arm or disarm audio provenance.
461 ///
462 /// Arming allocates both stores; disarming frees them. Mirrors
463 /// `Ppu::set_pixel_provenance`, including that re-arming an already-armed
464 /// APU is a no-op rather than a silent wipe — the frontend re-asserts the
465 /// arm every frame (a lesson from the pixel panel, whose edge-triggered
466 /// mirror desynced permanently the moment a ROM load installed a fresh
467 /// core).
468 #[cfg(feature = "debug-hooks")]
469 pub fn set_audio_provenance(&mut self, enabled: bool) {
470 if enabled {
471 if self.audio_prov.is_none() {
472 self.audio_prov = Some(alloc::boxed::Box::new(
473 crate::provenance::AudioProvenance::new(),
474 ));
475 }
476 } else {
477 self.audio_prov = None;
478 }
479 }
480
481 /// Whether audio provenance is armed.
482 #[cfg(feature = "debug-hooks")]
483 #[must_use]
484 pub const fn audio_provenance_armed(&self) -> bool {
485 self.audio_prov.is_some()
486 }
487
488 /// The per-register write attribution, or `None` when disarmed.
489 #[cfg(feature = "debug-hooks")]
490 #[must_use]
491 pub fn register_attribution(&self) -> Option<&crate::provenance::RegisterAttribution> {
492 self.audio_prov.as_ref().map(|p| &p.reg_attrib)
493 }
494
495 /// The per-CPU-cycle mix trace, or `None` when disarmed.
496 #[cfg(feature = "debug-hooks")]
497 #[must_use]
498 pub fn mix_trace(&self) -> Option<&crate::provenance::MixTrace> {
499 self.audio_prov.as_ref().map(|p| &p.mix_trace)
500 }
501
502 /// Begin a new frame's mix trace, anchored at `first_cycle`.
503 ///
504 /// The register attribution is deliberately NOT cleared here: "which
505 /// instruction last wrote `$4003`" is a question whose answer legitimately
506 /// predates the current frame, and clearing it every frame would report a
507 /// register nobody has touched this frame as never written.
508 #[cfg(feature = "debug-hooks")]
509 pub fn begin_audio_provenance_frame(&mut self, first_cycle: u64) {
510 if let Some(p) = self.audio_prov.as_mut() {
511 p.mix_trace.clear(first_cycle);
512 }
513 }
514
515 /// Forget the register attribution history. Called on a cold boot, where
516 /// the history it describes genuinely ended.
517 #[cfg(feature = "debug-hooks")]
518 pub fn clear_audio_provenance_history(&mut self) {
519 if let Some(p) = self.audio_prov.as_mut() {
520 p.reg_attrib.clear();
521 }
522 }
523
524 /// Attribute a write in `$4000-$4017` that the bus does NOT route through
525 /// [`Self::write_register`].
526 ///
527 /// Two addresses in the range are not APU registers and are handled
528 /// entirely on the bus: `$4014` (OAM DMA, which arms a burst) and `$4016`
529 /// (controller strobe, which is buffered to the next M2-low boundary).
530 /// `Bus::write` dispatches only `$4000-$4013 | $4015 | $4017` to
531 /// `write_register`, so the attribution recorded there can never see those
532 /// two — yet the table reserves slots for them, because the range is what
533 /// the bus already classifies as an APU write and punching a hole in it
534 /// would invite off-by-one arithmetic at every call site.
535 ///
536 /// Without this entry point those two slots would stay permanently empty
537 /// while the docs claimed they were tracked. This records the cause exactly
538 /// as `write_register` would, and dispatches nothing — the emulation of both
539 /// addresses stays wherever the bus already implements it.
540 #[cfg(feature = "debug-hooks")]
541 pub const fn record_bus_handled_register_write(&mut self, addr: u16, value: u8) {
542 if let Some(p) = self.audio_prov.as_mut() {
543 p.reg_attrib
544 .record(addr, p.attrib_pc, p.attrib_cycle, value);
545 }
546 }
547
548 /// Push the writing instruction's PC + cycle down, mirroring the PPU's
549 /// write-attribution context. Called once per instruction by the core.
550 #[cfg(feature = "debug-hooks")]
551 pub const fn set_attrib_context(&mut self, pc: u16, cycle: u64) {
552 // No-op when disarmed: nothing reads these, so skipping the stores keeps
553 // the disarmed per-instruction cost at one null test.
554 if let Some(p) = self.audio_prov.as_mut() {
555 p.attrib_pc = pc;
556 p.attrib_cycle = cycle;
557 }
558 }
559
560 /// Lift both stores out for a same-timeline restore (run-ahead), leaving the
561 /// APU disarmed. See [`crate::provenance::AudioProvenanceStash`] for why
562 /// this exists at all.
563 #[cfg(feature = "debug-hooks")]
564 #[must_use]
565 pub fn take_audio_provenance(&mut self) -> crate::provenance::AudioProvenanceStash {
566 crate::provenance::AudioProvenanceStash {
567 state: self.audio_prov.take(),
568 }
569 }
570
571 /// Put back stores taken by [`Self::take_audio_provenance`].
572 #[cfg(feature = "debug-hooks")]
573 pub fn put_audio_provenance(&mut self, stash: crate::provenance::AudioProvenanceStash) {
574 self.audio_prov = stash.state;
575 }
576
577 /// Reset (warm). Per nesdev: most APU state is preserved across reset
578 /// except `$4015` is cleared (channels disabled, DMC silenced).
579 ///
580 /// v2.0.0 beta.3 (A4 cycle-accurate reset, promoted to the only path in
581 /// beta.4): the 2A03
582 /// reset sequence behaves as if the LAST value written to `$4017` were
583 /// written again (blargg `apu_reset` spec) — the retained value is
584 /// re-issued through the normal `$4017` write path (pending mode + the
585 /// 3/4-cycle aligned delay + the mode-1 immediate quarter/half clock),
586 /// and the CPU's 8 clocked reset cycles then age the re-armed counter
587 /// so execution resumes ~9-12 cycles after the effective write (the
588 /// `4017_timing` window).
589 pub fn reset(&mut self) {
590 // Zero the sequencer + IRQ flags now; SCHEDULE the hardware
591 // `$4017` re-write to land 2 clocked cycles into the CPU's
592 // 8-cycle reset sequence (consumed in `tick_with_external`).
593 // Empirically calibrated against blargg `4017_timing`'s printed
594 // "delay after effective $4017 write" (accept window 6..=12,
595 // hardware-usual 9; the ROM quantizes in 2-cycle APU units): an
596 // immediate reset-start re-write measures 12 (the upper edge),
597 // a +3-cycle placement measures 6 (the lower edge), and +2
598 // lands mid-window at 8.
599 let last = self.frame_counter.reset_rewrite_4017();
600 self.reset_4017_value = last;
601 self.reset_4017_delay = 2;
602 self.write_register(0x4015, 0x00);
603 // v2.3.7 — that write went through the ordinary CPU path, which just
604 // attributed it to whatever instruction was last latched. No instruction
605 // caused it: this models the warm-reset silencing of the channels.
606 // Correct the origin so the panel reports hardware rather than naming an
607 // innocent PC. (Caught in review of the PR that added the feature.)
608 #[cfg(feature = "debug-hooks")]
609 if let Some(p) = self.audio_prov.as_mut() {
610 p.reg_attrib.record_reset(0x4015, p.attrib_cycle, 0x00);
611 }
612 self.pending_dmc_dma = false;
613 self.dmc_dma_is_load = false;
614 self.dmc_dma_short = false;
615 self.defer_dmc_reload_once = false;
616 self.pending_dmc_abort = false;
617 self.dmc_abort_delay = 0;
618 self.dmc_dma_cooldown = 0;
619 self.cannot_run_dmc_dma = 0;
620 self.dmc_reenable_period_block = false;
621 self.dmc_reload_suppress_outputs = 0;
622 self.dmc_dma_delay = 0;
623 self.dmc_need_halt = false;
624 self.dmc_need_dummy_read = false;
625 // W3-Stage-3: a warm reset silences the DMC immediately — collapse the
626 // delayed-application machinery to the applied-disabled state (the
627 // `write_register(0x4015, 0)` above latched a deferred disable).
628 {
629 self.dmc_delayed_4015 = 0;
630 self.dmc_delayed_status = false;
631 self.dmc_status_applied = false;
632 self.dmc_set_implicit_abort = false;
633 self.dmc_implicit_abort = false;
634 self.dmc_edge_arm_suppress = false;
635 self.dmc.bytes_remaining = 0;
636 }
637 self.blip.reset();
638 }
639
640 /// Set the per-channel enable mask (a UI playback overlay; see
641 /// [`Apu::channel_mask`]). Bit 0 = pulse 1, 1 = pulse 2, 2 = triangle,
642 /// 3 = noise, 4 = DMC, 5 = external/mapper audio. [`CHANNEL_MASK_ALL`] is
643 /// the determinism-safe default (byte-identical mixer output).
644 pub const fn set_channel_mask(&mut self, mask: u8) {
645 self.channel_mask = mask & CHANNEL_MASK_ALL;
646 }
647
648 /// Current per-channel enable mask.
649 #[must_use]
650 pub const fn channel_mask(&self) -> u8 {
651 self.channel_mask
652 }
653
654 /// v1.4.0 Workstream C — set the per-channel output gain (a UI mixing
655 /// overlay; see [`Apu::channel_gain`]). Index 0 = pulse 1, 1 = pulse 2,
656 /// 2 = triangle, 3 = noise, 4 = DMC, 5 = external/mapper audio. Each gain is
657 /// clamped to `0.0..=2.0`; a NaN (which `f32::clamp` would pass through,
658 /// turning every mixed sample into NaN) becomes unity. [`CHANNEL_GAIN_UNITY`]
659 /// (all `1.0`) is the determinism-safe default (byte-identical mixer output).
660 pub fn set_channel_gain(&mut self, gain: [f32; 6]) {
661 for (slot, g) in self.channel_gain.iter_mut().zip(gain.iter()) {
662 *slot = if g.is_nan() { 1.0 } else { g.clamp(0.0, 2.0) };
663 }
664 self.gain_is_unity = self.channel_gain == CHANNEL_GAIN_UNITY;
665 }
666
667 /// v2.1.3 — select the analog output-filter model (see
668 /// [`crate::mixer::FilterModel`]). Default [`crate::mixer::FilterModel::NesRf`]
669 /// is byte-identical to the pre-v2.1.3 output; the softer models drop the
670 /// aggressive 440 Hz high-pass for a fuller low end. Display/tonal only —
671 /// channel content is unchanged.
672 pub fn set_filter_model(&mut self, model: crate::mixer::FilterModel) {
673 self.filter_model = model;
674 self.blip.set_filter_model(model);
675 }
676
677 /// v2.9.8 — the analog output-filter model last selected with
678 /// [`Self::set_filter_model`] ([`crate::mixer::FilterModel::NesRf`] until
679 /// one is).
680 #[must_use]
681 pub const fn filter_model(&self) -> crate::mixer::FilterModel {
682 self.filter_model
683 }
684
685 /// v2.9.8 — carry the host's settings from `prev` onto this freshly built
686 /// APU, so a power cycle (which rebuilds the APU from [`Self::new`]) keeps
687 /// them.
688 ///
689 /// Until v2.9.8 the bus rebuilt the APU and re-applied only its own wiring
690 /// (the externally driven DMC and the alignment seed), so the channel mask,
691 /// the per-channel gain and the filter model reverted to their defaults
692 /// and every host had to push them again. Carried: [`Self::channel_mask`],
693 /// [`Self::channel_gain`] and [`Self::filter_model`]. The filter goes
694 /// through [`Self::set_filter_model`], which builds a fresh chain for the
695 /// model at this APU's sample rate -- the chain a fresh console gets when
696 /// a host selects the model, with no IIR history carried from the old
697 /// timeline. The sample rate is the caller's to pass to [`Self::new`];
698 /// the audio provenance stores are moved by the core, armed and emptied.
699 ///
700 /// With every setting at its default this leaves the APU byte-identical
701 /// to [`Self::new`]: the default model's chain is the one `new` builds.
702 pub fn adopt_settings_from(&mut self, prev: &Self) {
703 self.channel_mask = prev.channel_mask;
704 // Through the setter, never a field copy: it also refreshes the cached
705 // `gain_is_unity`, which the per-cycle mix reads instead of the gain.
706 // A bare `self.channel_gain = prev.channel_gain` would leave the fresh
707 // APU's `true` in place and mix at unity gain after a power cycle.
708 // `a_power_cycle_keeps_a_non_unity_gain_audible` pins it.
709 self.set_channel_gain(prev.channel_gain);
710 self.set_filter_model(prev.filter_model);
711 }
712
713 /// Current per-channel output gain. See [`Apu::set_channel_gain`].
714 #[must_use]
715 pub const fn channel_gain(&self) -> [f32; 6] {
716 self.channel_gain
717 }
718
719 /// Pulse 1 raw output volume (0..=15) — for tests.
720 #[must_use]
721 pub fn pulse1_out(&self) -> u8 {
722 self.pulse1.output()
723 }
724 /// Pulse 2 raw output volume.
725 #[must_use]
726 pub fn pulse2_out(&self) -> u8 {
727 self.pulse2.output()
728 }
729 /// Triangle raw output (0..=15).
730 #[must_use]
731 pub fn triangle_out(&self) -> u8 {
732 self.triangle.output()
733 }
734 /// Noise raw output (0..=15).
735 #[must_use]
736 pub fn noise_out(&self) -> u8 {
737 self.noise.output()
738 }
739 /// DMC raw output (0..=127).
740 #[must_use]
741 pub const fn dmc_out(&self) -> u8 {
742 self.dmc.output()
743 }
744
745 /// v2.1.6 "Expansion Audio" — the most recent RAW on-cart expansion-audio
746 /// sample (pre-[`Self::channel_gain`], the `last_external` field). `0.0`
747 /// when the loaded board has no expansion audio. Read-only display tap for
748 /// the frontend Audio Mixer expansion-channel scope / VU meter — it reads a
749 /// copy and never feeds back into synthesis, so it is determinism-neutral.
750 #[must_use]
751 pub const fn external_out(&self) -> f32 {
752 self.last_external
753 }
754
755 /// Frame IRQ pending?
756 #[must_use]
757 pub const fn frame_irq_pending(&self) -> bool {
758 self.frame_counter.irq_flag
759 }
760
761 /// DMC IRQ pending?
762 #[must_use]
763 pub const fn dmc_irq_pending(&self) -> bool {
764 self.dmc.irq_flag
765 }
766
767 /// Combined IRQ line — true if either source is asserting.
768 ///
769 /// Session-26 iter 5 (2026-05-23): the frame-counter contribution
770 /// is `irq_line_active` (the CPU's `IRQSource::FrameCounter`
771 /// registration), NOT `irq_flag` (the `$4015` bit 6 visibility).
772 /// The two are SEPARATE fields since iter 5 — see
773 /// [`FrameCounter::irq_flag`](crate::frame_counter::FrameCounter::irq_flag)
774 /// and [`FrameCounter::irq_line_active`](crate::frame_counter::FrameCounter::irq_line_active).
775 /// AccuracyCoin Tests I/J/K specifically test that `$4015` bit 6
776 /// is visible during inhibit (transient 2-cycle window at FC steps
777 /// 29828-29829) while NO CPU IRQ fires (Test M).
778 #[must_use]
779 pub const fn irq_line(&self) -> bool {
780 self.frame_counter.irq_line_active || self.dmc.irq_flag
781 }
782
783 /// Returns the frame-counter events fired by the most recent `tick` call.
784 ///
785 /// The bus reads this immediately after [`Self::tick_with_external`] to
786 /// fan-out the events to on-cart audio extensions (MMC5) whose envelope
787 /// and length-counter sub-units share the 2A03 frame-counter cadence.
788 /// The value is overwritten at the start of every `tick`, so observers
789 /// must consume it before the next tick.
790 #[must_use]
791 pub const fn last_frame_events(&self) -> FrameEvents {
792 self.last_frame_events
793 }
794
795 /// Drain all finalized audio samples (host sample rate, normalized to
796 /// approximately `[-0.5, 0.5]`).
797 pub fn drain_audio(&mut self) -> Vec<f32> {
798 self.blip.drain_all()
799 }
800
801 /// Drain into a slice; returns count copied.
802 pub fn drain_audio_into(&mut self, out: &mut [f32]) -> usize {
803 self.blip.drain(out)
804 }
805
806 /// Has a DMC DMA request been raised? The bus polls this each CPU cycle
807 /// (BEFORE issuing reads) so it can halt the CPU on the next read cycle.
808 #[must_use]
809 pub const fn dmc_dma_pending(&self) -> bool {
810 self.pending_dmc_dma
811 }
812
813 /// Whether the pending DMC DMA is a load DMA.
814 #[must_use]
815 pub const fn dmc_dma_is_load(&self) -> bool {
816 self.dmc_dma_is_load
817 }
818
819 /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the bus-side per-cycle DMC DMA
820 /// service-gate term — TriCNES `_6502` line 4218:
821 /// `DoDMCDMA && (APU_Status_DMC || APU_ImplicitAbortDMC4015)`. While
822 /// false, a pending (or halted in-flight) DMC DMA is NOT serviced and the
823 /// CPU resumes — the emergent explicit abort. `pending_dmc_abort` is the
824 /// implicit-abort override (the 1-cycle abort DMA runs regardless).
825 #[must_use]
826 pub const fn dmc_dma_serviceable(&self) -> bool {
827 self.dmc_status_applied || self.dmc_implicit_abort || self.pending_dmc_abort
828 }
829
830 /// Whether the pending DMC DMA should use the short 3-cycle service path.
831 #[must_use]
832 pub const fn dmc_dma_short(&self) -> bool {
833 self.dmc_dma_short
834 }
835
836 /// Whether the current DMC DMA get should make the fetched byte visible
837 /// before the get-cycle APU tick.
838 #[must_use]
839 pub const fn dmc_dma_deliver_before_tick(&self) -> bool {
840 self.dmc.loop_flag
841 && self.dmc.sample_length == 1
842 && self.dmc.rate_index == 0x0E
843 && self.dmc.bits_remaining == 1
844 && self.dmc.timer == 0
845 && !self.apu_phase
846 }
847
848 /// Defer the next immediate DMC reload request by one CPU tick.
849 pub const fn defer_next_dmc_reload_once(&mut self) {
850 self.defer_dmc_reload_once = true;
851 }
852
853 /// Has a one-cycle DMC abort halt been raised?
854 #[must_use]
855 pub const fn dmc_abort_pending(&self) -> bool {
856 self.pending_dmc_abort
857 }
858
859 /// Read-only accessor for the DMC abort-delay countdown (CPU cycles
860 /// until `pending_dmc_abort` flips to `true`). Exposed for the
861 /// Session-21 per-cycle DMC trace tooling (`crates/rustynes-core/src/
862 /// irq_trace.rs`) which records the scheduler's calibration state
863 /// for cross-diffing against Mesen2's `NesDmc.cpp`.
864 #[must_use]
865 pub const fn dmc_abort_delay(&self) -> u8 {
866 self.dmc_abort_delay
867 }
868
869 /// Read-only accessor for the DMC DMA cooldown countdown (CPU cycles
870 /// during which a newly-empty sample buffer must NOT raise a new
871 /// DMA request). See [`Self::dmc_abort_delay`].
872 #[must_use]
873 pub const fn dmc_dma_cooldown(&self) -> u8 {
874 self.dmc_dma_cooldown
875 }
876
877 /// Read-only accessor for the DMC DMA delay countdown (CPU cycles
878 /// until an initial-load DMA after `$4015` enable transitions from
879 /// "armed" to `pending_dmc_dma = true`). See [`Self::dmc_abort_delay`].
880 #[must_use]
881 pub const fn dmc_dma_delay(&self) -> u8 {
882 self.dmc_dma_delay
883 }
884
885 /// Diagnostic: the DMC channel's internal byte-timer countdown. Exposed
886 /// for the per-cycle DMC-DMA cross-diff tracing that pins the abort-context
887 /// reload-arm phase (the +4-cycle `A->B` interval divergence).
888 #[must_use]
889 pub const fn dmc_timer(&self) -> u16 {
890 self.dmc.timer()
891 }
892
893 /// Diagnostic: bits remaining in the DMC output shift register.
894 #[must_use]
895 pub const fn dmc_bits_remaining(&self) -> u8 {
896 self.dmc.bits_remaining()
897 }
898
899 /// Diagnostic: DMC output-unit silence flag.
900 #[must_use]
901 pub const fn dmc_silence(&self) -> bool {
902 self.dmc.silence()
903 }
904
905 /// Diagnostic: DMC sample buffer occupied.
906 #[must_use]
907 pub const fn dmc_buffer_full(&self) -> bool {
908 self.dmc.buffer_full()
909 }
910
911 /// Read-only accessor for the APU's two-cycle phase counter
912 /// (false = put, true = get; toggled every CPU tick by
913 /// `tick_with_external`). See [`Self::dmc_abort_delay`].
914 #[must_use]
915 pub const fn apu_phase(&self) -> bool {
916 self.apu_phase
917 }
918
919 /// v2.0.0 beta.1 (A1 one-clock collapse): assign the APU's cycle counter
920 /// from the CANONICAL bus cycle counter. Called by the bus's per-cycle
921 /// hook (`cpu_clock` → `apu_advance_one`) immediately before
922 /// [`Self::tick_with_external`], replacing the legacy independent
923 /// `cpu_cycle += 1` mirror. The bus increments its canonical counter
924 /// earlier in the same per-cycle hook, so the value assigned here equals
925 /// the post-increment value the legacy mirror produced — the
926 /// `one_clock_invariants` harness test pins the residue. Promoted to the
927 /// only path in v2.0.0 beta.4.
928 pub const fn set_canonical_cycle(&mut self, cycle: u64) {
929 self.cpu_cycle = cycle;
930 }
931
932 /// Read-only accessor for the APU-side cumulative CPU-cycle counter
933 /// (v2.0.0-beta.1 one-clock instrumentation).
934 ///
935 /// This is one of the five counters of the timebase substrate the
936 /// v2.0.0 "Timebase" rewrite collapses (ADR 0002 + the v2.0.0
937 /// master-clock plan): `Cpu::master_clock`, `Cpu::cycles`,
938 /// `SystemBus::cycle`, `SystemBus::ppu_clock`, and this field are
939 /// each advanced exactly once (or by one region divider) per CPU cycle
940 /// at different points *within* the cycle, and must never drift. The
941 /// RW-1 parity collapse already derives `apu_phase` / `put_cycle` from
942 /// `(cpu_cycle + parity_seed) & 1`; exposing the raw counter lets the
943 /// test harness assert the cross-chip affine invariants
944 /// (`one_clock_invariants.rs`) that gate the beta.1 counter collapse.
945 #[must_use]
946 pub const fn cpu_cycle(&self) -> u64 {
947 self.cpu_cycle
948 }
949
950 /// CM-1: seed the absolute `apu_phase` alignment (the parity of the CPU
951 /// cycles on which the APU — incl. the DMC byte-timer — clocks). RustyNES
952 /// starts `apu_phase = false`, so the DMC always arms on one CPU-cycle
953 /// parity; Mesen's DMC arms on its `_currentCycle` alignment (one cycle off,
954 /// giving span 3 vs RustyNES's 4). Seeding `true` flips the whole APU phase
955 /// by one CPU cycle to test the Mesen-matching arm parity. Broad impact:
956 /// also shifts pulse/noise/frame-counter. Default-off.
957 pub const fn seed_apu_phase(&mut self, phase: bool) {
958 self.apu_phase = phase;
959 }
960
961 /// Address the DMC wants to read. Valid only when `dmc_dma_pending()`
962 /// returns `true`.
963 #[must_use]
964 pub const fn dmc_dma_addr(&self) -> u16 {
965 self.dmc_dma_addr
966 }
967
968 /// v1.2 Sprint 3 (get/put scheduler, ADR-0007).
969 ///
970 /// Returns `true` while the DMC still needs an initial halt
971 /// cycle on the bus. Set by any code path that raises
972 /// `pending_dmc_dma`; the new `bus::service_dmc_dma`
973 /// implementation under the `dmc-get-put-scheduler` feature
974 /// flag clears it after processing the halt get-cycle.
975 #[must_use]
976 pub const fn dmc_need_halt(&self) -> bool {
977 self.dmc_need_halt
978 }
979
980 /// v1.2 Sprint 3 (get/put scheduler, ADR-0007).
981 ///
982 /// Returns `true` while the DMC still needs a dummy-read /
983 /// alignment cycle after the halt cycle. Cleared by the new
984 /// scheduler once the alignment cycle has been processed.
985 #[must_use]
986 pub const fn dmc_need_dummy_read(&self) -> bool {
987 self.dmc_need_dummy_read
988 }
989
990 /// v1.2 Sprint 3 — bus clears this after consuming the halt
991 /// get-cycle (`_needHalt = false` in Mesen2's `NesCpu.cpp`).
992 pub const fn clear_dmc_need_halt(&mut self) {
993 self.dmc_need_halt = false;
994 }
995
996 /// v1.2 Sprint 3 — bus clears this after consuming the
997 /// alignment cycle (`_needDummyRead = false` in Mesen2's
998 /// `NesCpu.cpp`).
999 pub const fn clear_dmc_need_dummy_read(&mut self) {
1000 self.dmc_need_dummy_read = false;
1001 }
1002
1003 /// Bus calls this when it has executed a DMC DMA fetch (post-halt) and
1004 /// is delivering the sample byte.
1005 pub fn complete_dmc_dma(&mut self, byte: u8) {
1006 self.dmc.deliver_sample(byte);
1007 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `DMCDMA_Get`
1008 // (Emulator.cs:4148-4160) — a non-looping sample's natural end clears
1009 // the APPLIED status immediately (not via the delayed slot). A
1010 // looping end restarts the sample (`deliver_sample` already did), so
1011 // `bytes_remaining > 0` and the status holds.
1012 if self.dmc.bytes_remaining == 0 && !self.dmc.loop_flag {
1013 self.dmc_status_applied = false;
1014 }
1015 let was_load = self.dmc_dma_is_load;
1016 // v2.0 abort-context reload-arm phase fix (`mc-r1-dmc-abort-timer-phase`).
1017 // In the Implicit-DMA-Abort `$4015` disable->re-enable context, this LOAD
1018 // DMA's `deliver_sample` (just above) lands ON the byte-timer boundary
1019 // cycle, where `clock_output` already ran at cycle-START and took the
1020 // still-empty buffer (silence) BEFORE the LOAD filled it. TriCNES's LOAD
1021 // GET completes 3 cyc BEFORE the boundary, so the boundary consumes the
1022 // buffer into the shifter and arms a RELOAD (inserting a 4-cyc reload DMA
1023 // RustyNES otherwise skips, deferring the reload chain by 4 -> A->B 580
1024 // not 576 -> GET-catch skew +4 -> Y=0). Detect the boundary-coincidence
1025 // (silence set + bits just reloaded to 8 + buffer now full + bytes
1026 // remaining) and retroactively load the delivered byte into the shifter
1027 // (un-silence) so the buffer empties and the per-cycle reload-arm fires
1028 // promptly this cycle (cooldown cleared below) — reproducing TriCNES's
1029 // boundary-coupled reload. The condition only ever holds in this race.
1030 let abort_boundary_race = was_load
1031 && self.dmc.silence
1032 && self.dmc.bits_remaining == 8
1033 && self.dmc.bytes_remaining > 0
1034 && self.dmc.consume_buffer_into_shifter_if_silent();
1035 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): this load-completion abort
1036 // scheduling is the FLOOR's implicit-abort model (a pre-computed
1037 // 1-cycle halt via `dmc_abort_delay` -> `pending_dmc_abort` -> the
1038 // read1 abort-cancel path). Under the delayed-status port the same
1039 // physics is EMERGENT (the `$4015`-enable pre-fire-window latch + the
1040 // consume-edge arm + the 1-cycle override kill), so the floor
1041 // scheduler is superseded — both active would double-fire ($500
1042 // idx[10,11] measured 05 vs KEY 01).
1043 self.pending_dmc_dma = false;
1044 self.dmc_dma_is_load = false;
1045 self.dmc_dma_short = false;
1046 self.dmc_dma_delay = 0;
1047 self.dmc_dma_cooldown = 4;
1048 // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES `DMCDMA_Get`
1049 // (`Emulator.cs:4168`) sets `CannotRunDMCDMARightNow = 2` after EVERY
1050 // DMC GET (load or reload). The exclusion is decremented by 2 per get
1051 // cycle in `tick_with_external` and blocks the looping-reload arm while
1052 // `== 2` — the canonical "a DMA cannot occur within 2 cycles of a
1053 // previous DMC DMA" rule the Implicit-DMA-Abort `$540` plateau brackets.
1054 self.cannot_run_dmc_dma = 2;
1055 // Abort-context fix: the LOAD just emptied the buffer into the shifter
1056 // (boundary race), so the next reload must arm promptly — don't let the
1057 // post-LOAD cooldown suppress it for 4 cycles (which would re-introduce
1058 // the +4). Clear the cooldown so `dmc_step_reload_arm` fires next cycle.
1059 if abort_boundary_race {
1060 self.dmc_dma_cooldown = 0;
1061 }
1062 // v1.2 Sprint 3 — safety-clear the get/put flags on
1063 // completion. Under the new scheduler the bus should have
1064 // cleared them on the prior cycles already; clearing here
1065 // protects against re-arming on the next DMC request.
1066 self.dmc_need_halt = false;
1067 self.dmc_need_dummy_read = false;
1068 // W3-Stage-2 (`mc-r1-dma-unified-collapse`): this guard's phase term
1069 // means "the GET landed on the off-phase half". The normal GET half is
1070 // `apu_phase`-true at floor but `apu_phase`-false under the end-flip,
1071 // so the off-phase test inverts — otherwise the (floor-dead) suppress
1072 // path would fire at EVERY collapse GET in the 1-byte-loop contexts.
1073 let off_phase_get = self.apu_phase;
1074 if was_load
1075 && self.dmc.loop_flag
1076 && self.dmc.sample_length == 1
1077 && self.dmc.bits_remaining == 1
1078 && self.dmc.timer == 0
1079 && self.dmc.sample_buffer.is_some()
1080 && off_phase_get
1081 {
1082 self.dmc_reload_suppress_outputs = 1;
1083 }
1084 }
1085
1086 /// Complete a DMC DMA get whose fetched byte is visible before the
1087 /// get-cycle APU tick.
1088 pub fn complete_dmc_dma_before_get_tick(&mut self, byte: u8) {
1089 let was_load = self.dmc_dma_is_load;
1090 self.dmc.deliver_sample(byte);
1091 // W3-Stage-3: see `complete_dmc_dma` — non-looping natural end clears
1092 // the applied status immediately.
1093 if self.dmc.bytes_remaining == 0 && !self.dmc.loop_flag {
1094 self.dmc_status_applied = false;
1095 }
1096 if was_load
1097 && self.dmc.bytes_remaining == 0
1098 && !self.dmc.loop_flag
1099 && self.dmc.bits_remaining == 1
1100 && self.dmc.timer == 0
1101 && self.dmc.sample_buffer.is_some()
1102 {
1103 self.dmc_abort_delay = 3;
1104 }
1105 if was_load && self.dmc.loop_flag && self.dmc.sample_length == 1 {
1106 self.defer_dmc_reload_once = true;
1107 }
1108 self.pending_dmc_dma = false;
1109 self.dmc_dma_is_load = false;
1110 self.dmc_dma_short = false;
1111 self.dmc_dma_delay = 0;
1112 self.dmc_dma_cooldown = 5;
1113 // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): see `complete_dmc_dma`.
1114 {
1115 self.cannot_run_dmc_dma = 2;
1116 }
1117 self.dmc_need_halt = false;
1118 self.dmc_need_dummy_read = false;
1119 }
1120
1121 /// Bus calls this after either consuming or suppressing a one-cycle DMC
1122 /// abort halt.
1123 pub const fn complete_dmc_abort(&mut self) {
1124 self.pending_dmc_abort = false;
1125 self.dmc_abort_delay = 0;
1126 }
1127
1128 /// v1.2 Sprint 3 iter 3 (get/put scheduler, ADR 0007) — DMC DMA
1129 /// abort with cancel semantics.
1130 ///
1131 /// Under the OLD scheduler, [`Self::complete_dmc_abort`] clears
1132 /// only the abort flag; the DMC DMA still fires afterward (the
1133 /// abort just inserts a 1-cycle halt). Under the get/put model
1134 /// the abort CANCELS the DMA entirely — no byte fetch, all
1135 /// flag state cleared — matching Mesen2's
1136 /// `processCycle::if(_abortDmcDma)` branch
1137 /// (`NesCpu.cpp:386-390`):
1138 ///
1139 /// ```text
1140 /// if(_abortDmcDma) {
1141 /// _dmcDmaRunning = false;
1142 /// _abortDmcDma = false;
1143 /// _needDummyRead = false;
1144 /// _needHalt = false;
1145 /// }
1146 /// ```
1147 ///
1148 /// This is the "Option C" semantic shift from the iter 3
1149 /// research audit: abort cancels the fetch rather than letting
1150 /// it complete after a wasted cycle. The new bus-side
1151 /// `service_dmc_dma` (under `dmc-get-put-scheduler` feature)
1152 /// calls this when it detects `dmc_abort_pending` mid-loop.
1153 pub const fn cancel_dmc_dma(&mut self) {
1154 self.pending_dmc_dma = false;
1155 self.pending_dmc_abort = false;
1156 self.dmc_abort_delay = 0;
1157 self.dmc_dma_short = false;
1158 self.dmc_dma_is_load = false;
1159 self.dmc_dma_delay = 0;
1160 self.dmc_need_halt = false;
1161 self.dmc_need_dummy_read = false;
1162 }
1163
1164 /// One CPU clock. Bus must NOT have halted the CPU for DMC DMA when
1165 /// calling this (the bus is responsible for performing the DMA fetch
1166 /// before resuming `tick()` calls).
1167 ///
1168 /// Standalone/test convenience: production (`SystemBus`) drives the
1169 /// canonical cycle counter via [`Self::set_canonical_cycle`] before each
1170 /// [`Self::tick_with_external`] (the v2.0.0 one-clock contract — the APU
1171 /// never self-increments). This helper self-advances the counter so
1172 /// standalone APU stepping (unit tests, the snapshot fixtures) keeps the
1173 /// one-cycle-per-tick behavior.
1174 pub fn tick(&mut self) {
1175 self.cpu_cycle = self.cpu_cycle.wrapping_add(1);
1176 self.tick_with_external(0.0);
1177 }
1178
1179 /// Same as `tick`, but accepts an additional pre-mixed audio sample
1180 /// from the cartridge (VRC6 / VRC7 / MMC5 / Sunsoft 5B / Namco 163 /
1181 /// FDS). The external value is added to the APU's own mix BEFORE the
1182 /// band-limited buffer push.
1183 ///
1184 /// The expected scale is ~ `[-0.5, 0.5]` (matching the APU mixer's
1185 /// own output range). The bus is responsible for converting whatever
1186 /// the mapper returns (currently `i16` from `Mapper::mix_audio`) into
1187 /// that range.
1188 pub fn tick_with_external(&mut self, external: f32) {
1189 // v2.0.0 beta.1 (A1 one-clock collapse, promoted to the only path in
1190 // beta.4): the APU's cycle counter is ASSIGNED from the canonical
1191 // bus counter (see `set_canonical_cycle`, called by the bus
1192 // immediately before this tick) instead of being an
1193 // independently-incremented lockstep mirror. The RW-1
1194 // `apu_phase`/`put_cycle` parity derivation below then reads from
1195 // the ONE counter.
1196
1197 // v2.0 RA-1: the DMC byte-timer + arms could clock HERE at cycle START
1198 // (on `apu_phase`), unified with the rest of the APU — Mesen
1199 // `ProcessCpuClock` at `StartCpuCycle`.
1200 //
1201 // v2.0 Program M (M-1 within-cycle order): the DMC byte-timer CLOCK +
1202 // reload-arm + reenable bookkeeping live at end-of-cycle (after the CPU's
1203 // bus access, in `dmc_tick_end`), matching Mesen `StartCpuCycle`->
1204 // `ProcessCpuClock` and TriCNES `_6502`->`_EmulateAPU` (CPU reads state ->
1205 // APU ticks/arms reload -> get/put flips). The reload arm thereby becomes
1206 // invisible to its own cycle -> first-service is the next (put) cycle ->
1207 // span-4. So the cycle-START DMC clock/arm paths below are never taken;
1208 // the LOAD delay-arm moves to the put phase of `dmc_tick_end` (the TriCNES
1209 // `DMCDMADelay` put-branch placement, Emulator.cs:1217).
1210
1211 if self.dmc_abort_delay > 0 {
1212 self.dmc_abort_delay -= 1;
1213 if self.dmc_abort_delay == 0 && !self.pending_dmc_abort {
1214 self.pending_dmc_abort = true;
1215 }
1216 }
1217 if self.dmc_dma_cooldown > 0 {
1218 self.dmc_dma_cooldown -= 1;
1219 }
1220
1221 // Triangle clocks at CPU rate.
1222 self.triangle.clock_timer();
1223
1224 // Pulse, noise, DMC clock at APU rate (every other CPU cycle).
1225 // RW-1 (`mc-r1-one-clock`): DERIVE `apu_phase` from the single per-cycle
1226 // counter + boot seed instead of a free-running toggle, so it shares ONE
1227 // source with `put_cycle` (and thus the DMA get/put parity + DMC
1228 // fire-phase) and can never drift. `cpu_cycle` was incremented above, so
1229 // `(cpu_cycle + parity_seed) & 1 == 1` reproduces the toggle-from-`false`
1230 // sequence exactly when `parity_seed == 0` (the floor config).
1231 {
1232 self.apu_phase = (self.cpu_cycle.wrapping_add(self.parity_seed) & 1) == 1;
1233 }
1234 // v2.0.0 beta.3 (A4 cycle-accurate reset): consume the scheduled
1235 // warm-reset `$4017` re-write N clocked cycles into the CPU's reset
1236 // sequence (see `Apu::reset` for the calibration). Runs after the
1237 // `apu_phase` derivation so the write's 3/4-cycle alignment delay
1238 // reads the current cycle's parity, exactly like a CPU-issued write.
1239 if self.reset_4017_delay > 0 {
1240 self.reset_4017_delay -= 1;
1241 if self.reset_4017_delay == 0 {
1242 let aligned = self.apu_phase;
1243 self.frame_counter.write(self.reset_4017_value, aligned);
1244 }
1245 }
1246 if self.apu_phase {
1247 self.pulse1.clock_timer();
1248 self.pulse2.clock_timer();
1249 self.noise.clock_timer();
1250 // F-2/M-1: the DMC byte-timer clock lives in `tick_dmc`
1251 // (end-of-cycle), not here at cycle START.
1252 }
1253
1254 // Frame counter (CPU clock). Latch the events so the bus can fan
1255 // them out to on-cart audio extensions (MMC5) after the tick.
1256 // Pass `apu_phase` AND `cpu_cycle` so the frame counter can
1257 // (a) compute APU-step timing as before and (b) mature any
1258 // pending lazy `$4015`-read IRQ-flag clear scheduled by a
1259 // previous read (Session-25, 2026-05-23 — see
1260 // `frame_counter::read_status` doc).
1261 let ev = self.frame_counter.tick(self.cpu_cycle, self.apu_phase);
1262 self.last_frame_events = ev;
1263 self.handle_frame_events(ev);
1264
1265 // v2.1.5 length halt/reload ordering: promote each channel's deferred
1266 // halt (`new_halt` -> `halt`) and pending length reload EVERY CPU cycle,
1267 // AFTER the half-frame clock in `handle_frame_events` and BEFORE the
1268 // mixer samples the channel outputs below. This realizes the 2A03's
1269 // "halt change takes effect after clocking length" and "reload ignored
1270 // during a non-zero length clock" rules (blargg `10.len_halt_timing` /
1271 // `11.len_reload_timing`; TetaNES `LengthCounter::reload` +
1272 // Mesen2 `_newHaltValue`). On the common cycle with no half-frame clock
1273 // the reload applies in-cycle (the count was untouched since the write),
1274 // so a plain length load / halt write remains byte-identical to an
1275 // immediate apply — only the write-lands-on-the-clock-cycle coincidence
1276 // the tests probe differs. The DMC has no length counter. See
1277 // `crates/rustynes-apu/src/length.rs`.
1278 self.pulse1.length.reload();
1279 self.pulse2.length.reload();
1280 self.triangle.length.reload();
1281 self.noise.length.reload();
1282
1283 // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`) reload-arm/reenable
1284 // bookkeeping and the `CannotRunDMCDMARightNow` exclusion decrement all
1285 // live at end-of-cycle (`dmc_tick_end`) under M-1, NOT here at cycle
1286 // START.
1287
1288 // Emit one mixed sample to the band-limited buffer. The external
1289 // (cartridge) audio is summed AFTER the internal non-linear mixer
1290 // since it's already a linear value.
1291 // Per-channel mute overlay. With the default `CHANNEL_MASK_ALL` every
1292 // `gate(..)` returns the raw output unchanged, so this is byte-identical
1293 // to the un-masked mix (the determinism contract — the oracle / test
1294 // ROMs never clear a bit). A cleared bit forces that channel's raw
1295 // output to 0 BEFORE the non-linear mixer, so it contributes nothing.
1296 let mask = self.channel_mask;
1297
1298 // v2.3.5 C1 — the DEFAULT-CONFIGURATION fast path.
1299 //
1300 // Every gate/scale below is inert at the shipped default: the
1301 // determinism contract says the oracle and the test ROMs never clear a
1302 // mask bit or change a gain, so `gate` returns its input unchanged and
1303 // `scale` returns `round(v * 1.0) == v`. The emulator was still paying,
1304 // every CPU cycle at 1.789 MHz, for a 6-wide `f32` array copy, five
1305 // integer mask tests, five float compares, and a sixth mask test for the
1306 // external sum -- to produce a result identical to the ungated mix.
1307 //
1308 // Same shape as the PPU fast dot path, which is the one core
1309 // optimization this project has adopted: hoist the
1310 // "is-this-the-default?" question out of the per-cycle body and take a
1311 // branch with none of the machinery. It is a strict specialization, not
1312 // an approximation -- `mix()` receives exactly the same five arguments
1313 // it would have received, so the output is byte-identical by
1314 // construction rather than by measurement. `apu_default_mix_matches_the_gated_path`
1315 // pins that across a 2,048-point sweep anyway.
1316 if mask == CHANNEL_MASK_ALL && self.gain_is_unity {
1317 self.last_external = external;
1318 let mixed = self.mixer.mix(
1319 self.pulse1.output(),
1320 self.pulse2.output(),
1321 self.triangle.output(),
1322 self.noise.output(),
1323 self.dmc.output(),
1324 ) + external;
1325 #[cfg(feature = "debug-hooks")]
1326 if self.audio_prov.is_some() {
1327 self.record_mix_armed(mixed, external);
1328 }
1329 self.blip.add_sample(mixed);
1330 // Nothing follows the general path's `add_sample` but comments --
1331 // the get/put flip moved to `dmc_tick_end` under M-2 -- so there is
1332 // no shared tail to run before returning. Verified by reading it,
1333 // not assumed: a missed tail here would desynchronise the two paths.
1334 return;
1335 }
1336
1337 let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
1338 // v1.4.0 Workstream C — per-channel gain (a UI mixing overlay). With the
1339 // default `CHANNEL_GAIN_UNITY` every `scale(..)` returns `round(v * 1.0)
1340 // == v` and `external * 1.0 == external`, so this is byte-identical to
1341 // the pre-gain mix (the determinism contract — the oracle / test ROMs
1342 // never change a gain). A gain != 1.0 scales that channel's contribution
1343 // before the non-linear mixer (gain 0.0 == a cleared mask bit). The
1344 // `gain` slice is checked-for-unity-and-skipped so the default path is
1345 // the exact integer-gate code as before.
1346 let gain = self.channel_gain;
1347 // `max` is the channel's native raw ceiling (pulse/tri/noise = 15, DMC =
1348 // 127); the scaled value is clamped to it so the non-linear mixer's
1349 // `pulse_table` (31) / `tnd_table` (203) index bounds always hold even at
1350 // gain 2.0. At gain 1.0 the value is returned unchanged (byte-identical).
1351 let scale = |bit: usize, v: u8, max: u8| {
1352 let g = gain[bit];
1353 if g == 1.0 {
1354 v
1355 } else {
1356 #[allow(
1357 clippy::cast_possible_truncation,
1358 clippy::cast_sign_loss,
1359 clippy::cast_precision_loss
1360 )]
1361 {
1362 roundf(f32::from(v) * g).clamp(0.0, f32::from(max)) as u8
1363 }
1364 }
1365 };
1366 // v2.1.6 — stash the RAW (pre-gain) external contribution for the
1367 // frontend expansion-channel scope/VU. Write-only from synthesis; never
1368 // read back into the mix, so it cannot alter deterministic output.
1369 self.last_external = external;
1370 let ext_gain = gain[5];
1371 let ext = if ext_gain == 1.0 {
1372 external
1373 } else {
1374 external * ext_gain
1375 };
1376 let mixed = self.mixer.mix(
1377 scale(0, gate(0, self.pulse1.output()), 15),
1378 scale(1, gate(1, self.pulse2.output()), 15),
1379 scale(2, gate(2, self.triangle.output()), 15),
1380 scale(3, gate(3, self.noise.output()), 15),
1381 scale(4, gate(4, self.dmc.output()), 127),
1382 ) + if mask & (1 << 5) != 0 { ext } else { 0.0 };
1383 #[cfg(feature = "debug-hooks")]
1384 if self.audio_prov.is_some() {
1385 // RAW `external`, not the gained `ext`, and not zero when the mask
1386 // bit clears it. The five channel fields are already the raw
1387 // pre-gate outputs, so recording a gain-scaled or mask-zeroed
1388 // expansion value would make ONE field follow the user's mixer
1389 // sliders while five describe the chip -- and would make this path
1390 // disagree with the fast path, which records the raw value. Review
1391 // caught the disagreement; this resolves it toward the documented
1392 // semantic rather than toward the local variable that happened to
1393 // be in scope.
1394 self.record_mix_armed(mixed, external);
1395 }
1396 self.blip.add_sample(mixed);
1397
1398 // v2.0 interleaved-DMA Phase A: toggle the global get/put flip-flop once
1399 // per CPU cycle, right after the APU tick (TriCNES `APU_PutCycle =
1400 // !APU_PutCycle` after `_EmulateAPU()`, `Emulator.cs:920`). Gated on
1401 // `dmc_driven_externally` so the default build never touches it
1402 // (byte-identical); under the R1 substrate this is the single
1403 // per-cycle get/put counter the interleaved DMA (Phase B) consumes.
1404 // RW-1 (`mc-r1-one-clock`): `put_cycle` is the COMPLEMENT of `apu_phase`,
1405 // derived from the same counter — not a second independent flip-flop.
1406 // In the floor config the two toggles already stayed perfectly
1407 // complementary (both flip once per `tick_with_external`); RW-1 makes
1408 // that structural so RW-2 has a SINGLE place to make the parity
1409 // OAM-DMA-aware. The bus's get/put decision (`get = !put_cycle`) and the
1410 // F-2 DMC clock (`!put_cycle`) then read this coherent value.
1411 // M-2 (`mc-r1-counter-collapse`): the get/put `put_cycle` flip moves to
1412 // END of the CPU cycle (`dmc_tick_end`), AFTER the bus access — the
1413 // references' "access -> APU tick -> get/put flip" order. So at the START
1414 // (here) `put_cycle` is LEFT at its prior-cycle value; the bus access this
1415 // cycle therefore reads `put_cycle = !apu_phase_{N-1} = apu_phase_N`,
1416 // one parity position later than the floor's `!apu_phase_N`. `apu_phase`
1417 // itself (the APU IRQ line / C1 phi2 sample source) still flips at start
1418 // (line ~927), so C1 is invariant.
1419 }
1420
1421 fn handle_frame_events(&mut self, ev: FrameEvents) {
1422 if ev.quarter {
1423 self.pulse1.clock_quarter_frame();
1424 self.pulse2.clock_quarter_frame();
1425 self.triangle.clock_quarter_frame();
1426 self.noise.clock_quarter_frame();
1427 }
1428 if ev.half {
1429 self.pulse1.clock_half_frame();
1430 self.pulse2.clock_half_frame();
1431 self.triangle.clock_half_frame();
1432 self.noise.clock_half_frame();
1433 }
1434 }
1435
1436 /// Visibility-delay promotion (called at END of cycle, after the CPU's bus
1437 /// access): a reload latched this cycle becomes visible to the NEXT cycle's
1438 /// DMA servicing (first-service on the put cycle => span 4), matching
1439 /// TriCNES `_EmulateAPU`-after-`_6502` invisible-arm ordering.
1440 pub fn promote_dmc_pending_next(&mut self) {
1441 if self.pending_dmc_dma_next {
1442 self.pending_dmc_dma_next = false;
1443 self.pending_dmc_dma = true;
1444 }
1445 }
1446
1447 /// v2.0 Program M (M-1 within-cycle order, `mc-r1-dmc-bytetimer-end`): clock
1448 /// the DMC byte-timer + arm the reload at END of cycle (after the CPU's bus
1449 /// access), the mirror of the cycle-START block in `tick_with_external` that
1450 /// `dmc_clock_at_start` now suppresses. Order matches `tick_with_external`:
1451 /// byte-timer clock (on this cycle's already-set `apu_phase`) -> reenable
1452 /// consume-edge clear -> reload-arm -> `cannot_run` decrement. The bus calls
1453 /// this from `cpu_clock_apu_dmc` (end-of-cycle), AFTER
1454 /// `promote_dmc_pending_next` so a reload latched here is invisible to its
1455 /// own cycle (promoted -> serviced the NEXT cycle = span-4, like the
1456 /// references). The LOAD delay-arm is NOT here — it stays at cycle-start.
1457 pub fn dmc_tick_end(&mut self) {
1458 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the 1-cycle implicit-abort
1459 // kill — TriCNES clears `APU_ImplicitAbortDMC4015` at the END of
1460 // `_6502` whenever the DMA is pending (Emulator.cs:9000-9003), i.e.
1461 // BEFORE `_EmulateAPU`'s boundary work. A flag set by the previous
1462 // cycle's consume edge therefore survives exactly one CPU access
1463 // (one serviced halt cycle if it was a read; none if a write — "it
1464 // won't run at all") and dies here.
1465 if self.pending_dmc_dma && self.dmc_implicit_abort {
1466 self.dmc_implicit_abort = false;
1467 }
1468 let d4015_bits_before = self.dmc.bits_remaining();
1469 let dmc_bits_before = d4015_bits_before;
1470 // The byte-timer-end flag composes only with the canonical apu_phase
1471 // clock (the `mc-r1-full-cpu` config); the cpu-rate / phase-minus1
1472 // diagnostic clock variants are not combined with it.
1473 // M-2 (`mc-r1-counter-collapse`): the get/put `put_cycle` flip moved to
1474 // end-of-cycle (one parity position later), so the GET decision
1475 // (`get = !put_cycle`) now reads the shifted parity. The DMC byte-timer
1476 // FIRE must follow the SAME shift or the GET de-syncs from the byte-timer
1477 // wrap (wedge). At entry `put_cycle == apu_phase` (the prior end-flip),
1478 // so clocking on `!self.put_cycle == !apu_phase` shifts the byte-timer by
1479 // one to stay locked to the shifted GET — ONE counter driving both.
1480 let timer_phase = !self.put_cycle;
1481 if timer_phase {
1482 self.dmc.clock_timer();
1483 }
1484 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the consume-edge transfer
1485 // (Emulator.cs:1163-1175) — at the shifter-consume edge (bits 1 -> 8
1486 // on this end-tick's byte-timer fire) a latched
1487 // `dmc_set_implicit_abort` becomes the live `dmc_implicit_abort`
1488 // service-gate override AND arms the DMA directly (TriCNES
1489 // `if (BytesRemaining > 0 || SetImplicit) { if (!DoDMCDMA &&
1490 // CannotRun != 2) { DoDMCDMA = true; Halt = true; } ... }` — the arm
1491 // fires regardless of the buffer state). The armed DMA runs for
1492 // exactly one read cycle under the override (the kill above), then
1493 // waits for the delayed status — the emergent 1-cycle implicit abort.
1494 if timer_phase
1495 && self.dmc_set_implicit_abort
1496 && self.dmc.bits_remaining() == 8
1497 && d4015_bits_before <= 1
1498 {
1499 self.dmc_implicit_abort = true;
1500 self.dmc_set_implicit_abort = false;
1501 if !self.pending_dmc_dma && self.cannot_run_dmc_dma != 2 {
1502 self.pending_dmc_dma = true;
1503 self.dmc_dma_is_load = false;
1504 self.dmc_dma_short = false;
1505 self.dmc_dma_addr = self.dmc.dma_addr();
1506 self.dmc_need_halt = true;
1507 self.dmc_need_dummy_read = true;
1508 }
1509 }
1510 // W3-Stage-2 (`mc-r1-dma-unified-collapse`): the TriCNES `DMCDMADelay`
1511 // put-branch — the `$4015`-enable LOAD delay counts down ONLY on the
1512 // put phase of this end-of-cycle tick (Emulator.cs:1217 sits in the
1513 // `else` of the get branch), arming the halt at the end of a PUT cycle
1514 // so the load's first halted cycle is always a GET (entry-on-get =
1515 // span 3) regardless of the write cycle's parity. The put phase here
1516 // is `!timer_phase` (the complement of the shifted byte-timer phase).
1517 if !timer_phase {
1518 self.dmc_step_delay_arm_put_end();
1519 }
1520 if self.dmc_reenable_period_block && self.dmc.bits_remaining() == 8 && dmc_bits_before <= 1
1521 {
1522 self.dmc_reenable_period_block = false;
1523 }
1524 // W3-Stage-4 (`mc-r1-dmc-delayed-4015` grid correction): the TriCNES
1525 // reload arm is consume-edge-quantized. A consume edge that lands ON
1526 // the GET-delivery cycle itself (the X=8/9 Implicit `$540` restart
1527 // race: the silent-restart load GET collides with the free-running
1528 // byte-timer boundary) is arm-BLOCKED by `CannotRunDMCDMARightNow ==
1529 // 2` (Emulator.cs:1165; the :1186 decrement runs later that same
1530 // end-tick, so `== 2` is only ever observable at the same-cycle
1531 // edge) — and TriCNES holds NO level request: the chain simply waits
1532 // for the NEXT consume edge (one full byte period). Our `needs_dma()`
1533 // is level-triggered and would re-arm 4 cycles later (cooldown
1534 // expiry) — one grid boundary early, the `$540[8,9]` cliff. Latch the
1535 // suppression at the blocked same-cycle edge; release at the next
1536 // consume edge right here (BEFORE the reload-arm step) so the
1537 // deferred arm fires exactly on-grid, like TriCNES's
1538 // `BytesRemaining > 0` edge arm.
1539 if timer_phase && self.dmc.bits_remaining() == 8 && d4015_bits_before <= 1 {
1540 if self.dmc_edge_arm_suppress {
1541 self.dmc_edge_arm_suppress = false;
1542 } else if self.cannot_run_dmc_dma == 2 && self.dmc.needs_dma() && !self.pending_dmc_dma
1543 {
1544 self.dmc_edge_arm_suppress = true;
1545 }
1546 }
1547 self.dmc_step_reload_arm();
1548 // M-2: the `cannot_run` decrement is TriCNES's get-cycle decrement; under
1549 // the collapse the get cycle is the shifted `timer_phase`, not raw
1550 // apu_phase.
1551 if timer_phase && self.cannot_run_dmc_dma > 0 {
1552 self.cannot_run_dmc_dma = self.cannot_run_dmc_dma.saturating_sub(2);
1553 }
1554 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES
1555 // `APU_DelayedDMC4015` countdown (Emulator.cs:1214-1224) — decremented
1556 // EVERY CPU cycle after the get/put branch work (the byte-timer /
1557 // reload-arm / load-delay above). On expiry the latched `$4015` DMC
1558 // status APPLIES: `APU_Status_DMC = APU_Status_DelayedDMC`, and a
1559 // disable zeroes `bytes_remaining` HERE rather than at the write. The
1560 // bus-side service gate reads `dmc_status_applied` per cycle, so an
1561 // in-flight DMA whose status drops stops being serviced — the
1562 // emergent explicit abort.
1563 if self.dmc_delayed_4015 > 0 {
1564 self.dmc_delayed_4015 -= 1;
1565 if self.dmc_delayed_4015 == 0 {
1566 self.dmc_status_applied = self.dmc_delayed_status;
1567 if !self.dmc_status_applied {
1568 self.dmc.bytes_remaining = 0;
1569 }
1570 }
1571 }
1572 // M-2 (`mc-r1-counter-collapse`): flip the get/put parity HERE at
1573 // end-of-cycle (after the CPU's bus access + the byte-timer/reload-arm
1574 // tick above), matching the references' "access -> APU tick -> get/put
1575 // flip" order. `put_cycle = !apu_phase` of the cycle that just ran; the
1576 // NEXT cycle's bus access reads this value. (Under bytetimer-end alone
1577 // this flip stays at cycle-start in `tick_with_external`.)
1578 {
1579 self.put_cycle = !self.apu_phase;
1580 }
1581 }
1582
1583 /// W3-Stage-2 (`mc-r1-dma-unified-collapse`): the TriCNES `DMCDMADelay`
1584 /// put-branch body — same arm as [`Self::dmc_step_delay_arm`] but ticked
1585 /// only on the put phase of `dmc_tick_end` (value units = put end-ticks,
1586 /// set to 2 at the `$4015` enable like TriCNES `DMCDMADelay = 2`).
1587 fn dmc_step_delay_arm_put_end(&mut self) {
1588 if self.dmc_dma_delay > 0 {
1589 self.dmc_dma_delay -= 1;
1590 if self.dmc_dma_delay == 0 && !self.pending_dmc_dma {
1591 self.pending_dmc_dma = true;
1592 self.dmc_dma_short = self.dmc_dma_is_load;
1593 self.dmc_dma_addr = self.dmc.dma_addr();
1594 self.dmc_need_halt = true;
1595 self.dmc_need_dummy_read = true;
1596 }
1597 }
1598 }
1599
1600 /// DMC delay-arm step: countdown the load-DMA delay and arm `pending_dmc_dma`
1601 /// when it expires. Extracted from `tick_with_external` so `tick_dmc` (F-2)
1602 /// can run it at end-of-cycle.
1603 fn dmc_step_delay_arm(&mut self) {
1604 if self.dmc_dma_delay > 0 {
1605 self.dmc_dma_delay -= 1;
1606 if self.dmc_dma_delay == 0 && !self.pending_dmc_dma {
1607 self.pending_dmc_dma = true;
1608 self.dmc_dma_short = self.dmc_dma_is_load;
1609 self.dmc_dma_addr = self.dmc.dma_addr();
1610 self.dmc_need_halt = true;
1611 self.dmc_need_dummy_read = true;
1612 }
1613 }
1614 }
1615
1616 /// DMC reload-arm step: arm a reload DMA when the sample buffer empties
1617 /// (subject to cooldown / suppress / defer). Extracted for `tick_dmc` (F-2).
1618 #[allow(clippy::too_many_lines)]
1619 fn dmc_step_reload_arm(&mut self) {
1620 // final lever #1 (`mc-r1-dmc-halt-subpos`): master-clock DMA-halt
1621 // sub-position. The reload byte-timer wraps and arms on the apu_phase
1622 // get cycle (so the CPU recognizes the halt at the NEXT read1 = one CPU
1623 // cycle too late -> the GET lands adjacent to the `LDA $4000` data read,
1624 // which sees the GET's $00 -> Y=3). TriCNES arms one CPU cycle EARLIER so
1625 // the GET preempts the operand-high fetch (re-driving $40 -> Y=4). On the
1626 // `!apu_phase` cycle IMMEDIATELY preceding the wrap, the byte-timer sits
1627 // at `timer==0 && bits_remaining==1` (the final output bit is one
1628 // apu-clock from emptying the byte). Pre-arm `pending_dmc_dma` HERE, one
1629 // CPU cycle early. Scoped EXACTLY to the X=10/11 boundary by the
1630 // `cannot_run_dmc_dma == 2` exclusion (post-LOAD-GET window) — fires 6x,
1631 // nowhere else — so steady-state GETs + SH* are untouched (context-local,
1632 // distinct from a global byte-timer phase shift that shatters SH*).
1633 // The pre-wrap `!apu_phase` cycle that uniquely marks the `$540` X=10/11
1634 // boundary: the reload byte-timer is at `timer==0 && bits_remaining==1`
1635 // (one apu-clock from emptying the byte), the LOAD has just FILLED the
1636 // buffer (`buffer_full` -> needs_dma still FALSE), and we are inside the
1637 // post-LOAD-GET `cannot_run == 2` exclusion. This is distinct from the
1638 // `$500`/`$520` X=10/11 blocks (Key1/Key2, already correct) whose buffer
1639 // is already empty at this point (no `buffer_full` pre-wrap cycle), so
1640 // pre-arming here leaves them untouched. Arm `pending_dmc_dma` one CPU
1641 // cycle early so the wrap-cycle's `read1` recognizes the halt (the GET
1642 // preempts the operand-high fetch -> $40 re-driven -> Y 3->4) instead of
1643 // the next read1 (GET adjacent to the data read -> $00 seen -> Y=3).
1644 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the halt-subpos boundary
1645 // pre-arm is a floor-unit expression of the same missing `$4015`
1646 // application delay (the Stage-2 residual map); under the
1647 // delayed-status port it is superseded by the emergent consume-edge
1648 // arm — both active double-fire on the X=10/11 entries.
1649 // Gate also on the visibility-delay latch so a reload cannot double-arm
1650 // while one is latched-but-not-yet-promoted (would cascade/wedge).
1651 let already = self.pending_dmc_dma || self.pending_dmc_dma_next;
1652 // v2.0 Phase 2 (`mc-r1-dmc-reenable-phase`): TriCNES gates the reload
1653 // arm on `CannotRunDMCDMARightNow != 2` (`Emulator.cs:1165`) — a reload
1654 // cannot arm on the get cycle immediately following a DMC GET. That
1655 // exclusion is hit ONLY at the Implicit-DMA-Abort X=10/11 `$4015`
1656 // re-enable boundary (the LOAD GET lands so the next byte-timer wrap
1657 // coincides with the window) — confirmed by the probe firing on exactly
1658 // those two entries. A full-period reload deferral there OVERSHOOTS
1659 // ($540[10,11] -> 00, Y=0) because RustyNES's start-clock + Option-buffer
1660 // structure shifts the whole chain a byte; TriCNES instead realigns the
1661 // byte-timer phase by ~1 cycle. So at the boundary we apply a ONE-SHOT
1662 // swept byte-timer phase shift (`REENABLE_BUMP`, env-tunable) that
1663 // realigns the looping-reload chain like TriCNES's re-enable, while the
1664 // bare `cannot_run == 2` gate still defers this cycle's arm.
1665 let cannot_run_now = self.cannot_run_dmc_dma == 2;
1666 // One-shot byte-timer realignment at the exclusion boundary. `period_block`
1667 // is the one-shot guard (set here, cleared at the next consume edge in
1668 // `tick_with_external`) so the bump is applied exactly once per boundary.
1669 if cannot_run_now
1670 && self.dmc.needs_dma()
1671 && !already
1672 && self.dmc_dma_delay == 0
1673 && self.dmc_reload_suppress_outputs == 0
1674 && self.dmc_dma_cooldown == 0
1675 && !self.defer_dmc_reload_once
1676 && !self.dmc_reenable_period_block
1677 {
1678 let bump = crate::dmc::REENABLE_BUMP.load(core::sync::atomic::Ordering::Relaxed);
1679 if bump != 0 {
1680 self.dmc.bump_timer_phase(bump);
1681 }
1682 self.dmc_reenable_period_block = true;
1683 }
1684 // W3-Stage-4: the consume-edge-quantization suppression (see
1685 // `dmc_tick_end`) — while latched, the level-held `needs_dma()` must
1686 // NOT arm; the deferred arm fires at the next consume edge.
1687 let edge_suppressed = self.dmc_edge_arm_suppress;
1688 if self.dmc.needs_dma()
1689 && !already
1690 && self.dmc_dma_delay == 0
1691 && !cannot_run_now
1692 && !edge_suppressed
1693 {
1694 if self.dmc_reload_suppress_outputs > 0
1695 || self.dmc_dma_cooldown > 0
1696 || self.defer_dmc_reload_once
1697 {
1698 self.defer_dmc_reload_once = false;
1699 } else {
1700 // Visibility-delay: a reload latches into `_next` (promoted next
1701 // cycle) so first-service lands on the put cycle (span 4). Loads
1702 // and the default keep direct `pending_dmc_dma` (first-service get).
1703 {
1704 self.pending_dmc_dma_next = true;
1705 }
1706 self.dmc_dma_is_load = false;
1707 self.dmc_dma_short = false;
1708 self.dmc_dma_addr = self.dmc.dma_addr();
1709 self.dmc_need_halt = true;
1710 self.dmc_need_dummy_read = true;
1711 }
1712 } else {
1713 self.defer_dmc_reload_once = false;
1714 }
1715 }
1716
1717 /// v2.0 F-2: advance ONLY the DMC byte-timer + DMA arm by one CPU cycle.
1718 /// The R1 bus calls this at END of cycle (after the access) when
1719 /// [`Self::set_dmc_driven_externally`] is set, so the DMC fire-phase matches
1720 /// the pre-v2.0.0 `tick_one_cpu_cycle` (the cycle DMASync's `$4000` conflict
1721 /// expects) while the rest of the APU — incl. the IRQ line — stays on the
1722 /// cycle-start `tick_with_external`. Order mirrors `tick_with_external`:
1723 /// delay-arm → APU-rate timer clock (via the `dmc_ext_phase` flip-flop) →
1724 /// reload-arm.
1725 pub fn tick_dmc(&mut self) {
1726 self.dmc_step_delay_arm();
1727 // Divergence A: clock the DMC byte-timer off the SHARED `put_cycle`
1728 // counter (the same flip-flop the interleaved DMA's get/put decision
1729 // uses) instead of a separate `dmc_ext_phase`, so the DMC fire-phase and
1730 // the get/put parity share ONE seed and can NEVER drift (TriCNES seeds
1731 // `APU_PutCycle` + the DMC timer together). The DMC clocks at the APU
1732 // rate (every other CPU cycle). Polarity `!put_cycle`: main clocks the
1733 // DMC on `apu_phase`-true (cycles 1,3,5 — odd); `put_cycle` is seeded so
1734 // its true-phase falls on EVEN cycles, so `!put_cycle` recovers main's
1735 // ODD-cycle DMC fire-phase (the DMASync-positioning alignment).
1736 if !self.put_cycle {
1737 self.dmc.clock_timer();
1738 }
1739 self.dmc_step_reload_arm();
1740 }
1741
1742 /// v2.0 interleaved-DMA Phase B: advance ONLY the DMC byte-timer clock (no
1743 /// delay/reload ARM), for a cycle of an interleaved DMC DMA span. The timer
1744 /// advances (so the variable-3/4-span feeds back into the next fire-cycle —
1745 /// divergence-A self-consistency) WITHOUT re-arming a new DMA mid-span (no
1746 /// cascade). Toggles the same `dmc_ext_phase` flip-flop as [`Self::tick_dmc`]
1747 /// so the every-other-cycle cadence stays consistent across normal + DMA
1748 /// cycles. (In the burst model this re-wedged; in the per-cycle interleaved
1749 /// model each DMA cycle is discrete and arm-gated, so it should hold.)
1750 pub fn tick_dmc_timer_only(&mut self) {
1751 // Divergence A: clock off the shared `put_cycle` counter (see `tick_dmc`).
1752 if !self.put_cycle {
1753 self.dmc.clock_timer();
1754 }
1755 }
1756
1757 /// v2.0 F-2: route the DMC byte-timer + arm to [`Self::tick_dmc`] instead of
1758 /// `tick_with_external`. Default `false` = byte-identical.
1759 pub const fn set_dmc_driven_externally(&mut self, on: bool) {
1760 self.dmc_driven_externally = on;
1761 }
1762
1763 /// v2.0 interleaved-DMA Phase A: the global get/put flip-flop (TriCNES
1764 /// `APU_PutCycle`). `true` = put cycle, `false` = get cycle.
1765 #[must_use]
1766 pub const fn put_cycle(&self) -> bool {
1767 self.put_cycle
1768 }
1769
1770 /// v2.0 interleaved-DMA Phase A: seed the global get/put flip-flop from an
1771 /// `APUAlignment` value (TriCNES `Emulator.cs:685/776`), the single seed the
1772 /// interleaved DMA (Phase B) will share with the DMC fire-phase (divergence
1773 /// A). The low bit selects the parity (TriCNES case 0/2 -> put, 1/3 -> get).
1774 ///
1775 /// Phase A seeds ONLY `put_cycle` and deliberately leaves `dmc_ext_phase`
1776 /// untouched, so the un-wedged feature-on behavior is preserved (the f2e
1777 /// experiment proved flipping `dmc_ext_phase` alone regresses). The exact
1778 /// `put_cycle` <-> `dmc_ext_phase` pairing is determined empirically in
1779 /// Phase B, when the bus first consumes `put_cycle` for the get/put decision.
1780 pub const fn seed_apu_alignment(&mut self, alignment: u8) {
1781 self.put_cycle = (alignment & 1) == 0;
1782 // RW-1 (`mc-r1-one-clock`): record the boot parity as the ONE seed both
1783 // `apu_phase` and `put_cycle` derive from. `alignment == 0` -> seed 0,
1784 // which reproduces the floor config (boot `apu_phase = false` +
1785 // put-on-even) exactly. Set at power-on, reset, and restore; constant
1786 // otherwise. The legacy `put_cycle` assignment above is harmless when the
1787 // flag is on (the next derivation overwrites it from `cpu_cycle`).
1788 {
1789 self.parity_seed = (alignment & 1) as u64;
1790 }
1791 }
1792
1793 /// CPU register write (`$4000-$4017` excluding `$4014`).
1794 pub fn write_register(&mut self, addr: u16, value: u8) {
1795 // v2.3.7 "Overtone" — attribute the write BEFORE dispatching it, so the
1796 // recorded value is what the CPU put on the bus rather than whatever a
1797 // channel decided to keep. One `Option` test when disarmed.
1798 #[cfg(feature = "debug-hooks")]
1799 if let Some(p) = self.audio_prov.as_mut() {
1800 p.reg_attrib
1801 .record(addr, p.attrib_pc, p.attrib_cycle, value);
1802 }
1803 match addr {
1804 0x4000 => self.pulse1.write_ctrl(value),
1805 0x4001 => self.pulse1.write_sweep(value),
1806 0x4002 => self.pulse1.write_timer_lo(value),
1807 0x4003 => self.pulse1.write_timer_hi(value),
1808 0x4004 => self.pulse2.write_ctrl(value),
1809 0x4005 => self.pulse2.write_sweep(value),
1810 0x4006 => self.pulse2.write_timer_lo(value),
1811 0x4007 => self.pulse2.write_timer_hi(value),
1812 0x4008 => self.triangle.write_linear(value),
1813 0x4009 => {} // unused
1814 0x400A => self.triangle.write_timer_lo(value),
1815 0x400B => self.triangle.write_timer_hi(value),
1816 0x400C => self.noise.write_ctrl(value),
1817 0x400D => {} // unused
1818 0x400E => self.noise.write_period(value),
1819 0x400F => self.noise.write_length(value),
1820 0x4010 => self.dmc.write_ctrl(value),
1821 0x4011 => self.dmc.write_dac(value),
1822 0x4012 => self.dmc.write_sample_addr(value),
1823 0x4013 => self.dmc.write_sample_length(value),
1824 0x4015 => self.write_status(value),
1825 0x4017 => {
1826 // $4017 also clears DMC IRQ? No — only $4015 clears DMC.
1827 // But writing $4017 with bit 6 set clears frame IRQ.
1828 // The frame counter handles the inhibit-clears-flag effect.
1829 // Apu-aligned: cycle is even when apu_phase will toggle to
1830 // true on the NEXT tick. Our `apu_phase` reflects the
1831 // *current* state after the tick. Per nesdev: "If the write
1832 // occurs during an APU clock (CPU cycle 1, 3, 5...) the
1833 // effects occur 3 CPU cycles after the write; if during a
1834 // non-APU clock, the effects occur 4 CPU cycles after."
1835 let aligned = self.apu_phase;
1836 self.frame_counter.write(value, aligned);
1837 }
1838 _ => {}
1839 }
1840 }
1841
1842 // W3-Stage-3: the delayed-4015 cfg arms (the latch dispatch + the
1843 // superseded-compensation gating) push the counted length just past the
1844 // clippy limit; the body is mostly per-feature cfg blocks.
1845 #[allow(clippy::too_many_lines)]
1846 fn write_status(&mut self, value: u8) {
1847 self.pulse1.length.set_enabled((value & 0x01) != 0);
1848 self.pulse2.length.set_enabled((value & 0x02) != 0);
1849 self.triangle.length.set_enabled((value & 0x04) != 0);
1850 self.noise.length.set_enabled((value & 0x08) != 0);
1851 let enable_dmc = (value & 0x10) != 0;
1852 let was_active = self.dmc.active();
1853 let implicit_stop_edge = enable_dmc
1854 && !was_active
1855 && !self.dmc.loop_flag
1856 && self.dmc.sample_length == 1
1857 && self.dmc.rate_index == 0x0E
1858 && self.dmc.bits_remaining == 1
1859 && self.dmc.sample_buffer.is_none();
1860 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES delayed-status
1861 // latch replaces the immediate `set_enabled` application — see
1862 // `latch_delayed_dmc_4015`.
1863 self.latch_delayed_dmc_4015(enable_dmc);
1864 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES gates the enable-side
1865 // LOAD-delay arm on `APU_Silent` (Emulator.cs:9519-9522 — "the sample
1866 // will only begin playing if the DMC is currently silent"; otherwise
1867 // the restart is picked up at the NEXT shifter-consume edge). Our
1868 // floor condition (`needs_dma()` alone) arms the load immediately
1869 // even while the output unit is still draining the prior looping
1870 // byte — in the Implicit Loop3/`$540` re-enable race that fires a
1871 // span-3 load-style DMA 2-4 sweep positions before the hardware's
1872 // boundary-quantized reload (the `03,03` lead-in + plateau-2-early).
1873 let load_arm = enable_dmc && !was_active && self.dmc.needs_dma() && self.dmc.silence();
1874 if load_arm {
1875 self.pending_dmc_dma = false;
1876 self.dmc_dma_is_load = true;
1877 self.dmc_dma_short = true;
1878 self.dmc_dma_addr = self.dmc.dma_addr();
1879 // Load DMAs attempt to halt on the get cycle during the second
1880 // APU cycle after `$4015` enables DMC. In this emulator
1881 // `apu_phase == true` is the get half of the current APU cycle.
1882 //
1883 // v2.0 R-1 core C-1: under R1 (`dmc_driven_externally`) the DMC
1884 // clocks on `!put_cycle` (F-2), so `apu_phase` is the WRONG phase
1885 // basis for the load-arm delay just as it is for the abort `cuo`
1886 // (P-2: the R1 load DMA fires 1-2 cyc early → the 1-byte abort
1887 // sample's narrow active window lands off the swept `$4015` disable
1888 // → `disable_was_active` 12 vs 76). Use the DMC's actual phase.
1889 // W3-Stage-2 (`mc-r1-dma-unified-collapse`): TriCNES
1890 // `DMCDMADelay = 2` — two put end-ticks of `dmc_tick_end` (the
1891 // write cycle's own end-tick counts when it lands on a put, the
1892 // "really like 2 : 3" parity absorption), so the halt arms at the
1893 // end of a PUT and the load enters on a GET regardless of the
1894 // write parity. Replaces the every-cycle `apu_phase ? 4 : 3`
1895 // countdown whose value bakes in the floor GET parity.
1896 {
1897 self.dmc_dma_delay = 2;
1898 }
1899 // W3-Stage-2: the implicit-stop-edge -1 is a CPU-cycle-unit
1900 // calibration of the every-cycle countdown; under the TriCNES
1901 // put-end-tick countdown (put units) it cannot be expressed and
1902 // TriCNES has no such adjustment — skip it (TriCNES-exact).
1903 let _ = implicit_stop_edge;
1904 } else if !enable_dmc {
1905 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the disable-side floor
1906 // compensations below (the scheduled explicit abort, the
1907 // pending-reload keep-alive reshaping, the load-delay zeroing and
1908 // the suppress reset) are SUPERSEDED by the delayed-status
1909 // application — TriCNES's `$4015` disable write does nothing else
1910 // DMC-wise; the abort is EMERGENT from the applied status gating
1911 // the per-cycle DMA service (`_6502` line 4218).
1912 } else if enable_dmc {
1913 self.dmc_reload_suppress_outputs = 0;
1914 }
1915 }
1916
1917 /// W3-Stage-3 (`mc-r1-dmc-delayed-4015`): the TriCNES `$4015` write
1918 /// handler's DMC-status section (Emulator.cs:9504-9548). IMMEDIATE at the
1919 /// write: the enable-side `StartDMCSample` (`set_enabled(true)` restarts
1920 /// only when `bytes_remaining == 0` — exactly `StartDMCSample`, line
1921 /// 9517) and the DMC IRQ-flag clear (line 9529). DEFERRED: the status-bit
1922 /// application + the disable-side `bytes_remaining` zeroing, latched into
1923 /// `dmc_delayed_status` and applied `put ? 3 : 4` end-ticks later (line
1924 /// 9512; the write cycle's own end-tick counts — "really like 2 : 3"). A
1925 /// second `$4015` write during the pending window resets the countdown
1926 /// with the new target (last write wins, as TriCNES).
1927 fn latch_delayed_dmc_4015(&mut self, enable_dmc: bool) {
1928 if enable_dmc {
1929 self.dmc.set_enabled(true);
1930 } else {
1931 self.dmc.irq_flag = false;
1932 }
1933 self.dmc_delayed_status = enable_dmc;
1934 self.dmc_delayed_4015 = if self.put_cycle { 3 } else { 4 };
1935 // The explicit don't-abort edge (Emulator.cs:9533-9537): the disable
1936 // coincides with "the APU cycle that fires a DMC DMA". TriCNES
1937 // `(timer == 2 && get) || (timer == rate && put)` in CPU-rate units
1938 // maps to our APU-rate byte-timer as `(timer == 0 && get)` (this
1939 // cycle's end-tick wraps) or `(timer == timer_period && put)` (the
1940 // wrap happened on the previous get half). Extend the delay to
1941 // `put ? 5 : 6` so the just-armed reload DMA runs to completion
1942 // before the disable zeroes `bytes_remaining` (EXPLICIT sweep
1943 // idx[7] = 04).
1944 if !enable_dmc {
1945 let firing_apu_cycle = if self.put_cycle {
1946 self.dmc.timer == self.dmc.timer_period
1947 } else {
1948 self.dmc.timer == 0
1949 };
1950 if firing_apu_cycle {
1951 self.dmc_delayed_4015 = if self.put_cycle { 5 } else { 6 };
1952 }
1953 }
1954 // The implicit-abort edge (Emulator.cs:9540-9545): an ENABLE that
1955 // lands one byte-timer fire BEFORE the shifter-consume edge —
1956 // TriCNES `(timer == 10 && get) || (timer == 8 && put)` = our
1957 // APU-rate `(4, get)/(3, put)` (uniform `(t - 2) / 2` mapping).
1958 // "Regardless of the buffer being empty, there will be a 1-cycle
1959 // DMA that gets aborted" — latched here, consumed at the consume
1960 // edge in `dmc_tick_end`.
1961 if enable_dmc {
1962 let pre_fire_window = if self.put_cycle {
1963 self.dmc.timer == 3
1964 } else {
1965 self.dmc.timer == 4
1966 };
1967 if pre_fire_window {
1968 self.dmc_set_implicit_abort = true;
1969 }
1970 }
1971 }
1972
1973 /// CPU register read (only `$4015` is meaningful). Reading clears the
1974 /// frame IRQ flag.
1975 pub fn read_status(&mut self) -> u8 {
1976 let mut v = 0u8;
1977 if self.pulse1.length.active() {
1978 v |= 0x01;
1979 }
1980 if self.pulse2.length.active() {
1981 v |= 0x02;
1982 }
1983 if self.triangle.length.active() {
1984 v |= 0x04;
1985 }
1986 if self.noise.length.active() {
1987 v |= 0x08;
1988 }
1989 // W3-Stage-3 (`mc-r1-dmc-delayed-4015`): TriCNES `Observe` `$4015`
1990 // (Emulator.cs:9127/9260 + the 9268 footnote) — bit 4 is
1991 // `bytes_remaining != 0 && APU_Status_DelayedDMC`: a read right after
1992 // a disable write must see bit 4 CLEAR even though `bytes_remaining`
1993 // is not zeroed until the delayed application ("LDA #0, STA $4015,
1994 // LDA $4015 ... needs to immediately have bit 4 cleared").
1995 if self.dmc.active() && self.dmc_delayed_status {
1996 v |= 0x10;
1997 }
1998 if self.frame_counter.irq_flag {
1999 v |= 0x40;
2000 }
2001 if self.dmc.irq_flag {
2002 v |= 0x80;
2003 }
2004 // Reading clears frame IRQ flag (NOT DMC IRQ). The clear is
2005 // SCHEDULED for a future CPU cycle (1 cycle delta on a "get"
2006 // cycle, 2 cycles on a "put") and matured by a subsequent
2007 // observation -- the canonical Mesen2 `GetIrqFlag` lazy
2008 // algorithm (Session-25, 2026-05-23). The pre-Session-25
2009 // immediate-on-get / defer-by-one-tick-on-put scheme failed
2010 // `AccuracyCoin :: APU Tests :: Frame Counter IRQ` Test 7.
2011 // See `docs/audit/session-25-sprint2-iter3-frame-counter-irq-2026-05-23.md`.
2012 let _ = self
2013 .frame_counter
2014 .read_status(self.cpu_cycle, self.apu_phase);
2015 v
2016 }
2017
2018 /// Clear the frame IRQ flag immediately for DMA no-op reads of `$4015`.
2019 ///
2020 /// The normal CPU-visible `$4015` read path keeps the put-cycle deferred
2021 /// clear needed by frame-counter timing tests. DMC DMA no-op repeats use
2022 /// this after sampling the status value so the halted-read side effect is
2023 /// visible before the CPU resumes the original `$4015` read.
2024 ///
2025 /// Session-26 iter 5: also deassert the CPU IRQ line driver
2026 /// (`irq_line_active`) since the DMA no-op read mirrors a CPU
2027 /// `$4015` read on the silicon — the IRQ source is removed from
2028 /// the CPU's `_irqSource` list synchronously.
2029 pub fn clear_frame_irq_immediate_for_dma(&mut self) {
2030 self.frame_counter.irq_flag = false;
2031 self.frame_counter.irq_line_active = false;
2032 }
2033}
2034
2035#[cfg(test)]
2036mod tests {
2037
2038 /// v2.3.5 C1 — the default-configuration fast path must be byte-identical
2039 /// to the gated path it skips.
2040 ///
2041 /// The specialization is only sound because `gate` is the identity when its
2042 /// mask bit is set and `scale` is the identity at gain 1.0. If either ever
2043 /// stops being the identity at the default, this silently changes shipped
2044 /// audio -- so assert the equivalence directly over a 2,048-point sweep of
2045 /// the output range rather than trusting the reasoning.
2046 #[test]
2047 fn apu_default_mix_matches_the_gated_path() {
2048 let apu = Apu::new(Region::Ntsc, 48_000);
2049 assert_eq!(apu.channel_mask, CHANNEL_MASK_ALL, "premise: default mask");
2050 assert_eq!(
2051 apu.channel_gain, CHANNEL_GAIN_UNITY,
2052 "premise: default gain"
2053 );
2054
2055 let mask = CHANNEL_MASK_ALL;
2056 let gain = CHANNEL_GAIN_UNITY;
2057 let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2058 let scale = |bit: usize, v: u8, max: u8| {
2059 let g = gain[bit];
2060 if g == 1.0 {
2061 v
2062 } else {
2063 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
2064 {
2065 roundf(f32::from(v) * g).clamp(0.0, f32::from(max)) as u8
2066 }
2067 }
2068 };
2069
2070 // 2,048 SELECTED combinations, not the full cross-product. The DMC axis
2071 // is swept exhaustively (0..=127) against 16 rotating phases of the
2072 // other four channels, which walks the `tnd_table` index range end to
2073 // end and visits every raw level each channel can take. The exhaustive
2074 // product would be 16^4 * 128 = 8.4M; this is a sweep, and the wording
2075 // says "sweep" rather than claiming enumeration.
2076 for dmc in 0u8..=127 {
2077 for lvl in 0u8..=15 {
2078 let (p1, p2, tri, noise) = (lvl, 15 - lvl, (lvl + 7) % 16, (lvl + 3) % 16);
2079 let gated = apu.mixer.mix(
2080 scale(0, gate(0, p1), 15),
2081 scale(1, gate(1, p2), 15),
2082 scale(2, gate(2, tri), 15),
2083 scale(3, gate(3, noise), 15),
2084 scale(4, gate(4, dmc), 127),
2085 ) + if mask & (1 << 5) != 0 { 0.25f32 } else { 0.0 };
2086 let fast = apu.mixer.mix(p1, p2, tri, noise, dmc) + 0.25f32;
2087 assert_eq!(
2088 gated.to_bits(),
2089 fast.to_bits(),
2090 "p1={p1} p2={p2} tri={tri} noise={noise} dmc={dmc}: \
2091 the fast path must be BIT-identical, not merely close"
2092 );
2093 }
2094 }
2095 }
2096
2097 /// The fast path must NOT be taken once the configuration stops being the
2098 /// default -- otherwise the mute/gain overlay would silently stop working.
2099 #[test]
2100 fn a_non_default_mask_or_gain_still_takes_the_gated_path() {
2101 let mut muted = Apu::new(Region::Ntsc, 48_000);
2102 muted.set_channel_mask(CHANNEL_MASK_ALL & !0x01); // mute pulse 1
2103 assert_ne!(muted.channel_mask, CHANNEL_MASK_ALL);
2104
2105 let mut quiet = Apu::new(Region::Ntsc, 48_000);
2106 quiet.set_channel_gain([0.5, 1.0, 1.0, 1.0, 1.0, 1.0]);
2107 assert_ne!(quiet.channel_gain, CHANNEL_GAIN_UNITY);
2108
2109 // Drive both far enough to produce output, and confirm a muted channel
2110 // actually changes the mix relative to the default.
2111 let mut plain = Apu::new(Region::Ntsc, 48_000);
2112 for a in [&mut plain, &mut muted, &mut quiet] {
2113 a.write_register(0x4015, 0x1F);
2114 a.write_register(0x4000, 0xBF);
2115 a.write_register(0x4002, 0xAA);
2116 a.write_register(0x4003, 0x08);
2117 for _ in 0..2_000 {
2118 a.tick();
2119 }
2120 }
2121 // The original assertion here was broken, and both review bots caught it:
2122 // it compared `plain.pulse1.output() == 0` against
2123 // `muted.channel_mask & 0x01 != 0`, which is `false` for a muted mask --
2124 // so it only passed when pulse 1 happened to be at output 0 on the
2125 // sampled tick. Phase-dependent, disconnected from the overlay it claimed
2126 // to test, and it never touched `quiet` at all. It could not fail on the
2127 // bug it existed to catch.
2128 //
2129 // Compare the EMITTED AUDIO instead, which is what the overlay is
2130 // supposed to change. Deliberately not `pulse1.output() != 0` even as a
2131 // premise check: that samples one instant, and a square wave spends half
2132 // its period at zero, so it is phase-dependent -- exactly the flaw that
2133 // made the original assertion vacuous. Accumulated samples have no such
2134 // dependence: if anything was audible, some sample is non-zero.
2135 assert_eq!(muted.channel_mask() & 0x01, 0, "premise: pulse 1 is muted");
2136
2137 let plain_audio = plain.drain_audio();
2138 let muted_audio = muted.drain_audio();
2139 let quiet_audio = quiet.drain_audio();
2140 assert!(!plain_audio.is_empty(), "premise: samples were emitted");
2141 assert!(
2142 plain_audio.iter().any(|s| *s != 0.0),
2143 "premise: the default configuration produced audible output"
2144 );
2145 assert_ne!(
2146 plain_audio, muted_audio,
2147 "a cleared mask bit must change the emitted audio"
2148 );
2149 assert_ne!(
2150 plain_audio, quiet_audio,
2151 "a non-unity gain must change the emitted audio"
2152 );
2153 }
2154 use super::*;
2155
2156 #[test]
2157 fn write_4015_enables_channels() {
2158 let mut a = Apu::new(Region::Ntsc, 44_100);
2159 a.write_register(0x4015, 0x0F);
2160 assert!(a.pulse1.length.enabled);
2161 assert!(a.pulse2.length.enabled);
2162 assert!(a.triangle.length.enabled);
2163 assert!(a.noise.length.enabled);
2164 assert!(!a.dmc.active());
2165 }
2166
2167 #[test]
2168 fn write_4015_clears_lengths_when_disabled() {
2169 let mut a = Apu::new(Region::Ntsc, 44_100);
2170 a.pulse1.length.enabled = true;
2171 a.pulse1.length.count = 10;
2172 a.write_register(0x4015, 0x00);
2173 assert_eq!(a.pulse1.length.count, 0);
2174 }
2175
2176 #[test]
2177 fn read_4015_clears_frame_irq_not_dmc_irq() {
2178 // Session-25 (2026-05-23): the canonical Mesen2 lazy-clear
2179 // algorithm SCHEDULES the frame-IRQ flag clear instead of
2180 // performing it immediately. A GET-cycle (`apu_phase=true`)
2181 // read schedules a clear at `cpu_cycle + 1`; a tick then
2182 // matures the schedule and the flag observable on the next
2183 // CPU cycle is `false`. DMC IRQ is untouched.
2184 let mut a = Apu::new(Region::Ntsc, 44_100);
2185 a.frame_counter.irq_flag = true;
2186 a.dmc.irq_flag = true;
2187 a.apu_phase = true; // GET cycle (1-cycle delta)
2188 let v = a.read_status();
2189 assert_eq!(v & 0xC0, 0xC0, "read returns the OLD flag (still set)");
2190 // The flag is STILL set right after the read; the clear is
2191 // scheduled for `cpu_cycle + 1`.
2192 assert!(a.frame_counter.irq_flag);
2193 assert_ne!(a.frame_counter.irq_flag_clear_cycle, 0);
2194 assert!(a.dmc.irq_flag);
2195 // Tick once -- the scheduled clear matures inside the tick.
2196 a.tick();
2197 assert!(!a.frame_counter.irq_flag, "flag matures inside tick");
2198 assert_eq!(a.frame_counter.irq_flag_clear_cycle, 0);
2199 // DMC IRQ is independently retained.
2200 assert!(a.dmc.irq_flag);
2201 }
2202
2203 #[test]
2204 fn read_4015_on_put_cycle_defers_irq_clear_by_two_cycles() {
2205 // Session-25 (2026-05-23): a PUT-cycle (`apu_phase=false`)
2206 // read schedules the clear at `cpu_cycle + 2` instead of
2207 // `cpu_cycle + 1`. This is the AccuracyCoin `APU Frame
2208 // Counter IRQ` Test 7 axis: the SLO ABS,X double-read of
2209 // `$4015` on a PUT-cycle first read sees the flag STILL SET
2210 // on the second read 1 CPU cycle later (the schedule has not
2211 // yet matured).
2212 let mut a = Apu::new(Region::Ntsc, 44_100);
2213 a.frame_counter.irq_flag = true;
2214 a.apu_phase = false; // PUT cycle (2-cycle delta)
2215 let v = a.read_status();
2216 assert_eq!(v & 0x40, 0x40, "first read returns the OLD flag");
2217 assert!(a.frame_counter.irq_flag, "flag stays set on put-cycle read");
2218 let scheduled = a.frame_counter.irq_flag_clear_cycle;
2219 assert_eq!(scheduled, a.cpu_cycle.wrapping_add(2));
2220 // A second read on the SAME put cycle still sees the set
2221 // flag (the schedule hasn't matured: cpu_cycle == cpu_cycle).
2222 let v2 = a.read_status();
2223 assert_eq!(
2224 v2 & 0x40,
2225 0x40,
2226 "second read on same cycle still sees flag set"
2227 );
2228 assert!(a.frame_counter.irq_flag);
2229 // Now advance ONE CPU cycle via a tick. cpu_cycle becomes
2230 // scheduled - 1. The schedule has NOT yet matured.
2231 a.tick();
2232 assert!(a.frame_counter.irq_flag, "flag still set after 1 tick");
2233 // Advance the SECOND CPU cycle. cpu_cycle now equals
2234 // scheduled. The tick matures the clear.
2235 a.tick();
2236 assert!(!a.frame_counter.irq_flag, "flag matures after 2 ticks");
2237 assert_eq!(a.frame_counter.irq_flag_clear_cycle, 0);
2238 }
2239
2240 #[test]
2241 fn tick_advances_cycle_counter() {
2242 let mut a = Apu::new(Region::Ntsc, 44_100);
2243 for _ in 0..100 {
2244 a.tick();
2245 }
2246 assert_eq!(a.cpu_cycle, 100);
2247 }
2248
2249 #[test]
2250 fn channel_mask_defaults_to_all_on() {
2251 let a = Apu::new(Region::Ntsc, 44_100);
2252 assert_eq!(a.channel_mask(), CHANNEL_MASK_ALL);
2253 }
2254
2255 #[test]
2256 fn channel_mask_set_clamps_to_known_bits() {
2257 let mut a = Apu::new(Region::Ntsc, 44_100);
2258 // Upper bits beyond the 6 defined channels are masked off.
2259 a.set_channel_mask(0xFF);
2260 assert_eq!(a.channel_mask(), CHANNEL_MASK_ALL);
2261 a.set_channel_mask(0x00);
2262 assert_eq!(a.channel_mask(), 0x00);
2263 a.set_channel_mask(0b0010_1010);
2264 assert_eq!(a.channel_mask(), 0b0010_1010);
2265 }
2266
2267 #[test]
2268 fn default_mask_mix_is_byte_identical_to_unmasked() {
2269 // The determinism contract: with the default all-on mask, the gating in
2270 // `tick_with_external` must reproduce the raw mixer output exactly.
2271 let m = Mixer::new();
2272 let mask = CHANNEL_MASK_ALL;
2273 let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2274 for &(p1, p2, tri, n, dmc) in &[
2275 (0u8, 0u8, 0u8, 0u8, 0u8),
2276 (15, 15, 15, 15, 127),
2277 (7, 3, 11, 4, 60),
2278 (1, 14, 8, 15, 1),
2279 ] {
2280 let raw = m.mix(p1, p2, tri, n, dmc);
2281 let gated = m.mix(
2282 gate(0, p1),
2283 gate(1, p2),
2284 gate(2, tri),
2285 gate(3, n),
2286 gate(4, dmc),
2287 );
2288 assert_eq!(raw, gated, "default mask must be byte-identical");
2289 }
2290 }
2291
2292 #[test]
2293 fn cleared_channel_bit_zeroes_its_contribution() {
2294 let m = Mixer::new();
2295 // Mute pulse 1 only (bit 0 cleared).
2296 let mask = CHANNEL_MASK_ALL & !0x01;
2297 let gate = |bit: u8, v: u8| if mask & (1 << bit) != 0 { v } else { 0 };
2298 let muted = m.mix(gate(0, 15), gate(1, 0), gate(2, 0), gate(3, 0), gate(4, 0));
2299 // Pulse 1 = 15 muted to 0 => identical to an all-silent mix.
2300 assert_eq!(muted, m.mix(0, 0, 0, 0, 0));
2301 // Pulse 2 (bit 1 still set) still contributes.
2302 let p2_on = m.mix(gate(0, 15), gate(1, 15), gate(2, 0), gate(3, 0), gate(4, 0));
2303 assert!(p2_on > 0.0);
2304 }
2305
2306 #[test]
2307 fn channel_gain_defaults_to_unity() {
2308 let a = Apu::new(Region::Ntsc, 44_100);
2309 assert_eq!(a.channel_gain(), CHANNEL_GAIN_UNITY);
2310 }
2311
2312 #[test]
2313 fn external_out_tracks_last_external_sample() {
2314 // v2.1.6 — the read-only expansion-audio display tap reflects the most
2315 // recent RAW value fed to `tick_with_external` and defaults to 0.0.
2316 let mut a = Apu::new(Region::Ntsc, 44_100);
2317 assert_eq!(a.external_out(), 0.0);
2318 a.tick_with_external(0.25);
2319 assert!((a.external_out() - 0.25).abs() < f32::EPSILON);
2320 a.tick_with_external(-0.1);
2321 assert!((a.external_out() - (-0.1)).abs() < f32::EPSILON);
2322 // The tap is a copy: a non-unity external gain does NOT change what the
2323 // scope observes (it always sees the raw chip contribution).
2324 a.set_channel_gain([1.0, 1.0, 1.0, 1.0, 1.0, 0.5]);
2325 a.tick_with_external(0.4);
2326 assert!((a.external_out() - 0.4).abs() < f32::EPSILON);
2327 }
2328
2329 #[test]
2330 fn channel_gain_set_clamps_to_range() {
2331 let mut a = Apu::new(Region::Ntsc, 44_100);
2332 a.set_channel_gain([3.0, -1.0, 0.5, 1.0, 2.0, 0.0]);
2333 // 3.0 -> 2.0 (ceiling), -1.0 -> 0.0 (floor), the rest unchanged.
2334 assert_eq!(a.channel_gain(), [2.0, 0.0, 0.5, 1.0, 2.0, 0.0]);
2335 }
2336
2337 /// A NaN gain (only reachable from a hand-edited config) must not reach the
2338 /// mixer: `f32::clamp` passes NaN through, and one NaN term makes the mixed
2339 /// sample NaN. It falls back to unity; the infinities already clamp.
2340 #[test]
2341 fn channel_gain_rejects_nan_and_clamps_infinities() {
2342 let mut a = Apu::new(Region::Ntsc, 44_100);
2343 a.set_channel_gain([f32::NAN, f32::INFINITY, f32::NEG_INFINITY, 1.0, 1.0, 1.0]);
2344 assert_eq!(a.channel_gain(), [1.0, 2.0, 0.0, 1.0, 1.0, 1.0]);
2345 }
2346
2347 #[test]
2348 fn unity_gain_produces_byte_identical_samples() {
2349 // The hard determinism requirement: a full run with the default unity
2350 // gains must produce a bit-identical band-limited output to a fresh APU.
2351 const EXT: [f32; 7] = [0.0, 0.01, 0.02, 0.03, 0.04, 0.05, 0.06];
2352 let mut a = Apu::new(Region::Ntsc, 44_100);
2353 let mut b = Apu::new(Region::Ntsc, 44_100);
2354 b.set_channel_gain(CHANNEL_GAIN_UNITY); // explicit unity == default
2355 // Drive both with an identical register + tick sequence.
2356 for step in 0..4_000u32 {
2357 let v = (step & 0xFF) as u8;
2358 a.write_register(0x4000 + (step % 0x14) as u16, v);
2359 b.write_register(0x4000 + (step % 0x14) as u16, v);
2360 let ext = EXT[(step % 7) as usize];
2361 a.tick_with_external(ext);
2362 b.tick_with_external(ext);
2363 }
2364 let mut out_a = [0.0f32; 4096];
2365 let mut out_b = [0.0f32; 4096];
2366 let na = a.drain_audio_into(&mut out_a);
2367 let nb = b.drain_audio_into(&mut out_b);
2368 assert_eq!(na, nb);
2369 assert_eq!(
2370 out_a[..na],
2371 out_b[..nb],
2372 "unity gain must be bit-identical to the default mix"
2373 );
2374 }
2375
2376 /// v2.9.8: the per-cycle mix reads the cached `gain_is_unity`, so every
2377 /// path that changes the gain must refresh it. A power cycle carries the
2378 /// gain into a fresh APU through `adopt_settings_from`; a bare field copy
2379 /// there left the flag `true` and mixed a 0.5-gain channel at unity.
2380 #[test]
2381 fn a_power_cycle_keeps_a_non_unity_gain_audible() {
2382 const EXT: [f32; 7] = [0.0, 0.1, -0.2, 0.3, -0.1, 0.05, 0.0];
2383 let gain = [0.5, 1.0, 1.0, 1.0, 0.25, 1.0];
2384 let mut old = Apu::new(Region::Ntsc, 44_100);
2385 old.set_channel_gain(gain);
2386 // The power-cycled APU: a fresh one that adopted the old one's settings.
2387 let mut cycled = Apu::new(Region::Ntsc, 44_100);
2388 cycled.adopt_settings_from(&old);
2389 assert!(!cycled.gain_is_unity, "the cached flag went stale");
2390 // Reference: a fresh APU given the same gain through the setter.
2391 let mut direct = Apu::new(Region::Ntsc, 44_100);
2392 direct.set_channel_gain(gain);
2393 for step in 0..4_000u32 {
2394 let v = (step & 0xFF) as u8;
2395 cycled.write_register(0x4000 + (step % 0x14) as u16, v);
2396 direct.write_register(0x4000 + (step % 0x14) as u16, v);
2397 let ext = EXT[(step % 7) as usize];
2398 cycled.tick_with_external(ext);
2399 direct.tick_with_external(ext);
2400 }
2401 let mut out_c = [0.0f32; 4096];
2402 let mut out_d = [0.0f32; 4096];
2403 let nc = cycled.drain_audio_into(&mut out_c);
2404 let nd = direct.drain_audio_into(&mut out_d);
2405 assert_eq!(nc, nd);
2406 assert_eq!(
2407 out_c[..nc],
2408 out_d[..nd],
2409 "a power-cycled APU must mix with the gain it carried"
2410 );
2411 }
2412
2413 #[test]
2414 fn zero_gain_matches_a_cleared_mask_bit() {
2415 // Gain 0.0 on a channel is equivalent to clearing that channel's mask
2416 // bit (both force the raw output to 0 before the non-linear mixer).
2417 let m = Mixer::new();
2418 // Pulse 1 raw 15, everything else silent; gain 0 on pulse 1.
2419 // round(15 * 0.0) == 0, so the mixer sees pulse1 = 0.
2420 let scaled = m.mix(0, 0, 0, 0, 0);
2421 assert_eq!(scaled, m.mix(0, 0, 0, 0, 0));
2422 // Sanity: a real attenuation (0.5) lands strictly between full and muted.
2423 #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
2424 let half = (15.0f32 * 0.5).round() as u8; // 8
2425 let full = m.mix(15, 0, 0, 0, 0);
2426 let attenuated = m.mix(half, 0, 0, 0, 0);
2427 assert!(attenuated > 0.0 && attenuated < full);
2428 }
2429
2430 #[test]
2431 fn frame_irq_after_29828_cycles() {
2432 let mut a = Apu::new(Region::Ntsc, 44_100);
2433 // Default: 4-step mode, IRQ enabled.
2434 for _ in 0..29828 {
2435 a.tick();
2436 }
2437 assert!(a.frame_irq_pending());
2438 }
2439
2440 #[test]
2441 fn mode1_inhibits_irq() {
2442 let mut a = Apu::new(Region::Ntsc, 44_100);
2443 // Write mode=1 + inhibit. After ~3 cycles delay, fire qf+hf.
2444 a.write_register(0x4017, 0xC0);
2445 for _ in 0..40_000 {
2446 a.tick();
2447 }
2448 // No IRQ ever raised in mode 1.
2449 assert!(!a.frame_irq_pending());
2450 }
2451
2452 #[test]
2453 fn dmc_writes_dac_directly() {
2454 let mut a = Apu::new(Region::Ntsc, 44_100);
2455 a.write_register(0x4011, 0x40);
2456 assert_eq!(a.dmc.dac, 0x40);
2457 }
2458
2459 #[test]
2460 fn enabling_dmc_starts_sample() {
2461 let mut a = Apu::new(Region::Ntsc, 44_100);
2462 a.write_register(0x4012, 0x00);
2463 a.write_register(0x4013, 0x10); // 0x101 bytes
2464 a.write_register(0x4015, 0x10);
2465 assert!(a.dmc.active());
2466 }
2467}