Skip to main content

rustynes_cpu/
cpu.rs

1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the 6502/2A03 core is RustyNES's own, but the unstable-store opcode group (SHA/SHX/SHY/SHS/TAS — the `SyaSxaAxa` family) is derived from Mesen2 (GPL-3.0-or-later), `Core/NES/NesCpu.h`. See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! Ricoh 2A03 CPU (6502 derivative without BCD mode).
6//!
7//! See `docs/cpu-6502.md` for the spec. The implementation here matches:
8//!
9//! - all 151 documented 6502 opcodes,
10//! - all 105 unofficial / illegal opcodes that real software depends on,
11//! - the 12 JAM / KIL / STP halt opcodes,
12//! - cycle counts including page-crossing penalties on indexed reads, the
13//!   `+1 if branch taken / +2 if branch crosses page` branch convention, and
14//!   the dummy-read / dummy-write cycles of read-modify-write opcodes,
15//! - NMI (edge), IRQ (level), and BRK with the documented BRK/IRQ B-flag
16//!   distinction,
17//! - the `JMP ($XXFF)` indirect page-bug.
18//!
19//! The CPU steps one *instruction* at a time (`Cpu::step`), returning the
20//! cycle count. Since the v2.0.0 one-clock scheduler (ADR 0002 / ADR 0029)
21//! every cycle is clocked in two halves: `start_cycle` catches the PPU up
22//! (`Bus::run_ppu_to`) and runs the bus's per-cycle work (`Bus::cpu_clock`),
23//! and `end_cycle` catches the PPU up to the cycle's end and ticks the DMC
24//! (`Bus::cpu_clock_apu_dmc`). A cycle with a bus access (`read1` /
25//! `write1`) performs it between the halves; an internal cycle (`idle_tick`)
26//! runs both halves with no access. There
27//! is no separate `tick()`, and the pre-v2.0.0 per-cycle `Bus::on_cpu_cycle`
28//! callback survives only as the default body of `Bus::cpu_clock` for
29//! simple test buses. (Until v2.7.5 this header still described that
30//! callback as the stepping interface; core audit §4.7.)
31
32// Truncating casts are intentional throughout: this module is byte-arithmetic
33// against the 6502's 8/16-bit register file. `as u8` / `as i8` is the
34// canonical encoding of the wrap behavior the hardware exhibits.
35#![allow(
36    clippy::cast_possible_truncation,
37    clippy::cast_lossless,
38    clippy::cast_possible_wrap,
39    clippy::cast_sign_loss
40)]
41
42use crate::bus::Bus;
43use crate::status::Status;
44
45/// Stack base address: the CPU stack lives at `$0100 + S`.
46const STACK_BASE: u16 = 0x0100;
47
48// v2.0 master-clock R1 substrate constants (Phase 2; `mc-r1-substrate`).
49// The PPU sub-cycle offset and the read/write master-clock split (`pre` in
50// start_cycle, `post` in end_cycle). Mesen `_ppuOffset=1` /
51// `_startClockCount`/`_endClockCount`. Master clocks per CPU cycle are NOT a
52// constant — they are the cartridge region's `cpu_divider` (NTSC 12 / PAL 16 /
53// Dendy 15), read from `bus.cpu_divider()` and fed to `read_split`/
54// `write_split`; the master-clock unit is shared with the bus's `run_ppu_to`,
55// which does the regioned dot conversion off `ppu_divider`.
56/// PPU sub-cycle offset: the PPU is run to `master_clock - PPU_OFFSET` in BOTH
57/// halves of every access (the double catch-up). Mesen `_ppuOffset = 1`.
58const PPU_OFFSET: u64 = 1;
59
60/// The effective PPU-sample offset for `run_ppu_to` (no BP sweep: the constant).
61#[inline]
62const fn ppu_sample_offset() -> u64 {
63    PPU_OFFSET
64}
65/// READ access master-clock split (`+= pre` in `start_cycle`, `+= post` in
66/// `end_cycle`); `pre + post = div`, the region's `cpu_divider`. Derived per
67/// region from `bus.cpu_divider()` so PAL (16) / Dendy (15) get the right
68/// CPU<->PPU phase; for the NTSC divisor 12 these are exactly (5, 7), so the
69/// NTSC path is byte-identical to the prior `const`s.
70#[cfg(not(feature = "phi2-write-sweep"))]
71#[inline]
72const fn read_split(div: u64) -> (u64, u64) {
73    let pre = div / 2 - PPU_OFFSET;
74    (pre, div - pre)
75}
76
77/// Sweepable `read_split` (feature `phi2-write-sweep`, v2.6.18 study only).
78///
79/// A 6502 SAMPLES a read at phi2 just as it commits a write there, so a phi2
80/// model has to move both. The first sweep moved writes alone, which changed
81/// the SPACING between a write and a following read rather than moving the
82/// access model as a unit -- see the plan's note on that measurement.
83#[cfg(feature = "phi2-write-sweep")]
84#[inline]
85fn read_split(div: u64) -> (u64, u64) {
86    let extra = u64::from(READ_PHI_OFFSET.load(core::sync::atomic::Ordering::Relaxed));
87    let back = u64::from(READ_PHI_BACKOFF.load(core::sync::atomic::Ordering::Relaxed));
88    // `+ extra` BEFORE `- PPU_OFFSET`: identical for every reachable value
89    // (`div >= 12`, `PPU_OFFSET == 1`), but it removes the conceptual
90    // question of an unsigned subtraction preceding the addition.
91    // `- back` places the access EARLIER than shipped, which the offsets alone
92    // cannot express; clamped to 1 so `pre` stays a real split.
93    // Every step is structurally safe rather than safe-by-current-constants:
94    // the subtraction of `PPU_OFFSET` saturates (raised in review -- it cannot
95    // underflow at `div >= 12`, but nothing in the expression says so), and the
96    // upper clamp bound is floored at 1 because `clamp` PANICS when min > max,
97    // which a hypothetical `div < 2` would produce.
98    let pre = (div / 2 + extra)
99        .saturating_sub(PPU_OFFSET)
100        .saturating_sub(back)
101        .clamp(1, div.saturating_sub(1).max(1));
102    (pre, div - pre)
103}
104
105/// Extra master clocks added to a READ's pre-access split (v2.6.18 study).
106/// 0 = shipped behaviour. NTSC: +4 puts the sample at dot 2.0 = phi2.
107#[cfg(feature = "phi2-write-sweep")]
108pub static READ_PHI_OFFSET: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
109
110/// Master clocks SUBTRACTED from a READ's pre-access split (v2.6.18 study).
111///
112/// The offsets alone can only place an access later. Reaching the CPU cycle's
113/// FIRST dot needs it earlier, and that cell had never been measured -- see
114/// `access_dot_derivation.rs`, which sweeps the grid this pair makes reachable.
115#[cfg(feature = "phi2-write-sweep")]
116pub static READ_PHI_BACKOFF: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
117/// WRITE access split — swapped (writes commit `2 * PPU_OFFSET` mc later than
118/// reads). NTSC divisor 12 → (7, 5), byte-identical to the prior `const`s.
119///
120/// A 6502 commits a write at phi2, the LAST of a CPU cycle's three PPU dots.
121/// At the shipped `pre` of 7 (minus `PPU_OFFSET`) the PPU has advanced 6 of
122/// 12 master clocks — 1.5 dots — so the commit lands mid-cycle instead.
123///
124/// **That is a known divergence, measured and deliberately NOT corrected at
125/// v2.6.18.** Moving the commit to phi2 is the right diagnosis and was the
126/// wrong change as applied: the write alone reads 141/144 on `AccuracyCoin`
127/// and fails `ppu_vbl_nmi/10-even_odd_timing`; the best combination found
128/// (phi2 + the dot-321 OAM2 increment + a one-stage `mask_for_skip_check`)
129/// reads 142/144 against the 143/144 that ships. Several PPU behaviours
130/// compensate for the placement — `mask_for_skip_check` says so in its own
131/// comment — and replacing a documented compensation with an undocumented one
132/// is worse than keeping it. The `phi2-write-sweep` feature keeps the knob so
133/// the next attempt re-measures rather than rebuilding the apparatus; see the
134/// *CLOSED* section of `to-dos/plans/v2.6.18-terminus-plan.md` for every
135/// number and the three conditions for reopening it.
136#[cfg(not(feature = "phi2-write-sweep"))]
137#[inline]
138const fn write_split(div: u64) -> (u64, u64) {
139    let pre = div / 2 + PPU_OFFSET;
140    (pre, div - pre)
141}
142
143/// Sweepable `write_split` (feature `phi2-write-sweep`, v2.6.18 study only).
144///
145/// Default 0 reproduces the `const` path above exactly, so enabling the
146/// feature without setting the knob is byte-identical. Kept behind a feature
147/// rather than always-on because `write_split` runs on EVERY write, and the
148/// shipped build should not pay an atomic load per write for a study.
149#[cfg(feature = "phi2-write-sweep")]
150#[inline]
151fn write_split(div: u64) -> (u64, u64) {
152    let extra = u64::from(WRITE_PHI_OFFSET.load(core::sync::atomic::Ordering::Relaxed));
153    let back = u64::from(WRITE_PHI_BACKOFF.load(core::sync::atomic::Ordering::Relaxed));
154    // Clamp so `pre` never reaches the cycle length: `post` must stay >= 1 or
155    // `end_cycle` would not advance the master clock at all. The lower clamp
156    // matters for the same reason once `back` can pull the access earlier.
157    // See `read_split` for why each step saturates and why the upper clamp
158    // bound is floored at 1.
159    let pre = (div / 2 + PPU_OFFSET + extra)
160        .saturating_sub(back)
161        .clamp(1, div.saturating_sub(1).max(1));
162    (pre, div - pre)
163}
164
165/// Extra master clocks added to a WRITE's pre-access split (v2.6.18 study).
166/// 0 = shipped behaviour.
167#[cfg(feature = "phi2-write-sweep")]
168pub static WRITE_PHI_OFFSET: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
169
170/// Master clocks SUBTRACTED from a WRITE's pre-access split (v2.6.18 study).
171/// The counterpart of [`READ_PHI_BACKOFF`]; see it for why subtraction is
172/// needed at all.
173#[cfg(feature = "phi2-write-sweep")]
174pub static WRITE_PHI_BACKOFF: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
175
176/// NMI vector low byte address (`$FFFA/B`).
177const NMI_VECTOR: u16 = 0xFFFA;
178
179/// Reset vector low byte address (`$FFFC/D`).
180const RESET_VECTOR: u16 = 0xFFFC;
181
182/// IRQ / BRK vector low byte address (`$FFFE/F`).
183const IRQ_VECTOR: u16 = 0xFFFE;
184
185/// 6502 CPU core.
186//
187// Multiple boolean state bits track distinct interrupt-pipeline stages
188// (jam, NMI pending, NMI armed, IRQ pending, IRQ armed).  These map directly
189// onto orthogonal hardware latches; collapsing them into an enum or bitflags
190// would obscure rather than clarify the model.
191#[allow(clippy::struct_excessive_bools)]
192#[derive(Debug, Clone)]
193pub struct Cpu {
194    /// Accumulator.
195    pub a: u8,
196    /// X index register.
197    pub x: u8,
198    /// Y index register.
199    pub y: u8,
200    /// Program counter.
201    pub pc: u16,
202    /// Stack pointer (low byte; effective address `0x0100 | s`).
203    pub s: u8,
204    /// Processor status.
205    pub p: Status,
206    /// Cumulative CPU cycle count.
207    pub cycles: u64,
208    /// `true` when the CPU has executed a JAM/KIL/STP and is waiting for reset.
209    pub jammed: bool,
210    /// Edge-detected NMI latch.  Real hardware samples the NMI line at the
211    /// second-to-last cycle of every instruction; if asserted, the NMI
212    /// sequence is queued for AFTER the current instruction completes.  Our
213    /// `step` model dispatches all bus operations atomically before ticking
214    /// cycles, so a write that itself raises NMI (e.g.  enabling NMI in
215    /// `$2000` while VBL is set) is observed by the bus's edge detector
216    /// during this instruction's cycle tally.  Hardware would not have
217    /// observed it at the second-to-last cycle (the write logically happens
218    /// at the LAST cycle), so the NMI is queued for the NEXT instruction's
219    /// sample point and serviced AFTER the next instruction completes.
220    /// We approximate that by promoting `pending_nmi` to `armed_nmi` after
221    /// each instruction; only `armed_nmi` actually services.
222    pub(crate) pending_nmi: bool,
223    /// NMI ready to be serviced before the next instruction starts.
224    pub(crate) armed_nmi: bool,
225    /// IRQ pending — captured edge of the level line that fires once the
226    /// I-flag is clear.  Same double-latch promotion as NMI.
227    pub(crate) pending_irq: bool,
228    /// IRQ ready to be serviced before the next instruction starts.
229    pub(crate) armed_irq: bool,
230    /// First tick (within the current instruction's tally loop) at which
231    /// NMI was sampled high; `u8::MAX` if not seen.  Used by the
232    /// second-to-last-cycle interrupt classification.
233    pub(crate) nmi_first_tick: u8,
234    /// First tick at which IRQ was sampled high; `u8::MAX` if not seen.
235    pub(crate) irq_first_tick: u8,
236    /// Snapshot of the I flag at the *start* of the current instruction.
237    /// Hardware samples IRQ near the end of the instruction with the old
238    /// I-flag value; CLI / SEI / PLP / RTI take effect at the very last
239    /// cycle, AFTER the IRQ sample point.  We arm IRQ only when this
240    /// snapshot says I was clear at sample time.
241    pub(crate) irq_sample_i_flag: bool,
242    /// Number of cycles emitted by the per-cycle helpers
243    /// (`read1`/`write1`/`idle_tick`) within the *current* instruction.
244    /// Reset to zero at the top of `step()`. Used by the trailing
245    /// "burn remaining cycles" loop so we can incrementally migrate
246    /// opcodes from atomic-dispatch + trailing-loop to fully per-cycle
247    /// emission.
248    pub(crate) cycles_emitted: u8,
249    /// When `true`, [`Cpu::idle_tick`] does NOT update `irq_first_tick`.
250    /// Used by branch opcodes to model the `branch_delays_irq` quirk:
251    /// real 6502 branches poll IRQ at the same point a 2-cycle untaken
252    /// branch would (the opcode-fetch cycle, which `step()` performs
253    /// before entering dispatch).  The operand fetch and any extra
254    /// taken / page-cross cycles do *not* re-sample IRQ.  The branch
255    /// dispatch sets this flag *before* the operand fetch and `step()`
256    /// clears it at the top of every instruction.
257    /// Since v2.6.7 the same rule defers NMI *dispatch*: while this flag and
258    /// `skip_irq_sample_q` are both set, `handle_interrupts` freezes the
259    /// dispatch copy `mc_prev_need_nmi`. The NMI edge latch (`mc_need_nmi`)
260    /// keeps running every cycle, so an edge is never lost, only recognised
261    /// after the branch.
262    pub(crate) skip_irq_sample: bool,
263    /// `skip_irq_sample` as it stood on the PREVIOUS cycle. The NMI dispatch
264    /// gate freezes on this rather than on the live flag, because the two
265    /// latches have different shapes: `mc_run_irq` is RECOMPUTED from the live
266    /// line every cycle, so freezing it in place holds its end-of-C1 value,
267    /// while `mc_need_nmi` is STICKY and it is the *copy* that has to freeze --
268    /// one cycle later, or the copy that captures end-of-C1 never happens.
269    pub(crate) skip_irq_sample_q: bool,
270
271    // === v2.0 master-clock R1 substrate (Phase 2; `mc-r1-substrate`) ===
272    /// The CPU's authoritative master clock (Mesen `_masterClock` / `TetaNES`
273    /// `Cpu::master_clock`). Advanced by `start_cycle`/`end_cycle`; the bus is
274    /// caught up to `master_clock - PPU_OFFSET` from BOTH halves (double
275    /// catch-up). Only live under `mc-r1-substrate`.
276    pub(crate) master_clock: u64,
277    /// NMI edge-recognition latch (set on a /NMI rising edge in
278    /// `handle_interrupts`; consumed by the cycle-5 hijack in
279    /// `service_interrupt`). Mesen `_needNmi`.
280    pub(crate) mc_need_nmi: bool,
281    /// One-cycle-delayed copy of `mc_need_nmi` (the dispatch + hijack gate).
282    /// Mesen `_prevNeedNmi`.
283    pub(crate) mc_prev_need_nmi: bool,
284    /// Live IRQ-recognition latch (`irq_level && !irq_sample_i_flag`,
285    /// recomputed every `end_cycle`). Mesen `_runIrq`.
286    pub(crate) mc_run_irq: bool,
287    /// One-cycle-delayed copy of `mc_run_irq` (the dispatch gate). Mesen
288    /// `_prevRunIrq`.
289    pub(crate) mc_prev_run_irq: bool,
290    /// Previous /NMI line level, for the φ2 rising-edge detector in
291    /// `handle_interrupts`. Mesen `_prevNmiFlag`.
292    pub(crate) mc_prev_nmi_line: bool,
293    /// v2.0.0 beta.2 (A2 scoping diagnostic): per-opcode count of cycles the
294    /// trailing burn-loop had to fill (`cycles - cycles_emitted`) — the exact
295    /// remaining busless-cycle surface the every-cycle-bus-access conversion
296    /// must turn into dummy reads of the held address. Read by the harness
297    /// `burn_probe` bin; never consulted by emulation.
298    #[cfg(feature = "cpu-instr-cycle-trace")]
299    pub burn_histogram: [u64; 256],
300}
301
302impl Default for Cpu {
303    fn default() -> Self {
304        Self::new()
305    }
306}
307
308/// Effective address + page-crossed flag, returned by addressing-mode resolvers.
309#[derive(Clone, Copy)]
310struct Operand {
311    addr: u16,
312    page_crossed: bool,
313}
314
315impl Cpu {
316    /// New CPU in "post-reset" state. Caller must invoke [`Cpu::reset`] with a
317    /// real bus before stepping (PC is undefined until reset reads `$FFFC/D`).
318    ///
319    /// This constructor is the convenience entry-point used by unit tests and
320    /// nestest fixtures that drive the CPU without going through a full
321    /// power-on path: `S=$FD`, `P=$24` (`UNUSED` + `INTERRUPT_DISABLE`). If the
322    /// caller subsequently invokes [`Cpu::reset`] the stack pointer will be
323    /// decremented by 3 (per the reset sequence), landing on `$FA` — that is
324    /// the input shape several `tests/opcodes.rs` fixtures expect.
325    ///
326    /// **For the real cold-boot path** (`Nes::from_rom`, `Nes::power_cycle`),
327    /// use [`Cpu::power_on`] instead, which seeds `S=$00`. After the 3-decrement
328    /// reset sequence that lands `S=$FD`, matching Mesen2's power-up state.
329    /// See `docs/audit/session-13-cpu-boot-fix-2026-05-21.md` for the reference
330    /// behaviour from `Core/NES/NesCpu.cpp::NesCpu::Reset(softReset=false)`.
331    #[must_use]
332    pub const fn new() -> Self {
333        Self {
334            a: 0,
335            x: 0,
336            y: 0,
337            pc: 0,
338            s: 0xFD,
339            p: Status::power_on(),
340            cycles: 0,
341            jammed: false,
342            pending_nmi: false,
343            armed_nmi: false,
344            pending_irq: false,
345            armed_irq: false,
346            nmi_first_tick: u8::MAX,
347            irq_first_tick: u8::MAX,
348            irq_sample_i_flag: true,
349            cycles_emitted: 0,
350            skip_irq_sample: false,
351            skip_irq_sample_q: false,
352            master_clock: 0,
353            mc_need_nmi: false,
354            mc_prev_need_nmi: false,
355            mc_run_irq: false,
356            mc_prev_run_irq: false,
357            mc_prev_nmi_line: false,
358            #[cfg(feature = "cpu-instr-cycle-trace")]
359            burn_histogram: [0; 256],
360        }
361    }
362
363    /// New CPU in real-hardware cold-boot state (`S=$00`).
364    ///
365    /// Real silicon comes up with the stack pointer in an undefined state;
366    /// the convention used by Mesen2 (and adopted here for trace parity) is
367    /// to treat power-up as `S=$00` and rely on the reset sequence's three
368    /// "phantom" decrements to wrap into `S=$FD`. See Mesen2
369    /// `Core/NES/NesCpu.cpp::Reset(softReset=false)`:
370    ///
371    /// ```cpp
372    /// if(softReset) {
373    ///     _state.SP -= 0x03;          // soft reset path
374    /// } else {
375    ///     _state.SP = 0xFD;           // power-up: direct assignment
376    /// }
377    /// ```
378    ///
379    /// `RustyNES` models that two-path behaviour by gating the SP delta through
380    /// the constructor: `Cpu::power_on() + reset()` ⇒ `$00 - 3 = $FD` (cold);
381    /// `cpu.reset()` again ⇒ `$FD - 3 = $FA` (subsequent soft reset).
382    ///
383    /// `P` is left at `$24` (`INTERRUPT_DISABLE` | `UNUSED`). Mesen2's trace
384    /// surface shows `P = $04` because it masks `UNUSED` out of the displayed
385    /// byte, but the bit is conventionally always set on a 6502 internally
386    /// (nesdev: "Bit 5: Always 1, the so-called 'unused' bit"); the trace
387    /// divergence on P is cosmetic.
388    #[must_use]
389    pub const fn power_on() -> Self {
390        let mut cpu = Self::new();
391        cpu.s = 0x00;
392        cpu
393    }
394
395    /// Returns `true` when the CPU has executed a JAM/KIL/STP.
396    #[must_use]
397    pub const fn is_jammed(&self) -> bool {
398        self.jammed
399    }
400
401    /// Read-only accessor for the CPU's authoritative master clock
402    /// (v2.0.0-beta.1 one-clock instrumentation).
403    ///
404    /// `master_clock` counts master-clock units (NTSC: 12 per CPU cycle,
405    /// PAL: 16, Dendy: 15) and is advanced only by `start_cycle` /
406    /// `end_cycle` (the asymmetric read 5/7 vs write 7/5 φ1/φ2 split on
407    /// NTSC). It is the counter the v2.0.0
408    /// "Timebase" rewrite (ADR 0002) promotes to the ONE canonical
409    /// timebase; the test harness asserts the affine relation
410    /// `master_clock == seed + cpu_divider * cycles` against the other
411    /// cycle counters (`one_clock_invariants.rs`) as the gate for the
412    /// beta.1 counter collapse.
413    #[must_use]
414    pub const fn master_clock(&self) -> u64 {
415        self.master_clock
416    }
417
418    /// Reset (warm boot).
419    ///
420    /// Real hardware: 8-cycle sequence with suppressed pushes, then PC loads
421    /// from the reset vector and the I flag is set. We model the cycle count
422    /// (advances `cycles` by 8 and fires `on_cpu_cycle` 8 times) without
423    /// mutating registers other than P (set I), S (decrement by 3), and PC.
424    ///
425    /// Matches Mesen2's `NesCpu::Reset()` 8-cycle post-power-up loop ("CPU
426    /// takes 8 cycles before it starts executing the ROM's code"). Combined
427    /// with the PPU power-up at (scanline=-1, dot=340) (see `Ppu::new`),
428    /// this closes the +344-dot PPU offset identified empirically in
429    /// Session-13 (docs/audit/session-13-cpu-boot-fix-2026-05-21.md).
430    pub fn reset<B: Bus>(&mut self, bus: &mut B) {
431        // Real hardware decrements S three times during reset (no actual
432        // pushes occur, but the decrements happen).
433        self.s = self.s.wrapping_sub(3);
434        self.p.insert(Status::INTERRUPT_DISABLE);
435        self.jammed = false;
436        self.pending_nmi = false;
437        self.armed_nmi = false;
438        self.pending_irq = false;
439        self.armed_irq = false;
440        self.nmi_first_tick = u8::MAX;
441        self.irq_first_tick = u8::MAX;
442        // R1/R3 cold-boot: advance master_clock by one CPU divider BEFORE the
443        // 8-cycle reset loop (Mesen `NesCpu::Reset()` `_masterClock += cpuDivider
444        // + cpuOffset`). Without it the first start_cycle's `run_ppu_to(mc-1)`
445        // would leave the PPU 3 dots behind. See R1 port plan / branch `acddd22`.
446        {
447            self.master_clock = self.master_clock.wrapping_add(bus.cpu_divider());
448        }
449        // V-axis: Mesen adds `cpuDivider + cpuOffset` (13), not just cpuDivider
450        // (12). The +PPU_OFFSET corrects R1's power-up CPU/PPU sub-cycle
451        // alignment to the reference, shifting every $2002 poll-exit to match
452        // hardware (nesdev `PPU_frame_timing`: the read sees the flag change iff
453        // it starts at/after the set tick). Default-off; A/B against Y/C1/6-10.
454        // 8-cycle reset sequence: 6 idle/internal cycles + 2 vector reads.
455        self.cycles_emitted = 0;
456        for _ in 0..6 {
457            self.idle_tick(bus);
458        }
459        let lo = self.read1(bus, RESET_VECTOR);
460        let hi = self.read1(bus, RESET_VECTOR + 1);
461        self.pc = u16::from(lo) | (u16::from(hi) << 8);
462    }
463
464    /// Force PC to `addr`. Used by the nestest harness which enters at
465    /// `$C000` rather than the reset vector.
466    pub const fn set_pc(&mut self, addr: u16) {
467        self.pc = addr;
468    }
469
470    /// Step one instruction (or service an interrupt). Returns the number of
471    /// CPU cycles consumed.
472    ///
473    /// On a JAM-state CPU this is a no-op returning 0.
474    ///
475    /// # Interrupt timing model
476    ///
477    /// Real 6502 hardware samples the NMI / IRQ lines at the *second-to-last*
478    /// cycle of every instruction and, if asserted there, queues the
479    /// interrupt to be serviced *after* the current instruction completes.
480    /// Our model dispatches all bus operations atomically before ticking
481    /// cycles, so a write that itself raises NMI (e.g. `STA $2000` enabling
482    /// NMI while VBL is set) appears to the bus's edge detector during the
483    /// FIRST cycle of the tally loop — earlier than hardware would observe
484    /// it.  Hardware places the actual write at the *last* cycle of the
485    /// instruction, so the second-to-last sample point would NOT see the new
486    /// line state; only the NEXT instruction's sample sees it.  We model
487    /// that by introducing a one-instruction promotion delay: edges captured
488    /// at end-of-step land in `pending_*` and, after the following step,
489    /// promote to `armed_*` which is the gate that actually triggers
490    /// service.  This passes `04-nmi_control` test 11 ("Immediate occurence
491    /// should be after NEXT instruction") without regressing the
492    /// instruction-count-insensitive tests like `02-vbl_set_time`,
493    /// `09-even_odd_frames`, or any `instr_test_v5` ROM (which never raise
494    /// NMI from within a single instruction).
495    #[allow(clippy::too_many_lines, clippy::missing_panics_doc)]
496    pub fn step<B: Bus>(&mut self, bus: &mut B) -> u8 {
497        if self.jammed {
498            return 0;
499        }
500        // v2.0 master-clock R1 UNIFIED interrupt dispatch (`mc-r1-substrate`):
501        // a SINGLE service sequence gated on the one-cycle-delayed `prev_*`
502        // copies (Mesen `_prevRunIrq || _prevNeedNmi`). The vector is chosen
503        // INSIDE `service_interrupt` by the live `mc_need_nmi` at cycle 5 (the
504        // NMI hijack); NMI priority is resolved there, not here. Setting
505        // `irq_sample_i_flag = true` BEFORE the service masks the φ2 sampler
506        // for the 7-cycle sequence (no re-entry). Clears `skip_irq_sample`
507        // (a prior taken-branch could have left it set, freezing the recompute)
508        // and defers any still-pending NMI by one instruction.
509        if self.mc_prev_run_irq || self.mc_prev_need_nmi {
510            self.armed_irq = false;
511            self.irq_sample_i_flag = true;
512            self.skip_irq_sample = false;
513            self.service_interrupt(bus, IRQ_VECTOR, false);
514            self.mc_prev_need_nmi = false;
515            self.promote_post_step_interrupts(7);
516            return 7;
517        }
518        // Service an armed interrupt before the next instruction.  NMI has
519        // priority over IRQ; both are mutually exclusive for a single
520        // service window.
521        // Once armed, the IRQ services unconditionally — the I-flag
522        // gating already happened at the sample point (second-to-last
523        // cycle of the prior instruction).  This is what produces the
524        // "CLI SEI should still allow one IRQ to fire" behavior:
525        // SEI's I=1 takes effect at end-of-SEI but the sample at SEI's
526        // second-to-last cycle saw I=0 (CLI cleared it) and queued the
527        // IRQ, which now fires regardless of the current I-flag.
528
529        // Per-instruction state for the per-cycle helpers
530        // (`read1`/`write1`/`idle_tick`).  These track the FIRST tick at
531        // which each interrupt line was seen high; hardware samples at the
532        // second-to-last cycle so seen < last_tick = arm now;
533        // seen == last_tick = defer one instruction (the next instruction's
534        // sample window catches it instead).
535        self.nmi_first_tick = u8::MAX;
536        self.irq_first_tick = u8::MAX;
537        self.cycles_emitted = 0;
538        // Cleared every instruction; set inside the branch dispatch arms
539        // (after the operand fetch / canonical IRQ poll) to suppress
540        // further IRQ sampling on the additional taken / page-cross
541        // branch cycles.
542        self.skip_irq_sample = false;
543        // Snapshot the I flag for this instruction.  CLI / SEI / PLP /
544        // RTI mutate `self.p` *during* the instruction, but the hardware
545        // IRQ sample reads the I value as it was at the start.  This is
546        // what produces the documented "exactly one instruction after
547        // CLI executes before IRQ is taken" delay.
548        self.irq_sample_i_flag = self.p.contains(Status::INTERRUPT_DISABLE);
549
550        #[cfg(feature = "cpu-instr-cycle-trace")]
551        bus.trace_instr(self.pc, self.cycles);
552
553        let opcode = self.fetch_pc(bus);
554        let mut cycles = 0u8;
555        self.dispatch(bus, opcode, &mut cycles);
556        // Burn whichever cycles the dispatch did NOT emit through helpers.
557        // As opcodes migrate to fully per-cycle emission, this loop runs
558        // for fewer iterations; eventually it can be removed entirely.
559        //
560        // v2.0.0 beta.2 (A2 scoping): the diagnostic histogram below records,
561        // per opcode, how many cycles the burn-loop had to fill — the exact
562        // empirical work list for the every-cycle-bus-access conversion (the
563        // remaining busless cycles that must become dummy reads of the held
564        // address). Default-off; the `burn_probe` harness bin prints it.
565        #[cfg(feature = "cpu-instr-cycle-trace")]
566        {
567            let burned = cycles.saturating_sub(self.cycles_emitted);
568            if burned > 0 {
569                self.burn_histogram[opcode as usize] =
570                    self.burn_histogram[opcode as usize].saturating_add(u64::from(burned));
571            }
572        }
573        // v2.0.0 beta.2 (A2, promoted to the only path in beta.4): every
574        // instruction cycle is a bus access — the resolvers + RMW arms emit
575        // the canonical dummy reads, so the burn-loop must never fire.
576        // Proven empirically at zero across AccuracyCoin, nestest, both
577        // blargg_nes_cpu_test5 suites, and cpu_timing_test6 (the full
578        // official + unofficial opcode space); this assert makes any future
579        // under-emitting dispatch arm fail loud in dev-profile runs instead
580        // of silently reintroducing a busless cycle.
581        debug_assert!(
582            self.cycles_emitted >= cycles,
583            "opcode ${opcode:02X} under-emitted: declared {cycles} cycles but emitted \
584             only {} — a busless burn-loop cycle would fill the gap (A2 regression; \
585             see the v2.0.0 plan Workstream A2)",
586            self.cycles_emitted
587        );
588        while self.cycles_emitted < cycles {
589            self.idle_tick(bus);
590        }
591        self.promote_post_step_interrupts(cycles);
592        cycles
593    }
594
595    /// Promote any per-instruction interrupt edges captured by the
596    /// per-cycle helpers into the `armed_*` / `pending_*` latches the
597    /// next [`Cpu::step`] consults.  Hardware samples interrupts at the
598    /// second-to-last cycle of an instruction; we approximate that with
599    /// "first sampled tick strictly before the last cycle = arm now,
600    /// else defer one instruction."
601    const fn promote_post_step_interrupts(&mut self, cycles: u8) {
602        // Promote any previously-pending interrupt (latched at the very
603        // last cycle of the prior instruction).
604        if self.pending_nmi {
605            self.armed_nmi = true;
606            self.pending_nmi = false;
607        }
608        if self.pending_irq {
609            self.armed_irq = true;
610            self.pending_irq = false;
611        }
612        let last_tick = cycles.saturating_sub(1);
613        if self.nmi_first_tick != u8::MAX {
614            if self.nmi_first_tick < last_tick {
615                self.armed_nmi = true;
616            } else {
617                self.pending_nmi = true;
618            }
619        }
620        if self.irq_first_tick != u8::MAX {
621            // IRQ is masked by the I-flag value as it was at the START of
622            // this instruction; CLI / SEI / PLP / RTI mutations take effect
623            // at end-of-instruction.  If IRQ was already disabled when we
624            // entered, the second-to-last-cycle sample sees I=1 and the
625            // edge is dropped (the next instruction's sample will pick it
626            // up if I has since cleared).
627            if !self.irq_sample_i_flag {
628                if self.irq_first_tick < last_tick {
629                    self.armed_irq = true;
630                } else {
631                    self.pending_irq = true;
632                }
633            }
634        }
635    }
636
637    fn fetch_pc<B: Bus>(&mut self, bus: &mut B) -> u8 {
638        let v = self.read1(bus, self.pc);
639        self.pc = self.pc.wrapping_add(1);
640        v
641    }
642
643    fn fetch_pc_u16<B: Bus>(&mut self, bus: &mut B) -> u16 {
644        let lo = self.fetch_pc(bus);
645        let hi = self.fetch_pc(bus);
646        u16::from(lo) | (u16::from(hi) << 8)
647    }
648
649    fn read_u16_with_wrap<B: Bus>(&mut self, bus: &mut B, addr: u16) -> u16 {
650        // Used by indirect modes to honor the 6502 page-wrap quirk.
651        let lo = self.read1(bus, addr);
652        let hi_addr = (addr & 0xFF00) | u16::from((addr as u8).wrapping_add(1));
653        let hi = self.read1(bus, hi_addr);
654        u16::from(lo) | (u16::from(hi) << 8)
655    }
656
657    fn push<B: Bus>(&mut self, bus: &mut B, value: u8) {
658        self.write1(bus, STACK_BASE | u16::from(self.s), value);
659        self.s = self.s.wrapping_sub(1);
660    }
661
662    fn pull<B: Bus>(&mut self, bus: &mut B) -> u8 {
663        self.s = self.s.wrapping_add(1);
664        self.read1(bus, STACK_BASE | u16::from(self.s))
665    }
666
667    fn push_u16<B: Bus>(&mut self, bus: &mut B, value: u16) {
668        self.push(bus, (value >> 8) as u8);
669        self.push(bus, (value & 0xFF) as u8);
670    }
671
672    fn pull_u16<B: Bus>(&mut self, bus: &mut B) -> u16 {
673        let lo = self.pull(bus);
674        let hi = self.pull(bus);
675        u16::from(lo) | (u16::from(hi) << 8)
676    }
677
678    // ------------------------------------------------------------------
679    // Per-cycle bus interleaving primitives.
680    //
681    // Real 6502 hardware reads or writes the bus *exactly once per CPU
682    // cycle*; "internal" cycles (ALU work, stack pointer increment, etc.)
683    // still tick the system clock without driving the address bus.
684    // `read1` / `write1` / `idle_tick` model that one-cycle granularity:
685    // each one ticks the bus exactly once and samples the NMI/IRQ lines
686    // at the end of the cycle, mirroring what the existing trailing
687    // tally loop in `step` did but with the polling now coupled to the
688    // *actual* memory access ordering.
689    //
690    // `step()` resets `cycles_emitted` to 0; each call here increments
691    // it.  The trailing burn-loop in `step` consumes whatever cycles
692    // the opcode declared but didn't emit through these helpers.
693    // ------------------------------------------------------------------
694
695    // === v2.0 master-clock R1 substrate core (Phase 2; `mc-r1-substrate`) ===
696
697    /// Start half of one CPU cycle: advance `master_clock` by the PRE split,
698    /// catch the PPU up to `master_clock - PPU_OFFSET`, then fire the bus's
699    /// per-cycle work (`cpu_clock`). After this the PPU is at the access's
700    /// exact master clock (so a `$2002`/`$2007` read sees on-time state).
701    fn start_cycle<B: Bus>(&mut self, bus: &mut B, for_read: bool) {
702        let div = bus.cpu_divider();
703        let pre = if for_read {
704            read_split(div).0
705        } else {
706            write_split(div).0
707        };
708        self.master_clock = self.master_clock.wrapping_add(pre);
709        bus.run_ppu_to(self.master_clock.saturating_sub(ppu_sample_offset()), false);
710        bus.cpu_clock();
711        // v2.0.0 beta.1 (A1 one-clock collapse, promoted to the only path in
712        // beta.4): `cycles` is ASSIGNED from the canonical bus cycle counter
713        // at this single per-cycle site instead of being independently
714        // incremented by every `read1`/`write1`/`idle_tick`/DMA-loop caller.
715        // `bus.cpu_clock()` above advanced the canonical counter for THIS
716        // cycle, so the assignment lands on the same post-increment value the
717        // legacy caller-side `+= 1` produced (the `one_clock_invariants`
718        // harness test pins the residue at zero).
719        self.cycles = bus.cycle_count();
720    }
721
722    /// End half of one CPU cycle: advance by the POST split, catch the PPU
723    /// up again (the double catch-up), then sample interrupts (φ2, the
724    /// T_last-1 rule).
725    ///
726    /// Every DMA cycle is a first-class `start_cycle`/`end_cycle` on the
727    /// unified-DMA path, advancing `master_clock` directly, so there is no
728    /// bus-side DMA span to fold in. The `take_dma_mc_consumed` fold that
729    /// once did that was retired at v2.0.0 beta.1 (it only ever mattered for
730    /// the pre-v2.0.0 bus-side burst engine) and its hook was removed at
731    /// v2.9.8 with the rest of that engine's dead code (ADR 0042).
732    fn end_cycle<B: Bus>(&mut self, bus: &mut B, for_read: bool) {
733        let div = bus.cpu_divider();
734        let post = if for_read {
735            read_split(div).1
736        } else {
737            write_split(div).1
738        };
739        self.master_clock = self.master_clock.wrapping_add(post);
740        bus.run_ppu_to(self.master_clock.saturating_sub(ppu_sample_offset()), true);
741        // F-2: tick the DMC byte-timer at END of cycle (after the access),
742        // matching main's DMC fire-phase for DMASync, BEFORE the φ2 interrupt
743        // sample so handle_interrupts sees the post-tick DMC IRQ line.
744        bus.cpu_clock_apu_dmc();
745        self.handle_interrupts(bus);
746        // Diagnostic trace hook (no-op unless the bus enables irq-timing-trace).
747        bus.trace_end_cycle();
748    }
749
750    /// φ2 interrupt sampler (Mesen `EndCpuCycle`): edge-detect /NMI into
751    /// `mc_need_nmi` (after copying the one-cycle-delayed `mc_prev_need_nmi`),
752    /// and recompute `mc_run_irq = irq_level && !irq_sample_i_flag` (after the
753    /// `mc_prev_run_irq` copy). The `step()`-top dispatch reads the `prev_*`
754    /// copies — i.e. second-to-last-cycle recognition. The I-mask uses the
755    /// start-of-instruction snapshot (`irq_sample_i_flag`), not live `self.p`,
756    /// so CLI/SEI/PLP delay their I-change one instruction.
757    #[allow(clippy::needless_pass_by_ref_mut)] // &mut B for signature parity
758    fn handle_interrupts<B: Bus>(&mut self, bus: &mut B) {
759        // EXPERIMENT (v2.6.7): the taken-branch poll rule applies to NMI too.
760        //
761        // `CPU_interrupts.xhtml`: "Interrupts are always polled before the
762        // second CPU cycle (the operand fetch), but NOT before the third CPU
763        // cycle on a taken branch." It says *interrupts*, not *IRQs*.
764        //
765        // The edge LATCH (`mc_need_nmi`, below) keeps running every cycle --
766        // that is hardware's always-on edge detector and must not change. What
767        // freezes is the DISPATCH gate, exactly as `mc_run_irq` already does
768        // while `skip_irq_sample` is set.
769        if !(self.skip_irq_sample && self.skip_irq_sample_q) {
770            self.mc_prev_need_nmi = self.mc_need_nmi;
771        }
772        self.skip_irq_sample_q = self.skip_irq_sample;
773        let nmi_level = bus.nmi_level();
774        if !self.mc_prev_nmi_line && nmi_level {
775            self.mc_need_nmi = true;
776        }
777        self.mc_prev_nmi_line = nmi_level;
778        self.mc_prev_run_irq = self.mc_run_irq;
779        // W1 (`mc-r1-branch-poll-points`): a taken branch polls IRQ ONCE —
780        // before C2 — so while `skip_irq_sample` is set (the branch dispatch
781        // arms set it after the C1 opcode fetch, before the C2 operand fetch)
782        // the recognition latch is FROZEN at its end-of-C1 value instead of
783        // recomputed from the live line every cycle. DMC-DMA halt cycles
784        // drained inside the branch's own `read1`/`idle_tick` therefore
785        // cannot make a freshly-asserted IRQ visible to THIS instruction
786        // (AccuracyCoin `Interrupt flag latency` Test A; TriCNES polls at
787        // C2-start only, plus a can-set poll at C4-start handled in
788        // `branch()`). The `mc_prev_run_irq` copy above still runs, so the
789        // held end-of-C1 value is what the next `step()` dispatch reads. NMI
790        // edge detection above is untouched (sampled every cycle, per
791        // hardware — the quirk is IRQ-only).
792        if self.skip_irq_sample {
793            return;
794        }
795        let irq_level = bus.irq_level();
796        self.mc_run_irq = irq_level && !self.irq_sample_i_flag;
797    }
798
799    /// Tick the bus once and sample interrupt lines, *without* a bus
800    /// access.  Models a 6502 internal cycle.
801    fn idle_tick<B: Bus>(&mut self, bus: &mut B) {
802        {
803            // F-2 re-coupling (`mc-r1-dmc-idle-halt`): a DMC DMA can halt the CPU
804            // on a 6502 INTERNAL cycle too — on hardware every cycle is a bus
805            // read, and Mesen's `ProcessPendingDma` runs on every `MemoryRead`
806            // (incl. dummy reads), NOT only instruction/operand reads. R1's
807            // `read1` loop only services on real reads, so the DMA waits through
808            // internal cycles (the `lat=4` idle-runs the per-fetch trace pinned
809            // as the period-jitter source). Service it here too, on the held
810            // (last-read) bus address. Default-off; the banked 6/10 path skips it.
811            // W3-Stage-1 (`mc-r1-dma-unified`): the unified-engine replacement
812            // for the idle DMC drain above — same loop shape, ONE engine. The
813            // bus supplies the held (last-read) address for the parked 6502
814            // address bus. Same budget accounting as the loop it replaces.
815            while bus.unified_dma_pending() {
816                self.cycles_emitted = self.cycles_emitted.saturating_add(1);
817                self.start_cycle(bus, true);
818                bus.unified_dma_cycle_idle();
819                self.end_cycle(bus, true);
820            }
821            // R1: a pure internal cycle — busless (idle_tick stays busless).
822            self.cycles_emitted = self.cycles_emitted.saturating_add(1);
823            self.start_cycle(bus, true);
824            self.end_cycle(bus, true);
825        }
826    }
827
828    /// Canonical cycle-2 PC dummy read for implied / accumulator /
829    /// transfer / flag instructions (per nesdev `6502_cpu.txt` + MOS
830    /// 6502 datasheet). Real silicon fetches the byte AFTER the opcode
831    /// during cycle 2 of these single-byte instructions and discards
832    /// it (the would-be operand). Without this dummy read, our emulator
833    /// instead "burns an idle cycle" via `idle_tick` for the second
834    /// cycle, which counts the cycle for time but produces no bus
835    /// access — diverging from real silicon's bus-access pattern.
836    ///
837    /// Wired into 22 dispatch arms (ASL/LSR/ROL/ROR A; CLC/SEC/CLI/SEI/
838    /// CLV/CLD/SED; TAX/TAY/TSX/TXA/TXS/TYA; INX/DEX/INY/DEY; NOP;
839    /// 6 unofficial 1-byte NOPs), **unconditionally**.
840    ///
841    /// This was gated behind a `cpu-implied-dummy-reads` cargo feature,
842    /// default-off pending the DMC-scheduler audit in
843    /// `docs/audit/sprint-2.3-implied-dummy-dmc-recon-2026-05-25.md`. That
844    /// gate no longer exists: the flag is declared in no manifest and there
845    /// is no `cfg` on this helper, so every build performs the dummy read.
846    /// The text describing an OFF branch that "compiles to a no-op" survived
847    /// the promotion and is removed here — it described the shipped default
848    /// as disabled when it is unconditional, which is the most misleading
849    /// shape a stale comment can take.
850    // Only `inline_always` still binds: `needless_pass_by_ref_mut`,
851    // `unused_self` and `missing_const_for_fn` were for the deleted OFF
852    // branch, and stripping all four re-lints with `inline_always` as the
853    // sole finding. Measured rather than reasoned, per the v2.3.9 sweep that
854    // found 25 of 29 `allow`s suppressing nothing.
855    #[inline(always)]
856    #[allow(clippy::inline_always)]
857    fn implied_dummy_read<B: Bus>(&mut self, bus: &mut B) {
858        let _ = self.read1(bus, self.pc);
859    }
860
861    /// Read a byte at `addr` *and* consume one CPU cycle (with bus tick
862    /// + interrupt sampling).
863    #[allow(clippy::too_many_lines)] // mc-r1 DMA-interleave arms push this past 100
864    fn read1<B: Bus>(&mut self, bus: &mut B, addr: u16) -> u8 {
865        {
866            // accuracycoin-100 Phase 2 (`mc-r1-dmc-abort-cancel`): a 1-byte
867            // non-looping implicit abort that matured during the prior APU tick
868            // is serviced HERE, before the (cancelled) reload could run. On a
869            // GET (read) cycle the abort is a 1-cycle DMA (one halt re-read) →
870            // CalculateDMADuration Y=1; on a PUT cycle it does NOT occur → Y=0
871            // ("the 1-cycle abort will not land on a write cycle"). Both clear
872            // the pending reload so the `dmc_dma_pending` loop below skips it.
873            if bus.dmc_abort_pending() {
874                if bus.dmc_abort_is_get_cycle() {
875                    self.cycles_emitted = self.cycles_emitted.saturating_add(1);
876                    self.start_cycle(bus, true);
877                    bus.dmc_abort_halt_step(addr);
878                    self.end_cycle(bus, true);
879                } else {
880                    bus.dmc_abort_cancel();
881                }
882            }
883            // W3-Stage-1 (`mc-r1-dma-unified`): the ONE DMA loop. It replaced
884            // three earlier loops (the standalone DMC drain, the sequential
885            // Stage-D OAM loop and the Program-M overlap loop), whose bus hooks
886            // were deprecated at v2.7.5 and removed at v2.9.8 (ADR 0042). Each
887            // iteration is one full R1 cycle (start_cycle -> the bus's
888            // unified TriCNES-dispatch cycle -> end_cycle), so every DMA
889            // cycle keeps the φ2 IRQ sample — the C1-safe shape. The
890            // engine itself (ONE driver for standalone DMC, standalone OAM,
891            // and the overlap) lives bus-side in `unified_dma_cycle`; the
892            // load-get-entry defer is folded into `unified_dma_pending`
893            // (pre-cycle, like the floor's while-gate) AND the engine's
894            // in-cycle entry gate (post-flip parity). A DMC DMA halts the CPU
895            // only on a READ cycle, which is why the loop lives here and in
896            // `idle_tick`, never in `write1`.
897            while bus.unified_dma_pending() {
898                // DMA halt cycles count against `cycles_emitted`.
899                self.cycles_emitted = self.cycles_emitted.saturating_add(1);
900                self.start_cycle(bus, true);
901                bus.unified_dma_cycle(addr);
902                self.end_cycle(bus, true);
903            }
904            // R1 clean access shape: start_cycle (PPU caught up to the
905            // access's exact mc + bus cpu_clock) → bus.read → end_cycle
906            // (double catch-up + φ2 interrupt sample). Mesen `MemoryRead`.
907            self.cycles_emitted = self.cycles_emitted.saturating_add(1);
908            self.start_cycle(bus, true);
909            let v = bus.read(addr);
910            self.end_cycle(bus, true);
911            v
912        }
913    }
914
915    /// Write `value` to `addr` *and* consume one CPU cycle (with bus
916    /// tick + interrupt sampling).
917    fn write1<B: Bus>(&mut self, bus: &mut B, addr: u16, value: u8) {
918        {
919            // accuracycoin-100 Phase 2: a CPU write cycle cannot be RDY-halted,
920            // so a 1-byte implicit abort matured before a write does NOT occur
921            // (Y=0). Cancel it with no halt cycle — this is the "will not land on
922            // a write cycle" half of the sweep that the read-only `read1` path
923            // can't reach.
924            if bus.dmc_abort_pending() {
925                bus.dmc_abort_cancel();
926            }
927            // R1 clean write shape (symmetric split — writes commit 2 mc
928            // later than reads). No interrupt sample latches here; end_cycle's
929            // handle_interrupts does the φ2 sample.
930            self.cycles_emitted = self.cycles_emitted.saturating_add(1);
931            self.start_cycle(bus, false);
932            bus.write(addr, value);
933            self.end_cycle(bus, false);
934        }
935    }
936
937    /// SH* unstable-store family helper (`SHA / SHX / SHY / SHS / TAS`,
938    /// opcodes `$9F / $93 / $9E / $9C / $9B`).
939    ///
940    /// Implements the 6502 unstable-store (SH*) algorithm — the
941    /// "unstable"/"highbyte" store opcodes (`value AND (high_byte + 1)`, with the
942    /// RDY/DMA quirk), pinned bit-for-bit by `AccuracyCoin`'s "Unofficial
943    /// Instructions: SH*" sub-test.
944    ///
945    /// Provenance: **derived from Mesen2's `SyaSxaAxa`** (`Core/NES/NesCpu.h`),
946    /// `GPL-3.0-or-later`. The `NESdev` community documents this behavior, but this
947    /// implementation was ported from Mesen2's — not written independently from
948    /// the documentation. The surrounding DMC-DMA interruption detection uses the
949    /// emulator's own bus cycle-count machinery. See NOTICE and
950    /// docs/originality-and-provenance.md (Section 1).
951    /// The algorithm:
952    ///
953    /// 1. Compute the page-crossed flag against `base + index_reg`.
954    /// 2. Perform a dummy read at the **unfixed** address
955    ///    (`base + index_reg - 0x100` if page-crossed, else
956    ///    `base + index_reg`).  This is the cycle DMC DMA can
957    ///    interrupt.
958    /// 3. Detect DMC-DMA interruption via `bus.cycle_count()`
959    ///    before/after the dummy read — if more than 1 bus cycle
960    ///    elapsed, a DMA fired.
961    /// 4. On page-cross, the address-high-byte is corrupted to
962    ///    `original_addr_high AND value_reg`.
963    /// 5. Compute the store value:
964    ///    - With DMA: just `value_reg` (the H+1 AND is suppressed
965    ///      because the DMC pulled the bus low).
966    ///    - Without DMA: `value_reg AND ((base >> 8) + 1)`.
967    /// 6. Write to the (possibly corrupted) final address.
968    ///
969    /// This shape is what `AccuracyCoin Unofficial Instructions: SH*`
970    /// sub-test 7 ("the cycle before the write had a DMA") brackets.
971    /// Pre-2026-05-23 `RustyNES` skipped the dummy read entirely and
972    /// always wrote `value_reg & (H+1)`, failing sub-test 7 across
973    /// all 5 SH* opcodes (error code 7).
974    fn sh_store<B: Bus>(&mut self, bus: &mut B, base: u16, index_reg: u8, value_reg: u8) {
975        let addr = base.wrapping_add(u16::from(index_reg));
976        let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
977
978        // Dummy read at the unfixed address (this is the cycle DMC
979        // DMA can halt).  We sample the bus-side cycle count before
980        // and after to detect interruption — DMC DMA service path
981        // advances `bus.cycle` by 3+ extra ticks while the CPU's
982        // own `Cpu::cycles` (which `idle_tick` increments) only goes
983        // up by 1 for the read itself.
984        let cyc_before = bus.cycle_count();
985        let dummy_addr = if page_crossed {
986            addr.wrapping_sub(0x100)
987        } else {
988            addr
989        };
990        let _dummy = self.read1(bus, dummy_addr);
991        let had_dma = bus.cycle_count().wrapping_sub(cyc_before) > 1;
992
993        let addr_high = (addr >> 8) as u8;
994        let addr_low = (addr & 0xFF) as u8;
995        let final_high = if page_crossed {
996            addr_high & value_reg
997        } else {
998            addr_high
999        };
1000
1001        let write_value = if had_dma {
1002            // DMC DMA interrupted the dummy read — bus latch was
1003            // overwritten by the DMC fetch, so the store value loses
1004            // its AND-with-(H+1) component.  Per Mesen2 `SyaSxaAxa`.
1005            value_reg
1006        } else {
1007            // Canonical "documented behavior 1" path: AND with
1008            // (base_high + 1).
1009            value_reg & ((base >> 8) as u8).wrapping_add(1)
1010        };
1011
1012        let final_addr = (u16::from(final_high) << 8) | u16::from(addr_low);
1013        self.write1(bus, final_addr, write_value);
1014    }
1015
1016    fn service_interrupt<B: Bus>(&mut self, bus: &mut B, vector: u16, brk: bool) {
1017        // Per-cycle interrupt sequence (7 cycles total when entered from
1018        // an interrupt edge, 6 from BRK because its opcode fetch already
1019        // burned cycle 1):
1020        //   C1: opcode fetch (BRK only — IRQ/NMI skip this and instead
1021        //       perform an extra dummy read in C2).
1022        //   C2: dummy read of PC+1 (BRK)  /  filler/internal read (IRQ/NMI).
1023        //   C3-C5: push PCH, PCL, P.
1024        //   C6-C7: read vector lo, hi.
1025        // `cycles_emitted` is reset here so the caller's accounting starts
1026        // from this routine's first tick (the opcode-fetch tick from a BRK
1027        // is harmless — the BRK arm sets *cycles = 0 to suppress the
1028        // trailing burn loop).
1029        self.cycles_emitted = 0;
1030        // Reset the per-instruction interrupt sample latches so any NMI
1031        // edge during the push sequence below is captured here.
1032        self.nmi_first_tick = u8::MAX;
1033        self.irq_first_tick = u8::MAX;
1034        // Two filler reads for IRQ/NMI; one for BRK (the opcode fetch
1035        // counted as the other).
1036        //
1037        // W3-Stage-3 Part B (`mc-r1-brk-padding-read`): BRK's C2 is the
1038        // canonical PADDING-BYTE read at PC+1 — a REAL bus access on
1039        // silicon, not an internal cycle. AccuracyCoin `Implied Dummy
1040        // Reads` error 31 brackets exactly this: the test choreographs a
1041        // BRK whose padding read lands on `$4015`, which must clear the
1042        // frame-counter IRQ flag (RTI/RTS already emit their canonical
1043        // reads under `cpu-stack-dummy-reads`; BRK was the one gap — with
1044        // it the whole test PASSES, one sub-check beyond Mesen2's error
1045        // 34). The dispatch arm has already advanced PC past the padding
1046        // byte, so it sits at `pc - 1`.
1047        if brk {
1048            let _ = self.read1(bus, self.pc.wrapping_sub(1));
1049        } else {
1050            // v2.0.0 beta.2 (A2 every-cycle-bus-access, promoted to the only
1051            // path in beta.4): canonical hardware IRQ/NMI cycles 1-2 are
1052            // DUMMY READS of the interrupted PC (the suppressed opcode fetch
1053            // + suppressed operand fetch — nesdev `6502_cpu.txt`; Mesen2
1054            // `NesCpu::IRQ` issues two `DummyRead`s). This is the C1-trio
1055            // canary path: the conversion keeps the exact same two-cycle
1056            // start/end structure (φ2 samples unchanged) and only adds the
1057            // bus access + held-address update; the cpu_interrupts_v2 5/5
1058            // strict gate + AccuracyCoin 139/139 hold (verified at the
1059            // beta.2 gate).
1060            let _ = self.read1(bus, self.pc);
1061            let _ = self.read1(bus, self.pc);
1062        }
1063        self.push_u16(bus, self.pc);
1064        let mut p = self.p | Status::UNUSED;
1065        if brk {
1066            p.insert(Status::BREAK);
1067        } else {
1068            p.remove(Status::BREAK);
1069        }
1070        self.push(bus, p.bits());
1071        self.p.insert(Status::INTERRUPT_DISABLE);
1072        // NMI hijacking: real 6502 latches the vector to read on the
1073        // CYCLE just before the vector reads; if NMI is asserted at that
1074        // point, BRK / IRQ both read $FFFA / $FFFB instead of the
1075        // declared vector.  We approximate "NMI asserted by now" with
1076        // "the NMI sample latch was hit during cycles 1..=5 of this
1077        // sequence."
1078        // R1: the hijack reads the DELAYED `mc_prev_need_nmi`, NOT the live
1079        // `mc_need_nmi` — an NMI edge latched ON the P-push cycle's φ2 sampler
1080        // must NOT hijack (the BRK/IRQ completes to its own vector and the NMI
1081        // is taken after one handler instruction). `mc_prev_need_nmi` is the
1082        // pre-edge value: 1 only if the NMI was pending BEFORE this cycle
1083        // (oracle-derived, cpu_interrupts_v2/2). Legacy uses `nmi_first_tick`.
1084        let effective_vector = if self.mc_prev_need_nmi && vector != NMI_VECTOR {
1085            self.mc_need_nmi = false;
1086            self.mc_prev_need_nmi = false;
1087            NMI_VECTOR
1088        } else {
1089            vector
1090        };
1091        // Phase 1.2 of Track C1 attempt 14: notify the bus of the vector
1092        // fetch BEFORE the low-byte read so the trace records the cycle
1093        // at which the CPU enters its vector-fetch micro-op (C6 of the
1094        // 7-cycle service sequence).  `is_nmi` distinguishes a clean NMI
1095        // service entry from an IRQ/BRK service entry that an NMI edge
1096        // has hijacked to `$FFFA` — both fetch from `$FFFA` but only one
1097        // has `vector == NMI_VECTOR` at this call site.
1098        bus.notify_irq_service(effective_vector, vector == NMI_VECTOR);
1099        let lo = self.read1(bus, effective_vector);
1100        let hi = self.read1(bus, effective_vector + 1);
1101        self.pc = u16::from(lo) | (u16::from(hi) << 8);
1102    }
1103
1104    // ------------------------------------------------------------------
1105    // Addressing-mode resolvers. Each returns the effective address plus a
1106    // page-crossed flag; the caller decides whether to add a cycle.
1107    // ------------------------------------------------------------------
1108
1109    fn addr_zp<B: Bus>(&mut self, bus: &mut B) -> Operand {
1110        Operand {
1111            addr: u16::from(self.fetch_pc(bus)),
1112            page_crossed: false,
1113        }
1114    }
1115
1116    fn addr_zp_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1117        let base = self.fetch_pc(bus);
1118        // v2.0.0 beta.2 (A2 every-cycle-bus-access): canonical 6502 cycle 3
1119        // reads the UN-indexed zero-page address while the index add
1120        // completes, then discards it (nesdev `6502_cpu.txt`; Mesen2 models
1121        // it as a real `MemoryRead`). Zero-page addresses are always RAM
1122        // ($0000-$00FF), so the read is register-side-effect-free — but it
1123        // parks a real address on the bus (the held address a DMA halt
1124        // re-reads) instead of leaving the cycle busless in the burn-loop.
1125        // The burn-probe histogram pinned this family as 99% of the
1126        // remaining busless surface ($95 STA zp,X alone = 8,955 of 9,795
1127        // burned cycles over the AccuracyCoin battery). Promoted to the only
1128        // path in v2.0.0 beta.4.
1129        let _ = self.read1(bus, u16::from(base));
1130        Operand {
1131            addr: u16::from(base.wrapping_add(self.x)),
1132            page_crossed: false,
1133        }
1134    }
1135
1136    fn addr_zp_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1137        let base = self.fetch_pc(bus);
1138        // A2: same canonical un-indexed dummy read as `addr_zp_x` (cycle 3
1139        // of LDX/STX zp,Y and the unofficial LAX/SAX zp,Y arms).
1140        let _ = self.read1(bus, u16::from(base));
1141        Operand {
1142            addr: u16::from(base.wrapping_add(self.y)),
1143            page_crossed: false,
1144        }
1145    }
1146
1147    fn addr_abs<B: Bus>(&mut self, bus: &mut B) -> Operand {
1148        Operand {
1149            addr: self.fetch_pc_u16(bus),
1150            page_crossed: false,
1151        }
1152    }
1153
1154    fn addr_abs_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1155        let base = self.fetch_pc_u16(bus);
1156        let addr = base.wrapping_add(u16::from(self.x));
1157        let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1158        if page_crossed {
1159            // Canonical 6502 page-cross dummy read at the unfixed
1160            // address: (base_hi << 8) | ((base_lo + X) & 0xFF). The
1161            // high byte hasn't been incremented yet. This read has
1162            // side effects on PPU registers (`$2002` clears VBlank,
1163            // `$2007` advances the buffer) and is the hardware oracle
1164            // AccuracyCoin's `CPU Behavior :: Dummy read cycles`
1165            // Test 1 brackets via `LDA $20F2, X` with X=$10 reading
1166            // $2002 through the mirror.
1167            let dummy = (base & 0xFF00) | (addr & 0x00FF);
1168            let _ = self.read1(bus, dummy);
1169        }
1170        Operand { addr, page_crossed }
1171    }
1172
1173    fn addr_abs_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1174        let base = self.fetch_pc_u16(bus);
1175        let addr = base.wrapping_add(u16::from(self.y));
1176        let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1177        if page_crossed {
1178            // See addr_abs_x for the page-cross dummy-read rationale.
1179            let dummy = (base & 0xFF00) | (addr & 0x00FF);
1180            let _ = self.read1(bus, dummy);
1181        }
1182        Operand { addr, page_crossed }
1183    }
1184
1185    // ABS,X / ABS,Y operands for read-modify-write opcodes (ASL, LSR, ROL,
1186    // ROR, INC, DEC, and the unofficial SLO/RLA/SRE/RRA/DCP/ISC). Canonical
1187    // 6502: the unfixed-address dummy read happens UNCONDITIONALLY at
1188    // cycle 4 (not just on page cross) because the CPU has 7 cycles to
1189    // fill and cannot know the fixed address until the high-byte add
1190    // completes. Reads with side effects (`$2002` clears VBlank, `$4015`
1191    // clears frame-IRQ, `$2007` advances buffer) therefore fire twice on
1192    // RMW ABS,X. Bracketed by AccuracyCoin's `Implied Dummy Reads`
1193    // test 2: `SLO $4015,X` with X=0 expects the dummy read to clear the
1194    // frame-IRQ flag so the subsequent real read returns 0.
1195    fn addr_abs_x_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1196        let base = self.fetch_pc_u16(bus);
1197        let addr = base.wrapping_add(u16::from(self.x));
1198        let dummy = (base & 0xFF00) | (addr & 0x00FF);
1199        let _ = self.read1(bus, dummy);
1200        addr
1201    }
1202
1203    fn addr_abs_y_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1204        let base = self.fetch_pc_u16(bus);
1205        let addr = base.wrapping_add(u16::from(self.y));
1206        let dummy = (base & 0xFF00) | (addr & 0x00FF);
1207        let _ = self.read1(bus, dummy);
1208        addr
1209    }
1210
1211    /// (zp),Y operand for the unofficial read-modify-write opcodes
1212    /// (SLO/RLA/SRE/RRA/DCP/ISB `(zp),Y` — `$13/$33/$53/$73/$D3/$F3`).
1213    ///
1214    /// v2.0.0 beta.2 (A2 every-cycle-bus-access, promoted to the only path
1215    /// in beta.4): canonical 6502 8-cycle (zp),Y RMW performs the
1216    /// unfixed-address dummy read UNCONDITIONALLY at cycle 5 (like RMW
1217    /// ABS,X/Y above — the CPU cannot know the fixed address until the
1218    /// high-byte add completes), not only on page cross. The burn-probe
1219    /// histogram pinned these six arms as the last instruction-dispatch
1220    /// busless cycles (25 of the original 9,795).
1221    fn addr_ind_y_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1222        // Delegate to the plain resolver (which already emits the
1223        // unfixed-address dummy read on a page cross), then emit the
1224        // RMW's unconditional cycle-5 dummy for the non-crossing case.
1225        // On a non-crossing access the canonical unfixed address
1226        // `(base & 0xFF00) | (addr & 0xFF)` EQUALS the final address
1227        // (the high byte needed no fix-up), so reading `o.addr` here is
1228        // the silicon-exact target — do not "fix" this to a separate
1229        // unfixed computation, they are identical by construction.
1230        let o = self.addr_ind_y(bus);
1231        if !o.page_crossed {
1232            let _ = self.read1(bus, o.addr);
1233        }
1234        o.addr
1235    }
1236
1237    fn addr_ind_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1238        let base = self.fetch_pc(bus);
1239        // A2: canonical (zp,X) cycle 3 — dummy read of the UN-indexed
1240        // pointer address while the X add completes (same silicon behavior
1241        // as `addr_zp_x`; zero-page, so register-side-effect-free).
1242        let _ = self.read1(bus, u16::from(base));
1243        let ptr = base.wrapping_add(self.x);
1244        let lo = self.read1(bus, u16::from(ptr));
1245        let hi = self.read1(bus, u16::from(ptr.wrapping_add(1)));
1246        Operand {
1247            addr: u16::from(lo) | (u16::from(hi) << 8),
1248            page_crossed: false,
1249        }
1250    }
1251
1252    fn addr_ind_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1253        let ptr = self.fetch_pc(bus);
1254        let lo = self.read1(bus, u16::from(ptr));
1255        let hi = self.read1(bus, u16::from(ptr.wrapping_add(1)));
1256        let base = u16::from(lo) | (u16::from(hi) << 8);
1257        let addr = base.wrapping_add(u16::from(self.y));
1258        let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1259        if page_crossed {
1260            // Page-cross dummy read at the unfixed address — same as
1261            // addr_abs_x/y. Canonical 6502 behavior for LDA (zp),Y on
1262            // page crossing.
1263            let dummy = (base & 0xFF00) | (addr & 0x00FF);
1264            let _ = self.read1(bus, dummy);
1265        }
1266        Operand { addr, page_crossed }
1267    }
1268
1269    // ------------------------------------------------------------------
1270    // Top-level dispatch.
1271    //
1272    // The 256-way match is the cleanest way to express the entire opcode
1273    // table; the doc-comments are intentionally absent at the arm level
1274    // because each one is a single line of the standard 6502 reference and
1275    // adding individual arm comments would overwhelm the readability of the
1276    // table.
1277    // ------------------------------------------------------------------
1278
1279    #[allow(
1280        clippy::cognitive_complexity,
1281        clippy::too_many_lines,
1282        clippy::match_same_arms
1283    )]
1284    fn dispatch<B: Bus>(&mut self, bus: &mut B, op: u8, cycles: &mut u8) {
1285        match op {
1286            // === Loads ===
1287            0xA9 => {
1288                let v = self.fetch_pc(bus);
1289                self.lda(v);
1290                *cycles = 2;
1291            }
1292            0xA5 => {
1293                let o = self.addr_zp(bus);
1294                self.lda_addr(bus, o.addr);
1295                *cycles = 3;
1296            }
1297            0xB5 => {
1298                let o = self.addr_zp_x(bus);
1299                self.lda_addr(bus, o.addr);
1300                *cycles = 4;
1301            }
1302            0xAD => {
1303                let o = self.addr_abs(bus);
1304                self.lda_addr(bus, o.addr);
1305                *cycles = 4;
1306            }
1307            0xBD => {
1308                let o = self.addr_abs_x(bus);
1309                self.lda_addr(bus, o.addr);
1310                *cycles = 4 + u8::from(o.page_crossed);
1311            }
1312            0xB9 => {
1313                let o = self.addr_abs_y(bus);
1314                self.lda_addr(bus, o.addr);
1315                *cycles = 4 + u8::from(o.page_crossed);
1316            }
1317            0xA1 => {
1318                let o = self.addr_ind_x(bus);
1319                self.lda_addr(bus, o.addr);
1320                *cycles = 6;
1321            }
1322            0xB1 => {
1323                let o = self.addr_ind_y(bus);
1324                self.lda_addr(bus, o.addr);
1325                *cycles = 5 + u8::from(o.page_crossed);
1326            }
1327
1328            0xA2 => {
1329                let v = self.fetch_pc(bus);
1330                self.ldx(v);
1331                *cycles = 2;
1332            }
1333            0xA6 => {
1334                let o = self.addr_zp(bus);
1335                let v = self.read1(bus, o.addr);
1336                self.ldx(v);
1337                *cycles = 3;
1338            }
1339            0xB6 => {
1340                let o = self.addr_zp_y(bus);
1341                let v = self.read1(bus, o.addr);
1342                self.ldx(v);
1343                *cycles = 4;
1344            }
1345            0xAE => {
1346                let o = self.addr_abs(bus);
1347                let v = self.read1(bus, o.addr);
1348                self.ldx(v);
1349                *cycles = 4;
1350            }
1351            0xBE => {
1352                let o = self.addr_abs_y(bus);
1353                let v = self.read1(bus, o.addr);
1354                self.ldx(v);
1355                *cycles = 4 + u8::from(o.page_crossed);
1356            }
1357
1358            0xA0 => {
1359                let v = self.fetch_pc(bus);
1360                self.ldy(v);
1361                *cycles = 2;
1362            }
1363            0xA4 => {
1364                let o = self.addr_zp(bus);
1365                let v = self.read1(bus, o.addr);
1366                self.ldy(v);
1367                *cycles = 3;
1368            }
1369            0xB4 => {
1370                let o = self.addr_zp_x(bus);
1371                let v = self.read1(bus, o.addr);
1372                self.ldy(v);
1373                *cycles = 4;
1374            }
1375            0xAC => {
1376                let o = self.addr_abs(bus);
1377                let v = self.read1(bus, o.addr);
1378                self.ldy(v);
1379                *cycles = 4;
1380            }
1381            0xBC => {
1382                let o = self.addr_abs_x(bus);
1383                let v = self.read1(bus, o.addr);
1384                self.ldy(v);
1385                *cycles = 4 + u8::from(o.page_crossed);
1386            }
1387
1388            // === Stores ===
1389            0x85 => {
1390                let o = self.addr_zp(bus);
1391                self.write1(bus, o.addr, self.a);
1392                *cycles = 3;
1393            }
1394            0x95 => {
1395                let o = self.addr_zp_x(bus);
1396                self.write1(bus, o.addr, self.a);
1397                *cycles = 4;
1398            }
1399            0x8D => {
1400                let o = self.addr_abs(bus);
1401                self.write1(bus, o.addr, self.a);
1402                *cycles = 4;
1403            }
1404            0x9D => {
1405                let o = self.addr_abs_x(bus);
1406                // Canonical 6502: STA absolute,X performs a dummy
1407                // read at cycle 4 even when no page is crossed (unlike
1408                // LDA where cycle 4 is the real read). `addr_abs_x`
1409                // already issues the dummy read at the unfixed address
1410                // when page-crossed; for the no-page-cross case we add
1411                // it here at the final address.
1412                if !o.page_crossed {
1413                    let _ = self.read1(bus, o.addr);
1414                }
1415                self.write1(bus, o.addr, self.a);
1416                *cycles = 5;
1417            }
1418            0x99 => {
1419                let o = self.addr_abs_y(bus);
1420                if !o.page_crossed {
1421                    let _ = self.read1(bus, o.addr);
1422                }
1423                self.write1(bus, o.addr, self.a);
1424                *cycles = 5;
1425            }
1426            0x81 => {
1427                let o = self.addr_ind_x(bus);
1428                self.write1(bus, o.addr, self.a);
1429                *cycles = 6;
1430            }
1431            0x91 => {
1432                let o = self.addr_ind_y(bus);
1433                // Canonical STA (zp),Y always dummy-reads at cycle 5
1434                // even when no page is crossed. `addr_ind_y` already
1435                // handles the page-cross dummy at the unfixed address;
1436                // add the no-page-cross dummy here at the final address.
1437                if !o.page_crossed {
1438                    let _ = self.read1(bus, o.addr);
1439                }
1440                self.write1(bus, o.addr, self.a);
1441                *cycles = 6;
1442            }
1443
1444            0x86 => {
1445                let o = self.addr_zp(bus);
1446                self.write1(bus, o.addr, self.x);
1447                *cycles = 3;
1448            }
1449            0x96 => {
1450                let o = self.addr_zp_y(bus);
1451                self.write1(bus, o.addr, self.x);
1452                *cycles = 4;
1453            }
1454            0x8E => {
1455                let o = self.addr_abs(bus);
1456                self.write1(bus, o.addr, self.x);
1457                *cycles = 4;
1458            }
1459
1460            0x84 => {
1461                let o = self.addr_zp(bus);
1462                self.write1(bus, o.addr, self.y);
1463                *cycles = 3;
1464            }
1465            0x94 => {
1466                let o = self.addr_zp_x(bus);
1467                self.write1(bus, o.addr, self.y);
1468                *cycles = 4;
1469            }
1470            0x8C => {
1471                let o = self.addr_abs(bus);
1472                self.write1(bus, o.addr, self.y);
1473                *cycles = 4;
1474            }
1475
1476            // === Transfers ===
1477            0xAA => {
1478                self.implied_dummy_read(bus);
1479                self.x = self.a;
1480                self.p.set_nz(self.x);
1481                *cycles = 2;
1482            }
1483            0xA8 => {
1484                self.implied_dummy_read(bus);
1485                self.y = self.a;
1486                self.p.set_nz(self.y);
1487                *cycles = 2;
1488            }
1489            0xBA => {
1490                self.implied_dummy_read(bus);
1491                self.x = self.s;
1492                self.p.set_nz(self.x);
1493                *cycles = 2;
1494            }
1495            0x8A => {
1496                self.implied_dummy_read(bus);
1497                self.a = self.x;
1498                self.p.set_nz(self.a);
1499                *cycles = 2;
1500            }
1501            0x9A => {
1502                self.implied_dummy_read(bus);
1503                self.s = self.x;
1504                *cycles = 2;
1505            }
1506            0x98 => {
1507                self.implied_dummy_read(bus);
1508                self.a = self.y;
1509                self.p.set_nz(self.a);
1510                *cycles = 2;
1511            }
1512
1513            // === Stack ===
1514            0x48 => {
1515                // PHA: C2 dummy read PC (the 6502 always reads the next byte on
1516                // the second cycle of a stack push), then the push.
1517                let _ = self.read1(bus, self.pc);
1518                self.push(bus, self.a);
1519                *cycles = 3;
1520            }
1521            0x08 => {
1522                let _ = self.read1(bus, self.pc);
1523                self.push(bus, (self.p | Status::BREAK | Status::UNUSED).bits());
1524                *cycles = 3;
1525            }
1526            0x68 => {
1527                // PLA: C2 dummy read PC, C3 dummy stack read (pre-increment),
1528                // then the pull.
1529                {
1530                    let _ = self.read1(bus, self.pc);
1531                    let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
1532                }
1533                self.a = self.pull(bus);
1534                self.p.set_nz(self.a);
1535                *cycles = 4;
1536            }
1537            0x28 => {
1538                {
1539                    let _ = self.read1(bus, self.pc);
1540                    let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
1541                }
1542                let v = self.pull(bus);
1543                let mut new_p = Status::from_bits_truncate(v);
1544                new_p.remove(Status::BREAK);
1545                new_p.insert(Status::UNUSED);
1546                self.p = new_p;
1547                *cycles = 4;
1548            }
1549
1550            // === Logical ===
1551            0x29 => {
1552                let v = self.fetch_pc(bus);
1553                self.and(v);
1554                *cycles = 2;
1555            }
1556            0x25 => {
1557                let o = self.addr_zp(bus);
1558                let v = self.read1(bus, o.addr);
1559                self.and(v);
1560                *cycles = 3;
1561            }
1562            0x35 => {
1563                let o = self.addr_zp_x(bus);
1564                let v = self.read1(bus, o.addr);
1565                self.and(v);
1566                *cycles = 4;
1567            }
1568            0x2D => {
1569                let o = self.addr_abs(bus);
1570                let v = self.read1(bus, o.addr);
1571                self.and(v);
1572                *cycles = 4;
1573            }
1574            0x3D => {
1575                let o = self.addr_abs_x(bus);
1576                let v = self.read1(bus, o.addr);
1577                self.and(v);
1578                *cycles = 4 + u8::from(o.page_crossed);
1579            }
1580            0x39 => {
1581                let o = self.addr_abs_y(bus);
1582                let v = self.read1(bus, o.addr);
1583                self.and(v);
1584                *cycles = 4 + u8::from(o.page_crossed);
1585            }
1586            0x21 => {
1587                let o = self.addr_ind_x(bus);
1588                let v = self.read1(bus, o.addr);
1589                self.and(v);
1590                *cycles = 6;
1591            }
1592            0x31 => {
1593                let o = self.addr_ind_y(bus);
1594                let v = self.read1(bus, o.addr);
1595                self.and(v);
1596                *cycles = 5 + u8::from(o.page_crossed);
1597            }
1598
1599            0x09 => {
1600                let v = self.fetch_pc(bus);
1601                self.ora(v);
1602                *cycles = 2;
1603            }
1604            0x05 => {
1605                let o = self.addr_zp(bus);
1606                let v = self.read1(bus, o.addr);
1607                self.ora(v);
1608                *cycles = 3;
1609            }
1610            0x15 => {
1611                let o = self.addr_zp_x(bus);
1612                let v = self.read1(bus, o.addr);
1613                self.ora(v);
1614                *cycles = 4;
1615            }
1616            0x0D => {
1617                let o = self.addr_abs(bus);
1618                let v = self.read1(bus, o.addr);
1619                self.ora(v);
1620                *cycles = 4;
1621            }
1622            0x1D => {
1623                let o = self.addr_abs_x(bus);
1624                let v = self.read1(bus, o.addr);
1625                self.ora(v);
1626                *cycles = 4 + u8::from(o.page_crossed);
1627            }
1628            0x19 => {
1629                let o = self.addr_abs_y(bus);
1630                let v = self.read1(bus, o.addr);
1631                self.ora(v);
1632                *cycles = 4 + u8::from(o.page_crossed);
1633            }
1634            0x01 => {
1635                let o = self.addr_ind_x(bus);
1636                let v = self.read1(bus, o.addr);
1637                self.ora(v);
1638                *cycles = 6;
1639            }
1640            0x11 => {
1641                let o = self.addr_ind_y(bus);
1642                let v = self.read1(bus, o.addr);
1643                self.ora(v);
1644                *cycles = 5 + u8::from(o.page_crossed);
1645            }
1646
1647            0x49 => {
1648                let v = self.fetch_pc(bus);
1649                self.eor(v);
1650                *cycles = 2;
1651            }
1652            0x45 => {
1653                let o = self.addr_zp(bus);
1654                let v = self.read1(bus, o.addr);
1655                self.eor(v);
1656                *cycles = 3;
1657            }
1658            0x55 => {
1659                let o = self.addr_zp_x(bus);
1660                let v = self.read1(bus, o.addr);
1661                self.eor(v);
1662                *cycles = 4;
1663            }
1664            0x4D => {
1665                let o = self.addr_abs(bus);
1666                let v = self.read1(bus, o.addr);
1667                self.eor(v);
1668                *cycles = 4;
1669            }
1670            0x5D => {
1671                let o = self.addr_abs_x(bus);
1672                let v = self.read1(bus, o.addr);
1673                self.eor(v);
1674                *cycles = 4 + u8::from(o.page_crossed);
1675            }
1676            0x59 => {
1677                let o = self.addr_abs_y(bus);
1678                let v = self.read1(bus, o.addr);
1679                self.eor(v);
1680                *cycles = 4 + u8::from(o.page_crossed);
1681            }
1682            0x41 => {
1683                let o = self.addr_ind_x(bus);
1684                let v = self.read1(bus, o.addr);
1685                self.eor(v);
1686                *cycles = 6;
1687            }
1688            0x51 => {
1689                let o = self.addr_ind_y(bus);
1690                let v = self.read1(bus, o.addr);
1691                self.eor(v);
1692                *cycles = 5 + u8::from(o.page_crossed);
1693            }
1694
1695            0x24 => {
1696                let o = self.addr_zp(bus);
1697                let v = self.read1(bus, o.addr);
1698                self.bit(v);
1699                *cycles = 3;
1700            }
1701            0x2C => {
1702                let o = self.addr_abs(bus);
1703                let v = self.read1(bus, o.addr);
1704                self.bit(v);
1705                *cycles = 4;
1706            }
1707
1708            // === Arithmetic ===
1709            0x69 => {
1710                let v = self.fetch_pc(bus);
1711                self.adc(v);
1712                *cycles = 2;
1713            }
1714            0x65 => {
1715                let o = self.addr_zp(bus);
1716                let v = self.read1(bus, o.addr);
1717                self.adc(v);
1718                *cycles = 3;
1719            }
1720            0x75 => {
1721                let o = self.addr_zp_x(bus);
1722                let v = self.read1(bus, o.addr);
1723                self.adc(v);
1724                *cycles = 4;
1725            }
1726            0x6D => {
1727                let o = self.addr_abs(bus);
1728                let v = self.read1(bus, o.addr);
1729                self.adc(v);
1730                *cycles = 4;
1731            }
1732            0x7D => {
1733                let o = self.addr_abs_x(bus);
1734                let v = self.read1(bus, o.addr);
1735                self.adc(v);
1736                *cycles = 4 + u8::from(o.page_crossed);
1737            }
1738            0x79 => {
1739                let o = self.addr_abs_y(bus);
1740                let v = self.read1(bus, o.addr);
1741                self.adc(v);
1742                *cycles = 4 + u8::from(o.page_crossed);
1743            }
1744            0x61 => {
1745                let o = self.addr_ind_x(bus);
1746                let v = self.read1(bus, o.addr);
1747                self.adc(v);
1748                *cycles = 6;
1749            }
1750            0x71 => {
1751                let o = self.addr_ind_y(bus);
1752                let v = self.read1(bus, o.addr);
1753                self.adc(v);
1754                *cycles = 5 + u8::from(o.page_crossed);
1755            }
1756
1757            0xE9 | 0xEB => {
1758                let v = self.fetch_pc(bus);
1759                self.sbc(v);
1760                *cycles = 2;
1761            }
1762            0xE5 => {
1763                let o = self.addr_zp(bus);
1764                let v = self.read1(bus, o.addr);
1765                self.sbc(v);
1766                *cycles = 3;
1767            }
1768            0xF5 => {
1769                let o = self.addr_zp_x(bus);
1770                let v = self.read1(bus, o.addr);
1771                self.sbc(v);
1772                *cycles = 4;
1773            }
1774            0xED => {
1775                let o = self.addr_abs(bus);
1776                let v = self.read1(bus, o.addr);
1777                self.sbc(v);
1778                *cycles = 4;
1779            }
1780            0xFD => {
1781                let o = self.addr_abs_x(bus);
1782                let v = self.read1(bus, o.addr);
1783                self.sbc(v);
1784                *cycles = 4 + u8::from(o.page_crossed);
1785            }
1786            0xF9 => {
1787                let o = self.addr_abs_y(bus);
1788                let v = self.read1(bus, o.addr);
1789                self.sbc(v);
1790                *cycles = 4 + u8::from(o.page_crossed);
1791            }
1792            0xE1 => {
1793                let o = self.addr_ind_x(bus);
1794                let v = self.read1(bus, o.addr);
1795                self.sbc(v);
1796                *cycles = 6;
1797            }
1798            0xF1 => {
1799                let o = self.addr_ind_y(bus);
1800                let v = self.read1(bus, o.addr);
1801                self.sbc(v);
1802                *cycles = 5 + u8::from(o.page_crossed);
1803            }
1804
1805            // === Compare ===
1806            0xC9 => {
1807                let v = self.fetch_pc(bus);
1808                self.cmp_with(self.a, v);
1809                *cycles = 2;
1810            }
1811            0xC5 => {
1812                let o = self.addr_zp(bus);
1813                let v = self.read1(bus, o.addr);
1814                self.cmp_with(self.a, v);
1815                *cycles = 3;
1816            }
1817            0xD5 => {
1818                let o = self.addr_zp_x(bus);
1819                let v = self.read1(bus, o.addr);
1820                self.cmp_with(self.a, v);
1821                *cycles = 4;
1822            }
1823            0xCD => {
1824                let o = self.addr_abs(bus);
1825                let v = self.read1(bus, o.addr);
1826                self.cmp_with(self.a, v);
1827                *cycles = 4;
1828            }
1829            0xDD => {
1830                let o = self.addr_abs_x(bus);
1831                let v = self.read1(bus, o.addr);
1832                self.cmp_with(self.a, v);
1833                *cycles = 4 + u8::from(o.page_crossed);
1834            }
1835            0xD9 => {
1836                let o = self.addr_abs_y(bus);
1837                let v = self.read1(bus, o.addr);
1838                self.cmp_with(self.a, v);
1839                *cycles = 4 + u8::from(o.page_crossed);
1840            }
1841            0xC1 => {
1842                let o = self.addr_ind_x(bus);
1843                let v = self.read1(bus, o.addr);
1844                self.cmp_with(self.a, v);
1845                *cycles = 6;
1846            }
1847            0xD1 => {
1848                let o = self.addr_ind_y(bus);
1849                let v = self.read1(bus, o.addr);
1850                self.cmp_with(self.a, v);
1851                *cycles = 5 + u8::from(o.page_crossed);
1852            }
1853
1854            0xE0 => {
1855                let v = self.fetch_pc(bus);
1856                self.cmp_with(self.x, v);
1857                *cycles = 2;
1858            }
1859            0xE4 => {
1860                let o = self.addr_zp(bus);
1861                let v = self.read1(bus, o.addr);
1862                self.cmp_with(self.x, v);
1863                *cycles = 3;
1864            }
1865            0xEC => {
1866                let o = self.addr_abs(bus);
1867                let v = self.read1(bus, o.addr);
1868                self.cmp_with(self.x, v);
1869                *cycles = 4;
1870            }
1871
1872            0xC0 => {
1873                let v = self.fetch_pc(bus);
1874                self.cmp_with(self.y, v);
1875                *cycles = 2;
1876            }
1877            0xC4 => {
1878                let o = self.addr_zp(bus);
1879                let v = self.read1(bus, o.addr);
1880                self.cmp_with(self.y, v);
1881                *cycles = 3;
1882            }
1883            0xCC => {
1884                let o = self.addr_abs(bus);
1885                let v = self.read1(bus, o.addr);
1886                self.cmp_with(self.y, v);
1887                *cycles = 4;
1888            }
1889
1890            // === Increments / decrements ===
1891            0xE6 => {
1892                let o = self.addr_zp(bus);
1893                self.inc_addr(bus, o.addr);
1894                *cycles = 5;
1895            }
1896            0xF6 => {
1897                let o = self.addr_zp_x(bus);
1898                self.inc_addr(bus, o.addr);
1899                *cycles = 6;
1900            }
1901            0xEE => {
1902                let o = self.addr_abs(bus);
1903                self.inc_addr(bus, o.addr);
1904                *cycles = 6;
1905            }
1906            0xFE => {
1907                let addr = self.addr_abs_x_rmw(bus);
1908                self.inc_addr(bus, addr);
1909                *cycles = 7;
1910            }
1911            0xC6 => {
1912                let o = self.addr_zp(bus);
1913                self.dec_addr(bus, o.addr);
1914                *cycles = 5;
1915            }
1916            0xD6 => {
1917                let o = self.addr_zp_x(bus);
1918                self.dec_addr(bus, o.addr);
1919                *cycles = 6;
1920            }
1921            0xCE => {
1922                let o = self.addr_abs(bus);
1923                self.dec_addr(bus, o.addr);
1924                *cycles = 6;
1925            }
1926            0xDE => {
1927                let addr = self.addr_abs_x_rmw(bus);
1928                self.dec_addr(bus, addr);
1929                *cycles = 7;
1930            }
1931            0xE8 => {
1932                self.implied_dummy_read(bus);
1933                self.x = self.x.wrapping_add(1);
1934                self.p.set_nz(self.x);
1935                *cycles = 2;
1936            }
1937            0xCA => {
1938                self.implied_dummy_read(bus);
1939                self.x = self.x.wrapping_sub(1);
1940                self.p.set_nz(self.x);
1941                *cycles = 2;
1942            }
1943            0xC8 => {
1944                self.implied_dummy_read(bus);
1945                self.y = self.y.wrapping_add(1);
1946                self.p.set_nz(self.y);
1947                *cycles = 2;
1948            }
1949            0x88 => {
1950                self.implied_dummy_read(bus);
1951                self.y = self.y.wrapping_sub(1);
1952                self.p.set_nz(self.y);
1953                *cycles = 2;
1954            }
1955
1956            // === Shifts ===
1957            0x0A => {
1958                self.implied_dummy_read(bus);
1959                self.a = self.asl_value(self.a);
1960                *cycles = 2;
1961            }
1962            0x06 => {
1963                let o = self.addr_zp(bus);
1964                self.asl_addr(bus, o.addr);
1965                *cycles = 5;
1966            }
1967            0x16 => {
1968                let o = self.addr_zp_x(bus);
1969                self.asl_addr(bus, o.addr);
1970                *cycles = 6;
1971            }
1972            0x0E => {
1973                let o = self.addr_abs(bus);
1974                self.asl_addr(bus, o.addr);
1975                *cycles = 6;
1976            }
1977            0x1E => {
1978                let addr = self.addr_abs_x_rmw(bus);
1979                self.asl_addr(bus, addr);
1980                *cycles = 7;
1981            }
1982
1983            0x4A => {
1984                self.implied_dummy_read(bus);
1985                self.a = self.lsr_value(self.a);
1986                *cycles = 2;
1987            }
1988            0x46 => {
1989                let o = self.addr_zp(bus);
1990                self.lsr_addr(bus, o.addr);
1991                *cycles = 5;
1992            }
1993            0x56 => {
1994                let o = self.addr_zp_x(bus);
1995                self.lsr_addr(bus, o.addr);
1996                *cycles = 6;
1997            }
1998            0x4E => {
1999                let o = self.addr_abs(bus);
2000                self.lsr_addr(bus, o.addr);
2001                *cycles = 6;
2002            }
2003            0x5E => {
2004                let addr = self.addr_abs_x_rmw(bus);
2005                self.lsr_addr(bus, addr);
2006                *cycles = 7;
2007            }
2008
2009            0x2A => {
2010                self.implied_dummy_read(bus);
2011                self.a = self.rol_value(self.a);
2012                *cycles = 2;
2013            }
2014            0x26 => {
2015                let o = self.addr_zp(bus);
2016                self.rol_addr(bus, o.addr);
2017                *cycles = 5;
2018            }
2019            0x36 => {
2020                let o = self.addr_zp_x(bus);
2021                self.rol_addr(bus, o.addr);
2022                *cycles = 6;
2023            }
2024            0x2E => {
2025                let o = self.addr_abs(bus);
2026                self.rol_addr(bus, o.addr);
2027                *cycles = 6;
2028            }
2029            0x3E => {
2030                let addr = self.addr_abs_x_rmw(bus);
2031                self.rol_addr(bus, addr);
2032                *cycles = 7;
2033            }
2034
2035            0x6A => {
2036                self.implied_dummy_read(bus);
2037                self.a = self.ror_value(self.a);
2038                *cycles = 2;
2039            }
2040            0x66 => {
2041                let o = self.addr_zp(bus);
2042                self.ror_addr(bus, o.addr);
2043                *cycles = 5;
2044            }
2045            0x76 => {
2046                let o = self.addr_zp_x(bus);
2047                self.ror_addr(bus, o.addr);
2048                *cycles = 6;
2049            }
2050            0x6E => {
2051                let o = self.addr_abs(bus);
2052                self.ror_addr(bus, o.addr);
2053                *cycles = 6;
2054            }
2055            0x7E => {
2056                let addr = self.addr_abs_x_rmw(bus);
2057                self.ror_addr(bus, addr);
2058                *cycles = 7;
2059            }
2060
2061            // === Branches ===
2062            //
2063            // The `branch_delays_irq` quirk: real 6502 branches poll IRQ
2064            // at the same point a 2-cycle untaken branch would — at the
2065            // opcode-fetch cycle (the canonical 2-cycle "second-to-last"
2066            // poll).  The operand-fetch cycle and any extra taken /
2067            // page-cross cycles do NOT re-sample IRQ.  We suppress IRQ
2068            // sampling for the remaining cycles of the instruction
2069            // immediately *before* the operand fetch — the opcode-fetch
2070            // sample (in `step()`) has already happened by this point.
2071            // See `docs/cpu-6502.md` §Interrupt logic and
2072            // <https://www.nesdev.org/wiki/CPU_interrupts>.
2073            0x10 => {
2074                self.skip_irq_sample = true;
2075                let off = self.fetch_pc(bus);
2076                *cycles = self.branch(bus, off, !self.p.contains(Status::NEGATIVE));
2077            }
2078            0x30 => {
2079                self.skip_irq_sample = true;
2080                let off = self.fetch_pc(bus);
2081                *cycles = self.branch(bus, off, self.p.contains(Status::NEGATIVE));
2082            }
2083            0x50 => {
2084                self.skip_irq_sample = true;
2085                let off = self.fetch_pc(bus);
2086                *cycles = self.branch(bus, off, !self.p.contains(Status::OVERFLOW));
2087            }
2088            0x70 => {
2089                self.skip_irq_sample = true;
2090                let off = self.fetch_pc(bus);
2091                *cycles = self.branch(bus, off, self.p.contains(Status::OVERFLOW));
2092            }
2093            0x90 => {
2094                self.skip_irq_sample = true;
2095                let off = self.fetch_pc(bus);
2096                *cycles = self.branch(bus, off, !self.p.contains(Status::CARRY));
2097            }
2098            0xB0 => {
2099                self.skip_irq_sample = true;
2100                let off = self.fetch_pc(bus);
2101                *cycles = self.branch(bus, off, self.p.contains(Status::CARRY));
2102            }
2103            0xD0 => {
2104                self.skip_irq_sample = true;
2105                let off = self.fetch_pc(bus);
2106                *cycles = self.branch(bus, off, !self.p.contains(Status::ZERO));
2107            }
2108            0xF0 => {
2109                self.skip_irq_sample = true;
2110                let off = self.fetch_pc(bus);
2111                *cycles = self.branch(bus, off, self.p.contains(Status::ZERO));
2112            }
2113
2114            // === Jumps / subroutine ===
2115            0x4C => {
2116                self.pc = self.fetch_pc_u16(bus);
2117                *cycles = 3;
2118            }
2119            0x6C => {
2120                let ptr = self.fetch_pc_u16(bus);
2121                self.pc = self.read_u16_with_wrap(bus, ptr);
2122                *cycles = 5;
2123            }
2124            0x20 => {
2125                // Canonical 6502 JSR cycle sequence — the high byte of
2126                // the target is read AFTER PC is pushed to the stack.
2127                // Wrong order is observable when JSR overwrites its own
2128                // operand via the pushed return address (AccuracyCoin
2129                // `CPU Behavior 2 :: JSR Edge Cases` Test 2 brackets
2130                // this exactly):
2131                //   C1: opcode fetch (already done by `tick` dispatcher)
2132                //   C2: fetch low byte of target → advances PC
2133                //   C3: dummy read from stack at $0100|S (no-op)
2134                //   C4: push PC high (PC is currently at the high-byte
2135                //       operand address, which is exactly the return
2136                //       address minus one)
2137                //   C5: push PC low
2138                //   C6: fetch high byte of target → PC = target
2139                let lo = self.fetch_pc(bus);
2140                let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2141                // self.pc now points at the high-byte operand; this is
2142                // the "return - 1" address JSR canonically pushes.
2143                let return_minus_one = self.pc;
2144                self.push(bus, (return_minus_one >> 8) as u8);
2145                self.push(bus, (return_minus_one & 0xFF) as u8);
2146                let hi = self.fetch_pc(bus);
2147                self.pc = u16::from(lo) | (u16::from(hi) << 8);
2148                *cycles = 6;
2149            }
2150            0x60 => {
2151                // Canonical 6502 RTS bus pattern (every cycle is a bus access):
2152                //   C1 opcode fetch (dispatcher) | C2 dummy read PC |
2153                //   C3 dummy stack read (pre-increment) |
2154                //   C4 pull PCL | C5 pull PCH | C6 dummy read at the return addr.
2155                // Default build burns C2/C3/C6 as `idle_tick` (no bus access);
2156                // `cpu-stack-dummy-reads` emits the canonical dummy reads — the
2157                // DC-6 Y=3-vs-4 fix. See the cell-trace cross-diff.
2158                {
2159                    let _ = self.read1(bus, self.pc);
2160                    let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2161                    let v = self.pull_u16(bus);
2162                    let _ = self.read1(bus, v);
2163                    self.pc = v.wrapping_add(1);
2164                }
2165                *cycles = 6;
2166            }
2167            0x40 => {
2168                // Canonical RTI bus pattern: C2 dummy read PC, C3 dummy stack
2169                // read (pre-increment) before the pulls. Default-off helper.
2170                {
2171                    let _ = self.read1(bus, self.pc);
2172                    let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2173                }
2174                let p = self.pull(bus);
2175                let mut new_p = Status::from_bits_truncate(p);
2176                new_p.remove(Status::BREAK);
2177                new_p.insert(Status::UNUSED);
2178                self.p = new_p;
2179                // RTI's I-flag change is observed by the IRQ sample
2180                // (unlike PLP / CLI / SEI which delay one instruction).
2181                self.irq_sample_i_flag = self.p.contains(Status::INTERRUPT_DISABLE);
2182                self.pc = self.pull_u16(bus);
2183                *cycles = 6;
2184            }
2185            0x00 => {
2186                // BRK is a 7-cycle interrupt with PC+2 pushed (PC already
2187                // advanced by fetch; advance one more for the padding byte).
2188                self.pc = self.pc.wrapping_add(1);
2189                self.service_interrupt(bus, IRQ_VECTOR, true);
2190                // R1/A2: suppress an NMI that became pending during/just-after
2191                // the BRK sequence so the FIRST instruction of the IRQ handler
2192                // runs before the NMI is taken (Mesen2 `NesCpu::BRK`
2193                // `_prevNeedNmi = false`; "needed for nmi_and_brk"). The NMI is
2194                // not lost — `mc_need_nmi` stays set and re-arms next cycle.
2195                {
2196                    self.mc_prev_need_nmi = false;
2197                }
2198                // service_interrupt already burned 7 cycles; do NOT double-count.
2199                *cycles = 0;
2200            }
2201            0xEA => {
2202                self.implied_dummy_read(bus);
2203                *cycles = 2;
2204            }
2205
2206            // === Flag manipulation ===
2207            0x18 => {
2208                self.implied_dummy_read(bus);
2209                self.p.remove(Status::CARRY);
2210                *cycles = 2;
2211            }
2212            0x38 => {
2213                self.implied_dummy_read(bus);
2214                self.p.insert(Status::CARRY);
2215                *cycles = 2;
2216            }
2217            0x58 => {
2218                self.implied_dummy_read(bus);
2219                self.p.remove(Status::INTERRUPT_DISABLE);
2220                *cycles = 2;
2221            }
2222            0x78 => {
2223                self.implied_dummy_read(bus);
2224                self.p.insert(Status::INTERRUPT_DISABLE);
2225                *cycles = 2;
2226            }
2227            0xB8 => {
2228                self.implied_dummy_read(bus);
2229                self.p.remove(Status::OVERFLOW);
2230                *cycles = 2;
2231            }
2232            0xD8 => {
2233                self.implied_dummy_read(bus);
2234                self.p.remove(Status::DECIMAL);
2235                *cycles = 2;
2236            }
2237            0xF8 => {
2238                self.implied_dummy_read(bus);
2239                self.p.insert(Status::DECIMAL);
2240                *cycles = 2;
2241            }
2242
2243            // === Unofficial NOP variants ===
2244            // Implied / 1-byte NOPs
2245            0x1A | 0x3A | 0x5A | 0x7A | 0xDA | 0xFA => {
2246                self.implied_dummy_read(bus);
2247                *cycles = 2;
2248            }
2249            // Immediate / zero-page DOP (double NOP) variants: skip 1 byte.
2250            0x80 | 0x82 | 0x89 | 0xC2 | 0xE2 => {
2251                let _ = self.fetch_pc(bus);
2252                *cycles = 2;
2253            }
2254            0x04 | 0x44 | 0x64 => {
2255                let o = self.addr_zp(bus);
2256                let _ = self.read1(bus, o.addr); // unofficial DOP dummy read
2257                *cycles = 3;
2258            }
2259            0x14 | 0x34 | 0x54 | 0x74 | 0xD4 | 0xF4 => {
2260                let o = self.addr_zp_x(bus);
2261                let _ = self.read1(bus, o.addr); // unofficial DOP dummy read
2262                *cycles = 4;
2263            }
2264            // Absolute "TOP" (triple NOP) — must dummy-read the target so
2265            // that PPU-mirror side-effects (e.g. clearing $2002.7) fire,
2266            // matching real silicon and AccuracyCoin's All-NOPs Test 2.
2267            0x0C => {
2268                let o = self.addr_abs(bus);
2269                let _ = self.read1(bus, o.addr);
2270                *cycles = 4;
2271            }
2272            0x1C | 0x3C | 0x5C | 0x7C | 0xDC | 0xFC => {
2273                let o = self.addr_abs_x(bus);
2274                let _ = self.read1(bus, o.addr); // dummy read on TOP
2275                *cycles = 4 + u8::from(o.page_crossed);
2276            }
2277
2278            // === Stable unofficial: LAX, SAX ===
2279            0xA7 => {
2280                let o = self.addr_zp(bus);
2281                let v = self.read1(bus, o.addr);
2282                self.lax(v);
2283                *cycles = 3;
2284            }
2285            0xB7 => {
2286                let o = self.addr_zp_y(bus);
2287                let v = self.read1(bus, o.addr);
2288                self.lax(v);
2289                *cycles = 4;
2290            }
2291            0xAF => {
2292                let o = self.addr_abs(bus);
2293                let v = self.read1(bus, o.addr);
2294                self.lax(v);
2295                *cycles = 4;
2296            }
2297            0xBF => {
2298                let o = self.addr_abs_y(bus);
2299                let v = self.read1(bus, o.addr);
2300                self.lax(v);
2301                *cycles = 4 + u8::from(o.page_crossed);
2302            }
2303            0xA3 => {
2304                let o = self.addr_ind_x(bus);
2305                let v = self.read1(bus, o.addr);
2306                self.lax(v);
2307                *cycles = 6;
2308            }
2309            0xB3 => {
2310                let o = self.addr_ind_y(bus);
2311                let v = self.read1(bus, o.addr);
2312                self.lax(v);
2313                *cycles = 5 + u8::from(o.page_crossed);
2314            }
2315            0xAB => {
2316                let v = self.fetch_pc(bus);
2317                self.lax(v);
2318                *cycles = 2;
2319            } // LAX immediate (often listed as ATX). We follow nestest behavior.
2320
2321            0x87 => {
2322                let o = self.addr_zp(bus);
2323                self.write1(bus, o.addr, self.a & self.x);
2324                *cycles = 3;
2325            }
2326            0x97 => {
2327                let o = self.addr_zp_y(bus);
2328                self.write1(bus, o.addr, self.a & self.x);
2329                *cycles = 4;
2330            }
2331            0x8F => {
2332                let o = self.addr_abs(bus);
2333                self.write1(bus, o.addr, self.a & self.x);
2334                *cycles = 4;
2335            }
2336            0x83 => {
2337                let o = self.addr_ind_x(bus);
2338                self.write1(bus, o.addr, self.a & self.x);
2339                *cycles = 6;
2340            }
2341
2342            // === DCP (DEC + CMP) ===
2343            0xC7 => {
2344                let o = self.addr_zp(bus);
2345                self.dcp_addr(bus, o.addr);
2346                *cycles = 5;
2347            }
2348            0xD7 => {
2349                let o = self.addr_zp_x(bus);
2350                self.dcp_addr(bus, o.addr);
2351                *cycles = 6;
2352            }
2353            0xCF => {
2354                let o = self.addr_abs(bus);
2355                self.dcp_addr(bus, o.addr);
2356                *cycles = 6;
2357            }
2358            0xDF => {
2359                let addr = self.addr_abs_x_rmw(bus);
2360                self.dcp_addr(bus, addr);
2361                *cycles = 7;
2362            }
2363            0xDB => {
2364                let addr = self.addr_abs_y_rmw(bus);
2365                self.dcp_addr(bus, addr);
2366                *cycles = 7;
2367            }
2368            0xC3 => {
2369                let o = self.addr_ind_x(bus);
2370                self.dcp_addr(bus, o.addr);
2371                *cycles = 8;
2372            }
2373            0xD3 => {
2374                let addr = self.addr_ind_y_rmw(bus);
2375                self.dcp_addr(bus, addr);
2376                *cycles = 8;
2377            }
2378
2379            // === ISC (INC + SBC) ===
2380            0xE7 => {
2381                let o = self.addr_zp(bus);
2382                self.isc_addr(bus, o.addr);
2383                *cycles = 5;
2384            }
2385            0xF7 => {
2386                let o = self.addr_zp_x(bus);
2387                self.isc_addr(bus, o.addr);
2388                *cycles = 6;
2389            }
2390            0xEF => {
2391                let o = self.addr_abs(bus);
2392                self.isc_addr(bus, o.addr);
2393                *cycles = 6;
2394            }
2395            0xFF => {
2396                let addr = self.addr_abs_x_rmw(bus);
2397                self.isc_addr(bus, addr);
2398                *cycles = 7;
2399            }
2400            0xFB => {
2401                let addr = self.addr_abs_y_rmw(bus);
2402                self.isc_addr(bus, addr);
2403                *cycles = 7;
2404            }
2405            0xE3 => {
2406                let o = self.addr_ind_x(bus);
2407                self.isc_addr(bus, o.addr);
2408                *cycles = 8;
2409            }
2410            0xF3 => {
2411                let addr = self.addr_ind_y_rmw(bus);
2412                self.isc_addr(bus, addr);
2413                *cycles = 8;
2414            }
2415
2416            // === SLO (ASL + ORA) ===
2417            0x07 => {
2418                let o = self.addr_zp(bus);
2419                self.slo_addr(bus, o.addr);
2420                *cycles = 5;
2421            }
2422            0x17 => {
2423                let o = self.addr_zp_x(bus);
2424                self.slo_addr(bus, o.addr);
2425                *cycles = 6;
2426            }
2427            0x0F => {
2428                let o = self.addr_abs(bus);
2429                self.slo_addr(bus, o.addr);
2430                *cycles = 6;
2431            }
2432            0x1F => {
2433                let addr = self.addr_abs_x_rmw(bus);
2434                self.slo_addr(bus, addr);
2435                *cycles = 7;
2436            }
2437            0x1B => {
2438                let addr = self.addr_abs_y_rmw(bus);
2439                self.slo_addr(bus, addr);
2440                *cycles = 7;
2441            }
2442            0x03 => {
2443                let o = self.addr_ind_x(bus);
2444                self.slo_addr(bus, o.addr);
2445                *cycles = 8;
2446            }
2447            0x13 => {
2448                let addr = self.addr_ind_y_rmw(bus);
2449                self.slo_addr(bus, addr);
2450                *cycles = 8;
2451            }
2452
2453            // === RLA (ROL + AND) ===
2454            0x27 => {
2455                let o = self.addr_zp(bus);
2456                self.rla_addr(bus, o.addr);
2457                *cycles = 5;
2458            }
2459            0x37 => {
2460                let o = self.addr_zp_x(bus);
2461                self.rla_addr(bus, o.addr);
2462                *cycles = 6;
2463            }
2464            0x2F => {
2465                let o = self.addr_abs(bus);
2466                self.rla_addr(bus, o.addr);
2467                *cycles = 6;
2468            }
2469            0x3F => {
2470                let addr = self.addr_abs_x_rmw(bus);
2471                self.rla_addr(bus, addr);
2472                *cycles = 7;
2473            }
2474            0x3B => {
2475                let addr = self.addr_abs_y_rmw(bus);
2476                self.rla_addr(bus, addr);
2477                *cycles = 7;
2478            }
2479            0x23 => {
2480                let o = self.addr_ind_x(bus);
2481                self.rla_addr(bus, o.addr);
2482                *cycles = 8;
2483            }
2484            0x33 => {
2485                let addr = self.addr_ind_y_rmw(bus);
2486                self.rla_addr(bus, addr);
2487                *cycles = 8;
2488            }
2489
2490            // === SRE (LSR + EOR) ===
2491            0x47 => {
2492                let o = self.addr_zp(bus);
2493                self.sre_addr(bus, o.addr);
2494                *cycles = 5;
2495            }
2496            0x57 => {
2497                let o = self.addr_zp_x(bus);
2498                self.sre_addr(bus, o.addr);
2499                *cycles = 6;
2500            }
2501            0x4F => {
2502                let o = self.addr_abs(bus);
2503                self.sre_addr(bus, o.addr);
2504                *cycles = 6;
2505            }
2506            0x5F => {
2507                let addr = self.addr_abs_x_rmw(bus);
2508                self.sre_addr(bus, addr);
2509                *cycles = 7;
2510            }
2511            0x5B => {
2512                let addr = self.addr_abs_y_rmw(bus);
2513                self.sre_addr(bus, addr);
2514                *cycles = 7;
2515            }
2516            0x43 => {
2517                let o = self.addr_ind_x(bus);
2518                self.sre_addr(bus, o.addr);
2519                *cycles = 8;
2520            }
2521            0x53 => {
2522                let addr = self.addr_ind_y_rmw(bus);
2523                self.sre_addr(bus, addr);
2524                *cycles = 8;
2525            }
2526
2527            // === RRA (ROR + ADC) ===
2528            0x67 => {
2529                let o = self.addr_zp(bus);
2530                self.rra_addr(bus, o.addr);
2531                *cycles = 5;
2532            }
2533            0x77 => {
2534                let o = self.addr_zp_x(bus);
2535                self.rra_addr(bus, o.addr);
2536                *cycles = 6;
2537            }
2538            0x6F => {
2539                let o = self.addr_abs(bus);
2540                self.rra_addr(bus, o.addr);
2541                *cycles = 6;
2542            }
2543            0x7F => {
2544                let addr = self.addr_abs_x_rmw(bus);
2545                self.rra_addr(bus, addr);
2546                *cycles = 7;
2547            }
2548            0x7B => {
2549                let addr = self.addr_abs_y_rmw(bus);
2550                self.rra_addr(bus, addr);
2551                *cycles = 7;
2552            }
2553            0x63 => {
2554                let o = self.addr_ind_x(bus);
2555                self.rra_addr(bus, o.addr);
2556                *cycles = 8;
2557            }
2558            0x73 => {
2559                let addr = self.addr_ind_y_rmw(bus);
2560                self.rra_addr(bus, addr);
2561                *cycles = 8;
2562            }
2563
2564            // === ANC, ALR, ARR, AXS ===
2565            0x0B | 0x2B => {
2566                let v = self.fetch_pc(bus);
2567                self.a &= v;
2568                self.p.set_nz(self.a);
2569                self.p.set(Status::CARRY, self.a & 0x80 != 0);
2570                *cycles = 2;
2571            }
2572            0x4B => {
2573                let v = self.fetch_pc(bus);
2574                self.a &= v;
2575                let new_carry = self.a & 0x01 != 0;
2576                self.a >>= 1;
2577                self.p.set_nz(self.a);
2578                self.p.set(Status::CARRY, new_carry);
2579                *cycles = 2;
2580            }
2581            0x6B => {
2582                let v = self.fetch_pc(bus);
2583                self.a &= v;
2584                let carry_in = self.p.contains(Status::CARRY);
2585                self.a = (self.a >> 1) | (u8::from(carry_in) << 7);
2586                self.p.set_nz(self.a);
2587                let bit6 = self.a & 0x40 != 0;
2588                let bit5 = self.a & 0x20 != 0;
2589                self.p.set(Status::CARRY, bit6);
2590                self.p.set(Status::OVERFLOW, bit6 ^ bit5);
2591                *cycles = 2;
2592            }
2593            0xCB => {
2594                let v = self.fetch_pc(bus);
2595                let ax = self.a & self.x;
2596                let (res, overflow) = ax.overflowing_sub(v);
2597                self.x = res;
2598                self.p.set(Status::CARRY, !overflow);
2599                self.p.set_nz(res);
2600                *cycles = 2;
2601            }
2602
2603            // === Unstable: XAA, LAS, TAS, SHA, SHX, SHY ===
2604            0x8B => {
2605                // XAA / ANE: A = (A | const) & X & operand. nestest expects this.
2606                let v = self.fetch_pc(bus);
2607                self.a = (self.a | 0xFF) & self.x & v;
2608                self.p.set_nz(self.a);
2609                *cycles = 2;
2610            }
2611            0xBB => {
2612                let o = self.addr_abs_y(bus);
2613                let v = self.read1(bus, o.addr);
2614                let res = self.s & v;
2615                self.a = res;
2616                self.x = res;
2617                self.s = res;
2618                self.p.set_nz(res);
2619                *cycles = 4 + u8::from(o.page_crossed);
2620            }
2621            0x9B => {
2622                // TAS / SHS / XAS abs,Y: S = A & X; then SHA-style write
2623                // using `S` as the value register.
2624                let base = self.fetch_pc_u16(bus);
2625                self.s = self.a & self.x;
2626                self.sh_store(bus, base, self.y, self.s);
2627                *cycles = 5;
2628            }
2629            0x9F => {
2630                // SHA abs,Y. value_reg = A & X.
2631                let base = self.fetch_pc_u16(bus);
2632                self.sh_store(bus, base, self.y, self.a & self.x);
2633                *cycles = 5;
2634            }
2635            0x93 => {
2636                // SHA (zp),Y. Indirect; base from zp-pointer-resolved
2637                // low/high bytes.  value_reg = A & X.
2638                let zp = self.fetch_pc(bus);
2639                let lo = self.read1(bus, u16::from(zp));
2640                let hi_byte = self.read1(bus, u16::from(zp.wrapping_add(1)));
2641                let base = u16::from(lo) | (u16::from(hi_byte) << 8);
2642                self.sh_store(bus, base, self.y, self.a & self.x);
2643                *cycles = 6;
2644            }
2645            0x9E => {
2646                // SHX abs,Y. value_reg = X.
2647                let base = self.fetch_pc_u16(bus);
2648                self.sh_store(bus, base, self.y, self.x);
2649                *cycles = 5;
2650            }
2651            0x9C => {
2652                // SHY abs,X. value_reg = Y. Index register is X here.
2653                let base = self.fetch_pc_u16(bus);
2654                self.sh_store(bus, base, self.x, self.y);
2655                *cycles = 5;
2656            }
2657
2658            // === JAM / KIL / STP ===
2659            0x02 | 0x12 | 0x22 | 0x32 | 0x42 | 0x52 | 0x62 | 0x72 | 0x92 | 0xB2 | 0xD2 | 0xF2 => {
2660                // v2.0.0 (every-cycle-bus-access): cycle 2 is a real dummy
2661                // read of the byte after the opcode before the CPU wedges —
2662                // the same silicon shape as the implied-opcode cycle-2 dummy
2663                // read. Caught post-promote by the burn-loop fail-loud
2664                // assert (this arm declared 2 cycles but emitted only the
2665                // opcode fetch — invisible to every probe workload, since no
2666                // test ROM executes a JAM).
2667                let _ = self.read1(bus, self.pc);
2668                self.jammed = true;
2669                *cycles = 2;
2670            }
2671        }
2672    }
2673
2674    // ------------------------------------------------------------------
2675    // Helpers / micro-ops.
2676    // ------------------------------------------------------------------
2677
2678    fn lda(&mut self, value: u8) {
2679        self.a = value;
2680        self.p.set_nz(value);
2681    }
2682
2683    fn lda_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2684        let v = self.read1(bus, addr);
2685        self.lda(v);
2686    }
2687
2688    fn ldx(&mut self, value: u8) {
2689        self.x = value;
2690        self.p.set_nz(value);
2691    }
2692
2693    fn ldy(&mut self, value: u8) {
2694        self.y = value;
2695        self.p.set_nz(value);
2696    }
2697
2698    fn and(&mut self, value: u8) {
2699        self.a &= value;
2700        self.p.set_nz(self.a);
2701    }
2702
2703    fn ora(&mut self, value: u8) {
2704        self.a |= value;
2705        self.p.set_nz(self.a);
2706    }
2707
2708    fn eor(&mut self, value: u8) {
2709        self.a ^= value;
2710        self.p.set_nz(self.a);
2711    }
2712
2713    fn bit(&mut self, value: u8) {
2714        let result = self.a & value;
2715        self.p.set(Status::ZERO, result == 0);
2716        self.p.set(Status::NEGATIVE, value & 0x80 != 0);
2717        self.p.set(Status::OVERFLOW, value & 0x40 != 0);
2718    }
2719
2720    fn adc(&mut self, value: u8) {
2721        let carry = u16::from(self.p.contains(Status::CARRY));
2722        let sum = u16::from(self.a) + u16::from(value) + carry;
2723        let result = sum as u8;
2724        self.p.set(Status::CARRY, sum > 0xFF);
2725        let overflow = ((self.a ^ result) & (value ^ result) & 0x80) != 0;
2726        self.p.set(Status::OVERFLOW, overflow);
2727        self.a = result;
2728        self.p.set_nz(self.a);
2729    }
2730
2731    fn sbc(&mut self, value: u8) {
2732        // SBC = ADC of inverted value.
2733        self.adc(value ^ 0xFF);
2734    }
2735
2736    fn cmp_with(&mut self, lhs: u8, rhs: u8) {
2737        let (r, borrow) = lhs.overflowing_sub(rhs);
2738        self.p.set(Status::CARRY, !borrow);
2739        self.p.set_nz(r);
2740    }
2741
2742    fn inc_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2743        let original = self.read1(bus, addr);
2744        // RMW dummy write: real 6502 writes the original byte back to the
2745        // same address before writing the modified value (visible at memory-
2746        // mapped registers like $4014 and $2007). See `docs/cpu-6502.md` and
2747        // nesdev wiki "Dummy writes".
2748        self.write1(bus, addr, original);
2749        let v = original.wrapping_add(1);
2750        self.write1(bus, addr, v);
2751        self.p.set_nz(v);
2752    }
2753
2754    fn dec_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2755        let original = self.read1(bus, addr);
2756        self.write1(bus, addr, original);
2757        let v = original.wrapping_sub(1);
2758        self.write1(bus, addr, v);
2759        self.p.set_nz(v);
2760    }
2761
2762    fn asl_value(&mut self, value: u8) -> u8 {
2763        self.p.set(Status::CARRY, value & 0x80 != 0);
2764        let r = value << 1;
2765        self.p.set_nz(r);
2766        r
2767    }
2768
2769    fn asl_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2770        let v = self.read1(bus, addr);
2771        // RMW dummy write — see `inc_addr`.
2772        self.write1(bus, addr, v);
2773        let r = self.asl_value(v);
2774        self.write1(bus, addr, r);
2775    }
2776
2777    fn lsr_value(&mut self, value: u8) -> u8 {
2778        self.p.set(Status::CARRY, value & 0x01 != 0);
2779        let r = value >> 1;
2780        self.p.set_nz(r);
2781        r
2782    }
2783
2784    fn lsr_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2785        let v = self.read1(bus, addr);
2786        self.write1(bus, addr, v);
2787        let r = self.lsr_value(v);
2788        self.write1(bus, addr, r);
2789    }
2790
2791    fn rol_value(&mut self, value: u8) -> u8 {
2792        let carry_in = u8::from(self.p.contains(Status::CARRY));
2793        self.p.set(Status::CARRY, value & 0x80 != 0);
2794        let r = (value << 1) | carry_in;
2795        self.p.set_nz(r);
2796        r
2797    }
2798
2799    fn rol_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2800        let v = self.read1(bus, addr);
2801        self.write1(bus, addr, v);
2802        let r = self.rol_value(v);
2803        self.write1(bus, addr, r);
2804    }
2805
2806    fn ror_value(&mut self, value: u8) -> u8 {
2807        let carry_in = u8::from(self.p.contains(Status::CARRY)) << 7;
2808        self.p.set(Status::CARRY, value & 0x01 != 0);
2809        let r = (value >> 1) | carry_in;
2810        self.p.set_nz(r);
2811        r
2812    }
2813
2814    fn ror_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2815        let v = self.read1(bus, addr);
2816        self.write1(bus, addr, v);
2817        let r = self.ror_value(v);
2818        self.write1(bus, addr, r);
2819    }
2820
2821    fn branch<B: Bus>(&mut self, bus: &mut B, offset: u8, condition: bool) -> u8 {
2822        if !condition {
2823            return 2;
2824        }
2825        // T-60-001 (2026-05-17 — 9th C1 attempt, branch axis): per
2826        // nesdev wiki §"CPU interrupts" §"Branch instructions", TAKEN
2827        // branches DELAY IRQ detection. Our `step()` samples IRQ at
2828        // the opcode-fetch (cycle 1 = tick 0) via `idle_tick`, then
2829        // each branch opcode sets `skip_irq_sample = true` before the
2830        // operand fetch to suppress sampling on the extra cycles. But
2831        // the cycle-1 sample is still recorded in `irq_first_tick`
2832        // and `promote_post_step_interrupts` will ARM the IRQ at the
2833        // end of this instruction (cycle-1 sample < last_tick on a
2834        // 3- or 4-cycle taken branch). That contradicts the
2835        // "branches delay IRQ" rule — IRQ should be deferred to the
2836        // NEXT instruction's poll. Drop the cycle-1 sample here on
2837        // taken branches; the next instruction's opcode fetch will
2838        // re-sample the (still-asserted, level-triggered) IRQ line
2839        // and arm it normally. NMI is edge-triggered and sampled on
2840        // every cycle the CPU is alive (per nesdev) — its first-tick
2841        // latch is intentionally NOT dropped.
2842        self.irq_first_tick = u8::MAX;
2843        // Canonical 6502 branch cycle sequence per nesdev wiki and
2844        // AccuracyCoin `CPU Behavior 2 :: Branch Dummy Reads` Test 4:
2845        //   C1: opcode fetch (done by `tick` dispatcher)
2846        //   C2: operand fetch (done by per-opcode `fetch_pc` before call)
2847        //   C3: dummy read of PC (the byte after the operand) — this is
2848        //       cycle 3 of the taken branch, and is observable as a
2849        //       second consecutive read of `$2002` mirror through which
2850        //       AccuracyCoin brackets the dummy.
2851        //   C4: (only if page-crossed) dummy read of (old_pch | new_pcl)
2852        //       — the unfixed-high-byte address before the high-byte
2853        //       carry propagates.
2854        let _ = self.read1(bus, self.pc); // C3 dummy
2855        let signed = offset as i8 as i16;
2856        let old_pc = self.pc;
2857        let new_pc = (self.pc as i32 + i32::from(signed)) as u16;
2858        let crossed = (old_pc & 0xFF00) != (new_pc & 0xFF00);
2859        if crossed {
2860            // W1 (`mc-r1-branch-poll-points`): a page-cross taken branch
2861            // polls a SECOND time at C4-start — TriCNES's
2862            // `PollInterrupts_CantDisableIRQ` in the BPL microcode
2863            // (`100thCoin/TriCNES` `Emulator.cs` at `94f1b117`): if the C2-start
2864            // poll already saw the IRQ this one cannot un-see it (can-SET-
2865            // not-clear). `mc_run_irq` is frozen across the branch's
2866            // remaining cycles by the `handle_interrupts` early-return, so
2867            // sample the live line here (state as of end-of-C3) and OR it in;
2868            // the end-of-C4 `mc_prev_run_irq` copy then exposes it to the
2869            // next `step()` dispatch.
2870            if !self.mc_run_irq {
2871                self.mc_run_irq = bus.irq_level() && !self.irq_sample_i_flag;
2872            }
2873            // C4 page-cross dummy read at the unfixed address.
2874            let dummy = (old_pc & 0xFF00) | (new_pc & 0x00FF);
2875            let _ = self.read1(bus, dummy);
2876        }
2877        self.pc = new_pc;
2878        if crossed { 4 } else { 3 }
2879    }
2880
2881    fn lax(&mut self, value: u8) {
2882        self.a = value;
2883        self.x = value;
2884        self.p.set_nz(value);
2885    }
2886
2887    fn dcp_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2888        let original = self.read1(bus, addr);
2889        // RMW dummy write.
2890        self.write1(bus, addr, original);
2891        let v = original.wrapping_sub(1);
2892        self.write1(bus, addr, v);
2893        self.cmp_with(self.a, v);
2894    }
2895
2896    fn isc_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2897        let original = self.read1(bus, addr);
2898        self.write1(bus, addr, original);
2899        let v = original.wrapping_add(1);
2900        self.write1(bus, addr, v);
2901        self.sbc(v);
2902    }
2903
2904    fn slo_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2905        let v = self.read1(bus, addr);
2906        self.write1(bus, addr, v);
2907        let r = self.asl_value(v);
2908        self.write1(bus, addr, r);
2909        self.a |= r;
2910        self.p.set_nz(self.a);
2911    }
2912
2913    fn rla_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2914        let v = self.read1(bus, addr);
2915        self.write1(bus, addr, v);
2916        let r = self.rol_value(v);
2917        self.write1(bus, addr, r);
2918        self.a &= r;
2919        self.p.set_nz(self.a);
2920    }
2921
2922    fn sre_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2923        let v = self.read1(bus, addr);
2924        self.write1(bus, addr, v);
2925        let r = self.lsr_value(v);
2926        self.write1(bus, addr, r);
2927        self.a ^= r;
2928        self.p.set_nz(self.a);
2929    }
2930
2931    fn rra_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2932        let v = self.read1(bus, addr);
2933        self.write1(bus, addr, v);
2934        let r = self.ror_value(v);
2935        self.write1(bus, addr, r);
2936        self.adc(r);
2937    }
2938}