rustynes_cpu/cpu.rs
1// SPDX-License-Identifier: GPL-3.0-or-later
2//
3// Provenance: the 6502/2A03 core is RustyNES's own, but the unstable-store opcode group (SHA/SHX/SHY/SHS/TAS — the `SyaSxaAxa` family) is derived from Mesen2 (GPL-3.0-or-later), `Core/NES/NesCpu.h`. See docs/originality-and-provenance.md (Section 1)
4// and NOTICE for the complete, audited derivation record.
5//! Ricoh 2A03 CPU (6502 derivative without BCD mode).
6//!
7//! See `docs/cpu-6502.md` for the spec. The implementation here matches:
8//!
9//! - all 151 documented 6502 opcodes,
10//! - all 105 unofficial / illegal opcodes that real software depends on,
11//! - the 12 JAM / KIL / STP halt opcodes,
12//! - cycle counts including page-crossing penalties on indexed reads, the
13//! `+1 if branch taken / +2 if branch crosses page` branch convention, and
14//! the dummy-read / dummy-write cycles of read-modify-write opcodes,
15//! - NMI (edge), IRQ (level), and BRK with the documented BRK/IRQ B-flag
16//! distinction,
17//! - the `JMP ($XXFF)` indirect page-bug.
18//!
19//! The CPU steps one *instruction* at a time (`Cpu::step`), returning the
20//! cycle count. Since the v2.0.0 one-clock scheduler (ADR 0002 / ADR 0029)
21//! every cycle is clocked in two halves: `start_cycle` catches the PPU up
22//! (`Bus::run_ppu_to`) and runs the bus's per-cycle work (`Bus::cpu_clock`),
23//! and `end_cycle` catches the PPU up to the cycle's end and ticks the DMC
24//! (`Bus::cpu_clock_apu_dmc`). A cycle with a bus access (`read1` /
25//! `write1`) performs it between the halves; an internal cycle (`idle_tick`)
26//! runs both halves with no access. There
27//! is no separate `tick()`, and the pre-v2.0.0 per-cycle `Bus::on_cpu_cycle`
28//! callback survives only as the default body of `Bus::cpu_clock` for
29//! simple test buses. (Until v2.7.5 this header still described that
30//! callback as the stepping interface; core audit §4.7.)
31
32// Truncating casts are intentional throughout: this module is byte-arithmetic
33// against the 6502's 8/16-bit register file. `as u8` / `as i8` is the
34// canonical encoding of the wrap behavior the hardware exhibits.
35#![allow(
36 clippy::cast_possible_truncation,
37 clippy::cast_lossless,
38 clippy::cast_possible_wrap,
39 clippy::cast_sign_loss
40)]
41
42use crate::bus::Bus;
43use crate::status::Status;
44
45/// Stack base address: the CPU stack lives at `$0100 + S`.
46const STACK_BASE: u16 = 0x0100;
47
48// v2.0 master-clock R1 substrate constants (Phase 2; `mc-r1-substrate`).
49// The PPU sub-cycle offset and the read/write master-clock split (`pre` in
50// start_cycle, `post` in end_cycle). Mesen `_ppuOffset=1` /
51// `_startClockCount`/`_endClockCount`. Master clocks per CPU cycle are NOT a
52// constant — they are the cartridge region's `cpu_divider` (NTSC 12 / PAL 16 /
53// Dendy 15), read from `bus.cpu_divider()` and fed to `read_split`/
54// `write_split`; the master-clock unit is shared with the bus's `run_ppu_to`,
55// which does the regioned dot conversion off `ppu_divider`.
56/// PPU sub-cycle offset: the PPU is run to `master_clock - PPU_OFFSET` in BOTH
57/// halves of every access (the double catch-up). Mesen `_ppuOffset = 1`.
58const PPU_OFFSET: u64 = 1;
59
60/// The effective PPU-sample offset for `run_ppu_to` (no BP sweep: the constant).
61#[inline]
62const fn ppu_sample_offset() -> u64 {
63 PPU_OFFSET
64}
65/// READ access master-clock split (`+= pre` in `start_cycle`, `+= post` in
66/// `end_cycle`); `pre + post = div`, the region's `cpu_divider`. Derived per
67/// region from `bus.cpu_divider()` so PAL (16) / Dendy (15) get the right
68/// CPU<->PPU phase; for the NTSC divisor 12 these are exactly (5, 7), so the
69/// NTSC path is byte-identical to the prior `const`s.
70#[cfg(not(feature = "phi2-write-sweep"))]
71#[inline]
72const fn read_split(div: u64) -> (u64, u64) {
73 let pre = div / 2 - PPU_OFFSET;
74 (pre, div - pre)
75}
76
77/// Sweepable `read_split` (feature `phi2-write-sweep`, v2.6.18 study only).
78///
79/// A 6502 SAMPLES a read at phi2 just as it commits a write there, so a phi2
80/// model has to move both. The first sweep moved writes alone, which changed
81/// the SPACING between a write and a following read rather than moving the
82/// access model as a unit -- see the plan's note on that measurement.
83#[cfg(feature = "phi2-write-sweep")]
84#[inline]
85fn read_split(div: u64) -> (u64, u64) {
86 let extra = u64::from(READ_PHI_OFFSET.load(core::sync::atomic::Ordering::Relaxed));
87 let back = u64::from(READ_PHI_BACKOFF.load(core::sync::atomic::Ordering::Relaxed));
88 // `+ extra` BEFORE `- PPU_OFFSET`: identical for every reachable value
89 // (`div >= 12`, `PPU_OFFSET == 1`), but it removes the conceptual
90 // question of an unsigned subtraction preceding the addition.
91 // `- back` places the access EARLIER than shipped, which the offsets alone
92 // cannot express; clamped to 1 so `pre` stays a real split.
93 // Every step is structurally safe rather than safe-by-current-constants:
94 // the subtraction of `PPU_OFFSET` saturates (raised in review -- it cannot
95 // underflow at `div >= 12`, but nothing in the expression says so), and the
96 // upper clamp bound is floored at 1 because `clamp` PANICS when min > max,
97 // which a hypothetical `div < 2` would produce.
98 let pre = (div / 2 + extra)
99 .saturating_sub(PPU_OFFSET)
100 .saturating_sub(back)
101 .clamp(1, div.saturating_sub(1).max(1));
102 (pre, div - pre)
103}
104
105/// Extra master clocks added to a READ's pre-access split (v2.6.18 study).
106/// 0 = shipped behaviour. NTSC: +4 puts the sample at dot 2.0 = phi2.
107#[cfg(feature = "phi2-write-sweep")]
108pub static READ_PHI_OFFSET: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
109
110/// Master clocks SUBTRACTED from a READ's pre-access split (v2.6.18 study).
111///
112/// The offsets alone can only place an access later. Reaching the CPU cycle's
113/// FIRST dot needs it earlier, and that cell had never been measured -- see
114/// `access_dot_derivation.rs`, which sweeps the grid this pair makes reachable.
115#[cfg(feature = "phi2-write-sweep")]
116pub static READ_PHI_BACKOFF: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
117/// WRITE access split — swapped (writes commit `2 * PPU_OFFSET` mc later than
118/// reads). NTSC divisor 12 → (7, 5), byte-identical to the prior `const`s.
119///
120/// A 6502 commits a write at phi2, the LAST of a CPU cycle's three PPU dots.
121/// At the shipped `pre` of 7 (minus `PPU_OFFSET`) the PPU has advanced 6 of
122/// 12 master clocks — 1.5 dots — so the commit lands mid-cycle instead.
123///
124/// **That is a known divergence, measured and deliberately NOT corrected at
125/// v2.6.18.** Moving the commit to phi2 is the right diagnosis and was the
126/// wrong change as applied: the write alone reads 141/144 on `AccuracyCoin`
127/// and fails `ppu_vbl_nmi/10-even_odd_timing`; the best combination found
128/// (phi2 + the dot-321 OAM2 increment + a one-stage `mask_for_skip_check`)
129/// reads 142/144 against the 143/144 that ships. Several PPU behaviours
130/// compensate for the placement — `mask_for_skip_check` says so in its own
131/// comment — and replacing a documented compensation with an undocumented one
132/// is worse than keeping it. The `phi2-write-sweep` feature keeps the knob so
133/// the next attempt re-measures rather than rebuilding the apparatus; see the
134/// *CLOSED* section of `to-dos/plans/v2.6.18-terminus-plan.md` for every
135/// number and the three conditions for reopening it.
136#[cfg(not(feature = "phi2-write-sweep"))]
137#[inline]
138const fn write_split(div: u64) -> (u64, u64) {
139 let pre = div / 2 + PPU_OFFSET;
140 (pre, div - pre)
141}
142
143/// Sweepable `write_split` (feature `phi2-write-sweep`, v2.6.18 study only).
144///
145/// Default 0 reproduces the `const` path above exactly, so enabling the
146/// feature without setting the knob is byte-identical. Kept behind a feature
147/// rather than always-on because `write_split` runs on EVERY write, and the
148/// shipped build should not pay an atomic load per write for a study.
149#[cfg(feature = "phi2-write-sweep")]
150#[inline]
151fn write_split(div: u64) -> (u64, u64) {
152 let extra = u64::from(WRITE_PHI_OFFSET.load(core::sync::atomic::Ordering::Relaxed));
153 let back = u64::from(WRITE_PHI_BACKOFF.load(core::sync::atomic::Ordering::Relaxed));
154 // Clamp so `pre` never reaches the cycle length: `post` must stay >= 1 or
155 // `end_cycle` would not advance the master clock at all. The lower clamp
156 // matters for the same reason once `back` can pull the access earlier.
157 // See `read_split` for why each step saturates and why the upper clamp
158 // bound is floored at 1.
159 let pre = (div / 2 + PPU_OFFSET + extra)
160 .saturating_sub(back)
161 .clamp(1, div.saturating_sub(1).max(1));
162 (pre, div - pre)
163}
164
165/// Extra master clocks added to a WRITE's pre-access split (v2.6.18 study).
166/// 0 = shipped behaviour.
167#[cfg(feature = "phi2-write-sweep")]
168pub static WRITE_PHI_OFFSET: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
169
170/// Master clocks SUBTRACTED from a WRITE's pre-access split (v2.6.18 study).
171/// The counterpart of [`READ_PHI_BACKOFF`]; see it for why subtraction is
172/// needed at all.
173#[cfg(feature = "phi2-write-sweep")]
174pub static WRITE_PHI_BACKOFF: core::sync::atomic::AtomicU8 = core::sync::atomic::AtomicU8::new(0);
175
176/// NMI vector low byte address (`$FFFA/B`).
177const NMI_VECTOR: u16 = 0xFFFA;
178
179/// Reset vector low byte address (`$FFFC/D`).
180const RESET_VECTOR: u16 = 0xFFFC;
181
182/// IRQ / BRK vector low byte address (`$FFFE/F`).
183const IRQ_VECTOR: u16 = 0xFFFE;
184
185/// 6502 CPU core.
186//
187// Multiple boolean state bits track distinct interrupt-pipeline stages
188// (jam, NMI pending, NMI armed, IRQ pending, IRQ armed). These map directly
189// onto orthogonal hardware latches; collapsing them into an enum or bitflags
190// would obscure rather than clarify the model.
191#[allow(clippy::struct_excessive_bools)]
192#[derive(Debug, Clone)]
193pub struct Cpu {
194 /// Accumulator.
195 pub a: u8,
196 /// X index register.
197 pub x: u8,
198 /// Y index register.
199 pub y: u8,
200 /// Program counter.
201 pub pc: u16,
202 /// Stack pointer (low byte; effective address `0x0100 | s`).
203 pub s: u8,
204 /// Processor status.
205 pub p: Status,
206 /// Cumulative CPU cycle count.
207 pub cycles: u64,
208 /// `true` when the CPU has executed a JAM/KIL/STP and is waiting for reset.
209 pub jammed: bool,
210 /// Edge-detected NMI latch. Real hardware samples the NMI line at the
211 /// second-to-last cycle of every instruction; if asserted, the NMI
212 /// sequence is queued for AFTER the current instruction completes. Our
213 /// `step` model dispatches all bus operations atomically before ticking
214 /// cycles, so a write that itself raises NMI (e.g. enabling NMI in
215 /// `$2000` while VBL is set) is observed by the bus's edge detector
216 /// during this instruction's cycle tally. Hardware would not have
217 /// observed it at the second-to-last cycle (the write logically happens
218 /// at the LAST cycle), so the NMI is queued for the NEXT instruction's
219 /// sample point and serviced AFTER the next instruction completes.
220 /// We approximate that by promoting `pending_nmi` to `armed_nmi` after
221 /// each instruction; only `armed_nmi` actually services.
222 pub(crate) pending_nmi: bool,
223 /// NMI ready to be serviced before the next instruction starts.
224 pub(crate) armed_nmi: bool,
225 /// IRQ pending — captured edge of the level line that fires once the
226 /// I-flag is clear. Same double-latch promotion as NMI.
227 pub(crate) pending_irq: bool,
228 /// IRQ ready to be serviced before the next instruction starts.
229 pub(crate) armed_irq: bool,
230 /// First tick (within the current instruction's tally loop) at which
231 /// NMI was sampled high; `u8::MAX` if not seen. Used by the
232 /// second-to-last-cycle interrupt classification.
233 pub(crate) nmi_first_tick: u8,
234 /// First tick at which IRQ was sampled high; `u8::MAX` if not seen.
235 pub(crate) irq_first_tick: u8,
236 /// Snapshot of the I flag at the *start* of the current instruction.
237 /// Hardware samples IRQ near the end of the instruction with the old
238 /// I-flag value; CLI / SEI / PLP / RTI take effect at the very last
239 /// cycle, AFTER the IRQ sample point. We arm IRQ only when this
240 /// snapshot says I was clear at sample time.
241 pub(crate) irq_sample_i_flag: bool,
242 /// Number of cycles emitted by the per-cycle helpers
243 /// (`read1`/`write1`/`idle_tick`) within the *current* instruction.
244 /// Reset to zero at the top of `step()`. Used by the trailing
245 /// "burn remaining cycles" loop so we can incrementally migrate
246 /// opcodes from atomic-dispatch + trailing-loop to fully per-cycle
247 /// emission.
248 pub(crate) cycles_emitted: u8,
249 /// When `true`, [`Cpu::idle_tick`] does NOT update `irq_first_tick`.
250 /// Used by branch opcodes to model the `branch_delays_irq` quirk:
251 /// real 6502 branches poll IRQ at the same point a 2-cycle untaken
252 /// branch would (the opcode-fetch cycle, which `step()` performs
253 /// before entering dispatch). The operand fetch and any extra
254 /// taken / page-cross cycles do *not* re-sample IRQ. The branch
255 /// dispatch sets this flag *before* the operand fetch and `step()`
256 /// clears it at the top of every instruction.
257 /// Since v2.6.7 the same rule defers NMI *dispatch*: while this flag and
258 /// `skip_irq_sample_q` are both set, `handle_interrupts` freezes the
259 /// dispatch copy `mc_prev_need_nmi`. The NMI edge latch (`mc_need_nmi`)
260 /// keeps running every cycle, so an edge is never lost, only recognised
261 /// after the branch.
262 pub(crate) skip_irq_sample: bool,
263 /// `skip_irq_sample` as it stood on the PREVIOUS cycle. The NMI dispatch
264 /// gate freezes on this rather than on the live flag, because the two
265 /// latches have different shapes: `mc_run_irq` is RECOMPUTED from the live
266 /// line every cycle, so freezing it in place holds its end-of-C1 value,
267 /// while `mc_need_nmi` is STICKY and it is the *copy* that has to freeze --
268 /// one cycle later, or the copy that captures end-of-C1 never happens.
269 pub(crate) skip_irq_sample_q: bool,
270
271 // === v2.0 master-clock R1 substrate (Phase 2; `mc-r1-substrate`) ===
272 /// The CPU's authoritative master clock (Mesen `_masterClock` / `TetaNES`
273 /// `Cpu::master_clock`). Advanced by `start_cycle`/`end_cycle`; the bus is
274 /// caught up to `master_clock - PPU_OFFSET` from BOTH halves (double
275 /// catch-up). Only live under `mc-r1-substrate`.
276 pub(crate) master_clock: u64,
277 /// NMI edge-recognition latch (set on a /NMI rising edge in
278 /// `handle_interrupts`; consumed by the cycle-5 hijack in
279 /// `service_interrupt`). Mesen `_needNmi`.
280 pub(crate) mc_need_nmi: bool,
281 /// One-cycle-delayed copy of `mc_need_nmi` (the dispatch + hijack gate).
282 /// Mesen `_prevNeedNmi`.
283 pub(crate) mc_prev_need_nmi: bool,
284 /// Live IRQ-recognition latch (`irq_level && !irq_sample_i_flag`,
285 /// recomputed every `end_cycle`). Mesen `_runIrq`.
286 pub(crate) mc_run_irq: bool,
287 /// One-cycle-delayed copy of `mc_run_irq` (the dispatch gate). Mesen
288 /// `_prevRunIrq`.
289 pub(crate) mc_prev_run_irq: bool,
290 /// Previous /NMI line level, for the φ2 rising-edge detector in
291 /// `handle_interrupts`. Mesen `_prevNmiFlag`.
292 pub(crate) mc_prev_nmi_line: bool,
293 /// v2.0.0 beta.2 (A2 scoping diagnostic): per-opcode count of cycles the
294 /// trailing burn-loop had to fill (`cycles - cycles_emitted`) — the exact
295 /// remaining busless-cycle surface the every-cycle-bus-access conversion
296 /// must turn into dummy reads of the held address. Read by the harness
297 /// `burn_probe` bin; never consulted by emulation.
298 #[cfg(feature = "cpu-instr-cycle-trace")]
299 pub burn_histogram: [u64; 256],
300}
301
302impl Default for Cpu {
303 fn default() -> Self {
304 Self::new()
305 }
306}
307
308/// Effective address + page-crossed flag, returned by addressing-mode resolvers.
309#[derive(Clone, Copy)]
310struct Operand {
311 addr: u16,
312 page_crossed: bool,
313}
314
315impl Cpu {
316 /// New CPU in "post-reset" state. Caller must invoke [`Cpu::reset`] with a
317 /// real bus before stepping (PC is undefined until reset reads `$FFFC/D`).
318 ///
319 /// This constructor is the convenience entry-point used by unit tests and
320 /// nestest fixtures that drive the CPU without going through a full
321 /// power-on path: `S=$FD`, `P=$24` (`UNUSED` + `INTERRUPT_DISABLE`). If the
322 /// caller subsequently invokes [`Cpu::reset`] the stack pointer will be
323 /// decremented by 3 (per the reset sequence), landing on `$FA` — that is
324 /// the input shape several `tests/opcodes.rs` fixtures expect.
325 ///
326 /// **For the real cold-boot path** (`Nes::from_rom`, `Nes::power_cycle`),
327 /// use [`Cpu::power_on`] instead, which seeds `S=$00`. After the 3-decrement
328 /// reset sequence that lands `S=$FD`, matching Mesen2's power-up state.
329 /// See `docs/audit/session-13-cpu-boot-fix-2026-05-21.md` for the reference
330 /// behaviour from `Core/NES/NesCpu.cpp::NesCpu::Reset(softReset=false)`.
331 #[must_use]
332 pub const fn new() -> Self {
333 Self {
334 a: 0,
335 x: 0,
336 y: 0,
337 pc: 0,
338 s: 0xFD,
339 p: Status::power_on(),
340 cycles: 0,
341 jammed: false,
342 pending_nmi: false,
343 armed_nmi: false,
344 pending_irq: false,
345 armed_irq: false,
346 nmi_first_tick: u8::MAX,
347 irq_first_tick: u8::MAX,
348 irq_sample_i_flag: true,
349 cycles_emitted: 0,
350 skip_irq_sample: false,
351 skip_irq_sample_q: false,
352 master_clock: 0,
353 mc_need_nmi: false,
354 mc_prev_need_nmi: false,
355 mc_run_irq: false,
356 mc_prev_run_irq: false,
357 mc_prev_nmi_line: false,
358 #[cfg(feature = "cpu-instr-cycle-trace")]
359 burn_histogram: [0; 256],
360 }
361 }
362
363 /// New CPU in real-hardware cold-boot state (`S=$00`).
364 ///
365 /// Real silicon comes up with the stack pointer in an undefined state;
366 /// the convention used by Mesen2 (and adopted here for trace parity) is
367 /// to treat power-up as `S=$00` and rely on the reset sequence's three
368 /// "phantom" decrements to wrap into `S=$FD`. See Mesen2
369 /// `Core/NES/NesCpu.cpp::Reset(softReset=false)`:
370 ///
371 /// ```cpp
372 /// if(softReset) {
373 /// _state.SP -= 0x03; // soft reset path
374 /// } else {
375 /// _state.SP = 0xFD; // power-up: direct assignment
376 /// }
377 /// ```
378 ///
379 /// `RustyNES` models that two-path behaviour by gating the SP delta through
380 /// the constructor: `Cpu::power_on() + reset()` ⇒ `$00 - 3 = $FD` (cold);
381 /// `cpu.reset()` again ⇒ `$FD - 3 = $FA` (subsequent soft reset).
382 ///
383 /// `P` is left at `$24` (`INTERRUPT_DISABLE` | `UNUSED`). Mesen2's trace
384 /// surface shows `P = $04` because it masks `UNUSED` out of the displayed
385 /// byte, but the bit is conventionally always set on a 6502 internally
386 /// (nesdev: "Bit 5: Always 1, the so-called 'unused' bit"); the trace
387 /// divergence on P is cosmetic.
388 #[must_use]
389 pub const fn power_on() -> Self {
390 let mut cpu = Self::new();
391 cpu.s = 0x00;
392 cpu
393 }
394
395 /// Returns `true` when the CPU has executed a JAM/KIL/STP.
396 #[must_use]
397 pub const fn is_jammed(&self) -> bool {
398 self.jammed
399 }
400
401 /// Read-only accessor for the CPU's authoritative master clock
402 /// (v2.0.0-beta.1 one-clock instrumentation).
403 ///
404 /// `master_clock` counts master-clock units (NTSC: 12 per CPU cycle,
405 /// PAL: 16, Dendy: 15) and is advanced only by `start_cycle` /
406 /// `end_cycle` (the asymmetric read 5/7 vs write 7/5 φ1/φ2 split on
407 /// NTSC). It is the counter the v2.0.0
408 /// "Timebase" rewrite (ADR 0002) promotes to the ONE canonical
409 /// timebase; the test harness asserts the affine relation
410 /// `master_clock == seed + cpu_divider * cycles` against the other
411 /// cycle counters (`one_clock_invariants.rs`) as the gate for the
412 /// beta.1 counter collapse.
413 #[must_use]
414 pub const fn master_clock(&self) -> u64 {
415 self.master_clock
416 }
417
418 /// Reset (warm boot).
419 ///
420 /// Real hardware: 8-cycle sequence with suppressed pushes, then PC loads
421 /// from the reset vector and the I flag is set. We model the cycle count
422 /// (advances `cycles` by 8 and fires `on_cpu_cycle` 8 times) without
423 /// mutating registers other than P (set I), S (decrement by 3), and PC.
424 ///
425 /// Matches Mesen2's `NesCpu::Reset()` 8-cycle post-power-up loop ("CPU
426 /// takes 8 cycles before it starts executing the ROM's code"). Combined
427 /// with the PPU power-up at (scanline=-1, dot=340) (see `Ppu::new`),
428 /// this closes the +344-dot PPU offset identified empirically in
429 /// Session-13 (docs/audit/session-13-cpu-boot-fix-2026-05-21.md).
430 pub fn reset<B: Bus>(&mut self, bus: &mut B) {
431 // Real hardware decrements S three times during reset (no actual
432 // pushes occur, but the decrements happen).
433 self.s = self.s.wrapping_sub(3);
434 self.p.insert(Status::INTERRUPT_DISABLE);
435 self.jammed = false;
436 self.pending_nmi = false;
437 self.armed_nmi = false;
438 self.pending_irq = false;
439 self.armed_irq = false;
440 self.nmi_first_tick = u8::MAX;
441 self.irq_first_tick = u8::MAX;
442 // R1/R3 cold-boot: advance master_clock by one CPU divider BEFORE the
443 // 8-cycle reset loop (Mesen `NesCpu::Reset()` `_masterClock += cpuDivider
444 // + cpuOffset`). Without it the first start_cycle's `run_ppu_to(mc-1)`
445 // would leave the PPU 3 dots behind. See R1 port plan / branch `acddd22`.
446 {
447 self.master_clock = self.master_clock.wrapping_add(bus.cpu_divider());
448 }
449 // V-axis: Mesen adds `cpuDivider + cpuOffset` (13), not just cpuDivider
450 // (12). The +PPU_OFFSET corrects R1's power-up CPU/PPU sub-cycle
451 // alignment to the reference, shifting every $2002 poll-exit to match
452 // hardware (nesdev `PPU_frame_timing`: the read sees the flag change iff
453 // it starts at/after the set tick). Default-off; A/B against Y/C1/6-10.
454 // 8-cycle reset sequence: 6 idle/internal cycles + 2 vector reads.
455 self.cycles_emitted = 0;
456 for _ in 0..6 {
457 self.idle_tick(bus);
458 }
459 let lo = self.read1(bus, RESET_VECTOR);
460 let hi = self.read1(bus, RESET_VECTOR + 1);
461 self.pc = u16::from(lo) | (u16::from(hi) << 8);
462 }
463
464 /// Force PC to `addr`. Used by the nestest harness which enters at
465 /// `$C000` rather than the reset vector.
466 pub const fn set_pc(&mut self, addr: u16) {
467 self.pc = addr;
468 }
469
470 /// Step one instruction (or service an interrupt). Returns the number of
471 /// CPU cycles consumed.
472 ///
473 /// On a JAM-state CPU this is a no-op returning 0.
474 ///
475 /// # Interrupt timing model
476 ///
477 /// Real 6502 hardware samples the NMI / IRQ lines at the *second-to-last*
478 /// cycle of every instruction and, if asserted there, queues the
479 /// interrupt to be serviced *after* the current instruction completes.
480 /// Our model dispatches all bus operations atomically before ticking
481 /// cycles, so a write that itself raises NMI (e.g. `STA $2000` enabling
482 /// NMI while VBL is set) appears to the bus's edge detector during the
483 /// FIRST cycle of the tally loop — earlier than hardware would observe
484 /// it. Hardware places the actual write at the *last* cycle of the
485 /// instruction, so the second-to-last sample point would NOT see the new
486 /// line state; only the NEXT instruction's sample sees it. We model
487 /// that by introducing a one-instruction promotion delay: edges captured
488 /// at end-of-step land in `pending_*` and, after the following step,
489 /// promote to `armed_*` which is the gate that actually triggers
490 /// service. This passes `04-nmi_control` test 11 ("Immediate occurence
491 /// should be after NEXT instruction") without regressing the
492 /// instruction-count-insensitive tests like `02-vbl_set_time`,
493 /// `09-even_odd_frames`, or any `instr_test_v5` ROM (which never raise
494 /// NMI from within a single instruction).
495 #[allow(clippy::too_many_lines, clippy::missing_panics_doc)]
496 pub fn step<B: Bus>(&mut self, bus: &mut B) -> u8 {
497 if self.jammed {
498 return 0;
499 }
500 // v2.0 master-clock R1 UNIFIED interrupt dispatch (`mc-r1-substrate`):
501 // a SINGLE service sequence gated on the one-cycle-delayed `prev_*`
502 // copies (Mesen `_prevRunIrq || _prevNeedNmi`). The vector is chosen
503 // INSIDE `service_interrupt` by the live `mc_need_nmi` at cycle 5 (the
504 // NMI hijack); NMI priority is resolved there, not here. Setting
505 // `irq_sample_i_flag = true` BEFORE the service masks the φ2 sampler
506 // for the 7-cycle sequence (no re-entry). Clears `skip_irq_sample`
507 // (a prior taken-branch could have left it set, freezing the recompute)
508 // and defers any still-pending NMI by one instruction.
509 if self.mc_prev_run_irq || self.mc_prev_need_nmi {
510 self.armed_irq = false;
511 self.irq_sample_i_flag = true;
512 self.skip_irq_sample = false;
513 self.service_interrupt(bus, IRQ_VECTOR, false);
514 self.mc_prev_need_nmi = false;
515 self.promote_post_step_interrupts(7);
516 return 7;
517 }
518 // Service an armed interrupt before the next instruction. NMI has
519 // priority over IRQ; both are mutually exclusive for a single
520 // service window.
521 // Once armed, the IRQ services unconditionally — the I-flag
522 // gating already happened at the sample point (second-to-last
523 // cycle of the prior instruction). This is what produces the
524 // "CLI SEI should still allow one IRQ to fire" behavior:
525 // SEI's I=1 takes effect at end-of-SEI but the sample at SEI's
526 // second-to-last cycle saw I=0 (CLI cleared it) and queued the
527 // IRQ, which now fires regardless of the current I-flag.
528
529 // Per-instruction state for the per-cycle helpers
530 // (`read1`/`write1`/`idle_tick`). These track the FIRST tick at
531 // which each interrupt line was seen high; hardware samples at the
532 // second-to-last cycle so seen < last_tick = arm now;
533 // seen == last_tick = defer one instruction (the next instruction's
534 // sample window catches it instead).
535 self.nmi_first_tick = u8::MAX;
536 self.irq_first_tick = u8::MAX;
537 self.cycles_emitted = 0;
538 // Cleared every instruction; set inside the branch dispatch arms
539 // (after the operand fetch / canonical IRQ poll) to suppress
540 // further IRQ sampling on the additional taken / page-cross
541 // branch cycles.
542 self.skip_irq_sample = false;
543 // Snapshot the I flag for this instruction. CLI / SEI / PLP /
544 // RTI mutate `self.p` *during* the instruction, but the hardware
545 // IRQ sample reads the I value as it was at the start. This is
546 // what produces the documented "exactly one instruction after
547 // CLI executes before IRQ is taken" delay.
548 self.irq_sample_i_flag = self.p.contains(Status::INTERRUPT_DISABLE);
549
550 #[cfg(feature = "cpu-instr-cycle-trace")]
551 bus.trace_instr(self.pc, self.cycles);
552
553 let opcode = self.fetch_pc(bus);
554 let mut cycles = 0u8;
555 self.dispatch(bus, opcode, &mut cycles);
556 // Burn whichever cycles the dispatch did NOT emit through helpers.
557 // As opcodes migrate to fully per-cycle emission, this loop runs
558 // for fewer iterations; eventually it can be removed entirely.
559 //
560 // v2.0.0 beta.2 (A2 scoping): the diagnostic histogram below records,
561 // per opcode, how many cycles the burn-loop had to fill — the exact
562 // empirical work list for the every-cycle-bus-access conversion (the
563 // remaining busless cycles that must become dummy reads of the held
564 // address). Default-off; the `burn_probe` harness bin prints it.
565 #[cfg(feature = "cpu-instr-cycle-trace")]
566 {
567 let burned = cycles.saturating_sub(self.cycles_emitted);
568 if burned > 0 {
569 self.burn_histogram[opcode as usize] =
570 self.burn_histogram[opcode as usize].saturating_add(u64::from(burned));
571 }
572 }
573 // v2.0.0 beta.2 (A2, promoted to the only path in beta.4): every
574 // instruction cycle is a bus access — the resolvers + RMW arms emit
575 // the canonical dummy reads, so the burn-loop must never fire.
576 // Proven empirically at zero across AccuracyCoin, nestest, both
577 // blargg_nes_cpu_test5 suites, and cpu_timing_test6 (the full
578 // official + unofficial opcode space); this assert makes any future
579 // under-emitting dispatch arm fail loud in dev-profile runs instead
580 // of silently reintroducing a busless cycle.
581 debug_assert!(
582 self.cycles_emitted >= cycles,
583 "opcode ${opcode:02X} under-emitted: declared {cycles} cycles but emitted \
584 only {} — a busless burn-loop cycle would fill the gap (A2 regression; \
585 see the v2.0.0 plan Workstream A2)",
586 self.cycles_emitted
587 );
588 while self.cycles_emitted < cycles {
589 self.idle_tick(bus);
590 }
591 self.promote_post_step_interrupts(cycles);
592 cycles
593 }
594
595 /// Promote any per-instruction interrupt edges captured by the
596 /// per-cycle helpers into the `armed_*` / `pending_*` latches the
597 /// next [`Cpu::step`] consults. Hardware samples interrupts at the
598 /// second-to-last cycle of an instruction; we approximate that with
599 /// "first sampled tick strictly before the last cycle = arm now,
600 /// else defer one instruction."
601 const fn promote_post_step_interrupts(&mut self, cycles: u8) {
602 // Promote any previously-pending interrupt (latched at the very
603 // last cycle of the prior instruction).
604 if self.pending_nmi {
605 self.armed_nmi = true;
606 self.pending_nmi = false;
607 }
608 if self.pending_irq {
609 self.armed_irq = true;
610 self.pending_irq = false;
611 }
612 let last_tick = cycles.saturating_sub(1);
613 if self.nmi_first_tick != u8::MAX {
614 if self.nmi_first_tick < last_tick {
615 self.armed_nmi = true;
616 } else {
617 self.pending_nmi = true;
618 }
619 }
620 if self.irq_first_tick != u8::MAX {
621 // IRQ is masked by the I-flag value as it was at the START of
622 // this instruction; CLI / SEI / PLP / RTI mutations take effect
623 // at end-of-instruction. If IRQ was already disabled when we
624 // entered, the second-to-last-cycle sample sees I=1 and the
625 // edge is dropped (the next instruction's sample will pick it
626 // up if I has since cleared).
627 if !self.irq_sample_i_flag {
628 if self.irq_first_tick < last_tick {
629 self.armed_irq = true;
630 } else {
631 self.pending_irq = true;
632 }
633 }
634 }
635 }
636
637 fn fetch_pc<B: Bus>(&mut self, bus: &mut B) -> u8 {
638 let v = self.read1(bus, self.pc);
639 self.pc = self.pc.wrapping_add(1);
640 v
641 }
642
643 fn fetch_pc_u16<B: Bus>(&mut self, bus: &mut B) -> u16 {
644 let lo = self.fetch_pc(bus);
645 let hi = self.fetch_pc(bus);
646 u16::from(lo) | (u16::from(hi) << 8)
647 }
648
649 fn read_u16_with_wrap<B: Bus>(&mut self, bus: &mut B, addr: u16) -> u16 {
650 // Used by indirect modes to honor the 6502 page-wrap quirk.
651 let lo = self.read1(bus, addr);
652 let hi_addr = (addr & 0xFF00) | u16::from((addr as u8).wrapping_add(1));
653 let hi = self.read1(bus, hi_addr);
654 u16::from(lo) | (u16::from(hi) << 8)
655 }
656
657 fn push<B: Bus>(&mut self, bus: &mut B, value: u8) {
658 self.write1(bus, STACK_BASE | u16::from(self.s), value);
659 self.s = self.s.wrapping_sub(1);
660 }
661
662 fn pull<B: Bus>(&mut self, bus: &mut B) -> u8 {
663 self.s = self.s.wrapping_add(1);
664 self.read1(bus, STACK_BASE | u16::from(self.s))
665 }
666
667 fn push_u16<B: Bus>(&mut self, bus: &mut B, value: u16) {
668 self.push(bus, (value >> 8) as u8);
669 self.push(bus, (value & 0xFF) as u8);
670 }
671
672 fn pull_u16<B: Bus>(&mut self, bus: &mut B) -> u16 {
673 let lo = self.pull(bus);
674 let hi = self.pull(bus);
675 u16::from(lo) | (u16::from(hi) << 8)
676 }
677
678 // ------------------------------------------------------------------
679 // Per-cycle bus interleaving primitives.
680 //
681 // Real 6502 hardware reads or writes the bus *exactly once per CPU
682 // cycle*; "internal" cycles (ALU work, stack pointer increment, etc.)
683 // still tick the system clock without driving the address bus.
684 // `read1` / `write1` / `idle_tick` model that one-cycle granularity:
685 // each one ticks the bus exactly once and samples the NMI/IRQ lines
686 // at the end of the cycle, mirroring what the existing trailing
687 // tally loop in `step` did but with the polling now coupled to the
688 // *actual* memory access ordering.
689 //
690 // `step()` resets `cycles_emitted` to 0; each call here increments
691 // it. The trailing burn-loop in `step` consumes whatever cycles
692 // the opcode declared but didn't emit through these helpers.
693 // ------------------------------------------------------------------
694
695 // === v2.0 master-clock R1 substrate core (Phase 2; `mc-r1-substrate`) ===
696
697 /// Start half of one CPU cycle: advance `master_clock` by the PRE split,
698 /// catch the PPU up to `master_clock - PPU_OFFSET`, then fire the bus's
699 /// per-cycle work (`cpu_clock`). After this the PPU is at the access's
700 /// exact master clock (so a `$2002`/`$2007` read sees on-time state).
701 fn start_cycle<B: Bus>(&mut self, bus: &mut B, for_read: bool) {
702 let div = bus.cpu_divider();
703 let pre = if for_read {
704 read_split(div).0
705 } else {
706 write_split(div).0
707 };
708 self.master_clock = self.master_clock.wrapping_add(pre);
709 bus.run_ppu_to(self.master_clock.saturating_sub(ppu_sample_offset()), false);
710 bus.cpu_clock();
711 // v2.0.0 beta.1 (A1 one-clock collapse, promoted to the only path in
712 // beta.4): `cycles` is ASSIGNED from the canonical bus cycle counter
713 // at this single per-cycle site instead of being independently
714 // incremented by every `read1`/`write1`/`idle_tick`/DMA-loop caller.
715 // `bus.cpu_clock()` above advanced the canonical counter for THIS
716 // cycle, so the assignment lands on the same post-increment value the
717 // legacy caller-side `+= 1` produced (the `one_clock_invariants`
718 // harness test pins the residue at zero).
719 self.cycles = bus.cycle_count();
720 }
721
722 /// End half of one CPU cycle: advance by the POST split, catch the PPU
723 /// up again (the double catch-up), then sample interrupts (φ2, the
724 /// T_last-1 rule).
725 ///
726 /// Every DMA cycle is a first-class `start_cycle`/`end_cycle` on the
727 /// unified-DMA path, advancing `master_clock` directly, so there is no
728 /// bus-side DMA span to fold in. The `take_dma_mc_consumed` fold that
729 /// once did that was retired at v2.0.0 beta.1 (it only ever mattered for
730 /// the pre-v2.0.0 bus-side burst engine) and its hook was removed at
731 /// v2.9.8 with the rest of that engine's dead code (ADR 0042).
732 fn end_cycle<B: Bus>(&mut self, bus: &mut B, for_read: bool) {
733 let div = bus.cpu_divider();
734 let post = if for_read {
735 read_split(div).1
736 } else {
737 write_split(div).1
738 };
739 self.master_clock = self.master_clock.wrapping_add(post);
740 bus.run_ppu_to(self.master_clock.saturating_sub(ppu_sample_offset()), true);
741 // F-2: tick the DMC byte-timer at END of cycle (after the access),
742 // matching main's DMC fire-phase for DMASync, BEFORE the φ2 interrupt
743 // sample so handle_interrupts sees the post-tick DMC IRQ line.
744 bus.cpu_clock_apu_dmc();
745 self.handle_interrupts(bus);
746 // Diagnostic trace hook (no-op unless the bus enables irq-timing-trace).
747 bus.trace_end_cycle();
748 }
749
750 /// φ2 interrupt sampler (Mesen `EndCpuCycle`): edge-detect /NMI into
751 /// `mc_need_nmi` (after copying the one-cycle-delayed `mc_prev_need_nmi`),
752 /// and recompute `mc_run_irq = irq_level && !irq_sample_i_flag` (after the
753 /// `mc_prev_run_irq` copy). The `step()`-top dispatch reads the `prev_*`
754 /// copies — i.e. second-to-last-cycle recognition. The I-mask uses the
755 /// start-of-instruction snapshot (`irq_sample_i_flag`), not live `self.p`,
756 /// so CLI/SEI/PLP delay their I-change one instruction.
757 #[allow(clippy::needless_pass_by_ref_mut)] // &mut B for signature parity
758 fn handle_interrupts<B: Bus>(&mut self, bus: &mut B) {
759 // EXPERIMENT (v2.6.7): the taken-branch poll rule applies to NMI too.
760 //
761 // `CPU_interrupts.xhtml`: "Interrupts are always polled before the
762 // second CPU cycle (the operand fetch), but NOT before the third CPU
763 // cycle on a taken branch." It says *interrupts*, not *IRQs*.
764 //
765 // The edge LATCH (`mc_need_nmi`, below) keeps running every cycle --
766 // that is hardware's always-on edge detector and must not change. What
767 // freezes is the DISPATCH gate, exactly as `mc_run_irq` already does
768 // while `skip_irq_sample` is set.
769 if !(self.skip_irq_sample && self.skip_irq_sample_q) {
770 self.mc_prev_need_nmi = self.mc_need_nmi;
771 }
772 self.skip_irq_sample_q = self.skip_irq_sample;
773 let nmi_level = bus.nmi_level();
774 if !self.mc_prev_nmi_line && nmi_level {
775 self.mc_need_nmi = true;
776 }
777 self.mc_prev_nmi_line = nmi_level;
778 self.mc_prev_run_irq = self.mc_run_irq;
779 // W1 (`mc-r1-branch-poll-points`): a taken branch polls IRQ ONCE —
780 // before C2 — so while `skip_irq_sample` is set (the branch dispatch
781 // arms set it after the C1 opcode fetch, before the C2 operand fetch)
782 // the recognition latch is FROZEN at its end-of-C1 value instead of
783 // recomputed from the live line every cycle. DMC-DMA halt cycles
784 // drained inside the branch's own `read1`/`idle_tick` therefore
785 // cannot make a freshly-asserted IRQ visible to THIS instruction
786 // (AccuracyCoin `Interrupt flag latency` Test A; TriCNES polls at
787 // C2-start only, plus a can-set poll at C4-start handled in
788 // `branch()`). The `mc_prev_run_irq` copy above still runs, so the
789 // held end-of-C1 value is what the next `step()` dispatch reads. NMI
790 // edge detection above is untouched (sampled every cycle, per
791 // hardware — the quirk is IRQ-only).
792 if self.skip_irq_sample {
793 return;
794 }
795 let irq_level = bus.irq_level();
796 self.mc_run_irq = irq_level && !self.irq_sample_i_flag;
797 }
798
799 /// Tick the bus once and sample interrupt lines, *without* a bus
800 /// access. Models a 6502 internal cycle.
801 fn idle_tick<B: Bus>(&mut self, bus: &mut B) {
802 {
803 // F-2 re-coupling (`mc-r1-dmc-idle-halt`): a DMC DMA can halt the CPU
804 // on a 6502 INTERNAL cycle too — on hardware every cycle is a bus
805 // read, and Mesen's `ProcessPendingDma` runs on every `MemoryRead`
806 // (incl. dummy reads), NOT only instruction/operand reads. R1's
807 // `read1` loop only services on real reads, so the DMA waits through
808 // internal cycles (the `lat=4` idle-runs the per-fetch trace pinned
809 // as the period-jitter source). Service it here too, on the held
810 // (last-read) bus address. Default-off; the banked 6/10 path skips it.
811 // W3-Stage-1 (`mc-r1-dma-unified`): the unified-engine replacement
812 // for the idle DMC drain above — same loop shape, ONE engine. The
813 // bus supplies the held (last-read) address for the parked 6502
814 // address bus. Same budget accounting as the loop it replaces.
815 while bus.unified_dma_pending() {
816 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
817 self.start_cycle(bus, true);
818 bus.unified_dma_cycle_idle();
819 self.end_cycle(bus, true);
820 }
821 // R1: a pure internal cycle — busless (idle_tick stays busless).
822 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
823 self.start_cycle(bus, true);
824 self.end_cycle(bus, true);
825 }
826 }
827
828 /// Canonical cycle-2 PC dummy read for implied / accumulator /
829 /// transfer / flag instructions (per nesdev `6502_cpu.txt` + MOS
830 /// 6502 datasheet). Real silicon fetches the byte AFTER the opcode
831 /// during cycle 2 of these single-byte instructions and discards
832 /// it (the would-be operand). Without this dummy read, our emulator
833 /// instead "burns an idle cycle" via `idle_tick` for the second
834 /// cycle, which counts the cycle for time but produces no bus
835 /// access — diverging from real silicon's bus-access pattern.
836 ///
837 /// Wired into 22 dispatch arms (ASL/LSR/ROL/ROR A; CLC/SEC/CLI/SEI/
838 /// CLV/CLD/SED; TAX/TAY/TSX/TXA/TXS/TYA; INX/DEX/INY/DEY; NOP;
839 /// 6 unofficial 1-byte NOPs), **unconditionally**.
840 ///
841 /// This was gated behind a `cpu-implied-dummy-reads` cargo feature,
842 /// default-off pending the DMC-scheduler audit in
843 /// `docs/audit/sprint-2.3-implied-dummy-dmc-recon-2026-05-25.md`. That
844 /// gate no longer exists: the flag is declared in no manifest and there
845 /// is no `cfg` on this helper, so every build performs the dummy read.
846 /// The text describing an OFF branch that "compiles to a no-op" survived
847 /// the promotion and is removed here — it described the shipped default
848 /// as disabled when it is unconditional, which is the most misleading
849 /// shape a stale comment can take.
850 // Only `inline_always` still binds: `needless_pass_by_ref_mut`,
851 // `unused_self` and `missing_const_for_fn` were for the deleted OFF
852 // branch, and stripping all four re-lints with `inline_always` as the
853 // sole finding. Measured rather than reasoned, per the v2.3.9 sweep that
854 // found 25 of 29 `allow`s suppressing nothing.
855 #[inline(always)]
856 #[allow(clippy::inline_always)]
857 fn implied_dummy_read<B: Bus>(&mut self, bus: &mut B) {
858 let _ = self.read1(bus, self.pc);
859 }
860
861 /// Read a byte at `addr` *and* consume one CPU cycle (with bus tick
862 /// + interrupt sampling).
863 #[allow(clippy::too_many_lines)] // mc-r1 DMA-interleave arms push this past 100
864 fn read1<B: Bus>(&mut self, bus: &mut B, addr: u16) -> u8 {
865 {
866 // accuracycoin-100 Phase 2 (`mc-r1-dmc-abort-cancel`): a 1-byte
867 // non-looping implicit abort that matured during the prior APU tick
868 // is serviced HERE, before the (cancelled) reload could run. On a
869 // GET (read) cycle the abort is a 1-cycle DMA (one halt re-read) →
870 // CalculateDMADuration Y=1; on a PUT cycle it does NOT occur → Y=0
871 // ("the 1-cycle abort will not land on a write cycle"). Both clear
872 // the pending reload so the `dmc_dma_pending` loop below skips it.
873 if bus.dmc_abort_pending() {
874 if bus.dmc_abort_is_get_cycle() {
875 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
876 self.start_cycle(bus, true);
877 bus.dmc_abort_halt_step(addr);
878 self.end_cycle(bus, true);
879 } else {
880 bus.dmc_abort_cancel();
881 }
882 }
883 // W3-Stage-1 (`mc-r1-dma-unified`): the ONE DMA loop. It replaced
884 // three earlier loops (the standalone DMC drain, the sequential
885 // Stage-D OAM loop and the Program-M overlap loop), whose bus hooks
886 // were deprecated at v2.7.5 and removed at v2.9.8 (ADR 0042). Each
887 // iteration is one full R1 cycle (start_cycle -> the bus's
888 // unified TriCNES-dispatch cycle -> end_cycle), so every DMA
889 // cycle keeps the φ2 IRQ sample — the C1-safe shape. The
890 // engine itself (ONE driver for standalone DMC, standalone OAM,
891 // and the overlap) lives bus-side in `unified_dma_cycle`; the
892 // load-get-entry defer is folded into `unified_dma_pending`
893 // (pre-cycle, like the floor's while-gate) AND the engine's
894 // in-cycle entry gate (post-flip parity). A DMC DMA halts the CPU
895 // only on a READ cycle, which is why the loop lives here and in
896 // `idle_tick`, never in `write1`.
897 while bus.unified_dma_pending() {
898 // DMA halt cycles count against `cycles_emitted`.
899 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
900 self.start_cycle(bus, true);
901 bus.unified_dma_cycle(addr);
902 self.end_cycle(bus, true);
903 }
904 // R1 clean access shape: start_cycle (PPU caught up to the
905 // access's exact mc + bus cpu_clock) → bus.read → end_cycle
906 // (double catch-up + φ2 interrupt sample). Mesen `MemoryRead`.
907 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
908 self.start_cycle(bus, true);
909 let v = bus.read(addr);
910 self.end_cycle(bus, true);
911 v
912 }
913 }
914
915 /// Write `value` to `addr` *and* consume one CPU cycle (with bus
916 /// tick + interrupt sampling).
917 fn write1<B: Bus>(&mut self, bus: &mut B, addr: u16, value: u8) {
918 {
919 // accuracycoin-100 Phase 2: a CPU write cycle cannot be RDY-halted,
920 // so a 1-byte implicit abort matured before a write does NOT occur
921 // (Y=0). Cancel it with no halt cycle — this is the "will not land on
922 // a write cycle" half of the sweep that the read-only `read1` path
923 // can't reach.
924 if bus.dmc_abort_pending() {
925 bus.dmc_abort_cancel();
926 }
927 // R1 clean write shape (symmetric split — writes commit 2 mc
928 // later than reads). No interrupt sample latches here; end_cycle's
929 // handle_interrupts does the φ2 sample.
930 self.cycles_emitted = self.cycles_emitted.saturating_add(1);
931 self.start_cycle(bus, false);
932 bus.write(addr, value);
933 self.end_cycle(bus, false);
934 }
935 }
936
937 /// SH* unstable-store family helper (`SHA / SHX / SHY / SHS / TAS`,
938 /// opcodes `$9F / $93 / $9E / $9C / $9B`).
939 ///
940 /// Implements the 6502 unstable-store (SH*) algorithm — the
941 /// "unstable"/"highbyte" store opcodes (`value AND (high_byte + 1)`, with the
942 /// RDY/DMA quirk), pinned bit-for-bit by `AccuracyCoin`'s "Unofficial
943 /// Instructions: SH*" sub-test.
944 ///
945 /// Provenance: **derived from Mesen2's `SyaSxaAxa`** (`Core/NES/NesCpu.h`),
946 /// `GPL-3.0-or-later`. The `NESdev` community documents this behavior, but this
947 /// implementation was ported from Mesen2's — not written independently from
948 /// the documentation. The surrounding DMC-DMA interruption detection uses the
949 /// emulator's own bus cycle-count machinery. See NOTICE and
950 /// docs/originality-and-provenance.md (Section 1).
951 /// The algorithm:
952 ///
953 /// 1. Compute the page-crossed flag against `base + index_reg`.
954 /// 2. Perform a dummy read at the **unfixed** address
955 /// (`base + index_reg - 0x100` if page-crossed, else
956 /// `base + index_reg`). This is the cycle DMC DMA can
957 /// interrupt.
958 /// 3. Detect DMC-DMA interruption via `bus.cycle_count()`
959 /// before/after the dummy read — if more than 1 bus cycle
960 /// elapsed, a DMA fired.
961 /// 4. On page-cross, the address-high-byte is corrupted to
962 /// `original_addr_high AND value_reg`.
963 /// 5. Compute the store value:
964 /// - With DMA: just `value_reg` (the H+1 AND is suppressed
965 /// because the DMC pulled the bus low).
966 /// - Without DMA: `value_reg AND ((base >> 8) + 1)`.
967 /// 6. Write to the (possibly corrupted) final address.
968 ///
969 /// This shape is what `AccuracyCoin Unofficial Instructions: SH*`
970 /// sub-test 7 ("the cycle before the write had a DMA") brackets.
971 /// Pre-2026-05-23 `RustyNES` skipped the dummy read entirely and
972 /// always wrote `value_reg & (H+1)`, failing sub-test 7 across
973 /// all 5 SH* opcodes (error code 7).
974 fn sh_store<B: Bus>(&mut self, bus: &mut B, base: u16, index_reg: u8, value_reg: u8) {
975 let addr = base.wrapping_add(u16::from(index_reg));
976 let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
977
978 // Dummy read at the unfixed address (this is the cycle DMC
979 // DMA can halt). We sample the bus-side cycle count before
980 // and after to detect interruption — DMC DMA service path
981 // advances `bus.cycle` by 3+ extra ticks while the CPU's
982 // own `Cpu::cycles` (which `idle_tick` increments) only goes
983 // up by 1 for the read itself.
984 let cyc_before = bus.cycle_count();
985 let dummy_addr = if page_crossed {
986 addr.wrapping_sub(0x100)
987 } else {
988 addr
989 };
990 let _dummy = self.read1(bus, dummy_addr);
991 let had_dma = bus.cycle_count().wrapping_sub(cyc_before) > 1;
992
993 let addr_high = (addr >> 8) as u8;
994 let addr_low = (addr & 0xFF) as u8;
995 let final_high = if page_crossed {
996 addr_high & value_reg
997 } else {
998 addr_high
999 };
1000
1001 let write_value = if had_dma {
1002 // DMC DMA interrupted the dummy read — bus latch was
1003 // overwritten by the DMC fetch, so the store value loses
1004 // its AND-with-(H+1) component. Per Mesen2 `SyaSxaAxa`.
1005 value_reg
1006 } else {
1007 // Canonical "documented behavior 1" path: AND with
1008 // (base_high + 1).
1009 value_reg & ((base >> 8) as u8).wrapping_add(1)
1010 };
1011
1012 let final_addr = (u16::from(final_high) << 8) | u16::from(addr_low);
1013 self.write1(bus, final_addr, write_value);
1014 }
1015
1016 fn service_interrupt<B: Bus>(&mut self, bus: &mut B, vector: u16, brk: bool) {
1017 // Per-cycle interrupt sequence (7 cycles total when entered from
1018 // an interrupt edge, 6 from BRK because its opcode fetch already
1019 // burned cycle 1):
1020 // C1: opcode fetch (BRK only — IRQ/NMI skip this and instead
1021 // perform an extra dummy read in C2).
1022 // C2: dummy read of PC+1 (BRK) / filler/internal read (IRQ/NMI).
1023 // C3-C5: push PCH, PCL, P.
1024 // C6-C7: read vector lo, hi.
1025 // `cycles_emitted` is reset here so the caller's accounting starts
1026 // from this routine's first tick (the opcode-fetch tick from a BRK
1027 // is harmless — the BRK arm sets *cycles = 0 to suppress the
1028 // trailing burn loop).
1029 self.cycles_emitted = 0;
1030 // Reset the per-instruction interrupt sample latches so any NMI
1031 // edge during the push sequence below is captured here.
1032 self.nmi_first_tick = u8::MAX;
1033 self.irq_first_tick = u8::MAX;
1034 // Two filler reads for IRQ/NMI; one for BRK (the opcode fetch
1035 // counted as the other).
1036 //
1037 // W3-Stage-3 Part B (`mc-r1-brk-padding-read`): BRK's C2 is the
1038 // canonical PADDING-BYTE read at PC+1 — a REAL bus access on
1039 // silicon, not an internal cycle. AccuracyCoin `Implied Dummy
1040 // Reads` error 31 brackets exactly this: the test choreographs a
1041 // BRK whose padding read lands on `$4015`, which must clear the
1042 // frame-counter IRQ flag (RTI/RTS already emit their canonical
1043 // reads under `cpu-stack-dummy-reads`; BRK was the one gap — with
1044 // it the whole test PASSES, one sub-check beyond Mesen2's error
1045 // 34). The dispatch arm has already advanced PC past the padding
1046 // byte, so it sits at `pc - 1`.
1047 if brk {
1048 let _ = self.read1(bus, self.pc.wrapping_sub(1));
1049 } else {
1050 // v2.0.0 beta.2 (A2 every-cycle-bus-access, promoted to the only
1051 // path in beta.4): canonical hardware IRQ/NMI cycles 1-2 are
1052 // DUMMY READS of the interrupted PC (the suppressed opcode fetch
1053 // + suppressed operand fetch — nesdev `6502_cpu.txt`; Mesen2
1054 // `NesCpu::IRQ` issues two `DummyRead`s). This is the C1-trio
1055 // canary path: the conversion keeps the exact same two-cycle
1056 // start/end structure (φ2 samples unchanged) and only adds the
1057 // bus access + held-address update; the cpu_interrupts_v2 5/5
1058 // strict gate + AccuracyCoin 139/139 hold (verified at the
1059 // beta.2 gate).
1060 let _ = self.read1(bus, self.pc);
1061 let _ = self.read1(bus, self.pc);
1062 }
1063 self.push_u16(bus, self.pc);
1064 let mut p = self.p | Status::UNUSED;
1065 if brk {
1066 p.insert(Status::BREAK);
1067 } else {
1068 p.remove(Status::BREAK);
1069 }
1070 self.push(bus, p.bits());
1071 self.p.insert(Status::INTERRUPT_DISABLE);
1072 // NMI hijacking: real 6502 latches the vector to read on the
1073 // CYCLE just before the vector reads; if NMI is asserted at that
1074 // point, BRK / IRQ both read $FFFA / $FFFB instead of the
1075 // declared vector. We approximate "NMI asserted by now" with
1076 // "the NMI sample latch was hit during cycles 1..=5 of this
1077 // sequence."
1078 // R1: the hijack reads the DELAYED `mc_prev_need_nmi`, NOT the live
1079 // `mc_need_nmi` — an NMI edge latched ON the P-push cycle's φ2 sampler
1080 // must NOT hijack (the BRK/IRQ completes to its own vector and the NMI
1081 // is taken after one handler instruction). `mc_prev_need_nmi` is the
1082 // pre-edge value: 1 only if the NMI was pending BEFORE this cycle
1083 // (oracle-derived, cpu_interrupts_v2/2). Legacy uses `nmi_first_tick`.
1084 let effective_vector = if self.mc_prev_need_nmi && vector != NMI_VECTOR {
1085 self.mc_need_nmi = false;
1086 self.mc_prev_need_nmi = false;
1087 NMI_VECTOR
1088 } else {
1089 vector
1090 };
1091 // Phase 1.2 of Track C1 attempt 14: notify the bus of the vector
1092 // fetch BEFORE the low-byte read so the trace records the cycle
1093 // at which the CPU enters its vector-fetch micro-op (C6 of the
1094 // 7-cycle service sequence). `is_nmi` distinguishes a clean NMI
1095 // service entry from an IRQ/BRK service entry that an NMI edge
1096 // has hijacked to `$FFFA` — both fetch from `$FFFA` but only one
1097 // has `vector == NMI_VECTOR` at this call site.
1098 bus.notify_irq_service(effective_vector, vector == NMI_VECTOR);
1099 let lo = self.read1(bus, effective_vector);
1100 let hi = self.read1(bus, effective_vector + 1);
1101 self.pc = u16::from(lo) | (u16::from(hi) << 8);
1102 }
1103
1104 // ------------------------------------------------------------------
1105 // Addressing-mode resolvers. Each returns the effective address plus a
1106 // page-crossed flag; the caller decides whether to add a cycle.
1107 // ------------------------------------------------------------------
1108
1109 fn addr_zp<B: Bus>(&mut self, bus: &mut B) -> Operand {
1110 Operand {
1111 addr: u16::from(self.fetch_pc(bus)),
1112 page_crossed: false,
1113 }
1114 }
1115
1116 fn addr_zp_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1117 let base = self.fetch_pc(bus);
1118 // v2.0.0 beta.2 (A2 every-cycle-bus-access): canonical 6502 cycle 3
1119 // reads the UN-indexed zero-page address while the index add
1120 // completes, then discards it (nesdev `6502_cpu.txt`; Mesen2 models
1121 // it as a real `MemoryRead`). Zero-page addresses are always RAM
1122 // ($0000-$00FF), so the read is register-side-effect-free — but it
1123 // parks a real address on the bus (the held address a DMA halt
1124 // re-reads) instead of leaving the cycle busless in the burn-loop.
1125 // The burn-probe histogram pinned this family as 99% of the
1126 // remaining busless surface ($95 STA zp,X alone = 8,955 of 9,795
1127 // burned cycles over the AccuracyCoin battery). Promoted to the only
1128 // path in v2.0.0 beta.4.
1129 let _ = self.read1(bus, u16::from(base));
1130 Operand {
1131 addr: u16::from(base.wrapping_add(self.x)),
1132 page_crossed: false,
1133 }
1134 }
1135
1136 fn addr_zp_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1137 let base = self.fetch_pc(bus);
1138 // A2: same canonical un-indexed dummy read as `addr_zp_x` (cycle 3
1139 // of LDX/STX zp,Y and the unofficial LAX/SAX zp,Y arms).
1140 let _ = self.read1(bus, u16::from(base));
1141 Operand {
1142 addr: u16::from(base.wrapping_add(self.y)),
1143 page_crossed: false,
1144 }
1145 }
1146
1147 fn addr_abs<B: Bus>(&mut self, bus: &mut B) -> Operand {
1148 Operand {
1149 addr: self.fetch_pc_u16(bus),
1150 page_crossed: false,
1151 }
1152 }
1153
1154 fn addr_abs_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1155 let base = self.fetch_pc_u16(bus);
1156 let addr = base.wrapping_add(u16::from(self.x));
1157 let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1158 if page_crossed {
1159 // Canonical 6502 page-cross dummy read at the unfixed
1160 // address: (base_hi << 8) | ((base_lo + X) & 0xFF). The
1161 // high byte hasn't been incremented yet. This read has
1162 // side effects on PPU registers (`$2002` clears VBlank,
1163 // `$2007` advances the buffer) and is the hardware oracle
1164 // AccuracyCoin's `CPU Behavior :: Dummy read cycles`
1165 // Test 1 brackets via `LDA $20F2, X` with X=$10 reading
1166 // $2002 through the mirror.
1167 let dummy = (base & 0xFF00) | (addr & 0x00FF);
1168 let _ = self.read1(bus, dummy);
1169 }
1170 Operand { addr, page_crossed }
1171 }
1172
1173 fn addr_abs_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1174 let base = self.fetch_pc_u16(bus);
1175 let addr = base.wrapping_add(u16::from(self.y));
1176 let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1177 if page_crossed {
1178 // See addr_abs_x for the page-cross dummy-read rationale.
1179 let dummy = (base & 0xFF00) | (addr & 0x00FF);
1180 let _ = self.read1(bus, dummy);
1181 }
1182 Operand { addr, page_crossed }
1183 }
1184
1185 // ABS,X / ABS,Y operands for read-modify-write opcodes (ASL, LSR, ROL,
1186 // ROR, INC, DEC, and the unofficial SLO/RLA/SRE/RRA/DCP/ISC). Canonical
1187 // 6502: the unfixed-address dummy read happens UNCONDITIONALLY at
1188 // cycle 4 (not just on page cross) because the CPU has 7 cycles to
1189 // fill and cannot know the fixed address until the high-byte add
1190 // completes. Reads with side effects (`$2002` clears VBlank, `$4015`
1191 // clears frame-IRQ, `$2007` advances buffer) therefore fire twice on
1192 // RMW ABS,X. Bracketed by AccuracyCoin's `Implied Dummy Reads`
1193 // test 2: `SLO $4015,X` with X=0 expects the dummy read to clear the
1194 // frame-IRQ flag so the subsequent real read returns 0.
1195 fn addr_abs_x_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1196 let base = self.fetch_pc_u16(bus);
1197 let addr = base.wrapping_add(u16::from(self.x));
1198 let dummy = (base & 0xFF00) | (addr & 0x00FF);
1199 let _ = self.read1(bus, dummy);
1200 addr
1201 }
1202
1203 fn addr_abs_y_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1204 let base = self.fetch_pc_u16(bus);
1205 let addr = base.wrapping_add(u16::from(self.y));
1206 let dummy = (base & 0xFF00) | (addr & 0x00FF);
1207 let _ = self.read1(bus, dummy);
1208 addr
1209 }
1210
1211 /// (zp),Y operand for the unofficial read-modify-write opcodes
1212 /// (SLO/RLA/SRE/RRA/DCP/ISB `(zp),Y` — `$13/$33/$53/$73/$D3/$F3`).
1213 ///
1214 /// v2.0.0 beta.2 (A2 every-cycle-bus-access, promoted to the only path
1215 /// in beta.4): canonical 6502 8-cycle (zp),Y RMW performs the
1216 /// unfixed-address dummy read UNCONDITIONALLY at cycle 5 (like RMW
1217 /// ABS,X/Y above — the CPU cannot know the fixed address until the
1218 /// high-byte add completes), not only on page cross. The burn-probe
1219 /// histogram pinned these six arms as the last instruction-dispatch
1220 /// busless cycles (25 of the original 9,795).
1221 fn addr_ind_y_rmw<B: Bus>(&mut self, bus: &mut B) -> u16 {
1222 // Delegate to the plain resolver (which already emits the
1223 // unfixed-address dummy read on a page cross), then emit the
1224 // RMW's unconditional cycle-5 dummy for the non-crossing case.
1225 // On a non-crossing access the canonical unfixed address
1226 // `(base & 0xFF00) | (addr & 0xFF)` EQUALS the final address
1227 // (the high byte needed no fix-up), so reading `o.addr` here is
1228 // the silicon-exact target — do not "fix" this to a separate
1229 // unfixed computation, they are identical by construction.
1230 let o = self.addr_ind_y(bus);
1231 if !o.page_crossed {
1232 let _ = self.read1(bus, o.addr);
1233 }
1234 o.addr
1235 }
1236
1237 fn addr_ind_x<B: Bus>(&mut self, bus: &mut B) -> Operand {
1238 let base = self.fetch_pc(bus);
1239 // A2: canonical (zp,X) cycle 3 — dummy read of the UN-indexed
1240 // pointer address while the X add completes (same silicon behavior
1241 // as `addr_zp_x`; zero-page, so register-side-effect-free).
1242 let _ = self.read1(bus, u16::from(base));
1243 let ptr = base.wrapping_add(self.x);
1244 let lo = self.read1(bus, u16::from(ptr));
1245 let hi = self.read1(bus, u16::from(ptr.wrapping_add(1)));
1246 Operand {
1247 addr: u16::from(lo) | (u16::from(hi) << 8),
1248 page_crossed: false,
1249 }
1250 }
1251
1252 fn addr_ind_y<B: Bus>(&mut self, bus: &mut B) -> Operand {
1253 let ptr = self.fetch_pc(bus);
1254 let lo = self.read1(bus, u16::from(ptr));
1255 let hi = self.read1(bus, u16::from(ptr.wrapping_add(1)));
1256 let base = u16::from(lo) | (u16::from(hi) << 8);
1257 let addr = base.wrapping_add(u16::from(self.y));
1258 let page_crossed = (base & 0xFF00) != (addr & 0xFF00);
1259 if page_crossed {
1260 // Page-cross dummy read at the unfixed address — same as
1261 // addr_abs_x/y. Canonical 6502 behavior for LDA (zp),Y on
1262 // page crossing.
1263 let dummy = (base & 0xFF00) | (addr & 0x00FF);
1264 let _ = self.read1(bus, dummy);
1265 }
1266 Operand { addr, page_crossed }
1267 }
1268
1269 // ------------------------------------------------------------------
1270 // Top-level dispatch.
1271 //
1272 // The 256-way match is the cleanest way to express the entire opcode
1273 // table; the doc-comments are intentionally absent at the arm level
1274 // because each one is a single line of the standard 6502 reference and
1275 // adding individual arm comments would overwhelm the readability of the
1276 // table.
1277 // ------------------------------------------------------------------
1278
1279 #[allow(
1280 clippy::cognitive_complexity,
1281 clippy::too_many_lines,
1282 clippy::match_same_arms
1283 )]
1284 fn dispatch<B: Bus>(&mut self, bus: &mut B, op: u8, cycles: &mut u8) {
1285 match op {
1286 // === Loads ===
1287 0xA9 => {
1288 let v = self.fetch_pc(bus);
1289 self.lda(v);
1290 *cycles = 2;
1291 }
1292 0xA5 => {
1293 let o = self.addr_zp(bus);
1294 self.lda_addr(bus, o.addr);
1295 *cycles = 3;
1296 }
1297 0xB5 => {
1298 let o = self.addr_zp_x(bus);
1299 self.lda_addr(bus, o.addr);
1300 *cycles = 4;
1301 }
1302 0xAD => {
1303 let o = self.addr_abs(bus);
1304 self.lda_addr(bus, o.addr);
1305 *cycles = 4;
1306 }
1307 0xBD => {
1308 let o = self.addr_abs_x(bus);
1309 self.lda_addr(bus, o.addr);
1310 *cycles = 4 + u8::from(o.page_crossed);
1311 }
1312 0xB9 => {
1313 let o = self.addr_abs_y(bus);
1314 self.lda_addr(bus, o.addr);
1315 *cycles = 4 + u8::from(o.page_crossed);
1316 }
1317 0xA1 => {
1318 let o = self.addr_ind_x(bus);
1319 self.lda_addr(bus, o.addr);
1320 *cycles = 6;
1321 }
1322 0xB1 => {
1323 let o = self.addr_ind_y(bus);
1324 self.lda_addr(bus, o.addr);
1325 *cycles = 5 + u8::from(o.page_crossed);
1326 }
1327
1328 0xA2 => {
1329 let v = self.fetch_pc(bus);
1330 self.ldx(v);
1331 *cycles = 2;
1332 }
1333 0xA6 => {
1334 let o = self.addr_zp(bus);
1335 let v = self.read1(bus, o.addr);
1336 self.ldx(v);
1337 *cycles = 3;
1338 }
1339 0xB6 => {
1340 let o = self.addr_zp_y(bus);
1341 let v = self.read1(bus, o.addr);
1342 self.ldx(v);
1343 *cycles = 4;
1344 }
1345 0xAE => {
1346 let o = self.addr_abs(bus);
1347 let v = self.read1(bus, o.addr);
1348 self.ldx(v);
1349 *cycles = 4;
1350 }
1351 0xBE => {
1352 let o = self.addr_abs_y(bus);
1353 let v = self.read1(bus, o.addr);
1354 self.ldx(v);
1355 *cycles = 4 + u8::from(o.page_crossed);
1356 }
1357
1358 0xA0 => {
1359 let v = self.fetch_pc(bus);
1360 self.ldy(v);
1361 *cycles = 2;
1362 }
1363 0xA4 => {
1364 let o = self.addr_zp(bus);
1365 let v = self.read1(bus, o.addr);
1366 self.ldy(v);
1367 *cycles = 3;
1368 }
1369 0xB4 => {
1370 let o = self.addr_zp_x(bus);
1371 let v = self.read1(bus, o.addr);
1372 self.ldy(v);
1373 *cycles = 4;
1374 }
1375 0xAC => {
1376 let o = self.addr_abs(bus);
1377 let v = self.read1(bus, o.addr);
1378 self.ldy(v);
1379 *cycles = 4;
1380 }
1381 0xBC => {
1382 let o = self.addr_abs_x(bus);
1383 let v = self.read1(bus, o.addr);
1384 self.ldy(v);
1385 *cycles = 4 + u8::from(o.page_crossed);
1386 }
1387
1388 // === Stores ===
1389 0x85 => {
1390 let o = self.addr_zp(bus);
1391 self.write1(bus, o.addr, self.a);
1392 *cycles = 3;
1393 }
1394 0x95 => {
1395 let o = self.addr_zp_x(bus);
1396 self.write1(bus, o.addr, self.a);
1397 *cycles = 4;
1398 }
1399 0x8D => {
1400 let o = self.addr_abs(bus);
1401 self.write1(bus, o.addr, self.a);
1402 *cycles = 4;
1403 }
1404 0x9D => {
1405 let o = self.addr_abs_x(bus);
1406 // Canonical 6502: STA absolute,X performs a dummy
1407 // read at cycle 4 even when no page is crossed (unlike
1408 // LDA where cycle 4 is the real read). `addr_abs_x`
1409 // already issues the dummy read at the unfixed address
1410 // when page-crossed; for the no-page-cross case we add
1411 // it here at the final address.
1412 if !o.page_crossed {
1413 let _ = self.read1(bus, o.addr);
1414 }
1415 self.write1(bus, o.addr, self.a);
1416 *cycles = 5;
1417 }
1418 0x99 => {
1419 let o = self.addr_abs_y(bus);
1420 if !o.page_crossed {
1421 let _ = self.read1(bus, o.addr);
1422 }
1423 self.write1(bus, o.addr, self.a);
1424 *cycles = 5;
1425 }
1426 0x81 => {
1427 let o = self.addr_ind_x(bus);
1428 self.write1(bus, o.addr, self.a);
1429 *cycles = 6;
1430 }
1431 0x91 => {
1432 let o = self.addr_ind_y(bus);
1433 // Canonical STA (zp),Y always dummy-reads at cycle 5
1434 // even when no page is crossed. `addr_ind_y` already
1435 // handles the page-cross dummy at the unfixed address;
1436 // add the no-page-cross dummy here at the final address.
1437 if !o.page_crossed {
1438 let _ = self.read1(bus, o.addr);
1439 }
1440 self.write1(bus, o.addr, self.a);
1441 *cycles = 6;
1442 }
1443
1444 0x86 => {
1445 let o = self.addr_zp(bus);
1446 self.write1(bus, o.addr, self.x);
1447 *cycles = 3;
1448 }
1449 0x96 => {
1450 let o = self.addr_zp_y(bus);
1451 self.write1(bus, o.addr, self.x);
1452 *cycles = 4;
1453 }
1454 0x8E => {
1455 let o = self.addr_abs(bus);
1456 self.write1(bus, o.addr, self.x);
1457 *cycles = 4;
1458 }
1459
1460 0x84 => {
1461 let o = self.addr_zp(bus);
1462 self.write1(bus, o.addr, self.y);
1463 *cycles = 3;
1464 }
1465 0x94 => {
1466 let o = self.addr_zp_x(bus);
1467 self.write1(bus, o.addr, self.y);
1468 *cycles = 4;
1469 }
1470 0x8C => {
1471 let o = self.addr_abs(bus);
1472 self.write1(bus, o.addr, self.y);
1473 *cycles = 4;
1474 }
1475
1476 // === Transfers ===
1477 0xAA => {
1478 self.implied_dummy_read(bus);
1479 self.x = self.a;
1480 self.p.set_nz(self.x);
1481 *cycles = 2;
1482 }
1483 0xA8 => {
1484 self.implied_dummy_read(bus);
1485 self.y = self.a;
1486 self.p.set_nz(self.y);
1487 *cycles = 2;
1488 }
1489 0xBA => {
1490 self.implied_dummy_read(bus);
1491 self.x = self.s;
1492 self.p.set_nz(self.x);
1493 *cycles = 2;
1494 }
1495 0x8A => {
1496 self.implied_dummy_read(bus);
1497 self.a = self.x;
1498 self.p.set_nz(self.a);
1499 *cycles = 2;
1500 }
1501 0x9A => {
1502 self.implied_dummy_read(bus);
1503 self.s = self.x;
1504 *cycles = 2;
1505 }
1506 0x98 => {
1507 self.implied_dummy_read(bus);
1508 self.a = self.y;
1509 self.p.set_nz(self.a);
1510 *cycles = 2;
1511 }
1512
1513 // === Stack ===
1514 0x48 => {
1515 // PHA: C2 dummy read PC (the 6502 always reads the next byte on
1516 // the second cycle of a stack push), then the push.
1517 let _ = self.read1(bus, self.pc);
1518 self.push(bus, self.a);
1519 *cycles = 3;
1520 }
1521 0x08 => {
1522 let _ = self.read1(bus, self.pc);
1523 self.push(bus, (self.p | Status::BREAK | Status::UNUSED).bits());
1524 *cycles = 3;
1525 }
1526 0x68 => {
1527 // PLA: C2 dummy read PC, C3 dummy stack read (pre-increment),
1528 // then the pull.
1529 {
1530 let _ = self.read1(bus, self.pc);
1531 let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
1532 }
1533 self.a = self.pull(bus);
1534 self.p.set_nz(self.a);
1535 *cycles = 4;
1536 }
1537 0x28 => {
1538 {
1539 let _ = self.read1(bus, self.pc);
1540 let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
1541 }
1542 let v = self.pull(bus);
1543 let mut new_p = Status::from_bits_truncate(v);
1544 new_p.remove(Status::BREAK);
1545 new_p.insert(Status::UNUSED);
1546 self.p = new_p;
1547 *cycles = 4;
1548 }
1549
1550 // === Logical ===
1551 0x29 => {
1552 let v = self.fetch_pc(bus);
1553 self.and(v);
1554 *cycles = 2;
1555 }
1556 0x25 => {
1557 let o = self.addr_zp(bus);
1558 let v = self.read1(bus, o.addr);
1559 self.and(v);
1560 *cycles = 3;
1561 }
1562 0x35 => {
1563 let o = self.addr_zp_x(bus);
1564 let v = self.read1(bus, o.addr);
1565 self.and(v);
1566 *cycles = 4;
1567 }
1568 0x2D => {
1569 let o = self.addr_abs(bus);
1570 let v = self.read1(bus, o.addr);
1571 self.and(v);
1572 *cycles = 4;
1573 }
1574 0x3D => {
1575 let o = self.addr_abs_x(bus);
1576 let v = self.read1(bus, o.addr);
1577 self.and(v);
1578 *cycles = 4 + u8::from(o.page_crossed);
1579 }
1580 0x39 => {
1581 let o = self.addr_abs_y(bus);
1582 let v = self.read1(bus, o.addr);
1583 self.and(v);
1584 *cycles = 4 + u8::from(o.page_crossed);
1585 }
1586 0x21 => {
1587 let o = self.addr_ind_x(bus);
1588 let v = self.read1(bus, o.addr);
1589 self.and(v);
1590 *cycles = 6;
1591 }
1592 0x31 => {
1593 let o = self.addr_ind_y(bus);
1594 let v = self.read1(bus, o.addr);
1595 self.and(v);
1596 *cycles = 5 + u8::from(o.page_crossed);
1597 }
1598
1599 0x09 => {
1600 let v = self.fetch_pc(bus);
1601 self.ora(v);
1602 *cycles = 2;
1603 }
1604 0x05 => {
1605 let o = self.addr_zp(bus);
1606 let v = self.read1(bus, o.addr);
1607 self.ora(v);
1608 *cycles = 3;
1609 }
1610 0x15 => {
1611 let o = self.addr_zp_x(bus);
1612 let v = self.read1(bus, o.addr);
1613 self.ora(v);
1614 *cycles = 4;
1615 }
1616 0x0D => {
1617 let o = self.addr_abs(bus);
1618 let v = self.read1(bus, o.addr);
1619 self.ora(v);
1620 *cycles = 4;
1621 }
1622 0x1D => {
1623 let o = self.addr_abs_x(bus);
1624 let v = self.read1(bus, o.addr);
1625 self.ora(v);
1626 *cycles = 4 + u8::from(o.page_crossed);
1627 }
1628 0x19 => {
1629 let o = self.addr_abs_y(bus);
1630 let v = self.read1(bus, o.addr);
1631 self.ora(v);
1632 *cycles = 4 + u8::from(o.page_crossed);
1633 }
1634 0x01 => {
1635 let o = self.addr_ind_x(bus);
1636 let v = self.read1(bus, o.addr);
1637 self.ora(v);
1638 *cycles = 6;
1639 }
1640 0x11 => {
1641 let o = self.addr_ind_y(bus);
1642 let v = self.read1(bus, o.addr);
1643 self.ora(v);
1644 *cycles = 5 + u8::from(o.page_crossed);
1645 }
1646
1647 0x49 => {
1648 let v = self.fetch_pc(bus);
1649 self.eor(v);
1650 *cycles = 2;
1651 }
1652 0x45 => {
1653 let o = self.addr_zp(bus);
1654 let v = self.read1(bus, o.addr);
1655 self.eor(v);
1656 *cycles = 3;
1657 }
1658 0x55 => {
1659 let o = self.addr_zp_x(bus);
1660 let v = self.read1(bus, o.addr);
1661 self.eor(v);
1662 *cycles = 4;
1663 }
1664 0x4D => {
1665 let o = self.addr_abs(bus);
1666 let v = self.read1(bus, o.addr);
1667 self.eor(v);
1668 *cycles = 4;
1669 }
1670 0x5D => {
1671 let o = self.addr_abs_x(bus);
1672 let v = self.read1(bus, o.addr);
1673 self.eor(v);
1674 *cycles = 4 + u8::from(o.page_crossed);
1675 }
1676 0x59 => {
1677 let o = self.addr_abs_y(bus);
1678 let v = self.read1(bus, o.addr);
1679 self.eor(v);
1680 *cycles = 4 + u8::from(o.page_crossed);
1681 }
1682 0x41 => {
1683 let o = self.addr_ind_x(bus);
1684 let v = self.read1(bus, o.addr);
1685 self.eor(v);
1686 *cycles = 6;
1687 }
1688 0x51 => {
1689 let o = self.addr_ind_y(bus);
1690 let v = self.read1(bus, o.addr);
1691 self.eor(v);
1692 *cycles = 5 + u8::from(o.page_crossed);
1693 }
1694
1695 0x24 => {
1696 let o = self.addr_zp(bus);
1697 let v = self.read1(bus, o.addr);
1698 self.bit(v);
1699 *cycles = 3;
1700 }
1701 0x2C => {
1702 let o = self.addr_abs(bus);
1703 let v = self.read1(bus, o.addr);
1704 self.bit(v);
1705 *cycles = 4;
1706 }
1707
1708 // === Arithmetic ===
1709 0x69 => {
1710 let v = self.fetch_pc(bus);
1711 self.adc(v);
1712 *cycles = 2;
1713 }
1714 0x65 => {
1715 let o = self.addr_zp(bus);
1716 let v = self.read1(bus, o.addr);
1717 self.adc(v);
1718 *cycles = 3;
1719 }
1720 0x75 => {
1721 let o = self.addr_zp_x(bus);
1722 let v = self.read1(bus, o.addr);
1723 self.adc(v);
1724 *cycles = 4;
1725 }
1726 0x6D => {
1727 let o = self.addr_abs(bus);
1728 let v = self.read1(bus, o.addr);
1729 self.adc(v);
1730 *cycles = 4;
1731 }
1732 0x7D => {
1733 let o = self.addr_abs_x(bus);
1734 let v = self.read1(bus, o.addr);
1735 self.adc(v);
1736 *cycles = 4 + u8::from(o.page_crossed);
1737 }
1738 0x79 => {
1739 let o = self.addr_abs_y(bus);
1740 let v = self.read1(bus, o.addr);
1741 self.adc(v);
1742 *cycles = 4 + u8::from(o.page_crossed);
1743 }
1744 0x61 => {
1745 let o = self.addr_ind_x(bus);
1746 let v = self.read1(bus, o.addr);
1747 self.adc(v);
1748 *cycles = 6;
1749 }
1750 0x71 => {
1751 let o = self.addr_ind_y(bus);
1752 let v = self.read1(bus, o.addr);
1753 self.adc(v);
1754 *cycles = 5 + u8::from(o.page_crossed);
1755 }
1756
1757 0xE9 | 0xEB => {
1758 let v = self.fetch_pc(bus);
1759 self.sbc(v);
1760 *cycles = 2;
1761 }
1762 0xE5 => {
1763 let o = self.addr_zp(bus);
1764 let v = self.read1(bus, o.addr);
1765 self.sbc(v);
1766 *cycles = 3;
1767 }
1768 0xF5 => {
1769 let o = self.addr_zp_x(bus);
1770 let v = self.read1(bus, o.addr);
1771 self.sbc(v);
1772 *cycles = 4;
1773 }
1774 0xED => {
1775 let o = self.addr_abs(bus);
1776 let v = self.read1(bus, o.addr);
1777 self.sbc(v);
1778 *cycles = 4;
1779 }
1780 0xFD => {
1781 let o = self.addr_abs_x(bus);
1782 let v = self.read1(bus, o.addr);
1783 self.sbc(v);
1784 *cycles = 4 + u8::from(o.page_crossed);
1785 }
1786 0xF9 => {
1787 let o = self.addr_abs_y(bus);
1788 let v = self.read1(bus, o.addr);
1789 self.sbc(v);
1790 *cycles = 4 + u8::from(o.page_crossed);
1791 }
1792 0xE1 => {
1793 let o = self.addr_ind_x(bus);
1794 let v = self.read1(bus, o.addr);
1795 self.sbc(v);
1796 *cycles = 6;
1797 }
1798 0xF1 => {
1799 let o = self.addr_ind_y(bus);
1800 let v = self.read1(bus, o.addr);
1801 self.sbc(v);
1802 *cycles = 5 + u8::from(o.page_crossed);
1803 }
1804
1805 // === Compare ===
1806 0xC9 => {
1807 let v = self.fetch_pc(bus);
1808 self.cmp_with(self.a, v);
1809 *cycles = 2;
1810 }
1811 0xC5 => {
1812 let o = self.addr_zp(bus);
1813 let v = self.read1(bus, o.addr);
1814 self.cmp_with(self.a, v);
1815 *cycles = 3;
1816 }
1817 0xD5 => {
1818 let o = self.addr_zp_x(bus);
1819 let v = self.read1(bus, o.addr);
1820 self.cmp_with(self.a, v);
1821 *cycles = 4;
1822 }
1823 0xCD => {
1824 let o = self.addr_abs(bus);
1825 let v = self.read1(bus, o.addr);
1826 self.cmp_with(self.a, v);
1827 *cycles = 4;
1828 }
1829 0xDD => {
1830 let o = self.addr_abs_x(bus);
1831 let v = self.read1(bus, o.addr);
1832 self.cmp_with(self.a, v);
1833 *cycles = 4 + u8::from(o.page_crossed);
1834 }
1835 0xD9 => {
1836 let o = self.addr_abs_y(bus);
1837 let v = self.read1(bus, o.addr);
1838 self.cmp_with(self.a, v);
1839 *cycles = 4 + u8::from(o.page_crossed);
1840 }
1841 0xC1 => {
1842 let o = self.addr_ind_x(bus);
1843 let v = self.read1(bus, o.addr);
1844 self.cmp_with(self.a, v);
1845 *cycles = 6;
1846 }
1847 0xD1 => {
1848 let o = self.addr_ind_y(bus);
1849 let v = self.read1(bus, o.addr);
1850 self.cmp_with(self.a, v);
1851 *cycles = 5 + u8::from(o.page_crossed);
1852 }
1853
1854 0xE0 => {
1855 let v = self.fetch_pc(bus);
1856 self.cmp_with(self.x, v);
1857 *cycles = 2;
1858 }
1859 0xE4 => {
1860 let o = self.addr_zp(bus);
1861 let v = self.read1(bus, o.addr);
1862 self.cmp_with(self.x, v);
1863 *cycles = 3;
1864 }
1865 0xEC => {
1866 let o = self.addr_abs(bus);
1867 let v = self.read1(bus, o.addr);
1868 self.cmp_with(self.x, v);
1869 *cycles = 4;
1870 }
1871
1872 0xC0 => {
1873 let v = self.fetch_pc(bus);
1874 self.cmp_with(self.y, v);
1875 *cycles = 2;
1876 }
1877 0xC4 => {
1878 let o = self.addr_zp(bus);
1879 let v = self.read1(bus, o.addr);
1880 self.cmp_with(self.y, v);
1881 *cycles = 3;
1882 }
1883 0xCC => {
1884 let o = self.addr_abs(bus);
1885 let v = self.read1(bus, o.addr);
1886 self.cmp_with(self.y, v);
1887 *cycles = 4;
1888 }
1889
1890 // === Increments / decrements ===
1891 0xE6 => {
1892 let o = self.addr_zp(bus);
1893 self.inc_addr(bus, o.addr);
1894 *cycles = 5;
1895 }
1896 0xF6 => {
1897 let o = self.addr_zp_x(bus);
1898 self.inc_addr(bus, o.addr);
1899 *cycles = 6;
1900 }
1901 0xEE => {
1902 let o = self.addr_abs(bus);
1903 self.inc_addr(bus, o.addr);
1904 *cycles = 6;
1905 }
1906 0xFE => {
1907 let addr = self.addr_abs_x_rmw(bus);
1908 self.inc_addr(bus, addr);
1909 *cycles = 7;
1910 }
1911 0xC6 => {
1912 let o = self.addr_zp(bus);
1913 self.dec_addr(bus, o.addr);
1914 *cycles = 5;
1915 }
1916 0xD6 => {
1917 let o = self.addr_zp_x(bus);
1918 self.dec_addr(bus, o.addr);
1919 *cycles = 6;
1920 }
1921 0xCE => {
1922 let o = self.addr_abs(bus);
1923 self.dec_addr(bus, o.addr);
1924 *cycles = 6;
1925 }
1926 0xDE => {
1927 let addr = self.addr_abs_x_rmw(bus);
1928 self.dec_addr(bus, addr);
1929 *cycles = 7;
1930 }
1931 0xE8 => {
1932 self.implied_dummy_read(bus);
1933 self.x = self.x.wrapping_add(1);
1934 self.p.set_nz(self.x);
1935 *cycles = 2;
1936 }
1937 0xCA => {
1938 self.implied_dummy_read(bus);
1939 self.x = self.x.wrapping_sub(1);
1940 self.p.set_nz(self.x);
1941 *cycles = 2;
1942 }
1943 0xC8 => {
1944 self.implied_dummy_read(bus);
1945 self.y = self.y.wrapping_add(1);
1946 self.p.set_nz(self.y);
1947 *cycles = 2;
1948 }
1949 0x88 => {
1950 self.implied_dummy_read(bus);
1951 self.y = self.y.wrapping_sub(1);
1952 self.p.set_nz(self.y);
1953 *cycles = 2;
1954 }
1955
1956 // === Shifts ===
1957 0x0A => {
1958 self.implied_dummy_read(bus);
1959 self.a = self.asl_value(self.a);
1960 *cycles = 2;
1961 }
1962 0x06 => {
1963 let o = self.addr_zp(bus);
1964 self.asl_addr(bus, o.addr);
1965 *cycles = 5;
1966 }
1967 0x16 => {
1968 let o = self.addr_zp_x(bus);
1969 self.asl_addr(bus, o.addr);
1970 *cycles = 6;
1971 }
1972 0x0E => {
1973 let o = self.addr_abs(bus);
1974 self.asl_addr(bus, o.addr);
1975 *cycles = 6;
1976 }
1977 0x1E => {
1978 let addr = self.addr_abs_x_rmw(bus);
1979 self.asl_addr(bus, addr);
1980 *cycles = 7;
1981 }
1982
1983 0x4A => {
1984 self.implied_dummy_read(bus);
1985 self.a = self.lsr_value(self.a);
1986 *cycles = 2;
1987 }
1988 0x46 => {
1989 let o = self.addr_zp(bus);
1990 self.lsr_addr(bus, o.addr);
1991 *cycles = 5;
1992 }
1993 0x56 => {
1994 let o = self.addr_zp_x(bus);
1995 self.lsr_addr(bus, o.addr);
1996 *cycles = 6;
1997 }
1998 0x4E => {
1999 let o = self.addr_abs(bus);
2000 self.lsr_addr(bus, o.addr);
2001 *cycles = 6;
2002 }
2003 0x5E => {
2004 let addr = self.addr_abs_x_rmw(bus);
2005 self.lsr_addr(bus, addr);
2006 *cycles = 7;
2007 }
2008
2009 0x2A => {
2010 self.implied_dummy_read(bus);
2011 self.a = self.rol_value(self.a);
2012 *cycles = 2;
2013 }
2014 0x26 => {
2015 let o = self.addr_zp(bus);
2016 self.rol_addr(bus, o.addr);
2017 *cycles = 5;
2018 }
2019 0x36 => {
2020 let o = self.addr_zp_x(bus);
2021 self.rol_addr(bus, o.addr);
2022 *cycles = 6;
2023 }
2024 0x2E => {
2025 let o = self.addr_abs(bus);
2026 self.rol_addr(bus, o.addr);
2027 *cycles = 6;
2028 }
2029 0x3E => {
2030 let addr = self.addr_abs_x_rmw(bus);
2031 self.rol_addr(bus, addr);
2032 *cycles = 7;
2033 }
2034
2035 0x6A => {
2036 self.implied_dummy_read(bus);
2037 self.a = self.ror_value(self.a);
2038 *cycles = 2;
2039 }
2040 0x66 => {
2041 let o = self.addr_zp(bus);
2042 self.ror_addr(bus, o.addr);
2043 *cycles = 5;
2044 }
2045 0x76 => {
2046 let o = self.addr_zp_x(bus);
2047 self.ror_addr(bus, o.addr);
2048 *cycles = 6;
2049 }
2050 0x6E => {
2051 let o = self.addr_abs(bus);
2052 self.ror_addr(bus, o.addr);
2053 *cycles = 6;
2054 }
2055 0x7E => {
2056 let addr = self.addr_abs_x_rmw(bus);
2057 self.ror_addr(bus, addr);
2058 *cycles = 7;
2059 }
2060
2061 // === Branches ===
2062 //
2063 // The `branch_delays_irq` quirk: real 6502 branches poll IRQ
2064 // at the same point a 2-cycle untaken branch would — at the
2065 // opcode-fetch cycle (the canonical 2-cycle "second-to-last"
2066 // poll). The operand-fetch cycle and any extra taken /
2067 // page-cross cycles do NOT re-sample IRQ. We suppress IRQ
2068 // sampling for the remaining cycles of the instruction
2069 // immediately *before* the operand fetch — the opcode-fetch
2070 // sample (in `step()`) has already happened by this point.
2071 // See `docs/cpu-6502.md` §Interrupt logic and
2072 // <https://www.nesdev.org/wiki/CPU_interrupts>.
2073 0x10 => {
2074 self.skip_irq_sample = true;
2075 let off = self.fetch_pc(bus);
2076 *cycles = self.branch(bus, off, !self.p.contains(Status::NEGATIVE));
2077 }
2078 0x30 => {
2079 self.skip_irq_sample = true;
2080 let off = self.fetch_pc(bus);
2081 *cycles = self.branch(bus, off, self.p.contains(Status::NEGATIVE));
2082 }
2083 0x50 => {
2084 self.skip_irq_sample = true;
2085 let off = self.fetch_pc(bus);
2086 *cycles = self.branch(bus, off, !self.p.contains(Status::OVERFLOW));
2087 }
2088 0x70 => {
2089 self.skip_irq_sample = true;
2090 let off = self.fetch_pc(bus);
2091 *cycles = self.branch(bus, off, self.p.contains(Status::OVERFLOW));
2092 }
2093 0x90 => {
2094 self.skip_irq_sample = true;
2095 let off = self.fetch_pc(bus);
2096 *cycles = self.branch(bus, off, !self.p.contains(Status::CARRY));
2097 }
2098 0xB0 => {
2099 self.skip_irq_sample = true;
2100 let off = self.fetch_pc(bus);
2101 *cycles = self.branch(bus, off, self.p.contains(Status::CARRY));
2102 }
2103 0xD0 => {
2104 self.skip_irq_sample = true;
2105 let off = self.fetch_pc(bus);
2106 *cycles = self.branch(bus, off, !self.p.contains(Status::ZERO));
2107 }
2108 0xF0 => {
2109 self.skip_irq_sample = true;
2110 let off = self.fetch_pc(bus);
2111 *cycles = self.branch(bus, off, self.p.contains(Status::ZERO));
2112 }
2113
2114 // === Jumps / subroutine ===
2115 0x4C => {
2116 self.pc = self.fetch_pc_u16(bus);
2117 *cycles = 3;
2118 }
2119 0x6C => {
2120 let ptr = self.fetch_pc_u16(bus);
2121 self.pc = self.read_u16_with_wrap(bus, ptr);
2122 *cycles = 5;
2123 }
2124 0x20 => {
2125 // Canonical 6502 JSR cycle sequence — the high byte of
2126 // the target is read AFTER PC is pushed to the stack.
2127 // Wrong order is observable when JSR overwrites its own
2128 // operand via the pushed return address (AccuracyCoin
2129 // `CPU Behavior 2 :: JSR Edge Cases` Test 2 brackets
2130 // this exactly):
2131 // C1: opcode fetch (already done by `tick` dispatcher)
2132 // C2: fetch low byte of target → advances PC
2133 // C3: dummy read from stack at $0100|S (no-op)
2134 // C4: push PC high (PC is currently at the high-byte
2135 // operand address, which is exactly the return
2136 // address minus one)
2137 // C5: push PC low
2138 // C6: fetch high byte of target → PC = target
2139 let lo = self.fetch_pc(bus);
2140 let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2141 // self.pc now points at the high-byte operand; this is
2142 // the "return - 1" address JSR canonically pushes.
2143 let return_minus_one = self.pc;
2144 self.push(bus, (return_minus_one >> 8) as u8);
2145 self.push(bus, (return_minus_one & 0xFF) as u8);
2146 let hi = self.fetch_pc(bus);
2147 self.pc = u16::from(lo) | (u16::from(hi) << 8);
2148 *cycles = 6;
2149 }
2150 0x60 => {
2151 // Canonical 6502 RTS bus pattern (every cycle is a bus access):
2152 // C1 opcode fetch (dispatcher) | C2 dummy read PC |
2153 // C3 dummy stack read (pre-increment) |
2154 // C4 pull PCL | C5 pull PCH | C6 dummy read at the return addr.
2155 // Default build burns C2/C3/C6 as `idle_tick` (no bus access);
2156 // `cpu-stack-dummy-reads` emits the canonical dummy reads — the
2157 // DC-6 Y=3-vs-4 fix. See the cell-trace cross-diff.
2158 {
2159 let _ = self.read1(bus, self.pc);
2160 let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2161 let v = self.pull_u16(bus);
2162 let _ = self.read1(bus, v);
2163 self.pc = v.wrapping_add(1);
2164 }
2165 *cycles = 6;
2166 }
2167 0x40 => {
2168 // Canonical RTI bus pattern: C2 dummy read PC, C3 dummy stack
2169 // read (pre-increment) before the pulls. Default-off helper.
2170 {
2171 let _ = self.read1(bus, self.pc);
2172 let _ = self.read1(bus, STACK_BASE | u16::from(self.s));
2173 }
2174 let p = self.pull(bus);
2175 let mut new_p = Status::from_bits_truncate(p);
2176 new_p.remove(Status::BREAK);
2177 new_p.insert(Status::UNUSED);
2178 self.p = new_p;
2179 // RTI's I-flag change is observed by the IRQ sample
2180 // (unlike PLP / CLI / SEI which delay one instruction).
2181 self.irq_sample_i_flag = self.p.contains(Status::INTERRUPT_DISABLE);
2182 self.pc = self.pull_u16(bus);
2183 *cycles = 6;
2184 }
2185 0x00 => {
2186 // BRK is a 7-cycle interrupt with PC+2 pushed (PC already
2187 // advanced by fetch; advance one more for the padding byte).
2188 self.pc = self.pc.wrapping_add(1);
2189 self.service_interrupt(bus, IRQ_VECTOR, true);
2190 // R1/A2: suppress an NMI that became pending during/just-after
2191 // the BRK sequence so the FIRST instruction of the IRQ handler
2192 // runs before the NMI is taken (Mesen2 `NesCpu::BRK`
2193 // `_prevNeedNmi = false`; "needed for nmi_and_brk"). The NMI is
2194 // not lost — `mc_need_nmi` stays set and re-arms next cycle.
2195 {
2196 self.mc_prev_need_nmi = false;
2197 }
2198 // service_interrupt already burned 7 cycles; do NOT double-count.
2199 *cycles = 0;
2200 }
2201 0xEA => {
2202 self.implied_dummy_read(bus);
2203 *cycles = 2;
2204 }
2205
2206 // === Flag manipulation ===
2207 0x18 => {
2208 self.implied_dummy_read(bus);
2209 self.p.remove(Status::CARRY);
2210 *cycles = 2;
2211 }
2212 0x38 => {
2213 self.implied_dummy_read(bus);
2214 self.p.insert(Status::CARRY);
2215 *cycles = 2;
2216 }
2217 0x58 => {
2218 self.implied_dummy_read(bus);
2219 self.p.remove(Status::INTERRUPT_DISABLE);
2220 *cycles = 2;
2221 }
2222 0x78 => {
2223 self.implied_dummy_read(bus);
2224 self.p.insert(Status::INTERRUPT_DISABLE);
2225 *cycles = 2;
2226 }
2227 0xB8 => {
2228 self.implied_dummy_read(bus);
2229 self.p.remove(Status::OVERFLOW);
2230 *cycles = 2;
2231 }
2232 0xD8 => {
2233 self.implied_dummy_read(bus);
2234 self.p.remove(Status::DECIMAL);
2235 *cycles = 2;
2236 }
2237 0xF8 => {
2238 self.implied_dummy_read(bus);
2239 self.p.insert(Status::DECIMAL);
2240 *cycles = 2;
2241 }
2242
2243 // === Unofficial NOP variants ===
2244 // Implied / 1-byte NOPs
2245 0x1A | 0x3A | 0x5A | 0x7A | 0xDA | 0xFA => {
2246 self.implied_dummy_read(bus);
2247 *cycles = 2;
2248 }
2249 // Immediate / zero-page DOP (double NOP) variants: skip 1 byte.
2250 0x80 | 0x82 | 0x89 | 0xC2 | 0xE2 => {
2251 let _ = self.fetch_pc(bus);
2252 *cycles = 2;
2253 }
2254 0x04 | 0x44 | 0x64 => {
2255 let o = self.addr_zp(bus);
2256 let _ = self.read1(bus, o.addr); // unofficial DOP dummy read
2257 *cycles = 3;
2258 }
2259 0x14 | 0x34 | 0x54 | 0x74 | 0xD4 | 0xF4 => {
2260 let o = self.addr_zp_x(bus);
2261 let _ = self.read1(bus, o.addr); // unofficial DOP dummy read
2262 *cycles = 4;
2263 }
2264 // Absolute "TOP" (triple NOP) — must dummy-read the target so
2265 // that PPU-mirror side-effects (e.g. clearing $2002.7) fire,
2266 // matching real silicon and AccuracyCoin's All-NOPs Test 2.
2267 0x0C => {
2268 let o = self.addr_abs(bus);
2269 let _ = self.read1(bus, o.addr);
2270 *cycles = 4;
2271 }
2272 0x1C | 0x3C | 0x5C | 0x7C | 0xDC | 0xFC => {
2273 let o = self.addr_abs_x(bus);
2274 let _ = self.read1(bus, o.addr); // dummy read on TOP
2275 *cycles = 4 + u8::from(o.page_crossed);
2276 }
2277
2278 // === Stable unofficial: LAX, SAX ===
2279 0xA7 => {
2280 let o = self.addr_zp(bus);
2281 let v = self.read1(bus, o.addr);
2282 self.lax(v);
2283 *cycles = 3;
2284 }
2285 0xB7 => {
2286 let o = self.addr_zp_y(bus);
2287 let v = self.read1(bus, o.addr);
2288 self.lax(v);
2289 *cycles = 4;
2290 }
2291 0xAF => {
2292 let o = self.addr_abs(bus);
2293 let v = self.read1(bus, o.addr);
2294 self.lax(v);
2295 *cycles = 4;
2296 }
2297 0xBF => {
2298 let o = self.addr_abs_y(bus);
2299 let v = self.read1(bus, o.addr);
2300 self.lax(v);
2301 *cycles = 4 + u8::from(o.page_crossed);
2302 }
2303 0xA3 => {
2304 let o = self.addr_ind_x(bus);
2305 let v = self.read1(bus, o.addr);
2306 self.lax(v);
2307 *cycles = 6;
2308 }
2309 0xB3 => {
2310 let o = self.addr_ind_y(bus);
2311 let v = self.read1(bus, o.addr);
2312 self.lax(v);
2313 *cycles = 5 + u8::from(o.page_crossed);
2314 }
2315 0xAB => {
2316 let v = self.fetch_pc(bus);
2317 self.lax(v);
2318 *cycles = 2;
2319 } // LAX immediate (often listed as ATX). We follow nestest behavior.
2320
2321 0x87 => {
2322 let o = self.addr_zp(bus);
2323 self.write1(bus, o.addr, self.a & self.x);
2324 *cycles = 3;
2325 }
2326 0x97 => {
2327 let o = self.addr_zp_y(bus);
2328 self.write1(bus, o.addr, self.a & self.x);
2329 *cycles = 4;
2330 }
2331 0x8F => {
2332 let o = self.addr_abs(bus);
2333 self.write1(bus, o.addr, self.a & self.x);
2334 *cycles = 4;
2335 }
2336 0x83 => {
2337 let o = self.addr_ind_x(bus);
2338 self.write1(bus, o.addr, self.a & self.x);
2339 *cycles = 6;
2340 }
2341
2342 // === DCP (DEC + CMP) ===
2343 0xC7 => {
2344 let o = self.addr_zp(bus);
2345 self.dcp_addr(bus, o.addr);
2346 *cycles = 5;
2347 }
2348 0xD7 => {
2349 let o = self.addr_zp_x(bus);
2350 self.dcp_addr(bus, o.addr);
2351 *cycles = 6;
2352 }
2353 0xCF => {
2354 let o = self.addr_abs(bus);
2355 self.dcp_addr(bus, o.addr);
2356 *cycles = 6;
2357 }
2358 0xDF => {
2359 let addr = self.addr_abs_x_rmw(bus);
2360 self.dcp_addr(bus, addr);
2361 *cycles = 7;
2362 }
2363 0xDB => {
2364 let addr = self.addr_abs_y_rmw(bus);
2365 self.dcp_addr(bus, addr);
2366 *cycles = 7;
2367 }
2368 0xC3 => {
2369 let o = self.addr_ind_x(bus);
2370 self.dcp_addr(bus, o.addr);
2371 *cycles = 8;
2372 }
2373 0xD3 => {
2374 let addr = self.addr_ind_y_rmw(bus);
2375 self.dcp_addr(bus, addr);
2376 *cycles = 8;
2377 }
2378
2379 // === ISC (INC + SBC) ===
2380 0xE7 => {
2381 let o = self.addr_zp(bus);
2382 self.isc_addr(bus, o.addr);
2383 *cycles = 5;
2384 }
2385 0xF7 => {
2386 let o = self.addr_zp_x(bus);
2387 self.isc_addr(bus, o.addr);
2388 *cycles = 6;
2389 }
2390 0xEF => {
2391 let o = self.addr_abs(bus);
2392 self.isc_addr(bus, o.addr);
2393 *cycles = 6;
2394 }
2395 0xFF => {
2396 let addr = self.addr_abs_x_rmw(bus);
2397 self.isc_addr(bus, addr);
2398 *cycles = 7;
2399 }
2400 0xFB => {
2401 let addr = self.addr_abs_y_rmw(bus);
2402 self.isc_addr(bus, addr);
2403 *cycles = 7;
2404 }
2405 0xE3 => {
2406 let o = self.addr_ind_x(bus);
2407 self.isc_addr(bus, o.addr);
2408 *cycles = 8;
2409 }
2410 0xF3 => {
2411 let addr = self.addr_ind_y_rmw(bus);
2412 self.isc_addr(bus, addr);
2413 *cycles = 8;
2414 }
2415
2416 // === SLO (ASL + ORA) ===
2417 0x07 => {
2418 let o = self.addr_zp(bus);
2419 self.slo_addr(bus, o.addr);
2420 *cycles = 5;
2421 }
2422 0x17 => {
2423 let o = self.addr_zp_x(bus);
2424 self.slo_addr(bus, o.addr);
2425 *cycles = 6;
2426 }
2427 0x0F => {
2428 let o = self.addr_abs(bus);
2429 self.slo_addr(bus, o.addr);
2430 *cycles = 6;
2431 }
2432 0x1F => {
2433 let addr = self.addr_abs_x_rmw(bus);
2434 self.slo_addr(bus, addr);
2435 *cycles = 7;
2436 }
2437 0x1B => {
2438 let addr = self.addr_abs_y_rmw(bus);
2439 self.slo_addr(bus, addr);
2440 *cycles = 7;
2441 }
2442 0x03 => {
2443 let o = self.addr_ind_x(bus);
2444 self.slo_addr(bus, o.addr);
2445 *cycles = 8;
2446 }
2447 0x13 => {
2448 let addr = self.addr_ind_y_rmw(bus);
2449 self.slo_addr(bus, addr);
2450 *cycles = 8;
2451 }
2452
2453 // === RLA (ROL + AND) ===
2454 0x27 => {
2455 let o = self.addr_zp(bus);
2456 self.rla_addr(bus, o.addr);
2457 *cycles = 5;
2458 }
2459 0x37 => {
2460 let o = self.addr_zp_x(bus);
2461 self.rla_addr(bus, o.addr);
2462 *cycles = 6;
2463 }
2464 0x2F => {
2465 let o = self.addr_abs(bus);
2466 self.rla_addr(bus, o.addr);
2467 *cycles = 6;
2468 }
2469 0x3F => {
2470 let addr = self.addr_abs_x_rmw(bus);
2471 self.rla_addr(bus, addr);
2472 *cycles = 7;
2473 }
2474 0x3B => {
2475 let addr = self.addr_abs_y_rmw(bus);
2476 self.rla_addr(bus, addr);
2477 *cycles = 7;
2478 }
2479 0x23 => {
2480 let o = self.addr_ind_x(bus);
2481 self.rla_addr(bus, o.addr);
2482 *cycles = 8;
2483 }
2484 0x33 => {
2485 let addr = self.addr_ind_y_rmw(bus);
2486 self.rla_addr(bus, addr);
2487 *cycles = 8;
2488 }
2489
2490 // === SRE (LSR + EOR) ===
2491 0x47 => {
2492 let o = self.addr_zp(bus);
2493 self.sre_addr(bus, o.addr);
2494 *cycles = 5;
2495 }
2496 0x57 => {
2497 let o = self.addr_zp_x(bus);
2498 self.sre_addr(bus, o.addr);
2499 *cycles = 6;
2500 }
2501 0x4F => {
2502 let o = self.addr_abs(bus);
2503 self.sre_addr(bus, o.addr);
2504 *cycles = 6;
2505 }
2506 0x5F => {
2507 let addr = self.addr_abs_x_rmw(bus);
2508 self.sre_addr(bus, addr);
2509 *cycles = 7;
2510 }
2511 0x5B => {
2512 let addr = self.addr_abs_y_rmw(bus);
2513 self.sre_addr(bus, addr);
2514 *cycles = 7;
2515 }
2516 0x43 => {
2517 let o = self.addr_ind_x(bus);
2518 self.sre_addr(bus, o.addr);
2519 *cycles = 8;
2520 }
2521 0x53 => {
2522 let addr = self.addr_ind_y_rmw(bus);
2523 self.sre_addr(bus, addr);
2524 *cycles = 8;
2525 }
2526
2527 // === RRA (ROR + ADC) ===
2528 0x67 => {
2529 let o = self.addr_zp(bus);
2530 self.rra_addr(bus, o.addr);
2531 *cycles = 5;
2532 }
2533 0x77 => {
2534 let o = self.addr_zp_x(bus);
2535 self.rra_addr(bus, o.addr);
2536 *cycles = 6;
2537 }
2538 0x6F => {
2539 let o = self.addr_abs(bus);
2540 self.rra_addr(bus, o.addr);
2541 *cycles = 6;
2542 }
2543 0x7F => {
2544 let addr = self.addr_abs_x_rmw(bus);
2545 self.rra_addr(bus, addr);
2546 *cycles = 7;
2547 }
2548 0x7B => {
2549 let addr = self.addr_abs_y_rmw(bus);
2550 self.rra_addr(bus, addr);
2551 *cycles = 7;
2552 }
2553 0x63 => {
2554 let o = self.addr_ind_x(bus);
2555 self.rra_addr(bus, o.addr);
2556 *cycles = 8;
2557 }
2558 0x73 => {
2559 let addr = self.addr_ind_y_rmw(bus);
2560 self.rra_addr(bus, addr);
2561 *cycles = 8;
2562 }
2563
2564 // === ANC, ALR, ARR, AXS ===
2565 0x0B | 0x2B => {
2566 let v = self.fetch_pc(bus);
2567 self.a &= v;
2568 self.p.set_nz(self.a);
2569 self.p.set(Status::CARRY, self.a & 0x80 != 0);
2570 *cycles = 2;
2571 }
2572 0x4B => {
2573 let v = self.fetch_pc(bus);
2574 self.a &= v;
2575 let new_carry = self.a & 0x01 != 0;
2576 self.a >>= 1;
2577 self.p.set_nz(self.a);
2578 self.p.set(Status::CARRY, new_carry);
2579 *cycles = 2;
2580 }
2581 0x6B => {
2582 let v = self.fetch_pc(bus);
2583 self.a &= v;
2584 let carry_in = self.p.contains(Status::CARRY);
2585 self.a = (self.a >> 1) | (u8::from(carry_in) << 7);
2586 self.p.set_nz(self.a);
2587 let bit6 = self.a & 0x40 != 0;
2588 let bit5 = self.a & 0x20 != 0;
2589 self.p.set(Status::CARRY, bit6);
2590 self.p.set(Status::OVERFLOW, bit6 ^ bit5);
2591 *cycles = 2;
2592 }
2593 0xCB => {
2594 let v = self.fetch_pc(bus);
2595 let ax = self.a & self.x;
2596 let (res, overflow) = ax.overflowing_sub(v);
2597 self.x = res;
2598 self.p.set(Status::CARRY, !overflow);
2599 self.p.set_nz(res);
2600 *cycles = 2;
2601 }
2602
2603 // === Unstable: XAA, LAS, TAS, SHA, SHX, SHY ===
2604 0x8B => {
2605 // XAA / ANE: A = (A | const) & X & operand. nestest expects this.
2606 let v = self.fetch_pc(bus);
2607 self.a = (self.a | 0xFF) & self.x & v;
2608 self.p.set_nz(self.a);
2609 *cycles = 2;
2610 }
2611 0xBB => {
2612 let o = self.addr_abs_y(bus);
2613 let v = self.read1(bus, o.addr);
2614 let res = self.s & v;
2615 self.a = res;
2616 self.x = res;
2617 self.s = res;
2618 self.p.set_nz(res);
2619 *cycles = 4 + u8::from(o.page_crossed);
2620 }
2621 0x9B => {
2622 // TAS / SHS / XAS abs,Y: S = A & X; then SHA-style write
2623 // using `S` as the value register.
2624 let base = self.fetch_pc_u16(bus);
2625 self.s = self.a & self.x;
2626 self.sh_store(bus, base, self.y, self.s);
2627 *cycles = 5;
2628 }
2629 0x9F => {
2630 // SHA abs,Y. value_reg = A & X.
2631 let base = self.fetch_pc_u16(bus);
2632 self.sh_store(bus, base, self.y, self.a & self.x);
2633 *cycles = 5;
2634 }
2635 0x93 => {
2636 // SHA (zp),Y. Indirect; base from zp-pointer-resolved
2637 // low/high bytes. value_reg = A & X.
2638 let zp = self.fetch_pc(bus);
2639 let lo = self.read1(bus, u16::from(zp));
2640 let hi_byte = self.read1(bus, u16::from(zp.wrapping_add(1)));
2641 let base = u16::from(lo) | (u16::from(hi_byte) << 8);
2642 self.sh_store(bus, base, self.y, self.a & self.x);
2643 *cycles = 6;
2644 }
2645 0x9E => {
2646 // SHX abs,Y. value_reg = X.
2647 let base = self.fetch_pc_u16(bus);
2648 self.sh_store(bus, base, self.y, self.x);
2649 *cycles = 5;
2650 }
2651 0x9C => {
2652 // SHY abs,X. value_reg = Y. Index register is X here.
2653 let base = self.fetch_pc_u16(bus);
2654 self.sh_store(bus, base, self.x, self.y);
2655 *cycles = 5;
2656 }
2657
2658 // === JAM / KIL / STP ===
2659 0x02 | 0x12 | 0x22 | 0x32 | 0x42 | 0x52 | 0x62 | 0x72 | 0x92 | 0xB2 | 0xD2 | 0xF2 => {
2660 // v2.0.0 (every-cycle-bus-access): cycle 2 is a real dummy
2661 // read of the byte after the opcode before the CPU wedges —
2662 // the same silicon shape as the implied-opcode cycle-2 dummy
2663 // read. Caught post-promote by the burn-loop fail-loud
2664 // assert (this arm declared 2 cycles but emitted only the
2665 // opcode fetch — invisible to every probe workload, since no
2666 // test ROM executes a JAM).
2667 let _ = self.read1(bus, self.pc);
2668 self.jammed = true;
2669 *cycles = 2;
2670 }
2671 }
2672 }
2673
2674 // ------------------------------------------------------------------
2675 // Helpers / micro-ops.
2676 // ------------------------------------------------------------------
2677
2678 fn lda(&mut self, value: u8) {
2679 self.a = value;
2680 self.p.set_nz(value);
2681 }
2682
2683 fn lda_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2684 let v = self.read1(bus, addr);
2685 self.lda(v);
2686 }
2687
2688 fn ldx(&mut self, value: u8) {
2689 self.x = value;
2690 self.p.set_nz(value);
2691 }
2692
2693 fn ldy(&mut self, value: u8) {
2694 self.y = value;
2695 self.p.set_nz(value);
2696 }
2697
2698 fn and(&mut self, value: u8) {
2699 self.a &= value;
2700 self.p.set_nz(self.a);
2701 }
2702
2703 fn ora(&mut self, value: u8) {
2704 self.a |= value;
2705 self.p.set_nz(self.a);
2706 }
2707
2708 fn eor(&mut self, value: u8) {
2709 self.a ^= value;
2710 self.p.set_nz(self.a);
2711 }
2712
2713 fn bit(&mut self, value: u8) {
2714 let result = self.a & value;
2715 self.p.set(Status::ZERO, result == 0);
2716 self.p.set(Status::NEGATIVE, value & 0x80 != 0);
2717 self.p.set(Status::OVERFLOW, value & 0x40 != 0);
2718 }
2719
2720 fn adc(&mut self, value: u8) {
2721 let carry = u16::from(self.p.contains(Status::CARRY));
2722 let sum = u16::from(self.a) + u16::from(value) + carry;
2723 let result = sum as u8;
2724 self.p.set(Status::CARRY, sum > 0xFF);
2725 let overflow = ((self.a ^ result) & (value ^ result) & 0x80) != 0;
2726 self.p.set(Status::OVERFLOW, overflow);
2727 self.a = result;
2728 self.p.set_nz(self.a);
2729 }
2730
2731 fn sbc(&mut self, value: u8) {
2732 // SBC = ADC of inverted value.
2733 self.adc(value ^ 0xFF);
2734 }
2735
2736 fn cmp_with(&mut self, lhs: u8, rhs: u8) {
2737 let (r, borrow) = lhs.overflowing_sub(rhs);
2738 self.p.set(Status::CARRY, !borrow);
2739 self.p.set_nz(r);
2740 }
2741
2742 fn inc_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2743 let original = self.read1(bus, addr);
2744 // RMW dummy write: real 6502 writes the original byte back to the
2745 // same address before writing the modified value (visible at memory-
2746 // mapped registers like $4014 and $2007). See `docs/cpu-6502.md` and
2747 // nesdev wiki "Dummy writes".
2748 self.write1(bus, addr, original);
2749 let v = original.wrapping_add(1);
2750 self.write1(bus, addr, v);
2751 self.p.set_nz(v);
2752 }
2753
2754 fn dec_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2755 let original = self.read1(bus, addr);
2756 self.write1(bus, addr, original);
2757 let v = original.wrapping_sub(1);
2758 self.write1(bus, addr, v);
2759 self.p.set_nz(v);
2760 }
2761
2762 fn asl_value(&mut self, value: u8) -> u8 {
2763 self.p.set(Status::CARRY, value & 0x80 != 0);
2764 let r = value << 1;
2765 self.p.set_nz(r);
2766 r
2767 }
2768
2769 fn asl_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2770 let v = self.read1(bus, addr);
2771 // RMW dummy write — see `inc_addr`.
2772 self.write1(bus, addr, v);
2773 let r = self.asl_value(v);
2774 self.write1(bus, addr, r);
2775 }
2776
2777 fn lsr_value(&mut self, value: u8) -> u8 {
2778 self.p.set(Status::CARRY, value & 0x01 != 0);
2779 let r = value >> 1;
2780 self.p.set_nz(r);
2781 r
2782 }
2783
2784 fn lsr_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2785 let v = self.read1(bus, addr);
2786 self.write1(bus, addr, v);
2787 let r = self.lsr_value(v);
2788 self.write1(bus, addr, r);
2789 }
2790
2791 fn rol_value(&mut self, value: u8) -> u8 {
2792 let carry_in = u8::from(self.p.contains(Status::CARRY));
2793 self.p.set(Status::CARRY, value & 0x80 != 0);
2794 let r = (value << 1) | carry_in;
2795 self.p.set_nz(r);
2796 r
2797 }
2798
2799 fn rol_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2800 let v = self.read1(bus, addr);
2801 self.write1(bus, addr, v);
2802 let r = self.rol_value(v);
2803 self.write1(bus, addr, r);
2804 }
2805
2806 fn ror_value(&mut self, value: u8) -> u8 {
2807 let carry_in = u8::from(self.p.contains(Status::CARRY)) << 7;
2808 self.p.set(Status::CARRY, value & 0x01 != 0);
2809 let r = (value >> 1) | carry_in;
2810 self.p.set_nz(r);
2811 r
2812 }
2813
2814 fn ror_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2815 let v = self.read1(bus, addr);
2816 self.write1(bus, addr, v);
2817 let r = self.ror_value(v);
2818 self.write1(bus, addr, r);
2819 }
2820
2821 fn branch<B: Bus>(&mut self, bus: &mut B, offset: u8, condition: bool) -> u8 {
2822 if !condition {
2823 return 2;
2824 }
2825 // T-60-001 (2026-05-17 — 9th C1 attempt, branch axis): per
2826 // nesdev wiki §"CPU interrupts" §"Branch instructions", TAKEN
2827 // branches DELAY IRQ detection. Our `step()` samples IRQ at
2828 // the opcode-fetch (cycle 1 = tick 0) via `idle_tick`, then
2829 // each branch opcode sets `skip_irq_sample = true` before the
2830 // operand fetch to suppress sampling on the extra cycles. But
2831 // the cycle-1 sample is still recorded in `irq_first_tick`
2832 // and `promote_post_step_interrupts` will ARM the IRQ at the
2833 // end of this instruction (cycle-1 sample < last_tick on a
2834 // 3- or 4-cycle taken branch). That contradicts the
2835 // "branches delay IRQ" rule — IRQ should be deferred to the
2836 // NEXT instruction's poll. Drop the cycle-1 sample here on
2837 // taken branches; the next instruction's opcode fetch will
2838 // re-sample the (still-asserted, level-triggered) IRQ line
2839 // and arm it normally. NMI is edge-triggered and sampled on
2840 // every cycle the CPU is alive (per nesdev) — its first-tick
2841 // latch is intentionally NOT dropped.
2842 self.irq_first_tick = u8::MAX;
2843 // Canonical 6502 branch cycle sequence per nesdev wiki and
2844 // AccuracyCoin `CPU Behavior 2 :: Branch Dummy Reads` Test 4:
2845 // C1: opcode fetch (done by `tick` dispatcher)
2846 // C2: operand fetch (done by per-opcode `fetch_pc` before call)
2847 // C3: dummy read of PC (the byte after the operand) — this is
2848 // cycle 3 of the taken branch, and is observable as a
2849 // second consecutive read of `$2002` mirror through which
2850 // AccuracyCoin brackets the dummy.
2851 // C4: (only if page-crossed) dummy read of (old_pch | new_pcl)
2852 // — the unfixed-high-byte address before the high-byte
2853 // carry propagates.
2854 let _ = self.read1(bus, self.pc); // C3 dummy
2855 let signed = offset as i8 as i16;
2856 let old_pc = self.pc;
2857 let new_pc = (self.pc as i32 + i32::from(signed)) as u16;
2858 let crossed = (old_pc & 0xFF00) != (new_pc & 0xFF00);
2859 if crossed {
2860 // W1 (`mc-r1-branch-poll-points`): a page-cross taken branch
2861 // polls a SECOND time at C4-start — TriCNES's
2862 // `PollInterrupts_CantDisableIRQ` in the BPL microcode
2863 // (`100thCoin/TriCNES` `Emulator.cs` at `94f1b117`): if the C2-start
2864 // poll already saw the IRQ this one cannot un-see it (can-SET-
2865 // not-clear). `mc_run_irq` is frozen across the branch's
2866 // remaining cycles by the `handle_interrupts` early-return, so
2867 // sample the live line here (state as of end-of-C3) and OR it in;
2868 // the end-of-C4 `mc_prev_run_irq` copy then exposes it to the
2869 // next `step()` dispatch.
2870 if !self.mc_run_irq {
2871 self.mc_run_irq = bus.irq_level() && !self.irq_sample_i_flag;
2872 }
2873 // C4 page-cross dummy read at the unfixed address.
2874 let dummy = (old_pc & 0xFF00) | (new_pc & 0x00FF);
2875 let _ = self.read1(bus, dummy);
2876 }
2877 self.pc = new_pc;
2878 if crossed { 4 } else { 3 }
2879 }
2880
2881 fn lax(&mut self, value: u8) {
2882 self.a = value;
2883 self.x = value;
2884 self.p.set_nz(value);
2885 }
2886
2887 fn dcp_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2888 let original = self.read1(bus, addr);
2889 // RMW dummy write.
2890 self.write1(bus, addr, original);
2891 let v = original.wrapping_sub(1);
2892 self.write1(bus, addr, v);
2893 self.cmp_with(self.a, v);
2894 }
2895
2896 fn isc_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2897 let original = self.read1(bus, addr);
2898 self.write1(bus, addr, original);
2899 let v = original.wrapping_add(1);
2900 self.write1(bus, addr, v);
2901 self.sbc(v);
2902 }
2903
2904 fn slo_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2905 let v = self.read1(bus, addr);
2906 self.write1(bus, addr, v);
2907 let r = self.asl_value(v);
2908 self.write1(bus, addr, r);
2909 self.a |= r;
2910 self.p.set_nz(self.a);
2911 }
2912
2913 fn rla_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2914 let v = self.read1(bus, addr);
2915 self.write1(bus, addr, v);
2916 let r = self.rol_value(v);
2917 self.write1(bus, addr, r);
2918 self.a &= r;
2919 self.p.set_nz(self.a);
2920 }
2921
2922 fn sre_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2923 let v = self.read1(bus, addr);
2924 self.write1(bus, addr, v);
2925 let r = self.lsr_value(v);
2926 self.write1(bus, addr, r);
2927 self.a ^= r;
2928 self.p.set_nz(self.a);
2929 }
2930
2931 fn rra_addr<B: Bus>(&mut self, bus: &mut B, addr: u16) {
2932 let v = self.read1(bus, addr);
2933 self.write1(bus, addr, v);
2934 let r = self.ror_value(v);
2935 self.write1(bus, addr, r);
2936 self.adc(r);
2937 }
2938}